From 170dd941b95098efd6d3c06452f866c0e2214833 Mon Sep 17 00:00:00 2001 From: sashatrask Date: Wed, 30 Sep 2026 20:30:56 +0300 Subject: [PATCH] Initial server source import --- .dockerignore | 237 + .env.postgres.example | 14 + .gitattributes | 12 + .gitignore | 44 + AGENTS.md | 15 + DOCKER_MIGRATION.md | 418 + DOCKER_READINESS_AUDIT.md | 364 + Dockerfile | 368 + QUARANTINE_AUDIT.md | 102 + RUNTIME_CHEATSHEET.md | 273 + WINDOWS_IMPORT.md | 285 + WORKER_OPERATOR_EXPERIENCE_HANDOFF.md | 398 + WORKSPACE.md | 40 + app/.streamlit/config.toml | 10 + app/CHEATSHEET.md | 289 + app/DETECTOR_NOTES.md | 351 + app/JSONL_RECONCILIATION.md | 13 + app/KEYCHECKERS.md | 192 + app/admin_api.py | 4555 +++ app/app.py | 13 + app/audit_github_tokens.py | 15 + app/capacity_model.py | 55 + app/child_bootstrap.py | 486 + app/config.linux.yaml | 1159 + app/console_runner.py | 7156 ++++ app/container_import.py | 1420 + app/container_import_config.py | 127 + app/container_projection_recovery.py | 416 + app/container_runtime.py | 594 + app/dashboard.py | 2581 ++ app/db_backend.py | 836 + app/docker_depth_experiment.py | 3898 ++ app/docker_depth_operator.py | 546 + app/docker_depth_report.py | 1164 + app/docker_shadow.py | 849 + app/host_agent_apply.py | 1119 + app/host_agent_client.py | 131 + app/host_agent_lifecycle.py | 1477 + app/host_agent_protocol.py | 315 + app/host_agent_reconcile.py | 94 + app/host_agent_runtime.py | 214 + app/host_agent_server.py | 167 + app/host_agent_state.py | 412 + app/janitor.py | 455 + app/jsonl_projector.py | 736 + app/keycheck_accounting_smoke.py | 71 + app/keycheck_candidates.py | 524 + app/keycheck_runner.py | 2842 ++ app/keycheckers/__init__.py | 1 + .../anthropic/anthropicKeycheck.py | 274 + app/keycheckers/aws/awsKeycheck.py | 607 + app/keycheckers/azure/azureKeycheck.py | 946 + app/keycheckers/deepseek/deepseekKeycheck.py | 292 + .../dockerhub/dockerhubKeycheck.py | 240 + app/keycheckers/gcp/gcpKeycheck.py | 789 + app/keycheckers/gemini/geminiKeycheck.py | 738 + app/keycheckers/github/githubKeycheck.py | 212 + app/keycheckers/gitlab/gitlabKeycheck.py | 178 + app/keycheckers/groq/groqKeycheck.py | 275 + .../huggingface/huggingfaceKeycheck.py | 142 + app/keycheckers/keycheck_common.py | 2465 ++ app/keycheckers/kimi/kimiKeycheck.py | 499 + app/keycheckers/openai/Keycheck.py | 572 + .../openrouter/OpenrouterKeycheck.py | 498 + app/keycheckers/provider_resolution.py | 189 + .../providerResolverKeycheck.py | 203 + app/keycheckers/qwen/qwenKeycheck.py | 773 + .../replicate/replicateKeycheck.py | 403 + app/keycheckers/xai/xaiKeycheck.py | 231 + app/keycheckers/zai/zaiKeycheck.py | 508 + app/lifecycle_authority.py | 768 + app/managed_files.py | 1440 + app/migrate_layout.py | 324 + app/migrate_observability_db.py | 137 + app/migrate_runtime_safety.py | 3504 ++ app/optimize_dashboard_db.py | 21 + app/owned_process.py | 1346 + app/paths.py | 237 + app/postgres_runtime.py | 1955 + app/process_identity.py | 449 + app/query_policy.py | 126 + app/remote_worker_bootstrap.py | 162 + app/remote_worker_client.py | 2625 ++ app/requirements-keycheckers.txt | 4 + app/requirements.txt | 10 + app/result_bundle.py | 757 + app/result_ingester.py | 605 + app/result_spool.py | 1073 + app/runtime_bootstrap.py | 143 + app/runtime_document.py | 1461 + app/runtime_document_io.py | 1442 + app/runtime_security.py | 1338 + app/scan_execution.py | 941 + app/scan_manager.py | 71 + app/scanner.py | 17045 +++++++++ app/scanner_db.py | 30695 ++++++++++++++++ app/scanner_error_policy_smoke.py | 423 + app/supervisor.py | 6562 ++++ app/supervisor_instance.py | 378 + app/sync_alive_github_tokens.py | 356 + app/target_identity.py | 199 + app/trufflehog-custom-detectors.yaml | 125 + app/ui_components.py | 12 + app/worker_api.py | 1137 + app/worker_assignment.py | 1001 + app/worker_assignment_runner.py | 1166 + app/worker_cli.py | 1243 + app/worker_contracts.py | 1165 + app/worker_local_state.py | 1396 + app/worker_package.py | 518 + app/worker_package_builder.py | 537 + app/worker_supervisor.py | 1103 + attach_runtime.ps1 | 11 + build_worker_release.cmd | 4 + build_worker_release.ps1 | 396 + cleanup_stale_agentui_vite.ps1 | 146 + compose.e2e.yaml | 202 + compose.edge.yaml | 56 + compose.shared-host.yaml | 62 + compose.snapshot-import.yaml | 22 + compose.windows-import.yaml | 7 + compose.yaml | 109 + deploy/capacity50/Dockerfile | 41 + deploy/capacity50/deploy.sh | 481 + .../release_stopped_pipeline_leases.py | 74 + deploy/capacity50/render_config.py | 73 + deploy/edge/Caddyfile | 133 + deploy/edge/Caddyfile.shared-host | 147 + deploy/edge/Dockerfile | 23 + deploy/edge/README.md | 253 + deploy/edge/admin-denylist.caddy | 1 + deploy/edge/automatic-tls.caddy | 1 + deploy/edge/entrypoint.sh | 71 + deploy/edge/host-caddy-shared.caddy | 26 + deploy/fail2ban/Dockerfile.edge-e2e | 61 + .../action.d-truf-caddy-admin-denylist.conf | 4 + deploy/fail2ban/edge_e2e_docker_shim.py | 194 + deploy/fail2ban/fail2ban.d-edge-e2e.local | 5 + .../fail2ban.d-truf-persistence.local | 3 + deploy/fail2ban/filter.d-truf-admin-auth.conf | 4 + deploy/fail2ban/jail.d-truf-admin-auth.local | 9 + deploy/fail2ban/truf_caddy_admin_denylist.py | 438 + deploy/host-agent/truf-host-agent.conf | 8 + deploy/host-agent/truf-host-agent.service | 50 + deploy/host-agent/truf-host-agent.socket | 16 + deploy/host-agent/truf_host_agent.py | 51 + deploy/host-agent/truf_host_agent_install.py | 571 + .../truf-caddy-admin-denylist-expire.service | 18 + .../truf-caddy-admin-denylist-expire.timer | 11 + deploy/worker/compose.yaml | 27 + deploy/worker/workerctl.ps1 | 3 + deploy/worker/workerctl.sh | 3 + deploy_capacity50.ps1 | 88 + docker-compose.postgres.yml | 40 + docker/Dockerfile.edge-e2e | 5 + docker/build-dependencies/README.md | 273 + docker/build-dependencies/requirements.in | 3 + docker/build-dependencies/requirements.lock | 40 + docker/requirements-test.in | 1 + docker/requirements-test.lock | 33 + docker/requirements-worker.in | 3 + docker/requirements-worker.lock | 365 + docker/requirements.in | 6 + docker/requirements.lock | 1216 + docker/test_verify.py | 196 + docker/test_windows_snapshot.py | 1097 + docker/verify.py | 909 + docker/verify_edge_e2e.py | 1164 + docker/verify_packaged_workers.py | 2072 ++ docker/windows_snapshot.py | 880 + docker/worker-package-pins.json | 43 + ...ub-discovery-retry-null-type-2026-09-25.md | 78 + ...tatus-scan-deadline-readback-2026-09-25.md | 50 + ...-windows-scan-timestamps-utc-2026-09-25.md | 72 + ...-oserror-mislabeled-local-io-2026-09-25.md | 50 + ...nd-to-end-scanner-validation-2026-09-22.md | 316 + docs/end-to-end-scanner-validation.md | 339 + docs/extended-live-validation-2026-09-26.md | 105 + docs/remote-worker-cheatsheet-docker-ru.md | 77 + docs/remote-worker-cheatsheet-linux-ru.md | 70 + docs/remote-worker-cheatsheet-windows-ru.md | 55 + docs/remote-worker-operations.md | 454 + docs/remote-worker-quickstart-ru.md | 98 + docs/session-handoff/CURRENT_STATE.md | 78 + docs/session-handoff/DECISIONS.md | 94 + docs/session-handoff/EVIDENCE_INDEX.md | 140 + docs/session-handoff/README.md | 35 + ...erator-experience-live-trace-2026-09-25.md | 336 + ...erator-experience-validation-2026-09-24.md | 269 + ...orker-parallelism-validation-2026-09-23.md | 473 + monitor_runtime_lag.ps1 | 262 + .../.openspec.yaml | 2 + .../design.md | 352 + .../proposal.md | 29 + .../docker-layer-content-scanning/spec.md | 218 + .../tasks.md | 46 + .../.openspec.yaml | 2 + .../add-layer-aware-docker-scanning/design.md | 254 + .../proposal.md | 31 + .../docker-layer-content-scanning/spec.md | 190 + .../add-layer-aware-docker-scanning/tasks.md | 45 + .../.openspec.yaml | 2 + .../baseline-evidence.md | 128 + .../add-minimal-remote-scan-workers/design.md | 147 + .../proposal.md | 29 + .../specs/distributed-scan-workers/spec.md | 145 + .../specs/restricted-public-access/spec.md | 75 + .../add-minimal-remote-scan-workers/tasks.md | 49 + .../changes/add-postman-source/.openspec.yaml | 2 + openspec/changes/add-postman-source/design.md | 80 + .../changes/add-postman-source/proposal.md | 32 + .../specs/postman-source/spec.md | 158 + openspec/changes/add-postman-source/tasks.md | 69 + .../.openspec.yaml | 2 + .../HANDOFF.md | 1549 + .../design.md | 155 + .../proposal.md | 32 + .../spec.md | 102 + .../specs/managed-runtime-editing/spec.md | 182 + .../multisource-worker-assignments/spec.md | 120 + .../specs/web-operations-console/spec.md | 142 + .../add-web-operations-control-plane/tasks.md | 92 + .../.openspec.yaml | 2 + .../add-worker-operator-experience/design.md | 206 + .../proposal.md | 38 + .../research.md | 207 + .../specs/worker-admin-experience/spec.md | 71 + .../specs/worker-diagnostics/spec.md | 86 + .../specs/worker-operator-supervisor/spec.md | 74 + .../specs/worker-progress-deadlines/spec.md | 79 + .../add-worker-operator-experience/tasks.md | 44 + .../cold-policy-stale-backlog/.openspec.yaml | 2 + .../cold-policy-stale-backlog/design.md | 80 + .../cold-policy-stale-backlog/proposal.md | 23 + .../specs/discovery-keyword-pruning/spec.md | 20 + .../specs/target-queue-policy-holds/spec.md | 78 + .../cold-policy-stale-backlog/tasks.md | 21 + .../.openspec.yaml | 2 + .../design.md | 78 + .../proposal.md | 24 + .../specs/openai-ecosystem-discovery/spec.md | 73 + .../tasks.md | 17 + .../.openspec.yaml | 2 + .../design.md | 51 + .../proposal.md | 23 + .../spec.md | 30 + .../tasks.md | 16 + .../.openspec.yaml | 2 + .../design.md | 82 + .../proposal.md | 27 + .../specs/keycheck-accounting/spec.md | 76 + .../tasks.md | 36 + .../.openspec.yaml | 2 + .../design.md | 70 + .../proposal.md | 27 + .../specs/dockerhub-search-pagination/spec.md | 68 + .../tasks.md | 21 + .../improve-core-scan-coverage/.openspec.yaml | 2 + .../improve-core-scan-coverage/design.md | 120 + .../improve-core-scan-coverage/proposal.md | 27 + .../docker-layer-graph-selection/spec.md | 49 + .../specs/git-ref-delta-scanning/spec.md | 68 + .../improve-core-scan-coverage/tasks.md | 20 + .../.openspec.yaml | 2 + .../design.md | 54 + .../proposal.md | 23 + .../specs/worker-operator-lifecycle/spec.md | 41 + .../tasks.md | 16 + .../validation.md | 79 + .../.openspec.yaml | 2 + .../design.md | 91 + .../proposal.md | 31 + .../dockerhub-incremental-discovery/spec.md | 164 + .../tasks.md | 30 + .../.openspec.yaml | 2 + .../design.md | 75 + .../proposal.md | 23 + .../specs/discovery-keyword-pruning/spec.md | 61 + .../specs/openai-discovery-coverage/spec.md | 37 + .../tasks.md | 19 + .../.openspec.yaml | 2 + .../rescan-updated-core-targets/design.md | 87 + .../rescan-updated-core-targets/proposal.md | 28 + .../specs/updated-target-rescan/spec.md | 87 + .../rescan-updated-core-targets/tasks.md | 24 + .../.openspec.yaml | 2 + .../resolve-ambiguous-key-providers/design.md | 78 + .../proposal.md | 27 + .../ambiguous-provider-resolution/spec.md | 52 + .../specs/zai-key-validation/spec.md | 49 + .../resolve-ambiguous-key-providers/tasks.md | 27 + .../.openspec.yaml | 2 + .../design.md | 68 + .../proposal.md | 24 + .../specs/openai-discovery-coverage/spec.md | 57 + .../tasks.md | 16 + .../retire-zero-alive-keywords/.openspec.yaml | 2 + .../retire-zero-alive-keywords/design.md | 81 + .../retire-zero-alive-keywords/proposal.md | 26 + .../specs/discovery-keyword-pruning/spec.md | 56 + .../specs/target-queue-policy-holds/spec.md | 37 + .../retire-zero-alive-keywords/tasks.md | 22 + .../.openspec.yaml | 2 + .../design.md | 134 + .../proposal.md | 27 + .../specs/docker-depth-experiment/spec.md | 252 + .../tasks.md | 52 + .../.openspec.yaml | 2 + .../design.md | 134 + .../proposal.md | 50 + .../docker-rank1-breadth-experiment/spec.md | 97 + .../tasks.md | 35 + .../changes/simplify-dashboard/.openspec.yaml | 2 + openspec/changes/simplify-dashboard/design.md | 52 + .../changes/simplify-dashboard/proposal.md | 29 + .../specs/single-page-observability/spec.md | 72 + openspec/changes/simplify-dashboard/tasks.md | 17 + .../.openspec.yaml | 2 + .../design.md | 89 + .../proposal.md | 28 + .../specs/provider-key-validation/spec.md | 48 + .../specs/scan-coverage-sizing/spec.md | 33 + .../specs/scanner-runtime-stability/spec.md | 40 + .../specs/source-discovery-targeting/spec.md | 48 + .../tasks.md | 48 + .../.openspec.yaml | 2 + .../design.md | 85 + .../proposal.md | 30 + .../specs/docker-scan-lifecycle/spec.md | 74 + .../tasks.md | 22 + .../.openspec.yaml | 2 + .../design.md | 55 + .../proposal.md | 27 + .../specs/gitlab-discovery-resilience/spec.md | 34 + .../tasks.md | 17 + .../.openspec.yaml | 2 + .../design.md | 63 + .../proposal.md | 28 + .../specs/gitlab-scan-lifecycle/spec.md | 56 + .../tasks.md | 19 + .../.openspec.yaml | 2 + .../design.md | 70 + .../proposal.md | 22 + .../specs/remote-assignment-capacity/spec.md | 60 + .../support-fifty-remote-assignments/tasks.md | 23 + .../validation.md | 27 + openspec/config.yaml | 20 + openspec/parking-lot.md | 96 + pytest.ini | 2 + start_core_runtime.ps1 | 28 + start_freeze_counters.ps1 | 122 + start_runtime.ps1 | 21 + stop_runtime.ps1 | 11 + tests/container_e2e.py | 3515 ++ tests/container_import_stop_e2e.py | 124 + tests/container_projection_recovery_e2e.py | 668 + tests/container_unit.py | 508 + tests/edge_e2e_backend.py | 433 + tests/edge_e2e_client.py | 428 + tests/fixtures/worker_tls_cert.pem | 20 + tests/fixtures/worker_tls_key.pem | 29 + tests/owned_process_helper.py | 159 + tests/packaged_worker_e2e_server.py | 626 + tests/parity_helpers.py | 126 + tests/requirements-browser.txt | 1 + tests/test_admin_api.py | 4182 +++ tests/test_admin_browser.py | 278 + tests/test_ambiguous_provider_resolution.py | 139 + tests/test_api_deadline_plumbing.py | 76 + tests/test_api_proxy_routing.py | 297 + tests/test_artifact_lifecycle_slots.py | 579 + tests/test_bounded_state_high_fixes.py | 741 + tests/test_capacity50_deploy.py | 77 + tests/test_container_e2e_helpers.py | 626 + tests/test_container_import.py | 1814 + tests/test_container_import_config.py | 265 + tests/test_container_migration_paths.py | 371 + tests/test_container_projection_recovery.py | 766 + tests/test_container_provider_portability.py | 507 + tests/test_container_runtime.py | 1452 + tests/test_container_security.py | 560 + ..._custom_provider_detector_compatibility.py | 168 + tests/test_dashboard_behavior.py | 310 + tests/test_dashboard_secret_guard.py | 61 + tests/test_db_backend_safety.py | 325 + .../test_direct_entrypoint_bytecode_policy.py | 343 + tests/test_discovery_only_cycle.py | 447 + tests/test_discovery_producer_supervisor.py | 238 + tests/test_discovery_request_budgets.py | 101 + tests/test_distributed_core_profile.py | 52 + tests/test_docker_codec_recovery.py | 712 + tests/test_docker_coverage_diagnostics.py | 313 + tests/test_docker_depth_cohort_holds.py | 1031 + tests/test_docker_depth_experiment.py | 566 + tests/test_docker_depth_experiment_schema.py | 898 + tests/test_docker_depth_operator.py | 316 + tests/test_docker_depth_report.py | 1086 + tests/test_docker_depth_resolver.py | 2834 ++ tests/test_docker_depth_resolver_refund.py | 115 + tests/test_docker_discovery_provenance.py | 753 + tests/test_docker_foundation.py | 326 + tests/test_docker_layer_scanning.py | 1833 + tests/test_docker_producer_cleanup.py | 112 + tests/test_docker_recovery_diagnostics.py | 611 + tests/test_docker_shadow_coverage.py | 43 + tests/test_docker_staging_bounds.py | 411 + tests/test_dockerhub_incremental_discovery.py | 972 + tests/test_dockerhub_search_pagination.py | 749 + tests/test_edge_deployment.py | 650 + tests/test_entrypoint_authority.py | 696 + tests/test_exact_git_scan_planning.py | 666 + tests/test_finding_pipeline_high_fixes.py | 482 + tests/test_gcp_keycheck_security.py | 143 + tests/test_gcp_rsa_adc_hardening.py | 195 + ...test_gemini_legacy_migration_durability.py | 187 + tests/test_gharchive_bounds.py | 187 + tests/test_git_checkout_recovery.py | 675 + tests/test_git_clone_authority.py | 316 + tests/test_git_diagnostic_coverage.py | 371 + tests/test_gitlab_resilience_fixes.py | 321 + tests/test_high_authority_bootstrap_fixes.py | 357 + tests/test_high_only_round6.py | 578 + tests/test_high_only_round7.py | 655 + tests/test_host_agent_apply.py | 849 + tests/test_host_agent_deploy.py | 291 + tests/test_host_agent_lifecycle.py | 1036 + tests/test_host_agent_linux.py | 87 + tests/test_host_agent_protocol.py | 529 + tests/test_host_agent_reconcile.py | 107 + tests/test_host_agent_runtime.py | 406 + tests/test_host_agent_state.py | 195 + tests/test_huggingface_long_paths.py | 427 + tests/test_import_suffix_manifest.py | 266 + tests/test_janitor_bounded.py | 281 + ...test_jsonl_generation_and_gemini_status.py | 337 + tests/test_keycheck_durability_fixes.py | 792 + tests/test_kimi_provider.py | 272 + tests/test_legacy_tools_safety.py | 329 + tests/test_managed_files.py | 1904 + tests/test_migration_runtime_safety.py | 1082 + tests/test_model_probe_preferences.py | 72 + tests/test_multisource_execution_snapshot.py | 192 + tests/test_native_binding_stability.py | 90 + ...test_observer_only_coordinated_shutdown.py | 161 + tests/test_openai_discovery_coverage.py | 306 + tests/test_operations_control.py | 513 + tests/test_operations_schema.py | 167 + tests/test_operations_service.py | 1052 + tests/test_outbox_publication_lease.py | 189 + tests/test_owned_process.py | 226 + tests/test_owned_process_boundary.py | 241 + tests/test_owned_process_linux.py | 566 + tests/test_package_postman_harvest_limits.py | 245 + tests/test_pipeline_cutover_invariants.py | 820 + tests/test_pipeline_postgres_integration.py | 10452 ++++++ tests/test_postgres_empty_initialization.py | 881 + tests/test_postgres_runtime.py | 1402 + ...t_postgres_runtime_validated_high_fixes.py | 267 + tests/test_process_identity_linux.py | 35 + tests/test_production_query_shapes.py | 1005 + tests/test_remote_direct_credentials.py | 210 + tests/test_remote_worker_db.py | 1489 + tests/test_resource_lifecycle_fixes.py | 260 + tests/test_result_bundle_v2.py | 708 + tests/test_result_spool.py | 396 + tests/test_result_spool_backpressure.py | 647 + tests/test_result_spool_temp_recovery.py | 143 + tests/test_runtime_bootstrap_authority.py | 536 + tests/test_runtime_document.py | 788 + tests/test_runtime_document_io.py | 1612 + tests/test_runtime_safety_layer.py | 2367 ++ tests/test_runtime_security.py | 634 + tests/test_scan_execution.py | 904 + tests/test_scan_slot_release_reliability.py | 251 + tests/test_scanner_queue_high_fixes.py | 623 + tests/test_scanner_result_sink.py | 957 + tests/test_supervisor_foreground_shutdown.py | 913 + .../test_supervisor_managed_postgres_gate.py | 237 + tests/test_supervisor_safety.py | 2565 ++ tests/test_supervisor_startup_rollback.py | 456 + tests/test_synthetic_llm_pipeline.py | 351 + tests/test_temp_owner_child_safety.py | 307 + tests/test_validated_high_lifecycle_fixes.py | 370 + tests/test_validated_high_scanner_fixes.py | 3006 ++ tests/test_worker_api.py | 2184 ++ tests/test_worker_api_runtime.py | 476 + tests/test_worker_assignment.py | 1126 + tests/test_worker_assignment_runner.py | 1774 + tests/test_worker_cli.py | 724 + tests/test_worker_contracts.py | 431 + tests/test_worker_documentation.py | 60 + tests/test_worker_local_state.py | 773 + tests/test_worker_observability_db.py | 643 + tests/test_worker_package.py | 613 + tests/test_worker_runner_handoff_linux.py | 67 + tests/test_worker_supervisor.py | 1083 + tests/test_zai_provider.py | 180 + tests/worker_lifecycle_fixture_server.py | 46 + 498 files changed, 261563 insertions(+) create mode 100644 .dockerignore create mode 100644 .env.postgres.example create mode 100644 .gitattributes create mode 100644 .gitignore create mode 100644 AGENTS.md create mode 100644 DOCKER_MIGRATION.md create mode 100644 DOCKER_READINESS_AUDIT.md create mode 100644 Dockerfile create mode 100644 QUARANTINE_AUDIT.md create mode 100644 RUNTIME_CHEATSHEET.md create mode 100644 WINDOWS_IMPORT.md create mode 100644 WORKER_OPERATOR_EXPERIENCE_HANDOFF.md create mode 100644 WORKSPACE.md create mode 100644 app/.streamlit/config.toml create mode 100644 app/CHEATSHEET.md create mode 100644 app/DETECTOR_NOTES.md create mode 100644 app/JSONL_RECONCILIATION.md create mode 100644 app/KEYCHECKERS.md create mode 100644 app/admin_api.py create mode 100644 app/app.py create mode 100644 app/audit_github_tokens.py create mode 100644 app/capacity_model.py create mode 100644 app/child_bootstrap.py create mode 100644 app/config.linux.yaml create mode 100644 app/console_runner.py create mode 100644 app/container_import.py create mode 100644 app/container_import_config.py create mode 100644 app/container_projection_recovery.py create mode 100644 app/container_runtime.py create mode 100644 app/dashboard.py create mode 100644 app/db_backend.py create mode 100644 app/docker_depth_experiment.py create mode 100644 app/docker_depth_operator.py create mode 100644 app/docker_depth_report.py create mode 100644 app/docker_shadow.py create mode 100644 app/host_agent_apply.py create mode 100644 app/host_agent_client.py create mode 100644 app/host_agent_lifecycle.py create mode 100644 app/host_agent_protocol.py create mode 100644 app/host_agent_reconcile.py create mode 100644 app/host_agent_runtime.py create mode 100644 app/host_agent_server.py create mode 100644 app/host_agent_state.py create mode 100644 app/janitor.py create mode 100644 app/jsonl_projector.py create mode 100644 app/keycheck_accounting_smoke.py create mode 100644 app/keycheck_candidates.py create mode 100644 app/keycheck_runner.py create mode 100644 app/keycheckers/__init__.py create mode 100644 app/keycheckers/anthropic/anthropicKeycheck.py create mode 100644 app/keycheckers/aws/awsKeycheck.py create mode 100644 app/keycheckers/azure/azureKeycheck.py create mode 100644 app/keycheckers/deepseek/deepseekKeycheck.py create mode 100644 app/keycheckers/dockerhub/dockerhubKeycheck.py create mode 100644 app/keycheckers/gcp/gcpKeycheck.py create mode 100644 app/keycheckers/gemini/geminiKeycheck.py create mode 100644 app/keycheckers/github/githubKeycheck.py create mode 100644 app/keycheckers/gitlab/gitlabKeycheck.py create mode 100644 app/keycheckers/groq/groqKeycheck.py create mode 100644 app/keycheckers/huggingface/huggingfaceKeycheck.py create mode 100644 app/keycheckers/keycheck_common.py create mode 100644 app/keycheckers/kimi/kimiKeycheck.py create mode 100644 app/keycheckers/openai/Keycheck.py create mode 100644 app/keycheckers/openrouter/OpenrouterKeycheck.py create mode 100644 app/keycheckers/provider_resolution.py create mode 100644 app/keycheckers/provider_resolver/providerResolverKeycheck.py create mode 100644 app/keycheckers/qwen/qwenKeycheck.py create mode 100644 app/keycheckers/replicate/replicateKeycheck.py create mode 100644 app/keycheckers/xai/xaiKeycheck.py create mode 100644 app/keycheckers/zai/zaiKeycheck.py create mode 100644 app/lifecycle_authority.py create mode 100644 app/managed_files.py create mode 100644 app/migrate_layout.py create mode 100644 app/migrate_observability_db.py create mode 100644 app/migrate_runtime_safety.py create mode 100644 app/optimize_dashboard_db.py create mode 100644 app/owned_process.py create mode 100644 app/paths.py create mode 100644 app/postgres_runtime.py create mode 100644 app/process_identity.py create mode 100644 app/query_policy.py create mode 100644 app/remote_worker_bootstrap.py create mode 100644 app/remote_worker_client.py create mode 100644 app/requirements-keycheckers.txt create mode 100644 app/requirements.txt create mode 100644 app/result_bundle.py create mode 100644 app/result_ingester.py create mode 100644 app/result_spool.py create mode 100644 app/runtime_bootstrap.py create mode 100644 app/runtime_document.py create mode 100644 app/runtime_document_io.py create mode 100644 app/runtime_security.py create mode 100644 app/scan_execution.py create mode 100644 app/scan_manager.py create mode 100644 app/scanner.py create mode 100644 app/scanner_db.py create mode 100644 app/scanner_error_policy_smoke.py create mode 100644 app/supervisor.py create mode 100644 app/supervisor_instance.py create mode 100644 app/sync_alive_github_tokens.py create mode 100644 app/target_identity.py create mode 100644 app/trufflehog-custom-detectors.yaml create mode 100644 app/ui_components.py create mode 100644 app/worker_api.py create mode 100644 app/worker_assignment.py create mode 100644 app/worker_assignment_runner.py create mode 100644 app/worker_cli.py create mode 100644 app/worker_contracts.py create mode 100644 app/worker_local_state.py create mode 100644 app/worker_package.py create mode 100644 app/worker_package_builder.py create mode 100644 app/worker_supervisor.py create mode 100644 attach_runtime.ps1 create mode 100644 build_worker_release.cmd create mode 100644 build_worker_release.ps1 create mode 100644 cleanup_stale_agentui_vite.ps1 create mode 100644 compose.e2e.yaml create mode 100644 compose.edge.yaml create mode 100644 compose.shared-host.yaml create mode 100644 compose.snapshot-import.yaml create mode 100644 compose.windows-import.yaml create mode 100644 compose.yaml create mode 100644 deploy/capacity50/Dockerfile create mode 100644 deploy/capacity50/deploy.sh create mode 100644 deploy/capacity50/release_stopped_pipeline_leases.py create mode 100644 deploy/capacity50/render_config.py create mode 100644 deploy/edge/Caddyfile create mode 100644 deploy/edge/Caddyfile.shared-host create mode 100644 deploy/edge/Dockerfile create mode 100644 deploy/edge/README.md create mode 100644 deploy/edge/admin-denylist.caddy create mode 100644 deploy/edge/automatic-tls.caddy create mode 100644 deploy/edge/entrypoint.sh create mode 100644 deploy/edge/host-caddy-shared.caddy create mode 100644 deploy/fail2ban/Dockerfile.edge-e2e create mode 100644 deploy/fail2ban/action.d-truf-caddy-admin-denylist.conf create mode 100644 deploy/fail2ban/edge_e2e_docker_shim.py create mode 100644 deploy/fail2ban/fail2ban.d-edge-e2e.local create mode 100644 deploy/fail2ban/fail2ban.d-truf-persistence.local create mode 100644 deploy/fail2ban/filter.d-truf-admin-auth.conf create mode 100644 deploy/fail2ban/jail.d-truf-admin-auth.local create mode 100644 deploy/fail2ban/truf_caddy_admin_denylist.py create mode 100644 deploy/host-agent/truf-host-agent.conf create mode 100644 deploy/host-agent/truf-host-agent.service create mode 100644 deploy/host-agent/truf-host-agent.socket create mode 100644 deploy/host-agent/truf_host_agent.py create mode 100644 deploy/host-agent/truf_host_agent_install.py create mode 100644 deploy/systemd/truf-caddy-admin-denylist-expire.service create mode 100644 deploy/systemd/truf-caddy-admin-denylist-expire.timer create mode 100644 deploy/worker/compose.yaml create mode 100644 deploy/worker/workerctl.ps1 create mode 100644 deploy/worker/workerctl.sh create mode 100644 deploy_capacity50.ps1 create mode 100644 docker-compose.postgres.yml create mode 100644 docker/Dockerfile.edge-e2e create mode 100644 docker/build-dependencies/README.md create mode 100644 docker/build-dependencies/requirements.in create mode 100644 docker/build-dependencies/requirements.lock create mode 100644 docker/requirements-test.in create mode 100644 docker/requirements-test.lock create mode 100644 docker/requirements-worker.in create mode 100644 docker/requirements-worker.lock create mode 100644 docker/requirements.in create mode 100644 docker/requirements.lock create mode 100644 docker/test_verify.py create mode 100644 docker/test_windows_snapshot.py create mode 100644 docker/verify.py create mode 100644 docker/verify_edge_e2e.py create mode 100644 docker/verify_packaged_workers.py create mode 100644 docker/windows_snapshot.py create mode 100644 docker/worker-package-pins.json create mode 100644 docs/defect-dockerhub-discovery-retry-null-type-2026-09-25.md create mode 100644 docs/defect-terminal-status-scan-deadline-readback-2026-09-25.md create mode 100644 docs/defect-windows-scan-timestamps-utc-2026-09-25.md create mode 100644 docs/defect-worker-network-oserror-mislabeled-local-io-2026-09-25.md create mode 100644 docs/end-to-end-scanner-validation-2026-09-22.md create mode 100644 docs/end-to-end-scanner-validation.md create mode 100644 docs/extended-live-validation-2026-09-26.md create mode 100644 docs/remote-worker-cheatsheet-docker-ru.md create mode 100644 docs/remote-worker-cheatsheet-linux-ru.md create mode 100644 docs/remote-worker-cheatsheet-windows-ru.md create mode 100644 docs/remote-worker-operations.md create mode 100644 docs/remote-worker-quickstart-ru.md create mode 100644 docs/session-handoff/CURRENT_STATE.md create mode 100644 docs/session-handoff/DECISIONS.md create mode 100644 docs/session-handoff/EVIDENCE_INDEX.md create mode 100644 docs/session-handoff/README.md create mode 100644 docs/worker-operator-experience-live-trace-2026-09-25.md create mode 100644 docs/worker-operator-experience-validation-2026-09-24.md create mode 100644 docs/worker-parallelism-validation-2026-09-23.md create mode 100644 monitor_runtime_lag.ps1 create mode 100644 openspec/changes/add-adaptive-docker-payload-scanning/.openspec.yaml create mode 100644 openspec/changes/add-adaptive-docker-payload-scanning/design.md create mode 100644 openspec/changes/add-adaptive-docker-payload-scanning/proposal.md create mode 100644 openspec/changes/add-adaptive-docker-payload-scanning/specs/docker-layer-content-scanning/spec.md create mode 100644 openspec/changes/add-adaptive-docker-payload-scanning/tasks.md create mode 100644 openspec/changes/add-layer-aware-docker-scanning/.openspec.yaml create mode 100644 openspec/changes/add-layer-aware-docker-scanning/design.md create mode 100644 openspec/changes/add-layer-aware-docker-scanning/proposal.md create mode 100644 openspec/changes/add-layer-aware-docker-scanning/specs/docker-layer-content-scanning/spec.md create mode 100644 openspec/changes/add-layer-aware-docker-scanning/tasks.md create mode 100644 openspec/changes/add-minimal-remote-scan-workers/.openspec.yaml create mode 100644 openspec/changes/add-minimal-remote-scan-workers/baseline-evidence.md create mode 100644 openspec/changes/add-minimal-remote-scan-workers/design.md create mode 100644 openspec/changes/add-minimal-remote-scan-workers/proposal.md create mode 100644 openspec/changes/add-minimal-remote-scan-workers/specs/distributed-scan-workers/spec.md create mode 100644 openspec/changes/add-minimal-remote-scan-workers/specs/restricted-public-access/spec.md create mode 100644 openspec/changes/add-minimal-remote-scan-workers/tasks.md create mode 100644 openspec/changes/add-postman-source/.openspec.yaml create mode 100644 openspec/changes/add-postman-source/design.md create mode 100644 openspec/changes/add-postman-source/proposal.md create mode 100644 openspec/changes/add-postman-source/specs/postman-source/spec.md create mode 100644 openspec/changes/add-postman-source/tasks.md create mode 100644 openspec/changes/add-web-operations-control-plane/.openspec.yaml create mode 100644 openspec/changes/add-web-operations-control-plane/HANDOFF.md create mode 100644 openspec/changes/add-web-operations-control-plane/design.md create mode 100644 openspec/changes/add-web-operations-control-plane/proposal.md create mode 100644 openspec/changes/add-web-operations-control-plane/specs/distributed-core-source-processing/spec.md create mode 100644 openspec/changes/add-web-operations-control-plane/specs/managed-runtime-editing/spec.md create mode 100644 openspec/changes/add-web-operations-control-plane/specs/multisource-worker-assignments/spec.md create mode 100644 openspec/changes/add-web-operations-control-plane/specs/web-operations-console/spec.md create mode 100644 openspec/changes/add-web-operations-control-plane/tasks.md create mode 100644 openspec/changes/add-worker-operator-experience/.openspec.yaml create mode 100644 openspec/changes/add-worker-operator-experience/design.md create mode 100644 openspec/changes/add-worker-operator-experience/proposal.md create mode 100644 openspec/changes/add-worker-operator-experience/research.md create mode 100644 openspec/changes/add-worker-operator-experience/specs/worker-admin-experience/spec.md create mode 100644 openspec/changes/add-worker-operator-experience/specs/worker-diagnostics/spec.md create mode 100644 openspec/changes/add-worker-operator-experience/specs/worker-operator-supervisor/spec.md create mode 100644 openspec/changes/add-worker-operator-experience/specs/worker-progress-deadlines/spec.md create mode 100644 openspec/changes/add-worker-operator-experience/tasks.md create mode 100644 openspec/changes/cold-policy-stale-backlog/.openspec.yaml create mode 100644 openspec/changes/cold-policy-stale-backlog/design.md create mode 100644 openspec/changes/cold-policy-stale-backlog/proposal.md create mode 100644 openspec/changes/cold-policy-stale-backlog/specs/discovery-keyword-pruning/spec.md create mode 100644 openspec/changes/cold-policy-stale-backlog/specs/target-queue-policy-holds/spec.md create mode 100644 openspec/changes/cold-policy-stale-backlog/tasks.md create mode 100644 openspec/changes/expand-openai-ecosystem-discovery/.openspec.yaml create mode 100644 openspec/changes/expand-openai-ecosystem-discovery/design.md create mode 100644 openspec/changes/expand-openai-ecosystem-discovery/proposal.md create mode 100644 openspec/changes/expand-openai-ecosystem-discovery/specs/openai-ecosystem-discovery/spec.md create mode 100644 openspec/changes/expand-openai-ecosystem-discovery/tasks.md create mode 100644 openspec/changes/fix-custom-provider-detector-compatibility/.openspec.yaml create mode 100644 openspec/changes/fix-custom-provider-detector-compatibility/design.md create mode 100644 openspec/changes/fix-custom-provider-detector-compatibility/proposal.md create mode 100644 openspec/changes/fix-custom-provider-detector-compatibility/specs/custom-provider-detection-compatibility/spec.md create mode 100644 openspec/changes/fix-custom-provider-detector-compatibility/tasks.md create mode 100644 openspec/changes/fix-keycheck-accounting-visibility/.openspec.yaml create mode 100644 openspec/changes/fix-keycheck-accounting-visibility/design.md create mode 100644 openspec/changes/fix-keycheck-accounting-visibility/proposal.md create mode 100644 openspec/changes/fix-keycheck-accounting-visibility/specs/keycheck-accounting/spec.md create mode 100644 openspec/changes/fix-keycheck-accounting-visibility/tasks.md create mode 100644 openspec/changes/harden-dockerhub-search-pagination/.openspec.yaml create mode 100644 openspec/changes/harden-dockerhub-search-pagination/design.md create mode 100644 openspec/changes/harden-dockerhub-search-pagination/proposal.md create mode 100644 openspec/changes/harden-dockerhub-search-pagination/specs/dockerhub-search-pagination/spec.md create mode 100644 openspec/changes/harden-dockerhub-search-pagination/tasks.md create mode 100644 openspec/changes/improve-core-scan-coverage/.openspec.yaml create mode 100644 openspec/changes/improve-core-scan-coverage/design.md create mode 100644 openspec/changes/improve-core-scan-coverage/proposal.md create mode 100644 openspec/changes/improve-core-scan-coverage/specs/docker-layer-graph-selection/spec.md create mode 100644 openspec/changes/improve-core-scan-coverage/specs/git-ref-delta-scanning/spec.md create mode 100644 openspec/changes/improve-core-scan-coverage/tasks.md create mode 100644 openspec/changes/improve-worker-operator-cheatsheets/.openspec.yaml create mode 100644 openspec/changes/improve-worker-operator-cheatsheets/design.md create mode 100644 openspec/changes/improve-worker-operator-cheatsheets/proposal.md create mode 100644 openspec/changes/improve-worker-operator-cheatsheets/specs/worker-operator-lifecycle/spec.md create mode 100644 openspec/changes/improve-worker-operator-cheatsheets/tasks.md create mode 100644 openspec/changes/improve-worker-operator-cheatsheets/validation.md create mode 100644 openspec/changes/optimize-dockerhub-discovery-rotation/.openspec.yaml create mode 100644 openspec/changes/optimize-dockerhub-discovery-rotation/design.md create mode 100644 openspec/changes/optimize-dockerhub-discovery-rotation/proposal.md create mode 100644 openspec/changes/optimize-dockerhub-discovery-rotation/specs/dockerhub-incremental-discovery/spec.md create mode 100644 openspec/changes/optimize-dockerhub-discovery-rotation/tasks.md create mode 100644 openspec/changes/prune-low-yield-discovery-keywords/.openspec.yaml create mode 100644 openspec/changes/prune-low-yield-discovery-keywords/design.md create mode 100644 openspec/changes/prune-low-yield-discovery-keywords/proposal.md create mode 100644 openspec/changes/prune-low-yield-discovery-keywords/specs/discovery-keyword-pruning/spec.md create mode 100644 openspec/changes/prune-low-yield-discovery-keywords/specs/openai-discovery-coverage/spec.md create mode 100644 openspec/changes/prune-low-yield-discovery-keywords/tasks.md create mode 100644 openspec/changes/rescan-updated-core-targets/.openspec.yaml create mode 100644 openspec/changes/rescan-updated-core-targets/design.md create mode 100644 openspec/changes/rescan-updated-core-targets/proposal.md create mode 100644 openspec/changes/rescan-updated-core-targets/specs/updated-target-rescan/spec.md create mode 100644 openspec/changes/rescan-updated-core-targets/tasks.md create mode 100644 openspec/changes/resolve-ambiguous-key-providers/.openspec.yaml create mode 100644 openspec/changes/resolve-ambiguous-key-providers/design.md create mode 100644 openspec/changes/resolve-ambiguous-key-providers/proposal.md create mode 100644 openspec/changes/resolve-ambiguous-key-providers/specs/ambiguous-provider-resolution/spec.md create mode 100644 openspec/changes/resolve-ambiguous-key-providers/specs/zai-key-validation/spec.md create mode 100644 openspec/changes/resolve-ambiguous-key-providers/tasks.md create mode 100644 openspec/changes/restore-openai-discovery-coverage/.openspec.yaml create mode 100644 openspec/changes/restore-openai-discovery-coverage/design.md create mode 100644 openspec/changes/restore-openai-discovery-coverage/proposal.md create mode 100644 openspec/changes/restore-openai-discovery-coverage/specs/openai-discovery-coverage/spec.md create mode 100644 openspec/changes/restore-openai-discovery-coverage/tasks.md create mode 100644 openspec/changes/retire-zero-alive-keywords/.openspec.yaml create mode 100644 openspec/changes/retire-zero-alive-keywords/design.md create mode 100644 openspec/changes/retire-zero-alive-keywords/proposal.md create mode 100644 openspec/changes/retire-zero-alive-keywords/specs/discovery-keyword-pruning/spec.md create mode 100644 openspec/changes/retire-zero-alive-keywords/specs/target-queue-policy-holds/spec.md create mode 100644 openspec/changes/retire-zero-alive-keywords/tasks.md create mode 100644 openspec/changes/run-bounded-docker-depth-experiment/.openspec.yaml create mode 100644 openspec/changes/run-bounded-docker-depth-experiment/design.md create mode 100644 openspec/changes/run-bounded-docker-depth-experiment/proposal.md create mode 100644 openspec/changes/run-bounded-docker-depth-experiment/specs/docker-depth-experiment/spec.md create mode 100644 openspec/changes/run-bounded-docker-depth-experiment/tasks.md create mode 100644 openspec/changes/run-bounded-rank1-breadth-experiment/.openspec.yaml create mode 100644 openspec/changes/run-bounded-rank1-breadth-experiment/design.md create mode 100644 openspec/changes/run-bounded-rank1-breadth-experiment/proposal.md create mode 100644 openspec/changes/run-bounded-rank1-breadth-experiment/specs/docker-rank1-breadth-experiment/spec.md create mode 100644 openspec/changes/run-bounded-rank1-breadth-experiment/tasks.md create mode 100644 openspec/changes/simplify-dashboard/.openspec.yaml create mode 100644 openspec/changes/simplify-dashboard/design.md create mode 100644 openspec/changes/simplify-dashboard/proposal.md create mode 100644 openspec/changes/simplify-dashboard/specs/single-page-observability/spec.md create mode 100644 openspec/changes/simplify-dashboard/tasks.md create mode 100644 openspec/changes/stabilize-and-widen-scanner-coverage/.openspec.yaml create mode 100644 openspec/changes/stabilize-and-widen-scanner-coverage/design.md create mode 100644 openspec/changes/stabilize-and-widen-scanner-coverage/proposal.md create mode 100644 openspec/changes/stabilize-and-widen-scanner-coverage/specs/provider-key-validation/spec.md create mode 100644 openspec/changes/stabilize-and-widen-scanner-coverage/specs/scan-coverage-sizing/spec.md create mode 100644 openspec/changes/stabilize-and-widen-scanner-coverage/specs/scanner-runtime-stability/spec.md create mode 100644 openspec/changes/stabilize-and-widen-scanner-coverage/specs/source-discovery-targeting/spec.md create mode 100644 openspec/changes/stabilize-and-widen-scanner-coverage/tasks.md create mode 100644 openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/.openspec.yaml create mode 100644 openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/design.md create mode 100644 openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/proposal.md create mode 100644 openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/specs/docker-scan-lifecycle/spec.md create mode 100644 openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/tasks.md create mode 100644 openspec/changes/stabilize-gitlab-discovery-retries/.openspec.yaml create mode 100644 openspec/changes/stabilize-gitlab-discovery-retries/design.md create mode 100644 openspec/changes/stabilize-gitlab-discovery-retries/proposal.md create mode 100644 openspec/changes/stabilize-gitlab-discovery-retries/specs/gitlab-discovery-resilience/spec.md create mode 100644 openspec/changes/stabilize-gitlab-discovery-retries/tasks.md create mode 100644 openspec/changes/stabilize-gitlab-trufflehog-lifecycle/.openspec.yaml create mode 100644 openspec/changes/stabilize-gitlab-trufflehog-lifecycle/design.md create mode 100644 openspec/changes/stabilize-gitlab-trufflehog-lifecycle/proposal.md create mode 100644 openspec/changes/stabilize-gitlab-trufflehog-lifecycle/specs/gitlab-scan-lifecycle/spec.md create mode 100644 openspec/changes/stabilize-gitlab-trufflehog-lifecycle/tasks.md create mode 100644 openspec/changes/support-fifty-remote-assignments/.openspec.yaml create mode 100644 openspec/changes/support-fifty-remote-assignments/design.md create mode 100644 openspec/changes/support-fifty-remote-assignments/proposal.md create mode 100644 openspec/changes/support-fifty-remote-assignments/specs/remote-assignment-capacity/spec.md create mode 100644 openspec/changes/support-fifty-remote-assignments/tasks.md create mode 100644 openspec/changes/support-fifty-remote-assignments/validation.md create mode 100644 openspec/config.yaml create mode 100644 openspec/parking-lot.md create mode 100644 pytest.ini create mode 100644 start_core_runtime.ps1 create mode 100644 start_freeze_counters.ps1 create mode 100644 start_runtime.ps1 create mode 100644 stop_runtime.ps1 create mode 100644 tests/container_e2e.py create mode 100644 tests/container_import_stop_e2e.py create mode 100644 tests/container_projection_recovery_e2e.py create mode 100644 tests/container_unit.py create mode 100644 tests/edge_e2e_backend.py create mode 100644 tests/edge_e2e_client.py create mode 100644 tests/fixtures/worker_tls_cert.pem create mode 100644 tests/fixtures/worker_tls_key.pem create mode 100644 tests/owned_process_helper.py create mode 100644 tests/packaged_worker_e2e_server.py create mode 100644 tests/parity_helpers.py create mode 100644 tests/requirements-browser.txt create mode 100644 tests/test_admin_api.py create mode 100644 tests/test_admin_browser.py create mode 100644 tests/test_ambiguous_provider_resolution.py create mode 100644 tests/test_api_deadline_plumbing.py create mode 100644 tests/test_api_proxy_routing.py create mode 100644 tests/test_artifact_lifecycle_slots.py create mode 100644 tests/test_bounded_state_high_fixes.py create mode 100644 tests/test_capacity50_deploy.py create mode 100644 tests/test_container_e2e_helpers.py create mode 100644 tests/test_container_import.py create mode 100644 tests/test_container_import_config.py create mode 100644 tests/test_container_migration_paths.py create mode 100644 tests/test_container_projection_recovery.py create mode 100644 tests/test_container_provider_portability.py create mode 100644 tests/test_container_runtime.py create mode 100644 tests/test_container_security.py create mode 100644 tests/test_custom_provider_detector_compatibility.py create mode 100644 tests/test_dashboard_behavior.py create mode 100644 tests/test_dashboard_secret_guard.py create mode 100644 tests/test_db_backend_safety.py create mode 100644 tests/test_direct_entrypoint_bytecode_policy.py create mode 100644 tests/test_discovery_only_cycle.py create mode 100644 tests/test_discovery_producer_supervisor.py create mode 100644 tests/test_discovery_request_budgets.py create mode 100644 tests/test_distributed_core_profile.py create mode 100644 tests/test_docker_codec_recovery.py create mode 100644 tests/test_docker_coverage_diagnostics.py create mode 100644 tests/test_docker_depth_cohort_holds.py create mode 100644 tests/test_docker_depth_experiment.py create mode 100644 tests/test_docker_depth_experiment_schema.py create mode 100644 tests/test_docker_depth_operator.py create mode 100644 tests/test_docker_depth_report.py create mode 100644 tests/test_docker_depth_resolver.py create mode 100644 tests/test_docker_depth_resolver_refund.py create mode 100644 tests/test_docker_discovery_provenance.py create mode 100644 tests/test_docker_foundation.py create mode 100644 tests/test_docker_layer_scanning.py create mode 100644 tests/test_docker_producer_cleanup.py create mode 100644 tests/test_docker_recovery_diagnostics.py create mode 100644 tests/test_docker_shadow_coverage.py create mode 100644 tests/test_docker_staging_bounds.py create mode 100644 tests/test_dockerhub_incremental_discovery.py create mode 100644 tests/test_dockerhub_search_pagination.py create mode 100644 tests/test_edge_deployment.py create mode 100644 tests/test_entrypoint_authority.py create mode 100644 tests/test_exact_git_scan_planning.py create mode 100644 tests/test_finding_pipeline_high_fixes.py create mode 100644 tests/test_gcp_keycheck_security.py create mode 100644 tests/test_gcp_rsa_adc_hardening.py create mode 100644 tests/test_gemini_legacy_migration_durability.py create mode 100644 tests/test_gharchive_bounds.py create mode 100644 tests/test_git_checkout_recovery.py create mode 100644 tests/test_git_clone_authority.py create mode 100644 tests/test_git_diagnostic_coverage.py create mode 100644 tests/test_gitlab_resilience_fixes.py create mode 100644 tests/test_high_authority_bootstrap_fixes.py create mode 100644 tests/test_high_only_round6.py create mode 100644 tests/test_high_only_round7.py create mode 100644 tests/test_host_agent_apply.py create mode 100644 tests/test_host_agent_deploy.py create mode 100644 tests/test_host_agent_lifecycle.py create mode 100644 tests/test_host_agent_linux.py create mode 100644 tests/test_host_agent_protocol.py create mode 100644 tests/test_host_agent_reconcile.py create mode 100644 tests/test_host_agent_runtime.py create mode 100644 tests/test_host_agent_state.py create mode 100644 tests/test_huggingface_long_paths.py create mode 100644 tests/test_import_suffix_manifest.py create mode 100644 tests/test_janitor_bounded.py create mode 100644 tests/test_jsonl_generation_and_gemini_status.py create mode 100644 tests/test_keycheck_durability_fixes.py create mode 100644 tests/test_kimi_provider.py create mode 100644 tests/test_legacy_tools_safety.py create mode 100644 tests/test_managed_files.py create mode 100644 tests/test_migration_runtime_safety.py create mode 100644 tests/test_model_probe_preferences.py create mode 100644 tests/test_multisource_execution_snapshot.py create mode 100644 tests/test_native_binding_stability.py create mode 100644 tests/test_observer_only_coordinated_shutdown.py create mode 100644 tests/test_openai_discovery_coverage.py create mode 100644 tests/test_operations_control.py create mode 100644 tests/test_operations_schema.py create mode 100644 tests/test_operations_service.py create mode 100644 tests/test_outbox_publication_lease.py create mode 100644 tests/test_owned_process.py create mode 100644 tests/test_owned_process_boundary.py create mode 100644 tests/test_owned_process_linux.py create mode 100644 tests/test_package_postman_harvest_limits.py create mode 100644 tests/test_pipeline_cutover_invariants.py create mode 100644 tests/test_pipeline_postgres_integration.py create mode 100644 tests/test_postgres_empty_initialization.py create mode 100644 tests/test_postgres_runtime.py create mode 100644 tests/test_postgres_runtime_validated_high_fixes.py create mode 100644 tests/test_process_identity_linux.py create mode 100644 tests/test_production_query_shapes.py create mode 100644 tests/test_remote_direct_credentials.py create mode 100644 tests/test_remote_worker_db.py create mode 100644 tests/test_resource_lifecycle_fixes.py create mode 100644 tests/test_result_bundle_v2.py create mode 100644 tests/test_result_spool.py create mode 100644 tests/test_result_spool_backpressure.py create mode 100644 tests/test_result_spool_temp_recovery.py create mode 100644 tests/test_runtime_bootstrap_authority.py create mode 100644 tests/test_runtime_document.py create mode 100644 tests/test_runtime_document_io.py create mode 100644 tests/test_runtime_safety_layer.py create mode 100644 tests/test_runtime_security.py create mode 100644 tests/test_scan_execution.py create mode 100644 tests/test_scan_slot_release_reliability.py create mode 100644 tests/test_scanner_queue_high_fixes.py create mode 100644 tests/test_scanner_result_sink.py create mode 100644 tests/test_supervisor_foreground_shutdown.py create mode 100644 tests/test_supervisor_managed_postgres_gate.py create mode 100644 tests/test_supervisor_safety.py create mode 100644 tests/test_supervisor_startup_rollback.py create mode 100644 tests/test_synthetic_llm_pipeline.py create mode 100644 tests/test_temp_owner_child_safety.py create mode 100644 tests/test_validated_high_lifecycle_fixes.py create mode 100644 tests/test_validated_high_scanner_fixes.py create mode 100644 tests/test_worker_api.py create mode 100644 tests/test_worker_api_runtime.py create mode 100644 tests/test_worker_assignment.py create mode 100644 tests/test_worker_assignment_runner.py create mode 100644 tests/test_worker_cli.py create mode 100644 tests/test_worker_contracts.py create mode 100644 tests/test_worker_documentation.py create mode 100644 tests/test_worker_local_state.py create mode 100644 tests/test_worker_observability_db.py create mode 100644 tests/test_worker_package.py create mode 100644 tests/test_worker_runner_handoff_linux.py create mode 100644 tests/test_worker_supervisor.py create mode 100644 tests/test_zai_provider.py create mode 100644 tests/worker_lifecycle_fixture_server.py diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..59882a7 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,237 @@ +# Deny by default. File-only exceptions keep credentials and caches out of the context. +** +!app/app.py +!app/audit_github_tokens.py +!app/capacity_model.py +!app/child_bootstrap.py +!app/console_runner.py +!app/container_import.py +!app/container_import_config.py +!app/container_projection_recovery.py +!app/container_runtime.py +!app/dashboard.py +!app/db_backend.py +!app/admin_api.py +!app/docker_depth_experiment.py +!app/docker_depth_operator.py +!app/docker_depth_report.py +!app/docker_shadow.py +!app/host_agent_client.py +!app/host_agent_apply.py +!app/host_agent_lifecycle.py +!app/host_agent_protocol.py +!app/host_agent_reconcile.py +!app/host_agent_runtime.py +!app/host_agent_server.py +!app/host_agent_state.py +!app/janitor.py +!app/jsonl_projector.py +!app/keycheck_accounting_smoke.py +!app/keycheck_candidates.py +!app/keycheck_runner.py +!app/keycheckers/__init__.py +!app/keycheckers/keycheck_common.py +!app/keycheckers/provider_resolution.py +!app/keycheckers/anthropic/anthropicKeycheck.py +!app/keycheckers/aws/awsKeycheck.py +!app/keycheckers/azure/azureKeycheck.py +!app/keycheckers/deepseek/deepseekKeycheck.py +!app/keycheckers/dockerhub/dockerhubKeycheck.py +!app/keycheckers/gcp/gcpKeycheck.py +!app/keycheckers/gemini/geminiKeycheck.py +!app/keycheckers/github/githubKeycheck.py +!app/keycheckers/gitlab/gitlabKeycheck.py +!app/keycheckers/groq/groqKeycheck.py +!app/keycheckers/huggingface/huggingfaceKeycheck.py +!app/keycheckers/kimi/kimiKeycheck.py +!app/keycheckers/openai/Keycheck.py +!app/keycheckers/openrouter/OpenrouterKeycheck.py +!app/keycheckers/provider_resolver/providerResolverKeycheck.py +!app/keycheckers/qwen/qwenKeycheck.py +!app/keycheckers/replicate/replicateKeycheck.py +!app/keycheckers/xai/xaiKeycheck.py +!app/keycheckers/zai/zaiKeycheck.py +!app/lifecycle_authority.py +!app/managed_files.py +!app/migrate_layout.py +!app/migrate_observability_db.py +!app/migrate_runtime_safety.py +!app/optimize_dashboard_db.py +!app/owned_process.py +!app/paths.py +!app/postgres_runtime.py +!app/process_identity.py +!app/query_policy.py +!app/result_bundle.py +!app/result_ingester.py +!app/result_spool.py +!app/runtime_bootstrap.py +!app/runtime_document.py +!app/runtime_document_io.py +!app/runtime_security.py +!app/scan_manager.py +!app/scanner.py +!app/scan_execution.py +!app/scanner_db.py +!app/scanner_error_policy_smoke.py +!app/supervisor.py +!app/supervisor_instance.py +!app/sync_alive_github_tokens.py +!app/target_identity.py +!app/ui_components.py +!app/worker_api.py +!app/worker_assignment.py +!app/worker_assignment_runner.py +!app/worker_cli.py +!app/worker_contracts.py +!app/worker_local_state.py +!app/worker_package.py +!app/worker_package_builder.py +!app/worker_supervisor.py +!app/remote_worker_bootstrap.py +!app/remote_worker_client.py +!app/requirements.txt +!app/requirements-keycheckers.txt +!app/config.linux.yaml +!app/trufflehog-custom-detectors.yaml +!app/.streamlit/config.toml +!start_runtime.ps1 +!start_core_runtime.ps1 +!stop_runtime.ps1 +!Dockerfile +!.dockerignore +!compose.yaml +!compose.edge.yaml +!compose.shared-host.yaml +!docs/remote-worker-quickstart-ru.md +!docs/remote-worker-cheatsheet-windows-ru.md +!docs/remote-worker-cheatsheet-linux-ru.md +!docs/remote-worker-cheatsheet-docker-ru.md +!deploy/edge/Dockerfile +!deploy/edge/Caddyfile +!deploy/edge/Caddyfile.shared-host +!deploy/edge/host-caddy-shared.caddy +!deploy/edge/entrypoint.sh +!deploy/edge/README.md +!deploy/edge/admin-denylist.caddy +!deploy/edge/automatic-tls.caddy +!deploy/fail2ban/truf_caddy_admin_denylist.py +!deploy/fail2ban/Dockerfile.edge-e2e +!deploy/fail2ban/edge_e2e_docker_shim.py +!deploy/fail2ban/fail2ban.d-edge-e2e.local +!deploy/fail2ban/filter.d-truf-admin-auth.conf +!deploy/fail2ban/jail.d-truf-admin-auth.local +!deploy/fail2ban/action.d-truf-caddy-admin-denylist.conf +!deploy/fail2ban/fail2ban.d-truf-persistence.local +!deploy/host-agent/truf_host_agent.py +!deploy/host-agent/truf_host_agent_install.py +!deploy/host-agent/truf-host-agent.conf +!deploy/host-agent/truf-host-agent.service +!deploy/host-agent/truf-host-agent.socket +!deploy/systemd/truf-caddy-admin-denylist-expire.service +!deploy/systemd/truf-caddy-admin-denylist-expire.timer +!docker/requirements.in +!docker/requirements.lock +!docker/Dockerfile.edge-e2e +!docker/requirements-test.in +!docker/requirements-test.lock +!docker/requirements-worker.in +!docker/requirements-worker.lock +!docker/worker-package-pins.json +!docker/verify.py +!docker/build-dependencies/requirements.in +!docker/build-dependencies/requirements.lock +!tests/container_unit.py +!tests/container_e2e.py +!tests/container_import_stop_e2e.py +!tests/container_projection_recovery_e2e.py +!tests/owned_process_helper.py +!tests/parity_helpers.py +!tests/packaged_worker_e2e_server.py +!tests/edge_e2e_backend.py +!tests/edge_e2e_client.py +!tests/test_synthetic_llm_pipeline.py +!tests/test_docker_foundation.py +!tests/test_db_backend_safety.py +!tests/test_owned_process.py +!tests/test_owned_process_linux.py +!tests/test_owned_process_boundary.py +!tests/test_runtime_bootstrap_authority.py +!tests/test_runtime_document.py +!tests/test_runtime_document_io.py +!tests/test_managed_files.py +!tests/test_operations_schema.py +!tests/test_operations_control.py +!tests/test_host_agent_protocol.py +!tests/test_host_agent_linux.py +!tests/test_host_agent_apply.py +!tests/test_host_agent_deploy.py +!tests/test_host_agent_lifecycle.py +!tests/test_host_agent_reconcile.py +!tests/test_host_agent_runtime.py +!tests/test_host_agent_state.py +!tests/test_supervisor_foreground_shutdown.py +!tests/test_supervisor_startup_rollback.py +!tests/test_supervisor_managed_postgres_gate.py +!tests/test_observer_only_coordinated_shutdown.py +!tests/test_supervisor_safety.py +!tests/test_discovery_producer_supervisor.py +!tests/test_distributed_core_profile.py +!tests/test_discovery_only_cycle.py +!tests/test_discovery_request_budgets.py +!tests/test_operations_service.py +!tests/test_postgres_runtime.py +!tests/test_container_security.py +!tests/test_runtime_security.py +!tests/test_postgres_empty_initialization.py +!tests/test_container_migration_paths.py +!tests/test_container_provider_portability.py +!tests/test_container_e2e_helpers.py +!tests/test_container_runtime.py +!tests/test_container_import.py +!tests/test_container_import_config.py +!tests/test_container_projection_recovery.py +!tests/test_result_bundle_v2.py +!tests/test_pipeline_cutover_invariants.py +!tests/test_custom_provider_detector_compatibility.py +!tests/test_scan_execution.py +!tests/test_worker_api.py +!tests/test_worker_api_runtime.py +!tests/test_worker_assignment.py +!tests/test_worker_cli.py +!tests/test_worker_contracts.py +!tests/test_worker_local_state.py +!tests/test_worker_package.py +!tests/test_worker_supervisor.py +!tests/test_multisource_execution_snapshot.py +!tests/test_remote_direct_credentials.py +!tests/test_docker_staging_bounds.py +!tests/test_remote_worker_db.py +!tests/test_admin_api.py +!tests/test_edge_deployment.py +!tests/fixtures/worker_tls_cert.pem +!tests/fixtures/worker_tls_key.pem + +# Defense in depth if the allowlist is expanded later. +**/.env* +**/secrets* +**/credentials* +**/*service-account* +**/__pycache__/ +**/.venv/ +**/venv/ +**/node_modules/ +**/*.py[cod] +**/*.db* +**/*.sqlite* +**/*.jsonl* +**/*.log* +.git/ +.opencode/ +runtime/ +tmp/ +state/ +logs/ +data/ +docker/imports/ +truf-cluster-authority-*/ diff --git a/.env.postgres.example b/.env.postgres.example new file mode 100644 index 0000000..17235c3 --- /dev/null +++ b/.env.postgres.example @@ -0,0 +1,14 @@ +TRUF_POSTGRES_DB=truf +TRUF_POSTGRES_USER=truf +TRUF_POSTGRES_PASSWORD=change-me-long-random-password +TRUF_POSTGRES_PORT=5432 + +# Optional tuning overrides. +TRUF_POSTGRES_MAX_CONNECTIONS=100 +TRUF_POSTGRES_SHARED_BUFFERS=512MB +TRUF_POSTGRES_EFFECTIVE_CACHE_SIZE=2GB +TRUF_POSTGRES_CHECKPOINT_TIMEOUT=15min +TRUF_POSTGRES_LOG_MIN_DURATION_STATEMENT=2000 + +# App DSN template. Keep the real password out of config snapshots/logs. +SCANNER_DB_URL=postgresql://truf:change-me-long-random-password@127.0.0.1:5432/truf diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..9d8659c --- /dev/null +++ b/.gitattributes @@ -0,0 +1,12 @@ +* text=auto +*.py text eol=lf +*.sh text eol=lf +*.yaml text eol=lf +*.yml text eol=lf +*.toml text eol=lf +*.md text eol=lf +*.txt text eol=lf +Dockerfile text eol=lf +.dockerignore text eol=lf +.gitignore text eol=lf +.gitattributes text eol=lf diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..1c079d8 --- /dev/null +++ b/.gitignore @@ -0,0 +1,44 @@ +# Local credentials and provider input pools. +.env* +!.env*.example +secrets.yaml* +secrets.*.yaml +!secrets.example.yaml +credentials*.json +*-service-account*.json +/app/keycheckers/**/*.txt + +# Runtime data, findings, queues, logs, and cluster authority. +/runtime/ +/tmp/ +/state/ +/logs/ +/docker/test-results/ +/docker/imports/ +/data/ +/truf-cluster-authority-*/ +/checked_*.txt +/todo_*.txt +/runner_state.json +*.db +*.db-* +*.sqlite +*.sqlite-* +*.sqlite3 +*.sqlite3-* +*.jsonl +*.jsonl.* +*.log +*.log.* + +# Reproducible dependencies and generated files. +__pycache__/ +*.py[cod] +.pytest_cache/ +.venv/ +venv/ +node_modules/ +build/ +dist/ +.coverage +htmlcov/ diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..d197b89 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,15 @@ +# Engineering Decisions + +## Provider Source Execution + +- Keep source adapters minimal. The server validates assignment shape, canonical target identity, and the immutable identity required by the protocol. +- Git planning may bind an exact commit. Docker planning may resolve a mutable tag to an immutable digest. These are identity operations, not provider-access proofs. +- The worker is the final authority for real provider access. It reports success, a permanent target failure, or a retryable provider failure; the server settles or retries from that result. +- Discovery credentials are not assignment fields unless a separately approved capability explicitly defines credential delivery. +- Do not add per-target server preflight requests, durable public-access proofs, proof TTL/freshness columns, access-evidence migrations, broad child-environment credential scrubbing, credential sandboxes, or post-hoc redaction pipelines by default. +- Before adding any such defensive or security-specific mechanism, stop and obtain explicit user approval. Record the approved behavior in an OpenSpec requirement and task before implementation. +- Do not treat existing defensive code as precedent for duplicating the same machinery for another source. +- The worker machine's ambient environment belongs to its operator. Assignment code must not silently rewrite HOME, XDG, Git, Docker, or provider environments merely to enforce a nominally tokenless assignment. +- Prefer direct, bounded provider-error classification over preventive infrastructure: authentication/access/not-found failures are permanent when target-scoped; rate limits, network failures, and provider 5xx responses are retryable. + +These rules apply to future source adapters and to changes in `worker_assignment.py`, `scan_execution.py`, `remote_worker_client.py`, `scanner.py`, `scanner_db.py`, and discovery producers. diff --git a/DOCKER_MIGRATION.md b/DOCKER_MIGRATION.md new file mode 100644 index 0000000..95bc408 --- /dev/null +++ b/DOCKER_MIGRATION.md @@ -0,0 +1,418 @@ +# Docker Development Copy + +Status: runnable Linux container deployment with a passed fresh offline E2E. +Production source/provider parity and original-data migration are not proven. +Original Windows installation: `D:\truf`. Development copy: `D:\truf-docker`. +Only the development copy is changed; the original `D:\truf` and storage on `S:` +remain untouched. No original STOP or START was performed for this migration. +Docker/Compose installation in WSL was user-approved and has occurred. Earlier +statements about unavailable Docker, no installations, and no images describe +historical stages, not the current deployment. + +The current acceptance evidence is from the isolated Docker runs on 2026-09-15. +The historical original-runtime lineage proof was not rerun. + +## Copy And Local History + +These are historical source-copy/baseline records, not a fresh Git inventory. + +- The user changed the initial full-snapshot request to a source-only copy. +- The retained source selection was 295 files, approximately 6.82 MiB before Git and migration edits. +- Application Python, tests, configuration, documentation, OpenSpec artifacts, and project scripts were retained. +- Databases, WAL/SHM files, results, queues, logs, runtime state, provider input pools, real secrets, native Windows bundles, and dependency caches were omitted or removed from the copy. +- The partial external-data copy `D:\truf-docker-data` was removed. Original external storage on `S:` was not modified. +- SHA-256 comparison verified the selected source files before migration edits. The clone's `.gitignore` was strengthened before staging. +- Initial local commit: `1b3c7fc`, `chore: snapshot source for Docker migration`, 292 tracked files on `main`. +- Three copied OpenCode package metadata files remain ignored by their original nested ignore policy. +- No remote, push, Git configuration change, or second commit was made. Migration edits remain separate from the baseline. + +## Safety Boundary + +Copied root PowerShell tools and direct canonical Python runtime/control CLI +launches retain their staging refusals. The supported container path is the +image's `tini -> python3 -u -I -S -B app/container_runtime.py` entrypoint, which +uses authenticated bootstrap dispatch rather than removing host safeguards. +Do not use copied Windows maintenance/import scripts to operate this deployment. + +The original `app/config.yaml` is preserved in the initial Git commit and in the +unchanged original installation. In this working tree it was renamed to +`app/config.linux.yaml`. The initial slice changed filesystem/deployment paths +only; subsequent container contracts fix private storage and control paths. +There is no automatically selected `app/config.yaml`; the container entrypoint +defaults explicitly to `/opt/truf/app/config.linux.yaml`. This production profile +is distinct from the verifier's deliberately narrowed `/data/config/e2e.yaml`. + +The image requires read-only application storage, UID/GID 10001 for runtime +commands, a private native Linux named volume at `/data`, and private tmpfs at +`/run/truf`. Provisioning alone runs as root, with only CHOWN, DAC_OVERRIDE, and +FOWNER added to the dropped capability set. Private application/data directories +are mode 0700 and files 0600. System executables remain root-owned and immutable; +they must not be chowned to the application user to satisfy private-file checks. + +PostgreSQL 16 runs under the supervisor in the same container, with generated +private credentials, loopback connectivity, and exact cluster identity binding. +External PostgreSQL authority is not implemented; changing a DSN or starting a +separate PostgreSQL service does not implement that backend. Never reuse physical +Windows PGDATA on Linux. Any original-data migration requires separately +approved logical export/import and validation. No host data bind mount, Docker +socket, privileged mode, or Docker-in-Docker is required for DockerHub scanning. + +## Implemented Foundation + +- Path defaults derive from this checkout instead of `D:\truf` or the current working directory. +- Generated filesystem templates use portable separators. Explicit YAML still overrides environment defaults. +- POSIX path resolution rejects Windows drive, UNC, device, and backslash syntax rather than silently joining it to a Linux directory. +- Generic path defaults retain the runtime-relative result-bundle directory and PostgreSQL's `runtime/postgres/data` suffix; the container profile explicitly selects the separate `/data` paths listed below. +- The Windows TruffleHog fallback is checked only on Windows. Managed PostgreSQL DSN precedence is unchanged. +- `.gitignore` excludes credentials and consumables. `.dockerignore` denies everything except reviewed, explicitly named build/source files, including nested provider modules. +- `.gitattributes` specifies LF for Linux-facing source/configuration files without changing global Git settings. +- Offline tests cover path behavior, the Linux profile, context allowlist, and refusal of copied control entrypoints. + +## Implemented Lifecycle Changes + +The lifecycle changes in `app/owned_process.py` and `app/supervisor.py` implement +Linux ownership and coordinated foreground shutdown. Native Linux ownership +tests and the fresh container E2E now pass within their selected scope; this is +not acceptance of every production source/provider or unsafe failure scenario. + +- Linux startup now reports complete kernel-derived identities, verifies the host and payload sessions, and requires a verified child subreaper. Unreadable or already-exited executables fail startup rather than receiving a PID-only identity. +- Nested observers have independent sessions. Cleanup signals the still-pinned payload group before reaping its leader, then drains adopted children. Completed adopted observers are also reaped while the root remains alive. +- Payload signal status is reproduced only after cleanup, including SIGKILL as a real negative subprocess return code. Administrative stop remains a distinct nonzero result. +- Linux failed-start cleanup retains its observer through repeated interruptions until exit is confirmed; it does not hard-kill that observer on a timer. +- A configured owner sends an explicit stop byte over its retained pipe, so another inherited writer cannot suppress cancellation. Forked proxy finalization only detaches its local descriptor, without signalling the owner's job or taking an inherited mutex. +- Startup status descriptors have a single atomic owner. An unclaimed reader is cancelled before waiting for host cleanup, avoiding double close, reused-FD reads, and a blocked status writer during interrupted thread startup. +- Windows keeps its Job Object containment and resource checks. Its isolated host now uses the base interpreter rather than a virtual-environment redirecting launcher, so the retained process and acknowledged identity have the same PID. +- The supervisor's POSIX SIGTERM callback only latches a shutdown request. Locked checkpoints close admission before activation or further starts, and preserve coordinated child-before-PostgreSQL teardown. +- Metadata publication and shutdown interruptions retain closed gates and unsafe authority in `FAILED_HOLD`. Partial activation uses full coordinated cleanup; failures remain nonzero even if a later cleanup retry succeeds. +- Foreground and background shutdown publish the same instance-bound receipt after fallible control/log cleanup. The waiting authenticated stopper or locked stale reconciliation removes metadata; foreground no longer deletes it before the stopper can verify completion. +- PostgreSQL stop and close share one remaining timeout. This is not a bound on the complete shutdown sequence: unsafe authority is retained indefinitely rather than released when a timer expires. + +Additional container work is implemented, rather than still pending: + +- Source-start rollback retains uncertain owners and closes admission; locked stale-metadata reconciliation uses exact identity rather than PID alone. +- Durable state stays on `/data`; recoverable control metadata is under `/run/truf/control` and does not survive recreation. +- Linux manifests distinguish private application files from root-owned native binaries; the obsolete Windows OpenRouter PowerShell dependency is not the Linux provider entrypoint. +- `Dockerfile` packages Python 3.12.14, PostgreSQL 16.15, Git, tini, and hash-locked Python dependencies in the production server. TruffleHog 3.97.4 is pinned only in worker and test targets; the remote-only production server does not contain it. +- Fresh provisioning, empty-cluster initialization, 27 schema migrations, final cutover, and initialization/identity markers are implemented. Partial initialization fails closed and is not automatically repaired or adopted. +- Compose applies a read-only root filesystem, dropped capabilities, no-new-privileges, 2 CPUs, 6 GiB memory, 512 PIDs, 256 MiB shared memory, and bounded log rotation. Tini forwards SIGTERM to the foreground runtime/supervisor, not indiscriminately to its process group. +- Dependency-aware health checks require authenticated ACTIVE control, READY PostgreSQL, required workers and durable pipeline leases, schema/cutover validity, the expected PG16 data directory, and writable nonfull persistent storage. + +Containment covers managed nested `OwnedProcess` trees, not arbitrary session +escapes or an observer independently killed by SIGKILL/OOM. The application +retains unsafe authority indefinitely in `FAILED_HOLD`, but Compose's stop grace +is only 10 minutes. Docker can then force termination; that is unsafe shutdown, +not successful coordinated cleanup. Two clean E2E stops do not prove safe +termination of `FAILED_HOLD` at that deadline. + +## Current Linux Paths + +These are the implemented image, volume, and tmpfs contracts. + +| Purpose | Path | +| --- | --- | +| Image application root | `/opt/truf` | +| Application code | `/opt/truf/app` | +| New Linux runtime state | `/data/runtime-linux` | +| Separately initialized Linux PostgreSQL cluster | `/data/postgres-linux` | +| Durable result bundles | `/data/scanner-result-bundles` | +| Scanner scratch | `/data/scanner-work` | +| Private imported provider credentials | `/data/config/secrets.yaml` | +| Generated PostgreSQL password | `/data/postgres-password` | +| Ephemeral control and authority state | `/run/truf/control`, `/run/truf/authority` | +| Worker/test TruffleHog executable | `/usr/local/bin/trufflehog` (absent from the production server) | + +The original physical cluster was PostgreSQL 16, but it was not retained in this +source-only copy. Do not mount Windows PostgreSQL data into a Linux server. +Initialize isolated development data; any later production-data migration needs +a separately approved logical export/import and validation procedure. + +## Development Linux/WSL Procedure + +This volume-only procedure is retained for isolated development and migration verification; +it is not the production edge installation procedure. Production operators must use +`deploy/edge/README.md`, including the fixed host-agent installer, active documents under +`/etc/truf/runtime`, immutable package manifests under `/etc/truf/worker-packages`, the +host-agent socket, and the combined base plus edge Compose invocation. + +The following are operator commands, not commands executed by this documentation +update. Use a Linux shell or WSL with Python 3, Git, Docker's Linux daemon, and +the Compose plugin (`docker compose`). Verified host versions were Docker 29.8.0 +and Compose 5.5.1. Keep Docker-managed named volumes on native Linux storage, not NTFS +or an original-runtime directory. Build steps need package/download network +access; the offline test runs below do not. Do not run the full legacy suite. + +### Build + +From the development checkout, define an explicitly scoped Compose helper. The +example uses the passwordless sudo Docker access used by the recorded verifier; +omit `sudo -n` if your account already has direct daemon access. Do not change +daemon permissions or install/reset WSL as part of these instructions. + +```sh +cd /mnt/d/truf-docker +dc() { sudo -n docker compose --project-name truf-docker --project-directory "$PWD" --env-file /dev/null --file compose.yaml "$@"; } +dc --profile test build runtime test +``` + +On native Linux, substitute the development checkout path for `/mnt/d/truf-docker`. +This creates `truf-local:runtime` and `truf-local:test`. All lifecycle commands +below must retain this project name and checkout so they use the same volume. +The explicit env file avoids implicitly loading a checkout `.env`; do not supply +unreviewed Docker/Compose environment overrides or proxy credentials. + +### Provision And Initialize + +Use these one-off commands only while the runtime is stopped. `--no-deps` avoids +implicitly starting other services. Provision is network-disabled, generates a +new private database password, and seeds an empty provider-secret mapping. A +valid already-provisioned layout is checked without regenerating credentials; +nonempty or partially provisioned layouts are refused. + +```sh +dc run --rm --no-deps --pull never -T provision +``` + +For an authorized production-profile run, import an existing private YAML mapping +from stdin. Replace the placeholder filename with an approved Linux-side secret +file, not a file in the original installation. Do not put secret values in command +arguments, the image, the checkout, or this document. This is not a Compose secret +mount: the locked, atomic import writes `/data/config/secrets.yaml` with private +ownership/mode and refuses an active runtime or PostgreSQL PID file. Imports can +also be repeated after confirmed shutdown for credential rotation. + +```sh +dc run --rm --no-deps --pull never -T runtime import-secrets < /absolute/private/provider-secrets.yaml +dc run --rm --no-deps --pull never -T runtime initialize +``` + +Initialization creates an independent PG16 cluster, starts maintenance mode, +applies base/schema migrations and final cutover, confirms PostgreSQL stopped, +then publishes the initialization marker. A matching initialized volume is not +reinitialized. On partial initialization, stop and inspect offline; do not delete +markers, change identity, or rerun repair scripts to force admission. + +### Start, Status, And Health + +Starting `compose.yaml` uses the production configuration and can launch enabled +sources and provider workers with real network access. It is NOT the offline E2E +procedure and must only be used with separately authorized targets/credentials. +`run` initializes if needed before execing the noninteractive autostart supervisor; +the explicit initialization step above makes that first-install phase visible. + +```sh +dc up --detach --no-deps --no-build --pull never runtime +dc ps runtime +dc exec -T runtime /usr/local/bin/python3 -I -S -B /opt/truf/app/container_runtime.py status +dc exec -T runtime /usr/local/bin/python3 -I -S -B /opt/truf/app/container_runtime.py health +``` + +`status` and `health` both call the same readiness function and return JSON on +success, nonzero on failure. They are not general stopped-runtime inventory +commands. Run them with `exec` in the existing runtime, not `compose run`, because +a new container has a different control tmpfs. During initial startup, readiness +can fail until activation and worker leases complete; Compose checks every 30 +seconds with a 15-second timeout, 240-second start period, and three retries. +Readiness is not proof that every provider works or that production load is safe. + +### Stop And Recreate + +```sh +dc stop --timeout 600 runtime +cid=$(dc ps --all --quiet runtime) +sudo -n docker container inspect --format 'status={{.State.Status}} exit={{.State.ExitCode}} oom={{.State.OOMKilled}} restarts={{.RestartCount}}' "$cid" +``` + +Require an exited container, exit code 0, and OOM false; investigate unexpected +restarts. A successful `compose stop` invocation alone is not graceful-exit proof. +Do not shorten the timeout, force-kill, remove the data volume, or treat a +10-minute forced termination as safe. Runtime restart policy is `on-failure:3`; +it is not a substitute for investigating uncertain ownership or partial state. + +To replace a confirmed-stopped container while retaining its existing data: + +```sh +dc up --detach --no-deps --no-build --pull never --force-recreate runtime +dc exec -T runtime /usr/local/bin/python3 -I -S -B /opt/truf/app/container_runtime.py health +``` + +Allow readiness to complete again. Never use `down --volumes` on data you intend +to retain. The old `docker-compose.postgres.yml` is a noncanonical manual-recovery +fixture, not a deployment or an external-authority implementation. + +### Offline Verification + +Use the reviewed container selection, not unrestricted pytest discovery: + +```sh +dc --profile test run --rm --no-deps --pull never -T test +python3 -I -S -B docker/test_verify.py +python3 -I -S -B docker/verify.py +``` + +The selected container suite runs without `/data` or provider credentials, with +network disabled and temporary fixtures in `/tmp`; native local child processes +and loopback control sockets are intentional. Only this unit-test service allows +execution from its 512 MiB `/tmp` for temporary venv and askpass fixtures. Both +the production and E2E services use the same 128 MiB `noexec,nosuid,nodev` `/tmp`. +`docker/test_verify.py` is a separate six-test stdlib regression suite, passing +on both Windows and WSL Linux; +its total is not silently added to the selected container-suite count. + +`docker/verify.py` requires already-built local runtime/test images and an already +Git-ignored `docker/test-results/latest.json`. It performs no builds or pulls and +does not edit ignore files. Run the verifier itself as the normal Linux user; it +tries direct Docker access, then `sudo -n docker`. Git is used for the read-only +evidence ignore guard, not for mutations. + +The verifier invokes only `compose.e2e.yaml`, generates a unique `truf-e2e-*` +project, pins the local images by ID, and uses fresh private `data` and `tools` +named volumes. All services have network disabled, proxy settings cleared, no +host data binds or published ports, and no automatic restarts. It preserves the +production runtime image/entrypoint but prepares a narrowed offline configuration: +a local Git fixture and real TruffleHog with verification disabled, then the real +scan/bundle/ingestion/projection pipeline and OpenAI worker. ONLY that worker's +HTTP transport is substituted with a synthetic 401 response; no live provider +request is made. Production source selection/provider parity is not tested by +this profile. Do not manually merge it into production Compose or rerun prepare +after recreation. + +The verifier checks healthy activation, pipeline lineage, one synthetic HTTP +request, graceful stop, recreation on the same volume, unchanged persisted +counts/hashes, no duplicates and no second HTTP request, health, and a second +graceful stop. Its aggregate check budget is 3600 seconds, each health wait at +most 240 seconds, each stop 600 seconds, and failure handling has a separate +720-second budget. Nonzero, forced, or OOM exits cannot pass. + +On success it removes only its ownership-verified containers and volumes; use +`python3 -I -S -B docker/verify.py --keep` instead to retain stopped test artifacts. +Failures retain artifacts after a guarded stop attempt, possibly with running +containers if ownership or stopping cannot be proven. Record the printed project +name for investigation; do not use broad prune/down/kill commands. Evidence is +written to `docker/test-results/latest.json` as counts, hashes, image IDs, statuses, +and durations without raw command logs or secret values. + +## Current Verified Evidence + +The fresh recorded E2E in `docker/test-results/latest.json` has `status.result` +and `status.checks` both `passed`, `status.cleanup` equal to `removed`, and zero +owned containers, networks, or volumes remaining. The final run on 2026-09-15 +took 54.374 seconds and repeated the earlier successful fresh-volume run. + +- Fresh PG16 initialization and all 27 migrations completed with identity/cutover checks; both authenticated health checks passed. +- Real local Git and native TruffleHog produced one finding, one scan, one result bundle/reservation, and one queue attempt through ingestion and projection, with zero pipeline quarantine/errors. +- The real OpenAI worker used only the offline synthetic 401 transport: one first HTTP request, one linked keycheck result/current state, two projection jobs, and three projection appends. +- Recreation used a new container on the same volume without rerunning prepare. Persisted artifact IDs, migration/cutover hashes, SQL summary, output bytes and projection ledger hashes matched; repeated keycheck made zero HTTP requests and introduced no duplicate rows/appends. +- Both graceful stops recorded container exit 0, OOM false, and zero restarts. Both separate read-only stopped-volume checks confirmed private PG storage, initialization, and absence of `postmaster.pid`. +- Shutdown receipt publication is inferred ONLY from the foreground zero-exit contract. The receipt itself was not read after stop because `/run/truf` tmpfs had disappeared; evidence labels this `exit_contract_only_tmpfs_removed`. +- Final selected container regression suite: 578 passed, seven Windows-only tests skipped on Linux, in 22.25 seconds. This includes all six fresh-volume regressions. Separately, `docker/test_verify.py` passed all six tests on both Windows and WSL Linux. These are scope-specific counts, not counts stored in the E2E JSON. +- All 157 Python files under `app`, `tests`, and `docker` parsed successfully. `git diff --check` passed; existing PowerShell LF/CRLF notices were not whitespace failures. No changes were staged or committed and no Git remote was added. + +Recorded image IDs (not a promise that mutable local tags still point to them): + +- Runtime: `sha256:92502a2581ebcabe79ddc28744dabcfb3000b99b7c0c5262c4c09aff9dc0267b`. +- Test: `sha256:4ce11325728ba3e58e6643c1c8e800f317179d5c7c50e7e80568b58f62dbdfd0`. + +## Remaining Limitations + +`DOCKER_READINESS_AUDIT.md` records the original Windows audit. Its historical +line references and pending foundation tasks are not a current implementation +checklist. The remaining acceptance boundaries are: + +1. `FAILED_HOLD` can outlive Docker's 10-minute grace and be terminated unsafely. No general crash/OOM/forced-stop recovery proof is claimed. +2. Production source/provider parity, live providers, real credentials, broad discovery, sustained load, throughput, and resource sizing have not been validated by the offline E2E. +3. There is no arm64 build/runtime proof, even though TruffleHog has an arm64 checksum entry. +4. External PostgreSQL authority is not implemented. Only a fresh supervisor-owned native Linux PG16 cluster is supported here. +5. There is no original Windows database migration, original-data equivalence, or production cutover proof. The historical read-only lineage record below is not such a migration proof and was not executed by this update. + +## Historical Verification (Superseded Status) + +The sections below preserve earlier scoped verification records. Their test +counts, file inventories, no-install/no-image statements, staging status, and +then-open container checks apply only to those stages and are superseded by the +current evidence above. At the earlier lifecycle stage Docker CLI was unavailable +and no image build/container/database migration had yet been performed. That is +no longer the current state. No historical original-runtime operation below is +claimed as an execution by this documentation update or by the fresh offline E2E. + +### WSL Ownership Verification + +Verified on 2026-09-14 in `Ubuntu-24.04`, WSL version 2, Linux kernel +`6.18.33.2-microsoft-standard-WSL2`, CPython 3.12.3, as unprivileged UID 1000: + +- The initial successful startup took approximately 44 seconds, exceeding the earlier 10-second probe limit. WSL reported automatic NAT-to-VirtioProxy networking fallback. No network setting was changed, and the tests needed no network access. +- All nine previously skipped `LinuxOwnedProcessIntegrationTests` first passed on the real kernel. They were then included in the expanded run: 38 tests passed, zero failures/errors/skips, in 3.682 seconds excluding WSL startup. +- The expanded selection also covers mocked failure paths, ordinary output/timeout behavior, static process-safety checks, isolated-host startup, an offline temporary virtual environment, and synthetic credential filtering. +- The first expanded run exposed a test-fixture issue: Ubuntu's standard-library `sitecustomize.py` shadowed the virtual environment's fixture in the positive control. The test now puts its own module first through a temporary `PYTHONPATH`, supplied to both control and isolated-host scenarios. Both startup-hook markers must appear in the control and remain absent for the isolated host. No application code was changed for this correction. +- The existing Windows selection was rerun after that test-only fix: 159 passed, nine Linux-only skips. Those nine skips are covered by the successful WSL run, not left untested. +- The Linux runner used only the standard-library `unittest`, `python3 -I -S -B`, an empty inherited environment via `env -i`, and explicit safe locale/path/temp settings. No packages were installed and no pytest plugins were loaded. It asserted the effective `tempfile` directory before collection; a 90-second test watchdog was separate from the longer WSL startup allowance. +- Code was read from `/mnt/d/truf-docker`; `HOME`, `TMPDIR`, `TEMP`, and `TMP` were confined to `/mnt/c/Users/PRO100~1/AppData/Local/Temp/opencode`. Fixtures used mounted Windows storage, not a new native Linux data volume. This does not validate Linux storage ownership, permissions, or container mounts. +- No canonical runtime CLI, original database, provider, Docker service, or copied recovery Compose fixture was launched. Staging refusals remain unchanged. + +Exact expanded `unittest` selection, with the clone's `tests` directory explicitly +added to the isolated runner's module search path: + +- `test_owned_process_linux` +- `test_owned_process.OwnedProcessTests` +- `test_owned_process.StaticProcessSafetyTests` +- `test_owned_process_boundary.OwnedProcessHostBoundaryTests` +- `test_owned_process_boundary.CredentialBoundaryTests.test_host_environment_strips_mixed_case_database_credentials_only` + +### Earlier Windows Lifecycle Verification + +Verified on 2026-09-14 with Windows CPython 3.12.3 after the lifecycle changes: + +- 159 selected tests passed; nine native Linux tests were skipped because they require a real Linux kernel and `/proc`. The passing selection includes native Windows Job/virtual-environment checks, mocked Linux kernel operations, mocked supervisor lifecycle scenarios, and the previous foundation checks. +- Regressions cover interrupted status-reader startup, cancellation before host reaping, duplicate control writers, foreign-proxy detachment, sticky failure results, receipt verification, shutdown ordering, and graceful TERM between activation/start checkpoints. +- The nine Linux-only fixtures cover live kernel identities and sessions, transitive cleanup on normal exit and actual pipe EOF, explicit stop with another writer retained, forked-proxy finalization, live adopted-child reaping, SIGKILL/SIGTERM status, and rejected startup after nested children are ready. Test signals use pidfds bound to the acknowledged payload identity. +- All 145 application/test Python files parsed successfully. The source-only artifact check found 301 files excluding `.git`, with no credential pools, databases, results, runtime/cache directories, or links. +- `git diff --check` passed. Seven existing PowerShell LF/CRLF warnings are not whitespace failures. The index and remote list remain empty; the only commit is still `1b3c7fc`. +- Independent scoped static reviews were followed by regression fixes and reruns. The last ownership review found no remaining concrete issue in the reviewed corrections; this is not a Linux conformance result. +- At that stage, a bounded WSL probe timed out without output; the subsequent WSL investigation and successful tests are recorded above. No WSL reset/install, image build, application launch, database access, or live provider test was performed by that lifecycle continuation. + +The combined selection was limited to these modules/node IDs: + +- `tests/test_owned_process_linux.py` +- `tests/test_owned_process.py` +- `tests/test_owned_process_boundary.py::OwnedProcessHostBoundaryTests` +- `tests/test_owned_process_boundary.py::CredentialBoundaryTests::test_host_environment_strips_mixed_case_database_credentials_only` +- `tests/test_supervisor_foreground_shutdown.py` +- `tests/test_supervisor_managed_postgres_gate.py` +- `tests/test_observer_only_coordinated_shutdown.py` +- `tests/test_docker_foundation.py` +- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_default_data_directory_is_unchanged` +- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_external_data_directory_expands_without_moving_runtime_assets` +- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_external_data_identity_mismatch_remains_fail_closed` +- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_external_data_identity_verifies_only_when_exactly_bound` +- `tests/test_supervisor_safety.py::ManagedConfigurationAuthorityTests::test_managed_dsn_overrides_config_database_urls` + +The test child used `python -X utf8 -B`, pytest `-q --tb=short -rs`, +`-p no:cacheprovider -o addopts= --confcutdir=tests`, disabled plugin autoload, +and empty `PYTHONPATH`, `PYTEST_ADDOPTS`, and `PYTEST_PLUGINS`. Runtime/DSN +overrides were removed only from that child's environment. All of `TMPDIR`, +`TEMP`, and `TMP` pointed at the approved temporary work area, and the runner +asserted the actual `tempfile` directory before collecting tests. + +### Previously Recorded Verification + +The following earlier results are retained as historical records. They are not +new original-runtime operations performed by this lifecycle continuation. + +Recorded on 2026-09-14 with Windows CPython 3.12.3: + +- 19 targeted offline tests passed: all 14 foundation tests, four existing PostgreSQL path/identity tests, and the existing managed-DSN precedence test. +- The existing external-data path fixture now uses portable separators too; intentional Windows-path rejection remains separately covered. +- Python CLI tests verify the complete literal refusal AST before invoking an isolated interpreter. PowerShell safety checks parse scripts without executing them, so a broken refusal cannot make the test control the host. +- All 143 Python application/test files parsed successfully; no syntax errors. +- Parsed YAML comparison against the initial Git commit found exactly 31 changed deployment-path fields; all other settings were identical. +- The final source tree contained 299 files, approximately 6.85 MiB excluding `.git`, with no real credentials/pools, databases, results, runtime/cache directories, or links found by the artifact check. +- The original supervisor and PostgreSQL were still running; no copy writers remained. There were no staged changes or Git remotes, and the only commit remained the initial baseline. +- A read-only original-runtime lineage proof joined one completed queue reservation through its bundle, scan, finding, keycheck candidate/result, projection jobs, appends, and stream generations. Exact event/hash relationships held, and both scan projections plus the keycheck projection matched their recorded byte offsets, lengths, record counts, and SHA-256 digests; no target or credential value was emitted. + +Tests ran with bytecode writes, pytest plugin autoload, and pytest cache disabled; +application/database environment overrides were removed from the test process. +Temporary test files were confined to the approved temporary work area. The +PowerShell review and Windows POSIX-path emulation do not prove Linux container +behavior, and the context allowlist check is not an actual Docker build. + +Do not run the entire existing test suite against this machine: it includes +native-process, socket, database, and live integration scenarios. diff --git a/DOCKER_READINESS_AUDIT.md b/DOCKER_READINESS_AUDIT.md new file mode 100644 index 0000000..716c3a1 --- /dev/null +++ b/DOCKER_READINESS_AUDIT.md @@ -0,0 +1,364 @@ +# Docker Readiness Audit + +Date: 2026-09-14. Scope: the application and runtime launch chain in `D:\truf`. + +This is an inspection report, not an implementation. Application code, configuration, secrets, databases and runtime data were not changed. The target assumed here is a Linux container. Windows containers, target CPU architecture, deployment host and resource budget have not been specified. + +## Verdict + +**The project is not ready to containerize unchanged. Adding a Dockerfile around the PowerShell launchers is insufficient.** There are both packaging gaps and concrete defects in the POSIX process/security paths. PostgreSQL is also part of a local process-ownership protocol, not just a replaceable connection URL. + +The first deployment should retain **one supervisor and its authenticated children in one container, with one runtime replica**. PostgreSQL can initially remain locally managed in that container, or become a separate service after an explicit external-database authority mode is implemented. Neither option is currently a configuration-only change. + +No deployment Dockerfile or `.dockerignore` was found in the inspected application/root. `docker-compose.postgres.yml` is explicitly a **noncanonical manual recovery fixture**, not the production runtime definition. DockerHub scanning in the application is unrelated to deployment packaging. + +Priority definitions: + +- **P0 / B01-B14:** resolve before a working, safely restartable Linux deployment. Some items need packaging/provisioning rather than application changes. +- **P1 / R01-R07:** resolve before unattended operation with persistent data. +- **Conditional / C01-C07:** required only for the stated feature or deployment choice. These are not all prerequisites for a headless, single-runtime deployment. + +## Runtime Map + +| Component | Actual entry points and role | +| --- | --- | +| Canonical startup | `app/runtime_bootstrap.py`, `app/child_bootstrap.py`; isolated interpreter startup and authenticated imports | +| Lifecycle | `app/supervisor.py`, `app/supervisor_instance.py`, `app/lifecycle_authority.py`; admission, ownership, control, manifests and shutdown | +| Subprocess containment | `app/owned_process.py`, `app/process_identity.py` | +| Scanning | `app/console_runner.py`, `app/scanner.py`; native TruffleHog and Git | +| Key checking | `app/keycheck_runner.py`, `app/keycheckers/`; Python provider processes, HTTP and AWS SDK | +| Database | `app/postgres_runtime.py`, `app/db_backend.py`, `app/scanner_db.py`; managed PostgreSQL and normalized runtime schema | +| Result pipeline | `app/result_bundle.py`, `app/result_ingester.py`, `app/jsonl_projector.py`, `app/janitor.py` | +| Optional UI | `app/dashboard.py`; supervised read-only Streamlit dashboard | +| Retired entry points | `app/app.py:1-13` and `app/scan_manager.py:8`; do not use as the container application | + +## P0: Deployment Blockers + +### B01. Replace Windows Path Assumptions, Not Just Environment Variables + +**Evidence:** `app/config.yaml:8-12,30-37,70-89,109-120,140-150`; `app/paths.py:7-8,17-18,46-59,70-110`. + +The configuration contains `D:\truf`, `S:\postgres-data`, `S:\scanner-result-bundles`, `S:\scanner-work`, `C:\Tools\trufflehog.exe` and backslash-based derived paths. POSIX treats a Windows drive path as relative and a backslash as an ordinary filename character. `os.path.normpath()` does not translate them. + +YAML `root_dir` wins over `SCANNER_ROOT_DIR`/`SCANNER_PROJECT_ROOT`; YAML `trufflehog_path` wins over `TRUFFLEHOG_PATH`. Several defaults remain Windows-specific even if those YAML values are removed. The managed PostgreSQL DSN has its own precedence and must remain consistent with the selected authority mode. + +**Isolated reproduction:** executing the actual path functions with POSIX path semantics, a config location under `/opt/truf/app`, and Linux environment overrides produced `/opt/truf/app/D:\truf` for the root and `/opt/truf/app/C:\Tools\trufflehog.exe` for TruffleHog. With empty YAML, the root became portable but the default log path still became `/srv/truf/runtime\logs`. No application was imported or started. + +**Correction:** supply a complete Linux config/profile and make path defaults platform-aware with joins or portable separators. Cover global paths, supervisor instance/status/lock paths, policy assets, caches and maintenance paths. Define and document precedence rather than assuming environment variables override YAML. Preserve the working Windows profile; do not replace backslashes indiscriminately in arbitrary settings or stored data. + +### B02. Use the Canonical Foreground Entrypoint + +**Evidence:** `start_runtime.ps1:18`; `start_core_runtime.ps1:23`; `app/runtime_bootstrap.py:24-31,115-129`; `app/supervisor.py:4468-4481,4878-4882,5000-5001,5277-5278`. + +The launchers use `--background`, return after starting a child, and therefore have the wrong lifetime for a container entrypoint. Interactive mode is not automatically disabled without a TTY; EOF can end its loop. Noninteractive mode without autostart is not sufficient either. + +**Correction:** use exec-style startup of the canonical supervisor, without daemonization, with explicit `--non-interactive --autostart`. Keep `-I -S -B` and the bootstrap entrypoint binding. Use `--no-dashboard` for the initial headless deployment. Do not launch workers directly or substitute `streamlit run app.py`. + +Illustrative command contract for the embedded-PostgreSQL option, **only after the other fixes and offline provisioning**; not executed during this audit: + +```text +python3 -u -I -S -B /opt/truf/app/runtime_bootstrap.py supervisor -- --runtime-bootstrap-entrypoint /opt/truf/app/supervisor.py --config /opt/truf/app/config.yaml --non-interactive --autostart --no-dashboard --no-clear --with-postgres +``` + +Select sources explicitly. `start_core_runtime.ps1` and `app/config.linux.yaml` use the exact distributed producer set `gitlab,dockerhub,huggingface`; operational workers such as `keychecks` are configured independently. `--once` is not a guarantee that the whole supervised pipeline is a terminating batch job (`app/supervisor.py:974`). + +### B03. Fix the POSIX OwnedProcess Identity Handshake + +**Evidence:** `app/owned_process.py:1006-1014`; `app/scanner.py:11479-11495,1825-1842`. + +The POSIX handshake returns payload/host identities without `creation_time`; the payload executable is taken from command text. The scanner passes that identity into its ownership marker, which requires `pid`, `creation_time` and `executable`. Access to the missing field can raise `KeyError` on the real native-scanner launch path. + +**Correction:** return complete, canonical, verified process identities before startup acknowledgement. Preserve the stdlib-only containment bootstrap and exact identity checks; a PID alone is insufficient. Add a real POSIX handshake-to-marker test, not a mock that supplies the missing field. + +### B04. Make Nested POSIX Process Containment Actually Contain the Tree + +**Evidence:** `app/owned_process.py:442-445,917-921,989-993`; `app/supervisor.py:1180-1189`; `app/keycheck_runner.py:2226-2233`; `app/scanner.py:11479`. + +An outer payload owns process group A. Its inner containment host inherits A, but the inner provider/native payload creates session/group B. Killing A can kill the inner host while leaving B running. The dead host can no longer reliably process its parent pipe and stop B. This affects source restarts and dependency-loss handling inside a still-running container, not only final container termination. + +**Correction:** implement nested containment with verified tree termination, such as isolated observer hosts with cascading stop/acknowledgement, or an appropriately designed cgroup mechanism. Do not replace identity-based ownership with name-based process killing. An init/reaper alone does not fix this defect. + +### B05. Integrate Container Signals and a Real Shutdown Budget + +**Evidence:** `app/supervisor.py:2338-2361,3447,3551,5293-5305`; `app/postgres_runtime.py:824`; `app/config.yaml:168-170,185`. + +The supervisor handles `KeyboardInterrupt` but does not register a SIGTERM handler. Docker's normal stop signal therefore is not wired into coordinated shutdown; PID 1 also has special Linux signal semantics. The shutdown path includes admission closure, pipeline draining, sequential child stops and PostgreSQL shutdown. A short container grace period can interrupt that protocol. Locally managed PostgreSQL is daemonized through `pg_ctl`, so reaping also needs attention. + +**Correction:** connect SIGTERM to the existing STOPPING/shutdown event flow, provide an init/reaper, and forward signals to the supervisor rather than indiscriminately to its whole process tree. Size `stop_grace_period` from the total measured shutdown deadline, not only the PostgreSQL timeout. Current configuration includes a 120-second PostgreSQL shutdown timeout and a 180-second background-shutdown timeout; neither proves that a particular total container grace is sufficient. + +An explicit SIGINT stop signal could be an interim tested workaround, not a substitute for the complete TERM/PID1/nested-process fix. An unconfirmed stop intentionally enters `FAILED_HOLD`; no finite grace period can guarantee a clean outcome there. Preserve authority, expose failure and require an escalation procedure instead of releasing locks optimistically. + +### B06. Make Restart Safe Across Container PID Reuse + +**Evidence:** `app/supervisor.py:5067`; related identity and instance handling in `app/supervisor_instance.py` and `app/process_identity.py`. + +Persisted, otherwise valid instance metadata combined with a reused PID can block a new foreground supervisor. PID reuse is particularly predictable across fresh container PID namespaces. + +**Correction:** reconcile stale instance state using exact identity under the correct authority lock, and/or place runtime-instance/control metadata in deliberately ephemeral storage. Keep persistent business data separate from per-instance PID, control, shutdown-receipt and session state. Never delete a lock or metadata merely because it is old. Verify that the previous runtime cannot still own the database before recovery. + +### B07. Provision Non-Root Ownership, Private Paths and a Usable Lock Root + +**Evidence:** `app/runtime_security.py:391-400,651-685,842-889,989-998`; `app/postgres_runtime.py:258-271`. + +POSIX private-file policy requires ownership by the effective UID and no group/other permissions. Lifecycle preflight is read-only and requires configured directories to exist already, including application/root paths, runtime data, bundle subdirectories and PostgreSQL paths. A fresh named volume or a default root-owned Docker secret does not automatically meet this contract. The PostgreSQL process inherits the runtime UID; Linux PostgreSQL cannot run as root. + +The authority lock root is hardcoded to `/var/lock/truf`. `/var/lock` is a symlink on many Linux images and conflicts with the no-symlink policy; creating it as an unprivileged user is another problem. + +**Correction:** choose a stable non-root UID/GID, provision all required paths and file ownership before normal startup, and use a real prepared private authority directory, for example `/run/truf/authority`. Make its location explicit instead of depending on a distribution's `/var/lock` layout. Data/config modes normally need owner-only access. Verify the actual behavior of named volumes, secret mounts and Docker Desktop mounts; do not solve this with `chmod 777` or privileged mode. + +Provisioning may need a separate controlled initialization step. The steady-state application should not require root, and startup should retain its fail-closed validation rather than silently repairing arbitrary mounted data. + +### B08. Separate Executable Trust Policy From Data-File Hardening + +**Evidence:** `app/runtime_security.py:651-685`; `app/lifecycle_authority.py:432-443,636-639`; `app/migrate_runtime_safety.py:2502-2503,2508-2521`. + +The current hardener sets every POSIX file to `0600`, removing native executable bits. Conversely, ordinary system-installed `0755`, root-owned Git/TruffleHog executables do not satisfy the current exact-private manifest policy. The offline hardener also hardens each file's parent and runtime trees; pointing it at a system binary can attempt to harden a shared system directory. + +**Correction:** define and verify executable permissions separately, preserving `x` and a trusted owner. Choose deliberately between private executable copies and an explicit immutable-system-binary trust policy. Account for the complete Git installation and PostgreSQL libraries/helpers, not just one binary. Do not run the existing recursive hardener over `/usr/bin` or blindly apply `0600` to a native runtime. Test hardening idempotency without breaking execution. + +### B09. Package the Code Authority and Isolated Import Layout Correctly + +**Evidence:** `app/lifecycle_authority.py:21,38-70,378-443`; `app/runtime_bootstrap.py:47-83`; `app/child_bootstrap.py:177-195`. + +The manifest unconditionally includes `../runtime/check-openrouter-keys.ps1`, `../start_runtime.ps1` and `../stop_runtime.ps1`. Excluding all PowerShell or all `runtime/` content before changing this contract can break authentication even on Linux. Application-tree symlinks and cached application bytecode are rejected. Creating an ordinary virtualenv under the application tree can introduce both. + +**Correction:** make the external authority-file set OS-aware, or retain these inert first-party files at the required relative locations until that change is made. Ship all required application modules and detector/policy assets, without application `.pyc`/`__pycache__` artifacts or symlinked application paths. Keep the dependency environment outside the inspected `app/` tree. Install dependencies for the exact interpreter used by isolated bootstrap; arbitrary `PYTHONPATH` and user-site packages are not a substitute. + +Use immutable releases with a full controlled restart. Live edits to mounted code/config are incompatible with manifest drift detection (`app/supervisor.py:3045`). + +### B10. Decide and Implement the PostgreSQL Authority Topology + +**Evidence:** `app/postgres_runtime.py:112-130,249-255,642-671,1335`; `app/db_backend.py:42-101`; `docker-compose.postgres.yml:1-20`. + +Current managed mode expects loopback, local PostgreSQL executables, a local data directory, bound cluster identity and an inspectable local postmaster process. Changing the DSN host to a Compose service name does not implement external PostgreSQL support. The URL parser also rejects query parameters, so appending `?sslmode=...` is not currently a supported TLS configuration route. + +| Option | Required work | +| --- | --- | +| Locally managed PostgreSQL in the runtime container | Preserve a shared PID/network namespace and lifecycle owner. Supply Linux PostgreSQL executables at the currently fixed `runtime/postgres/pgsql/bin` layout, or make the binary paths configurable. Keep PGDATA separate from binaries and use the compatible non-root UID. Include init/reaping and coordinated database shutdown. | +| Separate PostgreSQL container/service | Add an explicit external authority/backend mode across `postgres_runtime.py`, DSN validation, supervisor lifecycle and readiness. It must not require local postmaster PIDs/data paths/binaries or attempt local start/stop. Preserve authenticated endpoint/cluster identity checks, fencing and schema readiness. Define explicit TLS settings if required. | + +Simply disabling authority, process or endpoint checks is not an acceptable implementation. A pre-existing PostgreSQL instance can be observed without being owned; do not assume the supervisor will stop it. `maintenance-start` returns after startup and is not a PostgreSQL container service entrypoint (`app/postgres_runtime.py:1463`). + +Keep `docker-compose.postgres.yml` separate: it is profile-gated, uses `restart: no`, Windows bind paths and a deliberately noncanonical endpoint. Its `postgres:16` image is not evidence of the version required by the authoritative cluster. Do not silently promote this recovery database to production authority. + +### B11. Add an Explicit Offline Provisioning and Data-Migration Procedure + +**Evidence:** `app/postgres_runtime.py:321-356,400`; `app/scanner_db.py:238,5145-5177,20520-20530`; `app/migrate_runtime_safety.py:2971-3008`. + +`bootstrap_cluster_identity()` does not run `initdb`; it expects an existing cluster and executable set. It rejects supervisor metadata, `postmaster.pid` and a listening endpoint. The identity binds paths, binaries and cluster identity, so an old Windows identity file must not be reused as a Linux authority binding. + +Workers also require the runtime safety schema and a valid `postgres-normalized-v2-authority` final-cutover marker with evidence. A fresh PostgreSQL service reporting `pg_isready` is not an application-ready database. + +**Correction:** distinguish two offline phases. First, initialize/restore and bind the embedded cluster identity while the target PostgreSQL server is stopped. Then run database schema/cutover work with PostgreSQL available in controlled maintenance mode but all normal sources/pipeline workers stopped. Use `--initialize-base` for a genuinely fresh installation, not as a substitute for understanding an existing dataset. Complete the applicable normalization, projection reconciliation and final-cutover checks before admitting workers. + +Determine the real source/target PostgreSQL versions before transfer. Prefer a planned logical dump/restore for the Windows-to-Linux move unless another backup method is explicitly validated as compatible; do not assume copying Windows PGDATA works. Back up and transfer matching result bundles/projection data as well. Review legacy absolute Windows locators and use the applicable migration/reconciliation paths, not blanket database string replacement. + +`app/migrate_layout.py:18,280` and the legacy spool default in `app/migrate_runtime_safety.py:74` also contain host-specific paths; do not use their defaults as Linux provisioning instructions. Optional `pg_trgm` creation is attempted defensively, not a proven unconditional startup prerequisite. No live migration should be run until restore/rollback and exclusive ownership are established. + +### B12. Build a Complete, Reproducible Linux Dependency Set + +**Evidence:** `app/requirements.txt:1-9`; `app/requirements-keycheckers.txt:1-4`; `app/child_bootstrap.py:24-34,177-195`; `app/keycheck_runner.py:2168-2175`; `app/lifecycle_authority.py:242-291,378-390`; `app/scanner.py:11168,11403-11405,11932,12993,13012,13834`. + +- Installing only `requirements.txt` misses `boto3`/`botocore`; isolated bootstrap requires them for every keycheck-provider. Installing only `requirements-keycheckers.txt` misses `PyYAML`. Install the union or define complete, tested profiles. `zstandard` is currently required for every scanner bootstrap, not just an enabled Docker source. +- Pin a tested, patched CPython minor and dependency resolution, including a compatible boto3/botocore pair. The host has Python 3.12.3, but that is neither a recommended security patch level nor a Linux compatibility result. Archive extraction uses version-sensitive tarfile APIs; a strict minimum of 3.12 was not established because some security APIs were backported. +- Supply Linux TruffleHog and full Git for the target architecture, with release/checksum verification. The resolver currently prefers a present private `runtime/git/cmd/git.exe` without an OS check. Exclude Windows vendor binaries and make the Linux resolution explicit. Git and TruffleHog are required by the current global manifest even for a restricted source set. +- Ensure TruffleHog's subprocess `PATH` resolves the same intended Git installation as the manifest, including its HTTPS transport helper. Copying a lone `git` executable is insufficient. +- Validate native wheels/ABI and stdlib `ssl`, `sqlite3`, `zlib`, `bz2`, `lzma`, plus CA certificates. Check shared-library requirements of the selected TruffleHog and, if embedded, PostgreSQL build. Windows wheels and extensions cannot be reused. A glibc-based image is a simpler first target than assuming Alpine/musl compatibility. +- `psycopg[binary]` with a supported wheel does not automatically require `libpq-dev`/`pg_config`; source-build requirements depend on wheel availability. The zstandard CLI does not replace the Python package. Go/CGO are build dependencies only if the selected TruffleHog is compiled from source. +- Validate the actual TruffleHog CLI and output contract: the wrapper uses Git/Docker/filesystem/HuggingFace paths, archive flags and branch/SHA/local-development options. A successful version probe alone does not validate these. The parser also uses the `finished scanning` diagnostic to distinguish complete work from an incomplete command. + +### B13. Prevent Secrets and Host State From Entering the Image + +**Evidence:** root `.gitignore`; root/application file layout; `app/runtime_security.py:872-889,989-998`; manifest exceptions in `app/lifecycle_authority.py:66-70`. + +The working directory contains credential files, credential backups, databases, scan output, keycheck output, state, logs, temporary data and bundled Windows tools. Their contents were not read for this audit. `.gitignore` does not protect a Docker build context. + +**Correction:** create `.dockerignore` plus an allowlisted `COPY` strategy. Exclude real `.env*`, secret/backup/lock variants, databases including WAL/SHM, findings/results, queues, logs, state, scratch data, caches, local tool state and Windows vendor distributions. Account explicitly for the currently manifested first-party runtime scripts instead of blindly excluding them. Do not bake credentials into layers, build arguments or a committed Compose file. + +Provide non-secret example configuration and inject secrets at runtime. Test owner and mode compatibility under B07: a root-owned `0444` secret mount does not satisfy the current effective-UID private policy. Keep code/config read-only after provisioning where feasible. The optional credential-writeback workflow is covered separately in C04. + +### B14. Persist the Whole Data Pipeline and Preserve Filesystem Semantics + +**Evidence:** `app/config.yaml:11-19,34-37,79-81,109-120,140,148`; `app/result_bundle.py:67-85,331-354,388-403`; `app/jsonl_projector.py:109-153`; `app/keycheck_runner.py:2201-2203`. + +PostgreSQL is not the only durable store. Result reservations refer to payload bundles on disk. Bundles are flushed/fsynced and atomically published from `tmp` to `ready` under one root; the commit reference is relative to that root. Losing the bundle volume while keeping PostgreSQL can lose pending ingestion inputs. Output publication also has file-level state and locks. + +| Data class | Deployment treatment | +| --- | --- | +| PostgreSQL data | Durable volume; version-compatible backup/restore; one authority | +| Result bundles | Durable volume with `tmp`, `ready` and `quarantine` together; preserve atomic rename/fsync behavior | +| Results and keycheck outputs | Preserve JSONL, rotation/publication state and needed replay inputs; maintain owner-only access | +| Queue/state/resolver and SQLite caches | Classify individually; persist required resume state, distinguish rebuildable caches from authoritative PostgreSQL data | +| Work clones/download/extraction scratch | Separate bounded writable storage; do not assume it fits memory-backed tmpfs | +| Control/PID/session metadata | Deliberately per-instance storage or exact-identity reconciliation; do not restore stale runtime identity as business data | +| Logs | Bounded retention or a secure collector; do not grow the container writable layer indefinitely | +| Config and secrets | Separately provisioned/injected; not bundled into data/image backups indiscriminately | + +Do not mount bundle `tmp` on a different filesystem from `ready`. Validate ownership, no-symlink policy, locks and durable atomic publication on the actual storage driver. Do not assume Windows binds, SMB or NFS have the required POSIX behavior. Named volumes backed by a suitable native Linux filesystem are the safer first choice, but still require testing. + +Mounting only the old `runtime/` directory misses the configured `S:` locations. Root-level `scanner.db` and old output files are not automatically the authoritative deployment dataset. Establish the transfer inventory before copying. + +Current capacity settings include a 3 GiB bundle budget, a 192 MiB per-event cap, a 2 GiB projection backlog budget and a 20 GiB free-space floor. Provision space for concurrent work, PostgreSQL/WAL, bundles and outputs, or deliberately retune those policies. A small default container disk can refuse scans even while the process is healthy. Test a coordinated database-plus-bundle restore, not just `pg_dump` in isolation. + +## P1: Unattended Operation + +### R01. Retain Ownership When Failed Startup Cleanup Cannot Confirm Exit + +`app/supervisor.py:1200-1208` swallows errors from terminate/wait and clears the retained process reference. Preserve the owner/identity and enter the existing failed-hold path if rollback cannot prove that a child stopped. Otherwise a failed start can leave an untracked process. Verify this after the POSIX containment correction. + +### R02. Preserve Signal/OOM Exit Status + +`app/owned_process.py:930` attempts to reproduce a signalled payload exit through signal handling, but installing a handler for SIGKILL is invalid. A payload terminated with `-9` can be reported as host exit 127. Correct the signal-exit reproduction and test OOM/SIGKILL separately from ordinary program failures; do not label this as a container memory-policy fix by itself. + +### R03. Unify Foreground Shutdown Completion + +`app/supervisor.py:5327-5339` writes shutdown receipts only for background children, while POSIX inspection of a non-child process cannot retrieve its exit code. This can make the existing stop workflow report failure after a foreground container runtime has actually exited. Define one completion protocol for both launch modes. Until then, authenticated `--cmd shutdown` plus independently waiting for the supervisor/container to exit is different from trusting the shutdown acknowledgement alone. + +### R04. Add Dependency-Aware Health and Recovery Semantics + +`app/supervisor.py:5225,1285,3361-3367,3551` distinguishes activation, held workers, initial ingester readiness and failed-hold state. A live PID, ACTIVE handshake, Streamlit health response or PostgreSQL TCP response is not enough to certify the pipeline. + +Expose a read-only machine health result covering supervisor phase, authenticated database/cluster identity, schema/cutover readiness, ingester/projector heartbeat or singleton lease, configured required workers, storage/backlog health and any unrecoverable hold. Allow an honest startup period without admitting work prematurely. The initial source gate opens once; explicitly decide whether later dependency loss should close it or allow bounded asynchronous intake, and test that policy through outage, backlog exhaustion and recovery. + +Distinguish degraded readiness from a dead process. A Docker healthcheck alone does not restart an unhealthy container; a restart policy normally responds to process exit. Do not configure blind health-triggered replacement that discards a `FAILED_HOLD` ownership dispute. + +### R05. Replace Host Resource Assumptions With Container Budgets + +`app/scanner.py:291,1221,1279-1290,11460-11472`; `app/owned_process.py:965`; `app/config.yaml:90-95,145`. + +Windows Job memory/CPU/priority settings are not enforced by the POSIX branch. The configured TruffleHog Job memory limit is 4 GiB; putting that value in YAML does not create a Linux limit. CPU counts may describe the host rather than the effective quota. The optional bonus scan slot uses Windows resource counters and fails closed on Linux. + +Set explicit workload concurrency, cgroup CPU/memory/PID budgets and storage limits. Account for PostgreSQL, Python, native payloads, one containment-host process per owned job and threads. A whole-container memory cap is not equivalent to the old per-tree Windows Job cap. Explicitly disable the bonus slot initially or implement quota-aware Linux telemetry without weakening admission safety. Derive limits from representative tests rather than multiplying configured maxima into an asserted minimum RAM requirement. + +### R06. Make Logs Observable Without Depending on Ignored Python Variables + +`app/supervisor.py:240` and `app/keycheck_runner.py:2168-2175` launch isolated interpreters. `-I` ignores `PYTHONUNBUFFERED` and `PYTHONIOENCODING`; adding those variables to Compose is not a reliable buffering/encoding fix. Use explicit interpreter flags such as `-u` or configure streams, and verify child output under the container locale. Retain necessary file logs with rotation/collection and keep credential-bearing output private and redacted from generic health messages. + +### R07. Enforce Provider Process Deadlines + +`app/keycheck_runner.py:2263` waits for the provider process without a process-level timeout. A provider can outlive a scheduler deadline even when individual HTTP operations have timeouts. Add a bounded process deadline/watchdog using corrected owned-tree termination; preserve partial durable results and lease recovery. A liveness check must not silently treat a stuck provider as productive work. + +## Conditional Requirements + +### C01. Authenticated Git Needs a POSIX Askpass Helper + +**Applies when:** Git requests credentials, including relevant Git/HuggingFace/package paths. + +`app/scanner.py:11426-11433` unconditionally creates a Windows `git-askpass.cmd` using batch syntax when `TRUF_GIT_TOKEN` is set. Callers include `app/scanner.py:12384-12388,12600-12605`. Anonymous clones can hide the defect. + +Provide a POSIX helper with a correct interpreter/shebang, LF and private executable permissions; retain the Windows branch. Read the token from the controlled environment, not a credential-bearing URL or argv. If the work volume is `noexec`, a prepackaged trusted helper outside that scratch volume is preferable to weakening the whole volume. Coordinate this with executable hardening and immutable code policy. Test using fake credentials and a local/mocked Git interaction. + +### C02. Dashboard Publication Requires an Explicit Security Design + +**Applies when:** the UI must be accessed from outside the runtime container. + +`app/.streamlit/config.toml:1-4` sets `127.0.0.1:5000`; `app/supervisor.py:3608-3624` and `app/dashboard.py:2528-2538` independently reject non-loopback hosts. Dashboard launch also requires authenticated supervisor-child context. Changing only Streamlit configuration or publishing a Docker port will not make the in-container loopback listener reachable. + +Either add an explicit secured container-bind mode in both guards, or use a proxy/tunnel in the **same network namespace** that can reach the existing loopback listener. A normal separate bridge-network proxy cannot reach it. Add access control and TLS at the appropriate boundary; read-only database access does not make scan/credential observability safe for public exposure. Preserve the supervised launch contract. + +Keep the control interface `127.0.0.1:8765` private (`app/supervisor.py:3726-3733`; `app/config.yaml:182-183`). Use authenticated bootstrap commands such as `--cmd status`, `--cmd shutdown` and `--attach` through `docker exec` in the same container and UID. Do not publish port 8765 or broadly remove loopback restrictions. This entire UI exposure change can be deferred by using `--no-dashboard`. + +### C03. Restricted Egress, Proxies, Custom CA and IPv6 Need Explicit Support + +**Applies when:** deployment cannot use the existing direct outbound network behavior. + +- `app/scanner.py:374-375,560,11397` uses different routing for discovery and downloads/native Git/TruffleHog. Some paths deliberately remove proxy environment variables or use direct clients. Configured download-proxy flags do not themselves implement that routing. `HTTP_PROXY` alone is not enough for a proxy-only deployment. +- `app/scanner.py:480-484` accepts proxy formats that differ from `app/keycheckers/keycheck_common.py:2200`; OpenAI/Gemini/OpenRouter also have duplicated parsers. Unify or explicitly constrain all formats and fallback behavior, including escaped credentials and ambient environment proxies. Add PySocks/`requests[socks]` only if SOCKS is required; it is not currently declared. +- `app/scanner.py:13932` ignores ambient CA settings on the downloader path. Plumb the trusted CA explicitly for corporate interception/custom trust instead of disabling verification. `app/scanner.py:105` forces IPv4 by default; consider `SCANNER_FORCE_IPV4=0` only if the target network needs IPv6 and the path is tested. +- Allow the selected sources' API, registry/CDN, download and redirect destinations, plus configured provider/resolver endpoints. DNS/private-address protections can reject destinations (`app/scanner.py:9340-9389,9416`). Preserve SSRF safeguards while making any intended private-network exception explicit. Provider resolution may contact DeepSeek/Z.ai/Qwen/Kimi depending on configured order (`app/keycheckers/provider_resolution.py:64-156`). + +Use mocks/local fixtures for proxy, TLS, redirect and DNS tests. Do not use recovered credentials to test network readiness. + +### C04. Offline Credential Writeback Needs a Different Mount Contract + +**Applies when:** `sync_alive_github_tokens.py` will update the canonical secrets file. + +`app/sync_alive_github_tokens.py:120-138,162-175,257-262` requires verified stopped authority, an adjacent lock, a same-directory private temporary file and atomic replacement. A read-only secret can be suitable for normal runtime but not for this maintenance operation. A single-file bind mount also cannot be assumed to support replacement of its mountpoint. + +Choose a separate external/offline rotation workflow or a private writable containing directory for this maintenance mode. Preserve atomic publication and exact canonical path checks. Do not make all application code/secrets permanently writable merely to support an optional operation. + +### C05. Multiple Replicas or Split Workers Require New Coordination + +**Applies when:** scaling the runtime or moving authenticated workers into separate containers. + +`app/runtime_security.py:502` uses filesystem-scoped locks. `app/lifecycle_authority.py:641` relies on local process verification, metadata and control reachability. `app/jsonl_projector.py:140-153` has both a singleton database lease and a file lock; ingester/projector are not arbitrary scalable workers. + +Local lock paths in separate container filesystems do not provide a cross-container exclusion guarantee. Worker authentication also does not become remote authentication simply because a directory is mounted. A scale-out design needs explicit shared/distributed fencing, control/identity transport, data ownership and volume semantics. Until then, use one owner and one runtime replica, prevent the old host runtime from remaining active, and do not suggest `docker compose --scale` as an operational option. + +### C06. Decide Whether Windows Archive-Name Rules Remain Policy + +**Applies when:** Linux deployments should accept archive entries valid on POSIX but invalid on Windows. + +`app/scanner.py:13760-13775` still rejects Windows reserved names, colons and trailing dot/space on Linux. This is a policy limitation, not an unconditional container boot defect. Either document it unchanged or separate OS-specific name restrictions. Keep traversal, entry-type, size and expansion-budget protections intact. + +### C07. Extend the Manifest if First-Party Linux Native Modules Are Added + +**Applies when:** native `.so` application modules are introduced inside the authenticated application tree. + +`app/lifecycle_authority.py:21` includes `.pyd` but not Linux `.so` in application import suffixes. Add the appropriate native extension suffix policy and tests when such first-party modules exist. This is not a reason to add every installed dependency to the current application-code manifest or to block the present pure-Python application solely on this basis. + +## What Is Not Required + +- No Docker daemon, Docker socket mount, Docker CLI, DinD, privileged container or image-architecture emulation is needed for the inspected DockerHub scanning path. It reads image content rather than executing the image (`app/scanner.py:12980,13645`). +- `DOCKER_CONFIG` is an authentication input, not a need for Docker Desktop. Recovery intentionally rejects implicit keychain/helper assumptions; preserve the managed credential pool (`app/scanner.py:4790,13433-13456`). +- npm/PyPI content is scan data. Node.js and a browser are not runtime dependencies of these scanner/keychecker paths. AWS CLI, `gcloud` and `az` are not required merely because those providers are checked. +- Go is not needed in the final image when supplying a compatible prebuilt TruffleHog. GCP keychecker RSA handling does not establish a dependency on Google SDK/cryptography/openssl CLI (`app/keycheckers/gcp/gcpKeycheck.py:288`). +- Existing POSIX `/proc` identity support, `flock`, PostgreSQL command branches, activation/STOPPING states, leases and owned-versus-observed database semantics should be preserved and completed, not rewritten wholesale. +- A general path-case rename or repository-wide CRLF rewrite was not justified. Fix genuinely platform-specific helpers and configured paths instead. + +## Documentation and Operational Corrections + +Create a deployment Compose definition separate from the recovery fixture, an allowlisted image build, a non-secret Linux configuration example, an ownership/volume provisioning procedure, and a backup/restore/upgrade runbook. These are missing deployment deliverables, not files generated by this audit. + +Document the exact source profile and foreground lifecycle, explicit health semantics, stop/restart deadlines, singleton restriction, volume classes, secret maintenance, pinned versions and supported architecture. Remove Windows freeze-counter/diagnostic scripts from the Linux launch chain (`start_freeze_counters.ps1:27`); keeping a script as inert manifest data is different from executing it. + +Correct any assumption that provider checks are free/read-only readiness probes. `app/KEYCHECKERS.md:44,87,109` must be reconciled with actual provider defaults. Qwen and several other providers can perform generation by default (`app/keycheckers/qwen/qwenKeycheck.py:544`); AWS/Replicate/Azure paths can probe IAM, resources or RBAC, with additional optional model requests. TruffleHog `no-verification` does not disable the separate keychecker subsystem (`app/config.yaml:136`). Healthchecks and image smoke tests must not invoke those real credential checks. + +## Recommended Implementation Order + +1. Choose Linux distribution/CPU architecture, PostgreSQL topology, UI requirement, source set, egress policy, non-root UID and storage/resource budgets. Confirm which existing data is authoritative and define rollback. +2. Fix portable paths, POSIX identity/containment, signal/restart behavior and executable/ACL policy. Add focused offline Linux regression tests while preserving the current Windows contracts. +3. Assemble the pinned dependency/native-tool image and code-authority layout. Add `.dockerignore`, a foreground entrypoint contract, private provisioning and a separate deployment Compose definition. Keep one runtime replica. +4. Build and exercise a disposable Linux environment with fake credentials and local fixtures. Validate process lifetime, shutdown, restart, permissions, imports, native CLI contracts, health and resource limits before touching real data. +5. Implement the chosen PostgreSQL mode. Rehearse fresh initialization and a restored dataset, schema/cutover migration, bundle/projection reconciliation and coordinated recovery. Embedded identity binding needs a stopped target PostgreSQL; SQL migration needs PostgreSQL available with normal workers stopped. +6. Complete the conditional features actually needed: authenticated Git, secured UI, restricted-network support or credential maintenance. Defer unrelated scale-out work. +7. Perform a controlled real-data cutover only after backup/restore rehearsal, exclusive ownership and rollback checks. Start with conservative concurrency and verify health/backlog behavior before increasing load. + +## Acceptance Tests + +| Area | Required evidence before claiming support | +| --- | --- | +| Build and ABI | Build on each supported target architecture; resolved dependency check; stdlib/native imports through the intended isolated interpreter; TruffleHog/Git help/version and library compatibility | +| Authority image layout | No rejected application bytecode/symlinks; required manifest files/assets present; stable code/config hashes; dependencies outside the application tree | +| Paths and permissions | Linux path resolution for every configured directory/file; non-root fresh-volume provisioning; correct private owners/modes; executable bits survive hardening; usable real lock root | +| Process identity | Real POSIX host/payload handshake can create the scanner owner marker; exact identity survives normal lifecycle checks | +| Process containment | Nested provider/native grandchildren terminate on source restart, parent death and failed startup; no orphan payloads or unreaped zombies | +| Container lifecycle | Foreground no-TTY autostart; SIGTERM during active work; confirmed drain/stop; bounded ordinary shutdown; explicit failed-hold escalation; restart after reused PID/stale metadata | +| Database | Fresh schema and valid cutover; wrong cluster rejected; offline identity rebinding; authenticated outage/recovery; external mode, if chosen, has no local PG start/stop dependency | +| Durability | Container recreation preserves reservations/bundles/publications; interrupted atomic publication recovers safely; coordinated PG-plus-bundle backup can actually be restored | +| Limits | Full disk, low free-space floor, bounded backlog, constrained CPU/RAM/PIDs, OOM exit status, bonus-slot denial and provider process deadline | +| Network | Mocked direct/proxy/SOCKS-as-needed, parser formats, CA, IPv4/IPv6-as-needed, redirects and DNS/SSRF behavior | +| Git | POSIX authenticated askpass with fake credentials, private executable permissions and the selected `noexec` work-volume arrangement | +| Archives | gzip/zstd/PAX, malformed archives/missing decoders, entry-name policy and traversal/size protections on the chosen patched CPython | +| Optional UI | Reachable only by the intended secured path; supervised authentication intact; health distinguished from full pipeline readiness; control port not published | +| Safe probes | No provider generation, credential validation or production endpoint activity from build/health tests | + +Existing tests to extend/select carefully: + +- `tests/test_owned_process.py:79-82,98`: the identity assertion covers PID, and tree cleanup coverage is Windows-specific. +- `tests/test_temp_owner_child_safety.py:49-59`: a mock supplies complete identity and can hide the real POSIX handshake defect. +- `tests/test_supervisor_safety.py:978`: a mocked zero exit code can hide the foreground completion gap. +- `tests/test_pipeline_postgres_integration.py:69`: adapt `.exe` assumptions to a disposable Linux PostgreSQL setup, not the authoritative host database. +- `tests/test_runtime_security.py:227`: add actual container UID/mount/symlink/executable-policy cases. +- `tests/test_api_proxy_routing.py:62,92,215,248`: extend mocked routing, proxy-parser and trust behavior. +- `tests/test_docker_codec_recovery.py:72,130,149,175` and `tests/test_resource_lifecycle_fixes.py:191`: extend codec and cgroup/admission coverage. +- Some tests exercise installed/native/live paths, including `tests/test_docker_codec_recovery.py:666` and `tests/test_huggingface_long_paths.py:324`. Separate offline tests from explicitly opted-in integration tests; do not run the entire suite against existing data or credentials by default. + +## Verification Performed and Limits + +- Inspected the application, configuration, entrypoints, dependency manifests, security/process/DB/result-pipeline code, recovery Compose and relevant tests. Findings cite inspected file/line locations; line numbers may move with later edits. +- Reproduced the Windows-path/config-precedence failure using the actual pure path functions under POSIX path semantics, without application startup or filesystem mutation. +- Parsed all 61 application Python files with Python 3.12.3 using AST-only analysis: zero syntax errors. This is not an import, dependency, Linux execution or behavior test. +- Docker CLI was unavailable in this session: `docker version --format '{{json .Server}}'` failed because the command was not found. This does not prove that the host has no Docker installation or can never run containers. +- No Docker build, Compose deployment, Linux process integration test or full pytest suite was run. No supervisor, scanner, provider checker or PostgreSQL server was started. Secret contents, real credential checks and database migrations were not used for verification. +- Only this report was added. The findings identify the correction surface visible from repository inspection; target-platform tests may reveal additional issues. No claim of Docker readiness is made until the acceptance checks pass. diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..f911048 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,368 @@ +FROM python:3.12-slim-bookworm@sha256:782412e85d0f0984994c290652577d4018aff08145c85b262bb63dc0c7522254 AS python-base + +ENV PATH=/usr/local/bin:/usr/bin:/bin:/usr/lib/postgresql/16/bin \ + HOME=/data/home \ + LANG=C.UTF-8 \ + LC_ALL=C.UTF-8 \ + PYTHONDONTWRITEBYTECODE=1 + +RUN /usr/local/bin/python3 -I -S -B -c "import sys; assert sys.version_info[:3] == (3, 12, 14), sys.version" + +FROM python-base AS lock-generator + +COPY docker/build-dependencies/requirements.lock /tmp/compiler.lock +RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \ + --require-hashes --only-binary=:all: --no-compile --no-cache-dir \ + -r /tmp/compiler.lock \ + && rm /tmp/compiler.lock +WORKDIR /src +COPY app/requirements.txt app/requirements-keycheckers.txt ./app/ +COPY docker/requirements.in docker/requirements.lock docker/requirements-test.in docker/requirements-test.lock docker/requirements-worker.in docker/requirements-worker.lock ./docker/ +COPY docker/build-dependencies/requirements.in docker/build-dependencies/requirements.lock ./docker/build-dependencies/ +ENV CUSTOM_COMPILE_COMMAND="See docker/build-dependencies/README.md for the pinned Python 3.12.14 pip-tools generation command." +CMD ["python3", "-m", "piptools", "compile", "--generate-hashes", "--allow-unsafe", "--resolver=backtracking", "--strip-extras", "--no-emit-index-url", "--no-emit-trusted-host", "--index-url=https://pypi.org/simple", "--pip-args=--only-binary=:all:", "--output-file=docker/requirements.lock", "docker/requirements.in"] + +FROM python-base AS trufflehog-download + +ARG TARGETARCH +RUN python3 -I -S -B - "$TARGETARCH" <<'PY' +import hashlib +import os +import shutil +import sys +import tarfile +import urllib.request + +checksums = { + "amd64": "dc24007c2f233bd61c05beabeb44aa27ea9b43288166279209abe0458c5ce76b", + "arm64": "7e65e771d2a247964056aa5edba0f8ae3945895e5dce867fe0ffbc7b0128239a", +} +architecture = sys.argv[1] +if architecture not in checksums: + raise SystemExit("TruffleHog is pinned only for linux/amd64 and linux/arm64") +name = f"trufflehog_3.97.4_linux_{architecture}.tar.gz" +url = "https://github.com/trufflesecurity/trufflehog/releases/download/v3.97.4/" + name +digest = hashlib.sha256() +size = 0 +with urllib.request.urlopen(url, timeout=60) as response, open("/tmp/trufflehog.tar.gz", "wb") as output: + while chunk := response.read(1024 * 1024): + digest.update(chunk) + size += len(chunk) + output.write(chunk) +if digest.hexdigest() != checksums[architecture]: + raise SystemExit("TruffleHog archive SHA-256 mismatch") +if architecture == "amd64" and size != 34970205: + raise SystemExit("TruffleHog amd64 archive length mismatch") +with tarfile.open("/tmp/trufflehog.tar.gz", "r:gz") as archive: + member = archive.getmember("trufflehog") + if not member.isfile(): + raise SystemExit("TruffleHog archive executable must be a regular file") + with archive.extractfile(member) as source, open("/trufflehog", "wb") as output: + shutil.copyfileobj(source, output) +os.chmod("/trufflehog", 0o755) +os.unlink("/tmp/trufflehog.tar.gz") +print(f"Verified {name}: {size} bytes, sha256:{digest.hexdigest()}") +PY + +FROM python-base AS worker-dependencies + +COPY docker/requirements-worker.lock /tmp/requirements-worker.lock +RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \ + --require-hashes --only-binary=:all: --no-compile --no-cache-dir \ + --target /worker-dependencies -r /tmp/requirements-worker.lock \ + && PYTHONPATH=/worker-dependencies python3 -I -S -B - <<'PY' +import sys +sys.path.insert(0, "/worker-dependencies") +import requests +import yaml +import zstandard +PY +RUN rm /tmp/requirements-worker.lock + +FROM python-base AS worker-native-dependencies + +RUN <<'SH' +set -eu +rm -f /etc/apt/sources.list /etc/apt/sources.list.d/debian.sources +printf '%s\n' \ + 'Types: deb' \ + 'URIs: https://snapshot.debian.org/archive/debian/20260914T000000Z/' \ + 'Suites: bookworm bookworm-updates' \ + 'Components: main' \ + 'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \ + 'Check-Valid-Until: no' \ + '' \ + 'Types: deb' \ + 'URIs: https://snapshot.debian.org/archive/debian-security/20260914T000000Z/' \ + 'Suites: bookworm-security' \ + 'Components: main' \ + 'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \ + 'Check-Valid-Until: no' \ + > /etc/apt/sources.list.d/debian.sources +export DEBIAN_FRONTEND=noninteractive +apt-get -o Acquire::Retries=3 -o Acquire::https::Timeout=30 -o APT::Update::Error-Mode=any update +apt-get install -y --no-install-recommends \ + ca-certificates=20250419~deb12u1 \ + git=1:2.39.5-0+deb12u3 \ + tini=0.19.0-1+b3 +install -d -o 10001 -g 10001 -m 0700 /data /data/home +/usr/sbin/groupadd --gid 10001 truf +/usr/sbin/useradd --uid 10001 --gid 10001 --no-create-home --home-dir /data/home --shell /usr/sbin/nologin truf +install -d -o 0 -g 0 -m 0755 /worker-git/bin /worker-git/libexec /worker-git/share +cp -aL /usr/bin/git /worker-git/bin/git +cp -aL /usr/lib/git-core /worker-git/libexec/git-core +cp -aL /usr/share/git-core /worker-git/share/git-core +find /worker-git -type d -exec chmod 0755 {} + +find /worker-git -type f -exec chmod go-w {} + +test -x /worker-git/bin/git +test -x /worker-git/libexec/git-core/git-remote-https +rm -rf /var/lib/apt/lists/* /var/cache/apt/archives/* /var/log/apt/* +SH + +COPY --from=trufflehog-download --chown=0:0 --chmod=0755 /trufflehog /usr/local/bin/trufflehog + +FROM worker-native-dependencies AS worker-package-build + +ARG TARGETARCH +COPY --from=worker-dependencies --chown=0:0 /worker-dependencies /build/app/dependencies +COPY app/ /build/app/ +COPY docs/remote-worker-quickstart-ru.md /build/README_RU.md +COPY docs/remote-worker-cheatsheet-windows-ru.md /build/ +COPY docs/remote-worker-cheatsheet-linux-ru.md /build/ +COPY docs/remote-worker-cheatsheet-docker-ru.md /build/ +COPY docker/worker-package-pins.json /build/worker-package-pins.json +RUN case "$TARGETARCH" in \ + amd64) platform_tag=linux-x86_64 ;; \ + arm64) platform_tag=linux-aarch64 ;; \ + *) echo "unsupported worker architecture" >&2; exit 1 ;; \ + esac \ + && python3 -u -I -S -B /build/app/worker_package_builder.py assemble \ + --output /opt/truf-worker \ + --source-app /build/app \ + --dependencies /build/app/dependencies \ + --detector-policy /build/app/trufflehog-custom-detectors.yaml \ + --trufflehog /usr/local/bin/trufflehog \ + --git-root /worker-git \ + --git-executable bin/git \ + --platform-tag "$platform_tag" \ + --operator-readme /build/README_RU.md \ + --operator-cheatsheet /build/remote-worker-cheatsheet-windows-ru.md \ + --operator-cheatsheet /build/remote-worker-cheatsheet-linux-ru.md \ + --operator-cheatsheet /build/remote-worker-cheatsheet-docker-ru.md \ + --build-inputs /build/worker-package-pins.json + +FROM worker-native-dependencies AS worker + +ENV GIT_EXEC_PATH=/opt/truf-worker/runtime/git/libexec/git-core \ + GIT_TEMPLATE_DIR=/opt/truf-worker/runtime/git/share/git-core/templates +COPY --from=worker-package-build --chown=0:0 /opt/truf-worker /opt/truf-worker +RUN chown -R 10001:10001 /opt/truf-worker/app \ + && find /opt/truf-worker/app -type d -exec chmod 0700 {} + \ + && find /opt/truf-worker/app -type f -exec chmod 0600 {} + \ + && chown 10001:10001 /opt/truf-worker/worker-package.json \ + && chmod 0600 /opt/truf-worker/worker-package.json \ + && find /opt/truf-worker/bin /opt/truf-worker/runtime -type d -exec chmod 0755 {} + \ + && find /opt/truf-worker/bin /opt/truf-worker/runtime -type f -exec chmod go-w {} + \ + && test ! -e /opt/truf-worker/app/keycheck_runner.py \ + && test ! -d /opt/truf-worker/app/keycheckers \ + && test ! -e /usr/lib/postgresql \ + && test ! -e /usr/bin/psql +USER 10001:10001 +RUN /opt/truf-worker/bin/trufflehog --version >/dev/null \ + && /opt/truf-worker/runtime/git/bin/git --version >/dev/null \ + && /usr/local/bin/python3 -I -S -B - <<'PY' +import sys +sys.path[:0] = ['/opt/truf-worker/app', '/opt/truf-worker/app/dependencies'] +from importlib.util import find_spec +from worker_cli import parse_args +from worker_package import verify_worker_package + +package = verify_worker_package('/opt/truf-worker/worker-package.json') +assert package['manifest']['schema'] == 3 +assert package['manifest']['protocol_version'] == 2 +assert { + (item['source'], item['platform'], item['planning_kind']) + for item in package['manifest']['capabilities'] +} == { + ('gitlab', 'gitlab', 'exact_git_v1'), + ('dockerhub', 'docker', 'docker_direct_v1'), + ('huggingface', 'huggingface', 'huggingface_space_v1'), +} +assert set(package['runtime_trees']) == {'git'} +assert parse_args(['run', '--server', 'https://worker.example', '--token', 'x' * 32]).command == 'run' +assert all(find_spec(name) is None for name in ('httpx', 'psycopg', 'starlette', 'streamlit')) +PY +WORKDIR /data +ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/python3", "-u", "-I", "-S", "-B", "/opt/truf-worker/app/remote_worker_bootstrap.py", "--"] +CMD ["run"] + +FROM python-base AS native-dependencies + +ADD --checksum=sha256:0144068502a1eddd2a0280ede10ef607d1ec592ce819940991203941564e8e76 https://www.postgresql.org/media/keys/ACCC4CF8.asc /usr/share/keyrings/postgresql.asc + +RUN <<'SH' +set -eu +chmod 0644 /usr/share/keyrings/postgresql.asc +rm -f /etc/apt/sources.list /etc/apt/sources.list.d/debian.sources +printf '%s\n' \ + 'Types: deb' \ + 'URIs: https://snapshot.debian.org/archive/debian/20260914T000000Z/' \ + 'Suites: bookworm bookworm-updates' \ + 'Components: main' \ + 'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \ + 'Check-Valid-Until: no' \ + '' \ + 'Types: deb' \ + 'URIs: https://snapshot.debian.org/archive/debian-security/20260914T000000Z/' \ + 'Suites: bookworm-security' \ + 'Components: main' \ + 'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \ + 'Check-Valid-Until: no' \ + > /etc/apt/sources.list.d/debian.sources +printf '%s\n' \ + 'deb [signed-by=/usr/share/keyrings/postgresql.asc] https://apt-archive.postgresql.org/pub/repos/apt bookworm-pgdg-archive main' \ + > /etc/apt/sources.list.d/postgresql.list +# Only these exact PGDG packages may supplement the immutable Debian snapshot. +printf '%s\n' \ + 'Package: postgresql-16 postgresql-client-16 libpq5' \ + 'Pin: version 16.15-1.pgdg12+2' \ + 'Pin-Priority: 1001' \ + '' \ + 'Package: postgresql-common postgresql-client-common' \ + 'Pin: version 293.pgdg12+1' \ + 'Pin-Priority: 1001' \ + '' \ + 'Package: *' \ + 'Pin: origin apt-archive.postgresql.org' \ + 'Pin-Priority: -1' \ + > /etc/apt/preferences.d/postgresql +printf '#!/bin/sh\nexit 101\n' > /usr/sbin/policy-rc.d +chmod 0755 /usr/sbin/policy-rc.d +export DEBIAN_FRONTEND=noninteractive +apt-get -o Acquire::Retries=3 -o Acquire::https::Timeout=30 -o APT::Update::Error-Mode=any update +apt-get install -y --no-install-recommends \ + postgresql-common=293.pgdg12+1 \ + postgresql-client-common=293.pgdg12+1 +# Set this after common is installed but before installing any server package. +printf '\ncreate_main_cluster = false\n' >> /etc/postgresql-common/createcluster.conf +apt-get install -y --no-install-recommends \ + ca-certificates=20250419~deb12u1 \ + git=1:2.39.5-0+deb12u3 \ + tini=0.19.0-1+b3 \ + postgresql-16=16.15-1.pgdg12+2 \ + postgresql-client-16=16.15-1.pgdg12+2 \ + libpq5=16.15-1.pgdg12+2 +test ! -d /var/lib/postgresql/16/main +rm -f /etc/ssl/private/ssl-cert-snakeoil.key /etc/ssl/certs/ssl-cert-snakeoil.pem +rm -rf /var/lib/apt/lists/* /var/cache/apt/archives/* /var/log/apt/* +/usr/sbin/groupadd --gid 10001 truf +/usr/sbin/useradd --uid 10001 --gid 10001 --no-create-home --home-dir /data/home --shell /usr/sbin/nologin truf +install -d -o 10001 -g 10001 -m 0700 /data /data/home +SH + +RUN python3 -I -S -B - <<'PY' +import os +from pathlib import Path + +# Keep the package's complete Git helper tree; regular hard links preserve argv[0]. +for directory in (Path("/usr/local/bin"), Path("/usr/lib/git-core")): + for path in directory.iterdir(): + if path.is_symlink() and os.access(path, os.X_OK): + target = path.resolve(strict=True) + if not target.is_file() or target.stat().st_uid != 0: + raise SystemExit(f"Untrusted executable target: {path}") + path.unlink() + os.link(target, path) +# Prefer native PG16 clients over the distribution's symlinked version wrappers. +for target in Path("/usr/lib/postgresql/16/bin").iterdir(): + path = Path("/usr/bin") / target.name + if path.is_symlink(): + path.unlink() + os.link(target, path) +for name in ("/usr/local/bin/python3", "/usr/bin/git", "/usr/lib/git-core/git-remote-https", "/usr/bin/tini"): + path = Path(name) + details = path.lstat() + if path.is_symlink() or not path.is_file() or details.st_uid != 0 or details.st_mode & 0o022: + raise SystemExit(f"Untrusted native executable: {path}") +PY + +FROM native-dependencies AS dependencies + +COPY docker/requirements.lock /tmp/requirements.lock +RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \ + --require-hashes --only-binary=:all: --no-compile --no-cache-dir \ + -r /tmp/requirements.lock \ + && python3 -m pip --isolated check \ + && rm /tmp/requirements.lock +USER 10001:10001 + +# Both public targets inherit these exact runtime contents; the default stays runtime. +FROM dependencies AS runtime-base + +USER 0:0 +RUN install -d -o 10001 -g 10001 -m 0700 /opt/truf /opt/truf/app /opt/truf/tests +USER 10001:10001 +COPY --chown=10001:10001 app/ /opt/truf/app/ +RUN python3 -I -S -B - <<'PY' +import os +import stat + +for directory, directories, files in os.walk("/opt/truf/app", followlinks=False): + for path in [directory, *(os.path.join(directory, name) for name in directories + files)]: + details = os.lstat(path) + if not (stat.S_ISDIR(details.st_mode) or stat.S_ISREG(details.st_mode)): + raise SystemExit(f"Application links/special files are forbidden: {path}") + if os.path.basename(path) == "__pycache__" or path.endswith((".pyc", ".pyo")): + raise SystemExit(f"Application bytecode is forbidden: {path}") + if (details.st_uid, details.st_gid) != (10001, 10001): + raise SystemExit(f"Application ownership mismatch: {path}") + os.chmod(path, 0o700 if stat.S_ISDIR(details.st_mode) else 0o600) +PY +WORKDIR /opt/truf/app +ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/python3", "-u", "-I", "-S", "-B", "/opt/truf/app/container_runtime.py"] +CMD ["run"] + +FROM runtime-base AS test + +USER 0:0 +COPY --from=trufflehog-download --chown=0:0 --chmod=0755 /trufflehog /usr/local/bin/trufflehog +COPY docker/requirements-test.lock /tmp/requirements-test.lock +RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \ + --require-hashes --only-binary=:all: --no-compile --no-cache-dir \ + -r /tmp/requirements-test.lock \ + && python3 -m pip --isolated check \ + && rm /tmp/requirements-test.lock +USER 10001:10001 +COPY --chown=10001:10001 tests/ /opt/truf/tests/ +COPY --chown=10001:10001 --chmod=0600 .dockerignore /opt/truf/.dockerignore +COPY --chown=10001:10001 --chmod=0600 Dockerfile /opt/truf/Dockerfile +COPY --chown=10001:10001 --chmod=0600 start_runtime.ps1 start_core_runtime.ps1 stop_runtime.ps1 /opt/truf/ +COPY --chown=10001:10001 --chmod=0600 compose.yaml compose.edge.yaml /opt/truf/ +COPY --chown=10001:10001 deploy/ /opt/truf/deploy/ +COPY --chown=10001:10001 docker/ /opt/truf/docker/ +RUN python3 -I -S -B - <<'PY' +import os +from pathlib import Path +import stat + +edge_e2e = { + path.name for path in Path('/opt/truf/tests').glob('edge_e2e_*.py') +} +if edge_e2e != {'edge_e2e_backend.py', 'edge_e2e_client.py'}: + raise SystemExit(f'Unexpected edge E2E test-stage inputs: {sorted(edge_e2e)}') +for directory, directories, files in os.walk("/opt/truf/tests", followlinks=False): + for path in [directory, *(os.path.join(directory, name) for name in directories + files)]: + details = os.lstat(path) + if not (stat.S_ISDIR(details.st_mode) or stat.S_ISREG(details.st_mode)): + raise SystemExit(f"Test links/special files are forbidden: {path}") + if os.path.basename(path) == "__pycache__" or path.endswith((".pyc", ".pyo")): + raise SystemExit(f"Test bytecode is forbidden: {path}") + if (details.st_uid, details.st_gid) != (10001, 10001): + raise SystemExit(f"Test ownership mismatch: {path}") + os.chmod(path, 0o700 if stat.S_ISDIR(details.st_mode) else 0o600) +PY +WORKDIR /opt/truf +ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/python3", "-u", "-I", "-S", "-B", "/opt/truf/tests/container_unit.py"] +CMD [] + +FROM runtime-base AS runtime diff --git a/QUARANTINE_AUDIT.md b/QUARANTINE_AUDIT.md new file mode 100644 index 0000000..114599b --- /dev/null +++ b/QUARANTINE_AUDIT.md @@ -0,0 +1,102 @@ +# Pipeline Quarantine Audit + +Initial snapshot and remediation: `2026-08-15` + +## Current Impact + +- PostgreSQL quarantine rows: `177` +- Accounted capacity: `4354 items / 1,817,503,389 bytes` +- Configured admission limit: `10000 items / 1,073,741,824 bytes` +- New scan admission is closed because the byte limit is exceeded. +- `49,215` admission intents have already ended with `quarantine_admission_closed`. + +`pipeline: ready` means that PostgreSQL and workers are healthy. It does not mean that new scan admission is open. + +## Result Bundles + +Two rows account for `4004 items / 1,212,153,856 bytes`. + +| Quarantine ID | Source | Original error | Physical size | Read-only validation now | +|---|---|---|---:|---| +| `29` | Hugging Face `spaces` | Transient Windows `Permission denied` | `2,870 B` | Valid; 3 frames, no findings/errors/candidates | +| `91` | Docker Hub query `tokenizer` | Transient Windows `Permission denied` | `5,354 B` | Valid; 8 frames, 1 finding, 4 errors, no candidates | + +The bundle contents are valid. Their large capacity cost comes from worst-case reservations transferred into quarantine, not their physical file sizes. + +Current code now reports a temporarily unavailable private bundle as an availability error. Result ingester defers it instead of classifying it as invalid content. + +Recommendation: recover both bundles through an audited offline path rather than discard them. ID `91` contains one finding and must not be deleted without an explicit decision. + +## Keycheck Quarantine + +There are `175` pending keycheck quarantine rows. + +| Class | Rows | Distinct credentials | Assessment | +|---|---:|---:|---| +| Repeated DeepSeek unconsumed rechecks | `104` | `1` | Duplicate hourly retries; current state is now `NO_CONTEXT` | +| DeepSeek unconsumed findings | `15` | `6` | Legitimate historical candidates filtered by routing | +| Azure Foundry unconsumed | `15` | `12` | Historical provider-consumption issue | +| Hugging Face unconsumed | `7` | `5` | Historical provider-consumption issue | +| Replicate unconsumed | `6` | `3` | Historical provider-consumption issue | +| xAI unconsumed | `3` | `3` | Historical provider-consumption issue | +| DeepSeek route mismatch | `24` | `14` | Misrouted non-DeepSeek detectors; source findings remain in PostgreSQL | +| Azure route mismatch | `1` | `1` | Historical Azure Foundry route mismatch; current state is `UNKNOWN` | + +The active growth came from one DeepSeek credential whose current state was `NETWORK`. Hourly network retry created a fresh candidate, provider routing silently rejected it, and the candidate entered quarantine after three unconsumed attempts. + +## Preventive Changes + +- Non-DeepSeek routing decisions now complete as `NO_CONTEXT` instead of leaving a leased candidate unconsumed. +- DeepSeek has a canonical `deepseekNoContext.txt` status projection. +- Recheck generation now refuses to enqueue a credential while it has an unresolved `provider_candidate_unconsumed` or `candidate_provider_route_mismatch` quarantine row. +- Temporarily unavailable result bundle files are deferred instead of quarantined as validation failures. + +Verification: + +- Targeted tests: `60 passed`. +- Manual `recheck deepseek network`: code `0`, no candidates processed. +- DeepSeek unconsumed quarantine remained exactly `119`; no new row was created. + +## Approved Remediation + +1. Stop sources and acquire the offline migration guard. +2. Recover bundle IDs `29` and `91` through deterministic re-ingestion. +3. Discard the `104` duplicate DeepSeek retry rows after exact manifest review. +4. Discard the `24` confirmed DeepSeek route-mismatch rows after exact manifest review. +5. Requeue one current candidate per distinct credential from the remaining legitimate unconsumed groups; keep source attribution. +6. Review the single Azure route mismatch separately. +7. Resolve superseded duplicate candidate rows with an audited discard manifest. +8. Verify physical artifacts, capacity accounting, projections and reopened scan admission before restarting sources. + +All destructive decisions must use an exact private `truf-pipeline-quarantine-review-v1` manifest containing quarantine ID, reason code, payload hash and action. The offline review command requires both `--apply` and `--sources-stopped`. + +## Applied Result + +The user approved recovery plus selective cleanup. + +- Manifest: `runtime/control/quarantine-remediation-20260815.json` +- Manifest SHA-256: `c3143e6b0e15e07b6afacffd50b449444b9932e75001f633ac82b5b79e21fe3f` +- Reviewed: `177` +- Approved retry: `32` +- Audited discard: `145` +- Duplicate/conflicting reviews: `0` +- Final quarantine capacity: `0 items / 0 bytes` + +Bundle outcomes: + +| Quarantine ID | Final reservation | Final bundle | Target disposition | Preserved contents | +|---|---|---|---|---| +| `29` | `acknowledged` | `acknowledged` | `done` | Clean empty result | +| `91` | `acknowledged` | `acknowledged` | `deferred` | `1 finding`, `4 errors` | + +Keycheck outcomes from the retried unique credentials: + +| Service | Final current statuses | +|---|---| +| Azure | `12 FOUNDRY_UNRESOLVED`, `1 UNKNOWN` | +| DeepSeek | `6 NO_CONTEXT` | +| Hugging Face | `5 NO_CONTEXT` | +| Replicate | `3 NO_CONTEXT` | +| xAI | `3 NO_CONTEXT` | + +The malformed legacy provider candidates are now terminal current-state records instead of repeatedly deferred/quarantined candidates. Scan admission reopened and scanner workers returned to `3/3` active operation. diff --git a/RUNTIME_CHEATSHEET.md b/RUNTIME_CHEATSHEET.md new file mode 100644 index 0000000..65d5aac --- /dev/null +++ b/RUNTIME_CHEATSHEET.md @@ -0,0 +1,273 @@ +# Truf Runtime Cheatsheet + +## Быстрый старт + +Команды выполняются из `D:\truf` в PowerShell. + +```powershell +# Запустить весь canonical runtime +.\start_runtime.ps1 + +# Запустить explicit core set с keychecks, без dashboard +.\start_core_runtime.ps1 + +# Подключиться к интерактивной консоли supervisor +.\attach_runtime.ps1 + +# Координированно остановить весь runtime и PostgreSQL +.\stop_runtime.ps1 +``` + +Если PowerShell блокирует запуск скриптов: + +```powershell +powershell.exe -NoProfile -ExecutionPolicy Bypass -File .\start_runtime.ps1 +``` + +Для полного рестарта используй именно: + +```powershell +.\stop_runtime.ps1 +.\start_runtime.ps1 +``` + +Для полного рестарта в core-only режиме: + +```powershell +.\stop_runtime.ps1 +.\start_core_runtime.ps1 +``` + +Не используй `restart all` как замену полному рестарту: pipeline workers защищены от ручного рестарта, пока scanner sources работают. + +## Что запускается + +| Компонент | Назначение | +|---|---| +| PostgreSQL | Единственный authoritative storage | +| `result-ingester` | Переносит scan bundles в PostgreSQL | +| `jsonl-projector` | Создаёт compatibility JSONL projections | +| `janitor` | Обслуживает runtime queues и временные данные | +| `github` | GitHub scanner loop | +| `gitlab` | GitLab scanner loop | +| `huggingface` | Hugging Face scanner loop | +| `dockerhub` | Docker Hub scanner loop | +| `package_git` | Package/repository scanner loop | +| `keychecks` | Почасовой provider checker scheduler | + +Одновременно выполняется максимум `3` scan workers. Dashboard при обычном запуске выключен. + +## PowerShell-скрипты + +| Скрипт | Что делает | +|---|---| +| `start_runtime.ps1` | Запускает freeze diagnostics, проверяет identity PostgreSQL и поднимает background supervisor | +| `start_core_runtime.ps1` | Поднимает PostgreSQL, pipeline, janitor и три discovery-only producer; dashboard выключен | +| `stop_runtime.ps1` | Выполняет authenticated coordinated shutdown supervisor, children и PostgreSQL | +| `attach_runtime.ps1` | Открывает интерактивную supervisor-консоль; `quit` только отключает консоль | +| `start_freeze_counters.ps1` | Запускает Windows performance counters в `H:\truf-diagnostics` | +| `monitor_runtime_lag.ps1` | Пишет CPU/RAM/disk/runtime lag в CSV | +| `cleanup_stale_agentui_vite.ps1` | Отдельная уборка старых AgentUI/Vite процессов; без `-Apply` только dry run | +| `runtime\check-openrouter-keys.ps1` | Retired; намеренно завершается ошибкой | + +В проекте нет собственных `.bat`/`.cmd`. Найденные BAT внутри `runtime\postgres\pgsql\pgAdmin 4` принадлежат pgAdmin и для Truf не используются. + +## Full и Core-only режимы + +`start_runtime.ps1` использует allowlist `supervisor.enabled_sources` из `config.linux.yaml`. Сейчас этот allowlist уже равен distributed core set, поэтому оба start-скрипта запускают одинаковые discovery producer. + +`start_core_runtime.ps1` фиксирует core set прямо в wrapper и не зависит от будущего расширения default allowlist: + +```text +gitlab,dockerhub,huggingface +``` + +PostgreSQL, `result-ingester`, `jsonl-projector`, `janitor` и независимо включённый `keychecks` также запускаются. Dashboard не запускается. + +Чтобы сменить режим, сначала останови текущий supervisor через `.\stop_runtime.ps1`, затем запусти нужный start-скрипт. `stop_runtime.ps1` одинаков для обоих режимов. + +## Core Sources + +| Source | Что производит на сервере | +|---|---| +| `gitlab` | Ищет недавно активные GitLab projects и ставит их в очередь remote workers | +| `dockerhub` | Ищет Docker Hub images и ставит в очередь только immutable `repo@sha256:...` targets | +| `huggingface` | Ищет новейшие Hugging Face Spaces и ставит их в очередь remote workers | + +`result-ingester`, `jsonl-projector`, `janitor` и `keychecks` отображаются как отдельные system workers, но не являются discovery sources. GitHub и `package_git` остаются доступными legacy/manual source, однако в distributed core profile не входят. + +## Janitor + +Janitor обслуживает только scanner work area (`S:\scanner-work`), а не PostgreSQL и не provider status files. + +- Каждые `60` секунд ищет временные каталоги разрешённых типов. +- Рассматривает только каталоги старше `7200` секунд. +- Требует приватный `.scanner-owner.json` с точным process identity. +- Удаляет каталог только если owner и parent гарантированно мертвы. +- Не следует по symlink, junction или другим reparse points. +- Один проход ограничен `50` каталогами, `10000` entries, `1 GiB`, `30` секундами и depth `64`. +- Не имеет PostgreSQL credentials и не удаляет findings, keycheck history, current state или найденные секреты. + +Примеры диагностики: + +```powershell +# Один диагностический замер +.\monitor_runtime_lag.ps1 -Once + +# Свой файл и интервал +.\monitor_runtime_lag.ps1 -OutputPath H:\truf-diagnostics\lag.csv -IntervalSeconds 10 + +# Безопасный просмотр кандидатов на очистку Vite +.\cleanup_stale_agentui_vite.ps1 + +# Реальная очистка найденного точного набора +.\cleanup_stale_agentui_vite.ps1 -Apply +``` + +## Supervisor-команды + +Сначала запусти `.\attach_runtime.ps1`, затем используй команды ниже. + +| Команда | Назначение | +|---|---| +| `help` | Полная встроенная справка | +| `status` | Свежий status table | +| `watch` | Live status; `q` возвращает в prompt | +| `auth ` | Состояние auth pools | +| `logs [N]` | Последние `N` строк bounded-лога | +| `command ` | Фактическая child-команда, log и state paths | +| `start ` | Запустить остановленный source | +| `stop ` | Остановить и оставить остановленным | +| `restart ` | Перезапустить отдельный source | +| `pause ` | Остановить и отметить paused | +| `resume ` | Снять pause и запустить | +| `once ` | Один проход source с `--once` | +| `mode loop|once|repeat` | Изменить режим source | +| `set interval ` | Интервал repeat mode | +| `set restart on|off` | Автоматический restart после сбоя | +| `set restart_delay ` | Начальная задержка restart | +| `dashboard status|start|stop|restart` | Управление dashboard | +| `shutdown` | Полный coordinated shutdown | +| `quit` | В attach-режиме только отсоединиться | + +`reload` намеренно отключён. После изменения `config.yaml` или runtime-кода нужен полный `stop_runtime.ps1` + `start_runtime.ps1`. + +Source alias: `docker` означает `dockerhub`. + +## Статусы + +| Статус | Значение | +|---|---| +| `running` | Child сейчас работает | +| `waiting` | Ожидает следующего запуска/retry | +| `blocked` | Ждёт стабильной готовности PostgreSQL | +| `paused` | Остановлен командой `pause` | +| `done` | Успешный one-shot завершён | +| `failed` | Child завершился с ошибкой, restart выключен | + +`desired=running` показывает желаемое состояние. `rs` означает текущую серию ошибок / общее число automatic restarts. Старый `exit=1` рядом с уже `running` source относится к предыдущей попытке запуска. + +## Keycheck Recheck + +Формат: + +```text +recheck [type ...] [options] +``` + +Если type не указан, выполняется полный `--recheck-all` выбранного service. + +### Типы + +| Type | Что ставится в очередь | +|---|---| +| `network` | Текущие transient network statuses | +| `ratelimited` | Limited/rate-limited и связанные no-balance statuses | +| `unknown` | Unknown и no-context | +| `restricted` | Restricted | +| `nobalance` | No-balance/no-quota | +| `valid` или `alive` | Текущие alive credentials | +| `all` | Все известные credentials | +| `legacy-vertex` | Только GCP: импортировать и проверить отсутствующие legacy Vertex TXT credentials | + +### Опции + +| Опция | Значение | +|---|---| +| `--force` | Остановить уже работающий keycheck batch и начать этот | +| `--max-keys N` | Ограничить число credentials | +| `--proxy-file PATH` | Временно переопределить proxy file | +| `--no-resource-probe` | Отключить resource probe; сейчас используется Replicate | +| `--no-summary` | Не пересобирать summary/status projections после batch | + +`--input PATH` является legacy/offline compatibility option и в canonical PostgreSQL runtime не используется. + +### Примеры + +```text +# Повторить только network failures у всех providers +recheck all network + +# Перепроверить все текущие alive GCP credentials +recheck gcp valid + +# Полностью перепроверить Qwen +recheck qwen all + +# Проверить максимум 5 alive Replicate без resource probe +recheck replicate valid --max-keys 5 --no-resource-probe + +# Импортировать/дедуплицировать старые GCP Vertex TXT записи и проверить только их +recheck gcp legacy-vertex + +# Прервать текущий keycheck batch и запустить новый +recheck gcp valid --force +``` + +Scheduled keychecks запускаются раз в `3600` секунд. По умолчанию проверяются новые candidates и повторяются только `NETWORK`; alive/limited/unknown/restricted/no-balance автоматически каждый час не перепроверяются. + +## Текущие GCP Vertex Probes + +| Provider | Модели | Locations | Проверка | +|---|---|---|---| +| Google | `gemini-3.6-flash`, `gemini-3.1-pro-preview` | `global`, `us`, `eu` | `countTokens`, без генерации | +| Anthropic | `claude-opus-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-fable-5` | `global`, `us`, `eu`, `us-east5`, `europe-west1` | `rawPredict`, до 1 output token | + +Anthropic probe является реальным минимальным inference-вызовом и может иметь небольшой расход. + +## Куда идут данные + +1. Scanner sources создают result bundles. +2. `result-ingester` пишет findings и keycheck candidates в PostgreSQL. +3. Provider checker арендует candidate и выполняет API probe через `runtime\proxy.txt`. +4. Новый результат добавляется в append-only `keycheck_results`. +5. `keycheck_current_state` переключается на последний результат. +6. `jsonl-projector` создаёт compatibility JSONL. +7. Summary projection атомарно обновляет status TXT. + +PostgreSQL является source of truth. TXT/JSONL в `runtime\keychecks` являются compatibility projections, а не входом для обычного recheck. + +## Полезные пути + +| Путь | Назначение | +|---|---| +| `app\config.yaml` | Основная конфигурация runtime, sources и probes | +| `runtime\proxy.txt` | Proxy для provider checks | +| `runtime\logs\supervisor.status.txt` | Последний status snapshot | +| `runtime\logs\supervisor.log` | Supervisor log | +| `runtime\logs\keychecks.log` | Общий keycheck log | +| `runtime\keychecks\summary.tsv` | Текущий provider summary | +| `runtime\keychecks\alive_summary.tsv` | Краткий alive summary | +| `runtime\keychecks\` | Compatibility status/results files provider-а | +| `runtime\control\supervisor.instance.json` | Private control metadata; вручную не редактировать | +| `S:\postgres-data` | Canonical PostgreSQL cluster | +| `S:\scanner-work` | Scanner scratch/work area | + +## Безопасность + +- Не запускай provider scripts напрямую. +- Не передавай raw credentials через CLI. +- Не редактируй `supervisor.instance.json`. +- Для управления используй только authenticated supervisor. +- Для полного рестарта используй canonical start/stop scripts. +- Не удаляй PostgreSQL cluster или runtime queues вручную. diff --git a/WINDOWS_IMPORT.md b/WINDOWS_IMPORT.md new file mode 100644 index 0000000..899a140 --- /dev/null +++ b/WINDOWS_IMPORT.md @@ -0,0 +1,285 @@ +# Windows Snapshot Import + +## Current Status + +As of 2026-09-16, source capture succeeded, but the target import failed during +maintenance cleanup and was manually interrupted. The destination remains +**failed, unmarked and stopped**, not verified-stopped or ready for normal run. +The user then explicitly authorized removal of copied SQLite databases, +backups, `found_secrets` outputs and archival files, preserving Windows originals. + +**The staged snapshot is now intentionally incomplete: `files.tar` was deleted.** +Its retained manifest describes the original capture, not the reduced target. +Do not rewrite the manifest, retry import, restart the retained container, or +recapture/repopulate the removed copies automatically. The procedures below +describe the original full-snapshot workflow, not a resume procedure for this +pruned destination. Any future recovery needs a separately reviewed plan. + +| Execution evidence | Result | +| --- | --- | +| Source snapshot publication | Manifest published 2026-09-15T19:36:28.010915+00:00; source supervisor/PG stopped flags true | +| Approved manifest SHA-256 | `08344147133c37d4b6f404cf4fac3e59d58f94917f1fa58a77cbb68c36db7e8a` | +| Original capture inventory | 50,501 files, 34,851,776,467 bytes; 49,897 active and 604 archival files; 61 tables and 38 sequences | +| Retained PostgreSQL dump | `database.dump`, 2,619,119,892 bytes; not deleted or modified by cleanup | +| Failed import container | `63286fd554f832fd3a1f073e7c923977e209149e940b684cdbf35e4479ebd5ba`; exited 129, PID 0, restarts 0, restart policy `no` | +| Pinned runtime/cleanup image | `sha256:ecf1ee044fd6e936359a5955e0a42b452b8098a3f9d822272ab373b697761de2` | +| Cleanup verification | 635 original Windows files checked for presence/size and unchanged metadata; original PG control hashes unchanged; Windows `postgres.exe` count 0 | +| Retained destination verification | Metadata of 49,866 remaining inventory files and 1,883 PG files unchanged; PG control/config hashes unchanged; initialized marker absent; application remains stopped | +| Final verified-stopped acceptance | NOT ACHIEVED; cleanup does not repair the failed import | + +### Authorized Copy Cleanup + +Only these copied locations were removed on 2026-09-16: + +| Copied location/family | Files | Bytes removed | +| --- | ---: | ---: | +| `/data/windows-archive` including old SQLite backups and archived configurations | 604 | 10,827,425,254 | +| `/data/runtime-linux/results/scanner*.db` and associated WAL/SHM/journal files | 11 | 10,281,779,360 | +| `/data/runtime-linux/results/found_secrets.*`, including generations, manifest and publication ledger | 20 | 5,584,368,562 | +| Total from native `truf-docker_data` volume | 635 | 26,693,573,176 | +| Completed staging directory's `files.tar` on Windows D: | 1 | 34,917,959,680 | + +Original `D:\truf`, `S:\postgres-data` and source bundles were not deleted or +modified. Target PostgreSQL, its dump, translated configuration, credentials, +proxies, queues, other result streams, bundles, caches and their required +publication ledgers were retained. The one-off cleanup used a network-disabled +utility container with only the verified native target volume writable; it did +not run PostgreSQL, the importer, scanners, providers or application services. + +The volume gained approximately 26.69 GB of filesystem free space. Approximately +34.92 GB was freed on Windows D:. This did not compact the WSL VHDX on S: or +return all newly free ext4 blocks to the Windows host; S: reported +30,467,690,496 bytes free after cleanup. No WSL/storage reconfiguration was done. + +Removing `found_secrets` files does not reset PostgreSQL projector cursors or +pending append/rotation proofs. A future authorized startup must first address +that projection state explicitly; removing files or their SQLite ledger alone +is not a safe live-cursor reset. Do not erase PostgreSQL findings, counters or +pipeline evidence to make the removed files appear consistent. + +Reported regression evidence, not rerun by this documentation change: selected +Docker suite **663 passed, 7 Windows-only skipped**; synthetic snapshot tests +**58 passed**; pure config tests **12 passed**; host importer tests **66 passed, +1 POSIX-only skipped**. None is proof of this snapshot's capture or import E2E. + +## Scope And Paths + +The authorized operation copies the original logical database, proxies, secrets +and reviewed durable files. It does not move/delete the originals, migrate the +source schema, or execute providers, scanners, keycheckers or archived scripts. +Only exclusive PostgreSQL maintenance is allowed during capture/import. Leave +both the original supervisor and original PostgreSQL stopped after capture, +and the destination stopped after import. + +Current private staging directory: + +- Windows: `D:\truf-docker\docker\imports\windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3` +- WSL: `/mnt/d/truf-docker/docker/imports/windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3` +- Container: the same directory bound read-only at `/import`. + +A completed snapshot contains exactly `manifest.json`, `files.tar` and +`database.dump`. Do not add reports or other files inside it. Windows staging +remains private to the capturing account and SYSTEM; preserve its ACLs rather +than making it world-readable for Docker. `docker/imports/` is excluded from +Git and the image build context. Never put dump/tar contents, credentials, +proxy values, application data or raw logs in Git, images or terminal output. + +The directory listed above currently retains only `manifest.json` and +`database.dump` after the authorized cleanup. It is not a completed import input. + +The destination is the base Compose native Linux volume `truf-docker_data`, +mounted at `/data`, not a Windows bind mount. Paths in braces below are reviewed +families; optional archival inputs are copied only when present. + +| Original source | Destination within `/data` | +| --- | --- | +| `S:\postgres-data` via a full PG16 logical dump | `/data/postgres-linux`, independently initialized native Linux PG16 | +| `D:\truf\runtime\{results,queues,state,keychecks,postman_cache,result_spool}` | `/data/runtime-linux/{results,queues,state,keychecks,postman_cache,result_spool}` | +| `D:\truf\runtime\proxy.txt` | `/data/runtime-linux/proxy.txt` | +| `D:\truf\app\{secrets.yaml,trufflehog-custom-detectors.yaml}` | `/data/config/{secrets.yaml,trufflehog-custom-detectors.yaml}` | +| `D:\truf\app\config.yaml` | `/data/windows-archive/app/config.yaml`; translated profile at `/data/config/windows-import.yaml` | +| `S:\scanner-result-bundles\{tmp,ready,quarantine}` | `/data/scanner-result-bundles/{tmp,ready,quarantine}` | +| Reviewed archival inputs under `D:\truf` | `/data/windows-archive/` with their original relative paths | + +Archival scope includes `D:\truf\state`, `runtime\backups`, `runtime\imports`, +non-authority JSON reports from `runtime\control`, `app\.streamlit\config.toml`, +`.env.postgres`, `docker-compose.postgres.yml`, `runner_state.json`, +`runtime\keychecks.7z`, `runtime\orkey.txt`, `runtime\check-openrouter-keys.ps1`, +`runtime\*.md`, root `checked_*.txt`/`todo_*.txt`, root/app `scanner.db*`, +app `config.yaml.*`/`secrets.yaml.*`, root result projection families and +`*.publication-ledger.sqlite3*`. The legacy copy tree is also archival: +`D:\truf\runtime\keychecks \u2014 \u043a\u043e\u043f\u0438\u044f` +(Unicode escapes describe the actual folder name, not a literal shell path). +Within `runtime\state`, `scan_limiter*.db*`, `*.tmp*` and +`janitor.cursor.json` are archival only, never active Linux authority/state. + +Excluded: physical PGDATA/WAL, Windows PostgreSQL binaries/logs, live control +authority, locks/PIDs, `S:\scanner-work`, `runtime\downloads`, runtime git/traces/ +freeze-diagnostics, `gharchive_cache`, `.git`, `.opencode`, tests and code caches. +Ordinary logs are excluded outside retained result/keycheck projection families; +`scan_errors.log*` is deliberately durable data, not a diagnostic to display. +The manifest records the actual selected inventory and exclusion counts. + +The archived `.env.postgres` is never sourced or used for the target connection. +Provision generates `/data/postgres-password`; Linux uses this new +password and a different PG16 system identifier, not the source password or +physical cluster. Archived Windows configs are not executable runtime profiles. + +## Capture And Capacity Gates + +Main runs `D:\truf-docker\docker\windows_snapshot.py` using native PowerShell, +not context-mode or another Windows Job wrapper: original PostgreSQL correctly +refuses Job membership. The exporter temporarily starts only source maintenance +PostgreSQL and must confirm its stop before publication. Do not rerun capture +into the current attempt directory, reuse partial files, or kill a process that +is retaining authority while stop remains unconfirmed. + +After capture exits 0 and publishes its final manifest, main records its digest +from the trusted Windows path, then checks that the WSL-visible manifest has +the same digest. Do not replace the approved pin with a newly computed digest +merely to bypass a mismatch. Native PowerShell digest command: + +```powershell +(Get-FileHash -LiteralPath 'D:\truf-docker\docker\imports\windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3\manifest.json' -Algorithm SHA256).Hash.ToLowerInvariant() +``` + +Check Windows `D:` staging capacity and, independently, physical free space on +`S:`, which backs the Docker/WSL VHDX, and free space on native Linux `/data`. +Record the actual VHDX/daemon storage location; a large Linux `df` result does +not prove the Windows host can grow the VHDX. Budget its anticipated growth +while retaining at least **20 GiB physical free on S:**. The importer cannot +measure or enforce this host-side reserve. + +The Linux preflight requires `file_bytes + database_bytes + 20 GiB` free, using +source physical database size from manifest metadata. Older v1 metadata without +that size uses `max(24 GiB, 4 * dump_bytes)` as the database estimate. At least +**20 GiB must still be free on Linux after import**. Check both host and guest +capacity during and after restoration; compressed dump size alone is not a +capacity estimate. Do not delete original data to make space. + +## Offline Procedure + +Run these steps separately in WSL Bash only after main approves the capture and +capacity evidence. Use the existing Linux Docker daemon and already-built, +importer-integrated `truf-local:runtime` image; no builds or pulls here. Keep the +fixed project/directory below. Do not use the generic initialize/start procedure +in `DOCKER_MIGRATION.md` for this full-schema snapshot. + +```bash +TRUF_WINDOWS_SNAPSHOT='/mnt/d/truf-docker/docker/imports/windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3' +TRUF_WINDOWS_SNAPSHOT_SHA256='PENDING' + +dci() { + if [[ ! "${TRUF_WINDOWS_SNAPSHOT_SHA256:-}" =~ ^[0-9a-f]{64}$ ]]; then + printf '%s\n' 'STOP: set the approved lowercase manifest SHA-256.' >&2 + return 1 + fi + sudo -n env \ + TRUF_WINDOWS_SNAPSHOT="${TRUF_WINDOWS_SNAPSHOT:?Set the completed snapshot directory}" \ + TRUF_WINDOWS_SNAPSHOT_SHA256="$TRUF_WINDOWS_SNAPSHOT_SHA256" \ + docker compose \ + --project-name truf-docker \ + --project-directory /mnt/d/truf-docker \ + --env-file /dev/null \ + --file /mnt/d/truf-docker/compose.yaml \ + --file /mnt/d/truf-docker/compose.snapshot-import.yaml \ + "$@" +} + +sha256sum "$TRUF_WINDOWS_SNAPSHOT/manifest.json" +dci config --quiet +sudo -n docker image inspect --format '{{.Id}}' truf-local:runtime +sudo -n docker volume inspect --format '{{.Name}} {{.Driver}} {{.Mountpoint}}' truf-docker_data +sudo -n docker container inspect --format '{{.State.Status}}' truf-docker-snapshot-import +``` + +Replace `PENDING` with the previously approved pin, not a credential. The +`sudo -n env NAME=value ... docker compose` form explicitly passes the two +non-secret interpolation inputs even when sudo strips shell exports. Do not +use `sudo -E` or pass source connection/provider variables. `--env-file /dev/null` +prevents implicit checkout `.env` loading, but does not sanitize shell exports; +use a clean operator shell and no unreviewed Docker/Compose overrides. + +Main must separately confirm the image identity, **absence** of +`truf-docker_data`, and absence of the retained import container name before +provisioning. Only specific no-such-volume/no-such-container responses establish +absence; daemon/permission errors do not. If either already exists, stop for +review instead of adopting, overwriting or deleting it. + +```bash +dci run --rm --no-deps --pull never -T provision +``` + +Proceed only after successful provision. This unchanged base service creates +the private layout, generated password and empty placeholders, not an application +schema. **Do not call `initialize`, `import-secrets`, or normal `run` first.** +The importer uses `initialize-empty` internally and restores the entire custom +dump into a virgin schema before raw table/sequence comparison and permitted +target-only recovery/migrations. + +```bash +dci run --detach --no-deps --pull never -T \ + --name truf-docker-snapshot-import runtime +``` + +This is the single retained maintenance container: no `--rm`, no automatic +restart, no dependency startup, no healthcheck and `network_mode: none`. +The override preserves the base image/entrypoint, non-root UID/GID, read-only +rootfs, capabilities/security policy, native `/data` volume, tmpfs and resource +limits. It adds only read-only `/import`; `create_host_path: false` rejects a +missing source directory rather than silently creating one. + +Import starts with `/opt/truf/app/config.linux.yaml` in both the environment +and explicit `--config`. **Do not merge `compose.windows-import.yaml` here.** +The translated private profile does not exist at initial preflight; the importer +creates and selects it internally only after validating/extracting the snapshot. + +## Stopped Verification + +```bash +sudo -n docker container wait truf-docker-snapshot-import +sudo -n docker container inspect --format \ + 'status={{.State.Status}} exit={{.State.ExitCode}} oom={{.State.OOMKilled}} restarts={{.RestartCount}} network={{.HostConfig.NetworkMode}} restart={{.HostConfig.RestartPolicy.Name}} auto_remove={{.HostConfig.AutoRemove}}' \ + truf-docker-snapshot-import +``` + +Require `status=exited exit=0 oom=false restarts=0 network=none restart=no +auto_remove=false`. `wait` prints the container exit code; the command's own +exit status alone is not import success. Waiting may take hours or hold while +maintenance stop is uncertain. Do not impose a timeout that kills the container. + +Do not display raw `docker logs`, Compose logs, database logs, full environment +dumps or application data. Only structured importer numeric phase/count/byte +events and allowlisted aggregate/hash evidence are suitable for progress. +Phase 13 is emitted before final publication and is not success proof. + +Main must privately inspect the following evidence from the stopped retained +container, for example with `docker cp` into a separate owner-only evidence +directory outside `/import`. Do not start another runtime to inspect it; `health` +and `status` are live readiness actions, not stopped-import verification. + +- `/data/config/windows-import-manifest.json`: its byte SHA-256 equals the approved staging manifest pin; source stopped flags are true. The raw evidence, report and initialized marker all carry that same `manifest_sha256`. Report archive/dump hashes match this manifest and the verified staging files. +- `/data/config/windows-import-raw.json`: its SHA-256 matches report `raw_evidence_sha256`; `table_counts` equals manifest `database.table_counts`, `sequences_provided` is true and `sequences_verified` equals manifest `database.sequence_count`. The importer checks actual sequence values before transformations; this evidence records their verified count, not their values. Record only aggregate tables/rows/sequences. +- `/data/config/windows-import.yaml`: hash bytes without displaying values; SHA-256 matches report `config_sha256`. +- `/data/config/windows-import-report.json`: `status` is `verified-stopped`, `maintenance_stopped` is true, all `pipeline_after` counts are zero, final Linux reserve is at least 20 GiB, and cutover/migration/preserved-evidence checks succeeded. Record any reported fenced recovery or Postman rebasing; these may legitimately change final target counts or move incoming tmp/ready bundles after raw comparison. +- `/data/initialized.json`: exists as the last publication, has format `truf-container-data-v1`, `pg_major` 16 and the approved `manifest_sha256`; `import_report_sha256` matches the actual report bytes. Its system identifier equals report `linux_system_identifier` and differs from the source identifier. +- Confirm destination `/data/postgres-linux/postmaster.pid` is absent, original supervisor/PostgreSQL remain stopped, and host/guest space reserves still hold. Record all results in the PENDING table before declaring verified-stopped. + +## Failure And Release + +Any nonzero exit, OOM, missing/mismatched evidence or uncertain stop is not +verified-stopped. Retain the container, volume and private snapshot. A partial +import is failed/unmarked, not automatically resumable; early rejection can +leave no report. A report saying verified-stopped without a matching final +initialized marker and clean container exit is still insufficient. + +Do not automatically retry/restart, overwrite, delete, remove locks/markers, +initialize, prune, run `down --volumes`, or force-kill an authority-holding +importer. Review the retained state first; any new attempt needs an explicit +decision and separately approved fresh destination, not cleanup by this runbook. + +There is **no automatic normal run**. A future live run requires separate user +authorization after verified-stopped acceptance. Only then use base +`compose.yaml` plus `compose.windows-import.yaml`, without the snapshot override, +so runtime, health and status all select `/data/config/windows-import.yaml`. +Do not start either the original or destination supervisor as part of import. diff --git a/WORKER_OPERATOR_EXPERIENCE_HANDOFF.md b/WORKER_OPERATOR_EXPERIENCE_HANDOFF.md new file mode 100644 index 0000000..cac2ecb --- /dev/null +++ b/WORKER_OPERATOR_EXPERIENCE_HANDOFF.md @@ -0,0 +1,398 @@ +# Worker Operator Experience Handoff + +> Historical handoff. The current continuation entry point is +> `docs/session-handoff/README.md`. This file retains implementation and artifact +> provenance, but its stop point and immediate-next-actions section are obsolete. + +Last updated: 2026-09-25 + +This is the historical implementation record for the OpenSpec change +`add-worker-operator-experience`. For current continuation instructions, read +`docs/session-handoff/README.md`. Do not repeat completed production validation +or rebuild accepted artifacts unless a current verification fails. + +## User intent and constraints + +- Continue autonomously from this handoff and finish the change end to end. +- The user explicitly requested a file handoff because invoking conversation + compression appears to stop or destabilize all OpenCode sessions. Avoid + proactively invoking the compression tool in the continuation session. +- Workspace: `D:\truf-workers`. +- Use the configured SSH server named `sec` only. Never call or connect through + the configured server named `prod`. +- Never print or record tokens, credentials, the private admin prefix, raw + targets/findings, runtime YAML, or worker command lines containing auth data. +- Do not add a new masking, redaction, credential-sandbox, or other security + scope without explicit approval and an OpenSpec requirement. +- Do not remove or revert unrelated workspace files. The repository baseline is + entirely untracked (`git status --short` shows the whole tree as `??`), so Git + cannot provide a meaningful task-specific diff. +- Do not archive the OpenSpec change unless the user explicitly asks. Completing + tasks and reporting "ready to archive" is expected. + +## Current OpenSpec state + +- Change: `add-worker-operator-experience` +- Schema: `spec-driven` +- Artifact status: proposal, design, specs, and tasks are complete. +- Apply progress before final closure: 23/27 tasks complete. +- File: `openspec/changes/add-worker-operator-experience/tasks.md` +- Tasks 1.1 through 5.5 are checked. +- Remaining unchecked tasks: + - 6.1: validate the canonical from-zero operator guide. + - 6.2: complete test matrix, reproducible packages, manifest registration, + and documented identities. + - 6.3: bounded Windows and WSL/Docker production validation and restoration. + - 6.4: durable dated report with timings, watchdog evidence, snapshots, + transcript, known limits, rollout, and rollback. + +Do not check 6.1-6.4 until the remaining focused matrix, report, and strict +OpenSpec validation have passed. + +## Implemented scope + +The change now includes: + +- Versioned worker phase/events and canonical transitions. +- Monotonic sequence handling and JSON/NDJSON contracts. +- Unified bounded diagnostics with deterministic identities. +- PostgreSQL progress/diagnostic persistence and admin queries. +- Server-owned global/per-source assignment deadline policy. +- Private local state, logs, history, diagnostic artifacts, and retention. +- Status, attach, logs/history, JSON/NDJSON, bounded follow/tail CLI behavior. +- Cross-platform worker supervisor, control protocol, drain/stop, shutdown + receipts, stale instance handling, and recovery slots. +- Per-assignment contained runner, controller protocol, watchdog, timeout bundle + publication, restart adoption, and abandoned-root cleanup. +- Authenticated progress endpoint and diagnostic ingestion. +- Admin assignment/progress/diagnostic experience. +- Windows portable and Linux image packaging for the supervisor runtime. +- Canonical operator runbook in `docs/remote-worker-operations.md`. + +Important implementation files include: + +- `app/worker_contracts.py` +- `app/worker_local_state.py` +- `app/worker_supervisor.py` +- `app/worker_cli.py` +- `app/worker_assignment_runner.py` +- `app/remote_worker_client.py` +- `app/worker_api.py` +- `app/scanner_db.py` +- `app/admin_api.py` +- `app/worker_package.py` +- `app/worker_package_builder.py` +- `docker/verify_packaged_workers.py` +- Worker-related tests under `tests/` + +## Final correctness fixes + +### Recovered ready-bundle transition + +`WorkerSlot` could recover a published ready bundle while its persisted event +phase was still `assigned`. Upload code emitted `uploading` only from +`bundling/backoff`, then attempted the invalid transition +`assigned -> awaiting_receipt`. + +Fix in `app/remote_worker_client.py`: + +- Emit `UPLOADING` when the current event phase is `ASSIGNED`, as well as the + existing bundling/backoff cases. +- Regression in `tests/test_worker_api.py` validates the event sequence + `assigned -> uploading -> awaiting_receipt`. + +The focused worker API/local-state/supervisor suite passed 100 tests after this +fix. + +### Packaged E2E abandoned work invariant + +Completed runner roots are intentionally retained under `work/abandoned` for at +least 60 seconds; retention maintenance normally runs every 300 seconds. The E2E +harness incorrectly required the total work file count to be zero, causing a +false `linux_direct_claims_timeout` after Linux had correctly claimed both direct +assignments. + +Fix in `docker/verify_packaged_workers.py`: + +- Linux and Windows work-tree identities now include `active_entries`. +- Files/directories beneath top-level `abandoned` are retained but not active. +- Direct-assignment and final-cleanup predicates require zero active entries, + while preserving strict state and bundle identity checks. +- Outage marker waits also check worker liveness, so an exited worker fails + immediately rather than timing out after four minutes. + +### Windows `prepare-worker.ps1` ACL defect + +Testing a freshly extracted ZIP exposed a real release bug. The old generated +script ran `icacls ... /grant:r ... /T`; on descendants this produced +inheritance-only ACEs, returned success, and made packaged `python.exe` +inaccessible. + +Final fix in `app/worker_package_builder.py`: + +1. Set private inheritable full-control ACEs for the current user, SYSTEM, and + Administrators on the package root only. +2. Run `icacls (Join-Path $root '*') /inheritance:d /T /C` so each descendant + converts inherited ACLs to explicit protected ACLs with the correct file or + directory flags. + +Regression in `tests/test_worker_package.py` checks the generated script and, on +Windows, executes it and verifies `private_directory_ready(root)` plus +`private_file_ready(child)`. `tests/test_worker_package.py` passes 20 tests. + +Do not use the earlier `/reset /T` idea: inherited ACLs are not accepted because +runtime trust requires protected explicit ACLs. + +### Watchdog test timing stabilization + +The broad focused suite exposed two false failures because three tests created a +100 ms absolute watchdog deadline before runner protocol-root/state setup. Under +the complete Windows suite that setup could consume the deadline, exercising the +startup-deadline branch instead of the intended blocked-operation watchdog. + +Test-only changes in `tests/test_worker_assignment_runner.py`: + +- Affected tests: + - `test_watchdog_kills_while_state_persistence_is_blocked` + - `test_blocked_startup_gate_write_enters_preparing_timeout_result_path` + - `test_watchdog_kills_while_event_drain_is_blocked` +- Scan deadline: 1 second -> 2 seconds. +- Watchdog deadline: 0.1 second -> 1 second. +- Injected block: 0.4 second -> 1.4 seconds. +- Kill bound: 0.3 second -> 1.3 seconds. + +This preserves the independent watchdog assertion and does not weaken product +code. The exact three-test rerun passed: `3 passed in 5.17s`. + +## Accepted reproducible artifacts + +### Windows final pair: I and J + +Paths: + +- `build/operator-experience-validation/windows-i.zip` +- `build/operator-experience-validation/windows-i.zip.json` +- `build/operator-experience-validation/windows-j.zip` +- `build/operator-experience-validation/windows-j.zip.json` + +Both independently built archives are identical: + +- Bytes: `134850988` +- Archive SHA-256: + `6ea9290736a059f1e17d8e89d9cf83506fa4abe2ba2f3731a7422a7b0f386e97` +- Package manifest identity: + `78a962b2bd3fa411413c79e9a8ffb021608a08ff020b1ad851f4505ea634b2b6` +- Build-input identity: + `6991ebbce6ae758c2bdd19a6ae934335aa585a50f86b18ccde8d88bca40ce436` +- Raw `worker-package.json` SHA-256: + `e0b17d70fcb868fe39fac45ab6e05a17c6d40852e6034010fb63b6cab31f8a3c` + +Acceptance used a fresh extraction, not the builder output: + +- `build/pwe-final-i-extracted` +- The package's own corrected `prepare-worker.ps1` was run once. +- Direct package verification then passed. + +The older Windows G/H archives are obsolete for acceptance because they contain +the broken preparation script. Their package manifest identity happens to be the +same because the support script is outside that manifest, but their archive +identity is not accepted. Do not publish or register G/H as final Windows ZIPs. + +### Linux final pair: G and H + +Tags: + +- `truf-worker-test:operator-experience-final-3g` +- `truf-worker-test:operator-experience-final-3h` + +Both were built with provenance disabled and are reproducible: + +- Worker package identity: + `45588f2cf406b41b239cfa3b8a9dc83fe84b587229bc997b2729016e1f0dde42` +- Image manifest / accepted image ID: + `sha256:3a088f5743121d823aae132234a29730a84339cecbfda5fc601e8e942f9948c3` +- Config: + `sha256:687a1c4c51c1b962c7fa7ea0cc4b04d159e7ba4f94ef347940c9fb225f7cb87d` +- Raw `worker-package.json` SHA-256: + `ee926cce3c19e9e6094753f51fa902415bd7364c24fa649cd0c1b659c0aa4d60` + +Extracted final manifest: + +- `build/operator-experience-validation/linux-worker-package-g.json` + +Test image: + +- Tag: `truf-worker-test:operator-experience-final-3` +- ID: + `sha256:1a22c396dbf329e20caf77f88b7c7a310bda86befbcf3b917f10ded3720ee712` + +## Final packaged E2E + +Passed run: + +- Run ID: `35f3f52e232067c1` +- Safe summary: `build/pwe-35f3f52e232067c1/summary.json` +- Windows input: freshly extracted and prepared Windows I. +- Linux input: Linux G. +- Status: passed. +- Cleanup: complete. +- Foreign Docker state: unchanged. +- Windows and Linux normalized evidence matched. +- Restart, outage, durable bundle, direct assignment, direct bundle, receipt, + shutdown, local cleanup, and cross-platform evidence gates all passed. + +Do not copy raw target values from the summary into reports or chat. Only the +safe aggregate facts above are needed. + +All Docker resources from final and diagnosed failed runs were cleaned by exact +owned IDs/names. Some local `build/pwe-*` failure evidence directories remain and +are safe to leave. `build/pwe-final-g` may still have unusable ACLs after running +the old broken preparation script; do not use it. `build/pwe-final-i-extracted` +is the accepted extracted Windows directory. + +## Production validation evidence + +Bounded production validation was completed before final package acceptance and +production was restored afterward. + +Safe aggregate results: + +- Assignments issued: 34. +- Accepted: 33. +- One intentional expected expiry. +- Accepted assignments ingested, settled, and projected: 33. +- Unresolved, precommit, and quarantine counts: zero. +- Natural timeout evidence reservation: 1453. +- Full-stage progress/watchdog evidence reservation: 1455. + +Evidence files: + +- `build/operator-experience-validation/final-evidence.json` +- `build/operator-experience-validation/progress-v3-evidence.json` +- `build/operator-experience-validation/timeout-evidence.json` +- `build/operator-experience-validation/server-baseline.json` + +These files are the source for duration percentiles, phase/watchdog evidence, +diagnostic/admin snapshots, and reconciled counts in the final report. Derive +only aggregate/sanitized facts. Do not reproduce raw targets, findings, secrets, +or private route names. + +Final production state after restoration: + +- Operations controls: normal/open, revision 126. +- Standard WSL production worker user: enabled, assignment cap 1. +- Standard production device: enabled and not revoked. +- Temporary validation identities: disabled/revoked. +- Runtime canonical health: healthy. +- Edge remained up. + +Do not repeat production assignments merely to write the report. Existing +evidence is sufficient. + +## Registered trusted manifests + +Registration was completed only after the final packaged E2E passed, using SSH +server `sec` only. + +Remote paths: + +- `/etc/truf/worker-packages/linux-worker-package-v2.json` +- `/etc/truf/worker-packages/windows-worker-package-v3.json` + +Final remote SHA-256 values match the accepted manifests: + +- Linux: `ee926cce3c19e9e6094753f51fa902415bd7364c24fa649cd0c1b659c0aa4d60` +- Windows: `e0b17d70fcb868fe39fac45ab6e05a17c6d40852e6034010fb63b6cab31f8a3c` + +Both are `root:root` mode `0644`. Existing +`.pre-operator-experience` backups were preserved unchanged. Upload temp files +were removed. After registration, canonical runtime health succeeded and Docker +reported `truf-docker-runtime-1` healthy. No restart or config mutation was +needed. + +## Test state + +Completed checks: + +- Worker API/local-state/supervisor focused suite: 100 passed. +- Worker package tests after ACL fix: 20 passed. +- Exact three watchdog timing tests after stabilization: 3 passed. +- Full packaged Windows/Linux E2E: passed, run `35f3f52e232067c1`. +- Production health after final manifest registration: passed. + +The broad focused matrix was run before the watchdog test timing patch: + +```powershell +python -B -m pytest tests/test_worker_api.py tests/test_worker_api_runtime.py tests/test_worker_assignment.py tests/test_worker_assignment_runner.py tests/test_worker_cli.py tests/test_worker_contracts.py tests/test_worker_local_state.py tests/test_worker_observability_db.py tests/test_worker_package.py tests/test_worker_runner_handoff_linux.py tests/test_worker_supervisor.py tests/test_remote_worker_db.py tests/test_scan_execution.py tests/test_admin_api.py -q +``` + +Result before the timing-only patch: + +- 353 passed. +- 3 skipped. +- 2 false timing failures described above. + +The two failures and the nearby equivalent test pass after the patch, but the +complete 14-file command has not yet been rerun. This is the exact current stop +point. + +An unrestricted repository-wide pytest run is not a useful release gate in this +checkout because unrelated private/generated assets and platform assumptions are +absent. Its known baseline was `3031 passed, 134 skipped, 68 failed`. Do not try +to fix unrelated failures as part of this change. The focused change matrix, +packaged E2E, production proof, and strict OpenSpec validation are the gates. + +## Immediate next actions + +1. Rerun the exact 14-file focused matrix shown above. Expected result after the + timing patch is 355 passed and 3 skipped. If it fails, diagnose only genuine + worker-operator regressions; do not broaden scope. +2. Create the durable report: + `docs/worker-operator-experience-validation-2026-09-24.md`. +3. In the report, include only sanitized aggregate evidence: + - Scope and acceptance criteria. + - Final Windows I/J and Linux G/H identities from this handoff. + - Packaged E2E run `35f3f52e232067c1` and cleanup/foreign-state result. + - Production issued/accepted/reconciled counts. + - Duration percentiles derived from `final-evidence.json`. + - Watchdog/full-stage evidence from `progress-v3-evidence.json`. + - Natural timeout evidence from `timeout-evidence.json`. + - Diagnostic/admin snapshot facts without private content. + - Sanitized operator command transcript. + - Known limits, especially no public registry/auto-updater and intentional + abandoned-root retention. + - Rollout and rollback/restoration facts, controls revision 126, and final + healthy state. +4. Re-read `docs/remote-worker-operations.md` against task 6.1. It already covers + package acquisition/build, Windows preparation, install/first run, lifecycle, + status/attach/logs/history, phases/deadlines, diagnostics, drain/stop, + recovery, update, and removal. Make only a minimal correction if the final + artifact/report facts expose an actual gap. +5. Run strict validation: + + ```powershell + openspec validate add-worker-operator-experience --strict + ``` + +6. If the focused matrix, report, runbook review, and strict validation pass, + change only task checkboxes 6.1-6.4 in + `openspec/changes/add-worker-operator-experience/tasks.md` from `[ ]` to `[x]`. +7. Re-run `openspec instructions apply --change "add-worker-operator-experience" --json` + and confirm progress 27/27 with state `all_done`. +8. Give the user a concise completion result and say the change is ready to + archive. Do not archive it without an explicit request. + +## Report safety checklist + +Before saving or quoting the final report, verify it contains none of: + +- Tokens or credentials. +- Raw worker targets or findings. +- Runtime YAML or secret environment values. +- The private admin route prefix. +- Worker argv/auth command lines. +- Unbounded log or diagnostic bodies. + +Allowed report content includes hashes, aggregate counts, reservation numeric +IDs used as evidence references, phase names, durations/percentiles, safe test +counts, generic command names, and public artifact paths within this workspace. diff --git a/WORKSPACE.md b/WORKSPACE.md new file mode 100644 index 0000000..eac807f --- /dev/null +++ b/WORKSPACE.md @@ -0,0 +1,40 @@ +# Remote Worker Development Workspace + +This is an independent source-only snapshot of the current `D:\truf-docker` working tree, not a copy of its running system. + +## Snapshot + +- Source HEAD for provenance: `1b3c7fc4948c5cf2fc389065db3693a27b65300c`. +- Current modified and selected untracked source files are included; this snapshot is not equivalent to that commit alone. +- Copied 325 files, 8,150,338 bytes (about 7.8 MiB), with SHA-256 equality checked for every copied file. +- Source-side deletions are preserved, including the absence of `app/config.yaml`. +- No database, PGDATA, runtime directory, finding/keycheck output, logs, imports, caches, dependencies, or executable binaries were copied. +- No actual `.env`, secrets file, provider credential pool, or source Git history was copied. The tracked `.env.postgres.example` is only a template. +- Git was initialized independently. No commit, remote, runtime container, or Docker data volume was created for this workspace. +- Source `.opencode` skills, the old session handoff, and the loose operator note were not copied. Existing OpenSpec change artifacts remain unchanged and unarchived. + +## Active Plan + +`openspec/changes/add-minimal-remote-scan-workers/` contains the completed proposal, design, requirements, implementation checklist, and isolated worker implementation. It preserves existing scan and server-side keycheck logic. + +## Safe Local Checks + +Run from this directory: + +```powershell +python -I -S -B docker/test_verify.py -v +python -I -S -B tests/container_unit.py --check-selection +openspec validate add-minimal-remote-scan-workers --strict --no-interactive +``` + +The first two commands use the standard library only, do not import the application, and do not start Docker, PostgreSQL, a scanner, or provider checks. The selection check validates test declarations, not their execution. + +The original planning-stage verification passed. Current implementation evidence is recorded by the change checklist and isolated test outputs, including cross-platform packaged-client scans and the empty-database end-to-end gates. + +## Before Runtime Testing + +The inherited deployment files are SOURCE REFERENCES, not an isolated test setup. In particular, `compose.yaml` still names the production-style `truf-docker` project and shared `truf-local:*` image tags; `docker-compose.postgres.yml` and import overrides must not be used here. Do not run plain `docker compose up`, import a snapshot, invoke old native launchers, or run unrestricted pytest. + +The first implementation tasks must establish unique test project/image/volume names, neutral configuration, scrubbed inherited credentials/DSNs/proxies, disabled live discovery, and synthetic source/provider transports. Review existing `compose.e2e.yaml`, `docker/verify.py`, and `tests/container_unit.py` for reuse before adding any new test infrastructure. Never mount `D:\truf`, `D:\truf-docker`, their runtime directories, or existing Docker data volumes. A future test database must initialize empty and contain only synthetic fixtures. + +Local test implementation and worker packaging must use this workspace, not the active source or runtime. Production deployment/import and archiving unrelated changes require separate authorization. diff --git a/app/.streamlit/config.toml b/app/.streamlit/config.toml new file mode 100644 index 0000000..543ee06 --- /dev/null +++ b/app/.streamlit/config.toml @@ -0,0 +1,10 @@ +[server] +headless = true +address = "127.0.0.1" +port = 5000 + +[theme] +base = "light" + +[browser] +gatherUsageStats = false diff --git a/app/CHEATSHEET.md b/app/CHEATSHEET.md new file mode 100644 index 0000000..20baf08 --- /dev/null +++ b/app/CHEATSHEET.md @@ -0,0 +1,289 @@ +# Truf Runtime Operations + +All scanner, keycheck, dashboard, and PostgreSQL lifecycle mutation is owned by `supervisor.py`. Direct `console_runner.py`, mutating `keycheck_runner.py`, and legacy `app.py` controls are retired. + +PostgreSQL is the sole authority for scan results, queue completion, keycheck results, and keycheck current state. Scanner sources publish private version-2 bundles to `S:\scanner-result-bundles`; the singleton result ingester commits them transactionally. JSONL and status files are asynchronous, rebuildable compatibility projections and may lag without rolling back a committed scan. + +The scanner has `max_active_scans=3` guaranteed fair permits plus at most one memory-gated non-Docker bonus permit. A permit covers staging, TruffleHog, normalization, bundle fsync, and the atomic ready rename only. PostgreSQL ingestion, JSONL projection, and keychecks do not hold scan permits. + +Run commands from `D:\truf\app` unless a full path is shown. + +## Start And Stop + +Canonical production start: + +```powershell +..\start_runtime.ps1 +``` + +Canonical full coordinated shutdown: + +```powershell +..\stop_runtime.ps1 +``` + +Equivalent authenticated launch after cluster identity has been verified: + +```powershell +python -I -S -B runtime_bootstrap.py supervisor -- --runtime-bootstrap-entrypoint D:\truf\app\supervisor.py --config config.yaml --background --no-dashboard --with-postgres +``` + +Direct read/control operations remain supported by `supervisor.py`: + +```powershell +python supervisor.py --config config.yaml --background-status +..\attach_runtime.ps1 +python supervisor.py --config config.yaml --cmd "status" +python supervisor.py --config config.yaml --stop-background --with-postgres +``` + +In an attached prompt, `q` only detaches; `shutdown` requests full coordinated shutdown. Prefer `..\stop_runtime.ps1` for canonical full shutdown. + +The dashboard is currently disabled. To use the read-only dashboard, enable it in `config.yaml` and perform a coordinated runtime restart. The legacy scanner UI is intentionally retired. + +## Source Commands + +Use the foreground supervisor prompt or authenticated `--cmd` requests: + +```powershell +python supervisor.py --config config.yaml --cmd "status" +python supervisor.py --config config.yaml --cmd "start github" +python supervisor.py --config config.yaml --cmd "once gitlab" +python supervisor.py --config config.yaml --cmd "restart dockerhub" +python supervisor.py --config config.yaml --cmd "pause npm" +python supervisor.py --config config.yaml --cmd "resume npm" +python supervisor.py --config config.yaml --cmd "stop package_git" +python supervisor.py --config config.yaml --cmd "stop all" +python supervisor.py --config config.yaml --cmd "start all" +python supervisor.py --config config.yaml --cmd "logs pypi 80" +python supervisor.py --config config.yaml --cmd "command github" +``` + +`stop all` stops managed children while the supervisor and PostgreSQL remain running. `start all` includes `pypi`; do not use it when `pypi` must remain stopped. + +Configure discovery mode, queries, custom target files, timeouts, workers, and source-specific arguments in `config.yaml` before starting or restarting a source. Do not pass tokens or mutable scan options through a direct runner command. + +## Keychecks + +The supervisor manages keychecks as the `keychecks` pseudo-source: + +```powershell +python supervisor.py --config config.yaml --cmd "start keychecks" +python supervisor.py --config config.yaml --cmd "recheck all network" +python supervisor.py --config config.yaml --cmd "recheck gemini all --max-keys 100" +python supervisor.py --config config.yaml --cmd "recheck replicate valid --max-keys 25" +python supervisor.py --config config.yaml --cmd "recheck all --summary-only" +python supervisor.py --config config.yaml --cmd "logs keychecks 80" +``` + +Provider probe arguments belong under `keychecks.service_args` in `config.yaml`. Normal providers claim fenced PostgreSQL `keycheck_candidates` and commit `keycheck_results` plus `keycheck_current_state` directly. Files under `D:\truf\runtime\keychecks` are compatibility projections, not current-state authority. At most four provider children run concurrently, each with a bounded candidate slice so later services cannot starve. + +## Configuration And Secrets + +Primary files: + +```text +D:\truf\app\config.yaml +D:\truf\app\secrets.yaml +D:\truf\.env.postgres +D:\truf\runtime\proxy.txt +``` + +`config.yaml` contains paths, source settings, and auth-pool names. Actual source tokens belong in `secrets.yaml`; PostgreSQL credentials belong in `.env.postgres`. Managed children receive one canonical loopback PostgreSQL DSN after all configurable environment overrides. + +Important runtime paths: + +```text +D:\truf\runtime\results +S:\scanner-result-bundles +D:\truf\runtime\result_spool (legacy import compatibility only) +D:\truf\runtime\queues +D:\truf\runtime\state +D:\truf\runtime\logs +D:\truf\runtime\control +D:\truf\runtime\keychecks +D:\truf\runtime\postman_cache +D:\truf\runtime\postgres\data +D:\truf\tmp +``` + +Runtime startup performs read-only ACL/owner/reparse preflight and never repairs paths. + +### Remote Assignment Capacity + +`global.result_bundle_max_event_bytes` and `supervisor.worker_api.max_bundle_bytes` are hard per-bundle limits and remain 64 MiB. They are not admission reservations. Each unresolved remote assignment instead charges the persisted `global.remote_assignment_reserve_bytes` baseline of 2 MiB on both the bundle and projection byte axes; local scans retain their existing worst-case reservation behavior. + +Remote admission enforces `global.remote_assignment_max_active: 50` atomically in addition to each user's typed `active_assignment_cap`. The intended 50-assignment user must therefore have its cap set to 50 through the authenticated worker administration path. Lower either cap to reduce concurrency; do not raise the global cap above the validated maximum. + +A valid remote bundle larger than 2 MiB atomically expands its persisted bundle charge to actual bytes before the server returns an acceptance receipt. Temporary aggregate bundle exhaustion returns retryable capacity backpressure, and the worker must retry the identical durable upload. Projection serialization similarly expands a leased job to exact aggregate bytes before any append. If projection capacity is unavailable, the untouched job returns to `pending`; capacity backpressure alone never quarantines it. + +The production keycheck limits of 131,072 items and 128 MiB cover fifty baseline candidate reservations. Bundle and projection aggregate capacities and projection headroom remain independent safety bounds. Runtime-document validation rejects a hard bundle limit above 64 MiB, a remote baseline below 2 MiB or above the hard limit, a global cap above 50, and any aggregate axis that cannot hold all configured baselines. + +## Authority Model + +One cross-session lock is derived only from the canonical bundled PostgreSQL data directory. It is held by every lifecycle-owning foreground/background supervisor and by PostgreSQL bootstrap/verification, migration, reconciliation, and offline hardening. Changing control directory, instance file, or port cannot split authority; different data directories have independent locks. + +The configurable control-directory lock remains a secondary per-instance safety layer. Duplicate launch failure never sends coordinated shutdown to a different owner. + +Before spawn, the launcher captures exact config, supervisor, and code-manifest authority. The manifest also covers the result bundle, ingester, projector, keycheck candidate, and isolated janitor modules. + +The child completes canonical DSN validation and all controller/backend/source/keycheck/dashboard construction before publishing `ACTIVATING`. PostgreSQL, dashboard, and source ticks are forbidden until authenticated parent activation and exact post-activation recheck publish `ACTIVE`. + +If activation becomes uncertain, rollback uses authenticated shutdown and the full configured deadline. It never force-terminates an exact published candidate that may be active. An unconfirmed exact candidate is left running and reported rather than risking an orphaned PostgreSQL tree. + +Authenticated shutdown atomically publishes `STOPPING` and closes all start gates under the control lock before acknowledgement. Mutating commands and snapshot-triggered polls cannot start children after this transition. + +Code/config drift closes start gates and triggers safe shutdown. Authenticated shutdown remains available through private instance credentials, endpoint binding, and retained process identity even when on-disk code changed. Tokens, raw DSNs, and secret-derived hashes are not written to logs. + +## Child Authentication + +Managed scanner, result-ingester, JSONL-projector, keycheck, janitor, and dashboard children must prove all of the following before mutation. The janitor receives no database capability and persists only an exact-private bounded local cursor: + +1. Private per-instance metadata matches inherited instance credentials. +2. The retained supervisor process and config command line match metadata. +3. The supervisor handshake reports `ACTIVE`. +4. Config, supervisor, and complete code-manifest hashes match. +5. `SCANNER_DB_URL`, `DATABASE_URL`, and the inherited managed DSN name the same canonical loopback PostgreSQL authority. + +An environment marker by itself grants no authority. Empty or foreign DSNs fail before application files, `ScannerDB`, dependency checks, provider checks, or scanning. + +## PostgreSQL Maintenance + +One-time offline cluster identity binding, with all runtime processes stopped: + +```powershell +python postgres_runtime.py bootstrap --config config.yaml +``` + +Read-only identity verification: + +```powershell +python postgres_runtime.py verify --config config.yaml +``` + +Identity-verified maintenance start/stop, without scanner or keycheck children: + +```powershell +python postgres_runtime.py maintenance-start --config config.yaml +python postgres_runtime.py maintenance-stop --config config.yaml +``` + +### Docker Depth Experiment Operator + +Keep `sources.dockerhub.docker_depth_experiment.enabled: false` while reviewing and applying the cohort and hold. From `D:\truf\app`, use the existing private `runtime\state` directory and always stop maintenance PostgreSQL in a `finally` step if an operator command fails: + +```powershell +..\stop_runtime.ps1 +python postgres_runtime.py maintenance-start --config config.yaml +python docker_depth_operator.py --config config.yaml --status +python docker_depth_operator.py --config config.yaml --generate-cohort-manifest D:\truf\runtime\state\docker-depth-cohort-review.json +$cohortSha256 = Read-Host 'Reviewed cohort SHA-256' +python docker_depth_operator.py --config config.yaml --apply-cohort-manifest D:\truf\runtime\state\docker-depth-cohort-review.json --approve-sha256 $cohortSha256 --apply --sources-stopped +python docker_depth_operator.py --config config.yaml --generate-hold-manifest D:\truf\runtime\state\docker-depth-hold-review.json +$holdSha256 = Read-Host 'Reviewed hold SHA-256' +python docker_depth_operator.py --config config.yaml --apply-hold-manifest D:\truf\runtime\state\docker-depth-hold-review.json --approve-sha256 $holdSha256 --apply --sources-stopped +python postgres_runtime.py maintenance-stop --config config.yaml +``` + +Review the private files out of band and approve exactly the SHA-256 printed by their generation commands. The operator has no DSN option, never prints queries or targets, and never starts or stops PostgreSQL. After the hold apply and maintenance stop, changing only `enabled` from `false` to `true` is the separate activation decision; its semantic `config_sha256` must remain unchanged. Run `--status` between another maintenance start/stop pair to verify that hash before a separately approved `..\start_runtime.ps1`. + +An attempt-limit hold caused by the retired zero-graph resolver defect has a separate one-time reviewed recovery. Keep enabled config, stop sources, start maintenance PostgreSQL, review the private manifest, and approve only its exact printed SHA-256: + +```powershell +python docker_depth_operator.py --config config.yaml --generate-resolver-refund-manifest D:\truf\runtime\state\docker-depth-resolver-refund-review.json +$refundSha256 = Read-Host 'Reviewed resolver refund SHA-256' +python docker_depth_operator.py --config config.yaml --apply-resolver-refund-manifest D:\truf\runtime\state\docker-depth-resolver-refund-review.json --approve-sha256 $refundSha256 --apply --sources-stopped +``` + +A genuine remote `resolver_attempt_limit` hold uses a separate reviewed disposition. The manifest deterministically selects the next fresh repository or records terminal remote-unavailable scarcity when none remains: + +```powershell +python docker_depth_operator.py --config config.yaml --generate-resolver-disposition-manifest D:\truf\runtime\state\docker-depth-resolver-disposition-review.json +$dispositionSha256 = Read-Host 'Reviewed resolver disposition SHA-256' +python docker_depth_operator.py --config config.yaml --apply-resolver-disposition-manifest D:\truf\runtime\state\docker-depth-resolver-disposition-review.json --approve-sha256 $dispositionSha256 --apply --sources-stopped +``` + +Release is reviewed only after the experiment reaches `completed` under enabled config: + +```powershell +..\stop_runtime.ps1 +python postgres_runtime.py maintenance-start --config config.yaml +python docker_depth_operator.py --config config.yaml --generate-reactivation-manifest D:\truf\runtime\state\docker-depth-reactivation-review.json +$reactivationSha256 = Read-Host 'Reviewed reactivation SHA-256' +python docker_depth_operator.py --config config.yaml --apply-reactivation-manifest D:\truf\runtime\state\docker-depth-reactivation-review.json --approve-sha256 $reactivationSha256 --apply --sources-stopped +python postgres_runtime.py maintenance-stop --config config.yaml +``` + +Runtime-safety schema migration, with supervisor, dashboard, scanner, and keycheck sessions stopped: + +```powershell +python migrate_runtime_safety.py --config config.yaml --apply --sources-stopped +``` + +Final cutover refuses a nonempty legacy result spool, any `scan_publication_outbox` row, any legacy `raw_result_json` row, or a prepared legacy JSONL-ledger append. Runtime workers refuse to start until the migration records the singleton PostgreSQL cutover marker. + +Import bounded batches from the retired durable spool until `remaining=0`: + +```powershell +python migrate_runtime_safety.py --config config.yaml --import-legacy-spool --max-rows 1000 --apply --sources-stopped +``` + +Convert bounded legacy raw rows to normalized-v2 data, then rerun the normal migration to restore the cutover marker: + +```powershell +python migrate_runtime_safety.py --config config.yaml --backfill-normalized-results --max-rows 1000 --max-bytes 201326592 --max-seconds 30 --apply --sources-stopped +``` + +If the cutover gate reports legacy outbox rows, project only that bounded backlog offline before retrying migration: + +```powershell +python migrate_runtime_safety.py --config config.yaml --drain-legacy-outbox --max-rows 1000 --apply --sources-stopped +``` + +If a legacy event is too expensive for the per-finding ledger drain, first apply the additive schema (the command remains nonzero while the gate is closed), then transfer exact outbox references into the singleton projector queue without rewriting historical payloads: + +```powershell +python migrate_runtime_safety.py --config config.yaml --import-legacy-outbox-to-projection --max-rows 1000 --apply --sources-stopped +``` + +Build a full bounded, resumable PostgreSQL-derived projection in a dedicated empty directory. Repeat until `completed=true`; live projection files are never overwritten: + +```powershell +python migrate_runtime_safety.py --config config.yaml --rebuild-jsonl-output "D:\truf\runtime\rebuild" --max-rows 1000 --max-bytes 201326592 --max-seconds 30 --apply --sources-stopped +``` + +Quarantine review accepts only a private `truf-pipeline-quarantine-review-v1` manifest with exact `id`, `reason_code`, `payload_sha256`, and `action` (`discard`, deterministic `retry`, or Docker layer `rescan`) entries. `rescan` retires the stale bundle and returns its immutable target to the normal fresh-claim path: + +```powershell +python migrate_runtime_safety.py --config config.yaml --review-pipeline-quarantine "D:\review\quarantine.json" --max-rows 1000 --apply --sources-stopped +``` + +Bounded todo reconciliation: + +```powershell +python migrate_runtime_safety.py --config config.yaml --apply --sources-stopped --todo "D:\truf\runtime\queues\todo_github.txt" --source github --platform github --max-rows 1000 --max-bytes 4194304 --max-seconds 5 +``` + +Offline layout hardening creates required directories parent-first, recursively hardens runtime-owned trees, and hardens existing config, secrets, proxy, detector config, and PostgreSQL environment files: + +```powershell +python migrate_runtime_safety.py --config config.yaml --harden-runtime --apply --sources-stopped +``` + +Maintenance and runtime exclude each other through the same cluster authority lock. PostgreSQL sessions use `search_path=public`; `pg_catalog` keeps implicit precedence, authority built-ins are explicitly qualified, and PUBLIC `CREATE` on `public` fails closed. + +`D:\truf\docker-compose.postgres.yml` is a disabled, noncanonical manual-recovery fixture. It is behind the `noncanonical-manual-recovery` profile, has no restart policy, requires an explicit unused `TRUF_DOCKER_POSTGRES_PORT`, and binds data under `D:\truf`; never use it to start or replace the supervisor-owned cluster. Any legacy Docker named volume is intentionally left untouched. + +## Dashboard Secrecy + +Default dashboard queries and frames do not contain raw credentials. Raw scanner and validation values are available only behind explicit default-false per-session reveal controls with a local warning. Treat the database itself as sensitive because persisted findings still contain raw values. + +The dashboard binds to loopback only. Do not expose it through `0.0.0.0`, a reverse proxy, screenshots, or shared logs. + +## Cleanup + +The authenticated `janitor` child is the only stale-tree recovery worker. It does not import `scanner`, and it enforces exact PID/creation-time/executable identities plus enumeration, candidate, entry, byte, time, and depth budgets. Its exact-private local cursor rotates layouts after every inspected entry so a huge first layout cannot starve later trees. Unknown identity retains the tree. Sources may only attempt bounded cleanup of a directory they just used; there is no startup sweep, low-space sweep, periodic supervisor scanner import, or `atexit` cleanup. + +Definitively aborted unreferenced admission intents and deleted artifacts are retired in bounded keyset batches after 30 days. Retirement updates an aggregate SHA-256 chain and count before deleting exact rows; pending, committed/referenced, active, and open-quarantine authority is never eligible. + +Do not manually delete results, result spool events, queue state, PostgreSQL data, or control metadata. Use authenticated coordinated shutdown before offline maintenance. diff --git a/app/DETECTOR_NOTES.md b/app/DETECTOR_NOTES.md new file mode 100644 index 0000000..f0770aa --- /dev/null +++ b/app/DETECTOR_NOTES.md @@ -0,0 +1,351 @@ +# Detector Notes + +Working notes about TruffleHog detector behavior and local post-processing ideas. + +## GitHub / GitLab Noise + +Current TruffleHog source contains both modern and legacy detectors. + +GitHub v2 detects modern prefixed PATs: + +```text +(ghp|gho|ghu|ghs|ghr|github_pat)_[a-zA-Z0-9_]{36,255} +``` + +GitHub v1 detects legacy 40-character hex tokens near words such as `github`, `gh`, `pat`, or `token`: + +```text +(?:github|gh|pat|token).{0,40}([a-f0-9]{40}) +``` + +GitHubOauth2 detects a 20-character client id and a 40-character client secret near `github`: + +```text +client_id: [a-zA-Z0-9]{20} +client_secret: [a-f0-9]{40} +Raw = client_id +RawV2 = client_id + client_secret +``` + +GitLab v2 detects modern PATs: + +```text +glpat-[a-zA-Z0-9\-=_]{20,22} +``` + +GitLab v1 detects any 20-22 character token-like value near `gitlab` and skips `glpat-` so v2 can handle it: + +```text +gitlab ... ([a-zA-Z0-9\-=_]{20,22}) +``` + +Observed local results show high false-positive volume for unverified GitHub v1, GitHubOauth2, and GitLab v1 detections. The current scanner post-filter drops unverified GitHub/GitLab findings that do not match known modern token prefixes. This reduces noise but can hide real legacy/OAuth credentials if verification cannot run. + +TODO: Prefer a confidence model over hard dropping: + +```text +verified -> high confidence +modern prefix shape -> high/medium confidence +legacy GitHub v1 / GitHubOauth2 / GitLab v1 -> low confidence unless verified +``` + +Dashboard should hide low-confidence findings by default, but allow explicit review. + +## GCP + +### GCP service account JSON + +Detector: `GCP` + +TruffleHog detects JSON blobs containing `auth_provider_x509_cert_url` and parses service-account style credentials. + +Useful fields already present in the JSON: + +```text +type +project_id +private_key_id +private_key +client_email +client_id +auth_uri +token_uri +auth_provider_x509_cert_url +client_x509_cert_url +``` + +TruffleHog output behavior: + +```text +Raw = client_email, or full key JSON if client_email is missing +RawV2 = full cleaned credential JSON +Redacted = client_email +ExtraData.project = project_id +AnalysisInfo.principal = client_email +AnalysisInfo.type = type +``` + +Practical enrichment fields: + +```text +gcp_project_id +gcp_client_email +gcp_client_id +gcp_private_key_id +gcp_credential_type +``` + +This detector has enough context to verify/function without extra source-code lookup if `RawV2` is preserved. + +### GCP Application Default Credentials + +Detector: `GCPApplicationDefaultCredentials` + +TruffleHog detects ADC JSON containing `client_secret` and `.apps.googleusercontent.com` client IDs. + +Useful fields: + +```text +client_id +client_secret +refresh_token +type +``` + +TruffleHog output behavior: + +```text +Raw = client_id without .apps.googleusercontent.com suffix +RawV2 = client_id_without_suffix + refresh_token +Redacted = shortened refresh_token +ExtraData may contain verification details when verified +``` + +Risk: `RawV2` is concatenated and does not retain `client_secret` cleanly. The raw finding JSON may not be enough to reconstruct the original ADC JSON unless the source line/file is available. + +Practical enrichment fields: + +```text +gcp_client_id +gcp_refresh_token_redacted +gcp_credential_type +``` + +TODO: For ADC findings, use source context around the finding to parse the whole JSON and preserve `client_secret`/`refresh_token` as structured fields. + +### Google AQ authentication keys + +`AQ.` credentials are currently classified through the Gemini Developer API at +`generativelanguage.googleapis.com`. That result does not establish Vertex AI access. + +TODO: Add a separate Vertex AI Express probe for `AQ.` credentials against the supported +`aiplatform.googleapis.com` key-authenticated methods. Keep Gemini Developer API and Vertex +results independent, and do not infer access to the full project/location-scoped Vertex API +from the key prefix or from a successful Gemini Developer API check. + +## Azure + +### Azure Container Registry + +Detector: `AzureContainerRegistry` + +Detector finds registry hosts and ACR password-like values. + +Patterns: + +```text +registry: .azurecr.io +password: [a-zA-Z0-9+/]{42}+ACR[a-zA-Z0-9]{6} +``` + +TruffleHog output behavior: + +```text +Raw = password +RawV2 = {"username":"","password":""} +Redacted = registry name +``` + +Verification uses: + +```text +https://.azurecr.io/v2/ +BasicAuth(username=, password=) +``` + +Practical enrichment fields: + +```text +azure_acr_registry +azure_acr_login_server = .azurecr.io +``` + +This detector has enough context in `RawV2` to be useful. + +### Azure OpenAI + +Detector: `AzureOpenAI` + +Detector finds API keys and Azure OpenAI endpoints. + +Patterns: + +```text +endpoint: .openai.azure.com +key: 32 lowercase hex chars near api_key/openai_key keywords +``` + +TruffleHog output behavior: + +```text +Raw = api key +RawV2 = key:endpoint when endpoint is paired during verification or when only one endpoint exists +Redacted = shortened key +``` + +Verification calls: + +```text +https:///openai/deployments?api-version=2023-03-15-preview +Header: Api-Key: +``` + +Practical enrichment fields: + +```text +azure_openai_endpoint +azure_openai_resource_name +``` + +TODO: If RawV2 is empty, scan nearby source context for `.openai.azure.com` to pair keys with endpoints. + +### Azure DevOps PAT + +Detector: `AzureDevopsPersonalAccessToken` + +Detector finds a 52-character token and an organization-like string near `azure`. + +TruffleHog output behavior: + +```text +Raw = PAT +RawV2 = PAT + organization +``` + +Verification calls: + +```text +https://dev.azure.com//_apis/projects +BasicAuth(username="", password=) +``` + +Risk: `RawV2` is concatenated without delimiter, so organization extraction from RawV2 is ambiguous unless the token length is known. + +Practical enrichment fields: + +```text +azure_devops_org +``` + +TODO: Parse organization from raw finding JSON/source context rather than relying on concatenated RawV2 alone. + +## DockerHub + +Detector: `Dockerhub` + +DockerHub v2 detects modern PATs: + +```text +dckr_pat_[a-zA-Z0-9_-]{27} +``` + +DockerHub v1 detects UUID-like legacy tokens near `docker`. + +Both versions try to pair the token with nearby usernames or emails: + +```text +username: user/usr/-u/id nearby value or email address +Raw = token +RawV2 = username:token when username/email is found +``` + +Verification calls: + +```text +POST https://hub.docker.com/v2/users/login +{"username":"","password":""} +``` + +If verified, ExtraData can include: + +```text +hub_username +hub_email +hub_scope +2fa_required +``` + +Practical enrichment fields: + +```text +dockerhub_username +dockerhub_email +dockerhub_scope +dockerhub_2fa_required +``` + +Risk: A token without nearby username cannot be verified by this detector, but may still be useful if a username can be found elsewhere in the same package/repo. + +TODO: For unverified DockerHub PATs with empty RawV2, scan nearby context and package/repo metadata for plausible DockerHub usernames. + +## Proposed Enrichment Layer + +Add a post-processing enrichment layer after TruffleHog result parsing and before DB insert. + +Input: + +```text +finding JSON +target metadata +source file path/line if available +optional nearby source context +``` + +Output fields stored in DB/dashboard: + +```text +provider +credential_kind +credential_confidence +required_context_missing +principal +project_id +tenant_id +organization +registry +endpoint +username +email +scope +resource +``` + +Suggested confidence levels: + +```text +verified +structured_complete +prefix_shape_complete +token_only_missing_context +legacy_unverified +noisy_unverified +``` + +Priority implementation: + +1. Parse GCP service-account JSON from `RawV2`. +2. Parse Azure ACR `RawV2` JSON. +3. Parse Azure OpenAI `RawV2` as `key:endpoint` when available. +4. Parse DockerHub `RawV2` as `username:token` and ExtraData when verified. +5. Add low-confidence classification for GitHub/GitLab legacy detectors instead of hard-dropping them. +6. Optionally read nearby file context for detectors where RawV2 lacks required context. diff --git a/app/JSONL_RECONCILIATION.md b/app/JSONL_RECONCILIATION.md new file mode 100644 index 0000000..852229a --- /dev/null +++ b/app/JSONL_RECONCILIATION.md @@ -0,0 +1,13 @@ +# Offline JSONL Reconciliation + +PostgreSQL is authoritative. JSONL and provider status files are asynchronous, bounded, rebuildable compatibility projections and may lag. Normal keychecks never consume `found_secrets.jsonl`; reconciliation is explicit offline compatibility work only. + +Provider `*Checked.txt` files are rebuilt from PostgreSQL by the keycheck runner and are not append streams owned by the JSONL projector. + +1. Stop the supervisor and acquire the same cluster authority used by `migrate_runtime_safety.py`. +2. Make an immutable backup of the current file, every numbered segment, the manifest, the publication ledger, and any `*.torn-tail.bin` file. +3. Validate every retained segment as newline-terminated UTF-8 JSON. Quarantine, rather than concatenate, any final partial record. +4. Do not use keycheck `input_state.json` as a retention checkpoint. PostgreSQL candidates and current state are authoritative; immutable compatibility generations may be retired by the configured generation limit. +5. For pre-v2 multi-GiB history, do not raise online bounds or backfill it during startup. Preserve it and use a separately reviewed, bounded offline rebuild/import operation. +6. The singleton projector recovers prepared appends by exact generation, offset, length, and SHA-256; partial tails are quarantined and truncated to the prepared offset before retry. +7. Rotation renames the active generation atomically and never copies full history. A deterministic poison job is quarantined individually and later jobs continue. diff --git a/app/KEYCHECKERS.md b/app/KEYCHECKERS.md new file mode 100644 index 0000000..1a32173 --- /dev/null +++ b/app/KEYCHECKERS.md @@ -0,0 +1,192 @@ +# Keychecker Layout + +Normal input and authority: +```text +PostgreSQL keycheck_candidates fenced provider work queue +PostgreSQL keycheck_results authoritative history +PostgreSQL keycheck_current_state authoritative current classification +D:\truf\runtime\proxy.txt optional provider proxy input +``` + +`found_secrets.jsonl`, `*Results.jsonl`, and `*Checked.txt`/status files are projector-owned compatibility outputs. They may lag and normal providers do not read or write them. + +Shared helper: +```text +D:\truf\app\keycheckers\keycheck_common.py +``` + +`keycheck_runner.py --input-mode postgres` is the default managed mode. `--input` is accepted only with explicit `--input-mode jsonl` for reviewed offline/import compatibility. Provider completion inserts the result, updates current state, completes the exact candidate lease, releases candidate capacity, and creates its projection job in one PostgreSQL transaction. + +## DeepSeek + +```powershell +python supervisor.py --config config.yaml --cmd "recheck deepseek all --max-keys 100" +``` + +Output folder: +```text +deepseek\deepseekAlive.txt +deepseek\deepseekNoBalance.txt +deepseek\deepseekDead.txt +deepseek\deepseekLimited.txt +deepseek\deepseekNetwork.txt +deepseek\deepseekUnknown.txt +deepseek\deepseekChecked.txt +deepseek\deepseekResults.jsonl +``` + +## Qwen / DashScope + +```powershell +python supervisor.py --config config.yaml --cmd "recheck qwen all --max-keys 100" +``` + +The checker validates `QwenDashScope` findings with `GET /models` against the public DashScope OpenAI-compatible region endpoints. Coding Plan keys (`sk-sp-...`) use `https://coding-intl.dashscope.aliyuncs.com/v1` by default. Workspace-specific endpoints can be added with `--base-url` via `keychecks.service_args.qwen` or `QWEN_BASE_URLS`. + +Output folder: +```text +qwen\qwenAlive.txt +qwen\qwenNoBalance.txt +qwen\qwenNoContext.txt +qwen\qwenDead.txt +qwen\qwenLimited.txt +qwen\qwenRestricted.txt +qwen\qwenNetwork.txt +qwen\qwenUnknown.txt +qwen\qwenChecked.txt +qwen\qwenResults.jsonl +``` + +## Kimi / Moonshot AI + +```powershell +python supervisor.py --config config.yaml --cmd "recheck kimi all --max-keys 100" +``` + +The explicit `MOONSHOT_API_KEY` / `KIMI_API_KEY` detector is validated without generation by calling `GET /v1/users/me/balance` on the independent global and China Moonshot endpoints. + +Output folder: +```text +kimi\kimiAlive.txt +kimi\kimiNoBalance.txt +kimi\kimiDead.txt +kimi\kimiLimited.txt +kimi\kimiRestricted.txt +kimi\kimiNetwork.txt +kimi\kimiUnknown.txt +kimi\kimiChecked.txt +kimi\kimiResults.jsonl +``` + +## Groq + +```powershell +python supervisor.py --config config.yaml --cmd "recheck groq all --max-keys 100" +``` + +The checker validates TruffleHog `Groq` findings with `GET https://api.groq.com/openai/v1/models` and does not run generation probes. + +Output folder: +```text +groq\groqAlive.txt +groq\groqDead.txt +groq\groqLimited.txt +groq\groqRestricted.txt +groq\groqNetwork.txt +groq\groqUnknown.txt +groq\groqChecked.txt +groq\groqResults.jsonl +``` + +## Replicate / xAI / HuggingFace + +```powershell +python supervisor.py --config config.yaml --cmd "recheck replicate all --max-keys 100" +python supervisor.py --config config.yaml --cmd "recheck xai all --max-keys 100" +python supervisor.py --config config.yaml --cmd "recheck huggingface all --max-keys 100" +``` + +These checkers validate built-in TruffleHog findings through non-generating endpoints: Replicate account lookup, xAI model list, and HuggingFace whoami. + +## Anthropic + +```powershell +python supervisor.py --config config.yaml --cmd "recheck anthropic all --max-keys 100" +``` + +Output folder: +```text +anthropic\anthropicAlive.txt +anthropic\anthropicNoQuota.txt +anthropic\anthropicDead.txt +anthropic\anthropicLimited.txt +anthropic\anthropicRestricted.txt +anthropic\anthropicNetwork.txt +anthropic\anthropicUnknown.txt +anthropic\anthropicChecked.txt +anthropic\anthropicResults.jsonl +``` + +## AWS + +Default mode only checks STS identity: +```powershell +python supervisor.py --config config.yaml --cmd "recheck aws all --max-keys 100" +``` + +Optional Bedrock probing is configured under `keychecks.service_args.aws`, then run: +```powershell +python supervisor.py --config config.yaml --cmd "recheck aws all --max-keys 100" +``` + +Output folder: +```text +aws\awsAlive.txt +aws\awsBedrock.txt +aws\awsAdmin.txt +aws\awsCanary.txt +aws\awsQuarantined.txt +aws\awsAccessDenied.txt +aws\awsDead.txt +aws\awsNetwork.txt +aws\awsUnknown.txt +aws\awsChecked.txt +aws\awsResults.jsonl +``` + +Canary AWS credentials are detected before active AWS probes when TruffleHog provides `ExtraData.is_canary` / canary message. If metadata is absent, STS ARN containing `canarytokens` is also classified as `awsCanary.txt` and IAM/Bedrock probes are skipped. + +## Azure + +```powershell +python supervisor.py --config config.yaml --cmd "recheck azure all --max-keys 100" +``` + +This checks Azure service-principal findings from `DetectorName=Azure` using `tenantId`, `clientId`, `clientSecret` from `RawV2`. + +`DetectorName=AzureOpenAI` is placed into `azureOpenAIUnresolved.txt` unless an endpoint/resource name is available. + +Output folder: +```text +azure\azureAlive.txt +azure\azureDead.txt +azure\azureRestricted.txt +azure\azureNetwork.txt +azure\azureUnknown.txt +azure\azureOpenAIUnresolved.txt +azure\azureChecked.txt +azure\azureResults.jsonl +``` + +## Retry Flags + +Common flags: +```text +--retry-network +--retry-limited +--retry-unknown +--recheck-all +--max-keys N +``` + +Network/proxy failures are committed with the `network` status group and can be selected for a bounded PostgreSQL recheck. `*Network.txt` is only its asynchronous compatibility projection. diff --git a/app/admin_api.py b/app/admin_api.py new file mode 100644 index 0000000..2368029 --- /dev/null +++ b/app/admin_api.py @@ -0,0 +1,4555 @@ +import asyncio +import base64 +import binascii +import hashlib +import hmac +import html +import json +import logging +import re +import secrets +import threading +import time +import uuid +from datetime import datetime, timedelta, timezone +from urllib.parse import parse_qsl, quote, urlencode, urlsplit + +from starlette.concurrency import run_in_threadpool +from starlette.responses import Response, StreamingResponse +from starlette.routing import Match, Route + +from managed_files import ( + ManagedFileAccessError, ManagedFileDownload, ManagedFileIdentity, + ManagedFileListing, ManagedFileMutation, ManagedFileOperation, + ManagedFileRootRegistry, parse_managed_relative_path, +) +from scanner_db import ( + RuntimeControlRevisionConflictError, RuntimeControlTransitionError, + RuntimeOperationIdentityConflictError, +) +from runtime_document import ( + MAX_CONFIG_DOCUMENT_BYTES, MAX_SECRETS_DOCUMENT_BYTES, RuntimeDocumentError, + load_yaml_document, +) + + +ADMIN_PREFIX = '/admin-internal' +EDGE_MARKER_HEADER = 'x-truf-admin-edge' +OPERATOR_HEADER = 'x-truf-admin-operator' +OPERATOR_RE = re.compile(r'^[A-Za-z0-9_.-]{1,64}$') +SHA256_RE = re.compile(r'^[0-9a-f]{64}$') +DEFAULT_MAX_BODY_BYTES = 8 * 1024 +DEFAULT_SNAPSHOT_LIMIT = 200 +DEFAULT_REQUEUE_LIMIT = 100 +DEFAULT_OVERVIEW_TIMEOUT_SECONDS = 5 +DEFAULT_AUDIT_PAGE_LIMIT = 50 +DEFAULT_OPERATION_PAGE_LIMIT = 50 +DEFAULT_WORKER_PAGE_LIMIT = 25 +WORKER_PAGE_LIMITS = (25, 50, 100) +WORKER_FILTER_FIELDS = frozenset(( + 'source', 'worker', 'assignment', 'scan', 'phase', 'category', 'code', + 'retryable', 'window', 'limit', 'details', 'diagnostic_offset', + 'metric_offset', +)) +WORKER_FILTER_WINDOWS = { + '24h': timedelta(hours=24), + '7d': timedelta(days=7), + '30d': timedelta(days=30), + '90d': timedelta(days=90), + 'all': None, +} +KEY_RE = re.compile(r'^[A-Za-z0-9][A-Za-z0-9_.:@-]{0,127}$') +QUEUE_STATUSES = ( + 'pending', 'deferred', 'in_progress', 'done', 'failed', 'quarantined', 'cold', +) +CORE_PRODUCERS = ('gitlab', 'dockerhub', 'huggingface') +PRODUCER_IDS = tuple(f'discovery-producer:{source}' for source in CORE_PRODUCERS) +PRODUCER_ACTIONS = ('start', 'stop', 'restart', 'pause', 'resume', 'set-interval') +MAX_PRODUCER_INTERVAL_SECONDS = 365 * 24 * 60 * 60 +PIPELINE_SOURCE_IDS = ('result-ingester', 'jsonl-projector', 'janitor', 'worker-api') +MANAGED_SOURCE_ACTIONS = { + **{source_id: PRODUCER_ACTIONS for source_id in PRODUCER_IDS}, + **{ + source_id: ( + 'start', 'stop', 'restart', 'pause', 'resume', + 'set-restart', 'set-restart-delay', + ) + for source_id in PIPELINE_SOURCE_IDS + }, + 'keychecks': ( + 'start', 'stop', 'restart', 'pause', 'resume', 'once', 'set-mode', + 'set-interval', 'set-restart', 'set-restart-delay', + ), + 'docker-shadow': ('start', 'stop'), + 'dashboard': ('start', 'stop', 'restart'), +} +ALL_MANAGED_SOURCE_ACTIONS = frozenset(( + 'start', 'stop', 'restart', 'pause', 'resume', 'once', 'set-mode', + 'set-interval', 'set-restart', 'set-restart-delay', +)) +MAX_MANAGED_SOURCE_DELAY_SECONDS = 365 * 24 * 60 * 60 +MAX_MANAGED_SOURCE_LOG_LINES = 5000 +SECURITY_HEADERS = { + 'Cache-Control': 'no-store', + 'Referrer-Policy': 'same-origin', + 'Content-Security-Policy': ( + "default-src 'none'; style-src 'self'; script-src 'self'; form-action 'self'; " + "base-uri 'none'; frame-ancestors 'none'" + ), + 'X-Content-Type-Options': 'nosniff', + 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains', +} +logger = logging.getLogger(__name__) + + +class AdminAPIError(RuntimeError): + def __init__(self, status_code, message): + super().__init__(message) + self.status_code = int(status_code) + + +def _validate_key(value, label): + value = str(value or '') + if not KEY_RE.fullmatch(value): + raise AdminAPIError(400, f'{label} is invalid') + return value + + +def _validate_cap(value): + value = '' if value is None else str(value) + if not re.fullmatch(r'0|[1-9][0-9]{0,4}', value): + raise AdminAPIError(400, 'active assignment cap is invalid') + cap = int(value) + if cap > 10000: + raise AdminAPIError(400, 'active assignment cap is invalid') + return cap + + +def _validate_origin(origin): + origin = str(origin or '') + if not 1 <= len(origin) <= 512 or any(character.isspace() for character in origin): + raise ValueError('admin origin must be an exact HTTPS origin') + try: + parsed = urlsplit(origin) + parsed_port = parsed.port + except ValueError as exc: + raise ValueError('admin origin must be an exact HTTPS origin') from exc + if ( + parsed.scheme != 'https' or not parsed.hostname or parsed.username is not None + or parsed.password is not None or parsed.path or parsed.query or parsed.fragment + or parsed_port is not None and not 1 <= parsed_port <= 65535 + ): + raise ValueError('admin origin must be an exact HTTPS origin') + return origin + + +class AdminService: + def __init__( + self, db_url, origin, edge_marker, *, db_factory, + max_body_bytes=DEFAULT_MAX_BODY_BYTES, + snapshot_limit=DEFAULT_SNAPSHOT_LIMIT, + requeue_limit=DEFAULT_REQUEUE_LIMIT, + supervisor_metadata=None, + runtime_snapshot_provider=None, + source_action_provider=None, + dashboard_action_provider=None, + source_log_provider=None, + package_compatibility_provider=None, + runtime_config_path=None, + runtime_config_provider=None, + document_loader=None, + candidate_preview_provider=None, + candidate_save_provider=None, + candidate_verify_provider=None, + runtime_apply_provider=None, + managed_file_roots=None, + ): + self.db_url = str(db_url or '') + self.origin = _validate_origin(origin) + self.edge_marker = str(edge_marker or '') + if ( + not 32 <= len(self.edge_marker) <= 512 + or any(character.isspace() for character in self.edge_marker) + ): + raise ValueError('admin edge marker must be a 32..512 character secret') + self.db_factory = db_factory + self.max_body_bytes = int(max_body_bytes) + self.snapshot_limit = int(snapshot_limit) + self.requeue_limit = int(requeue_limit) + self.supervisor_metadata = dict(supervisor_metadata or {}) + self.runtime_snapshot_provider = runtime_snapshot_provider + self.source_action_provider = source_action_provider + self.dashboard_action_provider = dashboard_action_provider + self.source_log_provider = source_log_provider + self.package_compatibility_provider = package_compatibility_provider + self.runtime_config_path = str(runtime_config_path or '') + self.runtime_config_provider = runtime_config_provider + self.document_loader = document_loader + self.candidate_preview_provider = candidate_preview_provider + self.candidate_save_provider = candidate_save_provider + self.candidate_verify_provider = candidate_verify_provider + self.runtime_apply_provider = runtime_apply_provider + if managed_file_roots is None: + managed_file_roots = ManagedFileRootRegistry() + if not isinstance(managed_file_roots, ManagedFileRootRegistry): + raise ValueError('managed file root registry is invalid') + self.managed_file_roots = managed_file_roots + self._managed_file_operation_lock = threading.Lock() + self._queue_snapshot_lock = threading.Lock() + self._queue_snapshot_cache = None + self._queue_snapshot_retry_at = 0.0 + if not 1024 <= self.max_body_bytes <= 64 * 1024: + raise ValueError('admin body byte limit must be between 1024 and 65536') + if not 1 <= self.snapshot_limit <= 500: + raise ValueError('admin snapshot limit must be between 1 and 500') + if not 1 <= self.requeue_limit <= 500: + raise ValueError('admin requeue limit must be between 1 and 500') + self.csrf_token = secrets.token_urlsafe(48) + + def _call(self, method_name, *args, **kwargs): + db = self.db_factory(db_url=self.db_url, initialize=False) + if not db.enabled: + db.close() + raise RuntimeError('admin PostgreSQL connection is unavailable') + try: + return getattr(db, method_name)(*args, **kwargs) + finally: + db.close() + + def _worker_page_limit(self, limit): + limit = int(limit) + if limit not in WORKER_PAGE_LIMITS: + raise ValueError('worker page limit is out of range') + return min(limit, self.snapshot_limit) + + def snapshot(self, filters=None, *, limit=DEFAULT_WORKER_PAGE_LIMIT): + return self._call( + 'admin_remote_worker_snapshot', self._worker_page_limit(limit), + filters=dict(filters or {}), + ) + + def diagnostic_groups( + self, filters=None, *, occurrence_offset=0, + limit=DEFAULT_WORKER_PAGE_LIMIT, + ): + return self._call( + 'admin_worker_diagnostic_groups', self._worker_page_limit(limit), + filters=dict(filters or {}), occurrence_offset=occurrence_offset, + ) + + def duration_metrics( + self, filters=None, *, offset=0, limit=DEFAULT_WORKER_PAGE_LIMIT, + ): + filters = dict(filters or {}) + if 'since' not in filters: + filters['since'] = ( + datetime.now(timezone.utc) - WORKER_FILTER_WINDOWS['30d'] + ).isoformat(timespec='seconds') + return self._call( + 'admin_worker_duration_metrics', self._worker_page_limit(limit), + filters=filters, offset=offset, + ) + + def assignment_detail(self, reservation_id): + detail = self._call( + 'admin_worker_assignment_detail', reservation_id, + event_limit=self.snapshot_limit, diagnostic_limit=self.snapshot_limit, + ) + if detail is None: + raise AdminAPIError(404, 'worker assignment was not found') + try: + source = detail['assignment']['source'] + detail['current_effective_policy'] = next( + row for row in self.deadline_policy()['rows'] + if row['source'] == source + ) + except (KeyError, StopIteration, TypeError, RuntimeError): + detail['current_effective_policy'] = None + return detail + + def diagnostic_envelope(self, reservation_id, diagnostic_uid): + envelope = self._call( + 'admin_worker_diagnostic_envelope', reservation_id, diagnostic_uid, + ) + if envelope is None: + raise AdminAPIError(404, 'worker diagnostic was not found') + return envelope + + @staticmethod + def _deadline_policy(config): + try: + worker = config['supervisor']['worker_api'] + sources = config['sources'] + enabled = worker.get('sources') or ('gitlab', 'dockerhub', 'huggingface') + fallback = int(worker['assignment_ttl_seconds']) + overrides = dict(worker.get('assignment_ttl_seconds_by_source') or {}) + upload = int(worker['bundle_body_timeout_seconds']) + rows = [] + for source in enabled: + scan = int(sources[source]['timeout']) + assignment = int(overrides.get(source, fallback)) + required = scan + upload + 60 + rows.append({ + 'source': source, + 'scan_deadline_seconds': scan, + 'upload_deadline_seconds': upload, + 'assignment_deadline_seconds': assignment, + 'assignment_policy_source': ( + 'source override' if source in overrides else 'global fallback' + ), + 'required_minimum_seconds': required, + 'handoff_margin_seconds': 60, + 'relationship': ( + f'{assignment} >= {scan} + {upload} + 60' + ), + 'valid': assignment >= required, + 'future_assignments_only': True, + }) + return {'rows': rows, 'future_assignments_only': True} + except (KeyError, TypeError, ValueError, OverflowError) as exc: + raise RuntimeError('runtime deadline policy is unavailable') from exc + + def deadline_policy(self): + if not self.runtime_config_path: + raise RuntimeError('runtime deadline policy is unavailable') + provider = self.runtime_config_provider + if provider is None: + from runtime_document_io import load_managed_runtime_config + provider = load_managed_runtime_config + loaded = provider(self.runtime_config_path) + config = getattr(loaded, 'config', None) + if not isinstance(config, dict): + raise RuntimeError('runtime deadline policy is unavailable') + return self._deadline_policy(config) + + def deadline_policy_from_text(self, document_text): + try: + config = load_yaml_document( + document_text.encode('utf-8', errors='strict'), + max_bytes=MAX_CONFIG_DOCUMENT_BYTES, + ) + except (RuntimeDocumentError, UnicodeError) as exc: + raise RuntimeError('runtime deadline policy is unavailable') from exc + return self._deadline_policy(config) + + def runtime_document_observability(self, document_text): + return { + 'policy': self._overview_component( + 'candidate deadline policy', + lambda: self.deadline_policy_from_text(document_text), + ), + 'metrics': self._overview_component( + 'worker duration metrics', self.duration_metrics, + ), + } + + def _package_compatibility_snapshot(self): + provider = self.package_compatibility_provider + if provider is None: + raise RuntimeError('package compatibility is unavailable') + snapshot = provider() + if not isinstance(snapshot, dict) or set(snapshot) != { + 'profiles', 'required_capabilities', + }: + raise RuntimeError('package compatibility is malformed') + profiles = snapshot['profiles'] + required = snapshot['required_capabilities'] + if ( + not isinstance(profiles, list) or not 1 <= len(profiles) <= 16 + or not isinstance(required, list) or not 1 <= len(required) <= 16 + ): + raise RuntimeError('package compatibility is malformed') + + capability_fields = {'source', 'platform', 'planning_kind'} + + def capability(item): + if not isinstance(item, dict) or set(item) != capability_fields: + raise RuntimeError('package compatibility is malformed') + value = {key: item[key] for key in capability_fields} + if any( + type(part) is not str + or not re.fullmatch(r'[a-z][a-z0-9_]{0,63}', part) + for part in value.values() + ): + raise RuntimeError('package compatibility is malformed') + return value + + profile_fields = { + 'profile_name', 'protocol_version', 'bundle_format_version', + 'platform_tag', 'code_manifest_sha256', 'detector_policy_sha256', + 'sources', 'capabilities', + } + normalized_profiles = [] + for item in profiles: + if not isinstance(item, dict) or set(item) != profile_fields: + raise RuntimeError('package compatibility is malformed') + if ( + type(item['profile_name']) is not str + or not KEY_RE.fullmatch(item['profile_name']) + or type(item['protocol_version']) is not int + or item['protocol_version'] < 1 + or type(item['bundle_format_version']) is not int + or item['bundle_format_version'] < 1 + or type(item['platform_tag']) is not str + or not 1 <= len(item['platform_tag']) <= 128 + or not re.fullmatch(r'[a-f0-9]{64}', item['code_manifest_sha256']) + or not re.fullmatch(r'[a-f0-9]{64}', item['detector_policy_sha256']) + or not isinstance(item['sources'], list) + or not 1 <= len(item['sources']) <= 16 + or len(item['sources']) != len(set(item['sources'])) + or any( + type(source) is not str + or not re.fullmatch(r'[a-z][a-z0-9_]{0,63}', source) + for source in item['sources'] + ) + or not isinstance(item['capabilities'], list) + or not 1 <= len(item['capabilities']) <= 16 + ): + raise RuntimeError('package compatibility is malformed') + normalized_profiles.append({ + key: item[key] for key in profile_fields - {'sources', 'capabilities'} + } | { + 'sources': list(item['sources']), + 'capabilities': [capability(value) for value in item['capabilities']], + }) + return { + 'profiles': normalized_profiles, + 'required_capabilities': [capability(item) for item in required], + } + + def _runtime_snapshot(self): + if not self.supervisor_metadata: + raise RuntimeError('Supervisor metadata is unavailable') + provider = self.runtime_snapshot_provider + if provider is None: + from supervisor import get_runtime_snapshot + provider = get_runtime_snapshot + snapshot = provider( + self.supervisor_metadata, timeout=DEFAULT_OVERVIEW_TIMEOUT_SECONDS, + ) + if not ( + isinstance(snapshot, dict) + and isinstance(snapshot.get('runtime'), dict) + and isinstance(snapshot.get('postgres'), dict) + and isinstance(snapshot.get('pipeline'), dict) + and isinstance(snapshot.get('sources'), list) + and all(isinstance(item, dict) for item in snapshot['sources']) + ): + raise RuntimeError('Supervisor snapshot is malformed') + return snapshot + + def _queue_snapshot(self): + def copied(snapshot): + return dict( + snapshot, + counts=dict(snapshot.get('counts') or {}), + truncated_statuses=list(snapshot.get('truncated_statuses') or []), + ) + + with self._queue_snapshot_lock: + now = time.monotonic() + if now < self._queue_snapshot_retry_at: + if self._queue_snapshot_cache is None: + raise RuntimeError('queue snapshot is unavailable') + snapshot = copied(self._queue_snapshot_cache) + snapshot.update({ + 'degraded': True, + 'stale': True, + 'reason': 'bounded_count_retry_backoff', + 'retry_after_sec': max(1, int(self._queue_snapshot_retry_at - now)), + }) + return snapshot + + snapshot = self._call('admin_target_queue_health', CORE_PRODUCERS) + if not isinstance(snapshot, dict) or not isinstance(snapshot.get('counts'), dict): + raise RuntimeError('queue snapshot is malformed') + counts = snapshot['counts'] + if any( + status not in QUEUE_STATUSES or type(value) is not int or value < 0 + for status, value in counts.items() + ): + raise RuntimeError('queue snapshot is malformed') + truncated = snapshot.get('truncated_statuses') + if not isinstance(truncated, list) or any( + type(status) is not str or status not in QUEUE_STATUSES + for status in truncated + ): + raise RuntimeError('queue snapshot is malformed') + + retry_after = snapshot.get('retry_after_sec') + if type(retry_after) is not int or not 0 <= retry_after <= 3600: + raise RuntimeError('queue snapshot is malformed') + if snapshot.get('stale') is True: + self._queue_snapshot_retry_at = now + retry_after + if self._queue_snapshot_cache is not None: + cached = copied(self._queue_snapshot_cache) + cached.update({ + 'degraded': True, + 'stale': True, + 'reason': 'bounded_count_query_failed', + 'retry_after_sec': retry_after, + }) + return cached + if not counts: + raise RuntimeError('queue snapshot is unavailable') + return copied(snapshot) + + self._queue_snapshot_cache = copied(snapshot) + self._queue_snapshot_retry_at = 0.0 + return copied(snapshot) + + def _control_snapshot(self): + snapshot = self._call('runtime_drain_progress') + if not isinstance(snapshot, dict): + raise RuntimeError('control snapshot is malformed') + if ( + type(snapshot.get('revision')) is not int + or snapshot['revision'] < 0 + or any( + type(snapshot.get(key)) is not bool + for key in ( + 'discovery_paused', 'dispatch_paused', + 'effective_discovery_paused', 'effective_dispatch_paused', + ) + ) + or snapshot.get('drain_state') not in ('normal', 'draining', 'drained') + or any( + type(snapshot.get(key)) is not int or snapshot[key] < 0 + for key in ( + 'live_remote_assignments', 'precommit_result_bundles', + 'blocker_count', + ) + ) + or type(snapshot.get('actor')) is not str + or type(snapshot.get('updated_at')) is not str + ): + raise RuntimeError('control snapshot is malformed') + drain_active = snapshot['drain_state'] != 'normal' + if ( + snapshot['blocker_count'] != ( + snapshot['live_remote_assignments'] + + snapshot['precommit_result_bundles'] + ) + or snapshot['effective_discovery_paused'] is not ( + snapshot['discovery_paused'] or drain_active + ) + or snapshot['effective_dispatch_paused'] is not ( + snapshot['dispatch_paused'] or drain_active + ) + ): + raise RuntimeError('control snapshot is malformed') + return snapshot + + def _recent_operations(self): + rows = self._call('recent_runtime_operations', self.snapshot_limit) + fields = ( + 'operation_id', 'actor', 'action', 'target_ref', 'status', + 'safe_category', 'safe_detail', 'requested_at', 'completed_at', + 'updated_at', + ) + if not isinstance(rows, list) or len(rows) > self.snapshot_limit: + raise RuntimeError('operation snapshot is malformed') + result = [] + for row in rows: + if not isinstance(row, dict) or any( + row.get(key) is not None and type(row.get(key)) is not str + for key in fields + ): + raise RuntimeError('operation snapshot is malformed') + result.append({key: row.get(key) for key in fields}) + return result + + def operation_page(self, before=None): + before_updated_at = before_operation_id = None + if before is not None: + before_updated_at, before_operation_id = before + rows = self._call( + 'recent_runtime_operations', DEFAULT_OPERATION_PAGE_LIMIT + 1, + before_updated_at=before_updated_at, + before_operation_id=before_operation_id, + ) + fields = ( + 'operation_id', 'actor', 'action', 'target_ref', 'status', + 'safe_category', 'safe_detail', 'requested_at', 'completed_at', + 'updated_at', + ) + if not isinstance(rows, list) or len(rows) > DEFAULT_OPERATION_PAGE_LIMIT + 1: + raise RuntimeError('operation page is malformed') + operations = [] + for row in rows[:DEFAULT_OPERATION_PAGE_LIMIT]: + if not isinstance(row, dict) or any( + row.get(key) is not None and type(row.get(key)) is not str + for key in fields + ): + raise RuntimeError('operation page is malformed') + if not row.get('operation_id') or not row.get('updated_at'): + raise RuntimeError('operation page is malformed') + operations.append({key: row.get(key) for key in fields}) + next_before = None + if len(rows) > DEFAULT_OPERATION_PAGE_LIMIT and operations: + last = operations[-1] + next_before = (last['updated_at'], last['operation_id']) + return {'operations': operations, 'next_before': next_before} + + def operation_status(self, operation_id): + operation = self._call('runtime_operation', operation_id) + if operation is None: + raise AdminAPIError(404, 'Not Found') + return operation + + def audit_page(self, before_event_id=None): + return self._call( + 'runtime_audit_events', before_event_id=before_event_id, + limit=min(DEFAULT_AUDIT_PAGE_LIMIT, self.snapshot_limit), + ) + + @staticmethod + def _managed_file_traversal(traversal): + if traversal is None: + raise AdminAPIError(503, 'managed files are unavailable') + return traversal + + def list_managed_files(self, traversal, root_id, relative_path=None): + traversal = self._managed_file_traversal(traversal) + root = self.managed_file_roots.get(root_id) + if root is None: + raise AdminAPIError(404, 'managed file target was not found') + if not root.permissions.allow_list: + raise AdminAPIError(403, 'managed file operation is not allowed') + if relative_path is not None: + try: + parse_managed_relative_path(relative_path, root.limits) + except ManagedFileAccessError as exc: + _managed_file_error(exc) + try: + listing = traversal.list_directory(root_id, relative_path) + except ManagedFileAccessError as exc: + _managed_file_error(exc) + if not isinstance(listing, ManagedFileListing): + raise RuntimeError('managed file listing is malformed') + return listing + + def download_managed_file(self, traversal, root_id, relative_path): + traversal = self._managed_file_traversal(traversal) + self._managed_file_root(root_id, relative_path, ManagedFileOperation.READ) + try: + download = traversal.download_file(root_id, relative_path) + except ManagedFileAccessError as exc: + _managed_file_error(exc) + if not isinstance(download, ManagedFileDownload): + raise RuntimeError('managed file download is malformed') + return download + + @staticmethod + def _overview_component(name, callback): + try: + value = callback() + if value is None: + raise RuntimeError('overview component is unavailable') + return {'available': True, 'value': value} + except Exception as exc: + logger.warning('Admin overview component %s is unavailable: %s', name, type(exc).__name__) + return {'available': False, 'value': None} + + def overview(self): + return { + 'runtime': self._overview_component('runtime', self._runtime_snapshot), + 'queue': self._overview_component( + 'queue', self._queue_snapshot, + ), + 'control': self._overview_component( + 'control', self._control_snapshot, + ), + 'operations': self._overview_component( + 'operations', self._recent_operations, + ), + } + + def search_snapshot(self): + return { + 'runtime': self._overview_component('runtime', self._runtime_snapshot), + 'control': self._overview_component('control', self._control_snapshot), + } + + def workers_dispatch_snapshot( + self, filters=None, *, diagnostic_occurrence_offset=0, + metric_offset=0, page_limit=DEFAULT_WORKER_PAGE_LIMIT, + include_diagnostics=False, include_metrics=False, + ): + filters = dict(filters or {}) + return { + 'workers': self._overview_component( + 'workers', lambda: self.snapshot(filters, limit=page_limit), + ), + 'diagnostics': ( + self._overview_component( + 'diagnostics', lambda: self.diagnostic_groups( + filters, occurrence_offset=diagnostic_occurrence_offset, + limit=page_limit, + ), + ) if include_diagnostics else {'available': False, 'value': None} + ), + 'metrics': ( + self._overview_component( + 'metrics', lambda: self.duration_metrics( + filters, offset=metric_offset, limit=page_limit, + ), + ) if include_metrics else {'available': False, 'value': None} + ), + 'policy': self._overview_component('policy', self.deadline_policy), + 'control': self._overview_component('control', self._control_snapshot), + 'packages': self._overview_component( + 'packages', self._package_compatibility_snapshot, + ), + } + + def set_dispatch_paused(self, paused, expected_revision, actor, operation_id): + if type(paused) is not bool: + raise AdminAPIError(400, 'dispatch control state is invalid') + if type(expected_revision) is not int or expected_revision < 0: + raise AdminAPIError(400, 'control revision is invalid') + try: + return self._call( + 'set_runtime_dispatch_paused', paused, + expected_revision=expected_revision, actor=actor, + operation_id=operation_id, + ) + except ( + RuntimeControlRevisionConflictError, RuntimeControlTransitionError, + RuntimeOperationIdentityConflictError, + ) as exc: + raise AdminAPIError(409, 'dispatch control changed; refresh and retry') from exc + + def start_drain(self, expected_revision, actor, operation_id): + return self._drain_action( + 'start_runtime_drain', expected_revision, actor, operation_id, + ) + + def cancel_drain(self, expected_revision, actor, operation_id): + return self._drain_action( + 'cancel_runtime_drain', expected_revision, actor, operation_id, + ) + + def _drain_action(self, method, expected_revision, actor, operation_id): + if type(expected_revision) is not int or expected_revision < 0: + raise AdminAPIError(400, 'control revision is invalid') + try: + return self._call( + method, expected_revision=expected_revision, actor=actor, + operation_id=operation_id, + ) + except ( + RuntimeControlRevisionConflictError, RuntimeControlTransitionError, + RuntimeOperationIdentityConflictError, + ) as exc: + raise AdminAPIError(409, 'drain control changed; refresh and retry') from exc + + def set_discovery_paused(self, paused, expected_revision, actor, operation_id): + if type(paused) is not bool: + raise AdminAPIError(400, 'discovery control state is invalid') + if type(expected_revision) is not int or expected_revision < 0: + raise AdminAPIError(400, 'control revision is invalid') + try: + return self._call( + 'set_runtime_discovery_paused', paused, + expected_revision=expected_revision, actor=actor, + operation_id=operation_id, + ) + except ( + RuntimeControlRevisionConflictError, RuntimeControlTransitionError, + RuntimeOperationIdentityConflictError, + ) as exc: + raise AdminAPIError(409, 'discovery control changed; refresh and retry') from exc + + def _complete_producer_operation(self, operation_id, *, succeeded, outcome=None): + last_error = None + for attempt in range(3): + try: + return self._call( + 'complete_runtime_source_operation', operation_id, + succeeded=succeeded, outcome=outcome, + ) + except RuntimeOperationIdentityConflictError: + raise + except Exception as exc: + last_error = exc + if attempt < 2: + time.sleep(0.05) + raise last_error + + def _managed_source_entry(self, source_id): + if type(source_id) is not str or not KEY_RE.fullmatch(source_id) or source_id == 'all': + raise AdminAPIError(400, 'managed source action is invalid') + snapshot = self._runtime_snapshot() + matches = [ + item for item in snapshot.get('sources', []) + if isinstance(item, dict) and item.get('id') == source_id + ] + if len(matches) != 1: + raise AdminAPIError(400, 'managed source action is invalid') + allowed = matches[0].get('allowed_actions') + if ( + not isinstance(allowed, list) + or len(allowed) != len(set(allowed)) + or any(action not in ALL_MANAGED_SOURCE_ACTIONS for action in allowed) + ): + raise AdminAPIError(502, 'managed source state is invalid') + return matches[0] + + def managed_source_action( + self, source_id, source_action, actor, operation_id, *, + interval_seconds=None, mode=None, restart_enabled=None, + restart_delay_seconds=None, + ): + if source_id == 'dashboard': + allowed_actions = MANAGED_SOURCE_ACTIONS['dashboard'] + else: + allowed_actions = self._managed_source_entry(source_id).get('allowed_actions') + if source_action not in allowed_actions: + raise AdminAPIError(400, 'managed source action is invalid') + if source_action == 'set-interval': + if ( + type(interval_seconds) is not int + or not 1 <= interval_seconds <= MAX_PRODUCER_INTERVAL_SECONDS + ): + raise AdminAPIError(400, 'managed source interval is invalid') + elif interval_seconds is not None: + raise AdminAPIError(400, 'managed source action is invalid') + if source_action == 'set-mode': + if mode not in ('loop', 'once', 'repeat') or ( + source_id == 'keychecks' and mode == 'loop' + ): + raise AdminAPIError(400, 'managed source mode is invalid') + elif mode is not None: + raise AdminAPIError(400, 'managed source action is invalid') + if source_action == 'set-restart': + if type(restart_enabled) is not bool: + raise AdminAPIError(400, 'managed source restart setting is invalid') + elif restart_enabled is not None: + raise AdminAPIError(400, 'managed source action is invalid') + if source_action == 'set-restart-delay': + if ( + type(restart_delay_seconds) is not int + or not 1 <= restart_delay_seconds <= MAX_MANAGED_SOURCE_DELAY_SECONDS + ): + raise AdminAPIError(400, 'managed source restart delay is invalid') + elif restart_delay_seconds is not None: + raise AdminAPIError(400, 'managed source action is invalid') + + try: + operation_parameters = {} + if mode is not None: + operation_parameters['mode'] = mode + if restart_enabled is not None: + operation_parameters['restart_enabled'] = restart_enabled + if restart_delay_seconds is not None: + operation_parameters['restart_delay_seconds'] = restart_delay_seconds + created = self._call( + 'create_runtime_source_operation', operation_id=operation_id, + actor=actor, source_id=source_id, source_action=source_action, + interval_seconds=interval_seconds, + **operation_parameters, + ) + except RuntimeOperationIdentityConflictError as exc: + raise AdminAPIError(409, 'operation ID is already bound to another request') from exc + if created.get('replayed') is True: + if created.get('status') == 'succeeded': + resulting = created.get('resulting_identity') or {} + return {'outcome': resulting.get('outcome', 'completed')} + if created.get('status') == 'running': + raise AdminAPIError(409, 'managed source action is already in progress') + raise AdminAPIError(409, 'managed source action already completed with failure') + parameters = {} + if interval_seconds is not None: + parameters['interval_seconds'] = interval_seconds + if mode is not None: + parameters['mode'] = mode + if restart_enabled is not None: + parameters['restart_enabled'] = restart_enabled + if restart_delay_seconds is not None: + parameters['restart_delay_seconds'] = restart_delay_seconds + try: + if source_id == 'dashboard': + provider = self.dashboard_action_provider + if provider is None: + from supervisor import send_dashboard_action + provider = send_dashboard_action + result = provider( + self.supervisor_metadata, source_action, timeout=60, + ) + else: + provider = self.source_action_provider + if provider is None: + from supervisor import send_managed_source_action + provider = send_managed_source_action + result = provider( + self.supervisor_metadata, source_id, source_action, + timeout=60, **parameters, + ) + if ( + not isinstance(result, dict) + or result.get('outcome') not in ('completed', 'dependency-blocked') + ): + raise RuntimeError('managed source action response is invalid') + except Exception: + try: + self._complete_producer_operation(operation_id, succeeded=False) + except Exception as completion_error: + logger.error( + 'Managed source operation failure reconciliation failed: %s', + type(completion_error).__name__, + ) + raise AdminAPIError(502, 'managed source action failed') from None + try: + self._complete_producer_operation( + operation_id, succeeded=True, outcome=result['outcome'], + ) + except RuntimeOperationIdentityConflictError as exc: + raise AdminAPIError(409, 'managed source operation completion conflicted') from exc + return result + + def producer_action( + self, source_id, source_action, actor, operation_id, *, interval_seconds=None, + ): + if source_id not in PRODUCER_IDS or source_action not in PRODUCER_ACTIONS: + raise AdminAPIError(400, 'producer action is invalid') + return self.managed_source_action( + source_id, source_action, actor, operation_id, + interval_seconds=interval_seconds, + ) + + def managed_source_log(self, source_id, line_count): + self._managed_source_entry(source_id) + if type(line_count) is not int or not 1 <= line_count <= MAX_MANAGED_SOURCE_LOG_LINES: + raise AdminAPIError(400, 'managed source log line count is invalid') + provider = self.source_log_provider + if provider is None: + from supervisor import send_managed_source_log_tail + provider = send_managed_source_log_tail + try: + result = provider( + self.supervisor_metadata, source_id, line_count, timeout=60, + ) + except Exception: + raise AdminAPIError(502, 'managed source log tail failed') from None + if ( + not isinstance(result, dict) + or result.get('source_id') != source_id + or result.get('line_count') != len(result.get('lines') or []) + or not isinstance(result.get('lines'), list) + or any(type(line) is not str for line in result['lines']) + or type(result.get('response_truncated')) is not bool + ): + raise AdminAPIError(502, 'managed source log tail failed') + return { + 'source_id': source_id, + 'line_count': result['line_count'], + 'lines': list(result['lines']), + 'response_truncated': result['response_truncated'], + } + + @staticmethod + def _runtime_document_error(exc): + status = 409 if exc.category == 'reference' and exc.path == 'revision' else 400 + document = exc.document or 'document' + path = exc.path or 'root' + raise AdminAPIError( + status, f'{document} validation failed ({exc.category}) at {path}', + ) from exc + + def runtime_document_editor(self, document): + if document not in ('config', 'secrets') or not self.runtime_config_path: + raise AdminAPIError(503, 'runtime document editor is unavailable') + provider = self.document_loader + if provider is None: + from runtime_document_io import load_managed_runtime_editor_document + provider = load_managed_runtime_editor_document + try: + editor = provider(self.runtime_config_path, document) + except RuntimeDocumentError as exc: + self._runtime_document_error(exc) + if ( + getattr(editor, 'document', None) != document + or getattr(editor, 'source', None) not in ('active', 'candidate') + or type(getattr(editor, 'text', None)) is not str + or getattr(editor, 'state', None) is None + ): + raise AdminAPIError(503, 'runtime document editor is unavailable') + return editor + + def preview_runtime_document(self, document, document_text): + payload = None + provider = self.candidate_preview_provider + if provider is None: + from runtime_document_io import preview_managed_runtime_candidate + provider = preview_managed_runtime_candidate + try: + payload = document_text.encode('utf-8', errors='strict') + document_text = None + return provider( + self.runtime_config_path, document, payload, + ) + except RuntimeDocumentError as exc: + self._runtime_document_error(exc) + finally: + document_text = payload = None + + def save_runtime_document_candidate( + self, document, document_text, actor, operation_id, *, expected_hashes, + ): + candidate_bytes = None + try: + candidate_bytes = document_text.encode('utf-8', errors='strict') + document_text = None + return self._save_runtime_document_candidate_bytes( + document, candidate_bytes, actor, operation_id, + expected_hashes=expected_hashes, + ) + finally: + document_text = candidate_bytes = None + + def _save_runtime_document_candidate_bytes( + self, document, candidate_bytes, actor, operation_id, *, expected_hashes, + ): + preview = operation = revision = current = selected = None + + def complete_success(candidate_sha256, written): + for attempt in range(3): + try: + return self._call( + 'complete_runtime_document_operation', operation_id, + succeeded=True, candidate_sha256=candidate_sha256, + written=written, + ) + except Exception: + if attempt == 2: + raise AdminAPIError( + 503, 'runtime document completion is pending', + ) from None + + try: + action = f'runtime.{document}.save' + selected_key = f'candidate_{document}' + other_key = ( + 'candidate_secrets' if document == 'config' else 'candidate_config' + ) + proposed_sha256 = hashlib.sha256(candidate_bytes).hexdigest() + proposed_bytes = len(candidate_bytes) + operation = self._call('runtime_operation', operation_id) + if operation is not None: + identity = operation.get('expected_identity') or {} + selected_identity_key = f'{selected_key}_sha256' + other_identity_key = f'{other_key}_sha256' + submitted_selected = expected_hashes[selected_key] + identity_matches = ( + operation.get('actor') == actor + and operation.get('action') == action + and operation.get('target_kind') == 'runtime-document' + and operation.get('target_ref') == document + and hmac.compare_digest( + identity.get('active_config_sha256', ''), + expected_hashes['active_config'], + ) + and hmac.compare_digest( + identity.get('active_secrets_sha256', ''), + expected_hashes['active_secrets'], + ) + and hmac.compare_digest( + identity.get(other_identity_key, ''), + expected_hashes[other_key], + ) + and any( + hmac.compare_digest(identity.get(key, ''), submitted_selected) + for key in (selected_identity_key, 'candidate_after_sha256') + ) + and hmac.compare_digest( + identity.get('candidate_after_sha256', ''), proposed_sha256, + ) + and identity.get('candidate_after_bytes') == proposed_bytes + ) + if not identity_matches: + raise AdminAPIError( + 409, 'operation ID is already bound to another request', + ) + try: + operation = self._call( + 'create_runtime_document_operation', + operation_id=operation_id, actor=actor, action=action, + active_config_sha256=identity['active_config_sha256'], + active_secrets_sha256=identity['active_secrets_sha256'], + candidate_config_sha256=identity['candidate_config_sha256'], + candidate_secrets_sha256=identity['candidate_secrets_sha256'], + candidate_after_sha256=identity['candidate_after_sha256'], + candidate_before_bytes=identity['candidate_before_bytes'], + candidate_after_bytes=identity['candidate_after_bytes'], + candidate_before_present=identity['candidate_before_present'], + ) + except RuntimeOperationIdentityConflictError as exc: + raise AdminAPIError( + 409, 'operation ID is already bound to another request', + ) from exc + if operation.get('status') == 'succeeded': + return operation + if operation.get('status') != 'running': + raise AdminAPIError(409, 'runtime document save already has another result') + current = self.runtime_document_editor(document) + current_hashes = _runtime_document_hashes(current.state) + if any( + not hmac.compare_digest(current_hashes[key], identity[f'{key}_sha256']) + for key in ('active_config', 'active_secrets', other_key) + ): + raise AdminAPIError(409, 'runtime document revision changed') + selected = ( + current.state.candidate_config + if document == 'config' else current.state.candidate_secrets + ) + if selected.present and ( + hmac.compare_digest(selected.sha256, identity['candidate_after_sha256']) + and selected.byte_count == identity['candidate_after_bytes'] + ): + return complete_success( + selected.sha256, + written=( + not identity['candidate_before_present'] + or + identity[selected_identity_key] != selected.sha256 + or identity['candidate_before_bytes'] != selected.byte_count + ), + ) + if not ( + hmac.compare_digest(selected.sha256, identity[selected_identity_key]) + and selected.byte_count == identity['candidate_before_bytes'] + ): + raise AdminAPIError(409, 'runtime document revision changed') + expected_hashes = { + 'active_config': identity['active_config_sha256'], + 'active_secrets': identity['active_secrets_sha256'], + 'candidate_config': identity['candidate_config_sha256'], + 'candidate_secrets': identity['candidate_secrets_sha256'], + } + else: + provider = self.candidate_preview_provider + if provider is None: + from runtime_document_io import preview_managed_runtime_candidate + provider = preview_managed_runtime_candidate + try: + preview = provider(self.runtime_config_path, document, candidate_bytes) + except RuntimeDocumentError as exc: + self._runtime_document_error(exc) + current_hashes = _runtime_document_hashes(preview.state) + if any( + not hmac.compare_digest(current_hashes[key], expected_hashes[key]) + for key in current_hashes + ): + raise AdminAPIError(409, 'runtime document revision changed') + selected = ( + preview.state.candidate_config + if document == 'config' else preview.state.candidate_secrets + ) + try: + operation = self._call( + 'create_runtime_document_operation', + operation_id=operation_id, actor=actor, action=action, + active_config_sha256=expected_hashes['active_config'], + active_secrets_sha256=expected_hashes['active_secrets'], + candidate_config_sha256=expected_hashes['candidate_config'], + candidate_secrets_sha256=expected_hashes['candidate_secrets'], + candidate_after_sha256=preview.proposed.sha256, + candidate_before_bytes=selected.byte_count, + candidate_after_bytes=preview.proposed.byte_count, + candidate_before_present=selected.present, + ) + except RuntimeOperationIdentityConflictError as exc: + raise AdminAPIError( + 409, 'operation ID is already bound to another request', + ) from exc + provider = self.candidate_save_provider + if provider is None: + from runtime_document_io import save_managed_runtime_candidate + provider = save_managed_runtime_candidate + try: + revision = provider( + self.runtime_config_path, document, candidate_bytes, + expected_active_config_sha256=expected_hashes['active_config'], + expected_active_secrets_sha256=expected_hashes['active_secrets'], + expected_candidate_config_sha256=expected_hashes['candidate_config'], + expected_candidate_secrets_sha256=expected_hashes['candidate_secrets'], + ) + except RuntimeDocumentError as exc: + try: + current = self.runtime_document_editor(document) + except AdminAPIError: + current = None + if current is not None: + current_hashes = _runtime_document_hashes(current.state) + selected = ( + current.state.candidate_config + if document == 'config' else current.state.candidate_secrets + ) + other_key = ( + 'candidate_secrets' + if document == 'config' else 'candidate_config' + ) + if ( + all( + hmac.compare_digest( + current_hashes[key], expected_hashes[key], + ) + for key in ('active_config', 'active_secrets', other_key) + ) + and selected.present + and hmac.compare_digest( + selected.sha256, operation['expected_identity'][ + 'candidate_after_sha256' + ], + ) + and selected.byte_count == operation['expected_identity'][ + 'candidate_after_bytes' + ] + ): + identity = operation['expected_identity'] + return complete_success( + selected.sha256, + written=( + not identity['candidate_before_present'] + or identity[f'candidate_{document}_sha256'] + != selected.sha256 + or identity['candidate_before_bytes'] + != selected.byte_count + ), + ) + for _ in range(3): + try: + self._call( + 'complete_runtime_document_operation', operation_id, + succeeded=False, + ) + break + except Exception: + continue + self._runtime_document_error(exc) + complete_success(revision.proposed.sha256, revision.written) + return revision + finally: + candidate_bytes = preview = operation = revision = current = selected = None + complete_success = None + + def request_runtime_apply( + self, action, actor, operation_id, *, expected_hashes, + ): + if action not in ('apply-config', 'apply-secrets', 'apply-both'): + raise AdminAPIError(400, 'runtime apply action is invalid') + candidate_config = ( + expected_hashes['candidate_config'] + if action in ('apply-config', 'apply-both') else None + ) + candidate_secrets = ( + expected_hashes['candidate_secrets'] + if action in ('apply-secrets', 'apply-both') else None + ) + expected_identity = { + 'active_config_sha256': expected_hashes['active_config'], + 'active_secrets_sha256': expected_hashes['active_secrets'], + 'candidate_config_sha256': candidate_config, + 'candidate_secrets_sha256': candidate_secrets, + } + verified = None + operation = self._call('runtime_operation', operation_id) + if operation is None: + if self.runtime_apply_provider is None: + raise AdminAPIError(503, 'runtime apply agent is unavailable') + verify = self.candidate_verify_provider + if verify is None: + from runtime_document_io import verify_managed_runtime_candidates + verify = verify_managed_runtime_candidates + try: + verified = verify( + self.runtime_config_path, action, + expected_active_config_sha256=expected_hashes['active_config'], + expected_active_secrets_sha256=expected_hashes['active_secrets'], + expected_candidate_config_sha256=candidate_config, + expected_candidate_secrets_sha256=candidate_secrets, + ) + except RuntimeDocumentError as exc: + self._runtime_document_error(exc) + try: + operation = self._call( + 'create_runtime_operation', operation_id=operation_id, + actor=actor, action=action, expected_identity=expected_identity, + ) + except RuntimeOperationIdentityConflictError as exc: + raise AdminAPIError(409, 'operation ID is already bound to another request') from exc + if operation.get('status') in ('succeeded', 'failed', 'rolled_back', 'failed_hold'): + if operation.get('status') == 'succeeded': + return {'operation': operation, 'verification': verified} + raise AdminAPIError(409, 'runtime apply operation already has a terminal result') + if self.runtime_apply_provider is None: + raise AdminAPIError(503, 'runtime apply agent is unavailable') + try: + self.runtime_apply_provider( + operation_id=operation_id, + action=action, + active_config_sha256=expected_identity['active_config_sha256'], + active_secrets_sha256=expected_identity['active_secrets_sha256'], + candidate_config_sha256=expected_identity['candidate_config_sha256'], + candidate_secrets_sha256=expected_identity['candidate_secrets_sha256'], + ) + except Exception: + raise AdminAPIError(502, 'runtime apply dispatch failed') from None + return {'operation': operation, 'verification': verified} + + def _managed_file_root(self, root_id, relative_path, operation): + root = self.managed_file_roots.get(root_id) + if root is None: + raise AdminAPIError(404, 'managed file target was not found') + if not root.permissions.allows(operation): + raise AdminAPIError(403, 'managed file operation is not allowed') + try: + parse_managed_relative_path(relative_path, root.limits) + except ManagedFileAccessError as exc: + _managed_file_error(exc) + return root + + def _complete_managed_file_operation( + self, operation_id, *, succeeded, mutation=None, + before_sha256=None, before_byte_count=None, + after_sha256=None, after_byte_count=None, written=None, + ): + if mutation is not None: + before_sha256 = mutation.before.sha256 if mutation.before else None + before_byte_count = mutation.before.byte_count if mutation.before else None + after_sha256 = mutation.after.sha256 if mutation.after else None + after_byte_count = mutation.after.byte_count if mutation.after else None + written = mutation.written + arguments = { + 'succeeded': succeeded, + 'before_sha256': before_sha256, + 'before_byte_count': before_byte_count, + 'after_sha256': after_sha256, + 'after_byte_count': after_byte_count, + 'written': written, + } + if not succeeded: + arguments = {'succeeded': False} + last_error = None + for attempt in range(3): + try: + return self._call( + 'complete_runtime_managed_file_operation', operation_id, + **arguments, + ) + except RuntimeOperationIdentityConflictError: + raise + except Exception as exc: + last_error = exc + if attempt < 2: + time.sleep(0.05) + raise last_error + + @staticmethod + def _managed_file_observed_identity( + traversal, root_id, relative_path, operation_type, + *, require_private_sha256=None, + ): + try: + return traversal.mutation_file_identity( + root_id, relative_path, operation_type, + require_private_sha256=require_private_sha256, + ) + except ManagedFileAccessError as exc: + if exc.category == 'not_found': + return None + _managed_file_error(exc) + + def _complete_observed_managed_file_operation( + self, traversal, *, action, root_id, relative_path, operation_type, + operation_id, expected_sha256, proposed_sha256, proposed_byte_count, + ): + current = self._managed_file_observed_identity( + traversal, root_id, relative_path, operation_type, + require_private_sha256=( + proposed_sha256 + if action in ('files.create', 'files.replace') else None + ), + ) + if action == 'files.create': + if current is None: + return None + if ( + current.sha256 != proposed_sha256 + or current.byte_count != proposed_byte_count + ): + raise AdminAPIError(409, 'managed file state changed concurrently') + return self._complete_managed_file_operation( + operation_id, succeeded=True, + after_sha256=current.sha256, + after_byte_count=current.byte_count, written=True, + ) + if action == 'files.replace': + if current is None: + raise AdminAPIError(409, 'managed file state changed concurrently') + if ( + current.sha256 == proposed_sha256 + and current.byte_count == proposed_byte_count + ): + no_write = hmac.compare_digest(expected_sha256, proposed_sha256) + return self._complete_managed_file_operation( + operation_id, succeeded=True, + before_sha256=expected_sha256, + before_byte_count=current.byte_count if no_write else None, + after_sha256=current.sha256, + after_byte_count=current.byte_count, + written=not no_write, + ) + if not hmac.compare_digest(current.sha256, expected_sha256): + raise AdminAPIError(409, 'managed file state changed concurrently') + return None + if current is None: + return self._complete_managed_file_operation( + operation_id, succeeded=True, + before_sha256=expected_sha256, + before_byte_count=None, + after_sha256=None, after_byte_count=None, written=True, + ) + if not hmac.compare_digest(current.sha256, expected_sha256): + raise AdminAPIError(409, 'managed file state changed concurrently') + return None + + def _managed_file_mutation_locked_payload( + self, traversal, *, action, root_id, relative_path, actor, operation_id, + payload_box, expected_sha256=None, + ): + operation_type = ( + ManagedFileOperation.DELETE + if action == 'files.delete' else ManagedFileOperation.CREATE_REPLACE + ) + proposed_sha256 = proposed_byte_count = None + if action in ('files.create', 'files.replace'): + if len(payload_box) != 1 or type(payload_box[0]) is not bytes: + raise AdminAPIError(400, 'managed file content is invalid') + proposed_sha256 = hashlib.sha256(payload_box[0]).hexdigest() + proposed_byte_count = len(payload_box[0]) + if ( + action in ('files.replace', 'files.delete') + and not SHA256_RE.fullmatch(str(expected_sha256 or '')) + ): + raise AdminAPIError(400, 'managed file revision is invalid') + + existing = self._call('runtime_operation', operation_id) + if existing is None: + traversal = self._managed_file_traversal(traversal) + self._managed_file_root(root_id, relative_path, operation_type) + current = self._managed_file_observed_identity( + traversal, root_id, relative_path, operation_type, + ) + if action == 'files.create': + if current is not None: + raise AdminAPIError(409, 'managed file state changed concurrently') + elif ( + current is None + or not hmac.compare_digest(current.sha256, expected_sha256) + ): + raise AdminAPIError(409, 'managed file state changed concurrently') + try: + created = self._call( + 'create_runtime_managed_file_operation', + operation_id=operation_id, actor=actor, action=action, + root_id=root_id, relative_path=relative_path, + expected_sha256=expected_sha256, + proposed_sha256=proposed_sha256, + proposed_byte_count=proposed_byte_count, + ) + except RuntimeOperationIdentityConflictError as exc: + raise AdminAPIError( + 409, 'operation ID is already bound to another request', + ) from exc + if created.get('status') == 'succeeded': + return created + if created.get('status') != 'running': + raise AdminAPIError(409, 'managed file operation already has a terminal result') + + traversal = self._managed_file_traversal(traversal) + self._managed_file_root(root_id, relative_path, operation_type) + if created.get('replayed') is True: + completed = self._complete_observed_managed_file_operation( + traversal, action=action, root_id=root_id, + relative_path=relative_path, operation_type=operation_type, + operation_id=operation_id, expected_sha256=expected_sha256, + proposed_sha256=proposed_sha256, + proposed_byte_count=proposed_byte_count, + ) + if completed is not None: + return completed + + try: + if action == 'files.delete': + mutation = traversal.delete_file( + root_id, relative_path, expected_sha256=expected_sha256, + ) + else: + mutation = traversal.create_replace_file( + root_id, relative_path, payload_box[0], + expected_sha256=( + expected_sha256 if action == 'files.replace' else None + ), + ) + except ManagedFileAccessError as exc: + if ( + created.get('replayed') is True + and exc.category in ('hash_conflict', 'concurrent_change', 'not_found') + ): + try: + completed = self._complete_observed_managed_file_operation( + traversal, action=action, root_id=root_id, + relative_path=relative_path, operation_type=operation_type, + operation_id=operation_id, + expected_sha256=expected_sha256, + proposed_sha256=proposed_sha256, + proposed_byte_count=proposed_byte_count, + ) + if completed is not None: + return completed + except AdminAPIError as state_error: + if state_error.status_code >= 500: + raise + try: + self._complete_managed_file_operation( + operation_id, succeeded=False, + ) + except Exception as completion_error: + logger.error( + 'Managed file failure reconciliation failed: %s', + type(completion_error).__name__, + ) + _managed_file_error(exc) + try: + return self._complete_managed_file_operation( + operation_id, succeeded=True, mutation=mutation, + ) + except RuntimeOperationIdentityConflictError as exc: + raise AdminAPIError(409, 'managed file completion conflicted') from exc + except Exception as exc: + raise AdminAPIError(503, 'managed file completion is pending') from exc + finally: + mutation = None + + def _managed_file_mutation( + self, traversal, *, action, root_id, relative_path, actor, operation_id, + payload=None, expected_sha256=None, + ): + payload_box = [payload] if payload is not None else [] + payload = None + execution_db = None + execution_acquired = False + failure = None + try: + execution_db = self.db_factory( + db_url=self.db_url, initialize=False, + ) + if not execution_db.enabled: + raise AdminAPIError(503, 'managed files are unavailable') + try: + execution_db.acquire_runtime_managed_file_execution(operation_id) + except Exception as exc: + raise AdminAPIError(503, 'managed files are unavailable') from exc + execution_acquired = True + with self._managed_file_operation_lock: + return self._managed_file_mutation_locked_payload( + traversal, action=action, root_id=root_id, + relative_path=relative_path, actor=actor, + operation_id=operation_id, payload_box=payload_box, + expected_sha256=expected_sha256, + ) + except BaseException as exc: + failure = exc + raise + finally: + release_error = None + try: + if execution_acquired: + try: + execution_db.release_runtime_managed_file_execution( + operation_id, + ) + except BaseException as exc: + release_error = exc + finally: + try: + if execution_db is not None: + execution_db.close() + except BaseException as exc: + if ( + release_error is None + or isinstance(release_error, Exception) + and not isinstance(exc, Exception) + ): + release_error = exc + finally: + payload_box.clear() + payload = payload_box = execution_db = None + if release_error is not None: + if failure is None: + if not isinstance(release_error, Exception): + raise release_error + raise AdminAPIError( + 503, 'managed file completion is pending', + ) from release_error + logger.error( + 'Managed file execution lock release failed: %s', + type(release_error).__name__, + ) + + def create_managed_file( + self, traversal, root_id, relative_path, payload, actor, operation_id, + ): + try: + return self._managed_file_mutation( + traversal, action='files.create', root_id=root_id, + relative_path=relative_path, payload=payload, + actor=actor, operation_id=operation_id, + ) + finally: + payload = None + + def replace_managed_file( + self, traversal, root_id, relative_path, payload, expected_sha256, + actor, operation_id, + ): + try: + return self._managed_file_mutation( + traversal, action='files.replace', root_id=root_id, + relative_path=relative_path, payload=payload, + expected_sha256=expected_sha256, actor=actor, + operation_id=operation_id, + ) + finally: + payload = None + + def delete_managed_file( + self, traversal, root_id, relative_path, expected_sha256, + actor, operation_id, + ): + return self._managed_file_mutation( + traversal, action='files.delete', root_id=root_id, + relative_path=relative_path, expected_sha256=expected_sha256, + actor=actor, operation_id=operation_id, + ) + + def _complete_worker_admin_operation( + self, operation_id, *, succeeded, affected_count=None, + ): + last_error = None + for attempt in range(3): + try: + return self._call( + 'complete_runtime_worker_admin_operation', operation_id, + succeeded=succeeded, affected_count=affected_count, + ) + except RuntimeOperationIdentityConflictError: + raise + except Exception as exc: + last_error = exc + if attempt < 2: + time.sleep(0.05) + raise last_error + + def _worker_admin_mutation( + self, *, action, target_ref, parameters, actor, operation_id, callback, + success, affected_count=None, + ): + request_bytes = json.dumps( + {'action': action, 'target_ref': target_ref, 'parameters': parameters}, + sort_keys=True, separators=(',', ':'), ensure_ascii=True, + ).encode('ascii') + request_sha256 = hashlib.sha256(request_bytes).hexdigest() + try: + created = self._call( + 'create_runtime_worker_admin_operation', + operation_id=operation_id, actor=actor, action=action, + target_ref=target_ref, request_sha256=request_sha256, + ) + except RuntimeOperationIdentityConflictError as exc: + raise AdminAPIError(409, 'operation ID is already bound to another request') from exc + if created.get('replayed') is True: + if created.get('status') == 'succeeded': + raise AdminAPIError(409, 'worker administration action already completed') + if created.get('status') == 'running': + raise AdminAPIError(409, 'worker administration action is already in progress') + raise AdminAPIError(409, 'worker administration action already failed') + try: + result = callback() + applied = success(result) + except Exception: + try: + self._complete_worker_admin_operation(operation_id, succeeded=False) + except Exception as completion_error: + logger.error( + 'Worker admin failure reconciliation failed: %s', + type(completion_error).__name__, + ) + raise + try: + self._complete_worker_admin_operation( + operation_id, succeeded=applied, + affected_count=( + affected_count(result) if applied and affected_count else + 1 if applied else None + ), + ) + except RuntimeOperationIdentityConflictError as exc: + raise AdminAPIError(409, 'worker administration completion conflicted') from exc + return result + + def create_user(self, user_key, cap, actor, operation_id): + user_key = _validate_key(user_key, 'user key') + cap = _validate_cap(cap) + return self._worker_admin_mutation( + action='workers.user.create', target_ref=user_key, + parameters={'active_assignment_cap': cap}, actor=actor, + operation_id=operation_id, + callback=lambda: self._call('create_remote_worker_user', user_key, cap), + success=bool, + ) + + def set_user_cap(self, user_key, cap, actor, operation_id): + user_key = _validate_key(user_key, 'user key') + cap = _validate_cap(cap) + return self._worker_admin_mutation( + action='workers.user.set-cap', target_ref=user_key, + parameters={'active_assignment_cap': cap}, actor=actor, + operation_id=operation_id, + callback=lambda: self._call('set_remote_worker_user_cap', user_key, cap), + success=bool, + ) + + def set_user_disabled(self, user_key, disabled, actor, operation_id): + user_key = _validate_key(user_key, 'user key') + disabled = disabled is True + return self._worker_admin_mutation( + action='workers.user.disable' if disabled else 'workers.user.enable', + target_ref=user_key, parameters={}, actor=actor, + operation_id=operation_id, + callback=lambda: self._call( + 'set_remote_worker_user_disabled', user_key, disabled, + ), + success=bool, + ) + + def issue_device( + self, user_key, device_key, actor, operation_id, *, rotate=False, + ): + user_key = _validate_key(user_key, 'user key') + device_key = _validate_key(device_key, 'device key') + rotate = rotate is True + token_holder = {} + + def issue(): + token = secrets.token_urlsafe(48) + token_holder['token'] = token + token_sha256 = hashlib.sha256(token.encode('ascii')).hexdigest() + return self._call( + 'issue_remote_worker_device', user_key, device_key, token_sha256, + rotate=rotate, + ) + + result = self._worker_admin_mutation( + action='workers.device.rotate' if rotate else 'workers.device.issue', + target_ref=device_key, parameters={'user_key': user_key}, actor=actor, + operation_id=operation_id, callback=issue, success=bool, + ) + return result, token_holder.get('token') if result else None + + def set_device_revoked(self, device_key, revoked, actor, operation_id): + device_key = _validate_key(device_key, 'device key') + revoked = revoked is True + return self._worker_admin_mutation( + action='workers.device.revoke' if revoked else 'workers.device.unrevoke', + target_ref=device_key, parameters={}, actor=actor, + operation_id=operation_id, + callback=lambda: self._call( + 'set_remote_worker_device_revoked', device_key, revoked, + ), + success=bool, + ) + + def requeue(self, queue_ids, actor, operation_id): + if not queue_ids or len(queue_ids) > self.requeue_limit: + raise AdminAPIError(400, 'queue ID selection exceeds its bound') + selection_sha256 = hashlib.sha256( + ','.join(str(value) for value in queue_ids).encode('ascii') + ).hexdigest() + return self._worker_admin_mutation( + action='workers.queue.requeue', target_ref='deferred-queue', + parameters={ + 'selection_sha256': selection_sha256, + 'item_count': len(queue_ids), + }, actor=actor, operation_id=operation_id, + callback=lambda: self._call( + 'admin_requeue_deferred_targets', queue_ids, + max_items=self.requeue_limit, + ), + success=lambda result: type(result) is int and result >= 0, + affected_count=lambda result: result, + ) + + def discard_source_queue(self, source, actor, operation_id): + source = _validate_key(source, 'source') + if source not in CORE_PRODUCERS: + raise AdminAPIError(400, 'source is not a managed producer') + return self._worker_admin_mutation( + action='workers.queue.discard-source', target_ref=source, + parameters={'statuses': ['pending', 'deferred', 'cold']}, + actor=actor, operation_id=operation_id, + callback=lambda: self._call( + 'admin_discard_queued_source', source, + ), + success=lambda result: type(result) is int and result >= 0, + affected_count=lambda result: result, + ) + + +def _secure_response(body, status_code=200, media_type='text/plain'): + response = Response(body, status_code=status_code, media_type=media_type) + response.headers.update(SECURITY_HEADERS) + return response + + +def _secure_json_response(value, *, filename=None): + body = json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + allow_nan=False, + ) + headers = dict(SECURITY_HEADERS) + if filename: + headers['Content-Disposition'] = ( + "attachment; filename*=UTF-8''" + quote(filename, safe='') + ) + return Response(body, media_type='application/json', headers=headers) + + +def _diagnostic_json_response(envelope, diagnostic_uid): + body = envelope['canonical_json'] + headers = dict(SECURITY_HEADERS) + headers.update({ + 'Content-Disposition': ( + "attachment; filename*=UTF-8''" + + quote(f'{diagnostic_uid}.json', safe='') + ), + 'ETag': f'"{envelope["sha256"]}"', + 'Content-Length': str(len(body.encode('ascii'))), + }) + return Response(body, media_type='application/json', headers=headers) + + +def _single_header(request, name): + values = request.headers.getlist(name) + return values[0] if len(values) == 1 else None + + +def _trusted_operator(request, service): + marker = _single_header(request, EDGE_MARKER_HEADER) + operator = _single_header(request, OPERATOR_HEADER) + if marker is None or operator is None or not OPERATOR_RE.fullmatch(operator): + return None + try: + if not hmac.compare_digest(marker, service.edge_marker): + return None + except TypeError: + return None + return operator + + +async def _form_fields(request, service, expected, *, body_limit=None): + limit = service.max_body_bytes if body_limit is None else int(body_limit) + origin = _single_header(request, 'origin') + if origin is None or not hmac.compare_digest(origin, service.origin): + raise AdminAPIError(403, 'mutation authorization failed') + content_type = _single_header(request, 'content-type') + if content_type is None or content_type.split(';', 1)[0].strip().lower() != ( + 'application/x-www-form-urlencoded' + ): + raise AdminAPIError(415, 'urlencoded form body is required') + content_length = _single_header(request, 'content-length') + if content_length is not None: + try: + declared_length = int(content_length) + except (TypeError, ValueError, OverflowError): + raise AdminAPIError(400, 'Content-Length is invalid') from None + if declared_length < 0 or declared_length > limit: + raise AdminAPIError(413, 'form body exceeds its byte bound') + + body = bytearray() + chunk = encoded = pairs = fields = key = value = result = None + try: + async for chunk in request.stream(): + if len(body) + len(chunk) > limit: + raise AdminAPIError(413, 'form body exceeds its byte bound') + body.extend(chunk) + encoded = bytes(body).decode('ascii', errors='strict') + pairs = parse_qsl( + encoded, keep_blank_values=True, strict_parsing=True, + encoding='utf-8', errors='strict', max_num_fields=10, + ) + fields = {} + for key, value in pairs: + if key in fields: + raise AdminAPIError(400, 'form fields must not be repeated') + fields[key] = value + if set(fields) != set(expected): + raise AdminAPIError(400, 'form shape is invalid') + if not hmac.compare_digest(fields['csrf_token'], service.csrf_token): + raise AdminAPIError(403, 'mutation authorization failed') + result = fields + fields = None + return result + except (UnicodeDecodeError, UnicodeEncodeError, ValueError) as exc: + raise AdminAPIError(400, 'form body is invalid') from exc + finally: + body.clear() + if isinstance(fields, dict): + fields.clear() + chunk = encoded = pairs = fields = key = value = result = None + + +def _parse_queue_ids(value, limit): + value = str(value or '') + if not re.fullmatch(r'[1-9][0-9]*(?:,[1-9][0-9]*)*', value): + raise AdminAPIError(400, 'queue IDs must be comma-separated positive integers') + parts = value.split(',') + if any(len(item) > 19 for item in parts): + raise AdminAPIError(400, 'queue ID selection exceeds its bound') + ids = [int(item) for item in parts] + if ( + any(item > 9223372036854775807 for item in ids) + or len(ids) > limit or len(ids) != len(set(ids)) + ): + raise AdminAPIError(400, 'queue ID selection exceeds its bound') + return ids + + +def _parse_revision(value): + value = str(value or '') + if not re.fullmatch(r'0|[1-9][0-9]{0,18}', value): + raise AdminAPIError(400, 'control revision is invalid') + revision = int(value) + if revision > 9223372036854775806: + raise AdminAPIError(400, 'control revision is invalid') + return revision + + +def _parse_producer_interval(value): + value = str(value or '') + if not re.fullmatch(r'[1-9][0-9]{0,7}', value): + raise AdminAPIError(400, 'producer interval is invalid') + interval = int(value) + if interval > MAX_PRODUCER_INTERVAL_SECONDS: + raise AdminAPIError(400, 'producer interval is invalid') + return interval + + +def _parse_managed_source_delay(value, label): + value = str(value or '') + if not re.fullmatch(r'[1-9][0-9]{0,7}', value): + raise AdminAPIError(400, f'{label} is invalid') + delay = int(value) + if delay > MAX_MANAGED_SOURCE_DELAY_SECONDS: + raise AdminAPIError(400, f'{label} is invalid') + return delay + + +def _parse_operation_id(value): + value = str(value or '') + try: + parsed = uuid.UUID(value) + except (AttributeError, TypeError, ValueError) as exc: + raise AdminAPIError(400, 'operation ID is invalid') from exc + if parsed.int == 0 or str(parsed) != value: + raise AdminAPIError(400, 'operation ID is invalid') + return value + + +def _parse_audit_cursor(value): + value = str(value or '') + if not re.fullmatch(r'[1-9][0-9]{0,18}', value): + raise AdminAPIError(400, 'audit cursor is invalid') + cursor = int(value) + if cursor > 9223372036854775807: + raise AdminAPIError(400, 'audit cursor is invalid') + return cursor + + +def _operation_cursor(updated_at, operation_id): + payload = json.dumps( + [updated_at, operation_id], ensure_ascii=True, separators=(',', ':'), + ).encode('ascii') + return base64.urlsafe_b64encode(payload).rstrip(b'=').decode('ascii') + + +def _parse_operation_cursor(value): + value = str(value or '') + if not re.fullmatch(r'[A-Za-z0-9_-]{1,256}', value): + raise AdminAPIError(400, 'operation cursor is invalid') + try: + payload = base64.urlsafe_b64decode(value + '=' * (-len(value) % 4)) + parts = json.loads(payload.decode('ascii')) + if ( + not isinstance(parts, list) or len(parts) != 2 + or any(type(part) is not str for part in parts) + ): + raise ValueError + updated_at, operation_id = parts + parsed_at = datetime.fromisoformat(updated_at.replace('Z', '+00:00')) + if parsed_at.tzinfo is None or not updated_at: + raise ValueError + _parse_operation_id(operation_id) + if _operation_cursor(updated_at, operation_id) != value: + raise ValueError + return updated_at, operation_id + except (AdminAPIError, binascii.Error, UnicodeDecodeError, ValueError) as exc: + raise AdminAPIError(400, 'operation cursor is invalid') from exc + + +def _query_fields(request, allowed): + fields = {} + for key, value in request.query_params.multi_items(): + if key not in allowed or key in fields: + raise AdminAPIError(400, 'query shape is invalid') + fields[key] = value + return fields + + +def _parse_worker_filters(query): + selected = {key: str(value or '') for key, value in query.items()} + filters = {} + patterns = { + 'source': r'[a-z0-9][a-z0-9_.-]{0,63}', + 'phase': r'[a-z][a-z0-9_]{0,63}', + 'category': r'[a-z][a-z0-9_]{0,127}', + 'code': r'[A-Za-z0-9][A-Za-z0-9._:-]{0,255}', + } + for name, pattern in patterns.items(): + if selected.get(name): + if re.fullmatch(pattern, selected[name]) is None: + raise AdminAPIError(400, 'worker filter is invalid') + filters[name] = selected[name] + if selected.get('worker'): + worker = selected['worker'].strip() + if not 1 <= len(worker) <= 128 or '\x00' in worker: + raise AdminAPIError(400, 'worker filter is invalid') + selected['worker'] = worker + filters['worker'] = worker + if selected.get('assignment'): + if selected['assignment'] not in { + 'accepted', 'prebundle_failed', 'expired', 'unfinished', + }: + raise AdminAPIError(400, 'worker filter is invalid') + filters['assignment_outcome'] = selected['assignment'] + if selected.get('scan'): + if selected['scan'] not in { + 'clean', 'found', 'degraded', 'error', 'skipped', 'unavailable', + }: + raise AdminAPIError(400, 'worker filter is invalid') + filters['scan_outcome'] = selected['scan'] + if selected.get('retryable'): + if selected['retryable'] not in {'true', 'false'}: + raise AdminAPIError(400, 'worker filter is invalid') + filters['retryable'] = selected['retryable'] == 'true' + for name in ('diagnostic_offset', 'metric_offset'): + offset = selected.get(name) or '0' + if re.fullmatch(r'0|[1-9][0-9]{0,18}', offset) is None: + raise AdminAPIError(400, 'worker filter is invalid') + if int(offset) > 9223372036854775807: + raise AdminAPIError(400, 'worker filter is invalid') + selected[name] = offset + window = selected.get('window') or '30d' + selected['window'] = window + if window not in WORKER_FILTER_WINDOWS: + raise AdminAPIError(400, 'worker filter is invalid') + delta = WORKER_FILTER_WINDOWS[window] + if delta is not None: + filters['since'] = ( + datetime.now(timezone.utc) - delta + ).isoformat(timespec='seconds') + page_limit = selected.get('limit') or str(DEFAULT_WORKER_PAGE_LIMIT) + if page_limit not in {str(value) for value in WORKER_PAGE_LIMITS}: + raise AdminAPIError(400, 'worker filter is invalid') + selected['limit'] = page_limit + details = selected.get('details') or 'assignments' + if details not in {'assignments', 'diagnostics', 'metrics', 'all'}: + raise AdminAPIError(400, 'worker filter is invalid') + selected['details'] = details + return filters, selected + + +def _parse_assignment_id(value): + value = str(value or '') + if re.fullmatch(r'[1-9][0-9]{0,18}', value) is None: + raise AdminAPIError(404, 'worker assignment was not found') + result = int(value) + if result > 9223372036854775807: + raise AdminAPIError(404, 'worker assignment was not found') + return result + + +def _managed_file_error(exc): + status = { + 'invalid_path': 400, + 'invalid_hash': 400, + 'invalid_content': 400, + 'operation_not_allowed': 403, + 'unknown_root': 404, + 'not_found': 404, + 'unsafe_target': 404, + 'hash_conflict': 409, + 'concurrent_change': 409, + 'limit_exceeded': 413, + 'root_unavailable': 503, + 'filesystem_unavailable': 503, + 'closed': 503, + 'durability_uncertain': 503, + 'download_busy': 503, + }.get(getattr(exc, 'category', None), 503) + messages = { + 400: 'managed file request is invalid', + 403: 'managed file operation is not allowed', + 404: 'managed file target was not found', + 409: 'managed file state changed concurrently', + 413: 'managed file limit was exceeded', + 503: 'managed files are unavailable', + } + raise AdminAPIError(status, messages[status]) from exc + + +def _parse_managed_file_hash(value): + value = str(value or '') + if not SHA256_RE.fullmatch(value): + raise AdminAPIError(400, 'managed file revision is invalid') + return value + + +def _parse_managed_file_content(value): + raw = decoded = None + try: + raw = str(value or '').encode('ascii', errors='strict') + decoded = base64.b64decode(raw, altchars=b'-_', validate=True) + if base64.urlsafe_b64encode(decoded) != raw: + raise ValueError('noncanonical') + return decoded + except (binascii.Error, UnicodeEncodeError, ValueError) as exc: + raise AdminAPIError(400, 'managed file content is invalid') from exc + finally: + value = raw = decoded = None + + +def _managed_file_listing_location(root_id, relative_path): + parent = relative_path.rpartition('/')[0] or None + query = {'root_id': root_id} + if parent is not None: + query['relative_path'] = parent + return '../files?' + urlencode(query) + + +def _parse_document_hashes(fields, *, require_config=True, require_secrets=True): + names = ('active_config', 'active_secrets') + if require_config: + names += ('candidate_config',) + if require_secrets: + names += ('candidate_secrets',) + hashes = {} + for name in names: + value = str(fields.get(f'expected_{name}_sha256') or '') + if not SHA256_RE.fullmatch(value): + raise AdminAPIError(400, 'runtime document revision is invalid') + hashes[name] = value + return hashes + + +def _form(action, title, csrf_token, fields, relative_root, hidden=(), *, values=None): + values = dict(values or {}) + controls = [] + for field in fields: + name, label, input_type = field[:3] + bounds = '' + if len(field) == 5: + bounds = ( + f' min="{html.escape(str(field[3]))}"' + f' max="{html.escape(str(field[4]))}" step="1"' + ) + value = values.get(name) + value_attribute = ( + f' value="{html.escape(str(value))}"' if value is not None else '' + ) + controls.append( + f'' + ) + return ( + f'
' + f'

{html.escape(title)}

' + f'' + + ''.join( + f'' + for name, value in hidden + ) + + ''.join(controls) + + f'
' + ) + + +def _table(columns, rows): + header = ''.join(f'{html.escape(label)}' for key, label in columns) + rendered_rows = [] + for row in rows: + rendered_rows.append('' + ''.join( + f'{html.escape(str(row.get(key) if row.get(key) is not None else ""))}' + for key, _label in columns + ) + '') + return f'{header}{"".join(rendered_rows)}
' + + +def _admin_navigation(active, relative_root): + links = ( + ('workers', relative_root + '/', 'Workers / Dispatch'), + ('overview', relative_root + '/overview', 'Overview'), + ('search', relative_root + '/search', 'Search'), + ('supervisor', relative_root + '/supervisor', 'Supervisor'), + ('logs', relative_root + '/logs', 'Logs'), + ('config', relative_root + '/config', 'Config'), + ('secrets', relative_root + '/secrets', 'Secrets'), + ('files', relative_root + '/files', 'Files'), + ('operations', relative_root + '/operations', 'Operations'), + ('audit', relative_root + '/audit', 'Audit'), + ) + return '' + + +def _page_shell(title, subtitle, active, content, relative_root, script=None): + script_tag = ( + f'' if script else '' + ) + return f''' + +{html.escape(title)}{script_tag} +

{html.escape(title)}

{html.escape(subtitle)}

+{_admin_navigation(active, relative_root)}
{content}
''' + + +def _identity_json(value): + if value is None: + return '' + return json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + allow_nan=False, + ) + + +def _operation_link(operation_id, relative_root): + operation_id = html.escape(str(operation_id)) + href = html.escape(f'{relative_root}/operations/{operation_id}') + return f'{operation_id}' + + +def _render_operations_table(operations, relative_root): + headers = ( + 'Operation', 'Actor', 'Action', 'Target', 'Status', 'Category', + 'Requested', 'Completed', 'Updated', + ) + rows = [] + for operation in operations: + cells = [ + _operation_link(operation.get('operation_id', ''), relative_root), + ] + cells.extend( + html.escape(str(operation.get(key) or '')) + for key in ( + 'actor', 'action', 'target_ref', 'status', 'safe_category', + 'requested_at', 'completed_at', 'updated_at', + ) + ) + rows.append('' + ''.join(f'{cell}' for cell in cells) + '') + header = ''.join(f'{html.escape(label)}' for label in headers) + return f'{header}{"".join(rows)}
' + + +def _render_operations_page(page, relative_root='.'): + operations = list(page.get('operations') or []) + pagination = f'Newest operations' + next_before = page.get('next_before') + if next_before is not None: + pagination += ( + ' ' + ) + content = ( + '

Recent durable operations

' + '

Newest first. Open an operation to inspect its persisted status after a runtime restart.

' + + _render_operations_table(operations, relative_root) + + f'

{pagination}

' + ) + return _page_shell( + 'Operations', 'Bounded recent operation status from durable storage.', + 'operations', content, relative_root, + ) + + +def _render_operation_page(operation, relative_root='..'): + fields = ( + ('operation_id', 'Operation'), ('actor', 'Actor'), ('action', 'Action'), + ('target_kind', 'Target kind'), ('target_ref', 'Target'), + ('status', 'Status'), ('safe_category', 'Category'), + ('safe_detail', 'Detail'), ('expected_revision', 'Expected revision'), + ('resulting_revision', 'Resulting revision'), ('agent_state', 'Agent state'), + ('agent_result_sha256', 'Agent result SHA-256'), + ('requested_at', 'Requested'), ('started_at', 'Started'), + ('completed_at', 'Completed'), ('agent_reconciled_at', 'Agent reconciled'), + ('updated_at', 'Updated'), + ) + rows = ({'field': label, 'value': operation.get(key)} for key, label in fields) + expected = html.escape(_identity_json(operation.get('expected_identity'))) + resulting = html.escape(_identity_json(operation.get('resulting_identity'))) + content = ( + '

Durable status

' + + _table((('field', 'Field'), ('value', 'Value')), rows) + + '

Expected identity

' + + f'
{expected}
' + + '

Resulting identity

' + + f'
{resulting}
' + ) + return _page_shell( + 'Operation status', 'Persisted state remains queryable across runtime restarts.', + 'operations', content, relative_root, + ) + + +def _render_audit_page(page, relative_root='.'): + events = list(page.get('events') or []) + headers = ( + 'Event', 'Operation', 'Actor', 'Action', 'Target', 'Time', 'Result', + 'Category', 'Before identity', 'After identity', 'Before bytes', + 'After bytes', + ) + rows = [] + for event in events: + target = f'{event.get("target_kind") or ""}:{event.get("target_ref") or ""}' + cells = [ + html.escape(str(event.get('id') or '')), + _operation_link(event.get('operation_id', ''), relative_root), + html.escape(str(event.get('actor') or '')), + html.escape(str(event.get('action') or '')), + html.escape(target), + html.escape(str(event.get('created_at') or '')), + html.escape(str(event.get('result') or '')), + html.escape(str(event.get('safe_category') or '')), + '' + html.escape(_identity_json(event.get('before_identity'))) + '', + '' + html.escape(_identity_json(event.get('after_identity'))) + '', + html.escape(str(event.get('before_bytes') if event.get('before_bytes') is not None else '')), + html.escape(str(event.get('after_bytes') if event.get('after_bytes') is not None else '')), + ] + rows.append('' + ''.join(f'{cell}' for cell in cells) + '') + header = ''.join(f'{html.escape(label)}' for label in headers) + table = f'{header}{"".join(rows)}
' + next_before = page.get('next_before_event_id') + pagination = f'Newest events' + if next_before is not None: + pagination += ( + ' ' + ) + content = ( + '

Append-only events

' + '

Newest first. Each page is bounded and uses a stable event cursor.

' + + table + f'

{pagination}

' + ) + return _page_shell( + 'Audit', 'Content-free accepted and terminal operation evidence.', + 'audit', content, relative_root, + ) + + +def _managed_file_query(root_id, relative_path=None): + values = {'root_id': root_id} + if relative_path is not None: + values['relative_path'] = relative_path + return urlencode(values) + + +def _render_managed_file_forms(service, root, relative_path): + forms = [] + root_id = html.escape(root.root_id) + + def hidden(operation_id): + return ( + f'' + f'' + f'' + ) + + path_value = '' if relative_path is None else relative_path + '/' + path_input = ( + '' + ) + expected_input = ( + '' + ) + content_input = ( + '' + ) + if root.permissions.allow_create_replace: + forms.append( + '
' + '

Create file

' + hidden(str(uuid.uuid4())) + path_input + + content_input + '
' + ) + forms.append( + '
' + '

Replace file

' + hidden(str(uuid.uuid4())) + path_input + + expected_input + content_input + + '
' + ) + if root.permissions.allow_delete: + forms.append( + '
' + '

Delete file

' + hidden(str(uuid.uuid4())) + path_input + + expected_input + '
' + ) + if not forms: + return '

This root is read-only.

' + return ( + '

Mutation content is canonical URL-safe Base64 and is bounded by the ' + f'{service.max_body_bytes}-byte admin form limit.

' + '
' + ''.join(forms) + '
' + ) + + +def _render_files_page(service, *, root=None, relative_path=None, listing=None): + root_rows = [] + for configured in service.managed_file_roots.roots: + root_rows.append({ + 'root': configured.root_id, + 'list': configured.permissions.allow_list, + 'read': configured.permissions.allow_read, + 'create_replace': configured.permissions.allow_create_replace, + 'delete': configured.permissions.allow_delete, + 'path_bytes': configured.limits.max_relative_path_bytes, + 'entries': configured.limits.max_listing_entries, + 'file_bytes': configured.limits.max_file_bytes, + }) + roots = _table(( + ('root', 'Logical root'), ('list', 'List'), ('read', 'Read'), + ('create_replace', 'Create/replace'), ('delete', 'Delete'), + ('path_bytes', 'Path bytes'), ('entries', 'Listing entries'), + ('file_bytes', 'File bytes'), + ), root_rows) + root_links = ''.join( + '
  • ' + + html.escape(configured.root_id) + '
  • ' + for configured in service.managed_file_roots.roots + ) + content = ( + '

    Logical roots

    ' + '

    Only configured logical roots are exposed. Host paths are never accepted.

    ' + + roots + '
      ' + root_links + '
    ' + ) + if root is not None: + location = root.root_id + (f'/{relative_path}' if relative_path else '') + entries = [] + if listing is not None: + for entry in listing.entries: + child_path = ( + f'{relative_path}/{entry.name}' if relative_path else entry.name + ) + if entry.kind == 'directory': + name = ( + '' + html.escape(entry.name) + '/' + ) + elif root.permissions.allow_read: + name = ( + '' + html.escape(entry.name) + '' + ) + else: + name = html.escape(entry.name) + entries.append( + '' + name + '' + html.escape(entry.kind) + + '' + + html.escape('' if entry.byte_count is None else str(entry.byte_count)) + + '' + ) + listing_block = '

    Listing is not permitted for this root.

    ' + if listing is not None: + listing_block = ( + '' + '' + '' + ''.join(entries) + '
    NameKindBytes
    ' + ) + parent = '' + if relative_path: + parent_path = relative_path.rpartition('/')[0] or None + parent = ( + '

    Parent directory

    ' + ) + content += ( + '

    ' + html.escape(location) + '

    ' + parent + + listing_block + '

    Typed mutations

    ' + + _render_managed_file_forms(service, root, relative_path) + + '
    ' + ) + return _page_shell( + 'Files', 'Bounded descriptor-safe access by logical root.', + 'files', content, '.', + ) + + +def _filter_select(name, label, values, selected, *, include_empty=True): + options = [''] if include_empty else [] + for value, title in values: + current = ' selected' if selected.get(name) == value else '' + options.append( + f'' + ) + return ( + f'' + ) + + +def _render_worker_filters(selected, relative_root): + text_fields = ( + ('source', 'Source'), ('worker', 'Worker / device'), ('phase', 'Phase'), + ('category', 'Category'), ('code', 'Stable code'), + ) + controls = ''.join( + f'' + for name, label in text_fields + ) + controls += _filter_select('assignment', 'Assignment outcome', ( + ('accepted', 'Accepted'), ('prebundle_failed', 'Prebundle failed'), + ('expired', 'Expired'), ('unfinished', 'Unfinished'), + ), selected) + controls += _filter_select('scan', 'Scan outcome', ( + ('clean', 'Clean'), ('found', 'Found'), ('degraded', 'Degraded'), + ('error', 'Error'), ('skipped', 'Skipped'), + ('unavailable', 'Unavailable'), + ), selected) + controls += _filter_select('retryable', 'Retryability', ( + ('true', 'Retryable'), ('false', 'Not retryable'), + ), selected) + controls += _filter_select('window', 'Time window', ( + ('24h', 'Last 24 hours'), ('7d', 'Last 7 days'), + ('30d', 'Last 30 days'), ('90d', 'Last 90 days'), ('all', 'All'), + ), selected, include_empty=False) + controls += _filter_select('limit', 'Rows per heavy section', ( + ('25', '25 rows'), ('50', '50 rows'), ('100', '100 rows'), + ), selected, include_empty=False) + controls += _filter_select('details', 'Load detail sections', ( + ('assignments', 'Assignments only (fast)'), + ('diagnostics', 'Assignments + diagnostics'), + ('metrics', 'Assignments + durations'), + ('all', 'All detail sections'), + ), selected, include_empty=False) + return ( + f'
    {controls}
    ' + '
    ' + f'Clear filters' + '
    ' + ) + + +def _render_assignment_rows(assignments, relative_root): + headers = ( + 'Assignment', 'Source / target', 'Worker', 'Assignment outcome', + 'Scan outcome', 'Diagnostics', 'Phase / progress', 'Deadlines', + 'Slot / cap', 'Package identity', 'Ingestion / projection', + ) + rows = [] + for row in assignments: + reservation_id = int(row['reservation_id']) + terminal = row.get('finished_at') is not None + age_authority = 'assignment resolution' if terminal else 'current time' + deadline_label = 'at resolution' if terminal else 'remaining' + diagnostic_count = int(row.get('diagnostic_count') or 0) + protocol2 = str(row.get('protocol_version') or '') == '2' + if diagnostic_count: + diagnostic_state = ( + f'current stored: {diagnostic_count}: ' + f'{row.get("primary_diagnostic") or "none"}' + ) + elif protocol2: + diagnostic_state = 'current protocol-2: 0 diagnostics observed' + elif row.get('diagnostic_projection_version') == 1: + diagnostic_state = 'current projection: 0 diagnostics observed' + else: + diagnostic_state = 'legacy/unavailable' + if row.get('active_phase'): + phase = ( + f"latest persisted phase: {row['active_phase']}" + if terminal else f"current phase: {row['active_phase']}" + ) + phase_age = ( + f"{row['phase_age_seconds']}s" + if row.get('phase_age_seconds') is not None else 'unavailable' + ) + progress_age = ( + f"{row['last_progress_age_seconds']}s" + if row.get('last_progress_age_seconds') is not None else 'unavailable' + ) + elif protocol2: + phase = 'unavailable (no persisted progress)' + phase_age = progress_age = 'unavailable (no persisted progress)' + else: + phase = phase_age = progress_age = 'legacy/unavailable' + warning = row.get('scan_warning_summary') + warning_class = row.get('scan_warning_class') + if warning: + warning_prefix = f'persisted warning [{warning_class}]' if warning_class else 'persisted warning' + scan_outcome = f"{row.get('scan_outcome') or 'unavailable'}; {warning_prefix}: {warning}" + elif str(row.get('scan_outcome') or '').lower() == 'degraded': + scan_outcome = 'degraded; persisted warning detail unavailable' + else: + scan_outcome = str(row.get('scan_outcome') or 'unavailable') + package = '
    '.join(html.escape(str(value)) for value in ( + f"protocol {row.get('protocol_version') or 'unavailable'} / bundle {row.get('bundle_format_version') or 'unavailable'}", + f"platform {row.get('platform_tag') or 'unavailable'}", + f"code {row.get('code_manifest_sha256') or 'unavailable'}", + f"detector {row.get('detector_policy_sha256') or 'unavailable'}", + )) + deadlines = '
    '.join(html.escape(str(value)) for value in ( + f"scan: {row.get('scan_deadline_at') or 'legacy/unavailable'} ({row.get('scan_remaining_seconds') if row.get('scan_remaining_seconds') is not None else 'unavailable'}s {deadline_label})", + ( + f"result upload: {row['remote_result_upload_body_timeout_seconds']}s " + '(persisted at assignment issuance)' + if row.get('remote_result_upload_body_timeout_seconds') is not None + else 'result upload: legacy/unavailable (not persisted)' + ), + f"assignment: {row.get('assignment_deadline_at') or 'unavailable'} ({row.get('assignment_remaining_seconds') if row.get('assignment_remaining_seconds') is not None else 'unavailable'}s {deadline_label})", + )) + cells = ( + ('assignment', f'{reservation_id}'), + ('source_target', html.escape(f"{row.get('source') or ''}: {row.get('target') or ''}")), + ('worker', html.escape(f"{row.get('user_key') or ''} / {row.get('device_key') or ''}")), + ('assignment_outcome', html.escape(str(row.get('assignment_outcome') or 'unavailable'))), + ('scan_outcome', html.escape(scan_outcome)), + ('diagnostics', html.escape(diagnostic_state)), + ('progress', html.escape( + f'{phase}; phase age {phase_age}; progress age {progress_age}; ' + f'age authority: {age_authority}' + )), + ('deadlines', deadlines), + ('slot_cap', html.escape( + f"{row.get('slot_id') if row.get('slot_id') is not None else 'unavailable'} / {row.get('active_assignment_cap') or 0}" + )), + ('package', package), + ('pipeline', html.escape( + f"{row.get('ingestion_state') or 'unavailable'} / {row.get('projection_state') or 'unavailable'}" + )), + ) + rows.append( + f'' + + ''.join( + f'{cell}' for field, cell in cells + ) + '' + ) + header = ''.join(f'{html.escape(value)}' for value in headers) + return f'{header}{"".join(rows)}
    ' + + +def _render_diagnostic_groups(groups, relative_root): + if not groups: + return '

    No diagnostic occurrences match the selected filters.

    ' + rendered = [] + for group in groups: + occurrences = ''.join( + '
  • ' + f'' + f'assignment {item["reservation_id"]}: ' + f'{html.escape(item["occurred_at"])}; {html.escape(item["assignment_outcome"])} / ' + f'{html.escape(item["scan_outcome"])}; {html.escape(item["phase"])}; ' + f'{html.escape(item["category"] + "/" + item["code"])}; ' + f'retryable={html.escape(str(item["retryable"]).lower())}' + '
  • ' + for item in group['occurrences'] + ) + rendered.append( + '
    ' + f'

    {html.escape(group["fingerprint"])}

    ' + f'

    {group["count"]} total filtered occurrences across ' + f'{group["affected_assignment_count"]} assignments; ' + f'{group["page_occurrence_count"]} occurrences on this page. ' + f'Page assignments: ' + f'{html.escape(", ".join(str(value) for value in group["affected_assignments"]))}.

    ' + f'
      {occurrences}
    ' + ) + return ''.join(rendered) + + +def _render_diagnostic_group_status(snapshot, selected, relative_root): + matched = int(snapshot.get('matched_occurrence_count') or 0) + page_count = int(snapshot.get('page_occurrence_count') or 0) + offset = int(snapshot.get('occurrence_offset') or 0) + limit = int(snapshot.get('occurrence_limit') or 0) + status = ( + f'

    Matched occurrences: {matched}. Showing {page_count} occurrences ' + f'at deterministic occurrence offset {offset} with page limit {limit}. ' + 'Fingerprint counts and affected-assignment counts cover the full filtered set.

    ' + ) + + def link(label, target_offset): + query = { + key: value for key, value in selected.items() + if value and key != 'diagnostic_offset' + } + query['diagnostic_offset'] = str(target_offset) + return ( + f'' + f'{html.escape(label)}' + ) + + links = [] + previous_offset = snapshot.get('previous_occurrence_offset') + if previous_offset is not None: + links.append(link('Previous diagnostic occurrences', int(previous_offset))) + next_offset = snapshot.get('next_occurrence_offset') + if next_offset is not None: + links.append(link('Next diagnostic occurrences', int(next_offset))) + if links: + status += '
    ' + ''.join(links) + '
    ' + return status + + +def _render_duration_metrics(metrics): + rows = [] + for metric in metrics: + sufficient = bool(metric.get('sufficient')) + percentile = lambda name: ( + f"{metric[name]:.3f}" + if sufficient else 'insufficient' + ) + rows.append({ + **metric, + 'p50': percentile('p50_seconds'), + 'p95': percentile('p95_seconds'), + 'p99': percentile('p99_seconds'), + 'sample_label': ( + str(metric['sample_count']) if sufficient + else f"{metric['sample_count']} (minimum {metric['minimum_sample_count']})" + ), + }) + return _table(( + ('source', 'Source'), ('phase', 'Phase'), ('outcome', 'Outcome'), + ('p50', 'p50 seconds'), ('p95', 'p95 seconds'), ('p99', 'p99 seconds'), + ('sample_label', 'Samples'), + ), rows) + + +def _render_metric_status(snapshot, selected, relative_root): + total = int(snapshot.get('total_group_count') or 0) + page_count = int(snapshot.get('page_group_count') or 0) + offset = int(snapshot.get('metric_offset') or 0) + limit = int(snapshot.get('metric_limit') or 0) + status = ( + f'

    Total source / phase / outcome groups: {total}. Showing ' + f'{page_count} groups at offset {offset} with page limit {limit}.

    ' + ) + + def link(label, target_offset): + query = { + key: value for key, value in selected.items() + if value and key != 'metric_offset' + } + query['metric_offset'] = str(target_offset) + return ( + f'' + f'{html.escape(label)}' + ) + + links = [] + previous_offset = snapshot.get('previous_metric_offset') + if previous_offset is not None: + links.append(link('Previous duration groups', int(previous_offset))) + next_offset = snapshot.get('next_metric_offset') + if next_offset is not None: + links.append(link('Next duration groups', int(next_offset))) + if links: + status += '
    ' + ''.join(links) + '
    ' + return status + + +def _render_deadline_policy(policy): + return ( + '

    Future assignments only. Saving or applying a policy ' + 'does not alter deadlines on existing assignments.

    ' + + _table(( + ('source', 'Source'), ('scan_deadline_seconds', 'Scan deadline seconds'), + ('upload_deadline_seconds', 'Upload deadline seconds'), + ('assignment_deadline_seconds', 'Assignment deadline seconds'), + ('assignment_policy_source', 'Assignment value source'), + ('handoff_margin_seconds', 'Handoff margin seconds'), + ('required_minimum_seconds', 'Required minimum'), + ('relationship', 'Validation relationship'), ('valid', 'Valid'), + ), policy.get('rows') or []) + ) + + +def _render_workers_page(service, snapshot, notice='', issued_token=None, relative_root='.'): + snapshot = dict(snapshot or {}) + worker_snapshot = _component_value(snapshot, 'workers', dict) + diagnostic_snapshot = _component_value(snapshot, 'diagnostics', dict) + metric_snapshot = _component_value(snapshot, 'metrics', dict) + policy = _component_value(snapshot, 'policy', dict) + control = _component_value(snapshot, 'control', dict) + packages = _component_value(snapshot, 'packages', dict) + selected_filters = dict(snapshot.get('selected_filters') or {}) + users = list((worker_snapshot or {}).get('users') or []) + workers = list((worker_snapshot or {}).get('workers') or []) + assignments = list((worker_snapshot or {}).get('assignments') or []) + deferred = list((worker_snapshot or {}).get('deferred_queue') or []) + token_block = '' + if issued_token: + token_block = ( + '

    Device token

    ' + '

    Shown once. Store it now; only its SHA-256 digest was persisted.

    ' + f'{html.escape(issued_token)}
    ' + ) + notice_block = f'

    {html.escape(notice)}

    ' if notice else '' + forms = ''.join(( + _form('/users/create', 'Create user', service.csrf_token, ( + ('user_key', 'User key', 'text'), + ('active_assignment_cap', 'Assignment cap', 'number', 0, 10000), + ), relative_root, hidden=(('operation_id', str(uuid.uuid4())),)), + _form('/users/cap', 'Update user cap', service.csrf_token, ( + ('user_key', 'User key', 'text'), + ('active_assignment_cap', 'Assignment cap', 'number', 0, 10000), + ), relative_root, hidden=(('operation_id', str(uuid.uuid4())),)), + _form('/users/disable', 'Disable user', service.csrf_token, ( + ('user_key', 'User key', 'text'), + ), relative_root, hidden=(('operation_id', str(uuid.uuid4())),)), + _form('/users/enable', 'Enable user', service.csrf_token, ( + ('user_key', 'User key', 'text'), + ), relative_root, hidden=(('operation_id', str(uuid.uuid4())),)), + _form('/devices/issue', 'Issue device token', service.csrf_token, ( + ('user_key', 'User key', 'text'), ('device_key', 'Device key', 'text'), + ), relative_root, hidden=(('operation_id', str(uuid.uuid4())),)), + _form('/devices/rotate', 'Rotate device token', service.csrf_token, ( + ('user_key', 'User key', 'text'), ('device_key', 'Device key', 'text'), + ), relative_root, hidden=(('operation_id', str(uuid.uuid4())),)), + _form('/devices/revoke', 'Revoke device', service.csrf_token, ( + ('device_key', 'Device key', 'text'), + ), relative_root, hidden=(('operation_id', str(uuid.uuid4())),)), + _form('/devices/unrevoke', 'Unrevoke device', service.csrf_token, ( + ('device_key', 'Device key', 'text'), + ), relative_root, hidden=(('operation_id', str(uuid.uuid4())),)), + _form('/queue/requeue', 'Requeue deferred targets', service.csrf_token, ( + ('queue_ids', 'Queue IDs, comma separated', 'text'), + ), relative_root, hidden=(('operation_id', str(uuid.uuid4())),)), + _form( + '/queue/discard-source', 'Discard stale unassigned source backlog', + service.csrf_token, ( + ('source', 'Source', 'text'), + ('confirm_source', 'Type source again to confirm', 'text'), + ), relative_root, hidden=(('operation_id', str(uuid.uuid4())),), + ), + )) + if worker_snapshot is None: + worker_note = '

    Worker administration snapshot unavailable.

    ' + else: + worker_note = '' + + if control is None: + dispatch_content = '

    Dispatch control unavailable.

    ' + dispatch_forms = '' + else: + revision = control['revision'] + dispatch_content = _table(( + ('revision', 'Revision'), ('dispatch_paused', 'Explicit pause'), + ('effective_dispatch_paused', 'Effective pause'), + ('drain_state', 'Drain state'), + ('live_remote_assignments', 'Live assignments'), + ('precommit_result_bundles', 'Pre-commit bundles'), + ('blocker_count', 'Drain blockers'), ('actor', 'Last actor'), + ('updated_at', 'Updated'), + ), (control,)) + hidden = (('expected_revision', revision),) + dispatch_forms = '
    ' + ''.join(( + _form( + '/dispatch/pause', 'Pause new assignments', service.csrf_token, + (), relative_root, + hidden=hidden + (('operation_id', str(uuid.uuid4())),), + ), + _form( + '/dispatch/resume', 'Resume new assignments', service.csrf_token, + (), relative_root, + hidden=hidden + (('operation_id', str(uuid.uuid4())),), + ), + _form( + '/dispatch/drain/start', 'Start drain', service.csrf_token, + (), relative_root, + hidden=hidden + (('operation_id', str(uuid.uuid4())),), + ), + _form( + '/dispatch/drain/cancel', 'Cancel drain', service.csrf_token, + (), relative_root, + hidden=hidden + (('operation_id', str(uuid.uuid4())),), + ), + )) + '
    ' + + if packages is None: + package_content = '

    Package compatibility unavailable.

    ' + else: + profile_rows = [] + capability_rows = [] + for profile in packages['profiles']: + profile_rows.append({ + 'profile_name': profile['profile_name'], + 'protocol_version': profile['protocol_version'], + 'bundle_format_version': profile['bundle_format_version'], + 'platform_tag': profile['platform_tag'], + 'code_manifest_sha256': profile['code_manifest_sha256'], + 'detector_policy_sha256': profile['detector_policy_sha256'], + 'sources': ', '.join(profile['sources']), + }) + capability_rows.extend({ + 'profile_name': profile['profile_name'], **capability, + } for capability in profile['capabilities']) + required_rows = packages['required_capabilities'] + package_content = ( + '

    Trusted package profiles

    ' + + _table(( + ('profile_name', 'Profile'), ('protocol_version', 'Protocol'), + ('bundle_format_version', 'Bundle format'), + ('platform_tag', 'Platform tag'), ('sources', 'Sources'), + ('code_manifest_sha256', 'Code manifest SHA-256'), + ('detector_policy_sha256', 'Detector policy SHA-256'), + ), profile_rows) + + '

    Profile capabilities

    ' + + _table(( + ('profile_name', 'Profile'), ('source', 'Source'), + ('platform', 'Worker platform'), ('planning_kind', 'Planning kind'), + ), capability_rows) + + '

    Required capabilities

    ' + + _table(( + ('source', 'Source'), ('platform', 'Worker platform'), + ('planning_kind', 'Planning kind'), + ), required_rows) + ) + + assignment_table = _render_assignment_rows(assignments, relative_root) + loaded_details = selected_filters.get('details') or 'assignments' + if loaded_details not in {'diagnostics', 'all'}: + diagnostic_content = ( + '

    Not loaded. Select Assignments + diagnostics or All detail ' + 'sections in the filter above.

    ' + ) + elif diagnostic_snapshot is None: + diagnostic_content = '

    Diagnostic grouping unavailable.

    ' + else: + diagnostic_content = _render_diagnostic_group_status( + diagnostic_snapshot, selected_filters, relative_root, + ) + _render_diagnostic_groups( + diagnostic_snapshot.get('groups') or [], relative_root, + ) + if loaded_details not in {'metrics', 'all'}: + metrics_content = ( + '

    Not loaded. Select Assignments + durations or All detail ' + 'sections in the filter above.

    ' + ) + elif metric_snapshot is None: + metrics_content = '

    Duration metrics unavailable.

    ' + else: + metrics_content = _render_metric_status( + metric_snapshot, selected_filters, relative_root, + ) + _render_duration_metrics(metric_snapshot.get('metrics') or []) + policy_content = ( + '

    Effective deadline policy unavailable.

    ' + if policy is None else _render_deadline_policy(policy) + ) + for worker in workers: + active = [ + item for item in assignments + if item.get('device_key') == worker.get('device_key') + and item.get('assignment_outcome') == 'unfinished' + ] + worker['current_phases'] = worker.get('current_phases') or ( + ', '.join(sorted({ + str(item.get('active_phase') or 'legacy/unavailable') for item in active + })) or 'none' + ) + progress_ages = [ + int(item['last_progress_age_seconds']) for item in active + if item.get('last_progress_age_seconds') is not None + ] + if worker.get('latest_progress_age_seconds') is None: + worker['latest_progress_age_seconds'] = ( + min(progress_ages) if progress_ages else None + ) + worker['activity'] = ( + 'active progress' if worker.get('latest_progress_age_seconds') is not None + else 'active, progress unavailable' + if worker.get('active_slot_count') else 'no active assignment' + ) + + worker_scope = ( + '

    Observability scope: Filters apply independently to the compatible ' + 'fields in Assignments, Repeated diagnostics, Observed durations, Effective deadline ' + 'policy, and Deferred queue. Users, Workers, Dispatch, Package compatibility, and Typed ' + 'operations remain global.

    ' + ) + content = f'''{notice_block}{token_block}

    Filters

    +

    Every filter is independent. Diagnostic filters do not imply an assignment or scan outcome. Expensive diagnostic and duration sections are queried only when selected.

    {worker_scope}{_render_worker_filters(selected_filters, relative_root)}
    +

    Dispatch and drain

    +

    Pausing or draining blocks new assignment commits. Authenticated status, terminal reports, uploads, receipt replay, expiry, ingestion, projection, and maintenance remain available.

    {dispatch_content}{dispatch_forms}
    +

    Package compatibility

    Protocol-1 workers are completion-only; new claims require a matching trusted protocol-2 package profile.

    {package_content}
    +

    Users

    {worker_note}{_table((('user_key', 'User'), ('active_assignment_cap', 'Cap'), ('disabled', 'Disabled')), users)}
    +

    Workers

    {_table((('device_key', 'Device'), ('user_key', 'User'), ('active_assignment_cap', 'Cap'), ('active_slot_count', 'Active slots'), ('activity', 'Observed activity'), ('current_phases', 'Current phases'), ('latest_progress_age_seconds', 'Latest progress age seconds'), ('last_contact_at', 'Last API contact'), ('known_reasons', 'Known idle / backoff reason'), ('pending_local_recovery', 'Pending local recovery'), ('active_package_identity', 'Active package identity'), ('completed_count', 'Accepted'), ('failed_count', 'Prebundle failed'), ('expired_count', 'Expired'), ('revoked', 'Revoked')), workers)}
    +

    Assignments

    Assignment transport, scan outcome, diagnostics, progress, deadlines, slot/cap, package identity, ingestion, and projection are separate authoritative fields.

    {assignment_table}
    +

    Repeated diagnostics

    Grouping is deterministic and every bounded occurrence remains linked below its fingerprint.

    {diagnostic_content}
    +

    Observed durations

    Percentiles are observations only and never change policy automatically. Rows below five samples are explicitly insufficient.

    {metrics_content}
    +

    Effective deadline policy

    {policy_content}
    +

    Deferred queue

    {_table((('queue_id', 'Queue'), ('source', 'Source'), ('target', 'Target'), ('available_after', 'Available after')), deferred)}
    +

    Typed operations

    Discarding source backlog suspends only never-issued pending, deferred, and cold rows. Active, previously issued, and completed targets are preserved. If discovery sees a discarded target again, the same queue row is reactivated with fresh discovery data.

    {forms}
    ''' + return _page_shell( + 'Workers / Dispatch', 'Authoritative records only; no liveness inference.', + 'workers', content, relative_root, + ) + + +def _render_material(name, material): + if not isinstance(material, dict): + return f'

    {html.escape(name)}

    Not captured.

    ' + fields = ({ + 'encoding': material.get('encoding'), + 'original_size': material.get('original_size'), + 'stored_size': material.get('stored_size'), + 'sha256': material.get('sha256'), + 'truncated': material.get('truncated'), + 'state': 'truncated transformation' if material.get('truncated') else 'complete', + },) + head = str(material.get('head') or '') + tail = material.get('tail') + stored = html.escape(head) + if tail is not None: + omitted = max( + 0, int(material.get('original_size') or 0) + - int(material.get('stored_size') or 0), + ) + stored += ( + f'\n[... {omitted} original bytes omitted by the stored transformation ...]\n' + + html.escape(str(tail)) + ) + return ( + f'

    {html.escape(name)}

    ' + + _table(( + ('state', 'State'), ('encoding', 'Stored encoding'), + ('original_size', 'Original bytes'), ('stored_size', 'Stored bytes'), + ('sha256', 'Original SHA-256'), ('truncated', 'Truncated'), + ), fields) + + f'
    {stored}
    ' + ) + + +def _render_assignment_detail_page(detail, relative_root='..'): + assignment = detail['assignment'] + reservation_id = int(assignment['reservation_id']) + overview = ({ + **assignment, + **detail['deadlines'], + 'diagnostic_availability': detail['diagnostics']['availability'], + 'historical_upload_timeout': ( + f"{detail['deadlines']['upload_timeout_seconds']}s " + '(persisted at assignment issuance)' + if detail['deadlines'].get('upload_timeout_seconds') is not None + else 'legacy/unavailable (not persisted for this assignment)' + ), + },) + timeline = _table(( + ('timestamp', 'Timestamp'), ('kind', 'Kind'), ('label', 'Event'), + ('sequence', 'Sequence'), ('received_at', 'Server received'), + ), detail['timeline']) + durations = _table(( + ('phase', 'Phase'), ('duration_seconds', 'Duration seconds'), + ('outcome', 'Outcome'), ('complete', 'Complete'), + ('ended_at', 'Observed through'), ('authority', 'End authority'), + ), detail['durations']) + progress = detail['progress'] + progress_summary = _table(( + ('availability', 'Availability'), + ('current_phase', 'Current / latest phase'), + ('phase_started_at', 'Phase started'), + ('phase_age_seconds', 'Phase age seconds'), + ('last_progress_at', 'Last progress'), + ('last_progress_age_seconds', 'Last progress age seconds'), + ('age_authority', 'Age authority'), + ('total_event_count', 'Total persisted events'), + ('omitted_older_event_count', 'Older events omitted by bound'), + ), (progress,)) + scan = _table(tuple((name, label) for name, label in ( + ('available', 'Available'), ('target_scan_id', 'Target scan'), + ('status', 'Status'), ('started_at', 'Started'), ('ended_at', 'Ended'), + ('duration_seconds', 'Duration seconds'), ('findings_count', 'Findings'), + ('verified_findings_count', 'Verified findings'), ('error_count', 'Errors'), + ('skipped_reason', 'Skipped reason'), + ('first_error_summary', 'First error summary'), + ('warning_class', 'First persisted warning class'), + ('warning_summary', 'First persisted warning'), + )), (detail['scan'],)) + transport = _table(( + ('receipt_id', 'Receipt'), ('bundle_state', 'Bundle state'), + ('bundle_ready_at', 'Bundle received'), + ('bundle_committed_at', 'Ingestion committed'), + ('bundle_acknowledged_at', 'Bundle acknowledged'), + ('queue_status', 'Queue status'), ('queue_settled_at', 'Queue settled'), + ('projection_status', 'Projection status'), + ('projection_completed_at', 'Projection completed'), + ), (detail['transport'],)) + package = _table(tuple( + (key, key.replace('_', ' ').title()) for key in detail['package'] + ), (detail['package'],)) + current_policy = detail.get('current_effective_policy') + current_policy_html = ( + _render_deadline_policy({'rows': [current_policy]}) + if current_policy else + '

    Current effective source policy unavailable.

    ' + ) + + diagnostics = [] + for row in detail['diagnostics']['items']: + envelope = row['diagnostic'] + uid = str(row['diagnostic_uid']) + target_id = f'diagnostic-json-{uid}' + download = f'./{reservation_id}/diagnostics/{uid}.json' + materials = [] + http_context = envelope.get('http') or {} + process_context = envelope.get('process') or {} + if http_context: + materials.append(_render_material('HTTP body', http_context.get('body'))) + materials.append(_render_material('HTTP headers', http_context.get('headers'))) + if process_context: + materials.append(_render_material('Process stdout', process_context.get('stdout'))) + materials.append(_render_material('Process stderr', process_context.get('stderr'))) + if not materials: + materials.append('

    No body or process-log material was captured.

    ') + summary = ({ + 'diagnostic_uid': uid, + 'phase': row.get('phase'), 'kind': row.get('kind'), + 'category': row.get('category'), 'code': row.get('code'), + 'retryable': bool(row.get('retryable')), + 'summary': row.get('summary'), 'occurred_at': row.get('occurred_at'), + 'received_at': row.get('received_at'), + 'canonical_sha256': row.get('envelope_sha256'), + 'transformation': ( + 'schema validation plus canonical ASCII JSON with sorted keys and compact separators' + ), + },) + diagnostics.append( + f'
    ' + f'

    {html.escape(str(row.get("category")))} / ' + f'{html.escape(str(row.get("code")))}

    ' + + _table(tuple((key, key.replace('_', ' ').title()) for key in summary[0]), summary) + + '
    ' + f'' + f'Download canonical JSON' + '
    ' + f'' + + ''.join(materials) + '
    ' + ) + if not diagnostics: + diagnostics.append( + '

    No canonical diagnostic envelope is stored. ' + f'Availability: {html.escape(detail["diagnostics"]["availability"])}.

    ' + ) + + legacy = detail['legacy_evidence'] + legacy_rows = legacy.get('errors') or [] + legacy_content = ( + '

    These fields are legacy evidence, not a synthesized diagnostic envelope.

    ' + + _table(( + ('id', 'Error'), ('category', 'Category'), ('summary', 'Summary'), + ('created_at', 'Created'), + ), legacy_rows) + + ''.join( + '

    Exact stored raw_error

    ' + f'
    {html.escape(str(row.get("raw_error") or ""))}
    ' + for row in legacy_rows + ) + ) if legacy.get('available') else ( + '

    No legacy error evidence is stored. No envelope has been invented.

    ' + ) + truncation_notes = [] + for name in ('progress', 'diagnostics'): + if detail[name].get('truncated'): + if name == 'progress': + truncation_notes.append( + f'progress retained the latest events chronologically; ' + f'{detail[name].get("omitted_older_event_count", 0)} older events omitted' + ) + else: + truncation_notes.append(f'{name} reached the server result bound') + if legacy.get('truncated'): + truncation_notes.append('legacy evidence reached the server result bound') + truncation = ( + '

    ' + html.escape('; '.join(truncation_notes)) + '.

    ' + if truncation_notes else '' + ) + content = f''' +

    Assignment

    +

    Machine-readable detail

    +{_table((('reservation_id', 'Reservation'), ('queue_id', 'Queue'), ('source', 'Source'), ('target', 'Target'), ('user', 'User'), ('worker', 'Worker / device'), ('active_assignment_cap', 'Configured cap'), ('assignment_outcome', 'Assignment outcome'), ('scan_outcome', 'Scan outcome'), ('issued_at', 'Issued'), ('resolved_at', 'Resolved'), ('assignment_code', 'Assignment code'), ('assignment_detail', 'Assignment detail'), ('diagnostic_availability', 'Diagnostic availability'), ('scan_deadline_at', 'Scan deadline'), ('historical_upload_timeout', 'Historical result-upload timeout'), ('assignment_deadline_at', 'Assignment deadline')), overview)}
    +

    Progress authority

    {progress_summary}
    +

    Ordered timeline

    {truncation}{timeline}
    +

    Duration breakdown

    {durations}
    +

    Transport, receipt, ingestion, settlement, projection

    {transport}
    +

    Scan summary

    {scan}
    +

    Package identity

    {package}
    +

    Current effective source policy

    This is current policy context; the stored assignment deadline above remains immutable.

    {current_policy_html}
    +

    Diagnostics

    {''.join(diagnostics)}
    +

    Legacy evidence

    {legacy_content}
    ''' + return _page_shell( + f'Worker assignment {reservation_id}', + 'Authoritative timestamps and exact stored diagnostic material.', + 'workers', content, relative_root, script='../admin.js', + ) + + +def _component_value(snapshot, name, expected_type): + component = snapshot.get(name) if isinstance(snapshot, dict) else None + if not isinstance(component, dict) or component.get('available') is not True: + return None + value = component.get('value') + return value if isinstance(value, expected_type) else None + + +def _safe_count(value): + return value if type(value) is int and value >= 0 else '' + + +def _render_overview_page(snapshot, relative_root='.'): + runtime = _component_value(snapshot, 'runtime', dict) + queue = _component_value(snapshot, 'queue', dict) + control = _component_value(snapshot, 'control', dict) + operations = _component_value(snapshot, 'operations', list) + + if runtime is None: + runtime_content = '

    Runtime snapshot unavailable.

    ' + producers = [] + pipeline_workers = [] + pipeline = {} + else: + runtime_state = runtime.get('runtime') if isinstance(runtime.get('runtime'), dict) else {} + postgres = runtime.get('postgres') if isinstance(runtime.get('postgres'), dict) else {} + sources = runtime.get('sources') if isinstance(runtime.get('sources'), list) else [] + by_id = { + item.get('id'): item for item in sources + if isinstance(item, dict) and isinstance(item.get('id'), str) + } + runtime_content = _table(( + ('component', 'Component'), ('state', 'State'), ('ready', 'Ready'), + ('safe_error_category', 'Safe error'), + ), ( + { + 'component': 'Supervisor', 'state': runtime_state.get('phase', ''), + 'ready': ( + runtime_state.get('phase') == 'ACTIVE' + and runtime_state.get('start_gate_open') is True + and runtime_state.get('shutdown_requested') is False + and runtime_state.get('runtime_failed') is False + ), + 'safe_error_category': 'runtime_failed' if runtime_state.get('runtime_failed') else '', + }, + { + 'component': 'PostgreSQL', 'state': postgres.get('state', ''), + 'ready': postgres.get('ready', False), + 'safe_error_category': postgres.get('safe_error_category', ''), + }, + )) + producers = [] + for source in CORE_PRODUCERS: + item = by_id.get(f'discovery-producer:{source}', {}) + cycle = item.get('last_cycle_result') if isinstance(item.get('last_cycle_result'), dict) else {} + producers.append({ + 'source': source, 'role': item.get('role', 'unavailable'), + 'lifecycle_state': item.get('lifecycle_state', 'unavailable'), + 'last_cycle': cycle.get('status', ''), + 'last_success': item.get('last_successful_discovery_at', ''), + 'next_run': item.get('next_scheduled_run_at', ''), + 'safe_error_category': item.get('safe_error_category', ''), + }) + pipeline_workers = [] + for source_id in PIPELINE_SOURCE_IDS: + item = by_id.get(source_id, {}) + pipeline_workers.append({ + 'id': source_id, 'role': item.get('role', 'unavailable'), + 'lifecycle_state': item.get('lifecycle_state', 'unavailable'), + 'process_state': item.get('process_state', 'unavailable'), + 'desired_state': item.get('desired_state', ''), + 'safe_error_category': item.get('safe_error_category', ''), + }) + pipeline = runtime.get('pipeline') if isinstance(runtime.get('pipeline'), dict) else {} + + pipeline_rows = ({ + 'metric': 'Ingester lease', + 'state': pipeline.get('ingester_state', '') if type(pipeline.get('ingester_state')) is str else '', + 'ready': pipeline.get('ingester_ready', '') if type(pipeline.get('ingester_ready')) is bool else '', + }, { + 'metric': 'Projector lease', + 'state': pipeline.get('projector_state', '') if type(pipeline.get('projector_state')) is str else '', + 'ready': pipeline.get('projector_ready', '') if type(pipeline.get('projector_ready')) is bool else '', + }, { + 'metric': 'Final cutover', + 'state': ( + 'valid' if pipeline.get('cutover_ready') is True else + 'invalid' if pipeline.get('cutover_ready') is False else '' + ), + 'ready': pipeline.get('cutover_ready', '') if type(pipeline.get('cutover_ready')) is bool else '', + }) + + queue_rows = [] + if queue is not None and isinstance(queue.get('counts'), dict): + truncated = set(queue.get('truncated_statuses') or ()) + queue_rows = [ + { + 'status': status, 'count': _safe_count(queue['counts'].get(status, 0)), + 'truncated': status in truncated, + } + for status in QUEUE_STATUSES + ] + queue_note = ( + f'Bounded sample: {html.escape(str(queue.get("sampled_rows", "")))} rows; ' + f'degraded={html.escape(str(bool(queue.get("degraded"))))}; ' + f'stale={html.escape(str(bool(queue.get("stale"))))}.' + ) + else: + queue_note = 'Queue snapshot unavailable.' + + if control is None: + control_rows = [] + control_note = 'Control snapshot unavailable.' + else: + control_rows = ({ + 'revision': control.get('revision', ''), + 'discovery_paused': control.get('effective_discovery_paused', ''), + 'dispatch_paused': control.get('effective_dispatch_paused', ''), + 'drain_state': control.get('drain_state', ''), + 'assignments': _safe_count(control.get('live_remote_assignments')), + 'bundles': _safe_count(control.get('precommit_result_bundles')), + 'updated_at': control.get('updated_at', ''), + },) + control_note = '' + + operation_rows = [] + if operations is not None: + for item in operations: + if not isinstance(item, dict): + continue + operation_rows.append({ + key: item.get(key, '') for key in ( + 'operation_id', 'actor', 'action', 'target_ref', 'status', + 'safe_category', 'safe_detail', 'requested_at', 'completed_at', + 'updated_at', + ) + }) + operations_note = '' + else: + operations_note = 'Recent operations unavailable.' + + content = f''' +

    Runtime health

    {runtime_content}
    +

    Discovery producers

    {_table((('source', 'Source'), ('role', 'Role'), ('lifecycle_state', 'Lifecycle'), ('last_cycle', 'Last cycle'), ('last_success', 'Last success'), ('next_run', 'Next run'), ('safe_error_category', 'Safe error')), producers)}
    +

    Pipeline workers

    {_table((('id', 'ID'), ('role', 'Role'), ('lifecycle_state', 'Lifecycle'), ('process_state', 'Process'), ('desired_state', 'Desired'), ('safe_error_category', 'Safe error')), pipeline_workers)}
    +

    Pipeline authority

    {_table((('metric', 'Metric'), ('state', 'State'), ('ready', 'Ready')), pipeline_rows)}
    +

    Queue status

    {queue_note}

    {_table((('status', 'Status'), ('count', 'Bounded count'), ('truncated', 'Truncated')), queue_rows)}
    +

    Assignments, bundles and control

    {control_note}

    {_table((('revision', 'Revision'), ('discovery_paused', 'Discovery paused'), ('dispatch_paused', 'Dispatch paused'), ('drain_state', 'Drain'), ('assignments', 'Active assignments'), ('bundles', 'Pre-commit bundles'), ('updated_at', 'Updated')), control_rows)} +

    Bundle items: {html.escape(str(_safe_count(pipeline.get('bundle_items'))))}; bundle bytes: {html.escape(str(_safe_count(pipeline.get('bundle_bytes'))))}.

    +

    Recent operation outcomes

    {operations_note}

    {_table((('operation_id', 'Operation'), ('actor', 'Actor'), ('action', 'Action'), ('target_ref', 'Target'), ('status', 'Status'), ('safe_category', 'Category'), ('safe_detail', 'Detail'), ('requested_at', 'Requested'), ('completed_at', 'Completed'), ('updated_at', 'Updated')), operation_rows)}
    ''' + return _page_shell( + 'Runtime overview', 'Bounded structured health; unavailable components degrade independently.', + 'overview', content, relative_root, + ) + + +def _render_search_page(service, snapshot, relative_root='.'): + runtime = _component_value(snapshot, 'runtime', dict) + control = _component_value(snapshot, 'control', dict) + + if control is None: + control_content = '

    Discovery control unavailable.

    ' + control_forms = '' + else: + revision = control.get('revision', '') + control_content = _table(( + ('revision', 'Revision'), ('discovery_paused', 'Explicit pause'), + ('effective_discovery_paused', 'Effective pause'), + ('drain_state', 'Drain'), ('actor', 'Last actor'), + ('updated_at', 'Updated'), + ), ({ + 'revision': revision, + 'discovery_paused': control.get('discovery_paused', ''), + 'effective_discovery_paused': control.get('effective_discovery_paused', ''), + 'drain_state': control.get('drain_state', ''), + 'actor': control.get('actor', ''), + 'updated_at': control.get('updated_at', ''), + },)) + control_forms = '
    ' + ''.join(( + _form( + '/search/discovery/pause', 'Pause discovery', service.csrf_token, + (), relative_root, hidden=( + ('expected_revision', revision), ('operation_id', str(uuid.uuid4())), + ), + ), + _form( + '/search/discovery/resume', 'Resume discovery', service.csrf_token, + (), relative_root, hidden=( + ('expected_revision', revision), ('operation_id', str(uuid.uuid4())), + ), + ), + )) + '
    ' + + if runtime is None: + producer_content = '

    Producer state unavailable.

    ' + producer_forms = '' + else: + sources = runtime.get('sources') if isinstance(runtime.get('sources'), list) else [] + by_id = { + item.get('id'): item for item in sources + if isinstance(item, dict) and item.get('id') in PRODUCER_IDS + } + producer_rows = [] + forms = [] + for source_id in PRODUCER_IDS: + item = by_id.get(source_id, {}) + cycle = item.get('last_cycle_result') if isinstance( + item.get('last_cycle_result'), dict, + ) else {} + producer_rows.append({ + 'id': source_id, + 'lifecycle_state': item.get('lifecycle_state', 'unavailable'), + 'desired_state': item.get('desired_state', ''), + 'process_state': item.get('process_state', ''), + 'interval_seconds': item.get('interval_seconds', ''), + 'restart_enabled': item.get('restart_enabled', ''), + 'restart_count': item.get('restart_count', ''), + 'last_cycle': cycle.get('status', ''), + 'fetched': cycle.get('fetched_count', ''), + 'queued_new': cycle.get('queued_new_count', ''), + 'queued_updated': cycle.get('queued_updated_count', ''), + 'last_success': item.get('last_successful_discovery_at', ''), + 'next_run': item.get('next_scheduled_run_at', ''), + 'safe_error_category': item.get('safe_error_category', ''), + }) + label = source_id.split(':', 1)[1] + for action in ('start', 'stop', 'restart', 'pause', 'resume'): + forms.append(_form( + f'/search/producers/{action}', + f'{action.title()} {label}', service.csrf_token, (), relative_root, + hidden=( + ('source_id', source_id), ('operation_id', str(uuid.uuid4())), + ), + )) + forms.append(_form( + '/search/producers/interval', f'Set {label} interval', + service.csrf_token, + (( + 'interval_seconds', 'Interval seconds', 'number', 1, + MAX_PRODUCER_INTERVAL_SECONDS, + ),), + relative_root, hidden=( + ('source_id', source_id), ('operation_id', str(uuid.uuid4())), + ), + values={'interval_seconds': item.get('interval_seconds')}, + )) + producer_content = _table(( + ('id', 'Producer'), ('lifecycle_state', 'Lifecycle'), + ('desired_state', 'Desired'), ('process_state', 'Process'), + ('interval_seconds', 'Interval seconds'), + ('restart_enabled', 'Restart'), ('restart_count', 'Restarts'), + ('last_cycle', 'Last cycle'), ('fetched', 'Fetched'), + ('queued_new', 'Queued new'), ('queued_updated', 'Queued updated'), + ('last_success', 'Last success'), ('next_run', 'Next run'), + ('safe_error_category', 'Safe error'), + ), producer_rows) + producer_forms = '
    ' + ''.join(forms) + '
    ' + + content = f''' +

    Persistent discovery control

    Stored in PostgreSQL and preserved across runtime restarts.

    {control_content}{control_forms}
    +

    Discovery producers

    Lifecycle and interval changes affect the current Supervisor runtime only.

    {producer_content}{producer_forms}
    ''' + return _page_shell( + 'Search operations', 'Persistent discovery authority and exact producer controls.', + 'search', content, relative_root, + ) + + +def _render_supervisor_page(service, snapshot, relative_root='.'): + if not isinstance(snapshot, dict): + return _page_shell( + 'Supervisor', 'Structured managed-process state and typed controls.', + 'supervisor', '

    Supervisor state unavailable.

    ', + relative_root, + ) + runtime = snapshot.get('runtime') if isinstance(snapshot.get('runtime'), dict) else {} + postgres = snapshot.get('postgres') if isinstance(snapshot.get('postgres'), dict) else {} + dashboard = snapshot.get('dashboard') if isinstance(snapshot.get('dashboard'), dict) else {} + sources = [ + item for item in snapshot.get('sources', []) + if isinstance(item, dict) + and type(item.get('id')) is str + and KEY_RE.fullmatch(item['id']) + and isinstance(item.get('allowed_actions'), list) + and all(action in ALL_MANAGED_SOURCE_ACTIONS for action in item['allowed_actions']) + ] + content = '

    Runtime

    ' + _table(( + ('phase', 'Phase'), ('pid', 'PID'), ('start_gate_open', 'Start gate'), + ('shutdown_requested', 'Shutdown requested'), ('runtime_failed', 'Failed'), + ('postgres_state', 'PostgreSQL'), ('postgres_ready', 'PostgreSQL ready'), + ('postgres_error', 'PostgreSQL safe error'), + ), ({ + 'phase': runtime.get('phase', ''), 'pid': runtime.get('pid', ''), + 'start_gate_open': runtime.get('start_gate_open', ''), + 'shutdown_requested': runtime.get('shutdown_requested', ''), + 'runtime_failed': runtime.get('runtime_failed', ''), + 'postgres_state': postgres.get('state', ''), + 'postgres_ready': postgres.get('ready', ''), + 'postgres_error': postgres.get('safe_error_category', ''), + },)) + '
    ' + content += '

    Dashboard

    ' + _table(( + ('status', 'Status'), ('desired_state', 'Desired state'), + ('healthy', 'Healthy'), ('pid', 'PID'), ('safe_error_category', 'Safe error'), + ), (dashboard,)) + dashboard_forms = ''.join(_form( + f'/supervisor/dashboard/{action}', f'{action.title()} dashboard', + service.csrf_token, (), relative_root, + hidden=(('operation_id', str(uuid.uuid4())),), + ) for action in ('start', 'stop', 'restart')) + dashboard_summary = 'Dashboard controls' + if dashboard.get('status'): + dashboard_summary += f' - {dashboard["status"]}' + content += ( + '
    ' + + html.escape(dashboard_summary) + + '
    ' + + dashboard_forms + + '
    ' + ) + content += '

    Managed sources

    ' + _table(( + ('id', 'ID'), ('role', 'Role'), ('lifecycle_state', 'Lifecycle'), + ('desired_state', 'Desired'), ('process_state', 'Process'), ('pid', 'PID'), + ('mode', 'Mode'), ('interval_seconds', 'Interval'), + ('restart_enabled', 'Restart'), ('restart_delay_seconds', 'Restart delay'), + ('restart_count', 'Restarts'), ('restart_streak', 'Restart streak'), + ('safe_error_category', 'Safe error'), + ), sources) + panels = [] + for source in sources: + source_id = source['id'] + allowed = source.get('allowed_actions') + if not isinstance(allowed, list): + allowed = [] + forms = [] + for action in ('start', 'stop', 'restart', 'pause', 'resume', 'once'): + if action in allowed: + forms.append(_form( + f'/supervisor/sources/{action}', + f'{action.title()} {source_id}', service.csrf_token, (), relative_root, + hidden=( + ('source_id', source_id), ('operation_id', str(uuid.uuid4())), + ), + )) + if 'set-mode' in allowed: + for mode in ('loop', 'once', 'repeat'): + if source_id == 'keychecks' and mode == 'loop': + continue + forms.append(_form( + f'/supervisor/sources/mode/{mode}', + f'Set {source_id} mode {mode}', service.csrf_token, (), relative_root, + hidden=( + ('source_id', source_id), ('operation_id', str(uuid.uuid4())), + ), + )) + if 'set-interval' in allowed: + forms.append(_form( + '/supervisor/sources/interval', f'Set {source_id} interval', + service.csrf_token, (( + 'interval_seconds', 'Interval seconds', 'number', 1, + MAX_MANAGED_SOURCE_DELAY_SECONDS, + ),), relative_root, hidden=( + ('source_id', source_id), ('operation_id', str(uuid.uuid4())), + ), + values={'interval_seconds': source.get('interval_seconds')}, + )) + if 'set-restart' in allowed: + for enabled, label in ((True, 'Enable'), (False, 'Disable')): + route = 'enable' if enabled else 'disable' + forms.append(_form( + f'/supervisor/sources/restart-policy/{route}', + f'{label} {source_id} restart', service.csrf_token, (), relative_root, + hidden=( + ('source_id', source_id), ('operation_id', str(uuid.uuid4())), + ), + )) + if 'set-restart-delay' in allowed: + forms.append(_form( + '/supervisor/sources/restart-delay', + f'Set {source_id} restart delay', service.csrf_token, (( + 'restart_delay_seconds', 'Restart delay seconds', 'number', 1, + MAX_MANAGED_SOURCE_DELAY_SECONDS, + ),), relative_root, hidden=( + ('source_id', source_id), ('operation_id', str(uuid.uuid4())), + ), + values={ + 'restart_delay_seconds': source.get('restart_delay_seconds'), + }, + )) + summary_parts = [source_id] + for key in ('lifecycle_state', 'process_state'): + value = source.get(key) + if value: + summary_parts.append(str(value)) + panels.append( + '
    ' + + html.escape(' - '.join(summary_parts)) + + '
    ' + + ''.join(forms) + + '
    ' + ) + content += '
    ' + ''.join(panels) + '
    ' + return _page_shell( + 'Supervisor', 'Structured managed-process state and typed controls.', + 'supervisor', content, relative_root, + ) + + +def _render_logs_page(service, snapshot, tail=None, relative_root='.'): + if snapshot is None: + content = '

    Managed source list unavailable.

    ' + sources = [] + else: + sources = [ + item for item in snapshot.get('sources', []) + if isinstance(item, dict) + and type(item.get('id')) is str + and KEY_RE.fullmatch(item['id']) + and isinstance(item.get('allowed_actions'), list) + and all(action in ALL_MANAGED_SOURCE_ACTIONS for action in item['allowed_actions']) + ] + content = ( + '

    Managed source logs

    ' + '

    Only the active bounded log for an allowlisted source is available.

    ' + '
    ' + + ''.join(_form( + '/logs/tail', f'Tail {item["id"]}', service.csrf_token, (( + 'line_count', 'Newest lines', 'number', 1, + MAX_MANAGED_SOURCE_LOG_LINES, + ),), relative_root, hidden=(('source_id', item['id']),), + values={'line_count': 40}, + ) for item in sources) + + '
    ' + ) + if isinstance(tail, dict): + lines = '\n'.join(tail.get('lines') or []) + suffix = ' Response was truncated.' if tail.get('response_truncated') else '' + content += ( + '

    Tail result

    ' + + html.escape( + f'{tail.get("source_id", "")} · {tail.get("line_count", 0)} lines.{suffix}' + ) + + '

    ' + html.escape(lines) + '
    ' + ) + return _page_shell( + 'Managed logs', 'Bounded allowlisted log tails; no path or command input.', + 'logs', content, relative_root, + ) + + +def _runtime_document_hashes(state): + return { + 'active_config': state.active_config.sha256, + 'active_secrets': state.active_secrets.sha256, + 'candidate_config': state.candidate_config.sha256, + 'candidate_secrets': state.candidate_secrets.sha256, + } + + +def _render_runtime_document_page( + service, document, editor, *, document_text=None, preview=None, + notice='', relative_root='.', operation_id=None, observability=None, +): + state = preview.state if preview is not None else editor.state + text = editor.text if document_text is None else document_text + hashes = _runtime_document_hashes(state) + hidden = ''.join( + f'' + for name, value in hashes.items() + ) + operation_id = str(uuid.uuid4()) if operation_id is None else operation_id + editor_form = ( + f'
    ' + f'' + f'{hidden}' + f'' + '
    ' + f'
    ' + '
    ' + ) + identity_rows = ({ + 'active_config': hashes['active_config'], + 'active_secrets': hashes['active_secrets'], + 'candidate_config': hashes['candidate_config'], + 'candidate_secrets': hashes['candidate_secrets'], + 'config_candidate_present': state.candidate_config.present, + 'secrets_candidate_present': state.candidate_secrets.present, + },) + content = ( + (f'

    {html.escape(notice)}

    ' if notice else '') + + f'

    {html.escape(document.title())} candidate

    ' + + f'

    Editing source: {html.escape(editor.source)}. Save stages a private candidate and never activates it.

    ' + + _table(( + ('active_config', 'Active config SHA-256'), + ('active_secrets', 'Active secrets SHA-256'), + ('candidate_config', 'Candidate config SHA-256'), + ('candidate_secrets', 'Candidate secrets SHA-256'), + ('config_candidate_present', 'Config candidate'), + ('secrets_candidate_present', 'Secrets candidate'), + ), identity_rows) + + editor_form + '
    ' + ) + if preview is not None: + diff = preview.diff + if hasattr(diff, 'entries'): + rows = [ + { + 'path': entry.path, 'change': entry.change, + 'before': entry.before, 'after': entry.after, + 'redacted': entry.value_redacted, + } + for entry in diff.entries + ] + diff_html = _table(( + ('path', 'Path'), ('change', 'Change'), ('before', 'Before'), + ('after', 'After'), ('redacted', 'Redacted'), + ), rows) + diff_html += ( + f'

    Truncated: {html.escape(str(diff.truncated))}; ' + f'format-only change: {html.escape(str(diff.format_only_changed))}.

    ' + ) + else: + fields = ( + 'document_changed', 'semantic_changed', 'pools_before', 'pools_after', + 'pools_added', 'pools_removed', 'pools_changed', 'entries_before', + 'entries_after', 'entries_added', 'entries_removed', 'pools_reordered', + 'usernames_added', 'usernames_removed', 'usernames_changed', + 'tokens_changed', + ) + diff_html = _table(tuple((field, field.replace('_', ' ').title()) for field in fields), ({ + field: getattr(diff, field) for field in fields + },)) + content += '

    Validated preview

    ' + diff_html + '
    ' + if document == 'config': + observability = dict(observability or {}) + policy = _component_value(observability, 'policy', dict) + metrics = _component_value(observability, 'metrics', dict) + policy_html = ( + '

    Effective deadline policy unavailable.

    ' + if policy is None else _render_deadline_policy(policy) + ) + metrics_html = ( + '

    Duration metrics unavailable.

    ' + if metrics is None else ( + f'

    Total source / phase / outcome groups: ' + f'{int(metrics.get("total_group_count") or 0)}. ' + + ( + 'Additional groups are available through paginated Workers observability.

    ' + if metrics.get('has_next') else '

    ' + ) + + _render_duration_metrics(metrics.get('metrics') or []) + ) + ) + candidate_label = ( + 'Validated candidate effective values' if preview is not None + else f'{editor.source.title()} editor effective values' + ) + content += ( + f'

    {html.escape(candidate_label)}

    {policy_html}' + '

    Observed source / phase / outcome durations

    ' + '

    p50/p95/p99 values include sample counts and never mutate policy.

    ' + f'{metrics_html}
    ' + ) + relevant_candidate = ( + state.candidate_config if document == 'config' else state.candidate_secrets + ) + relevant_active = ( + state.active_config if document == 'config' else state.active_secrets + ) + can_apply_document = bool( + relevant_candidate.present + and relevant_candidate.sha256 != relevant_active.sha256 + ) + can_apply_both = bool( + state.candidate_config.present + and state.candidate_secrets.present + and ( + state.candidate_config.sha256 != state.active_config.sha256 + or state.candidate_secrets.sha256 != state.active_secrets.sha256 + ) + ) + apply_hidden = [ + ('operation_id', str(uuid.uuid4())), + ('expected_active_config_sha256', hashes['active_config']), + ('expected_active_secrets_sha256', hashes['active_secrets']), + (f'expected_candidate_{document}_sha256', relevant_candidate.sha256), + ] + apply_form = _form( + f'/{document}/apply', f'Apply {document} candidate', service.csrf_token, + (), relative_root, hidden=tuple(apply_hidden), + ) if can_apply_document else '' + both_form = _form( + '/runtime/apply-both', 'Apply both candidates', service.csrf_token, (), + relative_root, hidden=( + ('operation_id', str(uuid.uuid4())), + ('expected_active_config_sha256', hashes['active_config']), + ('expected_active_secrets_sha256', hashes['active_secrets']), + ('expected_candidate_config_sha256', hashes['candidate_config']), + ('expected_candidate_secrets_sha256', hashes['candidate_secrets']), + ), + ) if can_apply_both else '' + apply_controls = apply_form + both_form + apply_state = ( + f'
    {apply_controls}
    ' if apply_controls else + '

    No staged candidate differs from the active document. ' + 'Save a changed candidate before applying.

    ' + ) + content += ( + '

    Apply

    Apply requires the fixed host agent. ' + 'The operation is hash-bound and durable before dispatch.

    ' + f'{apply_state}
    ' + ) + return _page_shell( + f'{document.title()} editor', + 'Plaintext is rendered only inside this protected no-store editor.', + document, content, relative_root, + ) + + +def _redirect(location): + response = _secure_response('', status_code=303) + response.headers['Location'] = location + return response + + +class _ManagedFileStreamingResponse(StreamingResponse): + def __init__(self, snapshot, *args, **kwargs): + self._managed_file_snapshot = snapshot + super().__init__(*args, **kwargs) + + async def __call__(self, scope, receive, send): + try: + await super().__call__(scope, receive, send) + finally: + self._managed_file_snapshot.close() + + +def _managed_file_download_response(download, relative_path): + leaf_name = relative_path.rsplit('/', 1)[-1] + headers = dict(SECURITY_HEADERS) + headers['Content-Disposition'] = ( + "attachment; filename*=UTF-8''" + quote(leaf_name, safe='') + ) + headers['Content-Length'] = str(download.identity.byte_count) + headers['ETag'] = f'"{download.identity.sha256}"' + if download.snapshot is not None: + try: + return _ManagedFileStreamingResponse( + download.snapshot, download.snapshot.chunks(), + media_type='application/octet-stream', headers=headers, + ) + except BaseException: + download.snapshot.close() + raise + return Response( + download.content, media_type='application/octet-stream', headers=headers, + ) + + +async def _download_managed_file( + service, traversal, root_id, relative_path): + task = asyncio.create_task(asyncio.to_thread( + service.download_managed_file, + traversal, root_id, relative_path, + )) + try: + return await asyncio.shield(task) + except asyncio.CancelledError: + def close_snapshot(completed): + try: + download = completed.result() + except BaseException: + return + if download.snapshot is not None: + download.snapshot.close() + + task.add_done_callback(close_snapshot) + raise + + +ADMIN_CSS = ''' +:root { color-scheme: light; font-family: ui-monospace, Consolas, monospace; background: #f3f0e8; color: #18211d; } +body { margin: 0; } +header, main { box-sizing: border-box; width: 100%; max-width: 92rem; min-width: 0; margin: 0 auto; padding: 1.25rem; } +header { border-bottom: 4px solid #18211d; } +nav { display: flex; gap: .5rem; flex-wrap: wrap; } +nav a { color: inherit; border: 1px solid #6d746f; padding: .4rem .65rem; text-decoration: none; background: #fffdf6; } +nav a[aria-current="page"] { background: #18211d; color: #fffdf6; } +section, article { box-sizing: border-box; max-width: 100%; min-width: 0; } +section { margin: 1.5rem 0; overflow-x: auto; } +table { width: 100%; border-collapse: collapse; background: #fffdf6; } +th, td { border: 1px solid #9b9a8e; padding: .45rem; text-align: left; overflow-wrap: anywhere; } +.forms { display: grid; grid-template-columns: repeat(auto-fit, minmax(16rem, 1fr)); gap: .75rem; } +.control-panels { display: grid; gap: .75rem; margin-top: 1rem; } +.control-panel { border: 1px solid #6d746f; background: #e4eee8; } +.control-panel > summary { cursor: pointer; font-weight: bold; padding: .8rem; } +.control-panel[open] > summary { border-bottom: 1px solid #6d746f; } +.control-panel > .forms { padding: .8rem; } +form { border: 1px solid #6d746f; background: #fffdf6; padding: .8rem; } +label, input, select, button { display: block; box-sizing: border-box; width: 100%; margin: .45rem 0; } +input, select, button, textarea { font: inherit; padding: .45rem; box-sizing: border-box; } +textarea { display: block; width: 100%; resize: vertical; white-space: pre; overflow: auto; } +.button-row { display: grid; grid-template-columns: repeat(auto-fit, minmax(10rem, 1fr)); gap: .5rem; } +button { background: #174c3c; color: white; border: 0; cursor: pointer; } +.button-link { display: block; box-sizing: border-box; margin: .45rem 0; padding: .45rem; text-align: center; background: #fffdf6; border: 1px solid #174c3c; color: #174c3c; text-decoration: none; } +.filter-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(12rem, 1fr)); gap: .5rem; } +pre { max-height: 40rem; overflow: auto; white-space: pre-wrap; overflow-wrap: anywhere; background: #18211d; color: #fffdf6; padding: .8rem; } +.diagnostic, .diagnostic-group, .material { border: 1px solid #6d746f; padding: .8rem; margin: .8rem 0; background: #fffdf6; } +.diagnostic-group code { overflow-wrap: anywhere; } +.canonical-json { background: #f3f0e8; min-height: 12rem; } +.issued { border: 3px solid #8b2f24; padding: 1rem; background: #fff4df; } +.issued code { overflow-wrap: anywhere; } +.notice { border-left: 4px solid #174c3c; padding: .7rem; background: #e4eee8; } +.unavailable { color: #8b2f24; font-weight: bold; } +@media (max-width: 48rem) { + header, main { padding: .75rem; } + table { width: max-content; min-width: 100%; } + th, td { overflow-wrap: normal; word-break: normal; } +} +'''.strip() + + +ADMIN_JS = ''' +document.addEventListener('click', async (event) => { + const button = event.target.closest('[data-copy-target]'); + if (!button) return; + const target = document.getElementById(button.dataset.copyTarget); + if (!target) return; + await navigator.clipboard.writeText(target.value); + button.textContent = 'Copied canonical JSON'; +}); +'''.strip() + + +async def _dispatch_runtime_document(request, service, document, action): + fields = hashes = editor = preview = document_text = operation_id = None + try: + if action in ('preview', 'save'): + expected = { + 'csrf_token', 'document_text', 'operation_id', + 'expected_active_config_sha256', 'expected_active_secrets_sha256', + 'expected_candidate_config_sha256', 'expected_candidate_secrets_sha256', + } + max_document = ( + MAX_CONFIG_DOCUMENT_BYTES if document == 'config' + else MAX_SECRETS_DOCUMENT_BYTES + ) + fields = await _form_fields( + request, service, expected, body_limit=(max_document * 3) + 4096, + ) + document_text = fields.pop('document_text') + hashes = _parse_document_hashes(fields) + operation_id = _parse_operation_id(fields['operation_id']) + editor = await asyncio.to_thread(service.runtime_document_editor, document) + try: + if action == 'preview': + preview = await asyncio.to_thread( + service.preview_runtime_document, document, document_text, + ) + observability = None + if document == 'config': + observability = await asyncio.to_thread( + service.runtime_document_observability, document_text, + ) + return _secure_response( + _render_runtime_document_page( + service, document, editor, document_text=document_text, + preview=preview, + notice='Candidate is valid; nothing was saved.', + relative_root='..', operation_id=operation_id, + observability=observability, + ), + media_type='text/html', + ) + await asyncio.to_thread( + service.save_runtime_document_candidate, + document, document_text, request.state.admin_actor, + operation_id, expected_hashes=hashes, + ) + except AdminAPIError as exc: + if action == 'save': + editor = await asyncio.to_thread( + service.runtime_document_editor, document, + ) + observability = None + if document == 'config': + observability = await asyncio.to_thread( + service.runtime_document_observability, document_text, + ) + return _secure_response( + _render_runtime_document_page( + service, document, editor, document_text=document_text, + notice=str(exc), relative_root='..', + operation_id=operation_id, + observability=observability, + ), + status_code=exc.status_code, media_type='text/html', + ) + return _redirect(f'../{document}') + + apply_action = f'apply-{document}' if document != 'both' else 'apply-both' + candidate_fields = ( + {'expected_candidate_config_sha256'} if apply_action == 'apply-config' + else {'expected_candidate_secrets_sha256'} if apply_action == 'apply-secrets' + else {'expected_candidate_config_sha256', 'expected_candidate_secrets_sha256'} + ) + fields = await _form_fields( + request, service, + { + 'csrf_token', 'operation_id', 'expected_active_config_sha256', + 'expected_active_secrets_sha256', *candidate_fields, + }, + ) + hashes = _parse_document_hashes( + fields, + require_config=apply_action in ('apply-config', 'apply-both'), + require_secrets=apply_action in ('apply-secrets', 'apply-both'), + ) + operation_id = _parse_operation_id(fields['operation_id']) + await asyncio.to_thread( + service.request_runtime_apply, + apply_action, request.state.admin_actor, operation_id, + expected_hashes=hashes, + ) + return _redirect(f'../operations/{operation_id}') + finally: + if isinstance(fields, dict): + fields.clear() + fields = hashes = editor = preview = document_text = operation_id = None + + +async def _dispatch_managed_file_mutation(request, service, action): + fields = payload = encoded_content = operation_id = expected_sha256 = None + try: + expected = {'csrf_token', 'operation_id', 'root_id', 'relative_path'} + if action in ('create', 'replace'): + expected.add('content_base64') + if action in ('replace', 'delete'): + expected.add('expected_sha256') + fields = await _form_fields(request, service, expected) + operation_id = _parse_operation_id(fields['operation_id']) + if action in ('replace', 'delete'): + expected_sha256 = _parse_managed_file_hash(fields['expected_sha256']) + if action in ('create', 'replace'): + encoded_content = fields.pop('content_base64') + payload = _parse_managed_file_content(encoded_content) + encoded_content = None + traversal = getattr(request.app.state, 'managed_file_traversal', None) + if action == 'create': + await asyncio.to_thread( + service.create_managed_file, traversal, fields['root_id'], + fields['relative_path'], payload, request.state.admin_actor, + operation_id, + ) + elif action == 'replace': + await asyncio.to_thread( + service.replace_managed_file, traversal, fields['root_id'], + fields['relative_path'], payload, expected_sha256, + request.state.admin_actor, operation_id, + ) + else: + await asyncio.to_thread( + service.delete_managed_file, traversal, fields['root_id'], + fields['relative_path'], expected_sha256, + request.state.admin_actor, operation_id, + ) + return _redirect(_managed_file_listing_location( + fields['root_id'], fields['relative_path'], + )) + finally: + if isinstance(fields, dict): + fields.clear() + fields = payload = encoded_content = operation_id = expected_sha256 = None + + +async def _dispatch(request, service): + path = str(request.scope.get('path') or '') + raw_path = request.scope.get('raw_path') + try: + canonical_raw_path = path.encode('ascii') + except UnicodeEncodeError: + canonical_raw_path = None + if type(raw_path) is not bytes or raw_path != canonical_raw_path: + raise AdminAPIError(404, 'page not found') + method = request.method.upper() + allowed_query = set() + if method == 'GET' and path == f'{ADMIN_PREFIX}/audit': + allowed_query = {'before'} + elif method == 'GET' and path == f'{ADMIN_PREFIX}/operations': + allowed_query = {'before'} + elif method == 'GET' and path in (ADMIN_PREFIX, f'{ADMIN_PREFIX}/'): + allowed_query = WORKER_FILTER_FIELDS + elif method == 'GET' and path in ( + f'{ADMIN_PREFIX}/files', f'{ADMIN_PREFIX}/files/download', + ): + allowed_query = {'root_id', 'relative_path'} + query = _query_fields(request, allowed_query) + if method == 'GET': + if path in (ADMIN_PREFIX, f'{ADMIN_PREFIX}/'): + filters, selected = _parse_worker_filters(query) + snapshot = await asyncio.to_thread( + service.workers_dispatch_snapshot, filters, + diagnostic_occurrence_offset=int(selected['diagnostic_offset']), + metric_offset=int(selected['metric_offset']), + page_limit=int(selected['limit']), + include_diagnostics=selected['details'] in {'diagnostics', 'all'}, + include_metrics=selected['details'] in {'metrics', 'all'}, + ) + snapshot['selected_filters'] = selected + return _secure_response(_render_workers_page(service, snapshot), media_type='text/html') + if path == f'{ADMIN_PREFIX}/overview': + snapshot = await asyncio.to_thread(service.overview) + return _secure_response( + _render_overview_page(snapshot), media_type='text/html', + ) + if path == f'{ADMIN_PREFIX}/search': + snapshot = await asyncio.to_thread(service.search_snapshot) + return _secure_response( + _render_search_page(service, snapshot), media_type='text/html', + ) + if path in (f'{ADMIN_PREFIX}/supervisor', f'{ADMIN_PREFIX}/logs'): + try: + snapshot = await asyncio.to_thread(service._runtime_snapshot) + except Exception: + snapshot = None + renderer = _render_supervisor_page if path.endswith('/supervisor') else _render_logs_page + return _secure_response( + renderer(service, snapshot), media_type='text/html', + ) + if path in (f'{ADMIN_PREFIX}/config', f'{ADMIN_PREFIX}/secrets'): + document = path.rsplit('/', 1)[-1] + editor = await asyncio.to_thread(service.runtime_document_editor, document) + observability = None + if document == 'config': + observability = await asyncio.to_thread( + service.runtime_document_observability, editor.text, + ) + return _secure_response( + _render_runtime_document_page( + service, document, editor, observability=observability, + ), + media_type='text/html', + ) + if path == f'{ADMIN_PREFIX}/files': + if set(query) not in (set(), {'root_id'}, {'root_id', 'relative_path'}): + raise AdminAPIError(400, 'query shape is invalid') + root = listing = None + relative_path = query.get('relative_path') + if 'root_id' in query: + root = service.managed_file_roots.get(query['root_id']) + if root is None: + raise AdminAPIError(404, 'managed file target was not found') + if relative_path == '': + raise AdminAPIError(400, 'managed file request is invalid') + if relative_path is not None: + try: + parse_managed_relative_path(relative_path, root.limits) + except ManagedFileAccessError as exc: + _managed_file_error(exc) + if root.permissions.allow_list: + listing = await asyncio.to_thread( + service.list_managed_files, + getattr(request.app.state, 'managed_file_traversal', None), + root.root_id, relative_path, + ) + return _secure_response( + _render_files_page( + service, root=root, relative_path=relative_path, + listing=listing, + ), + media_type='text/html', + ) + if path == f'{ADMIN_PREFIX}/files/download': + if set(query) != {'root_id', 'relative_path'} or not query['relative_path']: + raise AdminAPIError(400, 'query shape is invalid') + download = await _download_managed_file( + service, + getattr(request.app.state, 'managed_file_traversal', None), + query['root_id'], query['relative_path'], + ) + return _managed_file_download_response(download, query['relative_path']) + if path == f'{ADMIN_PREFIX}/operations': + before = ( + _parse_operation_cursor(query['before']) if 'before' in query else None + ) + page = await asyncio.to_thread(service.operation_page, before) + return _secure_response( + _render_operations_page(page), media_type='text/html', + ) + if path.startswith(f'{ADMIN_PREFIX}/operations/'): + try: + operation_id = _parse_operation_id( + path[len(f'{ADMIN_PREFIX}/operations/'):], + ) + except AdminAPIError as exc: + raise AdminAPIError(404, 'Not Found') from exc + operation = await asyncio.to_thread( + service.operation_status, operation_id, + ) + return _secure_response( + _render_operation_page(operation), media_type='text/html', + ) + if path == f'{ADMIN_PREFIX}/audit': + before_event_id = ( + _parse_audit_cursor(query['before']) if 'before' in query else None + ) + page = await asyncio.to_thread( + service.audit_page, before_event_id, + ) + return _secure_response( + _render_audit_page(page), media_type='text/html', + ) + assignment_path = path[len(f'{ADMIN_PREFIX}/assignments/'):] + match = re.fullmatch(r'([1-9][0-9]{0,18})(\.json)?', assignment_path) + if match: + reservation_id = _parse_assignment_id(match.group(1)) + detail = await asyncio.to_thread( + service.assignment_detail, reservation_id, + ) + if match.group(2): + return _secure_json_response(detail) + return _secure_response( + _render_assignment_detail_page(detail), media_type='text/html', + ) + match = re.fullmatch( + r'([1-9][0-9]{0,18})/diagnostics/([a-f0-9]{64})\.json', + assignment_path, + ) + if match: + reservation_id = _parse_assignment_id(match.group(1)) + diagnostic_uid = match.group(2) + envelope = await asyncio.to_thread( + service.diagnostic_envelope, reservation_id, diagnostic_uid, + ) + return _diagnostic_json_response(envelope, diagnostic_uid) + if path == f'{ADMIN_PREFIX}/admin.css': + return _secure_response(ADMIN_CSS, media_type='text/css') + if path == f'{ADMIN_PREFIX}/admin.js': + return _secure_response(ADMIN_JS, media_type='text/javascript') + raise AdminAPIError(404, 'Not Found') + if method != 'POST': + raise AdminAPIError(405, 'Method Not Allowed') + + notice = '' + issued_token = None + document_routes = { + f'{ADMIN_PREFIX}/{document}/{action}': (document, action) + for document in ('config', 'secrets') + for action in ('preview', 'save', 'apply') + } + document_routes[f'{ADMIN_PREFIX}/runtime/apply-both'] = ('both', 'apply') + if path in document_routes: + document, action = document_routes[path] + return await _dispatch_runtime_document( + request, service, document, action, + ) + managed_file_routes = { + f'{ADMIN_PREFIX}/files/{action}': action + for action in ('create', 'replace', 'delete') + } + if path in managed_file_routes: + return await _dispatch_managed_file_mutation( + request, service, managed_file_routes[path], + ) + + dispatch_routes = { + f'{ADMIN_PREFIX}/dispatch/pause': ('dispatch', True), + f'{ADMIN_PREFIX}/dispatch/resume': ('dispatch', False), + f'{ADMIN_PREFIX}/dispatch/drain/start': ('drain', 'start'), + f'{ADMIN_PREFIX}/dispatch/drain/cancel': ('drain', 'cancel'), + } + if path in dispatch_routes: + fields = await _form_fields( + request, service, {'csrf_token', 'expected_revision', 'operation_id'}, + ) + kind, value = dispatch_routes[path] + arguments = ( + _parse_revision(fields['expected_revision']), + request.state.admin_actor, + _parse_operation_id(fields['operation_id']), + ) + if kind == 'dispatch': + await asyncio.to_thread( + service.set_dispatch_paused, value, *arguments, + ) + else: + operation = service.start_drain if value == 'start' else service.cancel_drain + await asyncio.to_thread(operation, *arguments) + return _redirect('../' if kind == 'dispatch' else '../../') + if path in ( + f'{ADMIN_PREFIX}/search/discovery/pause', + f'{ADMIN_PREFIX}/search/discovery/resume', + ): + fields = await _form_fields( + request, service, {'csrf_token', 'expected_revision', 'operation_id'}, + ) + await asyncio.to_thread( + service.set_discovery_paused, + path.endswith('/pause'), _parse_revision(fields['expected_revision']), + request.state.admin_actor, _parse_operation_id(fields['operation_id']), + ) + return _redirect('../../search') + producer_routes = { + f'{ADMIN_PREFIX}/search/producers/{action}': action + for action in ('start', 'stop', 'restart', 'pause', 'resume') + } + if path in producer_routes: + fields = await _form_fields( + request, service, {'csrf_token', 'source_id', 'operation_id'}, + ) + await asyncio.to_thread( + service.producer_action, fields['source_id'], producer_routes[path], + request.state.admin_actor, _parse_operation_id(fields['operation_id']), + ) + return _redirect('../../search') + if path == f'{ADMIN_PREFIX}/search/producers/interval': + fields = await _form_fields( + request, service, { + 'csrf_token', 'source_id', 'interval_seconds', 'operation_id', + }, + ) + await asyncio.to_thread( + service.producer_action, fields['source_id'], 'set-interval', + request.state.admin_actor, _parse_operation_id(fields['operation_id']), + interval_seconds=_parse_producer_interval(fields['interval_seconds']), + ) + return _redirect('../../search') + source_routes = { + f'{ADMIN_PREFIX}/supervisor/sources/{action}': (action, {}, '../../supervisor') + for action in ('start', 'stop', 'restart', 'pause', 'resume', 'once') + } + source_routes.update({ + f'{ADMIN_PREFIX}/supervisor/sources/mode/{mode}': ( + 'set-mode', {'mode': mode}, '../../../supervisor', + ) + for mode in ('loop', 'once', 'repeat') + }) + source_routes.update({ + f'{ADMIN_PREFIX}/supervisor/sources/restart-policy/{value}': ( + 'set-restart', {'restart_enabled': value == 'enable'}, + '../../../supervisor', + ) + for value in ('enable', 'disable') + }) + if path in source_routes: + fields = await _form_fields( + request, service, {'csrf_token', 'source_id', 'operation_id'}, + ) + source_action, parameters, redirect = source_routes[path] + await asyncio.to_thread( + service.managed_source_action, fields['source_id'], source_action, + request.state.admin_actor, _parse_operation_id(fields['operation_id']), + **parameters, + ) + return _redirect(redirect) + if path in ( + f'{ADMIN_PREFIX}/supervisor/sources/interval', + f'{ADMIN_PREFIX}/supervisor/sources/restart-delay', + ): + field_name = 'interval_seconds' if path.endswith('/interval') else 'restart_delay_seconds' + fields = await _form_fields( + request, service, {'csrf_token', 'source_id', 'operation_id', field_name}, + ) + delay = _parse_managed_source_delay(fields[field_name], field_name.replace('_', ' ')) + await asyncio.to_thread( + service.managed_source_action, fields['source_id'], + 'set-interval' if field_name == 'interval_seconds' else 'set-restart-delay', + request.state.admin_actor, _parse_operation_id(fields['operation_id']), + **{field_name: delay}, + ) + return _redirect('../../supervisor') + dashboard_routes = { + f'{ADMIN_PREFIX}/supervisor/dashboard/{action}': action + for action in ('start', 'stop', 'restart') + } + if path in dashboard_routes: + fields = await _form_fields( + request, service, {'csrf_token', 'operation_id'}, + ) + await asyncio.to_thread( + service.managed_source_action, 'dashboard', dashboard_routes[path], + request.state.admin_actor, _parse_operation_id(fields['operation_id']), + ) + return _redirect('../../supervisor') + if path == f'{ADMIN_PREFIX}/logs/tail': + fields = await _form_fields( + request, service, {'csrf_token', 'source_id', 'line_count'}, + ) + line_count = _parse_managed_source_delay(fields['line_count'], 'log line count') + if line_count > MAX_MANAGED_SOURCE_LOG_LINES: + raise AdminAPIError(400, 'log line count is invalid') + tail = await asyncio.to_thread( + service.managed_source_log, fields['source_id'], line_count, + ) + try: + snapshot = await asyncio.to_thread(service._runtime_snapshot) + except Exception: + snapshot = None + return _secure_response( + _render_logs_page(service, snapshot, tail=tail, relative_root='..'), + media_type='text/html', + ) + if path in (f'{ADMIN_PREFIX}/users/create', f'{ADMIN_PREFIX}/users/cap'): + fields = await _form_fields( + request, service, { + 'csrf_token', 'user_key', 'active_assignment_cap', 'operation_id', + }, + ) + operation = service.create_user if path.endswith('/create') else service.set_user_cap + result = await run_in_threadpool( + operation, fields['user_key'], fields['active_assignment_cap'], + request.state.admin_actor, _parse_operation_id(fields['operation_id']), + ) + if not result: + raise AdminAPIError(409 if path.endswith('/create') else 404, 'user operation was not applied') + return _redirect('../') + elif path in (f'{ADMIN_PREFIX}/users/disable', f'{ADMIN_PREFIX}/users/enable'): + fields = await _form_fields( + request, service, {'csrf_token', 'user_key', 'operation_id'}, + ) + disabled = path.endswith('/disable') + result = await run_in_threadpool( + service.set_user_disabled, fields['user_key'], disabled, + request.state.admin_actor, _parse_operation_id(fields['operation_id']), + ) + if not result: + raise AdminAPIError(404, 'user was not found') + return _redirect('../') + elif path in (f'{ADMIN_PREFIX}/devices/issue', f'{ADMIN_PREFIX}/devices/rotate'): + fields = await _form_fields( + request, service, { + 'csrf_token', 'user_key', 'device_key', 'operation_id', + }, + ) + rotate = path.endswith('/rotate') + result, issued_token = await run_in_threadpool( + service.issue_device, fields['user_key'], fields['device_key'], + request.state.admin_actor, _parse_operation_id(fields['operation_id']), + rotate=rotate, + ) + if not result: + raise AdminAPIError(409 if not rotate else 404, 'device token operation was not applied') + notice = 'Device token rotated.' if rotate else 'Device token issued.' + elif path in (f'{ADMIN_PREFIX}/devices/revoke', f'{ADMIN_PREFIX}/devices/unrevoke'): + fields = await _form_fields( + request, service, {'csrf_token', 'device_key', 'operation_id'}, + ) + revoked = path.endswith('/revoke') + result = await run_in_threadpool( + service.set_device_revoked, fields['device_key'], revoked, + request.state.admin_actor, _parse_operation_id(fields['operation_id']), + ) + if not result: + raise AdminAPIError(404, 'device was not found') + return _redirect('../') + elif path == f'{ADMIN_PREFIX}/queue/requeue': + fields = await _form_fields( + request, service, {'csrf_token', 'queue_ids', 'operation_id'}, + ) + queue_ids = _parse_queue_ids(fields['queue_ids'], service.requeue_limit) + count = await run_in_threadpool( + service.requeue, queue_ids, request.state.admin_actor, + _parse_operation_id(fields['operation_id']), + ) + if type(count) is not int or count < 0: + raise AdminAPIError(500, 'queue operation returned an invalid result') + return _redirect('../') + elif path == f'{ADMIN_PREFIX}/queue/discard-source': + fields = await _form_fields( + request, service, + {'csrf_token', 'source', 'confirm_source', 'operation_id'}, + ) + source = _validate_key(fields['source'], 'source') + if fields['confirm_source'] != source: + raise AdminAPIError(400, 'source confirmation does not match') + count = await run_in_threadpool( + service.discard_source_queue, source, request.state.admin_actor, + _parse_operation_id(fields['operation_id']), + ) + if type(count) is not int or count < 0: + raise AdminAPIError(500, 'queue operation returned an invalid result') + return _redirect('../') + else: + raise AdminAPIError(404, 'Not Found') + + snapshot = await asyncio.to_thread(service.workers_dispatch_snapshot) + return _secure_response( + _render_workers_page( + service, snapshot, notice=notice, issued_token=issued_token, relative_root='..', + ), + media_type='text/html', + ) + + +async def admin_endpoint(request): + service = request.app.state.admin_service + actor = _trusted_operator(request, service) + if actor is None: + return _secure_response('Not Found', status_code=404) + request.state.admin_actor = actor + try: + return await _dispatch(request, service) + except AdminAPIError as exc: + return _secure_response(str(exc), status_code=exc.status_code) + except Exception as exc: + logger.error('Admin request failed: %s', type(exc).__name__) + return _secure_response('Internal Server Error', status_code=500) + + +class _AdminRoute(Route): + def matches(self, scope): + match, child_scope = super().matches(scope) + if match == Match.PARTIAL and scope.get('type') == 'http': + return Match.FULL, child_scope + return match, child_scope + + async def handle(self, scope, receive, send): + await self.app(scope, receive, send) + + +def admin_routes(): + methods = ['GET', 'POST', 'PUT', 'PATCH', 'DELETE', 'OPTIONS', 'HEAD', 'TRACE'] + return [ + _AdminRoute(ADMIN_PREFIX, admin_endpoint, methods=methods), + _AdminRoute(f'{ADMIN_PREFIX}/{{admin_path:path}}', admin_endpoint, methods=methods), + ] diff --git a/app/app.py b/app/app.py new file mode 100644 index 0000000..bfebb8c --- /dev/null +++ b/app/app.py @@ -0,0 +1,13 @@ +"""Retired legacy mutation UI. + +Scanner lifecycle control is intentionally available only through supervisor.py. +The read-only observability UI remains dashboard.py. +""" + +import streamlit as st + + +st.set_page_config(page_title='Scanner UI Retired', page_icon='LOCK', layout='centered') +st.title('Legacy scanner controls are retired') +st.error('This UI cannot start, pause, resume, cancel, or configure scans.') +st.info('Use the authenticated supervisor commands for lifecycle control and dashboard.py for read-only observability.') diff --git a/app/audit_github_tokens.py b/app/audit_github_tokens.py new file mode 100644 index 0000000..5fc5e30 --- /dev/null +++ b/app/audit_github_tokens.py @@ -0,0 +1,15 @@ +"""Retired direct GitHub credential audit entrypoint.""" + +import sys + +sys.dont_write_bytecode = True + + +def main(): + raise SystemExit( + 'This direct credential audit is retired. Use authenticated supervisor-managed GitHub keychecks.' + ) + + +if __name__ == '__main__': + main() diff --git a/app/capacity_model.py b/app/capacity_model.py new file mode 100644 index 0000000..ec1921a --- /dev/null +++ b/app/capacity_model.py @@ -0,0 +1,55 @@ +MAX_RESULT_BUNDLE_BYTES = 64 * 1024 * 1024 +REMOTE_ASSIGNMENT_BASELINE_BYTES = 2 * 1024 * 1024 +REMOTE_ASSIGNMENT_MAX_ACTIVE = 50 + + +def validate_remote_assignment_capacity(config): + values = {} + fields = ( + 'result_bundle_max_event_bytes', + 'remote_assignment_reserve_bytes', + 'remote_assignment_max_active', + 'result_bundle_max_total_bytes', + 'projection_backlog_max_bytes', + 'projection_backlog_headroom_bytes', + 'keycheck_queue_max_items', + 'keycheck_queue_max_bytes', + 'keycheck_candidates_per_event', + 'keycheck_candidate_bytes_per_event', + ) + for name in fields: + value = config.get(name) + if type(value) is not int or value < 0: + raise ValueError(name) + values[name] = value + + hard_limit = values['result_bundle_max_event_bytes'] + reserve = values['remote_assignment_reserve_bytes'] + active = values['remote_assignment_max_active'] + if not REMOTE_ASSIGNMENT_BASELINE_BYTES <= reserve <= hard_limit: + raise ValueError('remote_assignment_reserve_bytes') + if not REMOTE_ASSIGNMENT_BASELINE_BYTES <= hard_limit <= MAX_RESULT_BUNDLE_BYTES: + raise ValueError('result_bundle_max_event_bytes') + if not 1 <= active <= REMOTE_ASSIGNMENT_MAX_ACTIVE: + raise ValueError('remote_assignment_max_active') + + required_bytes = active * reserve + if values['result_bundle_max_total_bytes'] < required_bytes: + raise ValueError('result_bundle_max_total_bytes') + projection_admission_bytes = ( + values['projection_backlog_max_bytes'] + - values['projection_backlog_headroom_bytes'] + ) + if projection_admission_bytes < required_bytes: + raise ValueError('projection_backlog_max_bytes') + if ( + values['keycheck_queue_max_items'] + < active * values['keycheck_candidates_per_event'] + ): + raise ValueError('keycheck_queue_max_items') + if ( + values['keycheck_queue_max_bytes'] + < active * values['keycheck_candidate_bytes_per_event'] + ): + raise ValueError('keycheck_queue_max_bytes') + return values diff --git a/app/child_bootstrap.py b/app/child_bootstrap.py new file mode 100644 index 0000000..ed84d23 --- /dev/null +++ b/app/child_bootstrap.py @@ -0,0 +1,486 @@ +"""Stdlib-only authentication boundary for supervised application children.""" + +import sys + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('supervised child bootstrap could not disable bytecode writes') + +import hashlib +import hmac +import json +import os +import runpy +import socket +import stat + + +MAX_METADATA_BYTES = 256 * 1024 +MAX_HANDSHAKE_BYTES = 1024 * 1024 +MAX_PYVENV_BYTES = 64 * 1024 +MANIFEST_SCHEMA = 5 +APPLICATION_IMPORT_SUFFIXES = ('.py', '.pyw', '.pyc', '.pyd') +CONTROL_SCHEMA = 1 +REQUIRED_DEPENDENCIES = { + 'supervisor': ('psycopg', 'yaml'), + 'postgres-runtime': ('psycopg', 'yaml'), + 'migrate-runtime-safety': ('psycopg', 'yaml'), + 'scanner': ('psycopg', 'requests', 'urllib3', 'yaml', 'zstandard'), + 'discovery-producer': ('psycopg', 'requests', 'urllib3', 'yaml', 'zstandard'), + 'docker-shadow': ('psycopg', 'requests', 'urllib3', 'yaml', 'zstandard'), + 'keycheck': ('psycopg', 'requests', 'yaml'), + 'dashboard': ('pandas', 'plotly', 'psycopg', 'streamlit', 'yaml'), + 'keycheck-provider': ('boto3', 'botocore', 'psycopg', 'requests', 'yaml'), + 'janitor': ('yaml',), + 'result-ingester': ('psycopg', 'yaml'), + 'jsonl-projector': ('psycopg', 'yaml'), + 'worker-api': ('psycopg', 'requests', 'starlette', 'urllib3', 'uvicorn', 'yaml', 'zstandard'), +} + + +class ChildRuntimeError(RuntimeError): + pass + + +ENV = { + 'instance_file': 'TRUF_SUPERVISOR_INSTANCE_FILE', + 'instance_id': 'TRUF_SUPERVISOR_INSTANCE_ID', + 'token': 'TRUF_SUPERVISOR_TOKEN', + 'config_sha256': 'TRUF_SUPERVISOR_CONFIG_SHA256', + 'supervisor_sha256': 'TRUF_SUPERVISOR_SHA256', + 'code_manifest_sha256': 'TRUF_SUPERVISOR_CODE_MANIFEST_SHA256', + 'dsn_sha256': 'TRUF_SUPERVISOR_DSN_SHA256', + 'kind': 'TRUF_SUPERVISOR_CHILD_KIND', +} + + +def _canonical(path): + return os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(path)))) + + +def _is_reparse_point(path): + details = os.lstat(path) + if stat.S_ISLNK(details.st_mode): + return True + attributes = getattr(details, 'st_file_attributes', 0) + reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0) + return bool(attributes & reparse_attribute) or getattr(os.path, 'isjunction', lambda _path: False)(path) + + +def _contained(path, roots): + for root in roots: + try: + if path != root and os.path.commonpath((root, path)) == root: + return True + except ValueError: + continue + return False + + +def _validated_site_directory(path, roots): + if not path or not os.path.isdir(path): + return '' + candidate = _canonical(path) + trusted_roots = tuple(_canonical(root) for root in roots if root) + if os.path.basename(candidate).lower() not in ('site-packages', 'dist-packages'): + raise RuntimeError(f'interpreter dependency path is not a site-packages directory: {candidate}') + if not _contained(candidate, trusted_roots): + raise RuntimeError(f'interpreter dependency path escapes its trusted root: {candidate}') + return candidate + + +def _append_site_directories(paths, roots): + existing = {_canonical(path) for path in sys.path if path} + for path in paths: + candidate = _validated_site_directory(path, roots) + if candidate and candidate not in existing: + # Direct insertion intentionally does not evaluate .pth hook lines. + sys.path.append(candidate) + existing.add(candidate) + + +def _venv_configuration(): + # Preserve a venv's bin/python symlink location while locating pyvenv.cfg. + executable_dir = os.path.dirname(os.path.normcase(os.path.abspath(sys.executable))) + roots = [executable_dir] + if os.path.basename(executable_dir).lower() in ('bin', 'scripts'): + roots.insert(0, os.path.dirname(executable_dir)) + for root in roots: + config_path = os.path.join(root, 'pyvenv.cfg') + if not os.path.isfile(config_path): + continue + details = os.stat(config_path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or details.st_size > MAX_PYVENV_BYTES: + raise RuntimeError('interpreter pyvenv.cfg is not a bounded regular file') + with open(config_path, 'rb') as handle: + payload = handle.read(MAX_PYVENV_BYTES + 1) + if len(payload) > MAX_PYVENV_BYTES: + raise RuntimeError('interpreter pyvenv.cfg exceeds its byte bound') + include_system = False + for raw_line in payload.decode('utf-8', errors='strict').splitlines(): + key, separator, value = raw_line.partition('=') + if separator and key.strip().lower() == 'include-system-site-packages': + include_system = value.strip().lower() in ('1', 'true', 'yes') + return _canonical(root), include_system + return '', True + + +def _venv_site_directories(root): + if os.name == 'nt': + return [os.path.join(root, 'Lib', 'site-packages')] + version = f'python{sys.version_info.major}.{sys.version_info.minor}' + return [ + os.path.join(root, library, version, name) + for library in ('lib', 'lib64') + for name in ('site-packages', 'dist-packages') + ] + + +def _system_site_directories(): + import sysconfig + + roots = tuple(dict.fromkeys((_canonical(sys.base_prefix), _canonical(sys.base_exec_prefix)))) + paths = sysconfig.get_paths(vars={ + 'base': sys.base_prefix, + 'platbase': sys.base_exec_prefix, + }) + return [paths.get('purelib'), paths.get('platlib')], roots + + +def _user_site_directories(): + if os.name == 'nt': + try: + import ctypes + + appdata = ctypes.create_unicode_buffer(32768) + if ctypes.windll.shell32.SHGetFolderPathW(None, 0x001A, None, 0, appdata) != 0: + return [], () + root = _canonical(os.path.join(appdata.value, 'Python')) + version = f'Python{sys.version_info.major}{sys.version_info.minor}' + return [os.path.join(root, version, 'site-packages')], (root,) + except (AttributeError, OSError, ValueError): + return [], () + try: + import pwd + + home = _canonical(pwd.getpwuid(os.getuid()).pw_dir) + except (ImportError, KeyError, OSError): + return [], () + version = f'python{sys.version_info.major}.{sys.version_info.minor}' + if sys.platform == 'darwin': + root = _canonical(os.path.join(home, 'Library', 'Python', f'{sys.version_info.major}.{sys.version_info.minor}')) + return [os.path.join(root, 'lib', 'python', 'site-packages')], (root,) + root = _canonical(os.path.join(home, '.local')) + return [ + os.path.join(root, 'lib', version, 'site-packages'), + os.path.join(root, 'lib', version, 'dist-packages'), + ], (root,) + + +def _missing_dependencies(kind): + import importlib.util + + return [name for name in REQUIRED_DEPENDENCIES[kind] if importlib.util.find_spec(name) is None] + + +def _enable_dependency_paths(kind): + venv_root, include_system = _venv_configuration() + if venv_root: + _append_site_directories(_venv_site_directories(venv_root), (venv_root,)) + if include_system: + system_paths, system_roots = _system_site_directories() + _append_site_directories(system_paths, system_roots) + if _missing_dependencies(kind): + user_paths, user_roots = _user_site_directories() + _append_site_directories(user_paths, user_roots) + missing = _missing_dependencies(kind) + if missing: + raise RuntimeError('required authenticated child dependencies are unavailable: ' + ', '.join(missing)) + + +def _sha256_file(path): + digest = hashlib.sha256() + with open(path, 'rb') as handle: + while True: + block = handle.read(1024 * 1024) + if not block: + return digest.hexdigest() + digest.update(block) + + +def _read_object(path): + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or details.st_size <= 0 or details.st_size > MAX_METADATA_BYTES: + raise RuntimeError('supervisor metadata is not a bounded regular file') + with open(path, 'rb') as handle: + payload = handle.read(MAX_METADATA_BYTES + 1) + if len(payload) > MAX_METADATA_BYTES: + raise RuntimeError('supervisor metadata exceeds its byte bound') + value = json.loads(payload.decode('utf-8')) + if not isinstance(value, dict): + raise RuntimeError('supervisor metadata root is invalid') + return value + + +def _manifest_digest(manifest): + payload = json.dumps(manifest, ensure_ascii=True, sort_keys=True, separators=(',', ':')).encode('utf-8') + return hashlib.sha256(payload).hexdigest() + + +def _application_code_files(root): + names = set() + + def raise_walk_error(exc): + raise RuntimeError(f'unable to inspect the application root: {exc}') from exc + + for current, directories, files in os.walk(root, followlinks=False, onerror=raise_walk_error): + for name in directories: + candidate = os.path.join(current, name) + if _is_reparse_point(candidate): + relative = os.path.relpath(candidate, root).replace(os.sep, '/') + raise RuntimeError(f'application directory reparse point is forbidden: {relative}') + relative_current = os.path.relpath(current, root) + in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep)) + suffixes = ('.pyc',) if in_cache else APPLICATION_IMPORT_SUFFIXES + for name in files: + source_path = os.path.abspath(os.path.join(current, name)) + if _is_reparse_point(source_path): + relative = os.path.relpath(source_path, root).replace(os.sep, '/') + raise RuntimeError(f'application file reparse point is forbidden: {relative}') + if not name.lower().endswith(suffixes): + continue + path = _canonical(source_path) + try: + contained = os.path.commonpath((root, path)) == root + except ValueError: + contained = False + if not contained: + raise RuntimeError('application Python authority escapes its root') + names.add(os.path.relpath(source_path, root).replace(os.sep, '/')) + return names + + +def _reject_cached_bytecode(root): + def raise_walk_error(exc): + raise RuntimeError(f'unable to inspect the application root: {exc}') from exc + + try: + root_details = os.lstat(root) + except OSError as exc: + raise RuntimeError(f'application root is unavailable: {root}') from exc + if _is_reparse_point(root): + raise RuntimeError(f'application root reparse point is forbidden: {root}') + if not stat.S_ISDIR(root_details.st_mode): + raise RuntimeError(f'application root is not a directory: {root}') + for current, directories, files in os.walk(root, followlinks=False, onerror=raise_walk_error): + for name in directories: + candidate = os.path.join(current, name) + if _is_reparse_point(candidate): + relative = os.path.relpath(candidate, root).replace(os.sep, '/') + if name.lower() == '__pycache__': + raise RuntimeError(f'application __pycache__ link is forbidden: {relative}') + raise RuntimeError(f'application directory reparse point is forbidden: {relative}') + relative_current = os.path.relpath(current, root) + in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep)) + for name in files: + candidate = os.path.join(current, name) + relative = os.path.relpath(candidate, root).replace(os.sep, '/') + if _is_reparse_point(candidate): + raise RuntimeError(f'application file reparse point is forbidden: {relative}') + if in_cache and name.lower().endswith('.pyc'): + raise RuntimeError(f'application __pycache__ bytecode is forbidden: {relative}') + + +def _verify_manifest(metadata, inherited): + manifest = metadata.get('code_manifest') + if not isinstance(manifest, dict) or manifest.get('schema') != MANIFEST_SCHEMA: + raise RuntimeError('unsupported child code manifest') + expected_digest = str(metadata.get('code_manifest_sha256') or '') + if not hmac.compare_digest(_manifest_digest(manifest), expected_digest): + raise RuntimeError('child code manifest digest mismatch') + if not hmac.compare_digest(expected_digest, inherited['code_manifest_sha256']): + raise RuntimeError('inherited child code manifest mismatch') + root_value = manifest.get('root') or '' + files = manifest.get('files') + executables = manifest.get('executables') + assets = manifest.get('assets') + if not root_value or not isinstance(files, dict) or not isinstance(executables, dict) or not isinstance(assets, dict): + raise RuntimeError('child code manifest is incomplete') + raw_root = os.path.abspath(os.fspath(root_value)) + _reject_cached_bytecode(raw_root) + root = _canonical(raw_root) + manifested_code = set() + for name, value in files.items(): + if not isinstance(value, dict): + raise RuntimeError('child code manifest file entry is invalid') + expected_path = _canonical(os.path.join(root, *str(name).split('/'))) + path = _canonical(value.get('path') or '') + if path != expected_path or not hmac.compare_digest(_sha256_file(path), str(value.get('sha256') or '')): + raise RuntimeError(f'child code authority drifted: {name}') + try: + contained = os.path.commonpath((root, path)) == root + except ValueError: + contained = False + if contained and str(name).lower().endswith(APPLICATION_IMPORT_SUFFIXES): + manifested_code.add(str(name).replace('\\', '/')) + current_code = _application_code_files(root) + if current_code != manifested_code: + added = sorted(current_code - manifested_code) + removed = sorted(manifested_code - current_code) + detail = added[0] if added else removed[0] if removed else 'unknown' + raise RuntimeError(f'application code authority file set drifted: {detail}') + for group_name, values in (('executable', executables), ('asset', assets)): + for name, value in values.items(): + if not isinstance(value, dict): + raise RuntimeError(f'child {group_name} authority entry is invalid') + path = _canonical(value.get('path') or '') + if not os.path.isabs(path) or not hmac.compare_digest(_sha256_file(path), str(value.get('sha256') or '')): + raise RuntimeError(f'child {group_name} authority drifted: {name}') + return root + + +def _handshake(metadata): + control = metadata.get('control') or {} + request = { + 'schema': CONTROL_SCHEMA, + 'instance_id': metadata['instance_id'], + 'token': metadata['token'], + 'action': 'handshake', + } + encoded = json.dumps(request, ensure_ascii=True, separators=(',', ':')).encode('utf-8') + b'\n' + chunks = [] + total = 0 + with socket.create_connection((control.get('host'), int(control.get('port') or 0)), timeout=3) as client: + client.settimeout(3) + client.sendall(encoded) + client.shutdown(socket.SHUT_WR) + while True: + chunk = client.recv(65536) + if not chunk: + break + total += len(chunk) + if total > MAX_HANDSHAKE_BYTES: + raise RuntimeError('supervisor handshake exceeds its byte bound') + chunks.append(chunk) + response = json.loads(b''.join(chunks).decode('utf-8')) + if ( + not isinstance(response, dict) + or response.get('schema') != CONTROL_SCHEMA + or response.get('instance_id') != metadata['instance_id'] + or response.get('ok') is not True + or not isinstance(response.get('result'), dict) + ): + raise RuntimeError('authenticated supervisor handshake failed') + return response['result'] + + +def _authenticate(kind): + inherited = {name: str(os.getenv(variable) or '') for name, variable in ENV.items()} + if not all(inherited.values()): + raise RuntimeError('direct mutation is retired; use an authenticated active supervisor command') + if inherited['kind'] != kind: + raise RuntimeError('supervised child kind does not match the bootstrap entrypoint') + instance_file = _canonical(inherited['instance_file']) + metadata = _read_object(instance_file) + if metadata.get('schema') != 2 or _canonical(metadata.get('instance_file') or '') != instance_file: + raise RuntimeError('supervisor child instance metadata is invalid') + for key in ('instance_id', 'token', 'config_sha256', 'supervisor_sha256', 'code_manifest_sha256'): + if not hmac.compare_digest(str(metadata.get(key) or ''), inherited[key]): + raise RuntimeError(f'supervisor child {key} authority mismatch') + if str(metadata.get('activation_state') or '').upper() != 'ACTIVE': + raise RuntimeError('supervisor is not ACTIVE; child launch is refused') + if not hmac.compare_digest(_sha256_file(metadata['config_path']), inherited['config_sha256']): + raise RuntimeError('supervisor config authority drifted') + if not hmac.compare_digest(_sha256_file(metadata['supervisor_path']), inherited['supervisor_sha256']): + raise RuntimeError('supervisor script authority drifted') + root = _verify_manifest(metadata, inherited) + dsn = str(os.getenv('TRUF_MANAGED_POSTGRES_DSN') or '') + if kind == 'janitor': + if dsn or os.getenv('SCANNER_DB_URL') or os.getenv('DATABASE_URL'): + raise RuntimeError('janitor child must not receive database mutation capability') + else: + dsn_digest = hashlib.sha256(dsn.encode('utf-8')).hexdigest() if dsn else '' + if not dsn or not hmac.compare_digest(dsn_digest, inherited['dsn_sha256']): + raise RuntimeError('managed PostgreSQL DSN authority mismatch') + for variable in ('SCANNER_DB_URL', 'DATABASE_URL'): + if not hmac.compare_digest(str(os.getenv(variable) or ''), dsn): + raise RuntimeError(f'{variable} does not match managed PostgreSQL authority') + handshake = _handshake(metadata) + if handshake.get('activation_state') != 'ACTIVE' or handshake.get('instance_id') != metadata['instance_id']: + raise RuntimeError('supervisor handshake did not confirm ACTIVE authority') + for key in ('config_sha256', 'supervisor_sha256', 'code_manifest_sha256'): + if not hmac.compare_digest(str(handshake.get(key) or ''), inherited[key]): + raise RuntimeError(f'supervisor handshake {key} mismatch') + if not hmac.compare_digest(str(handshake.get('canonical_dsn_sha256') or ''), inherited['dsn_sha256']): + raise RuntimeError('supervisor handshake PostgreSQL authority mismatch') + return root, metadata + + +def main(): + if not sys.flags.isolated or not sys.flags.no_site or not sys.flags.dont_write_bytecode: + raise RuntimeError('supervised child bootstrap requires isolated no-site bytecode-free startup (-I -S -B)') + if len(sys.argv) < 2: + raise SystemExit('supervised child bootstrap kind is required') + kind = str(sys.argv[1]).strip().lower() + root, metadata = _authenticate(kind) + _enable_dependency_paths(kind) + arguments = list(sys.argv[2:]) + if kind != 'keycheck-provider' and arguments[:1] == ['--']: + arguments.pop(0) + if kind in ('scanner', 'discovery-producer'): + entrypoint = os.path.join(root, 'console_runner.py') + elif kind == 'docker-shadow': + entrypoint = os.path.join(root, 'docker_shadow.py') + elif kind == 'keycheck': + entrypoint = os.path.join(root, 'keycheck_runner.py') + elif kind == 'dashboard': + entrypoint = os.path.join(root, 'dashboard.py') + elif kind == 'janitor': + entrypoint = os.path.join(root, 'janitor.py') + elif kind == 'result-ingester': + entrypoint = os.path.join(root, 'result_ingester.py') + elif kind == 'jsonl-projector': + entrypoint = os.path.join(root, 'jsonl_projector.py') + elif kind == 'worker-api': + entrypoint = os.path.join(root, 'worker_api.py') + elif kind == 'keycheck-provider': + if not arguments: + raise RuntimeError('keycheck provider bootstrap entrypoint is required') + relative = arguments.pop(0).replace('\\', '/') + if not arguments or arguments.pop(0) != '--': + raise RuntimeError('keycheck provider bootstrap separator is required') + if '--' in arguments: + raise RuntimeError('duplicate keycheck provider bootstrap separator') + entrypoint = _canonical(os.path.join(root, *relative.split('/'))) + provider_root = _canonical(os.path.join(root, 'keycheckers')) + try: + allowed = os.path.commonpath((provider_root, entrypoint)) == provider_root + except ValueError: + allowed = False + if not allowed or not relative.lower().endswith('.py'): + raise RuntimeError('keycheck provider bootstrap entrypoint is outside authority') + else: + raise RuntimeError('unsupported supervised child bootstrap kind') + entrypoint = _canonical(entrypoint) + files = (metadata.get('code_manifest') or {}).get('files') or {} + if not any(_canonical(value.get('path') or '') == entrypoint for value in files.values() if isinstance(value, dict)): + raise RuntimeError('child entrypoint is absent from immutable authority') + sys.path.insert(0, root) + try: + if kind == 'dashboard': + sys.argv = ['streamlit', 'run', entrypoint, *arguments] + runpy.run_module('streamlit', run_name='__main__', alter_sys=True) + else: + sys.argv = [entrypoint, *arguments] + runpy.run_path(entrypoint, run_name='__main__') + except Exception as exc: + raise ChildRuntimeError(str(exc)) from exc + + +if __name__ == '__main__': + try: + main() + except ChildRuntimeError as exc: + raise SystemExit(f'supervised child runtime failed: {exc}') from exc + except Exception as exc: + raise SystemExit(f'supervised child bootstrap rejected launch: {exc}') from exc diff --git a/app/config.linux.yaml b/app/config.linux.yaml new file mode 100644 index 0000000..15e47de --- /dev/null +++ b/app/config.linux.yaml @@ -0,0 +1,1159 @@ +# Linux container profile; host runtime entrypoints remain deliberately disabled. +# See DOCKER_MIGRATION.md. The Windows configuration is preserved in the initial Git commit. + +global: + loop: true # true = run forever; false = run one full pass over enabled sources + cooldown: 30 # seconds to sleep after one full pass over all enabled sources + backlog_poll_sec: 0.5 # immediately refill released scan slots while durable queue work remains + root_dir: "/opt/truf" # application image root, never the original Windows checkout + project_dir: "{root_dir}/app" + runtime_dir: "/data/runtime-linux" # new Linux state; do not reuse Windows control metadata + postgres_data_dir: "/data/postgres-linux" # independently initialized, never a Windows cluster copy + postgres_bin_dir: "/usr/lib/postgresql/16/bin" + result_bundle_dir: "/data/scanner-result-bundles" + result_bundle_max_event_bytes: 67108864 # hard limit per bundle; separate from remote reservation + remote_assignment_reserve_bytes: 2097152 # bundle and projection baseline per unresolved remote assignment + remote_assignment_max_active: 50 # global unresolved remote assignments across all users + result_bundle_max_items: 10000 + result_bundle_max_total_bytes: 3221225472 + result_bundle_min_free_bytes: 21474836480 + projection_backlog_max_items: 10000 + projection_backlog_max_bytes: 2147483648 + projection_backlog_headroom_bytes: 402653184 # one worst-case aggregate scan projection beyond all 3 physical slots + keycheck_queue_max_items: 131072 + keycheck_queue_max_bytes: 134217728 + pipeline_quarantine_max_items: 10000 + pipeline_quarantine_max_bytes: 1073741824 + pipeline_metadata_retention_days: 30 + pipeline_metadata_retirement_batch: 100 + keycheck_candidates_per_event: 2000 + keycheck_candidate_bytes_per_event: 2097152 + keycheck_result_projection_reserve_bytes: 3145728 + keycheck_recheck_batch_items: 10000 + legacy_result_spool_dir: "{runtime_dir}/result_spool" + legacy_result_spool_max_event_bytes: 201326592 + legacy_result_spool_max_events: 10000 + legacy_result_spool_max_total_bytes: 3221225472 + results_dir: "{runtime_dir}/results" # findings, scan result JSONL/logs, and scanner.db + queue_dir: "{runtime_dir}/queues" # todo_*.txt and checked_*.txt live here + keycheck_dir: "{runtime_dir}/keychecks" # keychecker outputs grouped by service + postman_cache_dir: "{runtime_dir}/postman_cache" # durable cached Postman collection/environment JSON + postman_cache_max_items: 100000 # aggregate content-addressed artifacts; capacity failure stops discovery + postman_cache_max_bytes: 21474836480 # artifact, metadata, and temporary bytes under the cache root + postman_cache_min_free_bytes: 21474836480 # preserve the shared 20 GiB work-volume reserve + postman_cache_lock_timeout_sec: 300 # match the bounded discovery window under concurrent cache publishers + postman_discovery_max_artifacts_per_cycle: 1000 # shared non-package download/cache attempt cap + postman_discovery_max_artifacts_per_page: 100 # bound one API page or GHArchive hour batch + postman_discovery_max_bytes_per_cycle: 1073741824 # aggregate downloaded artifact bytes + postman_discovery_max_elapsed_sec: 300 # includes network, cache scan, lock, and publication work + postman_package_harvest_max_artifacts: 100 # matching package artifacts examined/published per target + postman_package_harvest_max_bytes: 134217728 # aggregate matching artifact bytes considered per target + postman_package_harvest_max_elapsed_sec: 30 # optional package harvesting wall-clock deadline + postman_context_max_input_bytes: 16777216 + postman_context_max_nodes: 100000 + postman_context_max_depth: 64 + postman_context_max_scalar_bytes: 16777216 + postman_context_max_items: 50000 + context_enrichment_max_source_bytes: 16777216 # aggregate optional-context source reads per target + context_enrichment_max_findings: 2000 # stop optional enrichment without dropping later findings + context_enrichment_max_postman_comparisons: 200000 # hard cap on fallback substring comparisons + context_enrichment_max_elapsed_sec: 5 # aggregate optional-context wall-clock budget per target + trufflehog_diagnostic_max_lines: 2000 # stop parsing stderr after bounded diagnostic work + trufflehog_diagnostic_max_line_chars: 8192 + trufflehog_diagnostic_max_line_bytes: 8192 + trufflehog_diagnostic_max_errors: 200 + trufflehog_diagnostic_max_warnings: 200 + trufflehog_diagnostic_max_unclassified: 20 + keycheck_input_max_line_bytes: 16777216 # canonical found_secrets producer/consumer JSONL line limit + keycheck_candidate_artifact_max_items: 2000 # cap candidates derived from one scanned artifact + keycheck_candidate_artifact_max_bytes: 2097152 + keycheck_candidate_file_max_items: 100000 # aggregate bounded loader/writer limits + keycheck_candidate_file_max_bytes: 33554432 + keycheck_candidate_line_max_bytes: 8192 + gharchive_cache_dir: "{state_dir}/gharchive_cache" # shared validated hourly .json.gz cache for both GHArchive sources + gharchive_cache_max_items: 48 # aggregate retained hourly archives + gharchive_cache_max_bytes: 8589934592 # compressed artifacts, lock metadata, and temporary bytes + gharchive_cache_min_free_bytes: 5368709120 # preserve 5 GiB free on the runtime volume + gharchive_download_max_bytes: 536870912 # per-hour compressed response cap + gharchive_decompressed_max_bytes: 8589934592 # per-hour gzip expansion cap + gharchive_max_events: 5000000 # per-hour event/line count cap + gharchive_max_line_bytes: 8388608 # reject oversized individual JSON event lines + gharchive_cache_lock_timeout_sec: 600 # bounded aggregate/per-hour cross-process lock wait + state_dir: "{runtime_dir}/state" # runner state files + log_dir: "{runtime_dir}/logs" # supervisor/source/dashboard logs + control_dir: "/run/truf/control" # ephemeral container identity, never persisted across recreation + proxy_file: "{runtime_dir}/proxy.txt" # shared proxy list for checkers/scanners + api_proxy_enabled: true # discovery/metadata use proxy_file; artifact bodies explicitly bypass proxies + api_proxy_file: "{proxy_file}" # host:port:user:pass or full proxy URL; resolved from proxy_file by default + api_proxy_timeout: 5 # proxy connect cap; caller read timeout remains unchanged (fallback: 5 s) + api_proxy_max_retries: 100 # default total attempts; source-specific attempts/deadlines take precedence + api_proxy_retry_delay: 5 # seconds between API proxy retries + download_proxy_enabled: false # reserved: keep heavy downloads/scans direct for now + download_proxy_file: "" # reserved proxy list for package/artifact downloads when enabled later + max_active_scans: 1 # conservative default within the container's shared memory budget + opportunistic_scan_slots: 0 # host-memory-based opportunistic admission is disabled in containers + opportunistic_scan_sources: [github, gitlab, huggingface] + opportunistic_scan_reserve_overhead_bytes: 1073741824 # reserve Job cap plus 1 GiB process/staging overhead + opportunistic_scan_min_available_after_reserve_bytes: 4294967296 # preserve 4 GiB physical RAM after admission + opportunistic_scan_min_commit_after_reserve_bytes: 6442450944 # preserve 6 GiB commit headroom after admission + scan_limiter_db: "{state_dir}/scan_limiter.db" # separate SQLite DB for cross-process scan slot leasing + scan_slot_wait_sec: 0.5 # sleep between slot-acquire attempts when all scan slots are busy + scan_slot_wait_log_sec: 30 # log long waits at this interval + scan_slot_stale_sec: 7200 # clean slots older than this or owned by dead scanner PIDs + target_retry_max_attempts: 3 # total target attempts before a terminal failed queue state + target_retry_base_delay_sec: 3600 # transient target retry delay; doubles after each failed attempt + target_retry_max_delay_sec: 86400 # cap exponential target retry delay at 24 hours + target_timeout_retry_delay_sec: 21600 # timed-out targets use a non-terminal slow retry after 6 hours + target_claim_batch_size: 1 # fallback only; PostgreSQL slot-first dispatch claims exactly acquired capacity + admission_resolution_attempts: 300 # exact-token probes after an ambiguous PostgreSQL admission response + admission_resolution_seconds: 300 # cover 45s loss grace, 60s stable-ready gate, and reconnect margin + admission_resolution_retry_delay_sec: 1 # bounded pause between fast failed recovery probes + sync_file_queues: false # legacy todo/checked import is complete; PostgreSQL is authoritative + dockerhub_tag_cache_path: "{state_dir}/dockerhub_tag_cache.sqlite" # cache Docker Hub tag resolutions/rate limits + dockerhub_tag_cache_ttl_sec: 21600 # successful tag resolutions are reused for 6 hours + dockerhub_tag_negative_cache_ttl_sec: 3600 # empty/not-found tag lookups are retried after 1 hour + dockerhub_tag_rate_limit_cache_ttl_sec: 1800 # global Docker tag API cooldown fallback + dockerhub_tag_cache_max_rows: 50000 + dockerhub_tag_cache_max_age_sec: 604800 + dockerhub_tag_cache_max_bytes: 268435456 + dockerhub_tag_cache_min_free_bytes: 536870912 + database_path: "{results_dir}/scanner_active.db" # SQLite fallback observability DB when SCANNER_DB_URL is empty + database_url: "" # Postgres DSN comes from SCANNER_DB_URL; keep real credentials out of config + dashboard_db_path: "{database_path}" # SQLite dashboard fallback + dashboard_db_url: "" # optional dashboard Postgres DSN override; defaults to SCANNER_DB_URL when empty + dashboard_immutable_db: false # active DB is read-only via SQLite mode=ro, but not immutable because WAL changes + jsonl_rotation_enabled: true # rotate large runtime JSONL files instead of growing multi-GB active files + found_secrets_max_mb: 128 # rotate found_secrets.jsonl after this active-file size + scan_results_max_mb: 256 # rotate scan_results.jsonl after this active-file size + scan_errors_max_mb: 32 # bound each scan_errors.log segment + scan_errors_keep: 5 # retain at most this many rotated scan error segments + jsonl_lock_stale_sec: 300 # stale lock cleanup for cross-process JSONL rotation + jsonl_max_segments: 16 # hard cap; publication pauses until registered consumers catch up + jsonl_ledger_max_rows: 1000000 # durable O(1) publication identity bound + jsonl_ledger_max_bytes: 536870912 + jsonl_legacy_index_max_bytes: 16777216 # larger existing files require offline ledger reconciliation + jsonl_tail_scan_max_bytes: 8388608 + jsonl_torn_quarantine_max_bytes: 65536 + detectors: "" # empty = use all TruffleHog detectors; set IDs to limit intentionally + exclude_detectors: "github.v1,gitlab.v1,GitHubOauth2" # drop noisy legacy GitHub/GitLab detectors; keep modern prefixes + no_verification: true # pass --no-verification to TruffleHog; local checkers classify live/dead later + strict_git_provider_token_filter: true # drop unverified GitHub/GitLab detections that do not match known token prefixes + drop_detectors: "Privacy,URI,JDBC,Postgres,MongoDB,SQLServer,Box,ZohoCRM,Accuweather,Roaring,Flatio,LinkPreview,RailwayApp" # do not persist obvious non-keycheckable/generic noise detectors + versions_per_package: 3 # npm/PyPI: scan up to N recent versions per matching package + work_dir: "/data/scanner-work" # isolated scratch root for clone/download/extract folders + trufflehog_stdout_max_mb: 32 # hard file-backed streamed stdout byte bound per scan + trufflehog_stderr_max_mb: 8 # hard file-backed streamed diagnostic byte bound per scan + trufflehog_config: "{project_dir}/trufflehog-custom-detectors.yaml" # custom detectors loaded by TruffleHog --config + trufflehog_job_memory_limit_bytes: 4294967296 # aggregate Windows Job limit for each TruffleHog tree (4096 MiB) + trufflehog_windows_job_cpu_weight: 2 # normal priority with low relative Job weight; avoids broken BELOW_NORMAL startup + trufflehog_windows_memory_priority: 4 # reclaim scan pages before normal-priority desktop working sets + min_free_gb: 20 # minimum free space required in work_dir before starting new scans + state_file: "{state_dir}/runner_state.json" # stores current query index for each source + secrets_file: "/data/config/secrets.yaml" # runtime-owned private credentials, never baked into the image + stop_on_seen_pages: false # default false; enable per source where API pagination is sequential + seen_page_threshold: 2 # stop after this many consecutive all-known pages + min_pages_before_stop: 1 # always fetch at least this many pages before early stop + +supervisor: + enabled_sources: [gitlab, dockerhub, huggingface] # exact distributed discovery-producer profile + interactive: false # canonical foreground container supervisor, no terminal dependency + autostart: true # start only the explicit core allowlist above + poll_sec: 1.0 # how often supervisor checks child process/log state + heartbeat_sec: 60 # rewrite background status snapshot at least this often; 0 disables heartbeat + authority_check_interval_sec: 5 # detect code/config authority drift within this bound + pipeline_status_refresh_sec: 2 # cache authenticated pipeline status queries between loop ticks + postgres_health_interval_sec: 15 # bounded authenticated controller probe interval + postgres_ready_loss_grace_sec: 45 # tolerate transient loss for the same live authenticated postmaster + postgres_stable_ready_sec: 60 # dependency gate opens only after readiness remains stable this long + postgres_connect_timeout_sec: 5 # bounded controller authentication connect timeout + postgres_query_timeout_ms: 5000 # bounded controller identity/readiness query timeout + postgres_stop_timeout_sec: 60 # pg_ctl's bounded identity-verified coordinated stop timeout + postgres_start_settle_timeout_sec: 30 # bound late postmaster publication checks after pg_ctl start -W + postgres_shutdown_timeout_sec: 120 # total supervisor wait for the controller during coordinated shutdown + postgres_log_max_mb: 64 # collector/startup log segment bound + postgres_log_keep: 24 # maximum retained collector and startup log segments + refresh_sec: 5 # non-interactive stdout/status-loop interval + log_dir: "{log_dir}" # one appended child-process log per source + log_max_mb: 64 # live source output rotates at this active-segment bound + log_keep: 5 # keep this many rotated log segments per log file + control_dir: "{control_dir}" # hardened directory; links/junctions are rejected + instance_file: "{control_dir}/supervisor.instance.json" # private authenticated process/control identity + lock_file: "{control_dir}/supervisor.lock" # secondary per-instance lock; cluster authority is data-dir-derived and non-configurable + supervisor_log: "{log_dir}/supervisor.log" # stdout/stderr for background supervisor + status_file: "{log_dir}/supervisor.status.txt" # latest background supervisor status table + control_host: "127.0.0.1" # local only; do not expose externally + control_port: 8765 + background_start_timeout_sec: 20 # wait for matching private metadata and authenticated handshake + background_shutdown_timeout_sec: 180 # wait on the retained verified supervisor process handle + attach_poll_sec: 0.2 # --attach watch-mode poll interval + dashboard_log: "{log_dir}/dashboard.log" # stdout/stderr for supervisor-launched dashboard + state_dir: "{state_dir}" # per-source runner_state_*.json files to avoid parallel write races + per_source_state: true # true = supervisor sets RUNNER_STATE_FILE per child process + result_ingester: + enabled: true + poll_sec: 0.2 + lease_seconds: 300 + jsonl_projector: + enabled: true + poll_sec: 0.2 + lease_seconds: 300 + worker_api: + enabled: false # fail closed; configure profiles and private ingress before enabling + address: "127.0.0.1" # raw API must remain loopback/private and unexposed + port: 8766 + sources: [] # optional narrowing of exact protocol-2 package capability triples + auth_entries: {} # GitLab issuance plus optional legacy GitHub reconciliation auth entries + compatibility_profiles: {} # trusted package manifests; never inferred from clients + assignment_ttl_seconds: 86400 # immutable server-time deadline, default 24 hours + assignment_ttl_seconds_by_source: {} + max_bundle_bytes: 67108864 # hard-capped again by global result_bundle_max_event_bytes + reaper_interval_seconds: 60 + reaper_batch_size: 1000 + limit_concurrency: 64 + body_idle_timeout_seconds: 30 + json_body_timeout_seconds: 60 + bundle_body_timeout_seconds: 1800 + admin: + enabled: false # fail closed; available only behind authenticated Caddy + origin: "" # exact public HTTPS origin when enabled + edge_marker: "" # independent 256-bit secret shared only with Caddy + max_body_bytes: 8192 + snapshot_limit: 200 + requeue_limit: 100 + managed_file_roots: + runtime-keychecks: + path: "/data/runtime-linux/keychecks" + permissions: + list: true + read: true + create_replace: false + delete: false + limits: + max_relative_path_bytes: 1024 + max_component_bytes: 255 + max_path_depth: 16 + max_listing_entries: 500 + max_listing_bytes: 262144 + max_file_bytes: 67108864 + runtime-logs: + path: "/data/runtime-linux/logs" + permissions: + list: true + read: true + create_replace: false + delete: false + limits: + max_relative_path_bytes: 1024 + max_component_bytes: 255 + max_path_depth: 16 + max_listing_entries: 500 + max_listing_bytes: 262144 + max_file_bytes: 67108864 + runtime-results: + path: "/data/runtime-linux/results" + permissions: + list: true + read: true + create_replace: false + delete: false + limits: + max_relative_path_bytes: 1024 + max_component_bytes: 255 + max_path_depth: 16 + max_listing_entries: 500 + max_listing_bytes: 262144 + max_file_bytes: 268435456 + docker_shadow: + enabled: true # manual-only operator command; never autostarted + cohort_size: 50 + lease_seconds: 3600 + janitor: + enabled: true + interval_sec: 60 + minimum_age_sec: 7200 + max_candidates: 50 + max_entries: 10000 + max_bytes: 1073741824 + max_seconds: 30 + max_depth: 64 + interval: 300 # default delay before repeating a --once child source + restart_delay: 30 # initial restart delay after failures or unexpected exits + max_restart_delay: 600 # cap for exponential restart backoff + restart_reset_after: 300 # clear failure streak/old exit after this many stable seconds + dashboard: + enabled: false # true = supervisor also starts dashboard.py + address: "127.0.0.1" # local-only dashboard bind address + port: 5000 + startup_grace_sec: 30 # allow Streamlit to initialize before a failed health probe triggers restart + health_interval_sec: 2 # bounded asynchronous /_stcore/health probe interval + health_timeout_sec: 1 + restart_base_sec: 2 # exponential dashboard-only restart backoff + restart_max_sec: 60 + stable_health_sec: 60 # reset dashboard restart streak only after stable health + defaults: + enabled: true + once: false # false = child console_runner loops internally; true = one source cycle per child run + repeat: true # only relevant when once=true; repeat one-shot cycles after interval + restart: true # restart crashed/exited child processes + extra_args: [] # extra args passed to console_runner.py in config mode + sources: + github: + once: false + github_archive: + enabled: true # controlled broad GitHub discovery via GHArchive; start manually from supervisor + use_system_proxy: true + once: false + repeat: true + restart: true + interval: 3600 + github_archive_files: + enabled: true # controlled GHArchive changed-file fetch; start manually from supervisor + use_system_proxy: true + once: false + repeat: true + restart: true + interval: 1800 + github_gists: + enabled: true # controlled public gist discovery; start manually from supervisor + once: false + repeat: true + restart: true + interval: 1800 + gitlab: + once: false + github_actions: + enabled: false # paused after fresh and retained cohorts produced no strict-usable yield + once: false + repeat: true + restart: true + interval: 3600 + gitlab_ci: + enabled: true # show in supervisor; main source remains disabled for normal all-source runner + once: false + repeat: true + restart: true + interval: 3600 + dockerhub: + once: false + npm: + once: false + pypi: + once: false + package_git: + enabled: false + once: false + huggingface: + once: false + postman: + enabled: true # allows `supervisor --sources postman`; main sources.postman.enabled controls default all-source inclusion + once: false + env: + PYTHONIOENCODING: utf-8 # avoid Windows console codec crashes on Unicode repository paths + +keychecks: + enabled: true # supervisor manages this as pseudo-source "keychecks" + autostart: true # start keychecks when supervisor starts, even if scanner sources wait for manual start + input_mode: postgres # PostgreSQL candidate leases are authoritative; JSONL requires explicit compatibility mode + services: all # all or comma/list: openai,anthropic,qwen,kimi,github,... + interval: 3600 # repeat keycheck_runner every N seconds when in repeat/hourly mode + repeat: true + restart: false # do not auto-restart failed checker batch; wait for next interval/manual restart + max_keys: 0 # 0 = no per-run limit; set N for throttled hourly batches + scheduler_batch_keys: 1000 # per-provider process slice; max_keys=0 keeps rotating until empty/deadline + scheduler_workers: 1 # serialize provider handshakes; avoids transient control-plane startup failures + scheduler_deadline_sec: 1800 # aggregate work-conserving provider deadline + retry_network: true # retry transient network failures each scheduled run + retry_limited: false # set true to retry rate-limited/no-quota statuses hourly + retry_unknown: false + retry_restricted: false + retry_no_balance: false # retry no_balance/no_quota statuses when explicitly enabled + retry_valid: false # valid provider keys are re-probed only when explicitly requested + recheck_all: false # true forces all known keys to be checked again + env: + KEYCHECK_INPUT_TAIL_MB: "0" # first keycheck reads from offset 0; high-watermark state handles later appends + KEYCHECK_INPUT_MAX_LINE_BYTES: "16777216" # must match global.keycheck_input_max_line_bytes + KEYCHECK_CANDIDATE_MAX_UNCONSUMED_ATTEMPTS: "3" + KEYCHECK_DB_INGEST_TAIL_MB: "64" # optional tail size; initial DB ingest uses full bounded offsets unless KEYCHECK_DB_INGEST_TAIL_INITIAL=1 + KEYCHECK_RESULTS_MAX_MB: "32" # rotate per-service *Results.jsonl files into manifest segments + KEYCHECK_PROVIDER_RESOLUTION_ORDER: "deepseek,zai,qwen,kimi" + KEYCHECK_EVENT_MAP_BACKFILL_ROWS: "0" # one-time legacy backfill is complete; live ingest maintains this map atomically + KEYCHECK_UID_MAP_BACKFILL_ROWS: "0" # scanner writes finding_uid_map for all new findings + db_ingest: + enabled: false # explicit JSONL compatibility import only + repair_links: + enabled: false # new DB candidates carry exact finding attribution + service_args: # provider-specific probe flags passed by keycheck_runner + gemini: + - --probe-generation # call generateContent; RATE_LIMITED valid keys go to geminiAliveRateLimited.txt + aws: + - --probe-bedrock # after STS, probe Bedrock access using safe validation-style calls + - --bedrock-max-attempts + - "12" + azure: + - --probe-openai-route # after deployments list, probe Azure OpenAI chat route without generation + - --probe-foundry-route # probe Azure AI Foundry/MaaS route for configured models + - --foundry-models + - claude-opus-4-6,claude-fable-5 + - --timeout + - "8" + gcp: + - --probe-vertex # after OAuth, probe Vertex AI Gemini countTokens access + - --vertex-timeout + - "6" + - --vertex-max-attempts + - "6" + - --vertex-models + - gemini-3.6-flash,gemini-3.1-pro-preview + - --vertex-locations + - global,us,eu + - --vertex-anthropic-models + - claude-opus-5,claude-opus-4-7,claude-opus-4-6,claude-fable-5 + - --vertex-anthropic-locations + - global,us,eu,us-east5,europe-west1 + - --vertex-anthropic-max-attempts + - "20" + summary_tsv: "{keycheck_dir}/summary.tsv" + summary_json: "{keycheck_dir}/summary.json" + alive_summary_tsv: "{keycheck_dir}/alive_summary.tsv" + +query_policy: + rejected: # reviewed source-specific zero-alive evidence; queue rows remain auditable and reversible + - {source: dockerhub, query: OR, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1631, findings: 29137, unique_credentials: 36, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 36} + - {source: dockerhub, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1009, findings: 69486, unique_credentials: 30, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 229} + - {source: github, query: coding, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1696, findings: 251, unique_credentials: 11, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 0} + - {source: github, query: memory, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3190, findings: 254, unique_credentials: 11, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 8} + - {source: npm, query: OR, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1985, findings: 152, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 598} + - {source: npm, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 9446, findings: 5833, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3957} + - {source: npm, query: agents, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 8936, findings: 4573, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3346} + - {source: npm, query: ai, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 7676, findings: 49499, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3663} + - {source: npm, query: assistant, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3888, findings: 14831, unique_credentials: 2, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 899} + - {source: npm, query: benchmark, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2371, findings: 298, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1292} + - {source: npm, query: bot, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1784, findings: 2081, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 951} + - {source: npm, query: chat, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2048, findings: 381, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2087} + - {source: npm, query: chats, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1297, findings: 107, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 719} + - {source: npm, query: completion, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2529, findings: 1351, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1263} + - {source: npm, query: completions, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3058, findings: 1169, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1466} + - {source: npm, query: conversation, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1454, findings: 475, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 954} + - {source: npm, query: gemini, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 4922, findings: 3830, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 891} + - {source: npm, query: groq, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2050, findings: 333, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 484} + - {source: npm, query: langchain, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1264, findings: 169, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 908} + - {source: npm, query: llm, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3682, findings: 17976, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2333} + - {source: npm, query: mcp, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 8704, findings: 5572, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3524} + - {source: npm, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 6262, findings: 3288, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1359} + - {source: npm, query: openrouter, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1590, findings: 355, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1014} + - {source: npm, query: prompt, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1899, findings: 91, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1161} + - {source: npm, query: rag, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2242, findings: 1912, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 544} + - {source: npm, query: semantic, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1911, findings: 164, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1168} + - {source: npm, query: studio, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 4158, findings: 484, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1349} + - {source: npm, query: xai, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1390, findings: 2395, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 513} + - {source: package_git, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1851, findings: 5395, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 8791} + - {source: package_git, query: bot, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1610, findings: 7260, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 4486} + - {source: package_git, query: completion, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3149, findings: 10028, unique_credentials: 9, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1507} + - {source: package_git, query: llm, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1285, findings: 2116, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 5572} + - {source: package_git, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2053, findings: 14375, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 9968} + - {source: package_git, query: semantic, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3895, findings: 24064, unique_credentials: 22, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2414} + - {source: package_git, query: studio, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1028, findings: 2666, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 5709} + - {source: postman, query: XAI_API_KEY, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2935, findings: 83, unique_credentials: 1, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 306} + - {source: pypi, query: langchain, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1095, findings: 47, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 16} + - {source: pypi, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1253, findings: 731, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 66} + +sources: + github: + enabled: false # direct GitHub discovery is paused in favor of the Actions experiment + auth_pool: github_main # auth_pools. in secrets.yaml; leave empty to use env/legacy token + auth_rotation: per_cycle # per_cycle = use next token each source cycle; none = first available + retry_with_next_auth_on_rate_limit: true # switch token and retry when GitHub API rate-limits + rate_limit_cooldown: 3600 # seconds to cool down a rate-limited token if API gives no reset time + mode: recent # recent = updated recently; search = paginated search; custom = target_file URLs + queries: # one query is used per source cycle; state rotates through this list + - llm + - ai + - agent + - agents + - assistant + - bot + - chatbot + - rag + - semantic + - prompt + - completion + - completions + - mcp + - claude + - anthropic + - opus + - sonnet + - haiku + - vertex + - aiplatform + - bedrock + - foundry + - azure-openai + - openrouter + - langchain + - langgraph + - litellm + - crewai + - autogen + - semantic-kernel + - mistral + - groq + - cohere + - xai + - together + - perplexity + - gemini + - qwen + - kimi + - dashscope + - aistudio + - studio + - OR + - open + - chat + - chats + - conversation + - conversational + - dialogue + - OPENAI_API_KEY # bounded high-signal README integration query + - api.openai.com # bounded high-signal README endpoint query + - openai-agents # bounded OpenAI agent SDK/ecosystem query + - copilot + - ai-assistant + - ai-agent + - multi-agent + - workflow + - workflows + - llmops + - benchmark + - tokens + - function-calling + query_overrides: + OPENAI_API_KEY: + pages: 1 + per_page: 25 + max_targets: 5 + api.openai.com: + pages: 1 + per_page: 25 + max_targets: 5 + openai-agents: + pages: 1 + per_page: 50 + max_targets: 5 + pages: 5 # pages to fetch in search/recent mode + per_page: 100 # targets per API page, max is usually 100 + workers: 1 # one shared non-Docker scan slot for this source + timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes + recent_hours: 24 # recent mode: only repos updated within this many hours are discovered + max_repo_age_days: 90 # skip repos older than this by repo_age_field before queueing + repo_age_field: updated_at # metadata field for repo age: created_at, updated_at, pushed_at + max_commit_age_days: 90 # scan only recent git history by passing --since-commit boundary + commit_lookup_pages: 3 # API pages of commits to inspect when finding --since-commit boundary + skip_if_commit_lookup_fails: true # true = do not scan if recent commit boundary cannot be found + max_depth: 100 # TruffleHog --max-depth; extra cap on commit count scanned + exact_git_planning_enabled: true # resolve and bind the exact default/explicit ref head after claim + git_baseline_depth: 100 # bounded first/reset scan depth; deltas use only the covered base + git_ref_resolution_timeout_sec: 10 + git_ref_resolution_attempts: 2 + git_ref_resolution_max_bytes: 1048576 + scan_full_history: false # true = ignore max_depth; still uses max_commit_age_days when configured + sort_by: updated # GitHub search sort: updated, stars, forks, created + sort_order: desc # desc = newest/highest first; asc = oldest/lowest first + created_filter: any # optional GitHub created filter: any, today, week, month, year + stop_on_seen_pages: true # stop fetching pages after consecutive all-known pages + seen_page_threshold: 2 + min_pages_before_stop: 1 + updated_target_rescan_enabled: true # preserve pushed_at and rescan completed repos after newer pushes + updated_target_rescan_max_per_cycle: 1 # bounded rollout: at most one changed repo per discovery cycle + updated_target_rescan_cooldown_hours: 24 + + github_archive: + enabled: false # broad discovery source; supervisor exposes it for manual/controlled runs + use_system_proxy: true # direct GHArchive route is unstable; use inherited HTTP(S)_PROXY only for hourly dumps + auth_pool: github_main + auth_rotation: per_cycle + mode: archive # GHArchive hourly events -> GitHub repos -> recent git scan + queries: + - gharchive + archive_hours_back: 6 # read recent completed GHArchive hours + fetch_timeout: 120 # read timeout while downloading large hourly gzip archives + archive_max_repos_per_cycle: 500 # hard cap after event/repo dedupe/scoring + archive_rescan_cooldown_hours: 48 # allow rescanning active repos after cooldown; not forever-checked + archive_event_types: + - PushEvent + - CreateEvent + - PublicEvent + workers: 4 + timeout: 1800 + max_depth: 75 + scan_full_history: false + max_commit_age_days: 0 # avoid GitHub commit-boundary API lookups for broad source + commit_lookup_pages: 0 + skip_if_commit_lookup_fails: false + stop_on_seen_pages: false + + github_archive_files: + enabled: false # fetch suspicious changed files from GHArchive PushEvents + use_system_proxy: true + auth_pool: github_main + auth_rotation: per_cycle + mode: archive + queries: + - gharchive-files + archive_hours_back: 6 + archive_max_files_per_cycle: 100 + archive_max_commit_lookups: 200 + archive_event_types: + - PushEvent + workers: 1 + timeout: 300 + fetch_timeout: 120 + max_artifact_size_mb: 2 + stop_on_seen_pages: false + + github_gists: + enabled: false # public gist source; supervisor exposes it for manual/controlled runs + auth_pool: github_main + auth_rotation: per_cycle + mode: search + queries: + - gists + pages: 3 + per_page: 100 + workers: 3 + timeout: 300 + fetch_timeout: 20 + max_artifact_size_mb: 2 + stop_on_seen_pages: true + seen_page_threshold: 2 + min_pages_before_stop: 1 + + gitlab: + enabled: true # include the GitLab discovery producer in distributed runs + target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission + auth_pool: gitlab_main # auth_pools. in secrets.yaml; leave empty to use env/legacy token + auth_rotation: per_cycle # per_cycle = use next token each source cycle; none = first available + retry_with_next_auth_on_rate_limit: true # switch token and retry when GitLab API returns 429 + rate_limit_cooldown: 3600 # seconds to cool down a rate-limited token if API gives no reset time + discovery_request_attempts: 3 # bounded retries for idempotent project discovery transport failures + discovery_retry_delay: 5 # seconds between GitLab discovery transport attempts + external_trufflehog_lifecycle: true # bypass embedded overseer and require explicit completion for GitLab scans + mode: recent # recent = last_activity_after; search = paginated project search; custom = target_file URLs + queries: # one query is used per source cycle; state rotates through this list + - llm + - ai + - agent + - agents + - assistant + - bot + - chatbot + - rag + - openai-api # bounded metadata-friendly OpenAI API query + - openai-agents # bounded OpenAI agent SDK/ecosystem query + - librechat # bounded deployable chat application query + - semantic + - prompt + - completion + - completions + - mcp + - openrouter + - groq + - xai + - langchain + - litellm + - gemini + - qwen + - kimi + - dashscope + - aistudio + - studio + - OR + - open + - chat + - chats + - conversation + - conversational + - dialogue + - copilot + - coding + - ai-assistant + - ai-agent + - multi-agent + - workflow + - workflows + - llmops + - benchmark + - tokens + - memory + - function-calling + - hermes + - harnes + - flow + - helpdesk + - paperless + query_overrides: + openai-api: + pages: 1 + per_page: 50 + max_targets: 5 + openai-agents: + pages: 1 + per_page: 50 + max_targets: 5 + librechat: + pages: 1 + per_page: 25 + max_targets: 5 + hermes: + pages: 1 + per_page: 50 + max_targets: 5 + harnes: + pages: 1 + per_page: 50 + max_targets: 5 + flow: + pages: 1 + per_page: 50 + max_targets: 5 + helpdesk: + pages: 1 + per_page: 50 + max_targets: 5 + paperless: + pages: 1 + per_page: 50 + max_targets: 5 + pages: 10 # pages to fetch in search/recent mode + per_page: 100 # targets per API page, max is usually 100 + workers: 1 # one shared non-Docker scan slot for this source + timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes + recent_hours: 96 # recent mode: only projects active within this many hours are discovered + max_repo_age_days: 90 # skip projects older than this by repo_age_field before queueing + repo_age_field: last_activity_at # metadata field: created_at, updated_at, last_activity_at + max_commit_age_days: 90 # scan only recent git history by passing --since-commit boundary + commit_lookup_pages: 3 # API pages of commits to inspect when finding --since-commit boundary + skip_if_commit_lookup_fails: true # true = do not scan if recent commit boundary cannot be found + max_depth: 100 # TruffleHog --max-depth; extra cap on commit count scanned + exact_git_planning_enabled: true # resolve and bind the exact default/explicit ref head after claim + git_baseline_depth: 100 # bounded first/reset scan depth; deltas use only the covered base + git_ref_resolution_timeout_sec: 10 + git_ref_resolution_attempts: 2 + git_ref_resolution_max_bytes: 1048576 + scan_full_history: false # true = ignore max_depth; still uses max_commit_age_days when configured + sort_by: last_activity_at # GitLab order_by field: last_activity_at, created_at, updated_at, name, stars + sort_order: desc # desc = newest/highest first; asc = oldest/lowest first + visibility: public # public, internal, private; private/internal require token permissions + stop_on_seen_pages: true # stop fetching pages after consecutive all-known pages + seen_page_threshold: 2 + min_pages_before_stop: 1 + updated_target_rescan_enabled: true # preserve last_activity_at and rescan completed projects after new activity + updated_target_rescan_max_per_cycle: 1 + updated_target_rescan_cooldown_hours: 24 + + github_actions: + enabled: false # paused; queue and historical evidence remain intact + auth_pool: github_main + auth_rotation: per_cycle + mode: recent + queries: + - logs + ci_seed_sources: github,github_archive,package_git + ci_max_repos_per_cycle: 25 + ci_seed_scan_limit: 15000 + ci_seed_query_batch_size: 250 # bound each PostgreSQL history page on HDD + ci_soft_cooldown_days: 7 + ci_runs_per_repo: 5 + ci_lookback_days: 30 + ci_max_log_archive_mb: 150 + ci_max_log_file_mb: 50 + ci_scan_artifacts: true + ci_max_artifacts_per_run: 3 + ci_max_artifact_archive_mb: 150 + ci_max_artifact_file_mb: 50 + ci_max_artifact_files: 1000 + ci_failed_first: true + refresh_registry: true # keep discovering recent repositories while historical targets remain queued + target_claim_order: balanced # split work between fresh discoveries and the retained historical backlog + workers: 1 + timeout: 180 + + gitlab_ci: + enabled: false # controlled experiment: run manually from supervisor + auth_pool: gitlab_main + auth_rotation: per_cycle + mode: recent + queries: + - logs + ci_seed_sources: gitlab,package_git + ci_max_repos_per_cycle: 25 + ci_seed_scan_limit: 15000 + ci_seed_query_batch_size: 250 # bound each PostgreSQL history page on HDD + ci_soft_cooldown_days: 7 + ci_pipelines_per_project: 5 + ci_jobs_per_pipeline: 20 + ci_lookback_days: 30 + ci_max_trace_mb: 100 + ci_scan_artifacts: true + ci_max_artifacts_per_pipeline: 5 + ci_max_artifact_archive_mb: 150 + ci_max_artifact_file_mb: 50 + ci_max_artifact_files: 1000 + workers: 3 + timeout: 180 + + huggingface: + enabled: true # discover newest HuggingFace Spaces for remote workers + target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission + auth_pool: huggingface_main # optional auth_pools. in secrets.yaml or use HF_TOKEN/HUGGINGFACE_TOKEN + auth_rotation: per_cycle + mode: recent # recent/search both fetch newest spaces; custom = target_file space IDs + queries: + - spaces # placeholder query for state rotation; HuggingFace API fetch ignores it for now + pages: 100 # bounded cursor pages from the newest-lastModified Spaces API + per_page: 100 # newest-modified API page size is fixed at 100 by the runner + workers: 1 + timeout: 1800 + fetch_timeout: 15 + discovery_request_attempts: 3 # one transient page timeout must not restart the whole source + discovery_retry_delay: 5 + stop_on_seen_pages: true + seen_page_threshold: 2 + min_pages_before_stop: 1 + updated_target_rescan_enabled: true # preserve lastModified and revisit changed completed Spaces + updated_target_rescan_max_per_cycle: 1 + updated_target_rescan_cooldown_hours: 24 + + dockerhub: + enabled: true # include the DockerHub discovery producer in distributed runs + target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission + require_digest: true # only immutable repo@sha256:... image targets may reach TruffleHog + auth_pool: dockerhub_main # auth_pools. in secrets.yaml; all available accounts are rotated per image scan + auth_rotation: per_scan # DockerTokenManager rotates Docker accounts for each docker scan + retry_with_next_auth_on_rate_limit: true # rotate Hub/Registry requests to another account after 429/invalid auth + rate_limit_cooldown: 1800 # per-account cooldown; shared cooldown starts only after pool exhaustion + mode: search # search = Docker Hub search; recent = client-side recent tag filtering; custom = target_file images + refresh_registry: true # discover fresh images even while deferred targets remain queued + queries: # one query is used per source cycle; Docker Hub empty query returns no results + - llm + - ai + - agents + - assistant + - bot + - chatbot + - rag + - semantic + - prompt + - completion + - completions + - mcp + - openrouter + - groq + - xai + - langchain + - litellm + - gemini + - qwen + - kimi + - dashscope + - aistudio + - studio + - librechat # bounded deployable chat application query + - lobechat # bounded deployable chat application query + - openai-proxy # bounded OpenAI-compatible proxy query + - open + - chat + - chats + - conversation + - conversational + - dialogue + - copilot + - coding + - ai-assistant + - ai-agent + - multi-agent + - workflow + - workflows + - llmops + - benchmark + - tokens + - memory + - function-calling + - hermes + - harnes + - flow + - helpdesk + - paperless + - open-webui + - ragflow + - dify + - flowise + - crewai + - n8n + - langflow + - autogen + - browser-use + - openhands + - anythingllm + - agent-zero + query_overrides: + librechat: + max_targets: 10 + lobechat: + max_targets: 10 + openai-proxy: + max_targets: 10 + hermes: + max_targets: 10 + harnes: + max_targets: 10 + flow: + max_targets: 10 + helpdesk: + max_targets: 10 + paperless: + max_targets: 10 + pages: 30 # maximum Docker Hub search pages for every configured query + per_page: 100 # maximum Docker Hub results per search page + workers: 2 # allow two Docker image scans within the global three-slot limit + trufflehog_job_memory_limit_bytes: 6442450944 # Docker images get 6 GiB; other sources retain the 4 GiB global cap + timeout: 600 # bounded full-image scan window after disabling TruffleHog's embedded overseer + trufflehog_concurrency: 4 # bound per-image layer/chunk fan-out; two source workers remain available + recent_days: 7 # recent mode: keep images/tags updated within this many days + fetch_workers: 5 # parallel Docker Hub search page fetches before scanning + fetch_timeout: 15 # seconds per Docker Hub API request + tag_fetch_workers: 4 # bound the in-flight burst before a shared 429 cooldown becomes visible + tag_retry_count: 0 # do not retry individual transport failures during high-volume tag resolution + tag_retry_delay: 10 # base delay if bounded non-429 transport retries are enabled later + tag_resolve_limit: 100 # max old bare todo repos to resolve to tags per cycle; 0 = all + docker_platform_filter_enabled: true # skip tags only when complete metadata proves linux/amd64 is absent + docker_platform_os: linux + docker_platform_arch: amd64 + docker_platform_candidate_tags: 20 # inspect extra recent tags so an ARM-only latest tag does not hide an amd64 tag + docker_images_per_repository: 3 # ordinary FIFO resolver stays at the reviewed production depth + docker_content_scan_mode: canary # only prior durable full-image timeouts are eligible for layer fallback + docker_layer_canary_basis_points: 10000 # all prior durable full-image timeouts use bounded layer fallback + docker_adaptive_canary_basis_points: 0 # disabled until an exact aggregate shadow report passes every gate + docker_adaptive_gate_max_age_sec: 604800 # matching shadow evidence expires after seven days + docker_layer_config_max_bytes: 1048576 # image config is selected independently from layer budget + docker_layer_max_bytes: 268435456 # max compressed bytes for one selected application layer + docker_layer_image_max_bytes: 1073741824 # cumulative compressed layer budget per immutable image + docker_layer_max_layers: 8 # select application layers from top to base within the byte budget + docker_layer_archive_max_size_bytes: 268435456 # TruffleHog per-member extraction bound + docker_layer_archive_max_depth: 4 # required for an OCI wrapper plus compressed layer archive + docker_layer_archive_timeout_sec: 30 + docker_layer_blob_timeout_sec: 600 # shared transfer+filesystem scan deadline per durable checkpoint + docker_layer_filesystem_concurrency: 2 + docker_layer_blob_max_attempts: 3 # blob budget is authoritative; parent checkpoint attempts reset + docker_layer_blob_lease_sec: 1800 # exceeds the bounded 600-second execution plus handoff margin + docker_layer_min_free_bytes: 21474836480 # retain 20 GiB on the private work volume + docker_layer_checkpoint_delay_sec: 60 # resume the next selected blob without a full-image restart + docker_adaptive_checkpoint_max_blobs: 4 # scheduling-only bound for a future gated adaptive checkpoint + docker_adaptive_checkpoint_max_bytes: 536870912 # aggregate compressed bytes per adaptive checkpoint + docker_repository_refresh_interval_sec: 86400 # revisit each resolved repository daily for new immutable digests + docker_repository_refresh_max_per_cycle: 0 # disable periodic refresh of completed repository anchors + sort_by: updated_at # Docker Hub search sort field + sort_order: desc # desc = newest/highest first; asc = oldest/lowest first + + npm: + enabled: true # true = include npm registry in config-mode runs + mode: search # npm currently supports search mode + queries: # one query is used per source cycle; state rotates through this list + - chatbot + - litellm + - qwen + - kimi + - dashscope + - aistudio + - conversational + - dialogue + - copilot + - coding + - ai-assistant + - ai-agent + - multi-agent + - workflow + - workflows + - llmops + - tokens + - memory + - function-calling + pages: 30 # npm search pages to fetch for the current query + per_page: 50 # npm packages per search page + workers: 3 # parallel TruffleHog filesystem scans for downloaded packages + timeout: 300 # seconds before killing one package scan process tree + versions_per_package: 3 # scan latest N versions per package within max_version_age_days + max_version_age_days: 90 # skip package versions older than this many days + max_artifact_size_mb: 300 # skip/download-fail package tarballs larger than this + fetch_timeout: 20 # seconds per npm registry request + + pypi: + enabled: true # true = include PyPI registry in config-mode runs + mode: search # PyPI currently supports search mode + queries: # one query is used per source cycle; state rotates through this list + - llm + - ai + - agent + - agents + - assistant + - bot + - chatbot + - rag + - semantic + - prompt + - completion + - completions + - mcp + - openrouter + - groq + - xai + - litellm + - gemini + - qwen + - kimi + - dashscope + - aistudio + - studio + - OR + - chat + - chats + - conversation + - conversational + - dialogue + - copilot + - coding + - ai-assistant + - ai-agent + - multi-agent + - workflow + - workflows + - llmops + - benchmark + - tokens + - memory + - function-calling + pages: 40 # PyPI search pages to fetch for the current query + per_page: 50 # max package names to process per PyPI search page + workers: 3 # parallel TruffleHog filesystem scans for downloaded packages + timeout: 300 # seconds before killing one package scan process tree + versions_per_package: 3 # scan latest N release artifacts per package within max_version_age_days + max_version_age_days: 90 # skip package releases older than this many days + max_artifact_size_mb: 300 # skip/download-fail package artifacts larger than this + fetch_timeout: 20 # seconds per PyPI request + + package_git: + enabled: false # disabled after package-level discovery became duplicate-heavy + auth_pool: github_main # most package metadata points to GitHub; GitLab 401/403 falls back unauthenticated + auth_rotation: per_cycle + mode: search # package_git currently supports search mode and custom JSON/git URL targets + package_sources: # registries used for package -> repository discovery + - npm + - pypi + queries: + - ai + - agents + - assistant + - chatbot + - rag + - prompt + - completions + - mcp + - openrouter + - groq + - xai + - langchain + - litellm + - gemini + - qwen + - kimi + - dashscope + - aistudio + - OR + - chat + - chats + - conversation + - conversational + - dialogue + - copilot + - coding + - ai-assistant + - ai-agent + - multi-agent + - workflow + - workflows + - llmops + - benchmark + - tokens + - memory + - function-calling + pages: 10 # registry search pages per query for repo discovery + per_page: 50 # packages per search page + refresh_registry: true # merge cached repo candidates with fresh registry discovery each cycle + max_targets: 30 # bound one cycle so registry discovery cannot be starved by historical backlog + target_claim_order: balanced # split claims between oldest backlog and newest discovered repositories + workers: 3 # parallel TruffleHog git scans + timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes + versions_per_package: 3 # inspect repo metadata for up to N recent package versions + max_version_age_days: 90 # ignore package versions older than this during discovery + max_depth: 500 # git commit depth for package-derived repos + scan_full_history: false + max_commit_age_days: 0 # 0 avoids extra provider API commit-boundary lookup by default + commit_lookup_pages: 3 + skip_if_commit_lookup_fails: false + fetch_timeout: 20 # seconds per registry metadata request + + postman: + enabled: true # first run: enable manually for controlled backfill/tail scans + auth_pool: github_main # uses GitHub tokens for code search, commit lookup, and content download + auth_rotation: per_cycle # runner state still records a last auth; source-local pool rotates all tokens per request + mode: search # search = GitHub code search for Postman JSON artifacts; custom = target_file JSON targets + queries: + - anthropic + - gemini + - qwen + - kimi + - dashscope + - llm + - azure-openai + - openai.azure.com + - foundry + - services.ai.azure.com + - models.ai.azure.com + - DASHSCOPE_API_KEY + - QWEN_API_KEY + - MOONSHOT_API_KEY + - KIMI_API_KEY + - dashscope.aliyuncs.com + - api.moonshot.ai + - api.moonshot.cn + - GROQ_API_KEY + - api.groq.com + - OPENROUTER_API_KEY + - api.openrouter.ai + - REPLICATE_API_TOKEN + - api.replicate.com + - api.x.ai + - HF_TOKEN + - HUGGINGFACE_TOKEN + - ANTHROPIC_API_KEY + - api.anthropic.com + - rag + - agent + - assistant + search_kinds: + - all # Postman, Insomnia, Bruno, Thunder Client, Hoppscotch, and signature searches + pages: 2 # tail default; use 10 for one-time backfill up to GitHub's 1000-result cap + per_page: 100 + workers: 3 + timeout: 300 + fetch_timeout: 20 + max_targets: 0 # set a small value for smoke tests + max_file_age_days: 365 # backfill freshness filter by latest commit touching the file path; 0 disables + max_artifact_size_mb: 200 + postman_cache_dir: "{postman_cache_dir}" + github_code_search_rpm: 8 # safe per-token code search request rate; GitHub limit is about 10/min/token + all_tokens_cooldown: 1800 # fallback sleep when all GitHub tokens are rate-limited and no reset is known + stop_on_seen_pages: true # tail mode: stop after consecutive fully known pages + seen_page_threshold: 2 + min_pages_before_stop: 1 diff --git a/app/console_runner.py b/app/console_runner.py new file mode 100644 index 0000000..624a57f --- /dev/null +++ b/app/console_runner.py @@ -0,0 +1,7156 @@ +import sys + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('console runner could not disable bytecode writes') + +import argparse +import concurrent.futures +import copy +import contextlib +import hashlib +import json +import logging +import os +import secrets +import shutil +import threading +import time +from dataclasses import dataclass +from datetime import datetime, timedelta, timezone +from types import SimpleNamespace + + +if os.name == 'nt': + import ctypes + from ctypes import wintypes + + class _PROCESS_MEMORY_COUNTERS_EX(ctypes.Structure): + _fields_ = [ + ('cb', wintypes.DWORD), + ('PageFaultCount', wintypes.DWORD), + ('PeakWorkingSetSize', ctypes.c_size_t), + ('WorkingSetSize', ctypes.c_size_t), + ('QuotaPeakPagedPoolUsage', ctypes.c_size_t), + ('QuotaPagedPoolUsage', ctypes.c_size_t), + ('QuotaPeakNonPagedPoolUsage', ctypes.c_size_t), + ('QuotaNonPagedPoolUsage', ctypes.c_size_t), + ('PagefileUsage', ctypes.c_size_t), + ('PeakPagefileUsage', ctypes.c_size_t), + ('PrivateUsage', ctypes.c_size_t), + ] + + _P_PROCESS_MEMORY_COUNTERS_EX = ctypes.POINTER(_PROCESS_MEMORY_COUNTERS_EX) + _METRIC_KERNEL32 = ctypes.WinDLL('kernel32', use_last_error=True) + _METRIC_PSAPI = ctypes.WinDLL('psapi', use_last_error=True) + _METRIC_GET_CURRENT_PROCESS = _METRIC_KERNEL32.GetCurrentProcess + _METRIC_GET_CURRENT_PROCESS.argtypes = [] + _METRIC_GET_CURRENT_PROCESS.restype = wintypes.HANDLE + _GET_PROCESS_MEMORY_INFO = _METRIC_PSAPI.GetProcessMemoryInfo + _GET_PROCESS_MEMORY_INFO.argtypes = [ + wintypes.HANDLE, _P_PROCESS_MEMORY_COUNTERS_EX, wintypes.DWORD, + ] + _GET_PROCESS_MEMORY_INFO.restype = wintypes.BOOL +else: + _PROCESS_MEMORY_COUNTERS_EX = None + +from scanner import ( + ApiRequestError, + DockerContentTransferError, + DockerLayerInfrastructureError, + DockerRegistryResolutionError, + DockerResolverLeaseLostError, + DockerRemoteAccessError, + DockerHubDiscoveryTransportError, + GitLabDiscoveryTransportError, + RateLimitError, + ResultSinkError, + StagedResult, + acquire_scan_slot, + acquire_scan_slot_leases, + assign_finding_uids, + check_dependencies, + cleanup_pending_command_work_dirs, + cleanup_stale_temp_dirs, + configure_docker_accounts, + configure_docker_discovery_accounts, + configure_docker_discovery_tokens, + configure_docker_tokens, + csv_items, + docker_images_per_repository_limit, + docker_tag_resolution_is_conclusive, + drain_docker_auth_events, + fetch_github_archive_repos, + fetch_github_archive_file_targets, + fetch_github_gist_targets, + fetch_dockerhub_tags, + fetch_dockerhub_images, + fetch_dockerhub_search_page, + fetch_github_repo_items, + fetch_github_postman_targets, + fetch_github_repos, + fetch_gitlab_repo_items, + fetch_gitlab_repos, + fetch_huggingface_spaces, + fetch_npm_packages, + fetch_npm_package_git_repos, + fetch_pypi_packages, + fetch_pypi_package_git_repos, + fetch_recent_dockerhub_images, + fetch_recent_github_repo_items, + fetch_recent_github_repos, + fetch_docker_config_payload_classes, + fetch_recent_gitlab_repo_items, + fetch_recent_gitlab_repos, + restore_docker_endpoint_cooldowns, + get_trufflehog_cmd, + normalize_git_scan_resolution_target, + parse_github_repo_target, + parse_gitlab_project_target, + redact_secrets, + resolve_git_scan_target, + resolve_docker_content_manifest, + scan_config, + scan_limiter_enabled, + scan_slot_scope, + scan_target_result, + scan_targets_batch, + stage_result_bundle, + save_scan_result, + strip_nearby_context_for_persistence, + validate_postman_cache_artifact, + initialize_scanner_runtime, + write_structured_keycheck_candidates, +) +from scan_execution import QueueDispositionPolicy, stage_scan_result_in_scope +from scanner_db import ( + CI_SOFT_SKIP_REASONS, + DiscoveryPausedError, + DiscoveryRetryLeaseError, + ScanEventConflictError, + ScannerDB, + canonical_docker_layer_plan_bytes, + docker_adaptive_canary_selected, + docker_layer_canary_selected, + docker_layer_execution_policy_sha256, + docker_layer_selection_policy_sha256, + hash_file, + normalize_target as normalize_db_target, + queue_counts, + summarize_results, + validate_docker_adaptive_checkpoint, + validate_docker_layer_limits, +) +from result_spool import ( + ResultSpool, + SpoolBackpressureError, + SpoolContentionError, + SpoolTransientCapacityError, + prepare_scan_event, +) +from paths import apply_path_config, default_project_paths, resolve_optional_path +from docker_depth_experiment import ( + DOCKERHUB_DISCOVERY_ALGORITHM_VERSION, + DOCKERHUB_DISCOVERY_MAX_PAGES, + DOCKERHUB_DISCOVERY_MAX_PER_PAGE, + canonical_dockerhub_discovery_policy, + docker_depth_resolver_authority, + validate_docker_depth_config, + validate_dockerhub_discovery_policies, +) +from lifecycle_authority import ( + CHILD_KIND_ENV, + DISCOVERY_PRODUCER_ROLE, + DISCOVERY_PRODUCER_SOURCES, + LifecycleAuthorityError, + require_active_supervisor_child, +) +from process_identity import current_process_identity +from result_bundle import ( + ResultBundleReader, + bundle_partial_relative_path, + ensure_bundle_reservation_paths, +) +from target_identity import parse_docker_target +from runtime_security import ( + durable_unlink, + PrivatePathState, + PrivateFileLock, + ensure_private_directory, + harden_private_file, + inspect_private_relative_path, + preflight_lifecycle_paths, + read_private_json, + private_file_ready, + require_private_directory, + require_sensitive_runtime_paths, +) + + +logger = logging.getLogger(__name__) +SOURCE_INFRASTRUCTURE_HOLD_EXIT = 75 +DOCKERHUB_DEEP_INTERVAL = timedelta(hours=72) +DOCKERHUB_DISCOVERY_STATE_KEY = 'dockerhub_discovery' +DOCKERHUB_DISCOVERY_RETRY_LEASE_SECONDS = 300 +DOCKERHUB_DISCOVERY_ERROR_CATEGORIES = frozenset(( + 'account_pool_exhausted', 'auth_forbidden', 'auth_invalid', + 'auth_unavailable', 'invalid_payload', 'network', 'page_unavailable', + 'policy_mismatch', 'provider_cooldown', 'provider_unavailable', + 'query_removed', 'rate_limit', 'remote_transient', 'request_failed', + 'tail_unavailable', 'transport', +)) + + +class UnresolvedHandoffInfrastructureError(RuntimeError): + pass + + +def ci_seed_statement_timeout(exc): + if isinstance(exc, TimeoutError): + return True + return bool( + str(getattr(exc, 'sqlstate', '') or '') == '57014' + and 'statement timeout' in str(exc).lower() + ) + + +def commit_if_postgres(db): + try: + conn = getattr(db, 'conn', None) + if conn and getattr(conn, 'is_postgres', False): + conn.commit() + except Exception: + pass + + +def load_set_from_file(path): + if not os.path.exists(path): + return set() + + with open(path, 'r', encoding='utf-8') as f: + return {line.strip().lstrip('\ufeff') for line in f if line.strip()} + + +@contextlib.contextmanager +def projection_file_lock(path, timeout_sec=30): + lock_path = f'{path}.projection.lock' + parent = os.path.dirname(os.path.abspath(lock_path)) + ensure_private_directory(parent, reject_reparse=True) + deadline = time.monotonic() + max(0.1, float(timeout_sec)) + while True: + lock = PrivateFileLock(lock_path) + try: + lock.acquire() + break + except BlockingIOError: + if time.monotonic() >= deadline: + raise TimeoutError(f'timed out acquiring projection lock {lock_path}') + time.sleep(0.05) + try: + yield + finally: + lock.release() + + +def _write_lines_unlocked(path, lines): + parent = os.path.dirname(path) + if parent: + ensure_private_directory(parent, reject_reparse=True) + tmp_path = f'{path}.{os.getpid()}.{threading.get_ident()}.{time.time_ns()}.tmp' + with open(tmp_path, 'w', encoding='utf-8') as f: + for line in lines: + f.write(f"{line}\n") + f.flush() + try: + os.fsync(f.fileno()) + except OSError: + pass + os.replace(tmp_path, path) + harden_private_file(path) + + +def write_lines(path, lines): + with projection_file_lock(path): + _write_lines_unlocked(path, lines) + + +def _append_lines_unlocked(path, lines): + parent = os.path.dirname(path) + if parent: + ensure_private_directory(parent, reject_reparse=True) + with open(path, 'a', encoding='utf-8') as f: + for line in lines: + f.write(f"{line}\n") + f.flush() + try: + os.fsync(f.fileno()) + except OSError: + pass + harden_private_file(path) + + +def append_lines(path, lines): + with projection_file_lock(path): + _append_lines_unlocked(path, lines) + + +def skip_startup_cleanup(): + return str(os.getenv('SCANNER_SKIP_STARTUP_CLEANUP', '')).strip().lower() in ('1', 'true', 'yes', 'on') + + +def console_safe_text(value): + text = str(value) + encoding = getattr(sys.stdout, 'encoding', None) or 'utf-8' + return text.encode(encoding, errors='replace').decode(encoding, errors='replace') + + +def safe_print(value='', **kwargs): + print(console_safe_text(value), **kwargs) + + +def bool_config(value, default=False): + if value is None: + return default + if isinstance(value, bool): + return value + return str(value).strip().lower() in ('1', 'true', 'yes', 'on') + + +def normalize_target(target, platform): + return normalize_db_target(target, platform) + + +def read_custom_targets(path): + if not path: + raise ValueError('--target-file is required for custom mode') + if not os.path.exists(path): + raise FileNotFoundError(path) + + with open(path, 'r', encoding='utf-8') as f: + return [line.strip().lstrip('\ufeff') for line in f if line.strip()] + + +def read_queries(args): + queries = [] + if args.query_file: + if not os.path.exists(args.query_file): + raise FileNotFoundError(args.query_file) + with open(args.query_file, 'r', encoding='utf-8') as f: + queries.extend(line.strip() for line in f if line.strip()) + if args.query: + queries.extend(query.strip() for query in args.query.split(',') if query.strip()) + return list(dict.fromkeys(queries)) + + +def parse_datetime(value): + if not value: + return None + try: + normalized = str(value).replace('Z', '+00:00') + parsed = datetime.fromisoformat(normalized) + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.astimezone(timezone.utc) + except ValueError: + return None + + +def updated_target_rescan_enabled(args): + return bool(getattr(args, 'updated_target_rescan_enabled', False)) and args.platform in { + 'github', 'gitlab', 'huggingface', + } + + +def repo_item_discovery(item, args): + target = item.get('url') + if not target: + return None + if not updated_target_rescan_enabled(args) and args.platform != 'huggingface': + return target + remote_field = { + 'github': 'pushed_at', + 'gitlab': 'last_activity_at', + 'huggingface': 'updated_at', + }.get(args.platform) + remote_time = parse_datetime(item.get(remote_field)) if remote_field else None + discovery = { + 'target': target, + 'remote_modified_at': remote_time.isoformat(timespec='seconds') if remote_time else None, + } + if args.platform == 'huggingface': + for field in ('private', 'protected', 'gated', 'disabled'): + if field in item: + discovery[field] = item[field] + return discovery + + +def huggingface_discovery_target_is_restricted(item): + if not isinstance(item, dict): + return False + return any( + field in item and item[field] is not False and item[field] is not None + for field in ('private', 'protected', 'gated', 'disabled') + ) + + +def filter_repo_items_by_age(items, args): + max_days = int(getattr(args, 'max_repo_age_days', 0) or 0) + if max_days <= 0: + return [ + discovery for item in items + if (discovery := repo_item_discovery(item, args)) is not None + ] + + field = getattr(args, 'repo_age_field', None) or 'created_at' + cutoff = datetime.now(timezone.utc) - timedelta(days=max_days) + kept = [] + skipped_old = 0 + skipped_missing = 0 + + for item in items: + url = item.get('url') + if not url: + continue + value = item.get(field) + parsed = parse_datetime(value) + if not parsed: + skipped_missing += 1 + if updated_target_rescan_enabled(args): + discovery = repo_item_discovery(item, args) + if discovery is not None: + kept.append(discovery) + continue + if parsed >= cutoff: + discovery = repo_item_discovery(item, args) + if discovery is not None: + kept.append(discovery) + else: + skipped_old += 1 + + print( + f"Repo age filter: kept {len(kept)}, skipped {skipped_old} older than " + f"{max_days} days by {field}, skipped {skipped_missing} missing dates" + ) + return kept + + +def make_repo_candidate_callback(db=None, run_id=None, cycle_id=None, query=None): + if not db or not run_id: + return None + def callback(candidates): + return db.record_package_repo_candidates(run_id, cycle_id, query, candidates) + return callback + + +def fetch_package_git_targets_from_db(args, db=None): + if not db: + return [] + queries = read_queries(args) + package_sources = [item.strip().lower() for item in str(getattr(args, 'package_sources', 'npm,pypi') or '').split(',') if item.strip()] + targets = [] + seen = set() + for query in queries: + candidates = db.get_package_repo_candidates(query, package_sources, limit=max(1000, int(args.pages or 1) * int(args.per_page or 50) * 10)) + for candidate in candidates: + target = json.dumps(candidate, separators=(',', ':'), ensure_ascii=False) + normalized = normalize_target(target, 'package_git') + if normalized not in seen: + targets.append(target) + seen.add(normalized) + return targets + + +def source_list(value, default): + if value is None: + value = default + if isinstance(value, str): + return [item.strip() for item in value.split(',') if item.strip()] + return [str(item).strip() for item in value if str(item).strip()] + + +def ci_metadata_candidate_values(value): + if not value: + return [] + try: + data = json.loads(value) if isinstance(value, str) else value + except Exception: + return [] + + output = [] + interesting = {'repository', 'repo', 'repo_url', 'url', 'link', 'project', 'path_with_namespace'} + + def visit(item): + if isinstance(item, dict): + for key, child in item.items(): + if key in interesting and isinstance(child, str): + output.append(child) + visit(child) + elif isinstance(item, list): + for child in item: + visit(child) + + visit(data) + return output + + +def fetch_ci_repo_targets_from_db(args, db=None, provider='github'): + owned_db = None + if not db or not getattr(db, 'conn', None): + db_path = getattr(args, 'database_path', None) or os.getenv('SCANNER_DB_PATH') or os.getenv('SCAN_DB_PATH') + db_url = getattr(args, 'database_url', None) or os.getenv('SCANNER_DB_URL') or os.getenv('DATABASE_URL') + if not db_path: + db_path = os.path.join(getattr(args, 'save_dir', '') or '', 'scanner_active.db') + try: + owned_db = ScannerDB(db_path=db_path, db_url=db_url, initialize=False) + if not owned_db.enabled: + raise RuntimeError('scanner DB disabled') + db = owned_db + print(f'CI target discovery ({provider}): opened scanner DB {owned_db.db_display}') + except Exception as e: + print(f'CI target discovery ({provider}): scanner DB unavailable: {e}') + return [] + seed_sources = source_list(getattr(args, 'ci_seed_sources', None), 'github,gitlab,package_git') + limit = max(1, int(getattr(args, 'ci_max_repos_per_cycle', 100) or 100)) + scan_limit = max(1, int(getattr(args, 'ci_seed_scan_limit', 5000) or 5000)) + query_batch = min(500, max(25, int(getattr(args, 'ci_seed_query_batch_size', 250) or 250))) + platform = 'github_actions' if provider == 'github' else 'gitlab_ci' + soft_skip_reasons = CI_SOFT_SKIP_REASONS.get(platform, set()) + cooldown_days = int(getattr(args, 'ci_soft_cooldown_days', 7) or 0) + cooldown_keys = set() + cooldown_rows_loaded = 0 + cooldown_truncated = False + if cooldown_days > 0 and soft_skip_reasons: + cutoff = (datetime.now(timezone.utc) - timedelta(days=cooldown_days)).isoformat(timespec='seconds') + reason_predicate = ' OR '.join( + f"skipped_reason = '{reason}'" for reason in sorted(soft_skip_reasons) + ) + cooldown_scope = f"source = '{platform}' AND ({reason_predicate})" + cursor_ended_at = None + cursor_id = None + while cooldown_rows_loaded < scan_limit: + page_limit = min(query_batch, scan_limit - cooldown_rows_loaded) + cursor_clause = '' + params = [cutoff] + if cursor_ended_at is not None and cursor_id is not None: + cursor_clause = 'AND (ended_at, id) < (?, ?)' + params.extend((cursor_ended_at, cursor_id)) + params.append(page_limit) + try: + page = db.conn.execute(''' + SELECT id, target, skipped_reason, ended_at + FROM target_scans + WHERE status = 'skipped' + AND {} + AND ended_at >= ? {} + ORDER BY ended_at DESC, id DESC + LIMIT ? + '''.format(cooldown_scope, cursor_clause), params).fetchall() + except Exception: + if getattr(db.conn, 'is_postgres', False): + db.conn.rollback() + raise + if not page: + break + for row in page: + if provider == 'github': + repo, _ = parse_github_repo_target(row['target']) + else: + repo, _ = parse_gitlab_project_target(row['target']) + if repo: + cooldown_keys.add(repo.lower()) + cooldown_rows_loaded += len(page) + cursor_ended_at = page[-1]['ended_at'] + cursor_id = page[-1]['id'] + if len(page) < page_limit: + break + if cooldown_rows_loaded >= scan_limit: + probe_params = [cutoff, cursor_ended_at, cursor_id] + try: + cooldown_truncated = bool(db.conn.execute(''' + SELECT 1 + FROM target_scans + WHERE status = 'skipped' + AND {} + AND ended_at >= ? + AND (ended_at, id) < (?, ?) + ORDER BY ended_at DESC, id DESC + LIMIT 1 + '''.format(cooldown_scope), probe_params).fetchone()) + except Exception: + if getattr(db.conn, 'is_postgres', False): + db.conn.rollback() + raise + break + if cooldown_truncated: + safe_print( + f'Warning: CI cooldown history reached its bounded {scan_limit}-row limit; ' + 'older rows in the configured cooldown window were not loaded', + flush=True, + ) + if getattr(db.conn, 'is_postgres', False): + # Keyset pages keep each HDD transaction bounded. The per-source branch + # limit still preserves the exact global newest-first ordering. + rows = [] + cursor_ended_at = None + cursor_id = None + while len(rows) < scan_limit: + page_limit = min(query_batch, scan_limit - len(rows)) + cursor_clause = '' + params = [*seed_sources] + if cursor_ended_at is not None and cursor_id is not None: + cursor_clause = 'AND (ended_at < ? OR (ended_at = ? AND id < ?))' + params.extend((cursor_ended_at, cursor_ended_at, cursor_id)) + params.extend((page_limit, page_limit)) + try: + page = db.conn.execute(''' + SELECT ranked.id, ranked.source, ranked.query, ranked.target, + ranked.normalized_target, ranked.ended_at AS latest_seen + FROM (VALUES {}) AS seed(source) + CROSS JOIN LATERAL ( + SELECT id, source, query, target, normalized_target, ended_at + FROM target_scans + WHERE source = seed.source AND ended_at IS NOT NULL {} + ORDER BY ended_at DESC, id DESC + LIMIT ? + ) AS ranked + ORDER BY ranked.ended_at DESC, ranked.id DESC + LIMIT ? + '''.format(','.join('(?)' for _ in seed_sources), cursor_clause), params).fetchall() + except Exception as exc: + if not ci_seed_statement_timeout(exc): + raise + try: + db.conn.rollback() + except Exception as rollback_exc: + reset = getattr(db, '_reset_connection', None) + if not reset or not reset(): + raise exc from rollback_exc + safe_print( + f'Warning: CI target discovery ({provider}) degraded after a bounded database ' + 'statement timeout; no seed targets from this cycle were used', + flush=True, + ) + if owned_db: + owned_db.close() + return [] + if not page: + break + rows.extend(page) + cursor_ended_at = page[-1]['latest_seen'] + cursor_id = page[-1]['id'] + if len(page) < page_limit: + break + else: + rows = db.conn.execute(''' + SELECT source, query, target, normalized_target, ended_at AS latest_seen + FROM target_scans + WHERE source IN ({}) + ORDER BY ended_at DESC NULLS LAST + LIMIT ? + '''.format(','.join('?' for _ in seed_sources)), [*seed_sources, scan_limit]).fetchall() + seed_records = [] + seen_scan_targets = set() + for row in rows: + scan_key = ( + row['source'] or '', + row['query'] or '', + row['target'] or '', + row['normalized_target'] or '', + ) + if scan_key in seen_scan_targets: + continue + seen_scan_targets.add(scan_key) + candidates = [row['target'], row['normalized_target']] + try: + parsed_json = json.loads(str(row['target'] or '')) + except Exception: + parsed_json = None + if isinstance(parsed_json, dict): + candidates.append(parsed_json) + if parsed_json.get('repo_url'): + candidates.append(parsed_json.get('repo_url')) + seed_records.append({ + 'source': row['source'], + 'query': row['query'], + 'candidates': candidates, + }) + + # Lower-priority sources may fill only the remaining aggregate DB-row + # budget; they must not multiply ci_seed_scan_limit. + if 'package_git' in seed_sources and len(seed_records) < scan_limit: + remaining_seed_rows = scan_limit - len(seed_records) + package_rows = db.conn.execute(''' + SELECT package_source, package_name, package_version, query, repo_url, provider, last_seen_at + FROM package_repo_candidates + WHERE LOWER(provider) = ? + ORDER BY last_seen_at DESC + LIMIT ? + ''', [provider, remaining_seed_rows]).fetchall() + for row in package_rows: + seed_records.append({ + 'source': 'package_git', + 'query': row['query'], + 'candidates': [ + row['repo_url'], + { + 'repo_url': row['repo_url'], + 'provider': row['provider'], + 'package_source': row['package_source'], + 'name': row['package_name'], + 'version': row['package_version'], + }, + ], + }) + + use_finding_seeds = bool(getattr(args, 'ci_use_finding_seeds', False)) + finding_sources = [source for source in seed_sources if source in ('github', 'gitlab', 'package_git')] if use_finding_seeds else [] + if finding_sources and len(seed_records) < scan_limit: + per_source_finding_limit = max(limit * 50, scan_limit // max(1, len(finding_sources))) + for source_name in finding_sources: + remaining_seed_rows = scan_limit - len(seed_records) + if remaining_seed_rows <= 0: + break + finding_rows = db.conn.execute(''' + SELECT source, query, target, normalized_target, source_metadata_json, raw_finding_json, created_at + FROM findings + WHERE source = ? + ORDER BY id DESC + LIMIT ? + ''', [source_name, min(per_source_finding_limit, remaining_seed_rows)]).fetchall() + for row in finding_rows: + candidates = [row['target'], row['normalized_target']] + candidates.extend(ci_metadata_candidate_values(row['source_metadata_json'])) + candidates.extend(ci_metadata_candidate_values(row['raw_finding_json'])) + seed_records.append({ + 'source': row['source'], + 'query': row['query'], + 'candidates': candidates, + }) + + offered = [] + seen = set() + parsed = 0 + skipped_known = 0 + skipped_duplicate = 0 + skipped_unparseable = 0 + skipped_cooldown = 0 + for record in seed_records: + for candidate in record['candidates']: + if provider == 'github': + repo, url = parse_github_repo_target(candidate) + platform = 'github_actions' + else: + repo, url = parse_gitlab_project_target(candidate) + platform = 'gitlab_ci' + if not repo: + skipped_unparseable += 1 + continue + key = repo.lower() + parsed += 1 + if key in cooldown_keys: + skipped_cooldown += 1 + continue + target_payload = json.dumps({ + 'repo' if provider == 'github' else 'project': repo, + 'url': url, + 'seed_source': record['source'], + 'seed_query': record['query'], + }, separators=(',', ':'), ensure_ascii=False) + if key in seen: + skipped_duplicate += 1 + continue + seen.add(key) + offered.append((target_payload, normalize_target(target_payload, platform))) + if len(offered) >= scan_limit: + break + if len(offered) >= scan_limit: + break + + known = set() + lookup = getattr(db, 'known_target_normalizations_for', None) + if lookup: + for start in range(0, len(offered), 64): + batch = [target for target, _ in offered[start:start + 64]] + try: + known.update(lookup(platform, platform, batch) or ()) + except Exception as exc: + safe_print(f'Warning: CI known-target lookup failed open: {exc}') + eligible = [] + for target, normalized in offered: + if normalized in known: + skipped_known += 1 + continue + eligible.append(target) + targets = eligible[:limit] + print( + f'CI target discovery ({provider}): selected {len(targets)} target(s); ' + f'db_rows={len(seed_records)}, parsed={parsed}, known={len(known)}, ' + f'skipped_known={skipped_known}, skipped_duplicate={skipped_duplicate}, ' + f'skipped_unparseable={skipped_unparseable}, skipped_cooldown={skipped_cooldown}, ' + f'cooldown_rows={cooldown_rows_loaded}, cooldown_truncated={str(cooldown_truncated).lower()}, ' + f'seed_sources={",".join(seed_sources)}. ' + f'Increase ci_seed_scan_limit or add fresh seed repos if db_rows equals scan limit.' + ) + if owned_db: + owned_db.close() + else: + commit_if_postgres(db) + return targets + + +def _dockerhub_discovery_policy_values(pages, per_page, sort_by, sort_order): + try: + effective_pages = max(1, min(DOCKERHUB_DISCOVERY_MAX_PAGES, int(pages))) + effective_per_page = max( + 1, min(DOCKERHUB_DISCOVERY_MAX_PER_PAGE, int(per_page)), + ) + except (TypeError, ValueError, OverflowError): + raise ValueError('DockerHub discovery page policy is invalid') from None + return canonical_dockerhub_discovery_policy( + effective_pages, + effective_per_page, + str(sort_by or 'updated_at'), + str(sort_order or 'desc'), + ) + + +def dockerhub_discovery_policy(args): + return _dockerhub_discovery_policy_values( + getattr(args, 'pages', 1), + getattr(args, 'per_page', DOCKERHUB_DISCOVERY_MAX_PER_PAGE), + getattr(args, 'docker_sort_by', 'updated_at'), + getattr(args, 'sort_order', 'desc'), + ) + + +def configured_dockerhub_discovery_policies(source_config): + return { + effective['query']: { + key: value for key, value in effective.items() + if key not in ('query', 'max_targets') + } + for effective in validate_dockerhub_discovery_policies(source_config) + } + + +def managed_dockerhub_discovery(args, db, source_name): + return bool( + source_name == 'dockerhub' + and getattr(args, 'platform', None) == 'docker' + and getattr(args, 'mode', None) == 'search' + and db + and getattr(db, 'conn', None) + and getattr(db.conn, 'is_postgres', False) + and callable(getattr(db, 'persist_dockerhub_discovery_page', None)) + and callable(getattr(db, 'enqueue_discovery_retry', None)) + ) + + +def _require_discovery_provider_enabled(db): + control = db.runtime_control_state() + if control['effective_discovery_paused']: + raise DiscoveryPausedError(control) + return control + + +def _safe_dockerhub_discovery_category(error): + category = str(getattr(error, 'category', '') or '') + return category if category in DOCKERHUB_DISCOVERY_ERROR_CATEGORIES else 'page_unavailable' + + +def _validated_dockerhub_page(query, requested_page, args, policy=None): + policy = policy or dockerhub_discovery_policy(args) + result = fetch_dockerhub_search_page( + query, + requested_page, + per_page=policy['per_page'], + sort_by=policy['sort_by'], + sort_order=policy['sort_order'], + request_timeout=getattr(args, 'fetch_timeout', 15), + ) + valid = isinstance(result, dict) and result.get('page') == requested_page + repositories = result.get('repositories') if valid else None + total_count = result.get('total_count') if valid else None + if ( + not isinstance(repositories, list) + or isinstance(total_count, bool) + or (isinstance(total_count, float) and not total_count.is_integer()) + ): + valid = False + else: + try: + total_count = int(total_count) + except (TypeError, ValueError, OverflowError): + valid = False + result_count = len(repositories) if isinstance(repositories, list) else -1 + absolute_start = (requested_page - 1) * int(policy['per_page']) + absolute_bound = absolute_start + max(0, result_count) + if ( + not valid + or total_count < absolute_bound + or result_count > int(policy['per_page']) + or (result_count == 0 and total_count > absolute_start) + ): + raise DockerHubDiscoveryTransportError( + 'Docker Hub search page returned an invalid payload', + category='invalid_payload', remote_attempted=True, retryable=False, + ) + return result, repositories, total_count + + +def annotate_dockerhub_experiment_observation_args(args, experiment, cycle_id=None): + args.docker_depth_collection_only = bool( + experiment is not None and not experiment.enabled + ) + if experiment is None: + return args + args.dockerhub_discovery_ordered_queries = tuple(experiment.queries) + args.dockerhub_discovery_ordered_query_hash = str( + experiment.ordered_query_hash + ) + args.dockerhub_discovery_query_count = len(experiment.queries) + args.dockerhub_discovery_collection_generation = str( + experiment.collection_generation + ) + args.dockerhub_discovery_cycle_id = cycle_id + return args + + +def annotate_dockerhub_runtime_args( + args, experiment, discovery_policies, cycle_id=None, +): + annotate_dockerhub_experiment_observation_args(args, experiment, cycle_id) + if experiment is None: + return args + policy_hashes = { + policy['policy_sha256'] for policy in discovery_policies.values() + } + if len(policy_hashes) != 1: + raise RuntimeError('DockerHub experiment resolver policy authority is ambiguous') + args.docker_depth_experiment_authority = docker_depth_resolver_authority( + experiment, next(iter(policy_hashes)), + ) + return args + + +def _dockerhub_discovery_observation( + args, query, page_number, per_page, policy_sha256, pass_kind, + total_count=None, query_complete=False, +): + ordered_queries = getattr(args, 'dockerhub_discovery_ordered_queries', None) + ordered_query_hash = getattr( + args, 'dockerhub_discovery_ordered_query_hash', None, + ) + if ordered_queries is None and ordered_query_hash is None: + return None + if not isinstance(ordered_queries, tuple) or not ordered_queries: + raise RuntimeError('DockerHub experiment observation queries are unavailable') + if ( + not isinstance(ordered_query_hash, str) + or not _valid_dockerhub_policy_sha256(ordered_query_hash) + or len(set(ordered_queries)) != len(ordered_queries) + or query not in ordered_queries + or int(getattr(args, 'dockerhub_discovery_query_count', 0) or 0) + != len(ordered_queries) + ): + raise RuntimeError('DockerHub experiment observation identity is invalid') + return { + 'cycle_id': getattr(args, 'dockerhub_discovery_cycle_id', None), + 'query_ordinal': ordered_queries.index(query), + 'query_count': len(ordered_queries), + 'page_number': page_number, + 'page_limit': int(getattr(args, 'pages', 1)), + 'per_page': per_page, + 'total_count': total_count, + 'policy_sha256': policy_sha256, + 'pass_kind': pass_kind, + 'collection_generation': str( + getattr(args, 'dockerhub_discovery_collection_generation', '') or '' + ), + 'ordered_query_hash': ordered_query_hash, + 'query_complete': bool(query_complete), + } + + +def _confirmed_discovery_retry(db, source, query, policy_sha256, pass_kind, + work_kind, page_start, page_end, error, + observation=None): + retry_at = getattr(error, 'retry_at', None) + try: + observation_kwargs = ( + {'observation': observation} if observation is not None else {} + ) + result = db.enqueue_discovery_retry( + source, + query, + policy_sha256, + pass_kind, + work_kind, + page_start=page_start, + page_end=page_end, + available_after=retry_at, + error_category=_safe_dockerhub_discovery_category(error), + **observation_kwargs, + ) + confirmed = bool( + isinstance(result, dict) + and not isinstance(result.get('id'), bool) + and int(result.get('id') or 0) >= 1 + and result.get('status') in ('pending', 'leased') + ) + except Exception: + raise RuntimeError('DockerHub discovery retry delegation failed') from None + if not confirmed: + raise RuntimeError('DockerHub discovery retry delegation was not confirmed') + return result + + +def run_dockerhub_incremental_discovery( + args, db, source='dockerhub', *, partial_metrics=None, +): + policy = dockerhub_discovery_policy(args) + annotated_policy = str( + getattr(args, 'dockerhub_discovery_policy_sha256', '') or '' + ) + if annotated_policy and annotated_policy != policy['policy_sha256']: + raise RuntimeError('DockerHub discovery policy annotation is stale') + policy_sha256 = policy['policy_sha256'] + deep = bool(getattr(args, 'dockerhub_discovery_deep', False)) + pass_kind = 'deep' if deep else 'ordinary' + metrics = { + 'fetched_count': 0, + 'queued_new_count': 0, + 'queued_updated_count': 0, + 'discovery_pages_fetched': 0, + 'discovery_retry_enqueued_count': 0, + 'discovery_retry_inserted_count': 0, + 'discovery_retry_coalesced_count': 0, + 'discovery_preexisting_count': 0, + 'discovery_known_page_count': 0, + 'discovery_stopped_on_preexisting': False, + 'discovery_deep': deep, + 'discovery_pass_kind': pass_kind, + 'deep_dispatch_durable': False, + 'cycle_status': 'completed', + } + query = str(getattr(args, 'query', '') or '') + known_streak = 0 + new_this_pass = set() + knownness_disabled = False + + def publish_metrics(): + if isinstance(partial_metrics, dict): + partial_metrics.update(metrics) + + publish_metrics() + + def delegate(error, work_kind, page_start, page_end): + observation = _dockerhub_discovery_observation( + args, query, page_start, policy['per_page'], policy_sha256, + pass_kind, None, + ) + report = _confirmed_discovery_retry( + db, source, query, policy_sha256, pass_kind, + work_kind, page_start, page_end, error, observation, + ) + metrics['discovery_retry_enqueued_count'] += 1 + metrics['discovery_retry_inserted_count'] += int( + report.get('inserted_count', 0) or 0 + ) + metrics['discovery_retry_coalesced_count'] += int( + report.get('coalesced_count', 0) or 0 + ) + metrics['cycle_status'] = 'completed_with_retries' + publish_metrics() + + def admit(repositories, page_number, total_count, query_complete=False): + try: + observation = _dockerhub_discovery_observation( + args, query, page_number, policy['per_page'], policy_sha256, + pass_kind, total_count, query_complete, + ) + observation_kwargs = ( + {'observation': observation} if observation is not None else {} + ) + experiment_authority = getattr( + args, 'docker_depth_experiment_authority', None, + ) + if isinstance(experiment_authority, dict): + observation_kwargs.update({ + 'experiment_authority': experiment_authority, + 'final_cutover': True, + }) + report = db.persist_dockerhub_discovery_page( + source, query, repositories, **observation_kwargs, + ) + except DiscoveryPausedError: + raise + except Exception: + raise RuntimeError('DockerHub discovery page admission failed') from None + if not isinstance(report, dict): + raise RuntimeError('DockerHub discovery page admission was not confirmed') + counts = {} + for key in ( + 'attempted_count', 'normalized_count', 'preexisting_count', 'inserted_count', + ): + value = report.get(key) + if isinstance(value, bool): + raise RuntimeError('DockerHub discovery page admission report is invalid') + try: + counts[key] = int(value) + except (TypeError, ValueError, OverflowError): + raise RuntimeError( + 'DockerHub discovery page admission report is invalid' + ) from None + if counts[key] < 0: + raise RuntimeError('DockerHub discovery page admission report is invalid') + if ( + counts['attempted_count'] != len(repositories) + or counts['normalized_count'] > counts['attempted_count'] + or counts['preexisting_count'] > counts['normalized_count'] + or counts['inserted_count'] > counts['normalized_count'] + ): + raise RuntimeError('DockerHub discovery page admission report is invalid') + metrics['fetched_count'] += counts['attempted_count'] + metrics['queued_new_count'] += counts['inserted_count'] + metrics['discovery_preexisting_count'] += counts['preexisting_count'] + metrics['discovery_pages_fetched'] += 1 + publish_metrics() + return report, counts + + def observe_knownness(report, counts): + nonlocal known_streak, knownness_disabled + normalized = report.get('normalized_repositories') + preexisting = report.get('preexisting_repositories') + certain = bool( + isinstance(normalized, (set, frozenset)) + and isinstance(preexisting, (set, frozenset)) + and len(normalized) == counts['normalized_count'] + and len(preexisting) == counts['preexisting_count'] + and preexisting.issubset(normalized) + ) + if not certain: + known_streak = 0 + if isinstance(normalized, (set, frozenset)): + new_this_pass.update(normalized) + else: + knownness_disabled = True + return + all_preexisting = bool( + normalized + and normalized.issubset(preexisting) + and normalized.isdisjoint(new_this_pass) + ) + new_this_pass.update(normalized - preexisting) + known_streak = known_streak + 1 if all_preexisting else 0 + if all_preexisting: + metrics['discovery_known_page_count'] += 1 + + try: + _require_discovery_provider_enabled(db) + _, repositories, total_count = _validated_dockerhub_page(query, 1, args) + except DockerHubDiscoveryTransportError as error: + if not getattr(error, 'retryable', True): + raise + delegate(error, 'query', 1, policy['pages']) + metrics['deep_dispatch_durable'] = deep + publish_metrics() + return metrics + + expected_pages = max( + 1, + min(policy['pages'], (total_count + policy['per_page'] - 1) // policy['per_page']), + ) + report, counts = admit( + repositories, 1, total_count, + query_complete=not repositories or expected_pages == 1, + ) + if not repositories: + metrics['deep_dispatch_durable'] = deep + publish_metrics() + return metrics + observe_knownness(report, counts) + + if not deep and not knownness_disabled and known_streak >= 2: + metrics['discovery_stopped_on_preexisting'] = True + metrics['deep_dispatch_durable'] = deep + publish_metrics() + return metrics + + for page in range(2, expected_pages + 1): + try: + _require_discovery_provider_enabled(db) + _, repositories, page_total_count = _validated_dockerhub_page( + query, page, args, + ) + except DockerHubDiscoveryTransportError as error: + if not getattr(error, 'retryable', True): + raise + known_streak = 0 + if getattr(error, 'remote_attempted', True): + delegate(error, 'page', page, page) + continue + delegate(error, 'range', page, expected_pages) + break + + report, counts = admit( + repositories, page, page_total_count, + query_complete=not repositories or page >= expected_pages, + ) + if not repositories: + known_streak = 0 + break + observe_knownness(report, counts) + if not deep and not knownness_disabled and known_streak >= 2: + metrics['discovery_stopped_on_preexisting'] = True + break + + metrics['deep_dispatch_durable'] = deep + publish_metrics() + return metrics + + +def process_dockerhub_discovery_retry(args, db, source, configured_policies): + metrics = { + 'discovery_retry_claimed_count': 0, + 'discovery_retry_pages_fetched': 0, + 'discovery_retry_queued_new_count': 0, + 'discovery_retry_completed_count': 0, + 'discovery_retry_deferred_count': 0, + 'discovery_retry_held_count': 0, + 'discovery_retry_error_count': 0, + } + required = ( + 'claim_discovery_retries', 'persist_dockerhub_discovery_page', + 'finish_discovery_retry', 'renew_discovery_retry_lease', + 'update_discovery_retry', 'hold_discovery_retry', + ) + if ( + source != 'dockerhub' + or getattr(args, 'platform', None) != 'docker' + or getattr(args, 'mode', None) != 'search' + or not db + or not getattr(db, 'conn', None) + or not getattr(db.conn, 'is_postgres', False) + or any(not callable(getattr(db, name, None)) for name in required) + ): + return metrics + + policy_hashes = { + query: policy.get('policy_sha256') + for query, policy in configured_policies.items() + if isinstance(policy, dict) + } + lease_owner = f'dockerhub-retry:{os.getpid()}:{secrets.token_hex(8)}'[:128] + try: + claims = db.claim_discovery_retries( + source, policy_hashes, lease_owner, limit=1, + lease_seconds=DOCKERHUB_DISCOVERY_RETRY_LEASE_SECONDS, + ) + except Exception: + metrics['discovery_retry_error_count'] = 1 + logger.warning('DockerHub discovery retry claim failed') + return metrics + if not claims: + return metrics + + claim = claims[0] + metrics['discovery_retry_claimed_count'] = 1 + query = claim.get('query') if isinstance(claim, dict) else None + policy = configured_policies.get(query) if isinstance(query, str) else None + + def hold(category): + try: + db.hold_discovery_retry( + claim.get('id'), claim.get('lease_owner'), claim.get('lease_token'), + category, + ) + metrics['discovery_retry_held_count'] = 1 + except Exception: + metrics['discovery_retry_error_count'] = 1 + logger.warning('DockerHub discovery retry hold failed') + + if policy is None: + hold('query_removed') + return metrics + if claim.get('policy_sha256') != policy.get('policy_sha256'): + hold('policy_mismatch') + return metrics + + try: + work_kind = str(claim.get('work_kind') or '') + page_start = int(claim.get('page_start')) + stored_page_end = int(claim.get('page_end')) + next_page = int(claim.get('next_page')) + effective_end = min(stored_page_end, int(policy['pages'])) + valid_work = work_kind in ('query', 'page', 'range') + valid_work = valid_work and 1 <= page_start <= next_page <= stored_page_end <= 30 + valid_work = valid_work and (work_kind != 'query' or page_start == 1) + valid_work = valid_work and (work_kind != 'page' or page_start == stored_page_end) + except (TypeError, ValueError, OverflowError, KeyError): + valid_work = False + if not valid_work: + hold('invalid_payload') + return metrics + + if next_page > effective_end: + hold('invalid_payload') + return metrics + + page = next_page + remote_attempted_in_claim = False + + def defer(category, *, retry_at=None, refund_attempt=False): + try: + db.update_discovery_retry( + claim['id'], claim['lease_owner'], claim['lease_token'], category, + retry_at=retry_at, refund_attempt=refund_attempt, next_page=page, + ) + metrics['discovery_retry_deferred_count'] = 1 + return True + except Exception: + metrics['discovery_retry_error_count'] = 1 + logger.warning('DockerHub discovery retry deferral failed') + return False + + while page <= effective_end: + try: + db.renew_discovery_retry_lease( + claim['id'], claim['lease_owner'], claim['lease_token'], + DOCKERHUB_DISCOVERY_RETRY_LEASE_SECONDS, + ) + except Exception: + metrics['discovery_retry_error_count'] = 1 + logger.warning('DockerHub discovery retry lease renewal failed') + return metrics + try: + _require_discovery_provider_enabled(db) + _, repositories, total_count = _validated_dockerhub_page( + query, page, args, policy=policy, + ) + except DiscoveryPausedError: + defer('provider_cooldown', refund_attempt=True) + return metrics + except DockerHubDiscoveryTransportError as error: + if not getattr(error, 'retryable', True): + hold('invalid_payload') + return metrics + retry_at = getattr(error, 'retry_at', None) + refund_attempt = bool( + not remote_attempted_in_claim + and not getattr(error, 'remote_attempted', True) + and retry_at is not None + ) + defer( + _safe_dockerhub_discovery_category(error), + retry_at=retry_at if refund_attempt else None, + refund_attempt=refund_attempt, + ) + return metrics + except Exception: + logger.warning('DockerHub discovery retry acquisition failed') + defer('request_failed') + return metrics + + remote_attempted_in_claim = True + metrics['discovery_retry_pages_fetched'] += 1 + if work_kind == 'query' and page == 1: + effective_end = min( + effective_end, + max(1, (total_count + int(policy['per_page']) - 1) // int(policy['per_page'])), + ) + complete = not repositories or page >= effective_end + query_page_end = min( + int(policy['pages']), + max(1, (total_count + int(policy['per_page']) - 1) // int(policy['per_page'])), + ) + observation = _dockerhub_discovery_observation( + args, query, page, int(policy['per_page']), + str(claim['policy_sha256']), str(claim['pass_kind']), total_count, + query_complete=not repositories or page >= query_page_end, + ) + if observation is not None: + observation['cycle_id'] = claim.get('source_cycle_id') + observation_kwargs = ( + {'observation': observation} if observation is not None else {} + ) + experiment_authority = getattr( + args, 'docker_depth_experiment_authority', None, + ) + if isinstance(experiment_authority, dict): + observation_kwargs.update({ + 'experiment_authority': experiment_authority, + 'final_cutover': True, + }) + try: + report = db.persist_dockerhub_discovery_page( + source, + query, + repositories, + retry_id=claim['id'], + lease_owner=claim['lease_owner'], + lease_token=claim['lease_token'], + next_page=None if complete else page + 1, + complete=complete, + **observation_kwargs, + ) + if not isinstance(report, dict): + raise RuntimeError('discovery retry page admission was not confirmed') + inserted_count = report.get('inserted_count') + if isinstance(inserted_count, bool) or int(inserted_count) < 0: + raise RuntimeError('discovery retry page admission report is invalid') + progress_key = 'retry_completed_count' if complete else 'retry_progress_count' + if int(report.get(progress_key) or 0) != 1: + raise RuntimeError('discovery retry fence transition was not confirmed') + except DiscoveryPausedError: + raise + except ValueError: + hold('invalid_payload') + return metrics + except DiscoveryRetryLeaseError: + metrics['discovery_retry_error_count'] = 1 + logger.warning('DockerHub discovery retry page admission lost its lease') + return metrics + except Exception: + logger.warning('DockerHub discovery retry page admission failed') + defer('request_failed') + return metrics + metrics['discovery_retry_queued_new_count'] += int(inserted_count) + if complete: + metrics['discovery_retry_completed_count'] = 1 + return metrics + page += 1 + + return metrics + + +def fetch_targets(args, db=None, run_id=None, cycle_id=None, source_name=None): + if args.platform == 'docker' and args.docker_platform_filter_enabled and ( + str(args.docker_platform_os).lower(), str(args.docker_platform_arch).lower() + ) != ('linux', 'amd64'): + raise ValueError('This TruffleHog Docker runner supports platform filtering only for linux/amd64') + if args.mode == 'custom': + return read_custom_targets(args.target_file) + + stop_options = fetch_stop_options(args, db, source_name) + + if args.platform == 'github': + token = args.token or os.getenv('GITHUB_TOKEN') + use_metadata = ( + int(getattr(args, 'max_repo_age_days', 0) or 0) > 0 + or updated_target_rescan_enabled(args) + ) + raise_rate_limit = bool(getattr(args, 'raise_rate_limit', False)) + if args.mode == 'recent': + since = datetime.now() - timedelta(hours=args.recent_hours) + if use_metadata: + return filter_repo_items_by_age(fetch_recent_github_repo_items(args.query, since, token, raise_rate_limit, args.pages, args.per_page, **stop_options), args) + return fetch_recent_github_repos(args.query, since, token, raise_rate_limit, args.pages, args.per_page, **stop_options) + if use_metadata: + return filter_repo_items_by_age(fetch_github_repo_items( + args.query, + args.pages, + args.per_page, + token, + args.sort_by, + args.sort_order, + args.created_filter, + raise_rate_limit, + **stop_options, + ), args) + return fetch_github_repos( + args.query, + args.pages, + args.per_page, + token, + args.sort_by, + args.sort_order, + args.created_filter, + raise_rate_limit, + **stop_options, + ) + + if args.platform == 'github_archive': + targets = fetch_github_archive_repos( + args.archive_hours_back, + args.archive_max_repos_per_cycle, + normalize_archive_event_types(args.archive_event_types), + args.fetch_timeout, + getattr(args, 'gharchive_cache_dir', None), + ) + return filter_github_archive_targets_by_cooldown(targets, args, db) + + if args.platform == 'github_archive_files': + targets = fetch_github_archive_file_targets( + args.archive_hours_back, + args.archive_max_files_per_cycle, + normalize_archive_event_types(args.archive_event_types), + args.fetch_timeout, + getattr(args, 'postman_cache_dir', None), + getattr(args, 'max_artifact_size_mb', 2), + args.token or os.getenv('GITHUB_TOKEN'), + getattr(args, 'archive_max_commit_lookups', 200), + getattr(args, 'gharchive_cache_dir', None), + discovery_max_artifacts=getattr(args, 'postman_discovery_max_artifacts_per_cycle', None), + discovery_max_artifacts_per_page=getattr(args, 'postman_discovery_max_artifacts_per_page', None), + discovery_max_bytes=getattr(args, 'postman_discovery_max_bytes_per_cycle', None), + discovery_max_elapsed_sec=getattr(args, 'postman_discovery_max_elapsed_sec', None), + ) + return targets + + if args.platform == 'gitlab': + token = args.token or os.getenv('GITLAB_TOKEN') + use_metadata = ( + int(getattr(args, 'max_repo_age_days', 0) or 0) > 0 + or updated_target_rescan_enabled(args) + ) + raise_rate_limit = bool(getattr(args, 'raise_rate_limit', False)) + gitlab_options = { + **stop_options, + 'request_attempts': max(1, int(getattr(args, 'gitlab_discovery_request_attempts', 1) or 1)), + 'retry_delay': max(0, int(getattr(args, 'gitlab_discovery_retry_delay', 0) or 0)), + } + if args.mode == 'recent': + since = datetime.now() - timedelta(hours=args.recent_hours) + if use_metadata: + return filter_repo_items_by_age(fetch_recent_gitlab_repo_items(args.query, since, token, raise_rate_limit, args.pages, args.per_page, args.gitlab_visibility, **gitlab_options), args) + return fetch_recent_gitlab_repos(args.query, since, token, raise_rate_limit, args.pages, args.per_page, args.gitlab_visibility, **gitlab_options) + if use_metadata: + return filter_repo_items_by_age(fetch_gitlab_repo_items( + args.query, + args.pages, + args.per_page, + token, + args.gitlab_sort_by, + args.sort_order, + args.gitlab_visibility, + raise_rate_limit=raise_rate_limit, + **gitlab_options, + ), args) + return fetch_gitlab_repos( + args.query, + args.pages, + args.per_page, + token, + args.gitlab_sort_by, + args.sort_order, + args.gitlab_visibility, + raise_rate_limit, + **gitlab_options, + ) + + if args.platform == 'huggingface': + token = args.token or os.getenv('HF_TOKEN') or os.getenv('HUGGINGFACE_TOKEN') + preserve_metadata = True + spaces = fetch_huggingface_spaces( + args.pages, + token, + args.fetch_timeout, + return_metadata=preserve_metadata, + request_attempts=max( + 1, int(getattr(args, 'huggingface_discovery_request_attempts', 1) or 1), + ), + retry_delay=max( + 0, int(getattr(args, 'huggingface_discovery_retry_delay', 0) or 0), + ), + **stop_options, + ) + return [ + discovery for item in spaces + if (discovery := repo_item_discovery(item, args)) is not None + and not huggingface_discovery_target_is_restricted(discovery) + ] + + if args.platform == 'docker': + queries = read_queries(args) + if not queries: + print('Docker Hub search requires --query. Empty query returns no results from the Docker Hub API.') + return [] + + images = [] + seen = set() + queue_resolver = bool( + db and getattr(db, 'conn', None) + and getattr(db.conn, 'is_postgres', False) + and hasattr(db, 'claim_docker_resolutions') + ) + for query in queries: + print(f"Fetching Docker Hub targets for query: {query}") + if args.mode == 'recent': + since = datetime.now() - timedelta(days=args.recent_days) + query_images = fetch_recent_dockerhub_images( + query, since, args.per_page, args.pages, + args.docker_platform_filter_enabled, args.docker_platform_os, + args.docker_platform_arch, args.docker_platform_candidate_tags, + docker_images_per_repository_limit( + getattr(args, 'docker_images_per_repository', 1) + ), + not queue_resolver, + ) + else: + query_images = fetch_dockerhub_images( + query, + args.pages, + args.per_page, + args.docker_sort_by, + args.sort_order, + args.fetch_workers, + args.fetch_timeout, + not queue_resolver, + args.tag_fetch_workers, + args.tag_retry_count, + args.tag_retry_delay, + args.docker_platform_filter_enabled, + args.docker_platform_os, + args.docker_platform_arch, + args.docker_platform_candidate_tags, + docker_images_per_repository_limit( + getattr(args, 'docker_images_per_repository', 1) + ), + ) + for image in query_images: + if image not in seen: + images.append(image) + seen.add(image) + return images + + if args.platform == 'npm': + queries = read_queries(args) + if not queries: + print('npm search requires --query') + return [] + targets = [] + seen = set() + for query in queries: + print(f"Fetching npm targets for query: {query}") + query_targets = fetch_npm_packages( + query, + args.pages, + args.per_page, + args.max_version_age_days, + args.fetch_timeout, + args.versions_per_package, + make_repo_candidate_callback(db, run_id, cycle_id, query), + ) + for target in query_targets: + normalized = normalize_target(target, 'npm') + if normalized not in seen: + targets.append(target) + seen.add(normalized) + return targets + + if args.platform == 'pypi': + queries = read_queries(args) + if not queries: + print('PyPI search requires --query') + return [] + targets = [] + seen = set() + for query in queries: + print(f"Fetching PyPI targets for query: {query}") + query_targets = fetch_pypi_packages( + query, + args.pages, + args.per_page, + args.max_version_age_days, + args.fetch_timeout, + args.versions_per_package, + make_repo_candidate_callback(db, run_id, cycle_id, query), + ) + for target in query_targets: + normalized = normalize_target(target, 'pypi') + if normalized not in seen: + targets.append(target) + seen.add(normalized) + return targets + + if args.platform == 'package_git': + queries = read_queries(args) + if not queries: + print('package_git search requires --query') + return [] + cached_targets = fetch_package_git_targets_from_db(args, db) + refresh_registry = bool(getattr(args, 'refresh_registry', False)) + if cached_targets and not refresh_registry: + print(f'Using {len(cached_targets)} package git repo targets from scanner.db cache') + return cached_targets + if cached_targets: + print(f'Using {len(cached_targets)} cached package git repo target(s) and refreshing registry discovery') + else: + print('No package git repo candidates found in scanner.db cache; falling back to direct registry discovery') + package_sources = [item.strip().lower() for item in str(getattr(args, 'package_sources', 'npm,pypi') or '').split(',') if item.strip()] + targets = [] + seen = set() + for target in cached_targets: + normalized = normalize_target(target, 'package_git') + if normalized not in seen: + targets.append(target) + seen.add(normalized) + for query in queries: + print(f"Fetching package git repo targets for query: {query}") + query_targets = [] + if 'npm' in package_sources: + query_targets.extend(fetch_npm_package_git_repos( + query, + args.pages, + args.per_page, + args.max_version_age_days, + args.fetch_timeout, + args.versions_per_package, + )) + if 'pypi' in package_sources: + query_targets.extend(fetch_pypi_package_git_repos( + query, + args.pages, + args.per_page, + args.max_version_age_days, + args.fetch_timeout, + args.versions_per_package, + )) + for target in query_targets: + normalized = normalize_target(target, 'package_git') + if normalized not in seen: + targets.append(target) + seen.add(normalized) + return targets + + if args.platform == 'postman': + queries = read_queries(args) + if not queries: + print('Postman GitHub code search requires --query') + return [] + targets = [] + seen = set() + for query in queries: + print(f"Fetching Postman targets from GitHub code search for query: {query}") + query_targets = fetch_github_postman_targets( + query, + args.pages, + args.per_page, + token_entries=getattr(args, 'github_tokens', None), + token=args.token or os.getenv('GITHUB_TOKEN'), + search_kinds=getattr(args, 'search_kinds', 'collection,environment'), + cache_dir=getattr(args, 'postman_cache_dir', None), + max_file_age_days=getattr(args, 'max_file_age_days', 365), + max_artifact_size_mb=getattr(args, 'max_artifact_size_mb', 20), + request_timeout=getattr(args, 'fetch_timeout', 20), + code_search_rpm_per_token=getattr(args, 'github_code_search_rpm', 8), + all_tokens_cooldown=getattr(args, 'all_tokens_cooldown', 1800), + auth_status=getattr(args, 'github_auth_status', None), + discovery_max_artifacts=getattr(args, 'postman_discovery_max_artifacts_per_cycle', None), + discovery_max_artifacts_per_page=getattr(args, 'postman_discovery_max_artifacts_per_page', None), + discovery_max_bytes=getattr(args, 'postman_discovery_max_bytes_per_cycle', None), + discovery_max_elapsed_sec=getattr(args, 'postman_discovery_max_elapsed_sec', None), + **stop_options, + ) + for target in query_targets: + normalized = normalize_target(target, 'postman') + if normalized not in seen: + targets.append(target) + seen.add(normalized) + return targets + + if args.platform == 'github_gists': + token = args.token or os.getenv('GITHUB_TOKEN') + return fetch_github_gist_targets( + args.pages, + args.per_page, + getattr(args, 'gist_since', '') or None, + token, + getattr(args, 'postman_cache_dir', None), + getattr(args, 'max_artifact_size_mb', 2), + getattr(args, 'fetch_timeout', 20), + discovery_max_artifacts=getattr(args, 'postman_discovery_max_artifacts_per_cycle', None), + discovery_max_artifacts_per_page=getattr(args, 'postman_discovery_max_artifacts_per_page', None), + discovery_max_bytes=getattr(args, 'postman_discovery_max_bytes_per_cycle', None), + discovery_max_elapsed_sec=getattr(args, 'postman_discovery_max_elapsed_sec', None), + **stop_options, + ) + + if args.platform == 'github_actions': + return fetch_ci_repo_targets_from_db(args, db, 'github') + + if args.platform == 'gitlab_ci': + return fetch_ci_repo_targets_from_db(args, db, 'gitlab') + + raise ValueError(f'Unsupported platform: {args.platform}') + + +def known_targets_for_args(args, db=None, source_name=None): + postgres_configured = bool( + (db and getattr(db, 'postgres_required', False)) + or str(getattr(args, 'database_url', '') or '').lower().startswith(('postgresql://', 'postgres://')) + ) + if postgres_configured: + return set() + todo_file, checked_file = queue_files_for_args(args) + known = set() + paths = (todo_file,) if args.platform == 'github_archive' else (todo_file, checked_file) + for path in paths: + for target in load_set_from_file(path): + known.add(normalize_target(target, args.platform)) + return known + + +def fetch_stop_options(args, db=None, source_name=None): + if updated_target_rescan_enabled(args): + return {} + if not getattr(args, 'stop_on_seen_pages', False): + return {} + common = { + 'normalize_target': lambda target: normalize_target(target, args.platform), + 'stop_on_seen_pages': True, + 'seen_page_threshold': int(getattr(args, 'seen_page_threshold', 2) or 2), + 'min_pages_before_stop': int(getattr(args, 'min_pages_before_stop', 1) or 1), + } + if db and getattr(db, 'conn', None) and getattr(db.conn, 'is_postgres', False): + lookup = getattr(db, 'known_target_normalizations_for', None) + if not lookup: + return {} + source_key = source_name or args.platform + common['known_target_lookup'] = lambda targets: lookup(source_key, args.platform, targets) + return common + known = known_targets_for_args(args, db, source_name) + if not known: + return {} + common['known_targets'] = known + return common + + +def get_platform_token(args): + if args.platform in ('github', 'github_archive'): + return args.token or os.getenv('GITHUB_TOKEN') + if args.platform == 'gitlab': + return args.token or os.getenv('GITLAB_TOKEN') + if args.platform == 'package_git': + return args.token + if args.platform == 'huggingface': + return args.token or os.getenv('HF_TOKEN') or os.getenv('HUGGINGFACE_TOKEN') + if args.platform == 'postman': + return args.token or os.getenv('GITHUB_TOKEN') + if args.platform == 'github_gists': + return args.token or os.getenv('GITHUB_TOKEN') + if args.platform == 'github_actions': + return args.token or os.getenv('GITHUB_TOKEN') + if args.platform == 'gitlab_ci': + return args.token or os.getenv('GITLAB_TOKEN') + return None + + +def queue_files_for_args(args): + queue_dir = getattr(args, 'queue_dir', None) or args.save_dir + return ( + os.path.join(queue_dir, f'todo_{args.platform}.txt'), + os.path.join(queue_dir, f'checked_{args.platform}.txt'), + ) + + +def github_archive_recently_scanned(db, target, cooldown_hours): + if not db or not getattr(db, 'conn', None) or int(cooldown_hours or 0) <= 0: + return False + normalized = normalize_target(target, 'github_archive') + normalized_values = [normalized] + if '#' in normalized: + normalized_values.append(normalized.split('#', 1)[0]) + cutoff = (datetime.now() - timedelta(hours=int(cooldown_hours or 0))).isoformat(timespec='seconds') + row = db.conn.execute(''' + SELECT 1 + FROM target_scans + WHERE normalized_target IN ({}) + AND source = 'github_archive' + AND COALESCE(ended_at, started_at, created_at) >= ? + LIMIT 1 + '''.format(','.join('?' for _ in normalized_values)), [*normalized_values, cutoff]).fetchone() + commit_if_postgres(db) + return bool(row) + + +def filter_github_archive_targets_by_cooldown(targets, args, db): + cooldown_hours = int(getattr(args, 'archive_rescan_cooldown_hours', 0) or 0) + if cooldown_hours <= 0 or not db or not getattr(db, 'conn', None): + return targets + output = [] + skipped = 0 + for target in targets: + if github_archive_recently_scanned(db, target, cooldown_hours): + skipped += 1 + continue + output.append(target) + print(f'GHArchive cooldown filter: kept={len(output)} skipped_recent={skipped} cooldown_hours={cooldown_hours}') + return output + + +def queue_counts_for_args(args): + todo_file, checked_file = queue_files_for_args(args) + return queue_counts(todo_file, checked_file) + + +def refund_claim_setup(db, spool, reservation_id, claims, reason): + claims = [claim for claim in (claims or []) if claim is not None] + if not claims: + if reservation_id: + if spool.release_reservation(reservation_id) is False: + raise RuntimeError('unable to release exact result-spool reservation') + return True + if hasattr(db, 'refund_target_claims'): + refunded = db.refund_target_claims(claims, reason) + else: + refunded = all( + db.refund_target_claim( + claim.get('id') if isinstance(claim, dict) else claim['id'], + claim.get('lease_token') if isinstance(claim, dict) else claim['lease_token'], + reason, + ) + for claim in claims + ) + if not refunded: + raise RuntimeError('unable to atomically refund fenced target claims after infrastructure setup failure') + if reservation_id: + if spool.release_reservation(reservation_id) is False: + raise RuntimeError('claims were refunded but the exact result-spool reservation was not released') + return True + + +class PostClaimRefundGuard: + def __init__( + self, db, spool, reservation_id, claims, active_tokens, active_lock, + dispatch_leases=None, + ): + self.db = db + self.spool = spool + self.reservation_id = reservation_id + self.claims = [dict(claim) for claim in claims] + self.active_tokens = active_tokens + self.active_lock = active_lock + self.dispatch_leases = list(dispatch_leases or []) + + def release_unstarted_dispatch_leases(self): + for lease in self.dispatch_leases: + if lease.heartbeat_thread is None: + lease.release() + + def refund_undurable(self, reason): + if not self.reservation_id: + return True + with self.active_lock: + tokens = set(self.active_tokens) + claims = [ + claim for claim in self.claims + if claim.get('lease_token') in tokens + ] + if claims: + refund_claim_setup( + self.db, self.spool, self.reservation_id, claims, reason, + ) + else: + release = getattr(self.spool, 'release_reservation', None) + if release: + try: + release(self.reservation_id) + except Exception: + logger.exception('Unable to release consumed result-spool reservation') + self.release_unstarted_dispatch_leases() + self.reservation_id = None + return True + + def call(self, reason, function, *args, **kwargs): + try: + return function(*args, **kwargs) + except Exception as exc: + self.refund_undurable(f'{reason}: {exc}') + raise + + +def _claim_value(claim, name): + if isinstance(claim, dict): + return claim.get(name) + return claim[name] + + +def validate_recovered_claim_batch(db, rows, reservation_id, lease_owner, claim_limit): + rows = list(rows or []) + identities = [] + targets = set() + for row in rows: + queue_id = _claim_value(row, 'id') + lease_token = _claim_value(row, 'lease_token') + claim_batch = _claim_value(row, 'claim_batch') + owner = _claim_value(row, 'lease_owner') + target = str(_claim_value(row, 'target') or '') + if ( + queue_id is None or not lease_token or not target + or str(claim_batch or '') != str(reservation_id) + or str(owner or '') != str(lease_owner) + ): + raise RuntimeError('claim batch recovery returned inconsistent fenced row metadata') + identity = (int(queue_id), str(lease_token)) + if identity in identities or target in targets: + raise RuntimeError('claim batch recovery returned duplicate fenced rows') + identities.append(identity) + targets.add(target) + if len(rows) > int(claim_limit): + raise RuntimeError('claim batch recovery exceeded its reserved row limit') + expectation_loader = getattr(db, 'claim_recovery_expectation', None) + expected = expectation_loader(reservation_id, lease_owner) if expectation_loader else None + if expected is not None: + expected_identities = { + (int(_claim_value(claim, 'id')), str(_claim_value(claim, 'lease_token'))) + for claim in expected + } + if set(identities) != expected_identities: + raise RuntimeError('claim batch recovery did not return the exact pre-commit fenced row set') + return rows + + +def enqueue_discovered_targets( + args, fetched_targets, db, run_id, cycle_id, source_name=None, + partial_metrics=None, +): + source_key = source_name or args.platform + if not db or not getattr(db, 'conn', None) or not getattr(db.conn, 'is_postgres', False): + raise RuntimeError('Discovery admission requires PostgreSQL') + db.require_runtime_safety_schema() + if not run_id or not cycle_id: + raise RuntimeError('Postgres discovery admission requires valid run_id and cycle_id') + + discovery_by_normalized = {} + for item in fetched_targets or []: + if ( + args.platform == 'huggingface' + and huggingface_discovery_target_is_restricted(item) + ): + continue + target = item.get('target') or item.get('url') if isinstance(item, dict) else item + normalized = normalize_target(target, args.platform) + if not normalized: + continue + remote_value = item.get('remote_modified_at') if isinstance(item, dict) else None + remote_time = parse_datetime(remote_value) + existing = discovery_by_normalized.get(normalized) + if existing is None or ( + remote_time + and (existing['_remote_time'] is None or remote_time > existing['_remote_time']) + ): + record = { + 'target': target, + 'remote_modified_at': ( + remote_time.isoformat(timespec='seconds') if remote_time else None + ), + '_remote_time': remote_time, + } + discovery_by_normalized[normalized] = record + discovery_records = [ + {key: value for key, value in item.items() if key != '_remote_time'} + for item in discovery_by_normalized.values() + ] + discoveries = [item['target'] for item in discovery_records] + + eligible = discoveries + unresolved = [] + if args.platform == 'docker' and discoveries: + eligible = [] + visible = [] + for target in discoveries: + try: + eligible.append(parse_docker_target(target)['target']) + except (TypeError, ValueError): + visible.append(target) + eligible_seen = set() + deduped_eligible = [] + for target in eligible: + normalized = normalize_target(target, args.platform) + if normalized in eligible_seen: + continue + eligible_seen.add(normalized) + deduped_eligible.append(target) + eligible = deduped_eligible + eligible_normalized = { + normalize_target(target, args.platform) for target in eligible + } + unresolved_seen = set() + for target in visible: + normalized = normalize_target(target, args.platform) + if normalized in eligible_normalized or normalized in unresolved_seen: + continue + unresolved.append(target) + unresolved_seen.add(normalized) + + projected_targets = [*eligible, *unresolved] + expected = len(projected_targets) + queued_updated_count = 0 + if updated_target_rescan_enabled(args): + observer = getattr(db, 'observe_discovered_targets', None) + if not observer: + raise RuntimeError( + 'Updated-target rescan requires atomic Postgres discovery observation' + ) + observation = observer( + source_key, + args.platform, + args.query, + discovery_records, + rescan_limit=max( + 0, int(getattr(args, 'updated_target_rescan_max_per_cycle', 0) or 0), + ), + cooldown_seconds=max( + 0, + int(getattr(args, 'updated_target_rescan_cooldown_hours', 0) or 0) + * 3600, + ), + ) + queued_new_count = int(observation.get('queued_new_count', 0) or 0) + queued_updated_count = int(observation.get('queued_updated_count', 0) or 0) + if int(observation.get('attempted_count', -1)) != expected: + raise RuntimeError( + f'Unable to observe all Postgres discoveries for {source_key}: ' + f'{observation.get("attempted_count")}/{expected}' + ) + else: + if hasattr(db, 'known_target_normalizations_for'): + known_before = db.known_target_normalizations_for( + source_key, args.platform, projected_targets, + ) + else: + known_before = set() + queued_new_count = sum( + 1 for target in projected_targets + if normalize_target(target, args.platform) not in known_before + ) + enqueued = db.enqueue_targets( + source_key, + args.platform, + args.query, + eligible, + requeue_done=(args.platform == 'github_archive'), + unresolved_targets=unresolved, + discovery_admission=True, + ) + if enqueued != expected: + raise RuntimeError( + f'Unable to enqueue all Postgres discoveries for {source_key}: ' + f'{enqueued}/{expected}' + ) + + discovery_info = { + 'fetched_count': len(fetched_targets or []), + 'queued_new_count': queued_new_count, + 'queued_updated_count': queued_updated_count, + } + if isinstance(partial_metrics, dict): + partial_metrics.update(discovery_info) + return projected_targets, discovery_info + + +def prepare_targets( + args, + fetched_targets, + db=None, + run_id=None, + cycle_id=None, + source_name=None, + spool=None, + claim_limit_override=None, + dispatch_leases=None, + enqueue_only=False, + partial_metrics=None, + experiment_resolver_already_run=False, +): + todo_file, checked_file = queue_files_for_args(args) + source_key = source_name or args.platform + if db and getattr(db, 'postgres_required', False) and not getattr(db, 'conn', None): + raise RuntimeError('Required Postgres connection is unavailable') + postgres_db = bool(db and getattr(db, 'conn', None) and getattr(db.conn, 'is_postgres', False)) + project_files = not postgres_db or bool(getattr(args, 'sync_file_queues', True)) + if postgres_db: + projected_targets, admission = enqueue_discovered_targets( + args, fetched_targets, db, run_id, cycle_id, source_key, + partial_metrics=partial_metrics, + ) + queued_new_count = admission['queued_new_count'] + queued_updated_count = admission['queued_updated_count'] + + projection_error = '' + if project_files and projected_targets: + try: + os.makedirs(getattr(args, 'queue_dir', None) or args.save_dir, exist_ok=True) + with projection_file_lock(todo_file): + existing_projection = { + normalize_target(target, args.platform) for target in load_set_from_file(todo_file) + } + projection_rows = [ + target for target in projected_targets + if normalize_target(target, args.platform) not in existing_projection + ] + if projection_rows: + _append_lines_unlocked(todo_file, projection_rows) + except Exception as exc: + projection_error = str(exc) + safe_print(f'Warning: Postgres targets committed but todo projection failed: {exc}') + + if ( + args.platform == 'docker' + and not bool(getattr(args, 'docker_depth_collection_only', False)) + and hasattr(db, 'claim_docker_resolutions') + ): + resolve_due_docker_queue_targets_if_scan_queue_empty( + db, source_key, args, + include_experiment=not experiment_resolver_already_run, + ) + + if enqueue_only: + queue_info = queue_counts(todo_file, checked_file) if project_files else { + 'todo_count': 0, + 'checked_count': 0, + 'todo_file': None, + 'checked_file': None, + } + queue_info['db_queue'] = db.target_queue_counts(source_key) + queue_info.update({ + 'fetched_count': len(fetched_targets or []), + 'queued_new_count': queued_new_count, + 'queued_updated_count': queued_updated_count, + 'scan_requested_count': 0, + 'lease_owner': None, + 'lease_tokens': [], + 'lease_seconds': 0, + 'queue_claims': {}, + 'spool_reservation_id': None, + 'scan_slot_leases': [], + 'projection_error': projection_error, + }) + if queued_new_count: + print(f'Committed {queued_new_count} new Postgres target(s).') + if queued_updated_count: + print(f'Committed {queued_updated_count} updated Postgres target rescan(s).') + return [], todo_file, checked_file, queue_info + + worker_count = max(1, int(getattr(args, 'workers', 1) or 1)) + configured_batch = int(getattr(args, 'target_claim_batch_size', 0) or 0) + claim_limit = ( + max(1, int(claim_limit_override)) + if claim_limit_override is not None + else configured_batch if configured_batch > 0 else worker_count + ) + if args.max_targets: + claim_limit = min(claim_limit, int(args.max_targets)) + target_timeout = max(60, int(getattr(args, 'timeout', 0) or 0)) + lease_seconds = max(target_timeout + 900, 1800) + lease_owner = f'{source_key}:{os.getpid()}:{cycle_id or "cycle"}' + if spool is None: + spool = result_spool_for_args(args) + reservation_id = reserve_result_spool_claims( + spool, + db, + lease_owner, + claim_limit, + lease_seconds, + stop_event=getattr(args, 'result_spool_stop_event', None), + wait_seconds=float(getattr(args, 'result_spool_wait_sec', 1.0) or 1.0), + diagnostic_interval=float(getattr(args, 'result_spool_diagnostic_interval_sec', 30.0) or 30.0), + ) + claim_error = None + claim_rows_validated = False + try: + claimed_rows = db.claim_targets( + source_key, args.platform, claim_limit, lease_owner, lease_seconds, + max_attempts=int(getattr(args, 'target_retry_max_attempts', 3) or 3), + return_rows=True, claim_batch=reservation_id, + ) + except Exception as exc: + claimed_rows = None + claim_error = exc + if claimed_rows is None: + original_error = claim_error or RuntimeError( + getattr(db, 'last_error', '') or f'claim outcome is ambiguous for {source_key}' + ) + try: + claimed_rows = db.recover_claim_batch(reservation_id, lease_owner) + except Exception as recovery_error: + raise RuntimeError( + f'Unable to recover ambiguous claim batch for {source_key}; ' + f'original claim error: {original_error}; recovery error: {recovery_error}' + ) from original_error + claimed_rows = validate_recovered_claim_batch( + db, claimed_rows, reservation_id, lease_owner, claim_limit, + ) + claim_rows_validated = True + if not claimed_rows: + try: + released = spool.release_reservation(reservation_id) + if released is not True: + raise RuntimeError('exact result-spool reservation release was not confirmed') + except Exception as release_error: + raise RuntimeError( + f'FATAL durability error after zero-row claim recovery for {source_key}; ' + f'original claim error: {original_error}; reservation rollback error: {release_error}' + ) from original_error + raise original_error + if not claim_rows_validated: + claimed_rows = validate_recovered_claim_batch( + db, claimed_rows, reservation_id, lease_owner, claim_limit, + ) + setup_tokens = { + str(_claim_value(row, 'lease_token')) for row in claimed_rows + if _claim_value(row, 'lease_token') + } + setup_guard = PostClaimRefundGuard( + db, spool, reservation_id, claimed_rows, setup_tokens, threading.Lock(), + dispatch_leases=dispatch_leases, + ) + try: + reservation_id = spool.bind_claims(reservation_id, claimed_rows) + targets_to_scan = [row['target'] for row in claimed_rows] + leases = list(dispatch_leases or []) + scan_slot_leases = leases[:len(targets_to_scan)] + for extra_lease in leases[len(targets_to_scan):]: + extra_lease.release() + queue_claims = { + str(row['target']): { + 'id': row['id'], + 'attempts': int(row['attempts'] or 0), + 'lease_owner': row['lease_owner'], + 'lease_token': row['lease_token'], + 'claim_batch': row['claim_batch'], + } + for row in claimed_rows + } + queue_info = queue_counts(todo_file, checked_file) if project_files else { + 'todo_count': 0, + 'checked_count': 0, + 'todo_file': None, + 'checked_file': None, + } + queue_info['db_queue'] = db.target_queue_counts(source_key) + queue_info.update({ + 'fetched_count': len(fetched_targets or []), + 'queued_new_count': queued_new_count, + 'queued_updated_count': queued_updated_count, + 'scan_requested_count': len(targets_to_scan), + 'lease_owner': lease_owner, + 'lease_tokens': [row['lease_token'] for row in claimed_rows], + 'lease_seconds': lease_seconds, + 'queue_claims': queue_claims, + 'spool_reservation_id': reservation_id, + 'scan_slot_leases': scan_slot_leases, + 'projection_error': projection_error, + }) + except Exception as exc: + setup_guard.refund_undurable(f'infrastructure claim setup failure: {exc}') + raise + if queued_new_count: + print(f'Committed {queued_new_count} new Postgres target(s).') + if queued_updated_count: + print(f'Committed {queued_updated_count} updated Postgres target rescan(s).') + return targets_to_scan, todo_file, checked_file, queue_info + + os.makedirs(args.save_dir, exist_ok=True) + os.makedirs(getattr(args, 'queue_dir', None) or args.save_dir, exist_ok=True) + with projection_file_lock(todo_file): + checked = load_set_from_file(checked_file) + todo = load_set_from_file(todo_file) + checked_normalized = set() if args.platform == 'github_archive' else {normalize_target(target, args.platform) for target in checked} + todo_normalized = {normalize_target(target, args.platform) for target in todo} + new_targets = [] + for item in fetched_targets: + target = item.get('target') or item.get('url') if isinstance(item, dict) else item + normalized = normalize_target(target, args.platform) + if normalized not in checked_normalized and normalized not in todo_normalized: + new_targets.append(target) + todo.add(target) + todo_normalized.add(normalized) + if new_targets: + _append_lines_unlocked(todo_file, new_targets) + print(f'Queued {len(new_targets)} new targets.') + targets_to_scan = [ + target for target in sorted(todo) + if normalize_target(target, args.platform) not in checked_normalized + ] + if args.platform == 'docker': + targets_to_scan, todo_entries = resolve_docker_targets_for_scan(targets_to_scan, args) + _write_lines_unlocked(todo_file, sorted(set(todo_entries))) + queue_info = queue_counts(todo_file, checked_file) + queue_info.update({ + 'fetched_count': len(fetched_targets), + 'queued_new_count': len(new_targets), + 'queued_updated_count': 0, + 'scan_requested_count': len(targets_to_scan), + 'lease_owner': None, + 'lease_tokens': [], + 'lease_seconds': 0, + 'queue_claims': {}, + 'spool_reservation_id': None, + 'scan_slot_leases': [], + }) + return targets_to_scan, todo_file, checked_file, queue_info + + +def resolve_docker_targets_for_scan(targets, args, resolve_all=False): + def has_tag(target): + text = str(target or '').strip() + return '@' in text or ':' in text.rsplit('/', 1)[-1] + + tagged_targets = [target for target in targets if has_tag(target)] + bare_targets = [target for target in targets if not has_tag(target)] + + if not bare_targets: + return tagged_targets, tagged_targets + + resolve_limit = 0 if resolve_all else int(getattr(args, 'tag_resolve_limit', 100) or 0) + if resolve_limit > 0: + bare_to_resolve = bare_targets[:resolve_limit] + bare_to_keep = bare_targets[resolve_limit:] + else: + bare_to_resolve = bare_targets + bare_to_keep = [] + + print( + f'Resolving tags for {len(bare_to_resolve)} queued Docker repositories ' + f'({len(bare_to_keep)} deferred)...' + ) + + import concurrent.futures + max_workers = max(1, min(args.tag_fetch_workers, len(bare_to_resolve))) + resolved_targets = list(tagged_targets) + unresolved_targets = [] + with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor: + futures = { + executor.submit( + fetch_dockerhub_tags, + target, + None, + docker_images_per_repository_limit( + getattr(args, 'docker_images_per_repository', 1) + ), + args.tag_retry_count, + args.tag_retry_delay, + args.docker_platform_filter_enabled, + args.docker_platform_os, + args.docker_platform_arch, + args.docker_platform_candidate_tags, + True, + ): target + for target in bare_to_resolve + } + for future in concurrent.futures.as_completed(futures): + target = futures[future] + tags, status = future.result() + if tags: + resolved_targets.extend(tags) + if status != 'ok': + print(f'Preserving unresolved Docker repository {target}: tag resolution status={status}') + unresolved_targets.append(target) + + todo_entries = resolved_targets + unresolved_targets + bare_to_keep + return resolved_targets, todo_entries + + +def resolve_due_docker_queue_targets(db, source, args): + limit = max(1, min(100, int(getattr(args, 'tag_resolve_limit', 100) or 100))) + owner = f'docker-resolver:{source}:{os.getpid()}:{threading.get_ident()}' + periodic_limit = max( + 0, min(1, int(getattr(args, 'docker_repository_refresh_max_per_cycle', 0) or 0)), + ) + refresh_interval = max( + 0, int(getattr(args, 'docker_repository_refresh_interval_sec', 0) or 0), + ) + allowed_queries = getattr(args, 'configured_queries', None) + processed = 0 + stopped = False + + def process(row): + nonlocal processed + paused_error = None + try: + _require_discovery_provider_enabled(db) + outcome = fetch_dockerhub_tags( + row['target'], None, + docker_images_per_repository_limit( + getattr(args, 'docker_images_per_repository', 1) + ), + int(getattr(args, 'tag_retry_count', 2) or 2), + int(getattr(args, 'tag_retry_delay', 5) or 5), + bool(getattr(args, 'docker_platform_filter_enabled', True)), + str(getattr(args, 'docker_platform_os', 'linux')), + str(getattr(args, 'docker_platform_arch', 'amd64')), + int(getattr(args, 'docker_platform_candidate_tags', 20) or 20), + return_outcome=True, + ) + except DiscoveryPausedError as exc: + paused_error = exc + outcome = SimpleNamespace( + tags=(), status='global_cooldown', remote_attempted=False, + retry_at=None, error='Discovery paused by runtime control.', + ) + except Exception as exc: + outcome = SimpleNamespace( + tags=(), status='unknown', remote_attempted=True, + retry_at=None, error=str(exc), + ) + resolver_status = outcome.status + resolver_retry_at = outcome.retry_at + resolver_error = outcome.error or f'Docker tag resolution status={outcome.status}' + resolver_remote_attempted = outcome.remote_attempted + if not db.finish_docker_resolution( + source, row['id'], row['resolver_token'], list(outcome.tags), + resolver_error, + complete=docker_tag_resolution_is_conclusive(outcome.status), + retry_at=resolver_retry_at, + claim_attempt_consumed=( + False if paused_error is not None else not ( + resolver_status == 'global_cooldown' + and not resolver_remote_attempted + and bool(resolver_retry_at) + ) + ), + refresh_interval_sec=refresh_interval, + ): + raise RuntimeError(f'Unable to persist Docker resolver outcome for queue row {row["id"]}') + processed += 1 + if paused_error is not None: + raise paused_error + return bool( + resolver_retry_at + or resolver_status in ( + 'global_cooldown', 'rate_limited', 'auth_failed', + 'remote_transient', + ) + ) + + retry_budget = limit - 1 if periodic_limit and limit > 1 else limit + for _ in range(retry_budget): + rows = db.claim_docker_resolutions( + source, 1, owner, lease_seconds=300, + allowed_queries=allowed_queries, + ) + if not rows: + break + if process(rows[0]): + stopped = True + break + + if not stopped and periodic_limit: + rows = db.claim_docker_resolutions( + source, 1, owner, lease_seconds=300, + periodic_limit=1, periodic_only=True, + allowed_queries=allowed_queries, + ) + if rows: + stopped = process(rows[0]) + + while not stopped and processed < limit: + rows = db.claim_docker_resolutions( + source, 1, owner, lease_seconds=300, + allowed_queries=allowed_queries, + ) + if not rows: + break + stopped = process(rows[0]) + if processed: + print(f'Processed {processed} due PostgreSQL Docker resolver row(s).') + return processed + + +def resolve_due_docker_experiment_targets(db, source, args): + authority = getattr(args, 'docker_depth_experiment_authority', None) + if ( + source != 'dockerhub' + or not isinstance(authority, dict) + or authority.get('enabled') is not True + or bool(getattr(args, 'sync_file_queues', True)) + or not db + or not getattr(db, 'conn', None) + or not getattr(db.conn, 'is_postgres', False) + or not callable(getattr(db, 'claim_docker_depth_experiment_resolutions', None)) + or not callable(getattr(db, 'renew_docker_depth_experiment_resolution', None)) + or not callable(getattr(db, 'finish_docker_depth_experiment_resolution', None)) + ): + return 0, False + limit = max(1, min(100, int(getattr(args, 'tag_resolve_limit', 100) or 100))) + owner = f'docker-depth-resolver:{source}:{os.getpid()}:{threading.get_ident()}' + processed = 0 + stopped = False + for _ in range(limit): + rows = db.claim_docker_depth_experiment_resolutions( + source, 1, owner, lease_seconds=300, authority=authority, + final_cutover=True, + ) + if not rows: + break + row = rows[0] + paused_error = None + + def renew(): + _require_discovery_provider_enabled(db) + renewal = db.renew_docker_depth_experiment_resolution( + source, row['id'], row['resolver_generation'], + row['resolver_token'], resolver_owner=row['resolver_owner'], + lease_seconds=300, authority=authority, final_cutover=True, + ) + return bool(renewal.get('renewed')) + + try: + if not renew(): + raise DockerResolverLeaseLostError( + 'Docker depth resolver lease is no longer owned' + ) + outcome = fetch_dockerhub_tags( + row['target'], None, int(row['selection_limit']), + int(getattr(args, 'tag_retry_count', 2) or 2), + int(getattr(args, 'tag_retry_delay', 5) or 5), + bool(getattr(args, 'docker_platform_filter_enabled', True)), + str(getattr(args, 'docker_platform_os', 'linux')), + str(getattr(args, 'docker_platform_arch', 'amd64')), + int(getattr(args, 'docker_platform_candidate_tags', 20) or 20), + return_outcome=True, fresh_graph_evidence=True, + lease_renewal_callback=renew, + selector_version=authority['selector_version'], + ) + if not renew(): + raise DockerResolverLeaseLostError( + 'Docker depth resolver lease expired during remote resolution' + ) + except DockerResolverLeaseLostError: + stopped = True + break + except DiscoveryPausedError as exc: + paused_error = exc + outcome = SimpleNamespace( + tags=(), status='global_cooldown', remote_attempted=False, + retry_at=None, error='Discovery paused by runtime control.', + selection_records=(), selector_version='', selector_hash='', + candidate_distinct_graph_count=0, fresh_graph_evidence=False, + cache_bypassed=True, + ) + except Exception as exc: + outcome = SimpleNamespace( + tags=(), status='unknown', remote_attempted=True, + retry_at=None, error=str(exc), selection_records=(), + selector_version='', selector_hash='', + candidate_distinct_graph_count=0, + fresh_graph_evidence=False, cache_bypassed=True, + ) + result = db.finish_docker_depth_experiment_resolution( + source, row['id'], row['resolver_generation'], row['resolver_token'], + outcome, outcome.error or f'Docker tag resolution status={outcome.status}', + complete=docker_tag_resolution_is_conclusive(outcome.status), + resolver_owner=row['resolver_owner'], authority=authority, + retry_at=outcome.retry_at, + claim_attempt_consumed=( + False if paused_error is not None else not ( + outcome.status == 'global_cooldown' + and not outcome.remote_attempted + and bool(outcome.retry_at) + ) + ), + final_cutover=True, + ) + if result.get('committed'): + processed += 1 + if result.get('status') in ('stale', 'unavailable', 'invalid', 'error'): + raise RuntimeError( + f'Unable to persist Docker experiment resolver outcome for member {row["id"]}: ' + f'{result.get("status")}' + ) + if paused_error is not None: + raise paused_error + stopped = bool( + result.get('status') in ('conflict_retry', 'held') + or outcome.retry_at + or outcome.status in ('global_cooldown', 'rate_limited', 'auth_failed') + ) + if stopped: + break + if processed: + print(f'Processed {processed} fenced Docker depth experiment resolver row(s).') + return processed, stopped + + +def resolve_due_docker_queue_targets_if_scan_queue_empty( + db, source, args, *, include_experiment=True, +): + experiment_processed, experiment_stopped = (0, False) + if include_experiment: + experiment_processed, experiment_stopped = resolve_due_docker_experiment_targets( + db, source, args, + ) + if experiment_stopped: + return experiment_processed + backlog_probe = getattr(db, 'has_claimable_targets_v2', None) + if callable(backlog_probe): + has_scan_targets = backlog_probe( + source, + 'docker', + max_attempts=max(0, int(getattr(args, 'target_retry_max_attempts', 3) or 3)), + ) + if getattr(db, 'last_error', None): + raise RuntimeError( + f'Unable to inspect Docker scan backlog before resolver work: {db.last_error}' + ) + if has_scan_targets: + return experiment_processed + return experiment_processed + resolve_due_docker_queue_targets(db, source, args) + + +def mark_checked(results, todo_file, checked_file, platform): + completed_targets = [ + result.get('target', '') for result in results + if result.get('target') and str(result.get('skipped') or '') not in CI_SOFT_SKIP_REASONS.get(platform, set()) + ] + with projection_file_lock(todo_file): + checked_normalized = {normalize_target(target, platform) for target in load_set_from_file(checked_file)} + new_completed_targets = [] + for result in results: + target = result.get('target', '') + if not target: + continue + if str(result.get('skipped') or '') in CI_SOFT_SKIP_REASONS.get(platform, set()): + continue + normalized = normalize_target(target, platform) + if normalized not in checked_normalized: + new_completed_targets.append(target) + checked_normalized.add(normalized) + + if new_completed_targets: + _append_lines_unlocked(checked_file, new_completed_targets) + + completed_normalized = {normalize_target(target, platform) for target in completed_targets} + remaining = [ + target for target in load_set_from_file(todo_file) + if normalize_target(target, platform) not in completed_normalized + ] + _write_lines_unlocked(todo_file, sorted(remaining)) + + +def enqueue_targets_for_platform(queue_dir, platform, targets): + if not targets: + return 0 + os.makedirs(queue_dir, exist_ok=True) + todo_file = os.path.join(queue_dir, f'todo_{platform}.txt') + checked_file = os.path.join(queue_dir, f'checked_{platform}.txt') + with projection_file_lock(todo_file): + todo = load_set_from_file(todo_file) + checked = load_set_from_file(checked_file) + known = {normalize_target(target, platform) for target in todo} + known.update(normalize_target(target, platform) for target in checked) + new_targets = [] + for target in targets: + normalized = normalize_target(target, platform) + if normalized in known: + continue + new_targets.append(target) + known.add(normalized) + if new_targets: + _append_lines_unlocked(todo_file, new_targets) + return len(new_targets) + + +def collect_postman_targets(results): + targets = [] + for result in results or []: + for target in result.get('postman_targets') or []: + if target: + targets.append(target) + return targets + + +def publish_scan_payload(result): + try: + publication_result = copy.deepcopy(result) + candidate_result = copy.deepcopy(result) + if candidate_result.get('structured_keycheck_pending') and isinstance(candidate_result.get('postman'), dict): + postman_data = candidate_result['postman'] + candidate_path, _ = validate_postman_cache_artifact( + postman_data, + int(candidate_result.get('postman_max_artifact_size_mb') or 20), + expected_size=candidate_result.get('bytes'), + ) + write_structured_keycheck_candidates(candidate_path, postman_data) + if not save_scan_result(publication_result): + raise RuntimeError('one or more scanner JSONL/keycheck publications failed') + return True, '' + except Exception as exc: + return False, str(exc) + + +def drain_scan_publication_outbox(db, limit=100, require_v2_schema=True): + if not db or not getattr(db, 'conn', None) or not getattr(db.conn, 'is_postgres', False): + return 0 + lease_seconds = 300 + heartbeat_interval = min(60.0, lease_seconds / 3.0) + heartbeat_join_timeout = 12.0 + db_path = getattr(db, 'path', None) + db_url = getattr(db, 'url', None) + # Every enabled production ScannerDB has one of these identities. Their + # absence is tolerated for lightweight test doubles only. + canonical_db = bool(db_path or db_url) + + def fenced_finish(row_id, owner, delivered, error): + try: + return bool(db.finish_scan_publication(row_id, owner, delivered, error)) + except Exception as exc: + logger.error('Publication outbox fenced finish failed for row %s: %s', row_id, exc) + return False + + def stop_heartbeat(stop_event, thread, heartbeat_db): + if stop_event is not None: + stop_event.set() + alive = False + if thread is not None: + try: + alive = thread.is_alive() + if alive: + thread.join(heartbeat_join_timeout) + alive = thread.is_alive() + except Exception as exc: + alive = True + logger.error('Unable to join publication heartbeat thread: %s', exc) + if alive: + logger.error('Publication heartbeat did not stop within %.1f seconds', heartbeat_join_timeout) + elif heartbeat_db is not None: + try: + heartbeat_db.close() + except Exception as exc: + logger.error('Unable to close publication heartbeat DB: %s', exc) + return not alive + + delivered = 0 + for sequence in range(max(1, int(limit or 100))): + lease_owner = ( + f'publisher:{os.getpid()}:{threading.get_ident()}:{sequence}:' + f'{secrets.token_urlsafe(18)}' + ) + rows = db.claim_scan_publications(lease_owner, 1) + if not rows: + break + row = rows[0] + try: + result = json.loads(row['payload_json']) + except (TypeError, ValueError) as exc: + if not fenced_finish(row['id'], lease_owner, False, f'invalid publication payload: {exc}'): + if canonical_db: + logger.warning('Publication ownership handoff detected for row %s; stopping this drain pass', row['id']) + break + logger.critical('Publication outbox lease acknowledgement failed for row %s', row['id']) + raise RuntimeError(f'publication outbox lease acknowledgement failed for row {row["id"]}') + continue + + heartbeat_db = None + heartbeat_stop = None + heartbeat_thread = None + heartbeat_lost = None + heartbeat_failed = None + if canonical_db: + try: + heartbeat_stop = threading.Event() + heartbeat_lost = threading.Event() + heartbeat_failed = threading.Event() + heartbeat_db = ScannerDB(db_path=db_path, db_url=db_url, initialize=False) + if not heartbeat_db.enabled: + raise RuntimeError('dedicated publication heartbeat DB is unavailable') + if getattr(heartbeat_db.conn, 'is_postgres', False): + heartbeat_db.conn.execute("SELECT set_config('statement_timeout', '10000ms', false)") + heartbeat_db.conn.execute("SELECT set_config('lock_timeout', '5000ms', false)") + heartbeat_db.conn.commit() + if require_v2_schema: + heartbeat_db.require_runtime_safety_schema() + if not heartbeat_db.renew_scan_publication(row['id'], lease_owner, lease_seconds): + detail = getattr(heartbeat_db, 'last_error', '') + raise RuntimeError(detail or 'initial publication lease renewal did not confirm ownership') + + def renew_publication_lease(connection=heartbeat_db): + try: + while not heartbeat_stop.wait(heartbeat_interval): + if connection.renew_scan_publication(row['id'], lease_owner, lease_seconds): + continue + if getattr(connection, 'last_error', ''): + heartbeat_failed.set() + logger.error( + 'Publication lease heartbeat failed for row %s: %s', + row['id'], connection.last_error, + ) + else: + heartbeat_lost.set() + logger.warning('Publication lease ownership was lost for row %s', row['id']) + return + except Exception as exc: + heartbeat_failed.set() + logger.error('Publication lease heartbeat failed for row %s: %s', row['id'], exc) + finally: + try: + connection.close() + except Exception as exc: + logger.error('Unable to close publication heartbeat DB: %s', exc) + + heartbeat_thread = threading.Thread( + target=renew_publication_lease, + name=f'scan-publication-heartbeat-{row["id"]}', + daemon=True, + ) + heartbeat_thread.start() + except Exception as exc: + stopped = stop_heartbeat(heartbeat_stop, heartbeat_thread, heartbeat_db) + setup_error = f'publication lease heartbeat setup failed: {exc}' + acknowledged = fenced_finish(row['id'], lease_owner, False, setup_error) + if acknowledged: + logger.error('Publication row %s was requeued after heartbeat setup failure: %s', row['id'], exc) + else: + logger.warning( + 'Publication ownership handoff detected while handling setup failure for row %s', row['id'], + ) + if not stopped: + logger.error('Publication row %s heartbeat cleanup remains in progress', row['id']) + break + + try: + ok, error = publish_scan_payload(result) + except Exception as exc: + ok, error = False, str(exc) + + heartbeat_stopped = True + if canonical_db: + heartbeat_stopped = stop_heartbeat(heartbeat_stop, heartbeat_thread, heartbeat_db) + acknowledged = fenced_finish(row['id'], lease_owner, ok, error) + if not acknowledged: + if canonical_db: + logger.warning('Publication ownership handoff detected for row %s; stopping this drain pass', row['id']) + break + logger.critical('Publication outbox lease acknowledgement failed for row %s', row['id']) + raise RuntimeError(f'publication outbox lease acknowledgement failed for row {row["id"]}') + if ok: + delivered += 1 + if canonical_db and ( + not heartbeat_stopped or heartbeat_lost.is_set() or heartbeat_failed.is_set() + ): + logger.warning('Stopping publication drain after heartbeat ownership uncertainty for row %s', row['id']) + break + return delivered + + +def require_scan_publication_capacity(db, args, additional_items=1): + health_loader = getattr(db, 'scan_publication_backlog_health', None) + if health_loader is None: + return None + health = health_loader( + int(getattr(args, 'scan_outbox_max_pending_items', 10000)), + int(getattr(args, 'scan_outbox_max_pending_bytes', 1024 * 1024 * 1024)), + int(getattr(args, 'scan_outbox_max_pending_age_sec', 24 * 60 * 60)), + additional_items=additional_items, + ) + if not isinstance(health, dict) or not health.get('accepting'): + detail = (health or {}).get('reason') if isinstance(health, dict) else 'health query failed' + raise RuntimeError(f'scan publication backlog gate is closed: {detail or "configured bound reached"}') + return health + + +def result_spool_for_args(args): + runtime_dir = os.path.realpath(os.path.abspath(getattr(args, 'runtime_dir', '') or '')) + spool_dir = os.path.realpath(os.path.abspath(getattr(args, 'result_spool_dir', '') or '')) + if not runtime_dir or not spool_dir: + raise RuntimeError('Postgres queue mode requires runtime_dir and result_spool_dir') + try: + within_runtime = os.path.commonpath((runtime_dir, spool_dir)) == runtime_dir + except ValueError: + within_runtime = False + if not within_runtime or spool_dir == runtime_dir: + raise RuntimeError('result_spool_dir must be a dedicated directory under runtime_dir') + return ResultSpool( + spool_dir, + max_event_bytes=int(getattr(args, 'result_spool_max_event_bytes', 192 * 1024 * 1024)), + max_events=int(getattr(args, 'result_spool_max_events', 10000)), + max_total_bytes=int(getattr(args, 'result_spool_max_total_bytes', 2 * 1024 * 1024 * 1024)), + min_free_bytes=int(getattr(args, 'result_spool_min_free_bytes', 1024 * 1024 * 1024)), + ) + + +class ResultSpoolPublisherBusy(RuntimeError): + def __init__(self, state): + self.state = dict(state or {}) + super().__init__('result-spool publisher lease is held by another live database session') + + +class ResultSpoolPublisherHandoff(RuntimeError): + pass + + +def supervised_spool_stop_requested(): + instance_file = str(os.getenv('TRUF_SUPERVISOR_INSTANCE_FILE') or '') + inherited_id = str(os.getenv('TRUF_SUPERVISOR_INSTANCE_ID') or '') + if not instance_file or not inherited_id: + return False + try: + metadata = read_private_json(instance_file) + except (OSError, ValueError): + return False + if str(metadata.get('instance_id') or '') != inherited_id: + raise RuntimeError('supervisor instance identity changed while waiting on result-spool backpressure') + return str(metadata.get('activation_state') or '').upper() in ('STOPPING', 'FAILED_HOLD', 'INACTIVE') + + +def drain_result_spool(spool, db): + if not db or not getattr(db, 'conn', None) or not getattr(db.conn, 'is_postgres', False): + raise RuntimeError('result spool ingestion requires PostgreSQL') + db.require_runtime_safety_schema() + acquire = getattr(db, 'try_acquire_result_spool_publisher', None) + release = getattr(db, 'release_result_spool_publisher', None) + acquired = False + if acquire: + acquired = bool(acquire()) + if not acquired: + inspect_owner = getattr(db, 'result_spool_publisher_state', None) + state = inspect_owner() if inspect_owner else { + 'status': 'held', 'authenticated': True, + 'application_name': 'test-publisher', 'holder_identity': 'test', + } + if not isinstance(state, dict): + raise RuntimeError('result-spool publisher ownership lookup returned an invalid state') + if state.get('status') == 'free' and state.get('authenticated') is True: + raise ResultSpoolPublisherHandoff( + 'result-spool publisher released ownership before inspection' + ) + if ( + state.get('status') != 'held' + or state.get('authenticated') is not True + or not state.get('application_name') + or not state.get('holder_identity') + ): + raise RuntimeError('result-spool publisher ownership is unknown or unauthenticated') + raise ResultSpoolPublisherBusy(state) + outcomes = {} + legacy_records = None + try: + while True: + if hasattr(spool, 'next_pending_event'): + record = spool.next_pending_event() + else: + if legacy_records is None: + legacy_records = iter(spool.pending_events()) + record = next(legacy_records, None) + if record is None: + break + try: + outcome = db.ingest_scan_event(record.envelope) + except ScanEventConflictError as exc: + spool.quarantine_event(record.event_id, f'database scan-event hash conflict: {exc}') + raise RuntimeError(f'scan event {record.event_id} was quarantined after a database hash conflict') from exc + if not scan_outcome_matches(record, outcome): + raise RuntimeError(f'database did not confirm ingestion of scan event {record.event_id}') + if spool.acknowledge(record.event_id, record.event_hash) is not True: + raise RuntimeError(f'result-spool acknowledgement was not confirmed for event {record.event_id}') + outcomes[record.event_id] = outcome + spool.assert_claims_allowed() + return outcomes + finally: + if acquired and (not release or release() is not True): + raise RuntimeError('result-spool publisher lease release was not confirmed') + + +def wait_for_result_spool_ready( + spool, + db, + outcomes=None, + stop_event=None, + wait_seconds=1.0, + diagnostic_interval=30.0, +): + outcomes = outcomes if outcomes is not None else {} + wait_seconds = max(0.05, float(wait_seconds or 1.0)) + diagnostic_interval = max(wait_seconds, float(diagnostic_interval or 30.0)) + last_diagnostic = 0.0 + while True: + if (stop_event is not None and stop_event.is_set()) or supervised_spool_stop_requested(): + raise KeyboardInterrupt('coordinated shutdown while waiting for result-spool backpressure') + try: + drained = drain_result_spool(spool, db) + outcomes.update(drained) + if drained: + safe_print(f'Drained {len(drained)} prior durable result-spool event(s).', flush=True) + return outcomes + except ResultSpoolPublisherBusy as exc: + publisher = str(exc.state.get('application_name') or '')[:80] + state = None + reason = f'publisher={publisher}' + except ResultSpoolPublisherHandoff: + state = None + reason = 'publisher handoff observed; retrying acquisition' + except (SpoolBackpressureError, SpoolContentionError) as exc: + reason = str(exc)[:200] + try: + state = spool.backpressure_state() if hasattr(spool, 'backpressure_state') else None + except SpoolContentionError: + state = None + now = time.monotonic() + if now - last_diagnostic >= diagnostic_interval: + details = '' + if state: + details = ( + f" pending={state.get('pending_count', 0)}" + f" bytes={state.get('pending_bytes', 0)}" + f" oldest_age_sec={state.get('oldest_pending_age_sec', 0)}" + f" reservations={state.get('reservation_count', 0)}" + ) + safe_print(f'Result-spool backpressure; waiting in-process:{details} {reason}', flush=True) + last_diagnostic = now + if stop_event is not None: + if stop_event.wait(wait_seconds): + raise KeyboardInterrupt('coordinated shutdown while waiting for result-spool backpressure') + else: + time.sleep(wait_seconds) + + +def reserve_result_spool_claims( + spool, + db, + owner, + count, + lease_seconds, + stop_event=None, + wait_seconds=1.0, + diagnostic_interval=30.0, +): + last_capacity_diagnostic = 0.0 + while True: + if hasattr(spool, 'next_pending_event') or hasattr(spool, 'pending_events'): + wait_for_result_spool_ready( + spool, + db, + stop_event=stop_event, + wait_seconds=wait_seconds, + diagnostic_interval=diagnostic_interval, + ) + try: + return spool.reserve_claims(owner, count, lease_seconds) + except (SpoolBackpressureError, SpoolContentionError): + continue + except SpoolTransientCapacityError as exc: + snapshot_loader = getattr(spool, 'reservation_snapshot', None) + progress_loader = getattr(db, 'result_spool_reservation_progress', None) + if not snapshot_loader or not progress_loader: + raise RuntimeError('result-spool temporary capacity ownership cannot be verified') from exc + snapshot = snapshot_loader() + if not snapshot.get('reservations'): + if (stop_event is not None and stop_event.is_set()) or supervised_spool_stop_requested(): + raise KeyboardInterrupt('coordinated shutdown during result-spool capacity handoff') + if stop_event is not None: + if stop_event.wait(wait_seconds): + raise KeyboardInterrupt('coordinated shutdown during result-spool capacity handoff') + else: + time.sleep(max(0.05, wait_seconds)) + continue + progress = progress_loader(snapshot['reservations']) + if not isinstance(progress, dict) or progress.get('safe_progress') is not True: + raise RuntimeError('result-spool reservation progress is unsafe or indeterminate') from exc + now = time.monotonic() + if now - last_capacity_diagnostic >= max(wait_seconds, diagnostic_interval): + counts = progress.get('counts') or {} + safe_print( + 'Result-spool capacity reserved; waiting in-process: ' + f"reserved_bytes={snapshot.get('reserved_future_bytes', 0)} " + f"reservations={snapshot.get('reservation_count', 0)} " + f"exact_live={counts.get('exact_live', 0)} " + f"exact_expired={counts.get('exact_expired', 0)} " + f"stale={counts.get('stale_or_reassigned', 0)} " + f"next_progress_sec={progress.get('earliest_progress_in_sec', 0)}", + flush=True, + ) + last_capacity_diagnostic = now + if (stop_event is not None and stop_event.is_set()) or supervised_spool_stop_requested(): + raise KeyboardInterrupt('coordinated shutdown while waiting for result-spool capacity') + if stop_event is not None: + if stop_event.wait(wait_seconds): + raise KeyboardInterrupt('coordinated shutdown while waiting for result-spool capacity') + else: + time.sleep(max(0.05, wait_seconds)) + + +def scan_outcome_matches(record, outcome): + return bool( + outcome + and outcome.get('ingested') + and str(outcome.get('scan_event_id') or '') == str(record.event_id) + and str(outcome.get('scan_event_hash') or '') == str(record.event_hash) + ) + + +def recheck_active_lease_ownership(db, lease_owner, requested_tokens, active_tokens, active_lock): + with active_lock: + expected_before_query = set(requested_tokens).intersection(active_tokens) + if not expected_before_query: + return True + confirmed = set(db.active_target_lease_tokens(lease_owner, expected_before_query)) + with active_lock: + expected_after_query = expected_before_query.intersection(active_tokens) + return not expected_after_query.difference(confirmed) + + +def renew_active_result_spool_reservation( + spool, reservation_id, lease_seconds, reserved_tokens, active_lock, +): + with active_lock: + if not reservation_id or not reserved_tokens: + return True + return spool.renew_reservation(reservation_id, lease_seconds) is True + + +def write_reserved_result_spool_event( + spool, event, reservation_id, queue_id, lease_token, + reserved_tokens, active_tokens, active_lock, +): + # Keep the token transition and reservation consumption in one lock order. + # A heartbeat can then observe either the live reservation or its completed + # durable handoff, never the transient gap between those states. + with active_lock: + record = spool.write_event( + event, + reservation_id=reservation_id, + queue_id=queue_id, + lease_token=lease_token, + ) + reserved_tokens.discard(lease_token) + active_tokens.discard(lease_token) + return record + + +def first_error_line(result): + for error in result.get('errors', []): + for line in str(error).splitlines(): + line = line.strip() + if line: + try: + import json + payload = json.loads(line) + if payload.get('error'): + return str(payload.get('error'))[:300] + if payload.get('msg'): + return str(payload.get('msg'))[:300] + except Exception: + pass + return line[:300] + return '' + + +def target_retry_delay_sec(attempt, base_delay_sec=3600, max_delay_sec=86400): + attempt = max(1, int(attempt or 1)) + base_delay_sec = max(1, int(base_delay_sec or 3600)) + max_delay_sec = max(base_delay_sec, int(max_delay_sec or 86400)) + return min(max_delay_sec, base_delay_sec * (2 ** max(0, attempt - 1))) + + +def queue_error_disposition(db, source, platform, target, result, args, queue_claim=None): + queue_row = None if queue_claim else db.target_queue_item(source, platform, target) if db else None + attempts = int((queue_claim or {}).get('attempts') or 0) if queue_claim else int(queue_row['attempts'] or 0) if queue_row else 1 + max_attempts = max(1, int(getattr(args, 'target_retry_max_attempts', 3) or 3)) + timed_out = bool((result.get('scan_meta') or {}).get('command_timed_out')) or result.get('error_class') == 'timeout' + if timed_out: + if attempts >= max_attempts: + return 'failed', None, attempts, max_attempts + delay = max(60, int(getattr(args, 'target_timeout_retry_delay_sec', 21600) or 21600)) + available_after = (datetime.now(timezone.utc) + timedelta(seconds=delay)).isoformat(timespec='seconds') + return 'deferred', available_after, attempts, max_attempts + if result.get('source_failure'): + if not bool(result.get('retryable', True)) and attempts >= max_attempts: + return 'failed', None, attempts, max_attempts + delay = max(1, int(getattr(args, 'target_retry_max_delay_sec', 86400) or 86400)) + available_after = (datetime.now(timezone.utc) + timedelta(seconds=delay)).isoformat(timespec='seconds') + return 'deferred', available_after, attempts, max_attempts + retryable = bool(result.get('retryable', True)) + if not retryable or attempts >= max_attempts: + return 'failed', None, attempts, max_attempts + delay = target_retry_delay_sec( + attempts, + getattr(args, 'target_retry_base_delay_sec', 3600), + getattr(args, 'target_retry_max_delay_sec', 86400), + ) + available_after = (datetime.now(timezone.utc) + timedelta(seconds=delay)).isoformat(timespec='seconds') + return 'deferred', available_after, attempts, max_attempts + + +def queue_result_resets_attempts(result): + return bool( + (result.get('source_failure') and result.get('retryable', True)) + or docker_layer_result_resets_attempts(result) + ) + + +def docker_layer_result_resets_attempts(result): + if result.get('docker_layer_plan') is None or not result.get('retryable', False): + return False + execution = result.get('docker_layer_execution') + records = execution.get('blobs') if isinstance(execution, dict) else None + if not isinstance(records, list): + return False + descriptors = result['docker_layer_plan'].get('descriptors') + if not isinstance(descriptors, list) or not any( + item.get('coverage_state') in ('selected', 'shared_pending') + for item in descriptors if isinstance(item, dict) + ): + return False + return not any( + item.get('status') in ('retryable_failed', 'terminal_failed') + for item in records if isinstance(item, dict) + ) + + +def docker_layer_queue_disposition(result, args, attempts=None): + if result.get('docker_layer_plan') is None: + return None + if not result.get('errors'): + return 'done', None, False + if not bool(result.get('retryable', False)): + return 'failed', None, False + reset_attempts = docker_layer_result_resets_attempts(result) + max_attempts = max(1, int(getattr(args, 'target_retry_max_attempts', 3) or 3)) + if not reset_attempts and attempts is not None and int(attempts or 0) >= max_attempts: + return 'failed', None, False + delay = max(1, int(getattr(args, 'docker_layer_checkpoint_delay_sec', 60) or 60)) + available_after = ( + datetime.now(timezone.utc) + timedelta(seconds=delay) + ).isoformat(timespec='seconds') + return 'deferred', available_after, reset_attempts + + +def finding_summary(finding): + detector = finding.get('DetectorName', 'Unknown') + verified = 'verified' if finding.get('Verified', False) else 'unverified' + source = finding.get('SourceMetadata', {}).get('Data', {}) + git_source = source.get('Git', {}) if isinstance(source, dict) else {} + file_name = git_source.get('file', '') + line_number = git_source.get('line', '') + + location = '' + if file_name and line_number: + location = f' at {file_name}:{line_number}' + elif file_name: + location = f' at {file_name}' + return f'{detector} ({verified}){location}' + + +def print_result_report(results, save_dir): + results_file = os.path.join(save_dir, 'scan_results.jsonl') + secrets_file = os.path.join(save_dir, 'found_secrets.jsonl') + errors_file = os.path.join(save_dir, 'scan_errors.log') + + for index, result in enumerate(results, 1): + target = result.get('target', '') + findings = result.get('findings', []) + errors = result.get('errors', []) + warnings = result.get('warnings', []) + skipped = result.get('skipped') + + safe_print(f'\n[{index}/{len(results)}] Target: {target}') + if skipped: + safe_print(f' SKIPPED: {skipped}') + if findings: + safe_print(f' FOUND: {len(findings)} secret(s). Saved: {secrets_file}') + for finding in findings[:5]: + safe_print(f' - {finding_summary(finding)}') + if len(findings) > 5: + safe_print(f' - ... {len(findings) - 5} more') + if errors: + safe_print(f' ERROR: saved to {errors_file}') + safe_print(f' First error: {first_error_line(result)}') + if warnings: + safe_print(f' DEGRADED: {len(warnings)} non-fatal diagnostic(s)') + if not skipped and not findings and not errors and not warnings: + safe_print(' CLEAN: no secrets found') + + if findings or errors or warnings: + safe_print(f' Full non-clean result saved: {results_file}') + + +def prepare_scan_options(args, target_count, *, quiet=False): + if not quiet: + print(f'Scanning {target_count} targets with {args.workers} workers...') + print(f'TruffleHog timeout per target: {args.timeout}s') + print(f'Results directory: {args.save_dir}') + scan_config.drop_detectors = csv_items( + getattr(args, 'drop_detectors', getattr(scan_config, 'drop_detectors', [])) + ) + if scan_config.drop_detectors and not quiet: + print(f'Dropping detector findings before persistence: {",".join(scan_config.drop_detectors)}') + + scan_kwargs = { + 'timeout_sec': args.timeout, + 'detectors': args.detectors, + 'exclude_detectors': args.exclude_detectors, + 'no_verification': bool(getattr(args, 'no_verification', False)), + 'trufflehog_config': getattr(args, 'trufflehog_config', ''), + } + if args.platform == 'docker': + scan_kwargs['trufflehog_concurrency'] = int(getattr(args, 'trufflehog_concurrency', 0) or 0) + scan_kwargs['docker_recovery_limits'] = docker_layer_limits(args) + scan_kwargs['docker_recovery_min_free_bytes'] = int(getattr(args, 'docker_layer_min_free_bytes', 20 << 30)) + if scan_kwargs['trufflehog_concurrency'] > 0 and not quiet: + print(f"TruffleHog internal concurrency: {scan_kwargs['trufflehog_concurrency']}") + if args.platform == 'gitlab': + scan_kwargs['external_trufflehog_lifecycle'] = bool( + getattr(args, 'external_trufflehog_lifecycle', False) + ) + if args.platform in ('github', 'github_archive', 'gitlab', 'package_git') and not getattr(args, 'scan_full_history', False): + max_depth = int(getattr(args, 'max_depth', 0) or 0) + if max_depth > 0: + scan_kwargs['max_depth'] = max_depth + if not quiet: + print(f'TruffleHog git max depth: {max_depth} commits') + if args.platform in ('github', 'github_archive', 'gitlab', 'package_git'): + max_commit_age_days = int(getattr(args, 'max_commit_age_days', 0) or 0) + if max_commit_age_days > 0: + scan_kwargs['max_commit_age_days'] = max_commit_age_days + scan_kwargs['commit_lookup_pages'] = int(getattr(args, 'commit_lookup_pages', 3) or 3) + scan_kwargs['skip_if_commit_lookup_fails'] = bool(getattr(args, 'skip_if_commit_lookup_fails', True)) + if not quiet: + print(f'Git scan commit max age: {max_commit_age_days} days') + if args.platform in ('npm', 'pypi'): + scan_kwargs['max_artifact_size_mb'] = int(getattr(args, 'max_artifact_size_mb', 50) or 50) + if args.platform == 'postman': + scan_kwargs['max_artifact_size_mb'] = int(getattr(args, 'max_artifact_size_mb', 20) or 20) + if args.platform == 'github_actions': + scan_kwargs.update({ + 'ci_runs_per_repo': int(getattr(args, 'ci_runs_per_repo', 5) or 5), + 'ci_lookback_days': int(getattr(args, 'ci_lookback_days', 30) or 30), + 'ci_max_log_archive_mb': int(getattr(args, 'ci_max_log_archive_mb', 50) or 50), + 'ci_max_log_file_mb': int(getattr(args, 'ci_max_log_file_mb', 20) or 20), + 'ci_failed_first': bool(getattr(args, 'ci_failed_first', True)), + 'ci_scan_artifacts': bool(getattr(args, 'ci_scan_artifacts', False)), + 'ci_max_artifacts_per_run': int(getattr(args, 'ci_max_artifacts_per_run', 3) or 3), + 'ci_max_artifact_archive_mb': int(getattr(args, 'ci_max_artifact_archive_mb', 50) or 50), + 'ci_max_artifact_file_mb': int(getattr(args, 'ci_max_artifact_file_mb', 10) or 10), + 'ci_max_artifact_files': int(getattr(args, 'ci_max_artifact_files', 1000) or 1000), + 'ci_target_max_download_mb': int(getattr(args, 'ci_target_max_download_mb', 500) or 500), + 'fetch_timeout': int(getattr(args, 'fetch_timeout', 20) or 20), + }) + if args.platform == 'gitlab_ci': + scan_kwargs.update({ + 'ci_pipelines_per_project': int(getattr(args, 'ci_pipelines_per_project', 5) or 5), + 'ci_jobs_per_pipeline': int(getattr(args, 'ci_jobs_per_pipeline', 20) or 20), + 'ci_lookback_days': int(getattr(args, 'ci_lookback_days', 30) or 30), + 'ci_max_trace_mb': int(getattr(args, 'ci_max_trace_mb', 20) or 20), + 'ci_scan_artifacts': bool(getattr(args, 'ci_scan_artifacts', False)), + 'ci_max_artifacts_per_pipeline': int(getattr(args, 'ci_max_artifacts_per_pipeline', 5) or 5), + 'ci_max_artifact_archive_mb': int(getattr(args, 'ci_max_artifact_archive_mb', 50) or 50), + 'ci_max_artifact_file_mb': int(getattr(args, 'ci_max_artifact_file_mb', 10) or 10), + 'ci_max_artifact_files': int(getattr(args, 'ci_max_artifact_files', 1000) or 1000), + 'ci_target_max_download_mb': int(getattr(args, 'ci_target_max_download_mb', 500) or 500), + 'fetch_timeout': int(getattr(args, 'fetch_timeout', 20) or 20), + }) + token = get_platform_token(args) + if token: + scan_kwargs['token'] = token + return scan_kwargs + + +def validate_v2_capacity_model( + max_active_scans, max_event_bytes, projection_max_bytes, + projection_headroom_bytes, +): + slots = max(1, int(max_active_scans)) + event_bytes = max(1, int(max_event_bytes)) + per_scan_projection = event_bytes * 2 + headroom = max(per_scan_projection, int(projection_headroom_bytes)) + required = slots * per_scan_projection + headroom + if int(projection_max_bytes) < required: + raise RuntimeError( + 'projection capacity cannot cover every physical scan slot plus bounded backlog ' + f'headroom: configured={int(projection_max_bytes)} required={required}' + ) + return { + 'physical_slots': slots, + 'per_scan_projection_bytes': per_scan_projection, + 'headroom_bytes': headroom, + 'required_projection_bytes': required, + } + + +@dataclass(frozen=True) +class V2AdmissionOutcome: + claim: object + permit_released: bool + retry_without_claim: bool = False + reason: str | None = None + + +def reserve_v2_admission_with_recovery( + db_url, source, platform, producer_mapping, supervisor_instance_id, + declared_bundle_bytes, projection_bytes, candidate_items, candidate_bytes, + *, lease_seconds, max_attempts, capacity_limits, run_id, cycle_id, + reservation_token, bundle_id, scan_event_id, + resolution_attempts=8, resolution_seconds=30, retry_delay=0.2, + claim_order='oldest', docker_depth_authority=None, final_cutover=False, + db_factory=ScannerDB, stop_requested=supervised_spool_stop_requested, + sleep=time.sleep, release_permit=lambda: None, remote_assignment=None, + reserved_bundle_bytes=None, remote_max_active=50, +): + deadline = time.monotonic() + max(0.1, float(resolution_seconds)) + last_error = None + admission = db_factory(db_url=db_url, initialize=False) + try: + if not admission.enabled: + release_permit() + return V2AdmissionOutcome(None, permit_released=True) + admission.set_application_name(f'truf-admission:{source}') + claim = admission.reserve_and_claim_target( + source, platform, producer_mapping, supervisor_instance_id, + declared_bundle_bytes, projection_bytes, candidate_items, candidate_bytes, + lease_seconds=lease_seconds, max_attempts=max_attempts, + capacity_limits=capacity_limits, run_id=run_id, cycle_id=cycle_id, + reservation_token=reservation_token, bundle_id=bundle_id, + scan_event_id=scan_event_id, claim_order=claim_order, + docker_depth_authority=docker_depth_authority, + final_cutover=final_cutover, remote_assignment=remote_assignment, + reserved_bundle_bytes=reserved_bundle_bytes, + remote_max_active=remote_max_active, + ) + reason = ( + admission.admission_intent_resolution(reservation_token) + if claim is None else None + ) + return V2AdmissionOutcome( + claim, permit_released=False, reason=reason, + ) + except (KeyboardInterrupt, SystemExit): + raise + except Exception as exc: + last_error = exc + release_permit() + logger.warning( + 'Admission outcome is ambiguous; released physical permit and entered ' + 'bounded idempotent exact-token resolution: %s', + type(exc).__name__, + ) + finally: + admission.close() + + expected = { + 'bundle_id': str(bundle_id).lower(), + 'scan_event_id': str(scan_event_id).lower(), + 'source': str(source), + 'platform': str(platform), + 'producer_instance_id': str(supervisor_instance_id or ''), + 'producer_pid': int(producer_mapping['pid']), + 'producer_creation_time': str(producer_mapping['creation_time']), + 'producer_executable': str(producer_mapping['executable']), + 'declared_bundle_bytes': int(declared_bundle_bytes), + 'reserved_bundle_bytes': int( + declared_bundle_bytes + if reserved_bundle_bytes is None else reserved_bundle_bytes + ), + 'reserved_projection_items': 1, + 'reserved_projection_bytes': int(projection_bytes), + 'reserved_candidate_items': int(candidate_items), + 'reserved_candidate_bytes': int(candidate_bytes), + 'run_id': run_id, + 'cycle_id': cycle_id, + 'assignment_kind': 'local', + 'remote_user_id': None, + 'remote_device_id': None, + 'remote_effective_config_sha256': None, + 'remote_client_compat_sha256': None, + 'remote_execution_snapshot_json': None, + 'remote_execution_snapshot_sha256': None, + } + if remote_assignment is not None: + remote = ScannerDB._remote_assignment_mapping(remote_assignment) + expected.update({ + 'assignment_kind': 'remote', + 'remote_user_id': remote['user_id'], + 'remote_device_id': remote['device_id'], + 'remote_effective_config_sha256': remote['effective_config_sha256'], + 'remote_client_compat_sha256': remote['client_compat_sha256'], + 'remote_execution_snapshot_json': remote['execution_snapshot_json'], + 'remote_execution_snapshot_sha256': remote['execution_snapshot_sha256'], + }) + for attempt in range(max(1, int(resolution_attempts))): + if stop_requested(): + raise KeyboardInterrupt('coordinated shutdown during exact claim recovery') + recovery = db_factory(db_url=db_url, initialize=False) + try: + if not recovery.enabled: + raise RuntimeError('admission PostgreSQL connection is unavailable') + recovery.set_application_name(f'truf-admission-recovery:{source}') + claim = recovery.recover_result_reservation_claim( + reservation_token, expected, + ) + if claim: + logger.info( + 'Recovered exact admission reservation after an ambiguous response: %s', + claim['reservation_id'], + ) + return V2AdmissionOutcome(claim, permit_released=True) + return V2AdmissionOutcome(None, permit_released=True) + except (KeyboardInterrupt, SystemExit): + raise + except Exception as exc: + last_error = exc + logger.warning( + 'Admission serialized exact-token resolution remains unavailable (%s/%s): %s', + attempt + 1, max(1, int(resolution_attempts)), type(exc).__name__, + ) + finally: + recovery.close() + remaining = deadline - time.monotonic() + if attempt + 1 >= max(1, int(resolution_attempts)) or remaining <= 0: + break + sleep(min(remaining, min(1.0, max(0.05, float(retry_delay))))) + raise UnresolvedHandoffInfrastructureError( + 'admission outcome remained ambiguous after bounded exact-token resolution' + ) from last_error + + +def reacquire_scan_permit_bounded( + command, timeout_sec, wait_seconds, *, acquire=acquire_scan_slot, + stop_requested=supervised_spool_stop_requested, sleep=time.sleep, +): + if not scan_limiter_enabled(): + return None + deadline = time.monotonic() + max(0.1, float(wait_seconds)) + while time.monotonic() < deadline: + if stop_requested(): + raise KeyboardInterrupt('coordinated shutdown while reacquiring scan permit') + lease = acquire(command, timeout_sec, wait=False, start_heartbeat=False) + if lease is not None: + return lease + sleep(min(0.2, max(0.0, deadline - time.monotonic()))) + raise TimeoutError('bounded scan permit reacquisition expired') + + +def refund_v2_claim_after_no_handoff( + db, bundle_root, claim, producer_mapping, reason, +): + ready_relative = str(claim['ready_relative_path']).replace('\\', '/') + ready = inspect_private_relative_path(bundle_root, ready_relative) + if ready.state != PrivatePathState.ABSENT: + return False + partial_relative = bundle_partial_relative_path( + claim['bundle_id'], claim['reservation_token'], + ).replace(os.sep, '/') + partial = inspect_private_relative_path(bundle_root, partial_relative) + if partial.state == PrivatePathState.UNKNOWN: + return False + if partial.state == PrivatePathState.PRESENT: + if not private_file_ready(partial.path): + return False + try: + durable_unlink(partial.path) + except OSError: + return False + partial = inspect_private_relative_path(bundle_root, partial_relative) + if partial.state != PrivatePathState.ABSENT: + return False + return bool(db.refund_uncommitted_reservation( + claim['reservation_id'], producer_mapping, reason, + partial_absence_confirmed=True, + )) + + +class _GitResolutionFailure(Exception): + def __init__(self, error, invalid_target=False): + super().__init__(str(error)) + self.error = error + self.invalid_target = bool(invalid_target) + + +class _DockerPlanningFailure(Exception): + def __init__(self, error, *, retryable, source_failure, category): + super().__init__(str(error)) + self.error = error + self.retryable = bool(retryable) + self.source_failure = bool(source_failure) + self.category = str(category or 'docker_planning') + + +def docker_layer_limits(args): + return validate_docker_layer_limits({ + 'config_max_bytes': int(getattr(args, 'docker_layer_config_max_bytes', 1 << 20)), + 'layer_max_bytes': int(getattr(args, 'docker_layer_max_bytes', 256 << 20)), + 'image_max_bytes': int(getattr(args, 'docker_layer_image_max_bytes', 1 << 30)), + 'max_layers': int(getattr(args, 'docker_layer_max_layers', 8)), + 'archive_max_size_bytes': int( + getattr(args, 'docker_layer_archive_max_size_bytes', 256 << 20) + ), + 'archive_max_depth': int(getattr(args, 'docker_layer_archive_max_depth', 4)), + 'archive_timeout_sec': int(getattr(args, 'docker_layer_archive_timeout_sec', 30)), + 'blob_timeout_sec': int(getattr(args, 'docker_layer_blob_timeout_sec', 600)), + 'filesystem_concurrency': int( + getattr(args, 'docker_layer_filesystem_concurrency', 2) + ), + 'blob_max_attempts': int(getattr(args, 'docker_layer_blob_max_attempts', 3)), + }) + + +def docker_adaptive_checkpoint(args): + return validate_docker_adaptive_checkpoint({ + 'max_blobs': int(getattr(args, 'docker_adaptive_checkpoint_max_blobs', 4)), + 'max_bytes': int( + getattr(args, 'docker_adaptive_checkpoint_max_bytes', 512 << 20) + ), + }) + + +def docker_layer_effective_mode( + args, target, canary_eligible=False, *, adaptive_gate_passed=False, + selection_policy_sha256=None, +): + configured = str( + getattr(args, 'docker_content_scan_mode', 'full') or 'full' + ).strip().lower() + if configured not in ('full', 'canary', 'layer', 'adaptive-canary', 'adaptive'): + raise ValueError( + 'Docker content scan mode must be full, canary, layer, ' + 'adaptive-canary, or adaptive' + ) + basis_points = max(0, min( + 10000, int(getattr(args, 'docker_layer_canary_basis_points', 0) or 0), + )) + if configured in ('full', 'layer'): + return configured, basis_points + if configured == 'canary': + if basis_points == 0 or not canary_eligible: + return 'full', basis_points + image = str(parse_docker_target(target)['image']).lower() + manifest_digest = image.rsplit('@', 1)[-1] + effective = ( + 'layer' if docker_layer_canary_selected(manifest_digest, basis_points) else 'full' + ) + return effective, basis_points + + adaptive_basis_points = max(0, min( + 10000, int(getattr(args, 'docker_adaptive_canary_basis_points', 0) or 0), + )) + if not adaptive_gate_passed: + return 'full', adaptive_basis_points + if configured == 'adaptive': + return 'adaptive', 10000 + if adaptive_basis_points == 0: + return 'full', adaptive_basis_points + selection_policy_sha256 = selection_policy_sha256 or ( + docker_layer_selection_policy_sha256(docker_layer_limits(args)) + ) + image = str(parse_docker_target(target)['image']).lower() + manifest_digest = image.rsplit('@', 1)[-1] + effective = ( + 'adaptive' if docker_adaptive_canary_selected( + manifest_digest, selection_policy_sha256, adaptive_basis_points, + ) else 'full' + ) + return effective, adaptive_basis_points + + +def docker_layer_timeout_canary_eligible(db_url, source, claim): + eligibility_db = ScannerDB(db_url=db_url, initialize=False) + try: + if not eligibility_db.enabled: + raise RuntimeError('Docker layer canary eligibility database is unavailable') + eligibility_db.set_application_name(f'truf-docker-layer-canary:{source}') + return eligibility_db.docker_layer_timeout_canary_eligible( + claim['reservation_id'], claim['claim_lease_token'], + ) + finally: + eligibility_db.close() + + +def docker_adaptive_gate_report( + db_url, source, claim, scan_policy_sha256, execution_policy_sha256, + selection_policy_sha256, max_age_sec, +): + gate_db = ScannerDB(db_url=db_url, initialize=False) + try: + if not gate_db.enabled: + raise RuntimeError('Docker adaptive gate database is unavailable') + gate_db.set_application_name(f'truf-docker-adaptive-gate:{source}') + return gate_db.docker_adaptive_gate_report( + claim['reservation_id'], claim['claim_lease_token'], scan_policy_sha256, + execution_policy_sha256, selection_policy_sha256, + max_age_sec=max_age_sec, + ) + finally: + gate_db.close() + + +def docker_layer_scan_policy_sha256(args, scan_kwargs): + executable = shutil.which(get_trufflehog_cmd()) or get_trufflehog_cmd() + executable_sha256 = hash_file(executable) + if not executable_sha256: + raise RuntimeError('Docker layer scanner executable fingerprint is unavailable') + config_path = str(scan_kwargs.get('trufflehog_config') or '') + config_sha256 = hash_file(config_path) if config_path else None + if config_path and not config_sha256: + raise RuntimeError('Docker layer detector policy fingerprint is unavailable') + candidate_normalizer = os.path.join( + os.path.dirname(os.path.abspath(__file__)), 'keycheck_candidates.py', + ) + candidate_normalizer_sha256 = hash_file(candidate_normalizer) + if not candidate_normalizer_sha256: + raise RuntimeError('Docker routed identity policy fingerprint is unavailable') + recovery_limits = validate_docker_layer_limits( + scan_kwargs.get('docker_recovery_limits') or docker_layer_limits(args), + ) + payload = { + 'version': 'docker-content-scan-v3', + 'trufflehog_sha256': executable_sha256, + 'docker_recovery_execution': {key: recovery_limits[key] for key in ( + 'archive_max_size_bytes', 'archive_max_depth', 'archive_timeout_sec', 'filesystem_concurrency', + )}, + 'detectors': sorted(csv_items(scan_kwargs.get('detectors'))), + 'exclude_detectors': sorted(csv_items(scan_kwargs.get('exclude_detectors'))), + 'drop_detectors': sorted(csv_items(getattr(args, 'drop_detectors', []))), + 'no_verification': bool(scan_kwargs.get('no_verification', False)), + 'detector_config_sha256': config_sha256, + 'strict_git_provider_token_filter': bool( + getattr(scan_config, 'strict_git_provider_token_filter', True) + ), + 'docker_concurrency': max( + 0, + min(64, int(scan_kwargs.get('trufflehog_concurrency', 0) or 0)), + ), + 'candidate_normalizer_sha256': candidate_normalizer_sha256, + 'finding_identity_version': 'scanner-db-finding-identity-v1', + } + encoded = json.dumps( + payload, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + return hashlib.sha256(encoded).hexdigest() + + +def docker_planning_failure_result( + args, claim, scan_kwargs, failure, configured_mode, effective_mode, + canary_eligible=False, adaptive_gate=None, +): + adaptive_gate = adaptive_gate if isinstance(adaptive_gate, dict) else {} + now = datetime.now().isoformat() + message = redact_secrets(str(failure.error), [ + str(scan_kwargs.get('token') or ''), + str(getattr(args, 'token', '') or ''), + str(getattr(args, 'docker_token', '') or ''), + ]) + result = { + 'findings': [], + 'errors': [f'Docker layer planning failed: {message[:500]}'], + 'error_class': failure.category, + 'retryable': failure.retryable, + 'source_failure': failure.source_failure, + 'source_failure_category': failure.category if failure.source_failure else '', + 'source_failure_auth_related': failure.category == 'docker_auth', + 'scan_meta': { + 'docker_scan_assignment': { + 'configured_mode': configured_mode, + 'effective_mode': effective_mode, + 'canary_basis_points': max(0, min( + 10000, + int(getattr( + args, + 'docker_adaptive_canary_basis_points' + if configured_mode in ('adaptive-canary', 'adaptive') + else 'docker_layer_canary_basis_points', + 0, + ) or 0), + )), + 'canary_eligible': bool(canary_eligible), + 'adaptive_gate_passed': bool(adaptive_gate.get('passed')), + 'adaptive_gate_reason': str(adaptive_gate.get('reason') or 'not_configured'), + 'adaptive_gate_report_id': adaptive_gate.get('report_id'), + 'status': 'planning_failed', + }, + }, + 'target': str(claim['target']), + 'scan_type': 'docker', + 'scan_event_id': str(claim['scan_event_id']), + 'scan_started_at': now, + 'duration_sec': 0.0, + 'timestamp': now, + } + diagnostic_http = getattr(failure.error, 'diagnostic_http', None) + if isinstance(diagnostic_http, dict): + result['_diagnostic_http'] = dict(diagnostic_http) + assign_finding_uids(result) + return result + + +def resolve_and_bind_docker_claim( + args, db_url, source, claim, scan_kwargs, scan_policy_sha256, *, deadline=None, + adaptive=False, +): + if deadline is None: + deadline = time.monotonic() + max( + 0.001, float(getattr(args, 'timeout', 600) or 0.001), + ) + try: + resolved, bearer_auth = resolve_docker_content_manifest( + claim['target'], + getattr(args, 'docker_platform_os', 'linux'), + getattr(args, 'docker_platform_arch', 'amd64'), + deadline=deadline, + ) + except DockerRemoteAccessError as exc: + retryable, source_failure, category = { + 'rate_limited': (True, True, 'docker_rate_limit'), + 'auth_failed': (True, True, 'docker_auth'), + 'target_forbidden': (False, False, 'docker_target_forbidden'), + 'remote_transient': (True, True, 'remote_transient'), + }.get(exc.status, (True, True, 'remote_transient')) + raise _DockerPlanningFailure( + exc, retryable=retryable, source_failure=source_failure, + category=category, + ) from exc + except ApiRequestError as exc: + raise _DockerPlanningFailure( + exc, retryable=True, source_failure=True, category='remote_transient', + ) from exc + except (DockerRegistryResolutionError, ValueError) as exc: + raise _DockerPlanningFailure( + exc, retryable=False, source_failure=False, category='invalid_target', + ) from exc + + payload_classes = None + if adaptive: + try: + payload_classes, bearer_auth = fetch_docker_config_payload_classes( + resolved, bearer_auth, deadline=deadline, + min_free_bytes=max( + 0, int(getattr(args, 'docker_layer_min_free_bytes', 20 << 30) or 0), + ), + ) + except DockerLayerInfrastructureError as exc: + raise _DockerPlanningFailure( + exc, retryable=True, source_failure=bool(exc.source_failure), + category=str(exc.category or 'remote_transient'), + ) from exc + except DockerContentTransferError as exc: + raise _DockerPlanningFailure( + exc, retryable=bool(exc.retryable), source_failure=False, + category='docker_config_transfer', + ) from exc + + planning_db = ScannerDB(db_url=db_url, initialize=False) + try: + if not planning_db.enabled: + raise RuntimeError('Docker layer planning database is unavailable') + planning_db.set_application_name(f'truf-docker-layer-plan:{source}') + plan = planning_db.bind_docker_layer_plan( + claim['reservation_id'], claim['claim_lease_token'], resolved, + docker_layer_limits(args), scan_policy_sha256, + blob_lease_seconds=max( + int(getattr(args, 'docker_layer_blob_lease_sec', 1800) or 1800), + int(getattr(args, 'timeout', 600) or 600) + 300, + ), + payload_classes=payload_classes, + checkpoint=docker_adaptive_checkpoint(args) if adaptive else None, + ) + finally: + planning_db.close() + plan_bytes = canonical_docker_layer_plan_bytes(plan) + return { + 'plan': plan, + 'plan_sha256': hashlib.sha256(plan_bytes).hexdigest(), + 'bearer_auth': bearer_auth, + 'deadline': deadline, + 'min_free_bytes': max( + 0, int(getattr(args, 'docker_layer_min_free_bytes', 20 << 30) or 0), + ), + } + + +def git_resolution_failure_result(args, claim, scan_kwargs, failure): + error = failure.error + token = str(scan_kwargs.get('token') or '') + message = redact_secrets(str(error), [token]) + category = str(getattr(error, 'category', '') or '') + auth_related = bool(getattr(error, 'auth_related', False)) + if failure.invalid_target: + error_class = 'invalid_target' + category = 'invalid_target' + retryable = False + source_failure = False + elif isinstance(error, RateLimitError): + retryable = bool(getattr(error, 'retryable', True)) + source_failure = category != 'not_found' + error_class = 'source_auth' if auth_related else category or 'remote_transient' + else: + error_class = 'remote_transient' + category = category or 'remote_transient' + retryable = True + source_failure = True + auth_related = False + now = datetime.now().isoformat() + result = { + 'findings': [], + 'errors': [f'Exact Git ref resolution failed: {message}'], + 'error_class': error_class, + 'retryable': retryable, + 'source_failure': source_failure, + 'source_failure_category': category if source_failure else '', + 'source_failure_auth_related': auth_related if source_failure else False, + 'scan_meta': { + 'exact_git_resolution': { + 'status': 'failed', + 'provider': str(args.platform), + 'category': category, + 'retryable': retryable, + 'auth_related': auth_related, + }, + }, + 'target': str(claim['target']), + 'scan_type': str(args.platform), + 'scan_event_id': str(claim['scan_event_id']), + 'scan_started_at': now, + 'duration_sec': 0.0, + 'timestamp': now, + } + diagnostic_http = getattr(error, 'diagnostic_http', None) + if isinstance(diagnostic_http, dict): + result['_diagnostic_http'] = dict(diagnostic_http) + assign_finding_uids(result) + return result + + +def resolve_and_bind_git_claim( + args, db_url, source, claim, scan_kwargs, *, remote_credential=None, +): + try: + normalize_git_scan_resolution_target(claim['target'], args.platform) + except ValueError as exc: + raise _GitResolutionFailure(exc, invalid_target=True) from exc + try: + resolved = resolve_git_scan_target( + claim['target'], args.platform, scan_kwargs.get('token'), + request_attempts=max( + 1, int(getattr(args, 'git_ref_resolution_attempts', 2) or 2), + ), + timeout_sec=max( + 0.1, float(getattr(args, 'git_ref_resolution_timeout_sec', 10) or 10), + ), + max_response_bytes=max( + 1024, + int(getattr(args, 'git_ref_resolution_max_bytes', 1 << 20) or (1 << 20)), + ), + ) + except (RateLimitError, ApiRequestError, ValueError) as exc: + raise _GitResolutionFailure(exc) from exc + planning_db = ScannerDB(db_url=db_url, initialize=False) + try: + if not planning_db.enabled: + raise RuntimeError('exact Git planning database is unavailable') + planning_db.set_application_name(f'truf-git-plan:{source}') + binding_options = {} + if remote_credential is not None: + binding_options['remote_credential'] = remote_credential + return planning_db.bind_git_scan_plan( + claim['reservation_id'], claim['claim_lease_token'], resolved, + max(1, int(getattr(args, 'git_baseline_depth', 100) or 100)), + **binding_options, + ) + finally: + planning_db.close() + + +def run_discovery_cycle( + args, db, run_id, cycle_id, source_name=None, partial_metrics=None, +): + source = source_name or args.platform + expected_platforms = { + 'gitlab': 'gitlab', + 'dockerhub': 'docker', + 'huggingface': 'huggingface', + } + if source not in expected_platforms or args.platform != expected_platforms[source]: + raise ValueError('Discovery-only cycles support GitLab, DockerHub, and HuggingFace') + if not db or not getattr(db, 'conn', None) or not getattr(db.conn, 'is_postgres', False): + raise RuntimeError('Discovery-only cycles require PostgreSQL') + if not run_id or not cycle_id: + raise RuntimeError('Discovery-only cycles require valid run_id and cycle_id') + db.require_runtime_safety_schema() + db.require_final_cutover() + partial_metrics = partial_metrics if isinstance(partial_metrics, dict) else {} + + status = 'completed' + message = None + discovery_info = { + 'fetched_count': 0, + 'queued_new_count': 0, + 'queued_updated_count': 0, + } + control = db.runtime_control_state() + if control['effective_discovery_paused']: + status = 'paused' + message = 'Discovery paused by runtime control.' + else: + try: + _require_discovery_provider_enabled(db) + if managed_dockerhub_discovery(args, db, source): + collection_only = bool( + getattr(args, 'docker_depth_collection_only', False) + ) + experiment_stopped = False + if not collection_only: + _, experiment_stopped = resolve_due_docker_experiment_targets( + db, source, args, + ) + incremental = run_dockerhub_incremental_discovery( + args, db, source, partial_metrics=discovery_info, + ) + discovery_info.update(incremental) + status = str(discovery_info.get('cycle_status') or 'completed') + if not collection_only and not experiment_stopped: + resolve_due_docker_queue_targets(db, source, args) + else: + fetched_targets = fetch_targets(args, db, run_id, cycle_id, source) + _, discovery_info = enqueue_discovered_targets( + args, fetched_targets or [], db, run_id, cycle_id, source, + partial_metrics=partial_metrics, + ) + except DiscoveryPausedError: + status = 'paused' + message = 'Discovery paused before provider work or target admission.' + + metrics = { + **summarize_results([]), + **discovery_info, + 'scan_requested_count': 0, + 'staged_count': 0, + 'backlog_only': False, + 'docker_depth_collection_only': bool( + args.platform == 'docker' + and getattr(args, 'docker_depth_collection_only', False) + ), + 'cycle_status': status, + } + partial_metrics.update(metrics) + db.finish_source_cycle( + cycle_id, + status, + metrics, + { + 'todo_count': 0, + 'checked_count': 0, + 'todo_file': None, + 'checked_file': None, + }, + message, + ) + return metrics + + +def run_cycle_v2(args, db, run_id, cycle_id, source_name, partial_metrics=None): + source = source_name or args.platform + partial_metrics = partial_metrics if isinstance(partial_metrics, dict) else {} + docker_depth_collection_only = bool( + args.platform == 'docker' + and getattr(args, 'docker_depth_collection_only', False) + ) + db.require_runtime_safety_schema() + db.require_final_cutover() + bundle_root = require_private_directory( + getattr(args, 'result_bundle_dir', None) or scan_config.result_bundle_dir, + create=False, + ) + minimum_free = max(0, int(getattr(args, 'result_bundle_min_free_bytes', 20 * 1024 * 1024 * 1024))) + max_event_bytes = max(1, int(getattr(args, 'result_bundle_max_event_bytes', 64 * 1024 * 1024))) + max_targets = max(0, int(getattr(args, 'max_targets', 0) or 0)) + configured_slots = max(1, int(getattr(scan_config, 'max_active_scans', 3) or 3)) + capacity_slots = configured_slots + max(0, min(1, int(getattr( + scan_config, 'opportunistic_scan_slots', 0, + ) or 0))) + dispatch_limit = min(max(1, int(getattr(args, 'workers', 1) or 1)), configured_slots) + if max_targets: + dispatch_limit = min(dispatch_limit, max_targets) + validate_v2_capacity_model( + capacity_slots, + max_event_bytes, + int(getattr(args, 'projection_backlog_max_bytes', 2 * 1024 * 1024 * 1024)), + int(getattr(args, 'projection_backlog_headroom_bytes', max_event_bytes * 2)), + ) + producer_identity = current_process_identity() + producer_mapping = producer_identity.as_dict() + supervisor_instance_id = str(os.getenv('TRUF_SUPERVISOR_INSTANCE_ID') or '') + if not supervisor_instance_id: + raise RuntimeError('v2 source admission requires an authenticated supervisor instance ID') + + fetched_targets = [] + if args.platform == 'docker' and not docker_depth_collection_only: + resolve_due_docker_experiment_targets(db, source, args) + has_backlog = db.has_claimable_targets_v2( + source, args.platform, + max_attempts=int(getattr(args, 'target_retry_max_attempts', 3) or 3), + ) + if db.last_error: + raise RuntimeError(f'Unable to inspect target backlog for {source}: {db.last_error}') + todo_file, checked_file = queue_files_for_args(args) + discovery_info = { + 'fetched_count': 0, 'queued_new_count': 0, 'queued_updated_count': 0, + 'scan_requested_count': 0, + 'projection_error': '', + 'cycle_status': 'completed', + 'deep_dispatch_durable': False, + } + refresh_backlog = bool(getattr(args, 'refresh_registry', False)) + backlog_only = bool(has_backlog and not refresh_backlog) + if not has_backlog or refresh_backlog: + incremental_info = None + if managed_dockerhub_discovery(args, db, source): + incremental_info = run_dockerhub_incremental_discovery(args, db, source) + else: + fetched_targets = fetch_targets(args, db, run_id, cycle_id, source) + _, todo_file, checked_file, discovery_info = prepare_targets( + args, fetched_targets or [], db, run_id, cycle_id, source, enqueue_only=True, + partial_metrics=partial_metrics, experiment_resolver_already_run=True, + ) + if incremental_info is not None: + discovery_info.update(incremental_info) + partial_metrics.update({ + key: value for key, value in discovery_info.items() + if key in { + 'fetched_count', 'queued_new_count', 'queued_updated_count', + 'discovery_pages_fetched', 'discovery_retry_enqueued_count', + 'discovery_retry_inserted_count', 'discovery_retry_coalesced_count', + 'discovery_preexisting_count', 'discovery_known_page_count', + 'discovery_stopped_on_preexisting', 'discovery_deep', + 'discovery_pass_kind', 'deep_dispatch_durable', 'cycle_status', + } + }) + + scan_kwargs = prepare_scan_options(args, dispatch_limit) + event_scan_options = {key: value for key, value in scan_kwargs.items() if key != 'token'} + docker_configured_mode = str( + getattr(args, 'docker_content_scan_mode', 'full') or 'full' + ).strip().lower() + docker_canary_basis_points = max(0, min( + 10000, int(getattr(args, 'docker_layer_canary_basis_points', 0) or 0), + )) + docker_scan_policy_sha256 = None + docker_execution_policy_sha256 = None + docker_selection_policy_sha256 = None + docker_limits_value = None + docker_gate_max_age_sec = 604800 + if args.platform == 'docker': + if docker_configured_mode not in ( + 'full', 'canary', 'layer', 'adaptive-canary', 'adaptive', + ): + raise ValueError( + 'Docker content scan mode must be full, canary, layer, ' + 'adaptive-canary, or adaptive' + ) + docker_scan_policy_sha256 = docker_layer_scan_policy_sha256(args, scan_kwargs) + event_scan_options['scan_policy_sha256'] = docker_scan_policy_sha256 + if docker_configured_mode == 'layer' or ( + docker_configured_mode == 'canary' and docker_canary_basis_points > 0 + ) or docker_configured_mode in ('adaptive-canary', 'adaptive'): + docker_limits_value = docker_layer_limits(args) + if docker_configured_mode in ('adaptive-canary', 'adaptive'): + docker_execution_policy_sha256 = docker_layer_execution_policy_sha256( + docker_scan_policy_sha256, docker_limits_value, + ) + docker_selection_policy_sha256 = docker_layer_selection_policy_sha256( + docker_limits_value, + ) + docker_adaptive_checkpoint(args) + docker_gate_max_age_sec = int(getattr( + args, 'docker_adaptive_gate_max_age_sec', 604800, + )) + if not 60 <= docker_gate_max_age_sec <= 2592000: + raise ValueError( + 'Docker adaptive gate freshness must be between 60 and 2592000 seconds' + ) + capacity_limits = { + 'bundle_items': int(getattr(args, 'result_bundle_max_items', 10000)), + 'bundle_bytes': int(getattr(args, 'result_bundle_max_total_bytes', 3 * 1024 * 1024 * 1024)), + 'projection_items': int(getattr(args, 'projection_backlog_max_items', 10000)), + 'projection_bytes': int(getattr(args, 'projection_backlog_max_bytes', 2 * 1024 * 1024 * 1024)), + 'keycheck_items': int(getattr(args, 'keycheck_queue_max_items', 100000)), + 'keycheck_bytes': int(getattr(args, 'keycheck_queue_max_bytes', 512 * 1024 * 1024)), + 'quarantine_items': int(getattr(args, 'pipeline_quarantine_max_items', 10000)), + 'quarantine_bytes': int(getattr(args, 'pipeline_quarantine_max_bytes', 1024 * 1024 * 1024)), + } + candidate_items = int(getattr(args, 'keycheck_candidates_per_event', 2000)) + candidate_bytes = int(getattr(args, 'keycheck_candidate_bytes_per_event', 2 * 1024 * 1024)) + queue_policy = QueueDispositionPolicy( + target_retry_max_attempts=int(getattr(args, 'target_retry_max_attempts', 3) or 3), + target_retry_base_delay_sec=int(getattr(args, 'target_retry_base_delay_sec', 3600) or 3600), + target_retry_max_delay_sec=int(getattr(args, 'target_retry_max_delay_sec', 86400) or 86400), + target_timeout_retry_delay_sec=int(getattr(args, 'target_timeout_retry_delay_sec', 21600) or 21600), + docker_layer_checkpoint_delay_sec=int(getattr(args, 'docker_layer_checkpoint_delay_sec', 60) or 60), + ci_soft_cooldown_days=int(getattr(args, 'ci_soft_cooldown_days', 7) or 7), + soft_skip_reasons=tuple(sorted(CI_SOFT_SKIP_REASONS.get(args.platform, set()))), + ) + lease_seconds = max(1800, int(getattr(args, 'timeout', 0) or 0) + 900) + require_s_drive = bool(os.name == 'nt' and os.getenv('SCANNER_SUPERVISED') == '1') + + def reserve_with_exact_recovery(permit_box): + reservation_token = secrets.token_urlsafe(32) + bundle_id = secrets.token_hex(16) + scan_event_id = secrets.token_hex(16) + + def release_permit(): + lease = permit_box.get('lease') + if lease is not None: + lease.release() + permit_box['lease'] = None + + return reserve_v2_admission_with_recovery( + db.url, source, args.platform, producer_mapping, supervisor_instance_id, + max_event_bytes, max_event_bytes * 2, candidate_items, candidate_bytes, + lease_seconds=lease_seconds, + max_attempts=int(getattr(args, 'target_retry_max_attempts', 3) or 3), + capacity_limits=capacity_limits, run_id=run_id, cycle_id=cycle_id, + reservation_token=reservation_token, bundle_id=bundle_id, + scan_event_id=scan_event_id, + resolution_attempts=int(getattr(args, 'admission_resolution_attempts', 8) or 8), + resolution_seconds=float(getattr(args, 'admission_resolution_seconds', 30) or 30), + retry_delay=float( + getattr(args, 'admission_resolution_retry_delay_sec', 0.2) or 0.2 + ), + claim_order=str(getattr(args, 'target_claim_order', 'oldest') or 'oldest'), + docker_depth_authority=getattr( + args, 'docker_depth_experiment_authority', None, + ), + final_cutover=True, + release_permit=release_permit, + ) + + def notify_ready(claim, staged): + notification = ScannerDB(db_url=db.url, initialize=False) + try: + if not notification.enabled: + return False + notification.set_application_name(f'truf-bundle-notify:{source}') + return notification.mark_result_bundle_ready( + claim['reservation_id'], + { + 'bundle_id': staged.bundle_id, + 'scan_event_id': staged.scan_event_id, + 'scan_event_hash': staged.scan_event_hash, + 'relative_path': staged.relative_path, + 'actual_bytes': staged.actual_bytes, + 'frame_count': staged.frame_count, + 'finding_count': staged.finding_count, + 'error_count': staged.error_count, + 'candidate_count': staged.candidate_count, + }, + ) + except Exception as exc: + logger.warning( + 'Ready bundle DB notification failed after durable handoff event=%s: %s', + staged.scan_event_id, exc, + ) + return False + finally: + notification.close() + + def stage_claim(claim, lease): + claim_deadline = time.monotonic() + max( + 0.001, float(getattr(args, 'timeout', 600) or 0.001), + ) + with scan_slot_scope( + ['scan-target', args.platform], getattr(args, 'timeout', None), lease=lease, + ): + claim_scan_kwargs = scan_kwargs + if ( + args.platform in ('github', 'gitlab') + and bool(getattr(args, 'exact_git_planning_enabled', False)) + ): + try: + git_plan = resolve_and_bind_git_claim( + args, db.url, source, claim, scan_kwargs, + ) + except _GitResolutionFailure as failure: + result = git_resolution_failure_result( + args, claim, scan_kwargs, failure, + ) + else: + claim_scan_kwargs = dict(scan_kwargs) + claim_scan_kwargs['git_plan'] = git_plan + result = scan_target_result( + claim['target'], args.platform, claim['scan_event_id'], claim_scan_kwargs, + ) + elif args.platform == 'docker': + canary_eligible = False + adaptive_gate = {'passed': False, 'reason': 'not_configured'} + try: + if ( + docker_configured_mode == 'canary' + and docker_canary_basis_points > 0 + ): + canary_eligible = docker_layer_timeout_canary_eligible( + db.url, source, claim, + ) + if docker_configured_mode in ('adaptive-canary', 'adaptive'): + adaptive_gate = docker_adaptive_gate_report( + db.url, source, claim, docker_scan_policy_sha256, + docker_execution_policy_sha256, + docker_selection_policy_sha256, + docker_gate_max_age_sec, + ) + effective_mode, basis_points = docker_layer_effective_mode( + args, claim['target'], canary_eligible, + adaptive_gate_passed=bool(adaptive_gate.get('passed')), + selection_policy_sha256=docker_selection_policy_sha256, + ) + except ValueError as exc: + failure = _DockerPlanningFailure( + exc, retryable=False, source_failure=False, category='invalid_target', + ) + result = docker_planning_failure_result( + args, claim, scan_kwargs, failure, + docker_configured_mode, + 'adaptive' if docker_configured_mode in ('adaptive-canary', 'adaptive') + else 'layer', + canary_eligible, adaptive_gate, + ) + effective_mode = ( + 'adaptive' if docker_configured_mode in ('adaptive-canary', 'adaptive') + else 'layer' + ) + basis_points = max(0, min(10000, int(getattr( + args, + 'docker_adaptive_canary_basis_points' + if docker_configured_mode in ('adaptive-canary', 'adaptive') + else 'docker_layer_canary_basis_points', + 0, + ) or 0))) + else: + if effective_mode in ('layer', 'adaptive'): + try: + layer_work = resolve_and_bind_docker_claim( + args, db.url, source, claim, scan_kwargs, + docker_scan_policy_sha256, deadline=claim_deadline, + adaptive=effective_mode == 'adaptive', + ) + except _DockerPlanningFailure as failure: + result = docker_planning_failure_result( + args, claim, scan_kwargs, failure, + docker_configured_mode, effective_mode, canary_eligible, + adaptive_gate, + ) + else: + claim_scan_kwargs = dict(scan_kwargs) + claim_scan_kwargs['docker_layer_work'] = layer_work + result = scan_target_result( + claim['target'], args.platform, claim['scan_event_id'], + claim_scan_kwargs, + ) + else: + result = scan_target_result( + claim['target'], args.platform, claim['scan_event_id'], + claim_scan_kwargs, + ) + scan_meta = result.get('scan_meta') + if not isinstance(scan_meta, dict): + scan_meta = {} + result['scan_meta'] = scan_meta + scan_meta.setdefault('docker_scan_assignment', { + 'configured_mode': docker_configured_mode, + 'effective_mode': effective_mode, + 'canary_basis_points': basis_points, + 'canary_eligible': bool(canary_eligible), + 'adaptive_gate_passed': bool(adaptive_gate.get('passed')), + 'adaptive_gate_reason': str( + adaptive_gate.get('reason') or 'not_configured' + ), + 'adaptive_gate_report_id': adaptive_gate.get('report_id'), + 'selection_policy_sha256': docker_selection_policy_sha256, + 'status': 'executed', + }) + else: + result = scan_target_result( + claim['target'], args.platform, claim['scan_event_id'], claim_scan_kwargs, + ) + staged = stage_scan_result_in_scope( + result, claim, bundle_root, event_scan_options, queue_policy, + attempts=claim.get('attempts'), + candidate_max_items=candidate_items, + candidate_max_bytes=candidate_bytes, + require_s_drive=require_s_drive, + ) + # Notification is deliberately outside the permit; exact-path recovery handles failure. + notify_ready(claim, staged) + return staged + + refunded_claim = object() + + def recover_worker_failure(claim, exc): + ready_relative = str(claim['ready_relative_path']).replace('\\', '/') + inspection = inspect_private_relative_path(bundle_root, ready_relative) + try: + if inspection.state == PrivatePathState.PRESENT: + reader = ResultBundleReader( + inspection.path, max_event_bytes=max_event_bytes, + ) + validated = reader.validate() + metadata = reader.metadata() + staged = StagedResult( + target=str(claim['target']), scan_event_id=validated.scan_event_id, + bundle_id=validated.bundle_id, reservation_id=validated.reservation_id, + scan_event_hash=validated.scan_event_hash, actual_bytes=validated.actual_bytes, + relative_path=str(claim['ready_relative_path']).replace('\\', '/'), + frame_count=validated.frame_count, finding_count=validated.finding_count, + error_count=validated.error_count, candidate_count=validated.candidate_count, + queue_status=str(metadata.get('queue_status') or ''), + source_failure=bool(metadata.get('source_failure')), + source_failure_category=str(metadata.get('source_failure_category') or ''), + source_failure_auth_related=bool(metadata.get('source_failure_auth_related')), + first_error=str(metadata.get('first_error_summary') or ''), + ) + notify_ready(claim, staged) + return staged + if inspection.state == PrivatePathState.UNKNOWN: + logger.error( + 'Bundle handoff state is unknown; reservation retained id=%s error=%s', + claim['reservation_id'], inspection.detail, + ) + return None + refunded = refund_v2_claim_after_no_handoff( + db, bundle_root, claim, producer_mapping, + f'pre-handoff source failure: {type(exc).__name__}: {exc}', + ) + if not refunded: + logger.error( + 'Exact pre-handoff reservation refund was not confirmed; recovery retains it: %s', + claim['reservation_id'], + ) + return None + return refunded_claim + except (OSError, ValueError) as inspection_error: + logger.error( + 'Bundle handoff state is unknown; reservation retained for ingester recovery id=%s error=%s', + claim['reservation_id'], inspection_error, + ) + return None + + staged_count = 0 + source_failure_count = 0 + first_source_failure = None + pending = {} + claimed_count = 0 + admission_open = not docker_depth_collection_only + unresolved_handoff_error = None + with concurrent.futures.ThreadPoolExecutor(max_workers=dispatch_limit) as executor: + while pending or admission_open: + while admission_open and len(pending) < dispatch_limit and (not max_targets or claimed_count < max_targets): + if supervised_spool_stop_requested(): + admission_open = False + break + free_bytes = shutil.disk_usage(bundle_root).free + if minimum_free and free_bytes < minimum_free: + admission_open = False + break + lease = acquire_scan_slot( + ['dispatch-target', args.platform], getattr(args, 'timeout', None), + wait=True, start_heartbeat=False, + ) + permit_box = {'lease': lease} + try: + outcome = reserve_with_exact_recovery(permit_box) + except BaseException: + if permit_box.get('lease') is not None: + permit_box['lease'].release() + raise + claim = outcome.claim + if not claim: + if permit_box.get('lease') is not None: + permit_box['lease'].release() + if outcome.retry_without_claim: + continue + admission_open = False + break + try: + ensure_bundle_reservation_paths(bundle_root, claim) + except BaseException: + if permit_box.get('lease') is not None: + permit_box['lease'].release() + raise + if outcome.permit_released: + try: + permit_box['lease'] = reacquire_scan_permit_bounded( + ['dispatch-target', args.platform], getattr(args, 'timeout', None), + float(getattr(args, 'admission_permit_reacquire_seconds', 30) or 30), + ) + except BaseException as exc: + if recover_worker_failure(claim, exc) is not refunded_claim: + raise UnresolvedHandoffInfrastructureError( + 'confirmed admission claim could not reacquire a permit or refund exactly' + ) from exc + continue + lease = permit_box.get('lease') + if lease is None and scan_limiter_enabled(): + if recover_worker_failure( + claim, RuntimeError('confirmed claim has no physical scan permit'), + ) is not refunded_claim: + raise UnresolvedHandoffInfrastructureError( + 'confirmed admission claim lost its physical scan permit' + ) + continue + claimed_count += 1 + future = executor.submit(stage_claim, claim, lease) + pending[future] = claim + if not pending: + break + done, _ = concurrent.futures.wait( + tuple(pending), timeout=0.2, + return_when=concurrent.futures.FIRST_COMPLETED, + ) + for future in done: + claim = pending.pop(future) + try: + staged = future.result() + except Exception as exc: + staged = recover_worker_failure(claim, exc) + refunded = staged is refunded_claim + source_outage_refunded = bool( + refunded and isinstance(exc, DockerLayerInfrastructureError) + ) + if refunded: + staged = None + if source_outage_refunded: + source_failure_count += 1 + if first_source_failure is None: + first_source_failure = SimpleNamespace( + source_failure_category=exc.category, + source_failure_auth_related=exc.auth_related, + first_error=str(exc)[:300], + ) + admission_open = False + if staged is None: + if not source_outage_refunded: + logger.error('Target staging failed before durable handoff: %s', exc) + unresolved_handoff_error = UnresolvedHandoffInfrastructureError( + 'deterministic handoff cleanup remained unknown; source admission is closed' + ) + admission_open = False + if staged is not None: + staged_count += 1 + if staged.source_failure: + source_failure_count += 1 + if first_source_failure is None: + first_source_failure = staged + admission_open = False + safe_print( + f'Staged {staged_count} event(s): {staged.target} ' + f'findings={staged.finding_count} errors={staged.error_count}', + flush=True, + ) + + if unresolved_handoff_error is not None: + raise unresolved_handoff_error + + cycle_status = str(discovery_info.get('cycle_status') or 'completed') + if backlog_only: + cycle_status = 'backlog_only' + if source_failure_count: + cycle_status = 'source_failed' + metrics = { + 'authoritative_async': True, + 'fetched_count': discovery_info.get('fetched_count', len(fetched_targets)), + 'queued_new_count': discovery_info.get('queued_new_count', 0), + 'queued_updated_count': discovery_info.get('queued_updated_count', 0), + 'scan_requested_count': claimed_count, + 'staged_count': staged_count, + 'scanned_count': 0, + 'clean_count': 0, + 'found_count': 0, + 'skipped_count': 0, + 'error_count': 0, + 'findings_count': 0, + 'verified_findings_count': 0, + 'unique_secrets_count': 0, + 'unique_findings_count': 0, + 'backlog_only': bool(backlog_only), + 'docker_depth_collection_only': docker_depth_collection_only, + 'source_failure_count': source_failure_count, + 'source_failure_category': first_source_failure.source_failure_category if first_source_failure else '', + 'source_failure_auth_related': first_source_failure.source_failure_auth_related if first_source_failure else False, + 'source_failure_message': first_source_failure.first_error if first_source_failure else '', + 'source_failure_queue_ids': [], + 'cycle_status': cycle_status, + 'deep_dispatch_durable': bool(discovery_info.get('deep_dispatch_durable', False)), + } + for key in ( + 'discovery_pages_fetched', 'discovery_retry_enqueued_count', + 'discovery_retry_inserted_count', 'discovery_retry_coalesced_count', + 'discovery_preexisting_count', 'discovery_known_page_count', + 'discovery_stopped_on_preexisting', 'discovery_deep', + 'discovery_pass_kind', + ): + if key in discovery_info: + metrics[key] = discovery_info[key] + partial_metrics.update(metrics) + if cycle_id: + db.finish_source_cycle( + cycle_id, cycle_status, metrics, + {'todo_count': 0, 'checked_count': 0, 'todo_file': None, 'checked_file': None}, + metrics.get('source_failure_message') or None, + ) + safe_print( + f'Cycle staged {staged_count} durable bundle(s); PostgreSQL ingestion is asynchronous.', + flush=True, + ) + return metrics + + +def _run_cycle_legacy_compat( + args, db=None, run_id=None, cycle_id=None, source_name=None, partial_metrics=None, +): + source_for_db = source_name or args.platform + partial_metrics = partial_metrics if isinstance(partial_metrics, dict) else {} + postgres_db = bool(db and getattr(db, 'conn', None) and getattr(db.conn, 'is_postgres', False)) + spool = None + ingest_outcomes = {} + if postgres_db: + db.require_runtime_safety_schema() + spool = result_spool_for_args(args) + wait_for_result_spool_ready( + spool, + db, + outcomes=ingest_outcomes, + stop_event=getattr(args, 'result_spool_stop_event', None), + wait_seconds=float(getattr(args, 'result_spool_wait_sec', 1.0) or 1.0), + diagnostic_interval=float(getattr(args, 'result_spool_diagnostic_interval_sec', 30.0) or 30.0), + ) + drained = drain_scan_publication_outbox(db, 100) + if drained: + print(f'Published {drained} pending scanner outbox event(s).') + configured_slots = max(1, int(getattr(scan_config, 'max_active_scans', 1) or 1)) + dispatch_limit = min(max(1, int(getattr(args, 'workers', 1) or 1)), configured_slots) + if args.max_targets: + dispatch_limit = min(dispatch_limit, max(1, int(args.max_targets))) + require_scan_publication_capacity(db, args, additional_items=dispatch_limit) + else: + dispatch_limit = 0 + + slot_first_dispatch = bool(postgres_db and scan_limiter_enabled()) + + def has_backlog(): + present = db.has_claimable_targets( + source_for_db, + args.platform, + max_attempts=int(getattr(args, 'target_retry_max_attempts', 3) or 3), + ) + if db.last_error: + raise RuntimeError(f'Unable to inspect target backlog for {source_for_db}: {db.last_error}') + return present + + def claim_with_dispatch_slots(): + leases = acquire_scan_slot_leases( + ['dispatch-target', args.platform], dispatch_limit, getattr(args, 'timeout', None), + ) + if not leases: + raise RuntimeError('slot-first dispatch did not acquire a configured scan slot') + try: + return prepare_targets( + args, + [], + db, + run_id, + cycle_id, + source_for_db, + spool=spool, + claim_limit_override=len(leases), + dispatch_leases=leases, + ) + except BaseException: + for lease in leases: + if lease.heartbeat_thread is None: + lease.release() + raise + + fetched_targets = [] + backlog_only = False + prepared = None + initial_backlog = has_backlog() if slot_first_dispatch else False + if initial_backlog and args.platform == 'docker': + resolve_due_docker_experiment_targets(db, source_for_db, args) + if slot_first_dispatch and initial_backlog: + prepared = claim_with_dispatch_slots() + backlog_only = bool(prepared[0]) + + if not backlog_only: + fetched_targets = fetch_targets(args, db, run_id, cycle_id, source_for_db) + if not fetched_targets: + print('No targets fetched. This may be an empty result or an API limit/error.') + if slot_first_dispatch: + _, todo_file, checked_file, discovery_info = prepare_targets( + args, + fetched_targets or [], + db, + run_id, + cycle_id, + source_for_db, + spool=spool, + enqueue_only=True, + partial_metrics=partial_metrics, + ) + partial_metrics.update({ + 'fetched_count': discovery_info.get('fetched_count', len(fetched_targets)), + 'queued_new_count': discovery_info.get('queued_new_count', 0), + 'queued_updated_count': discovery_info.get('queued_updated_count', 0), + }) + if has_backlog(): + prepared = claim_with_dispatch_slots() + prepared[3]['fetched_count'] = len(fetched_targets or []) + prepared[3]['queued_new_count'] = discovery_info.get('queued_new_count', 0) + prepared[3]['queued_updated_count'] = discovery_info.get('queued_updated_count', 0) + prepared[3]['projection_error'] = discovery_info.get('projection_error', '') + else: + prepared = ([], todo_file, checked_file, discovery_info) + else: + prepared = prepare_targets( + args, fetched_targets or [], db, run_id, cycle_id, source_for_db, spool=spool, + partial_metrics=partial_metrics, + ) + + targets_to_scan, todo_file, checked_file, queue_info = prepared + partial_metrics.update({ + 'fetched_count': queue_info.get('fetched_count', len(fetched_targets)), + 'queued_new_count': queue_info.get('queued_new_count', 0), + 'queued_updated_count': queue_info.get('queued_updated_count', 0), + }) + if args.max_targets and len(targets_to_scan) > args.max_targets: + targets_to_scan = targets_to_scan[:args.max_targets] + queue_info['scan_requested_count'] = len(targets_to_scan) + + if not targets_to_scan: + print('No new targets to scan.') + metrics = { + 'fetched_count': queue_info.get('fetched_count', len(fetched_targets)), + 'queued_new_count': queue_info.get('queued_new_count', 0), + 'queued_updated_count': queue_info.get('queued_updated_count', 0), + 'scan_requested_count': 0, + 'backlog_only': backlog_only, + **summarize_results([]), + } + partial_metrics.update(metrics) + if db and cycle_id: + counts = queue_counts(todo_file, checked_file) if ( + not postgres_db or bool(getattr(args, 'sync_file_queues', True)) + ) else { + 'todo_count': 0, + 'checked_count': 0, + 'todo_file': None, + 'checked_file': None, + } + db.finish_source_cycle(cycle_id, 'completed', metrics, counts, 'No new targets to scan') + return metrics + + queue_event_meta = {} + queue_claims = {key: dict(value) for key, value in (queue_info.get('queue_claims') or {}).items()} + active_lease_tokens = {claim.get('lease_token') for claim in queue_claims.values() if claim.get('lease_token')} + reserved_lease_tokens = set(active_lease_tokens) + active_lease_lock = threading.Lock() + reservation_id = queue_info.get('spool_reservation_id') + scan_slot_leases = list(queue_info.get('scan_slot_leases') or []) + claim_guard = PostClaimRefundGuard( + db, spool, reservation_id, queue_claims.values(), active_lease_tokens, active_lease_lock, + dispatch_leases=scan_slot_leases, + ) if postgres_db else None + + if claim_guard: + scan_kwargs = claim_guard.call( + 'infrastructure scan option preparation failure', + prepare_scan_options, + args, + len(targets_to_scan), + ) + else: + scan_kwargs = prepare_scan_options(args, len(targets_to_scan)) + + def progress(completed, total, target): + if target != 'Scan completed': + safe_print(f'Progress: {completed}/{total} finished - {target}', flush=True) + + if claim_guard: + event_scan_options = claim_guard.call( + 'infrastructure scan event option preparation failure', + lambda: {key: value for key, value in scan_kwargs.items() if key != 'token'}, + ) + else: + event_scan_options = {key: value for key, value in scan_kwargs.items() if key != 'token'} + + def spool_result(result): + assign_finding_uids(result) + queue_claim = queue_claims.get(str(result.get('target', ''))) or {} + queue_id = queue_claim.get('id') + lease_token = queue_claim.get('lease_token') + if queue_id is None or not lease_token: + raise RuntimeError(f"completed target has no fenced queue claim: {result.get('target', '')}") + skipped_reason = str(result.get('skipped') or '') + attempts = int(queue_claim.get('attempts') or 0) + max_attempts = max(1, int(getattr(args, 'target_retry_max_attempts', 3) or 3)) + reset_attempts = False + if skipped_reason in CI_SOFT_SKIP_REASONS.get(args.platform, set()): + cooldown_days = max(1, int(getattr(args, 'ci_soft_cooldown_days', 7) or 7)) + available_after = (datetime.now(timezone.utc) + timedelta(days=cooldown_days)).isoformat(timespec='seconds') + queue_status = 'deferred' + queue_error = skipped_reason + reset_attempts = True + elif result.get('errors'): + layer_disposition = docker_layer_queue_disposition( + result, args, attempts=attempts, + ) + if layer_disposition is not None: + queue_status, available_after, reset_attempts = layer_disposition + else: + queue_status, available_after, attempts, max_attempts = queue_error_disposition( + db, source_for_db, args.platform, result.get('target', ''), result, args, queue_claim + ) + reset_attempts = queue_result_resets_attempts(result) + queue_error = first_error_line(result) + else: + queue_status = 'done' + available_after = None + queue_error = None + derived_targets = collect_postman_targets([result]) + strip_nearby_context_for_persistence(result) + try: + event = prepare_scan_event({ + 'version': 1, + 'scan_event_id': result.get('scan_event_id'), + 'run_id': run_id, + 'cycle_id': cycle_id, + 'source': source_for_db, + 'query': args.query, + 'target': result.get('target', ''), + 'result': result, + 'scan_options': event_scan_options, + 'queue_id': queue_id, + 'claim_lease_token': lease_token, + 'claim_lease_owner': queue_claim.get('lease_owner'), + 'queue_status': queue_status, + 'queue_error': queue_error, + 'available_after': available_after, + 'reset_attempts': reset_attempts, + 'derived_postman_targets': derived_targets, + }) + record = write_reserved_result_spool_event( + spool, event, reservation_id, queue_id, lease_token, + reserved_lease_tokens, active_lease_tokens, active_lease_lock, + ) + except Exception as exc: + result['persistence_failure'] = str(exc)[:500] + with active_lease_lock: + active_lease_tokens.discard(lease_token) + reserved_lease_tokens.discard(lease_token) + refunded = db.refund_target_claim( + queue_id, lease_token, + f'infrastructure persistence failure: {exc}', + ) + reservation_released = False + if refunded: + try: + reservation_released = spool.release_reserved_claim( + reservation_id, queue_id, lease_token, + ) + except Exception: + logger.exception('Unable to release failed result-spool reservation') + logger.critical( + 'COMPLETED RESULT COULD NOT BE PERSISTED target=%s queue_id=%s claim_refunded=%s reservation_released=%s error=%s', + result.get('target', ''), queue_id, refunded, reservation_released, exc, + ) + if not refunded: + raise RuntimeError( + f'infrastructure persistence failed and claim could not be refunded for queue row {queue_id}: {exc}' + ) from exc + raise RuntimeError(f'infrastructure result persistence failure: {exc}') from exc + queue_event_meta[record.event_id] = { + 'queue_status': queue_status, + 'attempts': attempts, + 'max_attempts': max_attempts, + } + wait_for_result_spool_ready( + spool, + db, + outcomes=ingest_outcomes, + stop_event=getattr(args, 'result_spool_stop_event', None), + wait_seconds=float(getattr(args, 'result_spool_wait_sec', 1.0) or 1.0), + diagnostic_interval=float(getattr(args, 'result_spool_diagnostic_interval_sec', 30.0) or 30.0), + ) + outcome = ingest_outcomes.get(record.event_id) + if outcome is None: + confirm = getattr(db, 'confirmed_scan_event', None) + outcome = confirm(record.event_id, record.event_hash) if confirm else None + if not scan_outcome_matches(record, outcome): + raise RuntimeError(f'database did not confirm durable handoff for scan event {record.event_id}') + ingest_outcomes[record.event_id] = outcome + + heartbeat_stop = claim_guard.call( + 'infrastructure heartbeat stop-token setup failure', threading.Event, + ) if claim_guard else threading.Event() + heartbeat_thread = None + heartbeat_db = None + heartbeat_failed = claim_guard.call( + 'infrastructure heartbeat failure-token setup failure', threading.Event, + ) if claim_guard else threading.Event() + lease_owner = queue_info.get('lease_owner') + lease_seconds = int(queue_info.get('lease_seconds') or 0) + if db and lease_owner and lease_seconds: + try: + heartbeat_db = ScannerDB(db_path=getattr(db, 'path', None), db_url=getattr(db, 'url', None), initialize=False) + if not heartbeat_db.enabled: + raise RuntimeError(f'Unable to open dedicated lease heartbeat DB for {source_for_db}') + set_application_name = getattr(heartbeat_db, 'set_application_name', None) + if set_application_name: + set_application_name(f'truf-heartbeat:{source_for_db}') + heartbeat_db.require_runtime_safety_schema() + if getattr(heartbeat_db.conn, 'is_postgres', False): + heartbeat_db.conn.execute("SELECT set_config('statement_timeout', '10000ms', false)") + heartbeat_db.conn.execute("SELECT set_config('lock_timeout', '5000ms', false)") + heartbeat_db.conn.commit() + except Exception as exc: + if heartbeat_db is not None: + heartbeat_db.close() + heartbeat_db = None + claim_guard.refund_undurable(f'infrastructure heartbeat setup failure: {exc}') + raise + + def renew_leases(): + interval = max(10, min(60, lease_seconds // 3)) + last_success = time.monotonic() + failure_grace = max(60, lease_seconds // 2) + while not heartbeat_stop.wait(interval): + with active_lease_lock: + current_tokens = list(active_lease_tokens) + if not current_tokens: + return + renewed = heartbeat_db.renew_target_leases(lease_owner, lease_seconds, current_tokens) + if renewed == len(current_tokens): + if reservation_id: + try: + if not renew_active_result_spool_reservation( + spool, reservation_id, lease_seconds, + reserved_lease_tokens, active_lease_lock, + ): + raise RuntimeError('reservation is absent or expired') + except Exception as exc: + logger.critical('Result-spool reservation renewal failed: %s', exc) + heartbeat_failed.set() + return + last_success = time.monotonic() + continue + if recheck_active_lease_ownership( + heartbeat_db, lease_owner, current_tokens, active_lease_tokens, active_lease_lock, + ): + if reservation_id: + try: + if not renew_active_result_spool_reservation( + spool, reservation_id, lease_seconds, + reserved_lease_tokens, active_lease_lock, + ): + raise RuntimeError('reservation is absent or expired') + except Exception as exc: + logger.critical('Result-spool reservation renewal failed: %s', exc) + heartbeat_failed.set() + return + last_success = time.monotonic() + continue + if heartbeat_db.last_error and time.monotonic() - last_success < failure_grace: + continue + heartbeat_failed.set() + return + + try: + candidate_thread = threading.Thread(target=renew_leases, name=f'lease-heartbeat-{source_for_db}', daemon=True) + candidate_thread.start() + heartbeat_thread = candidate_thread + except Exception as exc: + if heartbeat_db is not None: + heartbeat_db.close() + heartbeat_db = None + claim_guard.refund_undurable(f'infrastructure heartbeat thread start failure: {exc}') + raise + results = [] + batch_sink_error = None + try: + try: + results = scan_targets_batch( + targets_to_scan, + args.platform, + progress_callback=progress, + max_workers=args.workers, + persist_results=not bool(queue_info.get('lease_owner')), + result_sink=spool_result if postgres_db else None, + scan_slot_leases=scan_slot_leases, + sink_within_scan_slot=bool(postgres_db and scan_slot_leases), + **scan_kwargs, + ) + except ResultSinkError as exc: + results = exc.results + batch_sink_error = exc + except Exception as exc: + if claim_guard: + claim_guard.refund_undurable(f'infrastructure scan submission failure: {exc}') + raise + finally: + heartbeat_stop.set() + heartbeat_stuck = False + if heartbeat_thread: + # The Postgres heartbeat has a 10s statement timeout. Give an in-flight + # renewal time to finish instead of closing its connection underneath it. + heartbeat_thread.join(timeout=15) + heartbeat_stuck = heartbeat_thread.is_alive() + if heartbeat_db and not heartbeat_stuck: + heartbeat_db.close() + if heartbeat_stuck: + if claim_guard: + claim_guard.refund_undurable('infrastructure lease heartbeat shutdown failure') + raise RuntimeError(f'Lease heartbeat did not stop for {source_for_db}') + if claim_guard: + with active_lease_lock: + undurable_claims = bool(active_lease_tokens) + if undurable_claims: + claim_guard.refund_undurable('scan batch returned without a durable result for every fenced claim') + raise RuntimeError(f'Scan batch returned without durable results for every claim in {source_for_db}') + if spool: + wait_for_result_spool_ready( + spool, + db, + outcomes=ingest_outcomes, + stop_event=getattr(args, 'result_spool_stop_event', None), + wait_seconds=float(getattr(args, 'result_spool_wait_sec', 1.0) or 1.0), + diagnostic_interval=float(getattr(args, 'result_spool_diagnostic_interval_sec', 30.0) or 30.0), + ) + if reservation_id: + spool.release_reservation(reservation_id) + if heartbeat_failed.is_set(): + raise RuntimeError(f'Lease heartbeat lost ownership for {source_for_db}') + if batch_sink_error: + raise batch_sink_error + + db_queue = postgres_db + project_files = not db_queue or bool(getattr(args, 'sync_file_queues', True)) + results_for_checked = [result for result in results if not result.get('errors')] + if db_queue: + successful_results = [] + accepted_results = [] + for result in results: + event_id = str(result.get('scan_event_id') or '') + persisted = ingest_outcomes.get(event_id) + if not persisted or not persisted.get('ingested'): + raise RuntimeError(f'No confirmed database ingestion for scan event {event_id or ""}') + event_meta = queue_event_meta.get(event_id) or {} + queue_status = event_meta.get('queue_status') + accepted_results.append(result) + if persisted.get('stale'): + safe_print( + f"Preserved stale scan result for {result.get('target', '')}; " + 'newer queue ownership/status was left unchanged' + ) + elif queue_status in ('done', 'failed'): + successful_results.append(result) + if queue_status == 'failed' and persisted.get('queue_completion_applied'): + safe_print( + f"Target retry exhausted/non-retryable: {result.get('target', '')} " + f"class={result.get('error_class', 'unknown')} " + f"attempts={event_meta.get('attempts', 0)}/{event_meta.get('max_attempts', 0)}" + ) + results_for_checked = successful_results + drain_scan_publication_outbox(db, 100) + elif db and cycle_id: + accepted_results = results + recorded = db.record_target_results(run_id, cycle_id, source_for_db, args.query, results, scan_kwargs) + successful_results = [] + for result in results: + normalized = normalize_target(result.get('target', ''), source_for_db) + target_scan_id = (recorded or {}).get(normalized) + if target_scan_id is None: + target_scan_id = (recorded or {}).get(str(result.get('target', ''))) + if target_scan_id is not None and not result.get('errors'): + successful_results.append(result) + results_for_checked = successful_results + else: + accepted_results = results + + harvested_postman_targets = collect_postman_targets(accepted_results) + if harvested_postman_targets and project_files: + try: + queued_postman = enqueue_targets_for_platform(getattr(args, 'queue_dir', None) or args.save_dir, 'postman', harvested_postman_targets) + action = 'projected' if db_queue else 'queued' + print(f'Harvested {len(harvested_postman_targets)} Postman artifact(s); {action} {queued_postman} new Postman target(s).') + except Exception as exc: + if not db_queue: + raise + safe_print(f'Warning: harvested Postman targets were committed to Postgres but file projection failed: {exc}') + if project_files: + try: + mark_checked(results_for_checked, todo_file, checked_file, args.platform) + except Exception as exc: + if not db_queue: + raise + safe_print(f'Warning: queue completion was committed to Postgres but checked/todo projection failed: {exc}') + print_result_report(results, args.save_dir) + + metrics = summarize_results(results) + source_failures = [result for result in results if result.get('source_failure')] + source_failure_queue_ids = [] + for result in source_failures: + claim = (queue_info.get('queue_claims') or {}).get(str(result.get('target', ''))) or {} + if claim.get('id') is not None: + source_failure_queue_ids.append(claim['id']) + metrics.update({ + 'fetched_count': queue_info.get('fetched_count', len(fetched_targets)), + 'queued_new_count': queue_info.get('queued_new_count', 0), + 'queued_updated_count': queue_info.get('queued_updated_count', 0), + 'scan_requested_count': queue_info.get('scan_requested_count', len(targets_to_scan)), + 'backlog_only': backlog_only, + 'source_failure_count': len(source_failures), + 'source_failure_category': source_failures[0].get('source_failure_category') if source_failures else '', + 'source_failure_auth_related': bool(source_failures and source_failures[0].get('source_failure_auth_related')), + 'source_failure_message': first_error_line(source_failures[0]) if source_failures else '', + 'source_failure_queue_ids': source_failure_queue_ids, + }) + partial_metrics.update(metrics) + queue_after = queue_counts(todo_file, checked_file) if project_files else { + 'todo_count': 0, + 'checked_count': 0, + 'todo_file': None, + 'checked_file': None, + } + if db and cycle_id: + db.finish_source_cycle( + cycle_id, 'source_failed' if source_failures else 'completed', metrics, queue_after, + metrics.get('source_failure_message') or None, + ) + + print( + f"Cycle complete: {metrics['scanned_count']} scanned, " + f"{metrics['findings_count']} findings, {metrics['error_count']} errors, " + f"{metrics.get('degraded_count', 0)} degraded." + ) + return metrics + + +def run_cycle(args, db=None, run_id=None, cycle_id=None, source_name=None, partial_metrics=None): + source = source_name or args.platform + postgres_db = bool( + db and getattr(db, 'conn', None) and getattr(db.conn, 'is_postgres', False) + ) + if postgres_db and hasattr(db, 'reserve_and_claim_target'): + return run_cycle_v2(args, db, run_id, cycle_id, source, partial_metrics) + # SQLite and narrow test doubles retain the explicit v1 compatibility adapter. + return _run_cycle_legacy_compat(args, db, run_id, cycle_id, source, partial_metrics) + + +def load_config(path, *, managed_postgres=None, final_cutover=None): + try: + import yaml + except ImportError as e: + raise SystemExit('PyYAML is required for --config mode. Run: python -m pip install -r requirements.txt') from e + + if not os.path.exists(path): + raise FileNotFoundError(path) + with open(path, 'r', encoding='utf-8') as f: + loaded = yaml.safe_load(f) + if loaded is None: + loaded = {} + validated = validate_docker_depth_config( + loaded, + managed_postgres=managed_postgres, + final_cutover=final_cutover, + ) + return apply_path_config(validated.config, path) + + +def resolve_config_path(config_path, value): + if not value: + return value + if os.path.isabs(value): + return value + return os.path.join(os.path.dirname(os.path.abspath(config_path)), value) + + +def load_secrets(config, config_path): + global_config = config.get('global', {}) + secrets_file = global_config.get('secrets_file') + if not secrets_file: + return {} + path = resolve_config_path(config_path, secrets_file) + if not os.path.exists(path): + print(f'Info: secrets file {path} not found. Falling back to env/source tokens where configured.') + return {} + try: + import yaml + except ImportError as e: + raise SystemExit('PyYAML is required for secrets.yaml. Run: python -m pip install -r requirements.txt') from e + with open(path, 'r', encoding='utf-8') as f: + return yaml.safe_load(f) or {} + + +def parse_state_time(value): + if not value: + return None + try: + parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.astimezone(timezone.utc) + except ValueError: + return None + + +def utc_now(): + return datetime.now(timezone.utc) + + +def normalize_queries(value): + if value is None: + return [''] + if isinstance(value, str): + queries = [item.strip() for item in value.split(',')] + else: + queries = [str(item).strip() for item in value] + queries = [query for query in queries if query] + return list(dict.fromkeys(queries)) or [''] + + +def _valid_dockerhub_policy_sha256(value): + return bool( + isinstance(value, str) + and len(value) == 64 + and all(character in '0123456789abcdef' for character in value) + ) + + +def _valid_dockerhub_deep_record(record): + if not isinstance(record, dict): + return None + policy_sha256 = record.get('policy_sha256') + timestamp = record.get('last_dispatched_at') + if not _valid_dockerhub_policy_sha256(policy_sha256) or not isinstance(timestamp, str): + return None + try: + parsed = datetime.fromisoformat(timestamp.replace('Z', '+00:00')) + except (ValueError, OverflowError): + return None + if parsed.tzinfo is None or parsed.utcoffset() != timedelta(0): + return None + return { + 'policy_sha256': policy_sha256, + 'last_dispatched_at': timestamp, + '_parsed': parsed.astimezone(timezone.utc), + } + + +def prepare_dockerhub_discovery_state( + source_state, configured_policies, query, now=None, *, force_deep=False, +): + if not isinstance(force_deep, bool): + raise ValueError('DockerHub forced deep-pass authority is invalid') + configured_queries = set(configured_policies) + parent = source_state.get(DOCKERHUB_DISCOVERY_STATE_KEY) + raw_records = parent.get('deep_by_query') if isinstance(parent, dict) else None + raw_incomplete = ( + parent.get('incomplete_by_query') if isinstance(parent, dict) else None + ) + if not ( + isinstance(parent, dict) + and parent.get('schema') == 1 + and not isinstance(parent.get('schema'), bool) + and isinstance(raw_records, dict) + ): + raw_records = {} + raw_incomplete = {} + + records = {} + parsed_records = {} + incomplete = {} + for configured_query in configured_queries: + valid = _valid_dockerhub_deep_record(raw_records.get(configured_query)) + if valid is not None: + records[configured_query] = { + 'policy_sha256': valid['policy_sha256'], + 'last_dispatched_at': valid['last_dispatched_at'], + } + parsed_records[configured_query] = valid['_parsed'] + incomplete_record = ( + raw_incomplete.get(configured_query) + if isinstance(raw_incomplete, dict) else None + ) + configured_policy = configured_policies.get(configured_query) + if ( + isinstance(incomplete_record, dict) + and isinstance(configured_policy, dict) + and incomplete_record.get('policy_sha256') + == configured_policy.get('policy_sha256') + ): + incomplete[configured_query] = { + 'policy_sha256': incomplete_record['policy_sha256'], + } + source_state[DOCKERHUB_DISCOVERY_STATE_KEY] = { + 'schema': 1, + 'deep_by_query': records, + 'incomplete_by_query': incomplete, + } + + policy = configured_policies.get(query) + if not isinstance(policy, dict) or not _valid_dockerhub_policy_sha256( + policy.get('policy_sha256') + ): + raise ValueError('DockerHub discovery policy state is unavailable') + current = records.get(query) + now = (now or utc_now()).astimezone(timezone.utc) + last_dispatched = parsed_records.get(query) + deep_due = bool( + force_deep + or current is None + or current.get('policy_sha256') != policy['policy_sha256'] + or last_dispatched is None + or last_dispatched > now + or now - last_dispatched >= DOCKERHUB_DEEP_INTERVAL + or query in incomplete + ) + return { + 'deep': deep_due, + 'pass_kind': 'deep' if deep_due else 'ordinary', + 'policy_sha256': policy['policy_sha256'], + } + + +def mark_dockerhub_deep_dispatched(source_state, query, policy_sha256, now=None): + if not _valid_dockerhub_policy_sha256(policy_sha256): + raise ValueError('DockerHub discovery policy hash is invalid') + parent = source_state.get(DOCKERHUB_DISCOVERY_STATE_KEY) + if not isinstance(parent, dict) or parent.get('schema') != 1: + parent = {'schema': 1, 'deep_by_query': {}, 'incomplete_by_query': {}} + source_state[DOCKERHUB_DISCOVERY_STATE_KEY] = parent + records = parent.get('deep_by_query') + if not isinstance(records, dict): + records = {} + parent['deep_by_query'] = records + dispatched_at = (now or utc_now()).astimezone(timezone.utc).isoformat( + timespec='seconds', + ) + records[query] = { + 'policy_sha256': policy_sha256, + 'last_dispatched_at': dispatched_at, + } + + +def mark_dockerhub_discovery_incomplete(source_state, query, policy_sha256): + if not _valid_dockerhub_policy_sha256(policy_sha256): + raise ValueError('DockerHub discovery policy hash is invalid') + parent = source_state.get(DOCKERHUB_DISCOVERY_STATE_KEY) + if not isinstance(parent, dict) or parent.get('schema') != 1: + parent = {'schema': 1, 'deep_by_query': {}, 'incomplete_by_query': {}} + source_state[DOCKERHUB_DISCOVERY_STATE_KEY] = parent + incomplete = parent.get('incomplete_by_query') + if not isinstance(incomplete, dict): + incomplete = {} + parent['incomplete_by_query'] = incomplete + incomplete[query] = {'policy_sha256': policy_sha256} + + +def clear_dockerhub_discovery_incomplete(source_state, query): + parent = source_state.get(DOCKERHUB_DISCOVERY_STATE_KEY) + incomplete = parent.get('incomplete_by_query') if isinstance(parent, dict) else None + if isinstance(incomplete, dict): + incomplete.pop(query, None) + + +def annotate_dockerhub_discovery_args(args, annotation): + args.dockerhub_discovery_deep = bool(annotation['deep']) + args.dockerhub_discovery_pass_kind = str(annotation['pass_kind']) + args.dockerhub_discovery_policy_sha256 = str(annotation['policy_sha256']) + return args + + +def get_state_path(config, config_path): + env_state_file = os.getenv('RUNNER_STATE_FILE') or os.getenv('SCANNER_STATE_FILE') + if env_state_file: + return resolve_config_path(config_path, env_state_file) + + global_config = config.get('global', {}) + state_file = global_config.get('state_file') + if state_file: + return state_file + results_dir = global_config.get('results_dir') or scan_config.results_dir + return os.path.join(results_dir, 'runner_state.json') + + +def default_source_state(): + return { + 'query_index': 0, + 'auth_index': 0, + 'auth_status': {}, + 'auth_endpoint_status': {}, + 'last_auth': None, + 'last_query': None, + 'last_started_at': None, + 'last_completed_at': None, + 'last_status': None, + 'cycles': 0, + } + + +def load_state(path, config): + if os.path.exists(path): + try: + with open(path, 'r', encoding='utf-8') as f: + state = json.load(f) + except (OSError, json.JSONDecodeError): + state = {} + else: + state = {} + + state.setdefault('version', 1) + state.setdefault('sources', {}) + for source_name in config.get('sources', {}): + source_state = state['sources'].setdefault(source_name, default_source_state()) + for key, value in default_source_state().items(): + source_state.setdefault(key, value) + return state + + +def save_state(path, state): + parent = os.path.dirname(path) + if parent: + ensure_private_directory(parent, reject_reparse=True) + tmp_path = f'{path}.{os.getpid()}.{time.time_ns()}.tmp' + with open(tmp_path, 'w', encoding='utf-8') as f: + json.dump(state, f, indent=2, ensure_ascii=False) + delays = (0.05, 0.1, 0.2, 0.5, 1.0, 2.0, 3.0) + last_error = None + for attempt, delay in enumerate((*delays, None), 1): + try: + os.replace(tmp_path, path) + harden_private_file(path) + return + except PermissionError as e: + last_error = e + if delay is None: + break + time.sleep(delay) + except OSError as e: + if getattr(e, 'winerror', None) != 5 or delay is None: + raise + last_error = e + time.sleep(delay) + + try: + with open(path, 'w', encoding='utf-8') as f: + json.dump(state, f, indent=2, ensure_ascii=False) + except OSError as e: + raise e from last_error + finally: + try: + if os.path.exists(tmp_path): + os.remove(tmp_path) + except OSError: + pass + + +def enabled_source_names(config, selected_source=None): + sources = config.get('sources', {}) + if selected_source: + selected_source = 'dockerhub' if selected_source == 'docker' else selected_source + if selected_source not in sources: + raise SystemExit(f'Source {selected_source} is not present in config') + return [selected_source] + return [name for name, source in sources.items() if source.get('enabled', False)] + + +def auth_pool_entries(source_config, secrets): + pool_name = source_config.get('auth_pool') + if not pool_name: + return [], None + entries = (secrets.get('auth_pools') or {}).get(pool_name, []) + normalized = [] + for index, entry in enumerate(entries): + if not isinstance(entry, dict): + continue + entry = dict(entry) + entry.setdefault('name', f'{pool_name}_{index + 1}') + normalized.append(entry) + return normalized, pool_name + + +def auth_status_kind(status): + if not isinstance(status, dict): + return 'ok' + if status.get('disabled_until') == 'manual' or status.get('status') == 'dead' or status.get('disabled_reason') == 'auth_invalid': + return 'dead' + disabled_until = parse_state_time(status.get('disabled_until')) + if disabled_until and disabled_until > utc_now(): + return 'limited' + return 'ok' + + +def auth_entry_is_available(source_name, entry, state): + source_state = state['sources'].setdefault(source_name, default_source_state()) + auth_status = source_state.setdefault('auth_status', {}) + status = auth_status.get(entry.get('name'), {}) + if auth_status_kind(status) == 'dead': + return False + disabled_until = parse_state_time(status.get('disabled_until')) + return not disabled_until or disabled_until <= utc_now() + + +def refresh_auth_summary(source_name, source_config, state, secrets, current_auth_name=None): + entries, pool_name = auth_pool_entries(source_config, secrets) + source_state = state['sources'].setdefault(source_name, default_source_state()) + auth_status = source_state.setdefault('auth_status', {}) + counts = {'ok': 0, 'dead': 0, 'limited': 0} + rate_limit_errors = 0 + auth_invalid_errors = 0 + for entry in entries: + name = entry.get('name') + item = auth_status.setdefault(name, {}) + kind = auth_status_kind(item) + counts[kind] = counts.get(kind, 0) + 1 + rate_limit_errors += int(item.get('rate_limit_count', 0) or 0) + rate_limit_errors += int(item.get('secondary_rate_limit_count', 0) or 0) + auth_invalid_errors += int(item.get('auth_invalid_count', 0) or 0) + if current_auth_name is None: + current_auth_name = source_state.get('last_auth') + source_state['auth_summary'] = { + 'pool': pool_name or '', + 'current': current_auth_name or source_state.get('last_auth') or 'none', + 'total': len(entries), + 'ok': counts.get('ok', 0), + 'dead': counts.get('dead', 0), + 'limited': counts.get('limited', 0), + 'rate_limit_errors': rate_limit_errors, + 'auth_invalid_errors': auth_invalid_errors, + } + return source_state['auth_summary'] + + +def available_auth_entries(source_name, source_config, state, secrets): + entries, _ = auth_pool_entries(source_config, secrets) + return [entry for entry in entries if auth_entry_is_available(source_name, entry, state)] + + +def select_auth_entry(source_name, source_config, state, secrets): + entries, pool_name = auth_pool_entries(source_config, secrets) + if not entries: + return None + + available = [entry for entry in entries if auth_entry_is_available(source_name, entry, state)] + if not available: + return None + + source_state = state['sources'].setdefault(source_name, default_source_state()) + rotation = source_config.get('auth_rotation', 'per_cycle') + if rotation == 'none': + entry = available[0] + else: + start = int(source_state.get('auth_index', 0)) + entry = None + for offset in range(len(entries)): + candidate = entries[(start + offset) % len(entries)] + if candidate in available: + entry = candidate + source_state['auth_index'] = (start + offset + 1) % len(entries) + break + if entry is None: + entry = available[0] + + source_state['last_auth'] = entry.get('name') + return entry + + +def mark_auth_rate_limited(source_name, auth_entry, state, reset_at=None, cooldown=3600, category='rate_limit', message=None): + if not auth_entry: + return + source_state = state['sources'].setdefault(source_name, default_source_state()) + auth_status = source_state.setdefault('auth_status', {}) + name = auth_entry.get('name') + now = utc_now() + disabled_until = reset_at + if category == 'auth_invalid': + disabled_until = 'manual' + elif not disabled_until: + disabled_until = (now + timedelta(seconds=int(cooldown))).isoformat(timespec='seconds') + item = auth_status.setdefault(name, {}) + if ( + category != 'auth_invalid' + and ( + item.get('disabled_until') == 'manual' + or item.get('status') == 'dead' + or item.get('disabled_reason') == 'auth_invalid' + ) + ): + return + item['disabled_until'] = disabled_until + item['disabled_reason'] = category + item['status'] = 'dead' if category == 'auth_invalid' else 'limited' + item['last_error'] = str(message or '')[:500] + item['last_error_at'] = now.isoformat(timespec='seconds') + item['last_rate_limited_at'] = now.isoformat(timespec='seconds') + item['failures'] = int(item.get('failures', 0)) + 1 + if category == 'auth_invalid': + item['auth_invalid_count'] = int(item.get('auth_invalid_count', 0)) + 1 + item['dead_at'] = now.isoformat(timespec='seconds') + elif category == 'secondary_rate_limit': + item['secondary_rate_limit_count'] = int(item.get('secondary_rate_limit_count', 0)) + 1 + else: + item['rate_limit_count'] = int(item.get('rate_limit_count', 0)) + 1 + + +def clear_auth_rate_limit(source_name, auth_entry, state): + if not auth_entry: + return + source_state = state['sources'].setdefault(source_name, default_source_state()) + auth_status = source_state.setdefault('auth_status', {}) + item = auth_status.setdefault(auth_entry.get('name'), {}) + if item.get('disabled_until') == 'manual' or item.get('status') == 'dead': + return + item['disabled_until'] = None + item['disabled_reason'] = None + item['status'] = 'ok' + item['last_success_at'] = utc_now().isoformat(timespec='seconds') + item['success_count'] = int(item.get('success_count', 0)) + 1 + + +def auth_token_from_entry(entry): + return entry.get('token') if entry else None + + +def current_query_for_source(source_name, source_config, state): + queries = normalize_queries(source_config.get('queries')) + source_state = state['sources'].setdefault(source_name, default_source_state()) + index = int(source_state.get('query_index', 0)) % len(queries) + source_state['query_index'] = index + return queries[index], index, len(queries) + + +def advance_query_for_source(source_name, source_config, state): + queries = normalize_queries(source_config.get('queries')) + source_state = state['sources'].setdefault(source_name, default_source_state()) + source_state['query_index'] = (int(source_state.get('query_index', 0)) + 1) % len(queries) + + +def source_to_platform(source_name): + return 'docker' if source_name in ('docker', 'dockerhub') else source_name + + +def normalize_archive_event_types(value): + if value is None: + return ['PushEvent', 'CreateEvent', 'PublicEvent'] + if isinstance(value, str): + return [item.strip() for item in value.split(',') if item.strip()] + return [str(item).strip() for item in value if str(item).strip()] + + +def normalize_package_sources(value): + if value is None: + return 'npm,pypi' + if isinstance(value, str): + return ','.join(item.strip() for item in value.split(',') if item.strip()) + return ','.join(str(item).strip() for item in value if str(item).strip()) + + +def normalize_search_kinds(value): + if value is None: + return 'collection,environment' + if isinstance(value, str): + return ','.join(item.strip() for item in value.split(',') if item.strip()) + return ','.join(str(item).strip() for item in value if str(item).strip()) + + +def github_token_entries_for_postman(source_config, secrets): + entries, _ = auth_pool_entries(source_config, secrets) + token_entries = [] + for entry in entries: + if entry.get('token'): + token_entries.append({'name': entry.get('name'), 'token': entry.get('token')}) + if source_config.get('token'): + token_entries.append({'name': 'source_token', 'token': source_config.get('token')}) + return token_entries + + +def build_args_from_source_config(source_name, source_config, global_config, query, auth_entry=None): + query_overrides = source_config.get('query_overrides') or {} + if not isinstance(query_overrides, dict): + raise ValueError(f'{source_name} query_overrides must be a mapping') + query_override = query_overrides.get(query) + if query_override is not None: + if not isinstance(query_override, dict): + raise ValueError(f'{source_name} query override for {query!r} must be a mapping') + allowed = {'pages', 'per_page', 'max_targets'} + unknown = sorted(set(query_override) - allowed) + if unknown: + raise ValueError( + f'{source_name} query override for {query!r} has unsupported keys: ' + + ', '.join(unknown) + ) + effective_source_config = dict(source_config) + for key, value in query_override.items(): + try: + value = int(value) + except (TypeError, ValueError) as e: + raise ValueError( + f'{source_name} query override {key} for {query!r} must be an integer' + ) from e + minimum = 0 if key == 'max_targets' else 1 + if value < minimum: + raise ValueError( + f'{source_name} query override {key} for {query!r} must be >= {minimum}' + ) + effective_source_config[key] = value + source_config = effective_source_config + + platform = source_to_platform(source_name) + token = auth_token_from_entry(auth_entry) or source_config.get('token') + package_sources = normalize_package_sources(source_config.get('package_sources', global_config.get('package_sources', ['npm', 'pypi']))) + path_context = global_config + return SimpleNamespace( + platform=platform, + configured_queries=tuple(normalize_queries(source_config.get('queries'))), + mode=source_config.get('mode', 'recent'), + query=query, + query_file=None, + pages=int(source_config.get('pages', 1)), + per_page=int(source_config.get('per_page', 50)), + target_file=resolve_optional_path(source_config.get('target_file'), path_context), + token=token, + docker_username=(auth_entry or {}).get('username') or source_config.get('docker_username'), + docker_token=(auth_entry or {}).get('token') or source_config.get('docker_token'), + workers=int(source_config.get('workers', 6)), + timeout=int(source_config.get('timeout', scan_config.docker_timeout if platform == 'docker' else scan_config.git_timeout)), + detectors=str(source_config.get('detectors', global_config.get('detectors', scan_config.detectors))), + exclude_detectors=str(source_config.get('exclude_detectors', global_config.get('exclude_detectors', scan_config.exclude_detectors))), + drop_detectors=source_config.get('drop_detectors', global_config.get('drop_detectors', getattr(scan_config, 'drop_detectors', []))), + no_verification=bool(source_config.get('no_verification', global_config.get('no_verification', scan_config.no_verification))), + save_dir=global_config.get('results_dir', scan_config.results_dir), + runtime_dir=global_config.get('runtime_dir'), + result_bundle_dir=global_config.get('result_bundle_dir', scan_config.result_bundle_dir), + result_bundle_max_event_bytes=int(global_config.get('result_bundle_max_event_bytes', scan_config.result_bundle_max_event_bytes)), + remote_assignment_reserve_bytes=int(global_config.get('remote_assignment_reserve_bytes', 2 * 1024 * 1024)), + remote_assignment_max_active=int(global_config.get('remote_assignment_max_active', 50)), + result_bundle_max_items=int(global_config.get('result_bundle_max_items', 10000)), + result_bundle_max_total_bytes=int(global_config.get('result_bundle_max_total_bytes', 3 * 1024 * 1024 * 1024)), + result_bundle_min_free_bytes=int(global_config.get('result_bundle_min_free_bytes', 20 * 1024 * 1024 * 1024)), + projection_backlog_max_items=int(global_config.get('projection_backlog_max_items', 10000)), + projection_backlog_max_bytes=int(global_config.get('projection_backlog_max_bytes', 2 * 1024 * 1024 * 1024)), + projection_backlog_headroom_bytes=int(global_config.get( + 'projection_backlog_headroom_bytes', + 2 * int(global_config.get('result_bundle_max_event_bytes', scan_config.result_bundle_max_event_bytes)), + )), + keycheck_queue_max_items=int(global_config.get('keycheck_queue_max_items', 100000)), + keycheck_queue_max_bytes=int(global_config.get('keycheck_queue_max_bytes', 512 * 1024 * 1024)), + pipeline_quarantine_max_items=int(global_config.get('pipeline_quarantine_max_items', 10000)), + pipeline_quarantine_max_bytes=int(global_config.get('pipeline_quarantine_max_bytes', 1024 * 1024 * 1024)), + keycheck_candidates_per_event=int(global_config.get('keycheck_candidates_per_event', 2000)), + keycheck_candidate_bytes_per_event=int(global_config.get('keycheck_candidate_bytes_per_event', 2 * 1024 * 1024)), + result_spool_dir=global_config.get('result_spool_dir', scan_config.result_spool_dir), + result_spool_max_event_bytes=int(global_config.get('result_spool_max_event_bytes', scan_config.result_spool_max_event_bytes)), + result_spool_max_events=int(global_config.get('result_spool_max_events', scan_config.result_spool_max_events)), + result_spool_max_total_bytes=int(global_config.get('result_spool_max_total_bytes', scan_config.result_spool_max_total_bytes)), + result_spool_min_free_bytes=int(global_config.get('result_spool_min_free_bytes', scan_config.result_spool_min_free_bytes)), + result_spool_wait_sec=float(global_config.get('result_spool_wait_sec', 1.0)), + result_spool_diagnostic_interval_sec=float(global_config.get('result_spool_diagnostic_interval_sec', 30.0)), + scan_outbox_max_pending_items=int(global_config.get('scan_outbox_max_pending_items', scan_config.scan_outbox_max_pending_items)), + scan_outbox_max_pending_bytes=int(global_config.get('scan_outbox_max_pending_bytes', scan_config.scan_outbox_max_pending_bytes)), + scan_outbox_max_pending_age_sec=int(global_config.get('scan_outbox_max_pending_age_sec', scan_config.scan_outbox_max_pending_age_sec)), + database_path=global_config.get('database_path'), + database_url=global_config.get('database_url'), + queue_dir=global_config.get('queue_dir', getattr(scan_config, 'queue_dir', scan_config.results_dir)), + work_dir=global_config.get('work_dir', scan_config.work_dir), + trufflehog_path=global_config.get('trufflehog_path', scan_config.trufflehog_path), + trufflehog_config=resolve_optional_path( + source_config.get('trufflehog_config', global_config.get('trufflehog_config', getattr(scan_config, 'trufflehog_config', ''))), + path_context, + ), + trufflehog_job_memory_limit_bytes=int(source_config.get( + 'trufflehog_job_memory_limit_bytes', + global_config.get('trufflehog_job_memory_limit_bytes', scan_config.trufflehog_job_memory_limit_bytes), + )), + trufflehog_concurrency=int(source_config.get( + 'trufflehog_concurrency', global_config.get('trufflehog_concurrency', 0), + )), + loop=False, + cooldown=int(source_config.get('cooldown', global_config.get('cooldown', 300))), + max_cycles=0, + max_targets=int(source_config.get('max_targets', 0)), + target_retry_max_attempts=int(source_config.get('target_retry_max_attempts', global_config.get('target_retry_max_attempts', 3))), + target_retry_base_delay_sec=int(source_config.get('target_retry_base_delay_sec', global_config.get('target_retry_base_delay_sec', 3600))), + target_retry_max_delay_sec=int(source_config.get('target_retry_max_delay_sec', global_config.get('target_retry_max_delay_sec', 86400))), + target_timeout_retry_delay_sec=int(source_config.get('target_timeout_retry_delay_sec', global_config.get('target_timeout_retry_delay_sec', 21600))), + target_claim_batch_size=int(source_config.get('target_claim_batch_size', global_config.get('target_claim_batch_size', 0))), + target_claim_order=str(source_config.get('target_claim_order', global_config.get('target_claim_order', 'oldest'))).strip().lower(), + sync_file_queues=bool_config(source_config.get('sync_file_queues', global_config.get('sync_file_queues', True)), True), + recent_hours=int(source_config.get('recent_hours', 24)), + recent_days=int(source_config.get('recent_days', 7)), + max_version_age_days=int(source_config.get('max_version_age_days', 0)), + versions_per_package=int(source_config.get('versions_per_package', global_config.get('versions_per_package', 1))), + package_sources=package_sources, + refresh_registry=bool(source_config.get('refresh_registry', global_config.get('refresh_registry', False))), + updated_target_rescan_enabled=bool_config( + source_config.get('updated_target_rescan_enabled', False), False, + ), + updated_target_rescan_max_per_cycle=int( + source_config.get('updated_target_rescan_max_per_cycle', 0) or 0 + ), + updated_target_rescan_cooldown_hours=int( + source_config.get('updated_target_rescan_cooldown_hours', 0) or 0 + ), + search_kinds=normalize_search_kinds(source_config.get('search_kinds', global_config.get('search_kinds', ['collection', 'environment']))), + gist_since=source_config.get('gist_since', global_config.get('gist_since', '')), + postman_cache_dir=resolve_optional_path( + source_config.get('postman_cache_dir') or global_config.get('postman_cache_dir') or getattr(scan_config, 'postman_cache_dir', None), + path_context, + ), + postman_discovery_max_artifacts_per_cycle=int(source_config.get('postman_discovery_max_artifacts_per_cycle', global_config.get('postman_discovery_max_artifacts_per_cycle', getattr(scan_config, 'postman_discovery_max_artifacts_per_cycle', 1000)))), + postman_discovery_max_artifacts_per_page=int(source_config.get('postman_discovery_max_artifacts_per_page', global_config.get('postman_discovery_max_artifacts_per_page', getattr(scan_config, 'postman_discovery_max_artifacts_per_page', 100)))), + postman_discovery_max_bytes_per_cycle=int(source_config.get('postman_discovery_max_bytes_per_cycle', global_config.get('postman_discovery_max_bytes_per_cycle', getattr(scan_config, 'postman_discovery_max_bytes_per_cycle', 1024 * 1024 * 1024)))), + postman_discovery_max_elapsed_sec=float(source_config.get('postman_discovery_max_elapsed_sec', global_config.get('postman_discovery_max_elapsed_sec', getattr(scan_config, 'postman_discovery_max_elapsed_sec', 300.0)))), + gharchive_cache_dir=resolve_optional_path( + source_config.get('gharchive_cache_dir') or global_config.get('gharchive_cache_dir'), + path_context, + ), + max_artifact_size_mb=int(source_config.get('max_artifact_size_mb', global_config.get('max_artifact_size_mb', 50))), + max_file_age_days=int(source_config.get('max_file_age_days', global_config.get('max_file_age_days', 365))), + github_code_search_rpm=int(source_config.get('github_code_search_rpm', global_config.get('github_code_search_rpm', 8))), + all_tokens_cooldown=int(source_config.get('all_tokens_cooldown', global_config.get('all_tokens_cooldown', 1800))), + max_repo_age_days=int(source_config.get('max_repo_age_days', 0)), + repo_age_field=source_config.get('repo_age_field', 'created_at'), + max_commit_age_days=int(source_config.get('max_commit_age_days', 0)), + commit_lookup_pages=int(source_config.get('commit_lookup_pages', 3)), + skip_if_commit_lookup_fails=bool(source_config.get('skip_if_commit_lookup_fails', True)), + raise_rate_limit=bool(source_config.get('retry_with_next_auth_on_rate_limit', True)), + auth_name=(auth_entry or {}).get('name'), + max_depth=int(source_config.get('max_depth', 0)), + exact_git_planning_enabled=bool_config( + source_config.get( + 'exact_git_planning_enabled', + global_config.get('exact_git_planning_enabled', False), + ), + False, + ), + git_baseline_depth=int(source_config.get( + 'git_baseline_depth', source_config.get('max_depth', 100) or 100, + )), + git_ref_resolution_timeout_sec=float(source_config.get( + 'git_ref_resolution_timeout_sec', + global_config.get('git_ref_resolution_timeout_sec', 10), + )), + git_ref_resolution_attempts=int(source_config.get( + 'git_ref_resolution_attempts', + global_config.get('git_ref_resolution_attempts', 2), + )), + git_ref_resolution_max_bytes=int(source_config.get( + 'git_ref_resolution_max_bytes', + global_config.get('git_ref_resolution_max_bytes', 1 << 20), + )), + admission_resolution_attempts=int(source_config.get( + 'admission_resolution_attempts', + global_config.get('admission_resolution_attempts', 90), + )), + admission_resolution_seconds=float(source_config.get( + 'admission_resolution_seconds', + global_config.get('admission_resolution_seconds', 90), + )), + admission_resolution_retry_delay_sec=float(source_config.get( + 'admission_resolution_retry_delay_sec', + global_config.get('admission_resolution_retry_delay_sec', 1), + )), + scan_full_history=bool(source_config.get('scan_full_history', False)), + sort_by=source_config.get('sort_by', 'updated'), + sort_order=source_config.get('sort_order', 'desc'), + created_filter=source_config.get('created_filter', 'any'), + gitlab_sort_by=source_config.get('gitlab_sort_by', source_config.get('sort_by', 'last_activity_at')), + gitlab_visibility=source_config.get('gitlab_visibility', source_config.get('visibility', 'public')), + gitlab_discovery_request_attempts=int(source_config.get('discovery_request_attempts', 1)), + gitlab_discovery_retry_delay=int(source_config.get('discovery_retry_delay', 0)), + huggingface_discovery_request_attempts=int(source_config.get('discovery_request_attempts', 1)), + huggingface_discovery_retry_delay=int(source_config.get('discovery_retry_delay', 0)), + external_trufflehog_lifecycle=bool(source_config.get('external_trufflehog_lifecycle', False)), + docker_sort_by=source_config.get('docker_sort_by', source_config.get('sort_by', 'updated_at')), + fetch_workers=int(source_config.get('fetch_workers', global_config.get('fetch_workers', 8))), + fetch_timeout=int(source_config.get('fetch_timeout', global_config.get('fetch_timeout', 15))), + tag_fetch_workers=int(source_config.get('tag_fetch_workers', global_config.get('tag_fetch_workers', 4))), + tag_retry_count=int(source_config.get('tag_retry_count', global_config.get('tag_retry_count', 2))), + tag_retry_delay=int(source_config.get('tag_retry_delay', global_config.get('tag_retry_delay', 5))), + tag_resolve_limit=int(source_config.get('tag_resolve_limit', global_config.get('tag_resolve_limit', 100))), + docker_platform_filter_enabled=bool(source_config.get('docker_platform_filter_enabled', global_config.get('docker_platform_filter_enabled', True))), + docker_platform_os=str(source_config.get('docker_platform_os', global_config.get('docker_platform_os', 'linux'))), + docker_platform_arch=str(source_config.get('docker_platform_arch', global_config.get('docker_platform_arch', 'amd64'))), + docker_platform_candidate_tags=int(source_config.get('docker_platform_candidate_tags', global_config.get('docker_platform_candidate_tags', 20))), + docker_images_per_repository=docker_images_per_repository_limit( + source_config.get('docker_images_per_repository', global_config.get('docker_images_per_repository', 1)) + ), + docker_content_scan_mode=str(source_config.get( + 'docker_content_scan_mode', global_config.get('docker_content_scan_mode', 'full'), + )).strip().lower(), + docker_layer_canary_basis_points=int(source_config.get( + 'docker_layer_canary_basis_points', + global_config.get('docker_layer_canary_basis_points', 0), + )), + docker_adaptive_canary_basis_points=int(source_config.get( + 'docker_adaptive_canary_basis_points', + global_config.get('docker_adaptive_canary_basis_points', 0), + )), + docker_adaptive_gate_max_age_sec=int(source_config.get( + 'docker_adaptive_gate_max_age_sec', + global_config.get('docker_adaptive_gate_max_age_sec', 604800), + )), + docker_layer_config_max_bytes=int(source_config.get( + 'docker_layer_config_max_bytes', global_config.get('docker_layer_config_max_bytes', 1 << 20), + )), + docker_layer_max_bytes=int(source_config.get( + 'docker_layer_max_bytes', global_config.get('docker_layer_max_bytes', 256 << 20), + )), + docker_layer_image_max_bytes=int(source_config.get( + 'docker_layer_image_max_bytes', global_config.get('docker_layer_image_max_bytes', 1 << 30), + )), + docker_layer_max_layers=int(source_config.get( + 'docker_layer_max_layers', global_config.get('docker_layer_max_layers', 8), + )), + docker_layer_archive_max_size_bytes=int(source_config.get( + 'docker_layer_archive_max_size_bytes', + global_config.get('docker_layer_archive_max_size_bytes', 256 << 20), + )), + docker_layer_archive_max_depth=int(source_config.get( + 'docker_layer_archive_max_depth', global_config.get('docker_layer_archive_max_depth', 4), + )), + docker_layer_archive_timeout_sec=int(source_config.get( + 'docker_layer_archive_timeout_sec', global_config.get('docker_layer_archive_timeout_sec', 30), + )), + docker_layer_blob_timeout_sec=int(source_config.get( + 'docker_layer_blob_timeout_sec', global_config.get('docker_layer_blob_timeout_sec', 600), + )), + docker_layer_filesystem_concurrency=int(source_config.get( + 'docker_layer_filesystem_concurrency', + global_config.get('docker_layer_filesystem_concurrency', 2), + )), + docker_layer_blob_max_attempts=int(source_config.get( + 'docker_layer_blob_max_attempts', global_config.get('docker_layer_blob_max_attempts', 3), + )), + docker_layer_blob_lease_sec=int(source_config.get( + 'docker_layer_blob_lease_sec', global_config.get('docker_layer_blob_lease_sec', 1800), + )), + docker_layer_min_free_bytes=int(source_config.get( + 'docker_layer_min_free_bytes', global_config.get('docker_layer_min_free_bytes', 20 << 30), + )), + docker_layer_checkpoint_delay_sec=int(source_config.get( + 'docker_layer_checkpoint_delay_sec', + global_config.get('docker_layer_checkpoint_delay_sec', 60), + )), + docker_adaptive_checkpoint_max_blobs=int(source_config.get( + 'docker_adaptive_checkpoint_max_blobs', + global_config.get('docker_adaptive_checkpoint_max_blobs', 4), + )), + docker_adaptive_checkpoint_max_bytes=int(source_config.get( + 'docker_adaptive_checkpoint_max_bytes', + global_config.get('docker_adaptive_checkpoint_max_bytes', 512 << 20), + )), + docker_repository_refresh_interval_sec=max(0, int( + source_config.get( + 'docker_repository_refresh_interval_sec', + global_config.get('docker_repository_refresh_interval_sec', 86400), + ) or 0 + )), + docker_repository_refresh_max_per_cycle=max(0, min(1, int( + source_config.get( + 'docker_repository_refresh_max_per_cycle', + global_config.get('docker_repository_refresh_max_per_cycle', 0), + ) or 0 + ))), + archive_hours_back=int(source_config.get('archive_hours_back', global_config.get('archive_hours_back', 6))), + archive_max_repos_per_cycle=int(source_config.get('archive_max_repos_per_cycle', global_config.get('archive_max_repos_per_cycle', 200))), + archive_max_files_per_cycle=int(source_config.get('archive_max_files_per_cycle', global_config.get('archive_max_files_per_cycle', 300))), + archive_max_commit_lookups=int(source_config.get('archive_max_commit_lookups', global_config.get('archive_max_commit_lookups', 200))), + archive_rescan_cooldown_hours=int(source_config.get('archive_rescan_cooldown_hours', global_config.get('archive_rescan_cooldown_hours', 48))), + archive_event_types=normalize_archive_event_types(source_config.get('archive_event_types', global_config.get('archive_event_types'))), + ci_seed_sources=source_config.get('ci_seed_sources', global_config.get('ci_seed_sources', 'github,gitlab,package_git')), + ci_use_finding_seeds=bool(source_config.get('ci_use_finding_seeds', global_config.get('ci_use_finding_seeds', False))), + ci_max_repos_per_cycle=int(source_config.get('ci_max_repos_per_cycle', global_config.get('ci_max_repos_per_cycle', 50))), + ci_seed_scan_limit=int(source_config.get('ci_seed_scan_limit', global_config.get('ci_seed_scan_limit', 5000))), + ci_seed_query_batch_size=int(source_config.get('ci_seed_query_batch_size', global_config.get('ci_seed_query_batch_size', 250))), + ci_soft_cooldown_days=int(source_config.get('ci_soft_cooldown_days', global_config.get('ci_soft_cooldown_days', 7))), + ci_runs_per_repo=int(source_config.get('ci_runs_per_repo', global_config.get('ci_runs_per_repo', 5))), + ci_pipelines_per_project=int(source_config.get('ci_pipelines_per_project', global_config.get('ci_pipelines_per_project', 5))), + ci_jobs_per_pipeline=int(source_config.get('ci_jobs_per_pipeline', global_config.get('ci_jobs_per_pipeline', 20))), + ci_lookback_days=int(source_config.get('ci_lookback_days', global_config.get('ci_lookback_days', 30))), + ci_max_log_archive_mb=int(source_config.get('ci_max_log_archive_mb', global_config.get('ci_max_log_archive_mb', 50))), + ci_max_log_file_mb=int(source_config.get('ci_max_log_file_mb', global_config.get('ci_max_log_file_mb', 20))), + ci_max_trace_mb=int(source_config.get('ci_max_trace_mb', global_config.get('ci_max_trace_mb', 20))), + ci_failed_first=bool(source_config.get('ci_failed_first', global_config.get('ci_failed_first', True))), + ci_scan_artifacts=bool(source_config.get('ci_scan_artifacts', global_config.get('ci_scan_artifacts', False))), + ci_max_artifacts_per_run=int(source_config.get('ci_max_artifacts_per_run', global_config.get('ci_max_artifacts_per_run', 3))), + ci_max_artifacts_per_pipeline=int(source_config.get('ci_max_artifacts_per_pipeline', global_config.get('ci_max_artifacts_per_pipeline', 5))), + ci_max_artifact_archive_mb=int(source_config.get('ci_max_artifact_archive_mb', global_config.get('ci_max_artifact_archive_mb', 50))), + ci_max_artifact_file_mb=int(source_config.get('ci_max_artifact_file_mb', global_config.get('ci_max_artifact_file_mb', 10))), + ci_max_artifact_files=int(source_config.get('ci_max_artifact_files', global_config.get('ci_max_artifact_files', 1000))), + ci_target_max_download_mb=int(source_config.get('ci_target_max_download_mb', global_config.get('ci_target_max_download_mb', 500))), + stop_on_seen_pages=bool(source_config.get('stop_on_seen_pages', global_config.get('stop_on_seen_pages', False))), + seen_page_threshold=int(source_config.get('seen_page_threshold', global_config.get('seen_page_threshold', 2))), + min_pages_before_stop=int(source_config.get('min_pages_before_stop', global_config.get('min_pages_before_stop', 1))), + cleanup_temp_age_min=int(global_config.get('cleanup_temp_age_min', 120)), + cleanup_only=False, + ) + + +def apply_global_config(global_config): + scan_config.runtime_dir = global_config.get('runtime_dir', getattr(scan_config, 'runtime_dir', None)) + scan_config.work_dir = global_config.get('work_dir', scan_config.work_dir) + scan_config.results_dir = global_config.get('results_dir', scan_config.results_dir) + scan_config.result_bundle_dir = global_config.get('result_bundle_dir', scan_config.result_bundle_dir) + scan_config.result_bundle_max_event_bytes = int(global_config.get( + 'result_bundle_max_event_bytes', scan_config.result_bundle_max_event_bytes, + )) + scan_config.result_spool_dir = global_config.get('result_spool_dir', scan_config.result_spool_dir) + scan_config.result_spool_max_event_bytes = int(global_config.get('result_spool_max_event_bytes', scan_config.result_spool_max_event_bytes)) + scan_config.result_spool_max_events = int(global_config.get('result_spool_max_events', scan_config.result_spool_max_events)) + scan_config.result_spool_max_total_bytes = int(global_config.get('result_spool_max_total_bytes', scan_config.result_spool_max_total_bytes)) + scan_config.result_spool_min_free_bytes = int(global_config.get('result_spool_min_free_bytes', scan_config.result_spool_min_free_bytes)) + scan_config.scan_outbox_max_pending_items = int(global_config.get('scan_outbox_max_pending_items', scan_config.scan_outbox_max_pending_items)) + scan_config.scan_outbox_max_pending_bytes = int(global_config.get('scan_outbox_max_pending_bytes', scan_config.scan_outbox_max_pending_bytes)) + scan_config.scan_outbox_max_pending_age_sec = int(global_config.get('scan_outbox_max_pending_age_sec', scan_config.scan_outbox_max_pending_age_sec)) + scan_config.queue_dir = global_config.get('queue_dir', getattr(scan_config, 'queue_dir', scan_config.results_dir)) + scan_config.keycheck_dir = global_config.get('keycheck_dir', getattr(scan_config, 'keycheck_dir', None)) + scan_config.postman_cache_dir = global_config.get('postman_cache_dir', getattr(scan_config, 'postman_cache_dir', None)) + scan_config.postman_cache_max_items = int(global_config.get('postman_cache_max_items', getattr(scan_config, 'postman_cache_max_items', 100000))) + scan_config.postman_cache_max_bytes = int(global_config.get('postman_cache_max_bytes', getattr(scan_config, 'postman_cache_max_bytes', 20 * 1024 * 1024 * 1024))) + scan_config.postman_cache_min_free_bytes = int(global_config.get('postman_cache_min_free_bytes', getattr(scan_config, 'postman_cache_min_free_bytes', 5 * 1024 * 1024 * 1024))) + scan_config.postman_cache_lock_timeout_sec = int(global_config.get('postman_cache_lock_timeout_sec', getattr(scan_config, 'postman_cache_lock_timeout_sec', 30))) + scan_config.postman_discovery_max_artifacts_per_cycle = int(global_config.get('postman_discovery_max_artifacts_per_cycle', getattr(scan_config, 'postman_discovery_max_artifacts_per_cycle', 1000))) + scan_config.postman_discovery_max_artifacts_per_page = int(global_config.get('postman_discovery_max_artifacts_per_page', getattr(scan_config, 'postman_discovery_max_artifacts_per_page', 100))) + scan_config.postman_discovery_max_bytes_per_cycle = int(global_config.get('postman_discovery_max_bytes_per_cycle', getattr(scan_config, 'postman_discovery_max_bytes_per_cycle', 1024 * 1024 * 1024))) + scan_config.postman_discovery_max_elapsed_sec = float(global_config.get('postman_discovery_max_elapsed_sec', getattr(scan_config, 'postman_discovery_max_elapsed_sec', 300.0))) + scan_config.postman_package_harvest_max_artifacts = int(global_config.get('postman_package_harvest_max_artifacts', getattr(scan_config, 'postman_package_harvest_max_artifacts', 100))) + scan_config.postman_package_harvest_max_bytes = int(global_config.get('postman_package_harvest_max_bytes', getattr(scan_config, 'postman_package_harvest_max_bytes', 128 * 1024 * 1024))) + scan_config.postman_package_harvest_max_elapsed_sec = float(global_config.get('postman_package_harvest_max_elapsed_sec', getattr(scan_config, 'postman_package_harvest_max_elapsed_sec', 30.0))) + scan_config.postman_context_max_input_bytes = int(global_config.get('postman_context_max_input_bytes', getattr(scan_config, 'postman_context_max_input_bytes', 16 * 1024 * 1024))) + scan_config.postman_context_max_nodes = int(global_config.get('postman_context_max_nodes', getattr(scan_config, 'postman_context_max_nodes', 100000))) + scan_config.postman_context_max_depth = int(global_config.get('postman_context_max_depth', getattr(scan_config, 'postman_context_max_depth', 64))) + scan_config.postman_context_max_scalar_bytes = int(global_config.get('postman_context_max_scalar_bytes', getattr(scan_config, 'postman_context_max_scalar_bytes', 16 * 1024 * 1024))) + scan_config.postman_context_max_items = int(global_config.get('postman_context_max_items', getattr(scan_config, 'postman_context_max_items', 50000))) + scan_config.context_enrichment_max_source_bytes = int(global_config.get('context_enrichment_max_source_bytes', getattr(scan_config, 'context_enrichment_max_source_bytes', 16 * 1024 * 1024))) + scan_config.context_enrichment_max_findings = int(global_config.get('context_enrichment_max_findings', getattr(scan_config, 'context_enrichment_max_findings', 2000))) + scan_config.context_enrichment_max_postman_comparisons = int(global_config.get('context_enrichment_max_postman_comparisons', getattr(scan_config, 'context_enrichment_max_postman_comparisons', 200000))) + scan_config.context_enrichment_max_elapsed_sec = float(global_config.get('context_enrichment_max_elapsed_sec', getattr(scan_config, 'context_enrichment_max_elapsed_sec', 5.0))) + scan_config.trufflehog_diagnostic_max_lines = int(global_config.get('trufflehog_diagnostic_max_lines', getattr(scan_config, 'trufflehog_diagnostic_max_lines', 2000))) + scan_config.trufflehog_diagnostic_max_line_chars = int(global_config.get('trufflehog_diagnostic_max_line_chars', getattr(scan_config, 'trufflehog_diagnostic_max_line_chars', 8192))) + scan_config.trufflehog_diagnostic_max_line_bytes = int(global_config.get('trufflehog_diagnostic_max_line_bytes', getattr(scan_config, 'trufflehog_diagnostic_max_line_bytes', 8192))) + scan_config.trufflehog_diagnostic_max_errors = int(global_config.get('trufflehog_diagnostic_max_errors', getattr(scan_config, 'trufflehog_diagnostic_max_errors', 200))) + scan_config.trufflehog_diagnostic_max_warnings = int(global_config.get('trufflehog_diagnostic_max_warnings', getattr(scan_config, 'trufflehog_diagnostic_max_warnings', 200))) + scan_config.trufflehog_diagnostic_max_unclassified = int(global_config.get('trufflehog_diagnostic_max_unclassified', getattr(scan_config, 'trufflehog_diagnostic_max_unclassified', 20))) + scan_config.keycheck_input_max_line_bytes = int(global_config.get('keycheck_input_max_line_bytes', getattr(scan_config, 'keycheck_input_max_line_bytes', 16 * 1024 * 1024))) + scan_config.keycheck_candidate_artifact_max_items = int(global_config.get('keycheck_candidate_artifact_max_items', getattr(scan_config, 'keycheck_candidate_artifact_max_items', 2000))) + scan_config.keycheck_candidate_artifact_max_bytes = int(global_config.get('keycheck_candidate_artifact_max_bytes', getattr(scan_config, 'keycheck_candidate_artifact_max_bytes', 2 * 1024 * 1024))) + scan_config.keycheck_candidate_file_max_items = int(global_config.get('keycheck_candidate_file_max_items', getattr(scan_config, 'keycheck_candidate_file_max_items', 100000))) + scan_config.keycheck_candidate_file_max_bytes = int(global_config.get('keycheck_candidate_file_max_bytes', getattr(scan_config, 'keycheck_candidate_file_max_bytes', 32 * 1024 * 1024))) + scan_config.keycheck_candidate_line_max_bytes = int(global_config.get('keycheck_candidate_line_max_bytes', getattr(scan_config, 'keycheck_candidate_line_max_bytes', 8192))) + scan_config.gharchive_cache_dir = global_config.get('gharchive_cache_dir', getattr(scan_config, 'gharchive_cache_dir', None)) + scan_config.gharchive_cache_max_items = int(global_config.get('gharchive_cache_max_items', getattr(scan_config, 'gharchive_cache_max_items', 48))) + scan_config.gharchive_cache_max_bytes = int(global_config.get('gharchive_cache_max_bytes', getattr(scan_config, 'gharchive_cache_max_bytes', 8 * 1024 * 1024 * 1024))) + scan_config.gharchive_cache_min_free_bytes = int(global_config.get('gharchive_cache_min_free_bytes', getattr(scan_config, 'gharchive_cache_min_free_bytes', 5 * 1024 * 1024 * 1024))) + scan_config.gharchive_download_max_bytes = int(global_config.get('gharchive_download_max_bytes', getattr(scan_config, 'gharchive_download_max_bytes', 512 * 1024 * 1024))) + scan_config.gharchive_decompressed_max_bytes = int(global_config.get('gharchive_decompressed_max_bytes', getattr(scan_config, 'gharchive_decompressed_max_bytes', 8 * 1024 * 1024 * 1024))) + scan_config.gharchive_max_events = int(global_config.get('gharchive_max_events', getattr(scan_config, 'gharchive_max_events', 5000000))) + scan_config.gharchive_max_line_bytes = int(global_config.get('gharchive_max_line_bytes', getattr(scan_config, 'gharchive_max_line_bytes', 8 * 1024 * 1024))) + scan_config.gharchive_cache_lock_timeout_sec = int(global_config.get('gharchive_cache_lock_timeout_sec', getattr(scan_config, 'gharchive_cache_lock_timeout_sec', 600))) + scan_config.proxy_file = global_config.get('proxy_file', getattr(scan_config, 'proxy_file', None)) + scan_config.api_proxy_enabled = bool_config(global_config.get('api_proxy_enabled'), getattr(scan_config, 'api_proxy_enabled', False)) + scan_config.api_proxy_file = resolve_optional_path( + global_config.get('api_proxy_file') or scan_config.proxy_file, + global_config, + ) + scan_config.api_proxy_timeout = int(global_config.get('api_proxy_timeout', getattr(scan_config, 'api_proxy_timeout', 5))) + scan_config.api_proxy_max_retries = int(global_config.get('api_proxy_max_retries', getattr(scan_config, 'api_proxy_max_retries', 100))) + scan_config.api_proxy_retry_delay = int(global_config.get('api_proxy_retry_delay', getattr(scan_config, 'api_proxy_retry_delay', 5))) + scan_config.download_proxy_enabled = bool_config(global_config.get('download_proxy_enabled'), getattr(scan_config, 'download_proxy_enabled', False)) + scan_config.download_proxy_file = resolve_optional_path( + global_config.get('download_proxy_file') or getattr(scan_config, 'download_proxy_file', ''), + global_config, + ) if (global_config.get('download_proxy_file') or getattr(scan_config, 'download_proxy_file', '')) else '' + scan_config.max_active_scans = int(global_config.get('max_active_scans', getattr(scan_config, 'max_active_scans', 0))) + scan_config.opportunistic_scan_slots = max(0, min(1, int(global_config.get( + 'opportunistic_scan_slots', getattr(scan_config, 'opportunistic_scan_slots', 0), + )))) + scan_config.opportunistic_scan_sources = csv_items(global_config.get( + 'opportunistic_scan_sources', getattr(scan_config, 'opportunistic_scan_sources', []), + )) + scan_config.opportunistic_scan_reserve_overhead_bytes = max(0, int(global_config.get( + 'opportunistic_scan_reserve_overhead_bytes', + getattr(scan_config, 'opportunistic_scan_reserve_overhead_bytes', 1024 * 1024 * 1024), + ))) + scan_config.opportunistic_scan_min_available_after_reserve_bytes = max(0, int(global_config.get( + 'opportunistic_scan_min_available_after_reserve_bytes', + getattr(scan_config, 'opportunistic_scan_min_available_after_reserve_bytes', 4 * 1024 * 1024 * 1024), + ))) + scan_config.opportunistic_scan_min_commit_after_reserve_bytes = max(0, int(global_config.get( + 'opportunistic_scan_min_commit_after_reserve_bytes', + getattr(scan_config, 'opportunistic_scan_min_commit_after_reserve_bytes', 6 * 1024 * 1024 * 1024), + ))) + scan_config.scan_limiter_db = resolve_optional_path( + global_config.get('scan_limiter_db') or getattr(scan_config, 'scan_limiter_db', ''), + global_config, + ) + scan_config.scan_slot_wait_sec = float(global_config.get('scan_slot_wait_sec', getattr(scan_config, 'scan_slot_wait_sec', 0.5))) + scan_config.scan_slot_wait_log_sec = int(global_config.get('scan_slot_wait_log_sec', getattr(scan_config, 'scan_slot_wait_log_sec', 30))) + scan_config.scan_slot_stale_sec = int(global_config.get('scan_slot_stale_sec', getattr(scan_config, 'scan_slot_stale_sec', 7200))) + scan_config.low_space_cleanup_max_items = int(global_config.get('low_space_cleanup_max_items', getattr(scan_config, 'low_space_cleanup_max_items', 50))) + scan_config.drop_detectors = csv_items(global_config.get('drop_detectors', getattr(scan_config, 'drop_detectors', []))) + scan_config.jsonl_rotation_enabled = bool_config(global_config.get('jsonl_rotation_enabled'), getattr(scan_config, 'jsonl_rotation_enabled', False)) + scan_config.found_secrets_max_mb = int(global_config.get('found_secrets_max_mb', getattr(scan_config, 'found_secrets_max_mb', 512))) + scan_config.scan_results_max_mb = int(global_config.get('scan_results_max_mb', getattr(scan_config, 'scan_results_max_mb', 1024))) + scan_config.scan_errors_max_mb = int(global_config.get('scan_errors_max_mb', getattr(scan_config, 'scan_errors_max_mb', 64))) + scan_config.scan_errors_keep = int(global_config.get('scan_errors_keep', getattr(scan_config, 'scan_errors_keep', 5))) + scan_config.jsonl_lock_stale_sec = int(global_config.get('jsonl_lock_stale_sec', getattr(scan_config, 'jsonl_lock_stale_sec', 300))) + scan_config.jsonl_max_segments = int(global_config.get('jsonl_max_segments', getattr(scan_config, 'jsonl_max_segments', 16))) + scan_config.jsonl_ledger_max_rows = int(global_config.get('jsonl_ledger_max_rows', getattr(scan_config, 'jsonl_ledger_max_rows', 1000000))) + scan_config.jsonl_ledger_max_bytes = int(global_config.get('jsonl_ledger_max_bytes', getattr(scan_config, 'jsonl_ledger_max_bytes', 512 * 1024 * 1024))) + scan_config.jsonl_legacy_index_max_bytes = int(global_config.get('jsonl_legacy_index_max_bytes', getattr(scan_config, 'jsonl_legacy_index_max_bytes', 16 * 1024 * 1024))) + scan_config.jsonl_tail_scan_max_bytes = int(global_config.get('jsonl_tail_scan_max_bytes', getattr(scan_config, 'jsonl_tail_scan_max_bytes', 8 * 1024 * 1024))) + scan_config.jsonl_torn_quarantine_max_bytes = int(global_config.get('jsonl_torn_quarantine_max_bytes', getattr(scan_config, 'jsonl_torn_quarantine_max_bytes', 64 * 1024))) + scan_config.dockerhub_tag_cache_path = resolve_optional_path( + global_config.get('dockerhub_tag_cache_path') or getattr(scan_config, 'dockerhub_tag_cache_path', ''), + global_config, + ) if (global_config.get('dockerhub_tag_cache_path') or getattr(scan_config, 'dockerhub_tag_cache_path', '')) else '' + scan_config.dockerhub_tag_cache_ttl_sec = int(global_config.get('dockerhub_tag_cache_ttl_sec', getattr(scan_config, 'dockerhub_tag_cache_ttl_sec', 21600))) + scan_config.dockerhub_tag_negative_cache_ttl_sec = int(global_config.get('dockerhub_tag_negative_cache_ttl_sec', getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600))) + scan_config.dockerhub_tag_rate_limit_cache_ttl_sec = int(global_config.get('dockerhub_tag_rate_limit_cache_ttl_sec', getattr(scan_config, 'dockerhub_tag_rate_limit_cache_ttl_sec', 1800))) + scan_config.dockerhub_tag_cache_max_rows = int(global_config.get('dockerhub_tag_cache_max_rows', getattr(scan_config, 'dockerhub_tag_cache_max_rows', 50000))) + scan_config.dockerhub_tag_cache_max_age_sec = int(global_config.get('dockerhub_tag_cache_max_age_sec', getattr(scan_config, 'dockerhub_tag_cache_max_age_sec', 7 * 86400))) + scan_config.dockerhub_tag_cache_max_bytes = int(global_config.get('dockerhub_tag_cache_max_bytes', getattr(scan_config, 'dockerhub_tag_cache_max_bytes', 256 * 1024 * 1024))) + scan_config.dockerhub_tag_cache_min_free_bytes = int(global_config.get('dockerhub_tag_cache_min_free_bytes', getattr(scan_config, 'dockerhub_tag_cache_min_free_bytes', 512 * 1024 * 1024))) + scan_config.trufflehog_path = global_config.get('trufflehog_path', scan_config.trufflehog_path) + scan_config.trufflehog_config = resolve_optional_path( + global_config.get('trufflehog_config') or getattr(scan_config, 'trufflehog_config', ''), + global_config, + ) if (global_config.get('trufflehog_config') or getattr(scan_config, 'trufflehog_config', '')) else '' + scan_config.trufflehog_job_memory_limit_bytes = int(global_config.get( + 'trufflehog_job_memory_limit_bytes', + getattr(scan_config, 'trufflehog_job_memory_limit_bytes', 4096 * 1024 * 1024), + )) + scan_config.trufflehog_windows_job_cpu_weight = int(global_config.get( + 'trufflehog_windows_job_cpu_weight', + getattr(scan_config, 'trufflehog_windows_job_cpu_weight', 0), + )) + scan_config.trufflehog_windows_memory_priority = int(global_config.get( + 'trufflehog_windows_memory_priority', + getattr(scan_config, 'trufflehog_windows_memory_priority', 0), + )) + scan_config.trufflehog_stdout_max_mb = int(global_config.get( + 'trufflehog_stdout_max_mb', getattr(scan_config, 'trufflehog_stdout_max_mb', 32), + )) + scan_config.trufflehog_stderr_max_mb = int(global_config.get( + 'trufflehog_stderr_max_mb', getattr(scan_config, 'trufflehog_stderr_max_mb', 8), + )) + if 'min_free_gb' in global_config: + scan_config.min_free_gb = float(global_config['min_free_gb']) + if 'no_verification' in global_config: + scan_config.no_verification = bool(global_config['no_verification']) + if 'strict_git_provider_token_filter' in global_config: + scan_config.strict_git_provider_token_filter = bool(global_config['strict_git_provider_token_filter']) + + +def configure_source_auth(source_name, source_config, state=None, secrets=None, auth_entry=None): + if source_to_platform(source_name) != 'docker': + return + + if secrets is not None and state is not None and source_config.get('auth_pool'): + entries = available_auth_entries(source_name, source_config, state, secrets) + accounts = [ + { + 'name': entry.get('name'), + 'username': entry.get('username'), + 'token': entry.get('token'), + } + for entry in entries + if entry.get('username') and entry.get('token') + ] + configure_docker_discovery_accounts( + accounts, cooldown_sec=int(source_config.get('rate_limit_cooldown', 1800)), + ) + endpoint_status = state['sources'].setdefault( + source_name, default_source_state(), + ).get('auth_endpoint_status') + if endpoint_status: + restore_docker_endpoint_cooldowns(endpoint_status) + print(f"Docker auth pool: configured {len(accounts)} available account(s)") + return + + docker_token = ( + (auth_entry or {}).get('token') + or source_config.get('docker_token') + or source_config.get('token') + or os.getenv('DOCKERHUB_TOKEN') + or os.getenv('DOCKER_TOKEN') + or os.getenv('DOCKER_TOKENS') + ) + docker_username = (auth_entry or {}).get('username') or source_config.get('docker_username') or os.getenv('DOCKERHUB_USERNAME') or os.getenv('DOCKER_USERNAME') + configure_docker_discovery_tokens(docker_token, docker_username) + + +def persist_docker_auth_events(source_name, source_config, state, secrets): + if source_to_platform(source_name) != 'docker': + return + entries, _ = auth_pool_entries(source_config, secrets) + entries_by_name = {str(entry.get('name')): entry for entry in entries} + cooldown = int(source_config.get('rate_limit_cooldown', 1800)) + source_state = state['sources'].setdefault(source_name, default_source_state()) + endpoint_status = source_state.setdefault('auth_endpoint_status', {}) + for event in drain_docker_auth_events(): + entry = entries_by_name.get(str(event.get('name') or '')) + if not entry: + continue + category = str(event.get('category') or '') + endpoint = str(event.get('endpoint') or '') + if endpoint == 'hub_search' and category != 'auth_invalid': + account_status = endpoint_status.setdefault(endpoint, {}).setdefault( + str(entry.get('name')), {}, + ) + now = utc_now() + if category == 'ok': + account_status['disabled_until'] = None + account_status['disabled_reason'] = None + account_status['status'] = 'ok' + account_status['last_success_at'] = now.isoformat(timespec='seconds') + account_status['success_count'] = int( + account_status.get('success_count', 0) + ) + 1 + else: + account_status['disabled_until'] = event.get('reset_at') or ( + now + timedelta(seconds=cooldown) + ).isoformat(timespec='seconds') + account_status['disabled_reason'] = category or 'rate_limit' + account_status['status'] = 'limited' + account_status['last_error'] = str(event.get('message') or '')[:500] + account_status['last_error_at'] = now.isoformat(timespec='seconds') + account_status['failures'] = int(account_status.get('failures', 0)) + 1 + continue + if category == 'ok': + clear_auth_rate_limit(source_name, entry, state) + else: + mark_auth_rate_limited( + source_name, entry, state, + reset_at=event.get('reset_at'), cooldown=cooldown, + category=category or 'rate_limit', + message=event.get('message') or 'Docker authentication unavailable', + ) + refresh_auth_summary(source_name, source_config, state, secrets) + + +def run_configured_source( + source_name, config, state, state_path, secrets, db=None, run_id=None, + cycle_runner=None, +): + selected_cycle_runner = run_cycle if cycle_runner is None else cycle_runner + if not callable(selected_cycle_runner): + raise TypeError('Configured source cycle runner must be callable') + source_config = config['sources'][source_name] + global_config = config.get('global', {}) + query, query_index, query_count = current_query_for_source(source_name, source_config, state) + source_state = state['sources'].setdefault(source_name, default_source_state()) + + print(f'\n=== {source_name} source cycle ===') + print(f'Query: {query_index + 1}/{query_count} -> {query!r}') + print(f"Mode: {source_config.get('mode', 'recent')}") + + source_state['last_query'] = query + source_state['last_started_at'] = datetime.now().isoformat(timespec='seconds') + source_state['last_status'] = 'running' + save_state(state_path, state) + + auth_entry = select_auth_entry(source_name, source_config, state, secrets) + auth_name = auth_entry.get('name') if auth_entry else 'none' + refresh_auth_summary(source_name, source_config, state, secrets, auth_name) + print(f"Auth: {auth_name} (rotation={source_config.get('auth_rotation', 'per_cycle')})") + save_state(state_path, state) + + args = build_args_from_source_config(source_name, source_config, global_config, query, auth_entry) + dockerhub_policies = {} + dockerhub_annotation = None + dockerhub_experiment = None + if ( + source_name == 'dockerhub' + and args.platform == 'docker' + and args.mode == 'search' + ): + validated_depth = validate_docker_depth_config( + config, + managed_postgres=bool( + db and getattr(db, 'conn', None) + and getattr(db.conn, 'is_postgres', False) + ), + final_cutover=( + source_config.get( + 'sync_file_queues', global_config.get('sync_file_queues', True), + ) is False + ), + ) + dockerhub_experiment = validated_depth.experiment + dockerhub_policies = configured_dockerhub_discovery_policies(source_config) + annotate_dockerhub_runtime_args( + args, dockerhub_experiment, dockerhub_policies, + ) + if dockerhub_experiment is not None: + policy_hashes = { + policy['policy_sha256'] for policy in dockerhub_policies.values() + } + generation_check = getattr( + db, 'dockerhub_discovery_generation_complete', None, + ) + if not callable(generation_check): + raise RuntimeError( + 'DockerHub collection generation authority is unavailable' + ) + generation_complete = generation_check( + 'dockerhub', dockerhub_experiment.collection_generation, + dockerhub_experiment.ordered_query_hash, + len(dockerhub_experiment.queries), + next(iter(policy_hashes)), + ) + else: + generation_complete = True + dockerhub_annotation = prepare_dockerhub_discovery_state( + source_state, dockerhub_policies, query, + force_deep=not generation_complete, + ) + annotate_dockerhub_discovery_args(args, dockerhub_annotation) + mark_dockerhub_discovery_incomplete( + source_state, query, dockerhub_annotation['policy_sha256'], + ) + save_state(state_path, state) + scan_config.trufflehog_job_memory_limit_bytes = int(getattr( + args, 'trufflehog_job_memory_limit_bytes', scan_config.trufflehog_job_memory_limit_bytes, + )) + if source_to_platform(source_name) == 'postman': + source_state.setdefault('postman_auth_status', {}) + args.github_tokens = github_token_entries_for_postman(source_config, secrets) + args.github_auth_status = source_state['postman_auth_status'] + if args.github_tokens: + print(f"Postman GitHub auth pool: configured {len(args.github_tokens)} token(s)") + configure_source_auth(source_name, source_config, state, secrets, auth_entry) + cycle_id = None + if db and run_id: + cycle_id = db.start_source_cycle( + run_id, + source_name, + args.platform, + args.mode, + query, + query_index + 1, + query_count, + auth_name, + source_config, + queue_counts_for_args(args), + ) + if getattr(db.conn, 'is_postgres', False) and cycle_id is None: + raise RuntimeError(f'Unable to create Postgres source cycle for {source_name}; refusing to claim targets') + annotate_dockerhub_runtime_args( + args, dockerhub_experiment, dockerhub_policies, cycle_id, + ) + + all_auth_cooling_down = False + final_status = 'completed' + cycle_metrics = {} + attempt_finished = False + + def failure_metrics(**extra): + metrics = {'fetched_count': 0} + metrics.update(cycle_metrics) + metrics.update(extra) + return metrics + + while True: + try: + attempt_finished = False + cycle_metrics = selected_cycle_runner( + args, db, run_id, cycle_id, source_name, partial_metrics=cycle_metrics, + ) + persist_docker_auth_events(source_name, source_config, state, secrets) + attempt_finished = True + returned_status = str(cycle_metrics.get('cycle_status') or '') + if returned_status in ( + 'completed', 'completed_with_retries', 'query_invalid', 'failed', + 'source_failed', 'backlog_only', 'paused', + ): + final_status = returned_status + elif cycle_metrics.get('backlog_only'): + final_status = 'backlog_only' + else: + final_status = 'completed' + scanned = int(cycle_metrics.get('staged_count') or cycle_metrics.get('scanned_count', 0)) + if cycle_metrics.get('source_failure_count'): + failure_category = cycle_metrics.get('source_failure_category') or 'source_failed' + if failure_category == 'source_auth': + failure_category = 'auth_invalid' + raise RateLimitError( + source_name, + cycle_metrics.get('source_failure_message') or 'source-wide scanner failure', + category=failure_category, + auth_related=bool(cycle_metrics.get('source_failure_auth_related')), + ) + if source_to_platform(source_name) != 'docker' or not source_config.get('auth_pool'): + clear_auth_rate_limit(source_name, auth_entry, state) + refresh_auth_summary(source_name, source_config, state, secrets, auth_name) + break + except RateLimitError as e: + persist_docker_auth_events(source_name, source_config, state, secrets) + cooldown = int(source_config.get('rate_limit_cooldown', 3600)) + category = getattr(e, 'category', 'rate_limit') + print(f"{source_name} API {category}: {e}") + if not getattr(e, 'auth_related', True): + scanned = 0 + if final_status != 'source_failed': + final_status = 'query_invalid' if category == 'query_invalid' else 'failed' + if db and cycle_id and not attempt_finished: + db.finish_source_cycle(cycle_id, final_status, failure_metrics(), queue_counts_for_args(args), str(e)) + break + + mark_auth_rate_limited(source_name, auth_entry, state, e.reset_at, cooldown, category, str(e)) + refresh_auth_summary(source_name, source_config, state, secrets, auth_name) + save_state(state_path, state) + if not source_config.get('retry_with_next_auth_on_rate_limit', True): + if final_status != 'source_failed': + final_status = 'auth_failed' if category.startswith('auth_') else 'rate_limited' + scanned = 0 + if db and cycle_id and not attempt_finished: + db.finish_source_cycle(cycle_id, final_status, failure_metrics(), queue_counts_for_args(args), str(e)) + break + auth_entry = select_auth_entry(source_name, source_config, state, secrets) + if not auth_entry: + print(f"All {source_name} auth tokens are unavailable; skipping this source cycle.") + scanned = 0 + all_auth_cooling_down = True + if final_status != 'source_failed': + final_status = 'auth_failed' if category.startswith('auth_') else 'rate_limited' + refresh_auth_summary(source_name, source_config, state, secrets, 'none') + if db and cycle_id and not attempt_finished: + db.finish_source_cycle(cycle_id, final_status, failure_metrics(), queue_counts_for_args(args), str(e)) + break + auth_name = auth_entry.get('name') + refresh_auth_summary(source_name, source_config, state, secrets, auth_name) + print(f"Retrying {source_name} with auth {auth_name}") + args = build_args_from_source_config(source_name, source_config, global_config, query, auth_entry) + annotate_dockerhub_runtime_args( + args, dockerhub_experiment, dockerhub_policies, + ) + if dockerhub_annotation is not None: + annotate_dockerhub_discovery_args(args, dockerhub_annotation) + scan_config.trufflehog_job_memory_limit_bytes = int(getattr( + args, 'trufflehog_job_memory_limit_bytes', scan_config.trufflehog_job_memory_limit_bytes, + )) + if source_to_platform(source_name) == 'postman': + args.github_tokens = github_token_entries_for_postman(source_config, secrets) + args.github_auth_status = source_state.setdefault('postman_auth_status', {}) + configure_source_auth(source_name, source_config, state, secrets, auth_entry) + if db and run_id: + cycle_id = db.start_source_cycle( + run_id, source_name, args.platform, args.mode, query, + query_index + 1, query_count, auth_name, source_config, queue_counts_for_args(args), + ) + if getattr(db.conn, 'is_postgres', False) and cycle_id is None: + raise RuntimeError(f'Unable to create Postgres auth-retry cycle for {source_name}') + annotate_dockerhub_runtime_args( + args, dockerhub_experiment, dockerhub_policies, cycle_id, + ) + cycle_metrics = {} + except Exception as e: + persist_docker_auth_events(source_name, source_config, state, secrets) + if db and getattr(db, 'conn', None): + try: + db.conn.rollback() + except Exception: + pass + discovery_transport_failure = ( + isinstance(e, GitLabDiscoveryTransportError) and source_name == 'gitlab' + ) or ( + isinstance(e, DockerHubDiscoveryTransportError) + and source_to_platform(source_name) == 'docker' + ) + if ( + isinstance(e, DockerHubDiscoveryTransportError) + and not getattr(e, 'retryable', True) + ): + if db and cycle_id and not attempt_finished: + db.finish_source_cycle( + cycle_id, 'failed', failure_metrics(), + queue_counts_for_args(args), str(e), + ) + raise + if discovery_transport_failure: + scanned = 0 + final_status = 'failed' + cycle_metrics = failure_metrics(discovery_transport_failed=True) + print( + f'{source_name} API network: discovery cycle failed ' + f'after bounded retries: {e}' + ) + if db and cycle_id and not attempt_finished: + db.finish_source_cycle( + cycle_id, final_status, cycle_metrics, + queue_counts_for_args(args), str(e), + ) + break + if db and cycle_id and not attempt_finished: + db.finish_source_cycle(cycle_id, 'failed', failure_metrics(), queue_counts_for_args(args), str(e)) + raise + + source_state['last_completed_at'] = datetime.now().isoformat(timespec='seconds') + source_state['last_status'] = final_status + source_state['cycles'] = int(source_state.get('cycles', 0)) + 1 + source_state['last_scanned'] = scanned + if selected_cycle_runner is run_discovery_cycle: + source_state['last_cycle_result'] = { + 'status': final_status, + 'fetched_count': max(0, int(cycle_metrics.get('fetched_count', 0) or 0)), + 'queued_new_count': max(0, int(cycle_metrics.get('queued_new_count', 0) or 0)), + 'queued_updated_count': max(0, int(cycle_metrics.get('queued_updated_count', 0) or 0)), + } + if final_status in ('completed', 'completed_with_retries'): + source_state['last_discovery_success_at'] = source_state['last_completed_at'] + source_state['last_error_category'] = '' + else: + source_state['last_error_category'] = ( + final_status if final_status in ( + 'auth_failed', 'failed', 'paused', 'query_invalid', + 'rate_limited', 'source_failed', + ) else 'runtime_error' + ) + if ( + dockerhub_annotation is not None + and dockerhub_annotation['deep'] + and final_status in ('completed', 'completed_with_retries') + and cycle_metrics.get('deep_dispatch_durable') is True + ): + mark_dockerhub_deep_dispatched( + source_state, query, dockerhub_annotation['policy_sha256'], + ) + if ( + dockerhub_annotation is not None + and final_status in ('completed', 'completed_with_retries', 'query_invalid') + ): + clear_dockerhub_discovery_incomplete(source_state, query) + if final_status in ('completed', 'completed_with_retries', 'query_invalid'): + advance_query_for_source(source_name, source_config, state) + cycle_metrics['cycle_status'] = final_status + save_state(state_path, state) + if dockerhub_policies: + try: + retry_metrics = process_dockerhub_discovery_retry( + args, db, source_name, dockerhub_policies, + ) + cycle_metrics.update(retry_metrics) + except Exception: + logger.warning('DockerHub discovery retry lane failed') + try: + persist_docker_auth_events(source_name, source_config, state, secrets) + save_state(state_path, state) + except Exception: + logger.warning('DockerHub discovery retry auth-state persistence failed') + next_query, next_index, next_count = current_query_for_source(source_name, source_config, state) + print(f'{source_name} complete. Next query: {next_index + 1}/{next_count} -> {next_query!r}') + return cycle_metrics + + +def print_state(state_path, state): + print(f'State file: {state_path}') + print(json.dumps(state, indent=2, ensure_ascii=False)) + + +def current_process_private_bytes(): + if os.name != 'nt': + return 0 + try: + counters = _PROCESS_MEMORY_COUNTERS_EX() + counters.cb = ctypes.sizeof(counters) + if not _GET_PROCESS_MEMORY_INFO( + _METRIC_GET_CURRENT_PROCESS(), ctypes.byref(counters), counters.cb, + ): + return 0 + return int(counters.PrivateUsage) + except Exception: + return 0 + + +def run_discovery_producer_mode(args, config=None): + if config is None: + config = load_config(args.config, managed_postgres=True, final_cutover=True) + if args.show_state or args.cleanup_only or not args.once: + raise SystemExit('discovery-producer requires an authenticated one-shot source invocation') + source_name = str(args.source or '') + if source_name not in DISCOVERY_PRODUCER_SOURCES: + raise SystemExit('discovery-producer source is outside the canonical allowlist') + sources = config.get('sources') or {} + if source_name not in sources: + raise SystemExit(f'Source {source_name} is not present in config') + + global_config = config.get('global', {}) + preflight_lifecycle_paths( + args.config, config, authority_profile=DISCOVERY_PRODUCER_ROLE, + ) + apply_global_config(global_config) + require_sensitive_runtime_paths(global_config, create=False) + state_path = get_state_path(config, args.config) + state = load_state(state_path, config) + secrets = load_secrets(config, args.config) + refresh_auth_summary(source_name, sources[source_name], state, secrets) + save_state(state_path, state) + + db = ScannerDB( + scan_config.results_dir, + global_config.get('database_path') or global_config.get('db_path'), + db_url=global_config.get('database_url'), + initialize=False, + ) + if not db.enabled or not getattr(getattr(db, 'conn', None), 'is_postgres', False): + db.close() + raise RuntimeError('discovery-producer requires the managed PostgreSQL authority') + run_id = None + run_status = 'completed' + run_error = None + try: + set_application_name = getattr(db, 'set_application_name', None) + if set_application_name: + set_application_name(f'truf-discovery:{source_name}') + db.require_runtime_safety_schema() + db.require_final_cutover() + run_id = db.start_run( + DISCOVERY_PRODUCER_ROLE, + sys.argv, + selected_source=source_name, + selected_platform=source_to_platform(source_name), + config_path=args.config, + config_hash=hash_file(args.config), + enabled_sources=[source_name], + global_config=global_config, + ) + if run_id is None: + raise RuntimeError('Unable to create PostgreSQL discovery-producer run') + run_configured_source( + source_name, config, state, state_path, secrets, db, run_id, + cycle_runner=run_discovery_cycle, + ) + except BaseException as exc: + run_status = 'failed' + run_error = type(exc).__name__ + source_state = state['sources'].setdefault(source_name, default_source_state()) + source_state['last_status'] = 'failed' + source_state['last_error_category'] = 'runtime_error' + source_state['last_cycle_result'] = {'status': 'failed'} + save_state(state_path, state) + raise + finally: + if run_id is not None: + db.finish_run(run_id, run_status, run_error) + db.close() + + +def run_config_mode(args, config=None, runtime_role='scanner'): + if runtime_role == DISCOVERY_PRODUCER_ROLE: + return run_discovery_producer_mode(args, config=config) + if runtime_role != 'scanner': + raise SystemExit('unsupported console runtime role') + if config is None: + config = load_config( + args.config, + managed_postgres=bool(os.getenv('TRUF_MANAGED_POSTGRES_DSN')), + ) + global_config = config.get('global', {}) + preflight_lifecycle_paths(args.config, config) + apply_global_config(global_config) + + state_path = get_state_path(config, args.config) + state = load_state(state_path, config) + if args.show_state: + print_state(state_path, state) + return + + initialize_scanner_runtime(preflight_complete=True) + require_sensitive_runtime_paths(global_config, create=False) + secrets = load_secrets(config, args.config) + for source_name, source_config in (config.get('sources') or {}).items(): + refresh_auth_summary(source_name, source_config, state, secrets) + save_state(state_path, state) + + if args.cleanup_only: + raise SystemExit('Source-side cleanup is retired; use the authenticated janitor worker') + + if not check_dependencies(): + raise SystemExit(1) + + source_names = enabled_source_names(config, args.source) + if not source_names: + raise SystemExit('No enabled sources found in config') + + loop_enabled = bool(global_config.get('loop', True)) and not args.once + cooldown = int(global_config.get('cooldown', args.cooldown)) + db = ScannerDB( + scan_config.results_dir, + global_config.get('database_path') or global_config.get('db_path'), + db_url=global_config.get('database_url'), + initialize=bool(global_config.get('database_initialize', False)), + ) + if db.enabled: + set_application_name = getattr(db, 'set_application_name', None) + if set_application_name: + set_application_name( + f'truf-source:{source_names[0]}' if len(source_names) == 1 else 'truf-scanner' + ) + if getattr(db, 'postgres_required', False) and not db.enabled: + raise RuntimeError('Configured Postgres database is unavailable; refusing file-queue fallback') + if db.enabled and getattr(db.conn, 'is_postgres', False): + db.require_runtime_safety_schema() + run_id = db.start_run( + 'config', + sys.argv, + selected_source=args.source, + config_path=args.config, + config_hash=hash_file(args.config), + enabled_sources=source_names, + global_config=global_config, + ) if db.enabled else None + if db.enabled and getattr(db.conn, 'is_postgres', False) and run_id is None: + raise RuntimeError('Unable to create Postgres scanner run; refusing target claims') + run_status = 'completed' + run_error = None + try: + while True: + print(f'\n=== Config cycle started at {datetime.now().isoformat(timespec="seconds")} ===') + backlog_progress = False + for source_name in source_names: + if not config['sources'][source_name].get('enabled', False) and not args.source: + continue + cycle_metrics = run_configured_source( + source_name, config, state, state_path, secrets, db, run_id, + ) or {} + backlog_progress = backlog_progress or bool( + cycle_metrics.get('backlog_only') + and (cycle_metrics.get('staged_count') or cycle_metrics.get('scanned_count')) + ) + + if not loop_enabled: + break + sleep_seconds = ( + max(0.1, float(global_config.get('backlog_poll_sec', 0.5) or 0.5)) + if backlog_progress else cooldown + ) + print(f'Sleeping {sleep_seconds:g} seconds...') + time.sleep(sleep_seconds) + except KeyboardInterrupt: + run_status = 'stopped' + save_state(state_path, state) + print('Stopped. State preserved. Temp cleanup deferred to next run or --cleanup-only.') + except UnresolvedHandoffInfrastructureError as e: + run_status = 'failed' + run_error = str(e) + raise SystemExit(SOURCE_INFRASTRUCTURE_HOLD_EXIT) from e + except Exception as e: + run_status = 'failed' + run_error = str(e) + raise + finally: + if db.enabled and run_id: + db.finish_run(run_id, run_status, run_error) + db.close() + + +def parse_args(): + default_paths = default_project_paths() + parser = argparse.ArgumentParser(description='Console runner for TruffleHog scans.') + parser.add_argument('--config') + parser.add_argument('--source', choices=['github', 'github_archive', 'github_archive_files', 'github_gists', 'gitlab', 'docker', 'dockerhub', 'npm', 'pypi', 'package_git', 'huggingface', 'postman', 'github_actions', 'gitlab_ci']) + parser.add_argument('--once', action='store_true') + parser.add_argument('--show-state', action='store_true') + parser.add_argument('--platform', choices=['github', 'github_archive', 'github_archive_files', 'github_gists', 'gitlab', 'docker', 'dockerhub', 'npm', 'pypi', 'package_git', 'huggingface', 'postman', 'github_actions', 'gitlab_ci']) + parser.add_argument('--mode', choices=['recent', 'search', 'custom', 'archive'], default='recent') + parser.add_argument('--query', default='') + parser.add_argument('--query-file') + parser.add_argument('--pages', type=int, default=1) + parser.add_argument('--per-page', type=int, default=50) + parser.add_argument('--target-file') + parser.add_argument('--token') + parser.add_argument('--docker-username') + parser.add_argument('--docker-token') + parser.add_argument('--workers', type=int, default=6) + parser.add_argument('--timeout', type=int, default=None) + parser.add_argument('--detectors', default=scan_config.detectors) + parser.add_argument('--exclude-detectors', default=scan_config.exclude_detectors) + parser.add_argument('--drop-detectors', default=','.join(getattr(scan_config, 'drop_detectors', []))) + parser.add_argument('--no-verification', action='store_true', default=scan_config.no_verification) + parser.add_argument('--save-dir', default=scan_config.results_dir) + parser.add_argument('--runtime-dir', default=default_paths['runtime_dir']) + parser.add_argument('--result-bundle-dir', default=default_paths['result_bundle_dir']) + parser.add_argument('--result-bundle-max-event-bytes', type=int, default=scan_config.result_bundle_max_event_bytes) + parser.add_argument('--result-spool-dir', default=default_paths['result_spool_dir']) + parser.add_argument('--result-spool-max-event-bytes', type=int, default=scan_config.result_spool_max_event_bytes) + parser.add_argument('--result-spool-max-events', type=int, default=scan_config.result_spool_max_events) + parser.add_argument('--result-spool-max-total-bytes', type=int, default=scan_config.result_spool_max_total_bytes) + parser.add_argument('--result-spool-min-free-bytes', type=int, default=scan_config.result_spool_min_free_bytes) + parser.add_argument('--result-spool-wait-sec', type=float, default=1.0) + parser.add_argument('--result-spool-diagnostic-interval-sec', type=float, default=30.0) + parser.add_argument('--scan-outbox-max-pending-items', type=int, default=scan_config.scan_outbox_max_pending_items) + parser.add_argument('--scan-outbox-max-pending-bytes', type=int, default=scan_config.scan_outbox_max_pending_bytes) + parser.add_argument('--scan-outbox-max-pending-age-sec', type=int, default=scan_config.scan_outbox_max_pending_age_sec) + parser.add_argument('--queue-dir', default=default_paths['queue_dir']) + parser.add_argument('--work-dir', default=scan_config.work_dir) + parser.add_argument('--trufflehog-path', default=scan_config.trufflehog_path) + parser.add_argument('--trufflehog-config', default=getattr(scan_config, 'trufflehog_config', '')) + parser.add_argument('--trufflehog-concurrency', type=int, default=0) + parser.add_argument('--loop', action='store_true') + parser.add_argument('--cooldown', type=int, default=300) + parser.add_argument('--max-cycles', type=int, default=0) + parser.add_argument('--max-targets', type=int, default=0) + parser.add_argument('--target-retry-max-attempts', type=int, default=3) + parser.add_argument('--target-retry-base-delay-sec', type=int, default=3600) + parser.add_argument('--target-retry-max-delay-sec', type=int, default=86400) + parser.add_argument('--target-timeout-retry-delay-sec', type=int, default=21600) + parser.add_argument('--target-claim-batch-size', type=int, default=0) + parser.add_argument('--recent-hours', type=int, default=24) + parser.add_argument('--recent-days', type=int, default=7) + parser.add_argument('--max-version-age-days', type=int, default=0) + parser.add_argument('--versions-per-package', type=int, default=1) + parser.add_argument('--package-sources', default='npm,pypi') + parser.add_argument('--package-git-refresh-registry', action='store_true') + parser.add_argument('--search-kinds', default='collection,environment') + parser.add_argument('--gist-since', default='') + parser.add_argument('--postman-cache-dir', default=default_paths.get('postman_cache_dir')) + parser.add_argument('--gharchive-cache-dir', default=default_paths.get('gharchive_cache_dir')) + parser.add_argument('--max-artifact-size-mb', type=int, default=50) + parser.add_argument('--max-file-age-days', type=int, default=365) + parser.add_argument('--github-code-search-rpm', type=int, default=8) + parser.add_argument('--all-tokens-cooldown', type=int, default=1800) + parser.add_argument('--max-repo-age-days', type=int, default=0) + parser.add_argument('--repo-age-field', default='created_at') + parser.add_argument('--max-commit-age-days', type=int, default=0) + parser.add_argument('--commit-lookup-pages', type=int, default=3) + parser.add_argument('--no-skip-if-commit-lookup-fails', dest='skip_if_commit_lookup_fails', action='store_false') + parser.set_defaults(skip_if_commit_lookup_fails=True) + parser.add_argument('--max-depth', type=int, default=0) + parser.add_argument('--exact-git-planning', dest='exact_git_planning_enabled', action='store_true') + parser.add_argument('--git-baseline-depth', type=int, default=100) + parser.add_argument('--git-ref-resolution-timeout-sec', type=float, default=10) + parser.add_argument('--git-ref-resolution-attempts', type=int, default=2) + parser.add_argument('--git-ref-resolution-max-bytes', type=int, default=1 << 20) + parser.add_argument('--admission-resolution-attempts', type=int, default=90) + parser.add_argument('--admission-resolution-seconds', type=float, default=90) + parser.add_argument('--admission-resolution-retry-delay-sec', type=float, default=1) + parser.add_argument('--scan-full-history', action='store_true') + parser.add_argument('--sort-by', default='updated') + parser.add_argument('--sort-order', choices=['asc', 'desc'], default='desc') + parser.add_argument('--created-filter', choices=['any', 'today', 'week', 'month', 'year'], default='any') + parser.add_argument('--gitlab-sort-by', default='last_activity_at') + parser.add_argument('--gitlab-visibility', choices=['public', 'internal', 'private'], default='public') + parser.add_argument('--docker-sort-by', default='updated_at') + parser.add_argument('--fetch-workers', type=int, default=8) + parser.add_argument('--fetch-timeout', type=int, default=15) + parser.add_argument('--tag-fetch-workers', type=int, default=4) + parser.add_argument('--tag-retry-count', type=int, default=2) + parser.add_argument('--tag-retry-delay', type=int, default=5) + parser.add_argument('--tag-resolve-limit', type=int, default=100) + parser.add_argument('--no-docker-platform-filter', dest='docker_platform_filter_enabled', action='store_false') + parser.add_argument('--docker-platform-os', default='linux') + parser.add_argument('--docker-platform-arch', default='amd64') + parser.add_argument('--docker-platform-candidate-tags', type=int, default=20) + def docker_image_depth_argument(value): + try: + return docker_images_per_repository_limit(int(value, 10)) + except (TypeError, ValueError) as exc: + raise argparse.ArgumentTypeError( + 'docker image depth must be an integer from 1 through 10' + ) from exc + + parser.add_argument('--docker-images-per-repository', type=docker_image_depth_argument, default=1) + parser.add_argument( + '--docker-content-scan-mode', + choices=['full', 'canary', 'layer', 'adaptive-canary', 'adaptive'], + default='full', + ) + parser.add_argument('--docker-layer-canary-basis-points', type=int, default=0) + parser.add_argument('--docker-adaptive-canary-basis-points', type=int, default=0) + parser.add_argument('--docker-adaptive-gate-max-age-sec', type=int, default=604800) + parser.add_argument('--docker-layer-config-max-bytes', type=int, default=1 << 20) + parser.add_argument('--docker-layer-max-bytes', type=int, default=256 << 20) + parser.add_argument('--docker-layer-image-max-bytes', type=int, default=1 << 30) + parser.add_argument('--docker-layer-max-layers', type=int, default=8) + parser.add_argument('--docker-layer-archive-max-size-bytes', type=int, default=256 << 20) + parser.add_argument('--docker-layer-archive-max-depth', type=int, default=4) + parser.add_argument('--docker-layer-archive-timeout-sec', type=int, default=30) + parser.add_argument('--docker-layer-blob-timeout-sec', type=int, default=600) + parser.add_argument('--docker-layer-filesystem-concurrency', type=int, default=2) + parser.add_argument('--docker-layer-blob-max-attempts', type=int, default=3) + parser.add_argument('--docker-layer-blob-lease-sec', type=int, default=1800) + parser.add_argument('--docker-layer-min-free-bytes', type=int, default=20 << 30) + parser.add_argument('--docker-layer-checkpoint-delay-sec', type=int, default=60) + parser.add_argument('--docker-adaptive-checkpoint-max-blobs', type=int, default=4) + parser.add_argument('--docker-adaptive-checkpoint-max-bytes', type=int, default=512 << 20) + parser.add_argument('--docker-repository-refresh-interval-sec', type=int, default=86400) + parser.add_argument('--docker-repository-refresh-max-per-cycle', type=int, default=0) + parser.set_defaults(docker_platform_filter_enabled=True) + parser.add_argument('--archive-hours-back', type=int, default=6) + parser.add_argument('--archive-max-repos-per-cycle', type=int, default=200) + parser.add_argument('--archive-max-files-per-cycle', type=int, default=300) + parser.add_argument('--archive-max-commit-lookups', type=int, default=200) + parser.add_argument('--archive-rescan-cooldown-hours', type=int, default=48) + parser.add_argument('--archive-event-types', default='PushEvent,CreateEvent,PublicEvent') + parser.add_argument('--ci-seed-sources', default='github,gitlab,package_git') + parser.add_argument('--ci-use-finding-seeds', action='store_true') + parser.add_argument('--ci-max-repos-per-cycle', type=int, default=50) + parser.add_argument('--ci-seed-scan-limit', type=int, default=5000) + parser.add_argument('--ci-soft-cooldown-days', type=int, default=7) + parser.add_argument('--ci-runs-per-repo', type=int, default=5) + parser.add_argument('--ci-pipelines-per-project', type=int, default=5) + parser.add_argument('--ci-jobs-per-pipeline', type=int, default=20) + parser.add_argument('--ci-lookback-days', type=int, default=30) + parser.add_argument('--ci-max-log-archive-mb', type=int, default=50) + parser.add_argument('--ci-max-log-file-mb', type=int, default=20) + parser.add_argument('--ci-max-trace-mb', type=int, default=20) + parser.add_argument('--ci-scan-artifacts', action='store_true') + parser.add_argument('--ci-max-artifacts-per-run', type=int, default=3) + parser.add_argument('--ci-max-artifacts-per-pipeline', type=int, default=5) + parser.add_argument('--ci-max-artifact-archive-mb', type=int, default=50) + parser.add_argument('--ci-max-artifact-file-mb', type=int, default=10) + parser.add_argument('--ci-max-artifact-files', type=int, default=1000) + parser.add_argument('--ci-target-max-download-mb', type=int, default=500) + parser.add_argument('--no-ci-failed-first', dest='ci_failed_first', action='store_false') + parser.set_defaults(ci_failed_first=True) + parser.add_argument('--stop-on-seen-pages', action='store_true') + parser.add_argument('--seen-page-threshold', type=int, default=2) + parser.add_argument('--min-pages-before-stop', type=int, default=1) + parser.add_argument('--cleanup-temp-age-min', type=int, default=120) + parser.add_argument('--cleanup-only', action='store_true') + return parser.parse_args() + + +def main(): + args = parse_args() + if args.show_state and not args.config: + raise SystemExit('--show-state requires an explicit --config path') + config = None + if args.config and os.path.exists(args.config): + config = load_config( + args.config, + managed_postgres=bool(os.getenv('TRUF_MANAGED_POSTGRES_DSN')), + ) + runtime_role = str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() + if not args.show_state: + if not args.config: + raise SystemExit('Legacy direct scanner mode is retired; use canonical supervisor source commands.') + try: + require_active_supervisor_child( + args.config, child_kind=runtime_role, require_dsn=True, + ) + except LifecycleAuthorityError as exc: + raise SystemExit(str(exc)) from exc + if args.config: + run_config_mode(args, config=config, runtime_role=runtime_role) + return + + scan_config.work_dir = args.work_dir + scan_config.results_dir = args.save_dir + scan_config.result_spool_dir = args.result_spool_dir + scan_config.result_spool_max_event_bytes = args.result_spool_max_event_bytes + scan_config.result_spool_max_events = args.result_spool_max_events + scan_config.result_spool_max_total_bytes = args.result_spool_max_total_bytes + scan_config.result_spool_min_free_bytes = args.result_spool_min_free_bytes + scan_config.queue_dir = args.queue_dir + scan_config.postman_cache_dir = args.postman_cache_dir + scan_config.trufflehog_path = args.trufflehog_path + scan_config.trufflehog_config = args.trufflehog_config + security_paths = default_project_paths() + security_paths.update({ + 'runtime_dir': args.runtime_dir, + 'results_dir': args.save_dir, + 'result_spool_dir': args.result_spool_dir, + 'queue_dir': args.queue_dir, + 'work_dir': args.work_dir, + }) + require_sensitive_runtime_paths(security_paths, create=False) + initialize_scanner_runtime(preflight_complete=True) + if args.cleanup_only: + cleanup_pending_command_work_dirs(attempts=3, delay=0.2, log_failures=True) + cleanup_stale_temp_dirs(args.cleanup_temp_age_min) + print('Cleanup complete.') + return + + if skip_startup_cleanup(): + print('Startup temp cleanup skipped (supervised child).') + else: + cleanup_pending_command_work_dirs(attempts=3, delay=0.2, log_failures=True) + cleanup_stale_temp_dirs(args.cleanup_temp_age_min) + + if not args.platform: + raise SystemExit('--platform is required unless --cleanup-only is used') + + if args.platform == 'dockerhub': + args.platform = 'docker' + + if args.platform == 'docker': + docker_token = args.docker_token or args.token or os.getenv('DOCKERHUB_TOKEN') or os.getenv('DOCKER_TOKEN') or os.getenv('DOCKER_TOKENS') + docker_username = args.docker_username or os.getenv('DOCKERHUB_USERNAME') or os.getenv('DOCKER_USERNAME') + configure_docker_tokens(docker_token, docker_username) + + if args.timeout is None: + args.timeout = scan_config.docker_timeout if args.platform == 'docker' else scan_config.git_timeout + + if not check_dependencies(): + raise SystemExit(1) + + db = ScannerDB(args.save_dir) + if getattr(db, 'postgres_required', False) and not db.enabled: + raise RuntimeError('Configured Postgres database is unavailable; refusing file-queue fallback') + if db.enabled and getattr(db.conn, 'is_postgres', False): + db.require_runtime_safety_schema() + run_id = db.start_run( + 'legacy', + sys.argv, + selected_platform=args.platform, + enabled_sources=[args.platform], + global_config=vars(args), + ) if db.enabled else None + if db.enabled and getattr(db.conn, 'is_postgres', False) and run_id is None: + raise RuntimeError('Unable to create Postgres scanner run; refusing target claims') + run_status = 'completed' + run_error = None + cycle = 0 + try: + while True: + cycle += 1 + print(f'\n=== Cycle {cycle} started at {datetime.now().isoformat(timespec="seconds")} ===') + cycle_id = db.start_source_cycle( + run_id, + args.platform, + args.platform, + args.mode, + args.query, + cycle, + args.max_cycles or None, + None, + vars(args), + queue_counts_for_args(args), + ) if db.enabled and run_id else None + cycle_metrics = {} + try: + run_cycle( + args, db, run_id, cycle_id, args.platform, + partial_metrics=cycle_metrics, + ) + except RateLimitError as e: + if db.enabled and cycle_id: + metrics = {'fetched_count': 0} + metrics.update(cycle_metrics) + db.finish_source_cycle(cycle_id, 'rate_limited', metrics, queue_counts_for_args(args), str(e)) + raise + except Exception as e: + if db.enabled and getattr(db, 'conn', None): + try: + db.conn.rollback() + except Exception: + pass + if db.enabled and cycle_id: + metrics = {'fetched_count': 0} + metrics.update(cycle_metrics) + db.finish_source_cycle(cycle_id, 'failed', metrics, queue_counts_for_args(args), str(e)) + raise + + if not args.loop: + break + if args.max_cycles and cycle >= args.max_cycles: + break + + print(f'Sleeping {args.cooldown} seconds...') + time.sleep(args.cooldown) + except KeyboardInterrupt: + run_status = 'stopped' + print('Stopped. Temp cleanup deferred to next run or --cleanup-only.') + except Exception as e: + run_status = 'failed' + run_error = str(e) + raise + finally: + if db.enabled and run_id: + db.finish_run(run_id, run_status, run_error) + db.close() + + +if __name__ == '__main__': + main() diff --git a/app/container_import.py b/app/container_import.py new file mode 100644 index 0000000..9e38a08 --- /dev/null +++ b/app/container_import.py @@ -0,0 +1,1420 @@ +"""One-shot, offline Windows logical import into a freshly provisioned container. + +The caller passes its executing container_runtime module, after container and +default-environment preflight and installation of a flag-only SIGTERM handler. +SIGINT is temporarily made flag-only as well, including the child launch window. +Dispatch action import-snapshot, with required --manifest-sha256 HEX, to +import_snapshot(sys.modules[__name__], args.manifest_sha256) BEFORE initialize(). +Only /import/{manifest.json,files.tar,database.dump} is read. Nothing is resumed, +repaired, removed, or started under a supervisor. A failed import needs review. + +CLI contracts: postgres-runtime initialize-empty/maintenance-start/maintenance-stop +--config PATH; migrate-runtime-safety --config PATH --apply --sources-stopped +(never --initialize-base). If needed, the separate, fenced recovery action adds +--recover-stale-result-pipeline --max-rows 10000 --max-seconds 3600. +READY CLI handoffs are intentional: each CLI takes its own ClusterAuthorityLock. +initialize.lock remains held across those handoffs. Lifecycle children are NEVER +killed on a deadline/cancellation; uncertain stop retains ownership and retries. + +Phases: 1 preflight, 2 archive verification, 3 extraction, 4 config, 5 initdb, +6 maintenance start, 7 raw equality, 8 Postman, 9 recovery, 10 migration, +11 final checks, 12 confirmed stop, 13 publication. Codes: 1 rejected/failed, +2 review required (count only), 124 timeout, 130 cancellation. Helpers' diagnostic +text is suppressed, including migration SQL diagnostics via temporary defaults +on the new Linux role (reset after successful final verification; left suppressed +on a failed, stopped, unmarked import). Evidence is exclusive, +private, fsynced, and never reused: +config/windows-import-manifest.json, windows-import-raw.json, and +windows-import-report.json. initialized.json is the LAST publication, contains +the Linux system identifier, runtime FORMAT, and manifest_sha256. + +Safe diagnostics precede cleanup: import-snapshot-diagnostic followed by numeric +phase, stage, code, review_count, type_id, importer line (0 if unavailable), attempt. +Stages: 1 operation failure, 2 backend construction, 3 stop call, 4 stop result, +5 final probe call, 6 final probe state, 7 backend close, 8 stop CLI, 9 child wait. +Stages 5/6 mean the stop result already confirmed completed AND stopped. +Type IDs: 0 other, 1 Failure, 2 KeyboardInterrupt, 3 OSError, 4 ValueError, +5 TypeError, 6 RuntimeError (including subclasses, never their names or text). +After mutations, the first operation failure and first HOLD are also attempted +once as private, exclusive, fsynced config/windows-import-failure.json and +windows-import-hold.json. They are not completion proofs and are never replaced. +Later HOLD stages remain numeric events; failed diagnostic I/O cannot release +ownership or replace the original failure. Ordinary progress lines are unchanged. + +database.database_bytes supplies the actual physical source size. Older v1 +snapshots without it use max(24 GiB, dump * 4), rejecting estimates over 1 TiB, +plus all file bytes and a 20 GiB free reserve. +This does not measure physical free space on the Windows host's S: drive. + +Integration still needs the image's native PG16/psycopg/PyYAML and real mount +tests. Use /data/config/windows-import.yaml for subsequent container commands; +this module neither changes the default profile nor starts the supervisor. +database.sequence_states is the exporter's optional {schema: {sequence: +{last_value: int, is_called: bool}}} evidence. Partitioned/foreign/non-public +tables are refused because v1 cannot prove their complete row inventory. +Ordinary initialize-empty failure relies on that CLI's confirmed temporary-child +cleanup; abnormal child exit instead enters maintenance-stop's indefinite HOLD. +Do not force-kill a retained importer to bypass unconfirmed stop. +""" + +import contextlib +import hashlib +import json +import logging +import ntpath +import os +from pathlib import Path +import re +import shutil +import signal +import stat +import subprocess +import sys +import tarfile +import time +import unicodedata +from urllib.parse import unquote, urlsplit + + +IMPORT = Path('/import') +BLOCK = 1024 * 1024 +GIB = 1024 ** 3 +MAX_MANIFEST = 32 * BLOCK +MAX_CONFIG = 4 * BLOCK +MAX_ESTIMATE = 1024 * GIB +COUNT_TIMEOUT = 3600 +RESTORE_TIMEOUT = 12 * 3600 +MIGRATE_TIMEOUT = 6 * 3600 +LIFECYCLE_TIMEOUT = 600 +SNAPSHOT_FORMAT = 'truf-windows-snapshot-v1' +ACTIVE = {'results', 'queues', 'state', 'keychecks', 'postman_cache', 'result_spool'} +REQUIRED_FILES = { + 'windows-archive/app/config.yaml', 'config/secrets.yaml', + 'config/trufflehog-custom-detectors.yaml', 'runtime-linux/proxy.txt', +} +PLACEHOLDERS = {'runtime-linux/proxy.txt': b'', 'config/secrets.yaml': b'{}\n'} +FENCES = ( + 'lease_owner', 'lease_token', 'claim_batch', 'leased_at', 'lease_expires_at', + 'current_result_reservation_id', 'claim_event_id', 'resolver_token', +) +QUIET_PG_SETTINGS = ( + ('log_min_error_statement', 'panic'), ('log_min_messages', 'panic'), + ('log_statement', 'none'), ('log_min_duration_statement', '-1'), +) +# Fixed evidence only, not a general backup/repair facility. Recovery may append +# history and advance its own leases, but cannot rewrite these existing facts. +PRESERVED = { + 'pipeline_quarantine': ('id', None), + 'projection_streams': ('stream_name', None), + 'projection_cursors': ('stream_name', None), + 'projection_appends': ('id', None), + 'projection_append_audit': ('id', None), + 'projection_rotations': ('id', None), + 'target_scans': ('id', ( + 'id', 'scan_event_id', 'scan_event_hash', 'queue_id', 'claim_lease_token', + 'target', 'normalized_target', 'result_reservation_id', + )), + 'result_reservations': ('id', ( + 'id', 'reservation_token', 'bundle_id', 'scan_event_id', 'queue_id', + 'ready_relative_path', 'producer_instance_id', 'producer_pid', + 'producer_creation_time', 'producer_executable', + )), + 'result_bundles': ('reservation_id', ( + 'reservation_id', 'bundle_id', 'scan_event_id', 'scan_event_hash', + 'relative_path', 'actual_bytes', + )), + 'worker_progress_events': ('id', None), + 'worker_diagnostics': ('id', None), + 'runtime_operations': ('operation_id', None), + 'runtime_operations_control': ('id', None), + 'runtime_audit_events': ('id', None), +} + + +class Failure(Exception): + def __init__(self, code=1, count=0, *, uncertain=False): + self.code, self.count, self.uncertain = code, count, uncertain + super().__init__(code) + + +def _diagnostic(progress, stage, exc, attempt=0): + # Only local categories and importer line numbers cross the quiet boundary. + try: + code, count = (exc.code, exc.count) if isinstance(exc, Failure) else ( + 130 if isinstance(exc, KeyboardInterrupt) else 1, 0) + if type(code) is not int or code not in (1, 2, 124, 130): + code = 1 + if type(count) is not int or not 0 <= count <= 2 ** 63 - 1: + count = 0 + type_id = next((index for index, kind in enumerate( + (Failure, KeyboardInterrupt, OSError, ValueError, TypeError, RuntimeError), 1) + if isinstance(exc, kind)), 0) + line, traceback = 0, exc.__traceback__ + while traceback is not None: + if traceback.tb_frame.f_code.co_filename == __file__: + line = traceback.tb_lineno + traceback = traceback.tb_next + progress(12, attempt, 0, cleanup=True, diagnostic=(stage, code, count, type_id, line)) + except BaseException: + pass + + +def _checkpoint(runtime): + if runtime._shutdown_requested: + raise Failure(130) + + +@contextlib.contextmanager +def _quiet(): + previous = logging.root.manager.disable + with open(os.devnull, 'w', encoding='ascii') as sink: + try: + logging.disable(sys.maxsize) + with contextlib.redirect_stdout(sink), contextlib.redirect_stderr(sink): + yield + finally: + logging.disable(previous) + + +def _integer(value, minimum=0, maximum=2 ** 63 - 1): + if type(value) is not int or not minimum <= value <= maximum: + raise Failure() + return value + + +def _sha(value): + if not isinstance(value, str) or not re.fullmatch(r'[0-9a-f]{64}', value): + raise Failure() + return value + + +def _object(pairs): + result = {} + for key, value in pairs: + if key in result: + raise Failure() + result[key] = value + return result + + +def _json(payload): + def invalid(_value): + raise Failure() + + return json.loads(payload, object_pairs_hook=_object, parse_constant=invalid) + + +def _relative(name): + if (not isinstance(name, str) or not name or len(name.encode('utf-8')) > 4096 + or unicodedata.normalize('NFC', name) != name + or '\\' in name or ':' in name or any(ord(c) < 32 or ord(c) == 127 for c in name)): + raise Failure() + parts = name.split('/') + if (len(parts) > 64 or any(not p or p in ('.', '..') or p.endswith((' ', '.')) + or len(p.encode('utf-8')) > 255 for p in parts)): + raise Failure() + return parts + + +def _destination(name): + parts = _relative(name) + archived = parts[0] == 'windows-archive' and len(parts) > 1 + allowed = (archived or name in REQUIRED_FILES or + len(parts) > 2 and parts[0] == 'runtime-linux' and parts[1] in ACTIVE or + len(parts) > 2 and parts[0] == 'scanner-result-bundles' + and parts[1] in {'tmp', 'ready', 'quarantine'}) + if not allowed: + raise Failure() + folded = [p.casefold() for p in parts] + if any('.lock' in p or p.endswith('.pid') for p in folded): + raise Failure() + if not archived and any( + p in {'control', 'postgres', 'postgres-linux', 'postgres-data', 'pg_wal', + 'pg_xlog', 'pg_version', 'postmaster.opts', 'cluster_identity.json', + 'supervisor.instance.json', 'janitor.cursor.json'} + or p.startswith('scan_limiter') for p in folded + ): + raise Failure() + return parts + + +def _manifest(payload, expected): + if (not isinstance(expected, str) or not re.fullmatch(r'[0-9A-Fa-f]{64}', expected) + or not 0 < len(payload) <= MAX_MANIFEST + or hashlib.sha256(payload).hexdigest() != expected.lower()): + raise Failure() + value = _json(payload) + if not isinstance(value, dict) or value.get('format') != SNAPSHOT_FORMAT: + raise Failure() + source, database, archive = value['source'], value['database'], value['archive'] + if (source.get('supervisor_stopped') is not True or source.get('postgres_stopped') is not True + or ntpath.normcase(source.get('root', '')) != ntpath.normcase(r'D:\truf') + or ntpath.normcase(source.get('postgres_data_dir', '')) != ntpath.normcase(r'S:\postgres-data') + or ntpath.normcase(database.get('data_directory', '')) != ntpath.normcase(r'S:\postgres-data') + or _integer(database['version_num']) // 10000 != 16 + or not re.fullmatch(r'[1-9][0-9]{0,19}', database['system_identifier'])): + raise Failure() + _integer(database['port'], 1, 65535) + for name in (database['database_name'], database['user_name']): + _identifier(name) + for item in (archive, database): + _sha(item['sha256']) + _integer(item['bytes'], 6, 8 * MAX_ESTIMATE) + tables = database['table_counts'] + if not isinstance(tables, dict) or not tables: + raise Failure() + for name, count in tables.items(): + _identifier(name) + _integer(count) + sequences = database.get('sequence_states') + sequence_count = 0 + if 'sequence_states' in database: + if not isinstance(sequences, dict): + raise Failure() + for schema, states in sequences.items(): + _identifier(schema) + if schema == 'information_schema' or schema.startswith('pg_') or not isinstance(states, dict): + raise Failure() + for name, state in states.items(): + _identifier(name) + if (not isinstance(state, dict) or set(state) != {'last_value', 'is_called'} + or type(state['is_called']) is not bool): + raise Failure() + _integer(state['last_value'], -(2 ** 63)) + sequence_count += 1 + if 'sequence_count' in database and _integer(database['sequence_count']) != sequence_count: + raise Failure() + if 'database_bytes' in database: + _integer(database['database_bytes'], 1, MAX_ESTIMATE) + files, names, parents = {}, {}, {} + if not isinstance(value['files'], list): + raise Failure() + for item in value['files']: + if not isinstance(item, dict) or set(item) != {'path', 'size', 'sha256'}: + raise Failure() + name = item['path'] + parts = _destination(name) + folded = name.casefold() + if folded in names or folded in parents: + raise Failure() + for index in range(1, len(parts)): + parent = '/'.join(parts[:index]) + key = parent.casefold() + if key in names or key in parents and parents[key] != parent: + raise Failure() + parents[key] = parent + names[folded] = name + files[name] = {'size': _integer(item['size'], maximum=8 * MAX_ESTIMATE), + 'sha256': _sha(item['sha256'])} + if not REQUIRED_FILES.issubset(files): + raise Failure() + file_bytes = sum(item['size'] for item in files.values()) + if file_bytes > archive['bytes'] or files['windows-archive/app/config.yaml']['size'] > MAX_CONFIG: + raise Failure() + if 'counts' in value: + counts = {'files': len(files), 'file_bytes': file_bytes, + 'active_files': sum(not n.startswith('windows-archive/') for n in files), + 'archival_files': sum(n.startswith('windows-archive/') for n in files), + 'public_tables': len(tables)} + if sequences is not None: + counts['sequences'] = sequence_count + if any(_integer(value['counts'][k]) != v for k, v in counts.items()): + raise Failure() + return value, files + + +def _fingerprint(info): + return (info.st_dev, info.st_ino, info.st_mode, info.st_nlink, info.st_size, + info.st_mtime_ns, info.st_ctime_ns) + + +def _chain(path): + for parent in (*reversed(path.parents), path): + info = parent.lstat() + if stat.S_ISLNK(info.st_mode) or getattr(info, 'st_file_attributes', 0) & 0x400: + raise Failure() + + +def _regular(path, runtime=None): + _chain(path) + if runtime is not None: + runtime.private_path(path) + info = path.lstat() + if not stat.S_ISREG(info.st_mode) or info.st_nlink != 1: + raise Failure() + return _fingerprint(info) + + +@contextlib.contextmanager +def _input(path, runtime=None): + before = _regular(path, runtime) + descriptor = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK | getattr(os, 'O_BINARY', 0)) + with os.fdopen(descriptor, 'rb', buffering=0) as handle: + _unchanged(path, handle, before, runtime) + yield handle, before + _unchanged(path, handle, before, runtime) + + +def _unchanged(path, handle, before, runtime=None): + if _fingerprint(os.fstat(handle.fileno())) != before or _regular(path, runtime) != before: + raise Failure() + + +def _mounts(runtime): + _chain(IMPORT) + if not stat.S_ISDIR(IMPORT.lstat().st_mode): + raise Failure() + with open('/proc/self/mountinfo', 'rb') as handle: + payload = handle.read(BLOCK + 1) + if len(payload) > BLOCK: + raise Failure() + mounts = {} + for line in payload.splitlines(): + fields = line.split() + if len(fields) < 10 or b'-' not in fields: + raise Failure() + # Mountinfo escapes cannot disguise an exact fixed mount or a submount. + target = re.sub(rb'\\([0-7]{3})', lambda m: bytes([int(m[1], 8)]), fields[4]) + if target in mounts: + raise Failure() + mounts[target] = fields + if target.startswith((b'/import/', b'/data/')): + raise Failure() + incoming, destination = mounts.get(b'/import'), mounts.get(b'/data') + if (incoming is None or destination is None or b'ro' not in incoming[5].split(b',') + or incoming[0] == destination[0] or incoming[2:4] == destination[2:4] + or IMPORT.stat().st_dev == runtime.DATA.stat().st_dev + or not os.statvfs(IMPORT).f_flag & os.ST_RDONLY): + raise Failure() + if set(os.listdir(IMPORT)) != {'manifest.json', 'files.tar', 'database.dump'}: + raise Failure() + + +def _placeholder(runtime, name, before=None): + path = runtime.DATA / name + with _input(path, runtime) as (handle, fingerprint): + expected = PLACEHOLDERS[name] + if handle.read(len(expected) + 1) != expected or before is not None and fingerprint != before: + raise Failure() + return fingerprint + + +def _fresh(runtime, files=()): + allowed_dirs = set(runtime.DIRECTORIES) + allowed_files = {'.provisioned.json', '.provision.lock', 'initialize.lock', + 'postgres-password', *PLACEHOLDERS} + seen_dirs, seen_files = set(), set() + + def visit(path): + runtime.private_path(path, directory=True) + if path.stat().st_dev != runtime.DATA.stat().st_dev: + raise Failure() + with os.scandir(path) as entries: + for entry in entries: + child = path / entry.name + name = child.relative_to(runtime.DATA).as_posix() + if entry.is_dir(follow_symlinks=False): + if name not in allowed_dirs: + raise Failure() + seen_dirs.add(name) + visit(child) + else: + if name not in allowed_files: + raise Failure() + _regular(child, runtime) + seen_files.add(name) + + visit(runtime.DATA) + if seen_dirs != allowed_dirs or seen_files != allowed_files: + raise Failure() + existing = {name.casefold(): name for name in seen_dirs | seen_files} + for name in files: + if name.casefold() in existing and name not in PLACEHOLDERS: + raise Failure() + parts = name.split('/') + for index in range(1, len(parts)): + parent = '/'.join(parts[:index]) + found = existing.get(parent.casefold()) + if found is not None and (found != parent or found not in seen_dirs): + raise Failure() + marker = runtime._read_json(runtime.PROVISIONED) + if marker != {'format': runtime.FORMAT, 'uid': 10001, 'gid': 10001}: + raise Failure() + for name in ('.provision.lock', 'initialize.lock'): + if (runtime.DATA / name).stat().st_size != 0: + raise Failure() + return {name: _placeholder(runtime, name) for name in PLACEHOLDERS} + + +def _space(runtime, manifest): + database = manifest['database'] + estimate = database.get('database_bytes', max(24 * GIB, database['bytes'] * 4)) + _integer(estimate, 1, MAX_ESTIMATE) + needed = sum(item['size'] for item in manifest['files']) + estimate + 20 * GIB + free = shutil.disk_usage(runtime.DATA).free + if free < needed: + raise Failure() + return {'database_estimate_bytes': estimate, 'required_free_bytes': needed, + 'available_free_bytes': free, 'database_estimate_from_metadata': 'database_bytes' in database} + + +def _fsync_dir(path): + descriptor = os.open(path, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + + +@contextlib.contextmanager +def _new_file(runtime, path): + runtime.private_path(path.parent, directory=True) + descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW + | getattr(os, 'O_BINARY', 0), 0o600) + with os.fdopen(descriptor, 'wb') as handle: + _regular(path, runtime) + if not os.path.samestat(os.fstat(handle.fileno()), path.lstat()): + raise Failure() + yield handle + handle.flush() + os.fsync(handle.fileno()) + _regular(path, runtime) + _fsync_dir(path.parent) + + +def _write(runtime, path, payload): + with _new_file(runtime, path) as handle: + if handle.write(payload) != len(payload): + raise Failure() + return hashlib.sha256(payload).hexdigest() + + +def _encoded(value): + return (json.dumps(value, ensure_ascii=True, sort_keys=True, separators=(',', ':')) + '\n').encode('ascii') + + +class _TarInfo(tarfile.TarInfo): + def _proc_pax(self, archive): + if self.type != tarfile.XHDTYPE or not 0 < self.size <= 65536: + raise Failure() + return super()._proc_pax(archive) + + def _proc_gnulong(self, archive): + raise Failure() + + def _proc_sparse(self, archive): + raise Failure() + + def _proc_gnusparse_00(self, *args): + raise Failure() + + def _proc_gnusparse_01(self, *args): + raise Failure() + + def _proc_gnusparse_10(self, *args): + raise Failure() + + +class _HashReader: + def __init__(self, handle, size): + self.handle, self.limit = handle, size + self.digest, self.size = hashlib.sha256(), 0 + + def read(self, size): + block = self.handle.read(min(size, BLOCK)) + self.size += len(block) + if self.size > self.limit: + raise Failure() + self.digest.update(block) + return block + + +def _archive(runtime, handle, metadata, files, progress, placeholders=None): + handle.seek(0) + reader = _HashReader(handle, metadata['bytes']) + seen, size, copied = set(), 0, {} + with tarfile.open(fileobj=reader, mode='r|', tarinfo=_TarInfo) as archive: + for member in archive: + _checkpoint(runtime) + name = member.name + _destination(name) + if (not member.isreg() or member.type not in (tarfile.REGTYPE, tarfile.AREGTYPE) + or member.linkname or member.sparse is not None + or set(member.pax_headers) - {'path', 'size'} + or name not in files or name in seen or member.size != files[name]['size']): + raise Failure() + seen.add(name) + destination = runtime.DATA / name + replacement = placeholders is not None and name in placeholders + output = destination.with_name(destination.name + '.windows-import-partial') if replacement else destination + if placeholders is not None: + for parent in reversed(destination.parents): + if parent == runtime.DATA or runtime.DATA in parent.parents: + if not os.path.lexists(parent): + runtime.private_path(parent.parent, directory=True) + parent.mkdir(mode=0o700) + _fsync_dir(parent.parent) + runtime.private_path(parent, directory=True) + context = _new_file(runtime, output) if placeholders is not None else contextlib.nullcontext(None) + with archive.extractfile(member) as source, context as target: + digest, received = hashlib.sha256(), 0 + while True: + _checkpoint(runtime) + block = source.read(BLOCK) + if not block: + break + received += len(block) + digest.update(block) + if target is not None and target.write(block) != len(block): + raise Failure() + if received != member.size or digest.hexdigest() != files[name]['sha256']: + raise Failure() + if replacement: + _placeholder(runtime, name, placeholders[name]) + os.replace(output, destination) + _regular(destination, runtime) + _fsync_dir(destination.parent) + if placeholders is not None: + copied[name] = _regular(destination, runtime) + size += received + if len(seen) % 128 == 0: + progress(3 if placeholders is not None else 2, len(seen), size) + trailer = archive.offset + while reader.read(BLOCK): + _checkpoint(runtime) + if (seen != set(files) or reader.size != metadata['bytes'] + or reader.digest.hexdigest() != metadata['sha256'] + or not 1024 <= reader.size - trailer <= 2 * tarfile.RECORDSIZE): + raise Failure() + # tarfile stops at its first zero header. Do not accept a second hidden tar + # or nonzero data in its buffered trailer, even if the whole hash was signed. + handle.seek(trailer) + if any(handle.read(2 * tarfile.RECORDSIZE + 1)): + raise Failure() + progress(3 if placeholders is not None else 2, len(seen), size) + return copied + + +def _hash_input(runtime, path, handle, before, metadata, private=False, dump=False): + _unchanged(path, handle, before, runtime if private else None) + handle.seek(0) + digest, size, prefix = hashlib.sha256(), 0, b'' + while True: + _checkpoint(runtime) + block = handle.read(BLOCK) + if not block: + break + if not prefix: + prefix = block[:5] + size += len(block) + if size > metadata.get('bytes', metadata.get('size')): + raise Failure() + digest.update(block) + if (size != metadata.get('bytes', metadata.get('size')) or digest.hexdigest() != metadata['sha256'] + or dump and prefix != b'PGDMP'): + raise Failure() + _unchanged(path, handle, before, runtime if private else None) + handle.seek(0) + + +def _configuration(runtime): + import yaml + from container_import_config import translate_windows_config + + values = [] + for path in (runtime.DATA / 'windows-archive/app/config.yaml', runtime.DEFAULT_CONFIG): + with _input(path, runtime) as (handle, _before): + payload = handle.read(MAX_CONFIG + 1) + if len(payload) > MAX_CONFIG: + raise Failure() + values.append(yaml.safe_load(payload)) + translated, adjusted = translate_windows_config(*values) + global_config = translated['global'] + if global_config.get('database_url') or global_config.get('dashboard_db_url'): + raise Failure() + path = runtime.DATA / 'config/windows-import.yaml' + config_hash = _write(runtime, path, yaml.safe_dump(translated, allow_unicode=False).encode('utf-8')) + config = runtime.prepare_environment(path) + expected = {key: str(runtime.DATA / 'runtime-linux' / folder) for key, folder in ( + ('results_dir', 'results'), ('queue_dir', 'queues'), ('state_dir', 'state'), + ('log_dir', 'logs'), ('keycheck_dir', 'keychecks'), ('postman_cache_dir', 'postman_cache'), + ('result_spool_dir', 'result_spool'), ('legacy_result_spool_dir', 'result_spool'), + ('scan_limiter_db', 'state/scan_limiter.db'), + )} + expected.update(proxy_file=str(runtime.DATA / 'runtime-linux/proxy.txt'), + trufflehog_config=str(runtime.DATA / 'config/trufflehog-custom-detectors.yaml')) + if any(config['global'].get(key) != value for key, value in expected.items()): + raise Failure() + return path, config, adjusted, config_hash + + +def _credentials(runtime): + with _input(runtime.PASSWORD, runtime) as (handle, _before): + password = handle.read(130).decode('ascii').rstrip('\n') + if not re.fullmatch(r'[A-Za-z0-9_-]{32,128}', password): + raise Failure() + dsn = os.environ['SCANNER_DB_URL'] + parsed = urlsplit(dsn) + if (parsed.scheme != 'postgresql' or parsed.hostname != '127.0.0.1' or parsed.port != 5432 + or unquote(parsed.username or '') != 'truf' or unquote(parsed.path) != '/truf' + or unquote(parsed.password or '') != password or parsed.query or parsed.fragment + or any(os.environ.get(k) != dsn for k in ('DATABASE_URL', 'TRUF_MANAGED_POSTGRES_DSN'))): + raise Failure() + return dsn, password + + +def _command(runtime, command, timeout, progress, *, env=None, stdin=None, lifecycle=False, stopping=False): + if not stopping: + _checkpoint(runtime) + process = subprocess.Popen(command, stdin=subprocess.DEVNULL if stdin is None else stdin, + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + env=env, close_fds=True, start_new_session=True) + exited, failure, held = False, None, 0 + try: + deadline = time.monotonic() + timeout + while True: + try: + result = process.wait(timeout=1) + exited = True + break + except subprocess.TimeoutExpired: + pass + except BaseException: + failure = failure or Failure(130) + if not stopping and runtime._shutdown_requested: + failure = failure or Failure(130) + if time.monotonic() >= deadline: + failure = failure or Failure(124) + if failure is not None: + if lifecycle: + # A CLI can be retaining a not-yet-bookkept postmaster. Its + # own positive stop contract, not the parent, owns exit. + held += 1 + if held == 1 or held % 60 == 0: + if held == 1 and not stopping: + _diagnostic(progress, 1, failure) + _diagnostic(progress, 9, failure, held) + else: + try: + process.kill() + except OSError: + pass + except BaseException as exc: + failure = failure or Failure(130 if isinstance(exc, KeyboardInterrupt) else 1) + _diagnostic(progress, 9 if stopping else 1, exc) + finally: + # Even a broken progress pipe or an interrupt outside wait() cannot + # abandon a lifecycle child, nor a client with an open restore session. + while not exited: + if not lifecycle: + try: + process.kill() + except BaseException: + pass + try: + result = process.wait(timeout=1) + exited = True + except BaseException: + pass + if lifecycle and result < 0: + raise Failure(failure.code if failure is not None else 1, uncertain=True) + if failure is not None and not stopping: + raise failure + if result != 0: + raise Failure() + if not stopping: + _checkpoint(runtime) + + +def _identifier(name): + if not isinstance(name, str) or not name or '\x00' in name or len(name.encode('utf-8')) > 63: + raise Failure() + return '"' + name.replace('"', '""') + '"' + + +@contextlib.contextmanager +def _connect(runtime, *, readonly=True): + import psycopg + from psycopg.rows import dict_row + + dsn, _password = _credentials(runtime) + options = ( + f'-c search_path=public -c statement_timeout={COUNT_TIMEOUT * 1000} ' + '-c lock_timeout=10000 -c idle_in_transaction_session_timeout=3600000 ' + '-c row_security=off -c log_min_error_statement=panic -c log_statement=none ' + '-c log_min_messages=panic -c log_min_duration_statement=-1 ' + '-c default_transaction_read_only=' + ('on' if readonly else 'off') + ) + connection = psycopg.connect(dsn, connect_timeout=10, row_factory=dict_row, + options=options, application_name='truf-container-import', + tcp_user_timeout=60000) + try: + yield connection + finally: + connection.close() + + +def _identity(runtime, identity, source): + if (identity['pg_major'] != 16 or identity['data_directory'] != str(runtime.DATA / 'postgres-linux') + or identity['database'] != 'truf' or identity['user'] != 'truf' or identity['port'] != 5432 + or not re.fullmatch(r'[1-9][0-9]{0,19}', identity['system_identifier']) + or identity['system_identifier'] == source['system_identifier']): + raise Failure() + + +def _online(connection, identity): + row = connection.execute("""SELECT current_database() AS database, current_user AS user_name, + current_setting('data_directory') AS data_directory, current_setting('port')::int AS port, + current_setting('server_version_num')::int AS version_num, pg_is_in_recovery() AS in_recovery, + (SELECT system_identifier::text FROM pg_catalog.pg_control_system()) AS system_identifier, + (SELECT rolsuper FROM pg_catalog.pg_roles WHERE rolname = current_user) AS superuser, + (SELECT count(*) FROM pg_catalog.pg_stat_activity + WHERE backend_type = 'client backend' AND pid <> pg_backend_pid()) AS other_clients""").fetchone() + expected = {'database': identity['database'], 'user_name': identity['user'], + 'data_directory': identity['data_directory'], 'port': identity['port'], + 'system_identifier': identity['system_identifier'], 'in_recovery': False, + 'superuser': True, 'other_clients': 0} + if (not row or any(row.get(k) != v for k, v in expected.items()) + or _integer(row['version_num']) // 10000 != 16): + raise Failure() + return row['version_num'] + + +def _database_equality(runtime, connection, database, identity, progress, *, empty=False): + version = _online(connection, identity) + if empty: + row = connection.execute("""SELECT + (SELECT count(*) FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n + ON n.oid = c.relnamespace WHERE n.nspname <> 'information_schema' AND n.nspname !~ '^pg_') + + (SELECT count(*) FROM pg_catalog.pg_proc p JOIN pg_catalog.pg_namespace n + ON n.oid = p.pronamespace WHERE n.nspname <> 'information_schema' AND n.nspname !~ '^pg_') + + (SELECT count(*) FROM pg_catalog.pg_type t JOIN pg_catalog.pg_namespace n + ON n.oid = t.typnamespace WHERE n.nspname <> 'information_schema' AND n.nspname !~ '^pg_') + + (SELECT count(*) FROM pg_catalog.pg_namespace + WHERE nspname NOT IN ('public', 'information_schema') AND nspname !~ '^pg_') AS count""").fetchone() + if row['count'] != 0: + raise Failure() + return {} + tables = connection.execute("""SELECT n.nspname AS schema_name, c.relname AS name, c.relkind AS kind + FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r','p','f') AND n.nspname <> 'information_schema' AND n.nspname !~ '^pg_' + ORDER BY n.nspname, c.relname""").fetchall() + if (any(t['schema_name'] != 'public' or t['kind'] != 'r' for t in tables) + or {t['name'] for t in tables} != set(database['table_counts'])): + raise Failure() + counts = {} + for table in tables: + _checkpoint(runtime) + name = table['name'] + count = connection.execute('SELECT count(*) AS count FROM ONLY "public".' + _identifier(name)).fetchone()['count'] + if _integer(count) != database['table_counts'][name]: + raise Failure() + counts[name] = count + progress(7, len(counts), 0) + sequence_count = 0 + if 'sequence_states' in database: + sequences = connection.execute("""SELECT n.nspname AS schema_name, c.relname AS name + FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind = 'S' AND n.nspname <> 'information_schema' AND n.nspname !~ '^pg_' + ORDER BY n.nspname, c.relname""").fetchall() + expected = {(schema, name): state for schema, states in database['sequence_states'].items() + for name, state in states.items()} + if {(row['schema_name'], row['name']) for row in sequences} != set(expected): + raise Failure() + for (schema, name), state in expected.items(): + _checkpoint(runtime) + row = connection.execute('SELECT last_value, is_called FROM ' + _identifier(schema) + + '.' + _identifier(name)).fetchone() + if dict(row) != state: + raise Failure() + sequence_count += 1 + connection.commit() + _online(connection, identity) + connection.commit() + return {'table_counts': counts, 'tables': len(counts), 'rows': sum(counts.values()), + 'sequences_verified': sequence_count, 'sequences_provided': 'sequence_states' in database, + 'version_num': version} + + +def _postman_target(runtime, target, normalized, files): + from target_identity import postman_target_identity + + if not isinstance(target, str) or len(target.encode('utf-8')) > BLOCK: + raise Failure(2, 1) + value = _json(target) + if not isinstance(value, dict): + raise Failure(2, 1) + digest = _sha(str(value.get('sha256', '')).strip().lower()) + if normalized != 'postman:sha256:' + digest or postman_target_identity(value) != normalized: + raise Failure(2, 1) + changed, offered = False, False + for key in ('cache_path', 'local_path'): + if key not in value: + continue + offered = True + path = value[key] + if not isinstance(path, str) or not path: + raise Failure(2, 1) + windows = path.replace('\\', '/') + prefix = 'd:/truf/runtime/postman_cache/' + linux = str(runtime.DATA / 'runtime-linux/postman_cache').replace('\\', '/') + '/' + if windows.casefold().startswith(prefix): + relative = windows[len(prefix):] + elif path.startswith(linux): + relative = path[len(linux):] + else: + raise Failure(2, 1) + _relative(relative) + name = 'runtime-linux/postman_cache/' + relative + # The manifest rejects casefold collisions. Windows path case may differ + # from the preserved filename; never rename the copied artifact itself. + entry = files.get(name.casefold()) + if entry is None or entry['sha256'] != digest or entry['size'] <= 0: + raise Failure(2, 1) + for size_key in ('size', 'bytes'): + if size_key in value and (type(value[size_key]) not in (int, str) + or int(value[size_key]) != entry['size']): + raise Failure(2, 1) + newpath = runtime.DATA / entry['path'] + with _input(newpath, runtime) as (handle, before): + _hash_input(runtime, newpath, handle, before, entry, private=True) + newvalue = str(newpath) + changed |= newvalue != value[key] + value[key] = newvalue + if not offered or postman_target_identity(value) != normalized: + raise Failure(2, 1) + return json.dumps(value, ensure_ascii=True, sort_keys=True, separators=(',', ':')) if changed else target + + +def _rebase_postman(runtime, connection, files): + from target_identity import postman_target_identity + + cache = {name.casefold(): dict(entry, path=name) for name, entry in files.items() + if name.startswith('runtime-linux/postman_cache/')} + adjusted, reviewed, unsafe, after = 0, 0, 0, 0 + while True: + _checkpoint(runtime) + rows = connection.execute("""SELECT * FROM public.target_queue + WHERE platform IN ('postman','github_gists','github_archive_files') + AND status IN ('pending','deferred','in_progress') AND id > %s + ORDER BY id LIMIT 128 FOR UPDATE""", (after,)).fetchall() + if not rows: + break + for row in rows: + after = _integer(row['id'], 1) + reviewed += 1 + try: + if (row['status'] not in ('pending', 'deferred') + or any(row[name] is not None for name in FENCES) + or row['resolver_state'] not in (None, 'resolved')): + raise Failure(2, 1) + normalized = row['normalized_target'] + if row['platform'] != 'postman': + # ScannerDB normalizes these platforms as the entire lower- + # cased JSON, not the artifact digest. Changing a locator + # would change queue identity, even for the same artifact. + if (not isinstance(row['target'], str) or len(row['target'].encode('utf-8')) > BLOCK + or normalized != row['target'].strip().lower()): + raise Failure(2, 1) + normalized = postman_target_identity(row['target']) + replacement = _postman_target(runtime, row['target'], normalized, cache) + if row['platform'] != 'postman' and replacement != row['target']: + raise Failure(2, 1) + except (Failure, ValueError, TypeError, OSError): + _checkpoint(runtime) + unsafe += 1 + continue + if replacement != row['target']: + # All rows are locked in this transaction. Only the target JSON + # changes: no timestamp, identity, attempt, token, or history edit. + result = connection.execute("""UPDATE public.target_queue SET target = %s + WHERE id = %s AND target = %s AND normalized_target = %s + AND status IN ('pending','deferred') AND current_result_reservation_id IS NULL + AND lease_token IS NULL AND claim_event_id IS NULL""", + (replacement, row['id'], row['target'], row['normalized_target'])) + if result.rowcount != 1: + raise Failure() + adjusted += 1 + if unsafe: + connection.rollback() + raise Failure(2, unsafe) + _checkpoint(runtime) + connection.commit() + return {'postman_reviewed': reviewed, 'postman_adjusted': adjusted} + + +def _preserved(runtime, connection, previous=None): + evidence = {} + for table, (key, selected) in PRESERVED.items(): + _checkpoint(runtime) + if previous is not None and table not in previous: + continue + old = previous.get(table) if previous is not None else None + if old is None: + columns = [row['name'] for row in connection.execute("""SELECT a.attname AS name + FROM pg_catalog.pg_attribute a JOIN pg_catalog.pg_class c ON c.oid = a.attrelid + JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = 'public' AND c.relname = %s AND a.attnum > 0 AND NOT a.attisdropped + ORDER BY a.attname""", (table,)).fetchall()] + if not columns: + continue + if key not in columns: + raise Failure(2, 1) + # Additive migration may introduce a table/column not in this + # snapshot. The preservation proof covers every preexisting fact. + columns = [name for name in selected if name in columns] if selected else columns + cutoff = connection.execute('SELECT max(' + _identifier(key) + ') AS cutoff FROM public.' + + _identifier(table)).fetchone()['cutoff'] + else: + columns, cutoff = old['columns'], old['cutoff'] + digest, count = hashlib.sha256(), 0 + if cutoff is not None: + selection = ','.join(_identifier(column) for column in columns) + # Named cursor keeps millions of historical identifiers off both + # the client's buffered result set and the diagnostic channels. + with connection.cursor(name='windows_import_evidence') as cursor: + cursor.execute('SELECT pg_catalog.encode(pg_catalog.sha256(pg_catalog.convert_to(' + 'pg_catalog.row_to_json(e)::text, \'UTF8\')), \'hex\') AS digest FROM ' + '(SELECT ' + selection + ' FROM public.' + _identifier(table) + + ' WHERE ' + _identifier(key) + ' <= %s ORDER BY ' + _identifier(key) + ') e', (cutoff,)) + for row in cursor: + _checkpoint(runtime) + digest.update(_sha(row['digest']).encode('ascii')) + count += 1 + item = {'columns': columns, 'cutoff': cutoff, 'count': count, 'sha256': digest.hexdigest()} + if old is not None and item != old: + raise Failure(2, 1) + evidence[table] = item + connection.commit() + return evidence + + +def _pipeline_counts(connection): + queries = { + 'worker_leases': "SELECT count(*) AS count FROM public.pipeline_leases WHERE state NOT IN ('released','failed')", + 'result_reservations': "SELECT count(*) AS count FROM public.result_reservations WHERE state IN ('scanning','ready','ingesting','db_committed')", + 'queue_leases': "SELECT count(*) AS count FROM public.target_queue q LEFT JOIN public.result_reservations r ON r.id=q.current_result_reservation_id WHERE q.status='in_progress' OR r.state IN ('scanning','ready','ingesting','db_committed')", + 'blob_leases': "SELECT count(*) AS count FROM public.docker_content_blobs WHERE state IN ('leased','submitted') OR lease_reservation_id IS NOT NULL", + } + counts = {name: _integer(connection.execute(sql).fetchone()['count']) for name, sql in queries.items()} + connection.commit() + return counts + + +def _projection_files(runtime, connection, files): + rows = connection.execute("""SELECT s.stream_name, c.stream_name AS cursor_stream_name, + s.base_relative_path, s.current_generation, c.generation, c.committed_offset + FROM public.projection_streams s + FULL JOIN public.projection_cursors c ON c.stream_name = s.stream_name""").fetchall() + scan_streams = {'scan_results': 'scan_results.jsonl', 'found_secrets': 'found_secrets.jsonl', + 'scan_errors': 'scan_errors.log'} + stream_names = {row.get('stream_name') for row in rows} + if not scan_streams.keys() <= stream_names or len(stream_names) != len(rows): + raise Failure(2, len(rows)) + for row in rows: + stream_name = row.get('stream_name') + if (not isinstance(stream_name, str) or row.get('cursor_stream_name') != stream_name + or _integer(row.get('current_generation')) != _integer(row.get('generation'))): + raise Failure(2, 1) + offset = _integer(row.get('committed_offset')) + relative = row.get('base_relative_path') + _relative(relative) + root = 'runtime-linux/results/' + expected = scan_streams.get(stream_name) + status = False + if expected is None: + match = re.fullmatch(r'keycheck:([a-z0-9][a-z0-9_.-]{0,63}):(results|status)', stream_name) + if not match: + raise Failure(2, 1) + status = match[2] == 'status' + suffix = 'Checked.txt' if status else 'Results.jsonl' + expected = f'{match[1]}/{match[1]}{suffix}' + root = 'runtime-linux/keychecks/' + if relative != expected: + raise Failure(2, 1) + name = root + relative + path = runtime.DATA / name + metadata = files.get(name) + if status: + # Replacement status snapshots do not advance historical append cursors. + if metadata is None: + raise Failure(2, 1) + with _input(path, runtime) as (handle, before): + _hash_input(runtime, path, handle, before, metadata, private=True) + else: + if offset != (metadata['size'] if metadata is not None else 0): + raise Failure(2, 1) + if metadata is not None: + if _regular(path, runtime)[4] != metadata['size']: + raise Failure() + elif os.path.lexists(path): + raise Failure(2, 1) + connection.commit() + + +def _backend_stop(pg, config, backend, progress): + retries = 0 + while True: + stage = 3 + try: + result = pg.maintenance_stop(config, backend=backend) + stage = 4 + if result.completed is not True or result.stopped is not True: + raise Failure() + stage = 5 + probe = backend.probe() + stage = 6 + if probe.kind != pg.ProbeKind.STOPPED: + raise Failure() + return + except BaseException as exc: + retries += 1 + _diagnostic(progress, stage, exc, retries) + try: + time.sleep(2) + except BaseException: + pass + + +@contextlib.contextmanager +def _authority(runtime, config, source, progress, *, stopped=False): + import postgres_runtime as pg + from runtime_security import ClusterAuthorityLock + + dsn, _password = _credentials(runtime) + lock = ClusterAuthorityLock(config, endpoint_dsn=dsn) + lock.acquire() + backend = None + try: + backend = pg.PostgresBackend(config) + if stopped: + _backend_stop(pg, config, backend, progress) + elif backend.probe().kind != pg.ProbeKind.READY: + raise Failure() + identity = pg.verify_cluster_identity(config) + _identity(runtime, identity, source) + yield identity + if not stopped and backend.probe().kind != pg.ProbeKind.READY: + raise Failure() + except BaseException as exc: + _diagnostic(progress, 1, exc) + retries = 0 + while backend is None: + try: + backend = pg.PostgresBackend(config) + except BaseException as stop_exc: + retries += 1 + _diagnostic(progress, 2, stop_exc, retries) + try: + time.sleep(2) + except BaseException: + pass + _backend_stop(pg, config, backend, progress) + raise + finally: + if backend is not None: + retries = 0 + while True: + try: + backend.close() + break + except BaseException as exc: + retries += 1 + _diagnostic(progress, 7, exc, retries) + _backend_stop(pg, config, backend, progress) + lock.release() + + +def _stop_cli(runtime, path, progress): + retries = 0 + while True: + try: + _command(runtime, runtime._bootstrap_command('postgres-runtime', 'maintenance-stop', '--config', str(path)), + LIFECYCLE_TIMEOUT, progress, lifecycle=True, stopping=True) + return + except BaseException as exc: + retries += 1 + _diagnostic(progress, 8, exc, retries) + try: + time.sleep(2) + except BaseException: + pass + + +def _restore(runtime, path, config, manifest, files, dump, fingerprint, progress, report): + try: + _command(runtime, runtime._bootstrap_command('postgres-runtime', 'initialize-empty', '--config', str(path)), + LIFECYCLE_TIMEOUT, progress, lifecycle=True) + except Failure as exc: + _diagnostic(progress, 1, exc) + if exc.uncertain: + # A killed init CLI did not fulfill its ownership contract. Without + # an identity the stop CLI must HOLD, not adopt or release this data. + _stop_cli(runtime, path, progress) + report['maintenance_stopped'] = True + raise + # initialize-empty itself retains its temporary postmaster until stopped, + # including failure. A failed unbound initdb is not adopted by maintenance. + started = False + try: + _checkpoint(runtime) + progress(6) + started = True + _command(runtime, runtime._bootstrap_command('postgres-runtime', 'maintenance-start', '--config', str(path)), + LIFECYCLE_TIMEOUT, progress, lifecycle=True) + with _authority(runtime, config, manifest['database'], progress) as identity: + report['linux_system_identifier'] = identity['system_identifier'] + progress(7) + with _connect(runtime) as connection: + _database_equality(runtime, connection, manifest['database'], identity, progress, empty=True) + _hash_input(runtime, IMPORT / 'database.dump', dump, fingerprint, manifest['database'], dump=True) + _dsn, password = _credentials(runtime) + from runtime_security import require_trusted_native_executable + + restore_binary = require_trusted_native_executable('/usr/lib/postgresql/16/bin/pg_restore') + environment = {k: v for k, v in os.environ.items() if not k.upper().startswith('PG')} + environment.update( + PGHOST='127.0.0.1', PGHOSTADDR='127.0.0.1', PGPORT='5432', PGUSER='truf', + PGDATABASE='truf', PGPASSWORD=password, PGPASSFILE=os.devnull, PGSERVICEFILE=os.devnull, + PGSSLMODE='disable', PGGSSENCMODE='disable', PGCONNECT_TIMEOUT='10', + PGAPPNAME='truf-container-import-restore', PGCLIENTENCODING='UTF8', LC_ALL='C', LANG='C', + PGOPTIONS=f'-c statement_timeout={RESTORE_TIMEOUT * 1000} -c lock_timeout=60000 ' + '-c idle_in_transaction_session_timeout=3600000 -c row_security=off ' + '-c log_min_error_statement=panic -c log_min_messages=panic ' + '-c log_statement=none -c log_min_duration_statement=-1', + ) + _command(runtime, [restore_binary, '--dbname=truf', '--no-password', + '--single-transaction', '--exit-on-error', '--no-owner', '--no-acl', '--no-tablespaces'], + RESTORE_TIMEOUT, progress, env=environment, stdin=dump) + _hash_input(runtime, IMPORT / 'database.dump', dump, fingerprint, manifest['database'], dump=True) + with _connect(runtime) as connection: + raw = _database_equality(runtime, connection, manifest['database'], identity, progress) + report['raw'] = raw + preserved = _preserved(runtime, connection) + _projection_files(runtime, connection, files) + report['pipeline_before'] = _pipeline_counts(connection) + report['raw_evidence_sha256'] = _write(runtime, runtime.DATA / 'config/windows-import-raw.json', + _encoded({'manifest_sha256': report['manifest_sha256'], **raw})) + progress(8) + with _connect(runtime, readonly=False) as connection: + report.update(_rebase_postman(runtime, connection, files)) + # The existing migration CLI clears PGOPTIONS. Database-scoped + # defaults on this freshly generated Linux role also suppress + # server log DETAIL/STATEMENT values, not just child stderr. + for setting, value in QUIET_PG_SETTINGS: + connection.execute('ALTER ROLE "truf" IN DATABASE "truf" SET ' + + setting + " TO '" + value + "'") + connection.commit() + report['recovery_applied'] = any(report['pipeline_before'].values()) + if report['recovery_applied']: + progress(9) + _command(runtime, runtime._bootstrap_command( + 'migrate-runtime-safety', '--config', str(path), '--apply', '--sources-stopped', + '--recover-stale-result-pipeline', '--max-rows', '10000', '--max-seconds', '3600', + ), COUNT_TIMEOUT + 600, progress) + # Bundle ingestion can insert derived_postman_targets carrying the + # original Windows locators. Recheck after that intentional handoff. + with _authority(runtime, config, manifest['database'], progress): + with _connect(runtime, readonly=False) as connection: + _online(connection, identity) + for name, count in _rebase_postman(runtime, connection, files).items(): + # Reviewed counts are visits; adjusted counts are writes. + report[name] += count + # Both CLIs own their own authority; no importer SQL sessions survive + # the handoff (the migration explicitly rejects other client sessions). + progress(10) + _command(runtime, runtime._bootstrap_command( + 'migrate-runtime-safety', '--config', str(path), '--apply', '--sources-stopped', + ), MIGRATE_TIMEOUT, progress) + progress(11) + with _authority(runtime, config, manifest['database'], progress): + with _connect(runtime) as connection: + _online(connection, identity) + _preserved(runtime, connection, preserved) + report['preserved'] = {name: {k: item[k] for k in ('count', 'sha256')} + for name, item in preserved.items()} + report['pipeline_after'] = _pipeline_counts(connection) + if any(report['pipeline_after'].values()): + raise Failure(2, sum(report['pipeline_after'].values())) + from scanner_db import ScannerDB + + db = ScannerDB(db_url=_credentials(runtime)[0], initialize=False) + try: + if not db.enabled or not db.conn.is_postgres: + raise Failure() + db.conn.execute('SET default_transaction_read_only = on') + db.conn.execute(f'SET statement_timeout = {COUNT_TIMEOUT * 1000}') + db.conn.commit() + db.require_runtime_safety_schema() + db.require_final_cutover() + cutover = db.final_cutover_status() + report['cutover_sha256'] = _sha(cutover['evidence_sha256']) + report['migration_rows'] = _integer(db.conn.execute( + 'SELECT count(*) AS count FROM runtime_schema_migrations').fetchone()['count']) + db.conn.commit() + finally: + db.close() + with _connect(runtime, readonly=False) as connection: + for setting, _value in QUIET_PG_SETTINGS: + connection.execute('ALTER ROLE "truf" IN DATABASE "truf" RESET ' + setting) + connection.commit() + return identity + except BaseException as exc: + _diagnostic(progress, 1, exc) + raise + finally: + if started: + try: + progress(12, cleanup=True) + except BaseException: + pass + _stop_cli(runtime, path, progress) + report['maintenance_stopped'] = True + + +def import_snapshot(runtime, manifest_sha256) -> int: + """Import once, returning only a safe code; never let diagnostics escape.""" + output, errors = sys.stdout, sys.stderr + phase, report, mutations = 1, {}, False + + def progress(number, count=0, size=0, *, cleanup=False, diagnostic=None): + nonlocal phase + if not cleanup: + _checkpoint(runtime) + phase = number + if diagnostic is not None: + stage, failure_code, review_count, type_id, line = diagnostic + event = dict(phase=phase, stage=stage, code=failure_code, review_count=review_count, + type_id=type_id, line=line, attempt=count) + key = 'failure' if stage == 1 else 'hold' + first = key not in report + if first: + report[key] = event + if first or stage != 1: + try: + print('import-snapshot-diagnostic', *event.values(), file=errors, flush=True) + except BaseException: + pass + if first and mutations: + try: + _write(runtime, runtime.DATA / ('config/windows-import-' + key + '.json'), _encoded(event)) + except BaseException: + pass + try: + print('import-snapshot', number, count, size, file=output, flush=True) + except BaseException: + if not cleanup: + raise + + code, review_count, signal_installed = 1, 0, False + try: + previous_interrupt = signal.signal(signal.SIGINT, lambda *_: setattr(runtime, '_shutdown_requested', True)) + signal_installed = True + with _quiet(): + from runtime_security import PrivateFileLock, write_private_json_exclusive + + _checkpoint(runtime) + lock_path = runtime.INITIALIZE_LOCK + runtime.private_path(lock_path.parent, directory=True) + if os.path.lexists(lock_path): + _regular(lock_path, runtime) + with PrivateFileLock(str(lock_path)), contextlib.ExitStack() as stack: + try: + progress(1) + _mounts(runtime) + opened = {name: stack.enter_context(_input(IMPORT / name)) + for name in ('manifest.json', 'files.tar', 'database.dump')} + manifest_handle, manifest_fingerprint = opened['manifest.json'] + payload = manifest_handle.read(MAX_MANIFEST + 1) + manifest, files = _manifest(payload, manifest_sha256) + placeholders = _fresh(runtime, files) + report = {'manifest_sha256': manifest_sha256.lower(), + 'archive_sha256': manifest['archive']['sha256'], + 'database_sha256': manifest['database']['sha256'], + 'files': len(files), 'file_bytes': sum(f['size'] for f in files.values()), + **_space(runtime, manifest)} + archive, archive_fingerprint = opened['files.tar'] + dump, dump_fingerprint = opened['database.dump'] + progress(2) + _archive(runtime, archive, manifest['archive'], files, progress) + _unchanged(IMPORT / 'files.tar', archive, archive_fingerprint) + _hash_input(runtime, IMPORT / 'database.dump', dump, dump_fingerprint, manifest['database'], dump=True) + _unchanged(IMPORT / 'manifest.json', manifest_handle, manifest_fingerprint) + if _fresh(runtime, files) != placeholders: + raise Failure() + _space(runtime, manifest) + progress(3) + mutations = True + _write(runtime, runtime.DATA / 'config/windows-import-manifest.json', payload) + copied = _archive(runtime, archive, manifest['archive'], files, progress, placeholders) + _unchanged(IMPORT / 'files.tar', archive, archive_fingerprint) + progress(4) + path, config, adjusted, config_hash = _configuration(runtime) + report.update(adjusted_keys=adjusted, config_sha256=config_hash) + os.environ.update(TRUF_DB_STATEMENT_TIMEOUT_MS=str(COUNT_TIMEOUT * 1000), + TRUF_DB_LOCK_TIMEOUT_MS='10000', TRUF_DB_IDLE_TRANSACTION_TIMEOUT_MS='3600000') + progress(5) + identity = _restore(runtime, path, config, manifest, files, dump, dump_fingerprint, progress, report) + # Reacquire the stopped endpoint and retain it through the + # last publication, rather than treating a CLI exit as a marker. + with _authority(runtime, config, manifest['database'], progress, stopped=True) as stopped_identity: + if identity != stopped_identity: + raise Failure() + checked = 0 + for name, before in copied.items(): + _checkpoint(runtime) + # Only the explicitly permitted fenced recovery can + # consume/move incoming tmp/ready bundle artifacts. + if (report.get('recovery_applied') and name.startswith( + ('scanner-result-bundles/tmp/', 'scanner-result-bundles/ready/'))): + continue + destination = runtime.DATA / name + if _regular(destination, runtime) != before: + with _input(destination, runtime) as (handle, current): + _hash_input(runtime, destination, handle, current, files[name], private=True) + checked += 1 + if checked % 1024 == 0: + progress(11, checked, 0) + report['preserved_files'] = checked + report['remaining_free_bytes'] = shutil.disk_usage(runtime.DATA).free + if report['remaining_free_bytes'] < 20 * GIB: + raise Failure() + for name, (handle, before) in opened.items(): + _unchanged(IMPORT / name, handle, before) + stack.close() + progress(13) + report.update(status='verified-stopped', maintenance_stopped=True) + report_hash = _write(runtime, runtime.DATA / 'config/windows-import-report.json', _encoded(report)) + _checkpoint(runtime) + write_private_json_exclusive(str(runtime.INITIALIZED), { + 'format': runtime.FORMAT, 'system_identifier': identity['system_identifier'], + 'pg_major': identity['pg_major'], 'manifest_sha256': manifest_sha256.lower(), + 'import_report_sha256': report_hash, + }) + code = 0 + except BaseException as exc: + _diagnostic(progress, 1, exc) + failure = report.get('failure', {}) + code, review_count = failure.get('code', 1), failure.get('review_count', 0) + if mutations and not os.path.lexists(runtime.DATA / 'config/windows-import-report.json'): + report.update(status='failed-unmarked', phase=phase, code=code, review_count=review_count) + _write(runtime, runtime.DATA / 'config/windows-import-report.json', _encoded(report)) + raise + except BaseException as exc: + _diagnostic(progress, 1, exc) + failure = report.get('failure', {}) + code, review_count = failure.get('code', 1), failure.get('review_count', 0) + finally: + if signal_installed: + signal.signal(signal.SIGINT, previous_interrupt) + if code: + print('import-snapshot', phase, code, review_count, file=errors, flush=True) + else: + summary = {name: report[name] for name in ( + 'manifest_sha256', 'archive_sha256', 'database_sha256', 'files', 'file_bytes', + 'config_sha256', 'adjusted_keys', 'cutover_sha256', 'migration_rows', + 'postman_reviewed', 'postman_adjusted', + )} + summary.update({key: report['raw'][key] for key in ('tables', 'rows', 'sequences_verified')}) + print(json.dumps(summary, ensure_ascii=True, sort_keys=True), file=output, flush=True) + return code diff --git a/app/container_import_config.py b/app/container_import_config.py new file mode 100644 index 0000000..ee8dfc9 --- /dev/null +++ b/app/container_import_config.py @@ -0,0 +1,127 @@ +"""Pure translation of the reviewed Windows config; YAML and file copies are caller-owned.""" + +from copy import deepcopy +import ntpath + + +# Only these reviewed absolute Windows paths have known container replacements. +_FIXED_GLOBAL_PATHS = { + 'root_dir': (r'D:\truf', '/opt/truf'), + 'project_dir': (r'D:\truf\app', '/opt/truf/app'), + 'runtime_dir': (r'D:\truf\runtime', '/data/runtime-linux'), + 'postgres_data_dir': (r'S:\postgres-data', '/data/postgres-linux'), + 'postgres_bin_dir': (r'D:\truf\runtime\postgres\pgsql\bin', '/usr/lib/postgresql/16/bin'), + 'result_bundle_dir': (r'S:\scanner-result-bundles', '/data/scanner-result-bundles'), + 'work_dir': (r'S:\scanner-work', '/data/scanner-work'), + 'control_dir': (r'D:\truf\runtime\control', '/run/truf/control'), + 'trufflehog_path': (r'C:\Tools\trufflehog.exe', '/usr/local/bin/trufflehog'), + 'proxy_file': (r'D:\truf\runtime\proxy.txt', '/data/runtime-linux/proxy.txt'), + 'secrets_file': (r'D:\truf\app\secrets.yaml', '/data/config/secrets.yaml'), + 'trufflehog_config': ( + r'D:\truf\app\trufflehog-custom-detectors.yaml', + '/data/config/trufflehog-custom-detectors.yaml', + ), +} +_PATH_FIELDS = { + 'global': ( + 'result_spool_dir', 'legacy_result_spool_dir', 'results_dir', 'queue_dir', + 'state_dir', 'log_dir', 'keycheck_dir', 'postman_cache_dir', 'gharchive_cache_dir', + 'database_path', 'state_file', 'api_proxy_file', 'download_proxy_file', + 'dashboard_db_path', 'scan_limiter_db', 'dockerhub_tag_cache_path', + ), + 'supervisor': ('log_dir', 'supervisor_log', 'status_file', 'dashboard_log', 'state_dir'), + 'keychecks': ('input', 'proxy_file', 'keycheck_dir', 'summary_tsv', 'summary_json', 'alive_summary_tsv'), +} +_SOURCE_PATH_FIELDS = ('target_file', 'trufflehog_config', 'postman_cache_dir', 'gharchive_cache_dir') +_WINDOWS_KNOBS = ( + 'trufflehog_job_memory_limit_bytes', + 'trufflehog_windows_job_cpu_weight', + 'trufflehog_windows_memory_priority', +) + + +def translate_windows_config(original, baseline): + """Return an independent config and sorted, changed dotted key paths (never values). + + ``baseline`` is the parsed config.linux.yaml, not a general merge source. + Fixed paths must match the container contract. Other known path fields keep + relative paths/templates with Linux separators; unreviewed Windows absolute + paths raise ValueError naming only the key. No environment, filesystem, + runtime, database, or YAML operations are performed. + """ + if not isinstance(original, dict) or not isinstance(baseline, dict): + raise TypeError('original and baseline must be dictionaries') + config = deepcopy(original) + adjusted = set() + + def assign(mapping, key, value, prefix): + if key not in mapping or mapping[key] != value: + mapping[key] = deepcopy(value) + adjusted.add(prefix + '.' + key) + + def path_value(value, key_path, approved=None): + if value is None: + return value + if not isinstance(value, str): + raise ValueError('Expected path string at ' + key_path) + if ntpath.splitdrive(value)[0] or value.startswith('\\'): + if approved is None or ntpath.normcase(ntpath.normpath(value)) != ntpath.normcase(ntpath.normpath(approved[0])): + raise ValueError('Unsupported Windows path at ' + key_path) + return approved[1] + return value.replace('\\', '/') + + linux_paths = {key: pair[1] for key, pair in _FIXED_GLOBAL_PATHS.items()} + control = _FIXED_GLOBAL_PATHS['control_dir'] + supervisor_paths = {'control_dir': control} + for key, filename in (('instance_file', 'supervisor.instance.json'), ('lock_file', 'supervisor.lock')): + supervisor_paths[key] = (control[0] + '\\' + filename, control[1] + '/' + filename) + + for section, fixed_paths in (('global', _FIXED_GLOBAL_PATHS), ('supervisor', supervisor_paths)): + mapping = config.setdefault(section, {}) + for key, pair in fixed_paths.items(): + key_path = section + '.' + key + value = path_value(mapping.get(key), key_path, pair) + if key == 'trufflehog_path' and value not in (None, '', 'trufflehog', 'trufflehog.exe', pair[1]): + raise ValueError('Unsupported executable path at ' + key_path) + # The copied original policy intentionally replaces the image policy. + if key != 'trufflehog_config': + try: + baseline_path = baseline[section][key].format_map(linux_paths) + except (KeyError, AttributeError, ValueError): + raise ValueError('Invalid Linux baseline path at ' + key_path) from None + if baseline_path != pair[1]: + raise ValueError('Invalid Linux baseline path at ' + key_path) + assign(mapping, key, pair[1], section) + + groups = [(section, config.get(section, {}), fields) for section, fields in _PATH_FIELDS.items()] + groups.extend(('sources.' + name, source, _SOURCE_PATH_FIELDS) + for name, source in config.get('sources', {}).items()) + for prefix, mapping, fields in groups: + for key in fields: + if key not in mapping: + continue + approved = None + if key in ('proxy_file', 'api_proxy_file', 'download_proxy_file'): + approved = _FIXED_GLOBAL_PATHS['proxy_file'] + elif key == 'trufflehog_config': + approved = _FIXED_GLOBAL_PATHS[key] + value = path_value(mapping[key], prefix + '.' + key, approved) + if key == 'trufflehog_config' and value in ( + '{project_dir}/trufflehog-custom-detectors.yaml', 'trufflehog-custom-detectors.yaml', + ): + value = linux_paths[key] + assign(mapping, key, value, prefix) + + for key in ('max_active_scans', 'opportunistic_scan_slots') + _WINDOWS_KNOBS: + assign(config['global'], key, baseline['global'][key], 'global') + for key in ('interactive', 'autostart', 'control_host', 'control_port'): + assign(config['supervisor'], key, baseline['supervisor'][key], 'supervisor') + assign(config['supervisor'].setdefault('dashboard', {}), 'enabled', + baseline['supervisor']['dashboard']['enabled'], 'supervisor.dashboard') + for name, source in config.get('sources', {}).items(): + source_baseline = baseline.get('sources', {}).get(name, {}) + for key in _WINDOWS_KNOBS: + if key in source or key in source_baseline: + assign(source, key, source_baseline.get(key, baseline['global'][key]), 'sources.' + name) + + return config, sorted(adjusted) diff --git a/app/container_projection_recovery.py b/app/container_projection_recovery.py new file mode 100644 index 0000000..1e82bcb --- /dev/null +++ b/app/container_projection_recovery.py @@ -0,0 +1,416 @@ +"""Explicit maintenance-only loss acknowledgement for the reviewed copied output. + +No CLI, startup hook, connection creation, PostgreSQL lifecycle, or output rebuild. +The caller must retain initialize.lock and ClusterAuthorityLock through this call +AND subsequent positively verified maintenance stop, including every exception or +uncertain commit. It must keep the target isolated with no workers/network clients. +Use an idle, writable, autocommit=True psycopg connection with dict_row and quiet server log +settings. The supplied runtime is the prepared container_runtime module, not its +main()/initialize() entrypoint. Never call this during normal initialized startup. + +The immutable PREPARED journal describes intent, not a fabricated completed rename. +An exact before state can be applied; an exact after state is a read-only retry. +Partial journals and later output require review, never automatic cleanup/rewind. +""" + +import hashlib +import json +import os +from pathlib import Path +import re +import time +from datetime import datetime, timezone + + +DATA = Path('/data') +RUN = Path('/run/truf') +APPROVED_MANIFEST_SHA256 = '08344147133c37d4b6f404cf4fac3e59d58f94917f1fa58a77cbb68c36db7e8a' +JOURNAL_NAME = 'found-secrets-loss-g13-g14.prepared.json' +FORMAT = 'truf-found-secrets-loss-g13-g14-v1' +LOSS = 'Previously published copied output intentionally lost; all PostgreSQL history retained. No rotation or rename occurred.' +OLD_GENERATION, NEW_GENERATION, OLD_OFFSET = 13, 14, 97783145 +APPEND_COUNT, ROTATION_COUNT = 38024, 13 +MAX_JOURNAL = 1024 * 1024 +STREAM_COLUMNS = {'stream_name', 'base_relative_path', 'current_generation', 'rotation_bytes', + 'max_generations', 'created_at', 'updated_at'} +CURSOR_COLUMNS = {'stream_name', 'generation', 'committed_offset', 'last_append_id', 'last_job_id', + 'last_event_id', 'last_event_hash', 'updated_at'} + + +class ProjectionRecoveryError(RuntimeError): + """Safe diagnostic only; caller still owns maintenance/stop authority.""" + + +def _encoded(value): + return (json.dumps(value, ensure_ascii=True, sort_keys=True, + separators=(',', ':'), allow_nan=False) + '\n').encode('ascii') + + +def recover_found_secrets_projection(runtime, connection, *, system_identifier, manifest_sha256, + initialize_lock, authority_lock): + """Apply only the pinned g13 loss transition, or recognize its exact retry. + + Caller-owned locks must be acquired runtime_security lock objects for the fixed + target. This function never closes the connection or releases those locks. + Its own file lock and transaction-scoped advisory/table locks exclude writers. + Returned 'committed'/'already-committed' is not target readiness or stop proof. + journal_sha256 hashes the complete immutable file, not just its inner record. + """ + stage = 'preflight' + deadline = time.monotonic() + 10800 + try: + from container_import import _fsync_dir, _identifier, _input, _integer, _json, _manifest, _regular, _write, MAX_MANIFEST + from runtime_security import ClusterAuthorityLock, PrivateFileLock + + runtime.require_container() + if (runtime.DATA != DATA or runtime.RUN != RUN + or runtime.INITIALIZED != DATA / 'initialized.json' + or runtime.INITIALIZE_LOCK != DATA / 'initialize.lock' + or manifest_sha256 != APPROVED_MANIFEST_SHA256 + or not isinstance(system_identifier, str) + or re.fullmatch(r'[1-9][0-9]{0,19}', system_identifier) is None + or connection.closed or connection.broken or connection.autocommit is not True + or int(connection.info.transaction_status) != 0): + raise ValueError() + config_dir, results = DATA / 'config', DATA / 'runtime-linux/results' + identity_path = DATA / 'runtime-linux/postgres/cluster_identity.json' + manifest_path = config_dir / 'windows-import-manifest.json' + journal_path = config_dir / JOURNAL_NAME + projector_lock = results / '.jsonl-projector.lock' + endpoint = _encoded({'database': 'truf', 'host': '127.0.0.1', 'port': 5432, + 'schema': 'public'}).decode('ascii').strip() + data_hash = hashlib.sha256(str(DATA / 'postgres-linux').encode('utf-8')).hexdigest() + endpoint_hash = hashlib.sha256(endpoint.encode('ascii')).hexdigest() + + def files_stopped(): + if (time.monotonic() >= deadline or getattr(runtime, '_shutdown_requested', True) is not False + or not isinstance(initialize_lock, PrivateFileLock) or initialize_lock.acquired is not True + or initialize_lock.path != os.path.normcase(str(runtime.INITIALIZE_LOCK)) + or not isinstance(authority_lock, ClusterAuthorityLock) or authority_lock.acquired is not True + or authority_lock.data_directory != str(DATA / 'postgres-linux') + or authority_lock.endpoint_identity != endpoint + or authority_lock.path != str(DATA / 'runtime-linux/postgres' / f'.cluster-authority-{data_hash}.lock') + or authority_lock.endpoint_path != str(RUN / 'authority' / f'endpoint-{endpoint_hash}.lock')): + raise ValueError() + for directory in (DATA, config_dir, results, identity_path.parent, + DATA / 'runtime-linux/logs', RUN, RUN / 'control'): + runtime.private_path(directory, directory=True) + _regular(runtime.INITIALIZE_LOCK, runtime) + for path in (runtime.INITIALIZED, RUN / 'control/supervisor.instance.json', + RUN / 'control/supervisor.pid', DATA / 'runtime-linux/logs/supervisor.instance.json', + DATA / 'runtime-linux/logs/supervisor.pid'): + try: + path.lstat() + except FileNotFoundError: + continue + raise ValueError() + with os.scandir(results) as entries: + for index, entry in enumerate(entries): + if index >= 100000 or entry.name.casefold() == 'found_secrets' or entry.name.casefold().startswith('found_secrets.'): + raise ValueError() + + def read_private(path, limit): + with _input(path, runtime) as (handle, before): + if not 0 < before[4] <= limit: + raise ValueError() + raw = handle.read(limit + 1) + if len(raw) != before[4]: + raise ValueError() + return raw, _json(raw) + + files_stopped() + identity_raw, identity = read_private(identity_path, MAX_JOURNAL) + manifest_raw, _ = read_private(manifest_path, MAX_MANIFEST) + manifest, _ = _manifest(manifest_raw, manifest_sha256) + expected_identity = {'pg_major': 16, 'system_identifier': system_identifier, + 'data_directory': str(DATA / 'postgres-linux'), 'database': 'truf', + 'user': 'truf', 'port': 5432} + if (not isinstance(identity, dict) or any(identity.get(k) != v for k, v in expected_identity.items()) + or manifest['database']['system_identifier'] == system_identifier + or 'sequence_states' not in manifest['database']): + raise ValueError() + binding = {'system_identifier': system_identifier, 'manifest_sha256': manifest_sha256, + 'identity_sha256': hashlib.sha256(identity_raw).hexdigest()} + + def online(): + if time.monotonic() >= deadline or runtime._shutdown_requested: + raise ValueError() + connection.execute('SELECT pg_catalog.pg_stat_clear_snapshot()') + row = connection.execute("""SELECT pg_catalog.current_database() AS database, + current_user AS user_name, pg_catalog.current_setting('data_directory') AS data_directory, + pg_catalog.current_setting('port')::int AS port, + pg_catalog.current_setting('server_version_num')::int AS version_num, + pg_catalog.pg_is_in_recovery() AS in_recovery, + (SELECT system_identifier::text FROM pg_catalog.pg_control_system()) AS system_identifier, + (SELECT rolsuper FROM pg_catalog.pg_roles WHERE rolname = current_user) AS superuser, + (SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE backend_type = 'client backend' + AND pid <> pg_catalog.pg_backend_pid()) AS other_clients, + pg_catalog.current_schema() AS schema_name, + pg_catalog.current_setting('search_path') AS search_path, + pg_catalog.current_setting('transaction_read_only') AS read_only, + pg_catalog.current_setting('fsync') AS fsync, + pg_catalog.current_setting('full_page_writes') AS full_page_writes, + EXISTS (SELECT 1 FROM pg_catalog.pg_namespace n CROSS JOIN LATERAL pg_catalog.aclexplode( + COALESCE(n.nspacl, pg_catalog.acldefault('n', n.nspowner))) acl + WHERE n.nspname = 'public' AND acl.grantee = 0 AND acl.privilege_type = 'CREATE') AS public_create""").fetchone() + expected = {'database': 'truf', 'user_name': 'truf', 'data_directory': str(DATA / 'postgres-linux'), + 'port': 5432, 'system_identifier': system_identifier, 'in_recovery': False, + 'superuser': True, 'other_clients': 0, 'schema_name': 'public', + 'search_path': 'public', 'read_only': 'off', 'public_create': False, + 'fsync': 'on', 'full_page_writes': 'on'} + if (not isinstance(row, dict) or any(row.get(k) != v for k, v in expected.items()) + or _integer(row.get('version_num')) // 10000 != 16): + raise ValueError() + + def metadata(): + rows = connection.execute("""SELECT s.stream_name, c.stream_name AS cursor_stream_name, + pg_catalog.row_to_json(s) AS stream, pg_catalog.row_to_json(c) AS cursor + FROM public.projection_streams s FULL JOIN public.projection_cursors c + ON c.stream_name = s.stream_name""").fetchall() + seen, found = set(), None + scans = {'scan_results': 'scan_results.jsonl', 'found_secrets': 'found_secrets.jsonl', + 'scan_errors': 'scan_errors.log'} + for row in rows: + name, stream, cursor = row['stream_name'], row['stream'], row['cursor'] + if (not isinstance(name, str) or name in seen or row['cursor_stream_name'] != name + or not isinstance(stream, dict) or set(stream) != STREAM_COLUMNS + or not isinstance(cursor, dict) or set(cursor) != CURSOR_COLUMNS + or stream['stream_name'] != name or cursor['stream_name'] != name + or _integer(stream['current_generation']) != _integer(cursor['generation'])): + raise ValueError() + seen.add(name) + _integer(cursor['committed_offset']) + _integer(stream['rotation_bytes'], 1) + _integer(stream['max_generations']) + expected_path = scans.get(name) + if expected_path is None: + match = re.fullmatch(r'keycheck:([a-z0-9][a-z0-9_.-]{0,63}):(results|status)', name) + if not match: + raise ValueError() + suffix = 'Results.jsonl' if match[2] == 'results' else 'Checked.txt' + expected_path = f'{match[1]}/{match[1]}{suffix}' + if stream['base_relative_path'] != expected_path: + raise ValueError() + if name == 'found_secrets': + found = {'stream': stream, 'cursor': cursor} + if len(seen) != 34 or not scans.keys() <= seen: + raise ValueError() + return found + + tables = sorted(manifest['database']['table_counts']) + sequences = manifest['database']['sequence_states'] + if set(sequences) != {'public'}: + raise ValueError() + + def preserved(): + proof = {} + for table in tables: + if time.monotonic() >= deadline or runtime._shutdown_requested: + raise ValueError() + where = " WHERE t.stream_name <> 'found_secrets'" if table in ('projection_streams', 'projection_cursors') else '' + digest, count = hashlib.sha256(), 0 + with connection.cursor(name='found_loss_digest') as cursor: + cursor.itersize = 1000 + cursor.execute("SELECT pg_catalog.encode(pg_catalog.sha256(pg_catalog.convert_to(" + "pg_catalog.row_to_json(t)::text, 'UTF8')), 'hex') COLLATE \"C\" AS digest FROM public." + + _identifier(table) + ' AS t' + where + ' ORDER BY digest') + for row in cursor: + value = row['digest'] + if not isinstance(value, str) or re.fullmatch(r'[0-9a-f]{64}', value) is None: + raise ValueError() + digest.update(value.encode('ascii')) + count += 1 + if count % 1000 == 0 and (time.monotonic() >= deadline or runtime._shutdown_requested): + raise ValueError() + if count != manifest['database']['table_counts'][table] - int(bool(where)): + raise ValueError() + proof[table] = {'rows': count, 'sha256': digest.hexdigest()} + rows = connection.execute("""SELECT c.relname AS name FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = 'public' AND c.relkind = 'S' ORDER BY c.relname""").fetchall() + if {row['name'] for row in rows} != set(sequences['public']): + raise ValueError() + for row in rows: + state = connection.execute('SELECT last_value, is_called FROM public.' + _identifier(row['name'])).fetchone() + if _encoded(state) != _encoded(sequences['public'][row['name']]): + raise ValueError() + return {'tables': proof, 'sequences': {'count': len(rows), + 'sha256': hashlib.sha256(_encoded(sequences)).hexdigest()}} + + with PrivateFileLock(str(projector_lock)): + _regular(projector_lock, runtime) + with connection.transaction(): + stage = 'database-fences' + connection.execute('SET TRANSACTION ISOLATION LEVEL READ COMMITTED') + online() + connection.execute("SET LOCAL lock_timeout = '5s'") + connection.execute("SET LOCAL statement_timeout = '10800s'") + connection.execute("SET LOCAL temp_file_limit = '4GB'") + connection.execute("SET LOCAL work_mem = '128MB'") + connection.execute('SET LOCAL max_parallel_workers_per_gather = 0') + connection.execute("SET LOCAL row_security = off") + connection.execute('SET LOCAL synchronous_commit = on') + for setting, value in (('log_min_error_statement', 'panic'), ('log_min_messages', 'panic'), + ('log_statement', 'none'), ('log_min_duration_statement', '-1')): + connection.execute('SET LOCAL ' + setting + " = '" + value + "'") + for arguments in ((1414681926, 1785753445), (1414681926, 1768842867), (781273968142991337,)): + lock = connection.execute('SELECT pg_catalog.pg_try_advisory_xact_lock(' + + ','.join(['%s'] * len(arguments)) + ') AS locked', arguments).fetchone() + if not lock or lock['locked'] is not True: + raise ValueError() + catalog_sql = """SELECT c.relname AS name, c.relkind AS kind FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = 'public' AND c.relkind IN ('r','p','f') ORDER BY c.relname""" + catalog = connection.execute(catalog_sql).fetchall() + if len(catalog) != len(tables) or {row['name'] for row in catalog} != set(tables) or any(row['kind'] != 'r' for row in catalog): + raise ValueError() + connection.execute('LOCK TABLE ' + ','.join('public.' + _identifier(table) for table in tables) + + ' IN EXCLUSIVE MODE NOWAIT') + online() + stage = 'history-gates' + gates = connection.execute("""WITH a AS ( + SELECT count(*) AS appends, max(a.generation) AS append_highwater, + count(*) FILTER (WHERE a.state IS DISTINCT FROM 'appended' OR j.id IS NULL + OR j.status IS DISTINCT FROM 'completed' OR j.capacity_released IS DISTINCT FROM 1 + OR j.job_kind IS DISTINCT FROM 'scan_event' OR (j.required_stream_mask & 2) IS DISTINCT FROM 2 + OR a.event_id IS DISTINCT FROM j.event_id OR a.event_hash IS DISTINCT FROM j.event_hash + OR a.generation IS NULL OR a.generation NOT BETWEEN 0 AND 13 + OR a.byte_offset IS NULL OR a.byte_offset < 0 OR a.byte_length IS NULL OR a.byte_length < 0 + OR a.record_count IS NULL OR a.record_count < 0 + OR (a.generation = 13 AND a.byte_offset::numeric + a.byte_length::numeric > 97783145)) AS bad_appends + FROM public.projection_appends a LEFT JOIN public.projection_jobs j ON j.id = a.job_id + WHERE a.stream_name = 'found_secrets'), r AS ( + SELECT count(*) AS rotations, max(to_generation) AS rotation_highwater, + count(*) FILTER (WHERE state IS DISTINCT FROM 'completed' OR from_generation IS NULL OR from_generation < 0 + OR to_generation IS DISTINCT FROM from_generation + 1 OR to_generation > 13 + OR source_bytes IS NULL OR source_bytes < 0 + OR segment_relative_path IS DISTINCT FROM + 'found_secrets.g' || lpad(from_generation::text, 6, '0') || '.jsonl') AS bad_rotations + FROM public.projection_rotations WHERE stream_name = 'found_secrets') + SELECT a.*, r.*, (SELECT count(*) FROM public.projection_append_audit) AS audits, + (SELECT count(*) FROM public.projection_jobs WHERE (required_stream_mask & 2) <> 0 + AND status <> 'completed') AS pending, + (SELECT count(*) FROM public.result_reservations + WHERE state IN ('scanning','ready','ingesting','db_committed')) AS reservations, + (SELECT count(*) FROM public.target_queue q LEFT JOIN public.result_reservations r + ON r.id = q.current_result_reservation_id WHERE q.status = 'in_progress' + OR r.state IN ('scanning','ready','ingesting','db_committed')) AS queue_leases, + (SELECT count(*) FROM public.docker_content_blobs + WHERE state IN ('leased','submitted') OR lease_reservation_id IS NOT NULL) AS blob_leases, + (SELECT count(*) FROM pg_catalog.pg_trigger WHERE NOT tgisinternal + AND tgrelid IN ('public.projection_streams'::regclass,'public.projection_cursors'::regclass)) AS triggers, + (SELECT count(*) FROM pg_catalog.pg_rewrite + WHERE ev_class IN ('public.projection_streams'::regclass,'public.projection_cursors'::regclass)) AS rules + FROM a CROSS JOIN r""").fetchone() + expected_gates = {'appends': APPEND_COUNT, 'append_highwater': 13, 'bad_appends': 0, + 'rotations': ROTATION_COUNT, 'rotation_highwater': 13, 'bad_rotations': 0, + 'audits': 0, 'pending': 0, 'reservations': 0, 'queue_leases': 0, + 'blob_leases': 0, 'triggers': 0, 'rules': 0} + if _encoded(gates) != _encoded(expected_gates): + raise ValueError() + current = metadata() + proof = preserved() + stage = 'journal' + try: + journal_path.lstat() + except FileNotFoundError: + journal = None + else: + raw, journal = read_private(journal_path, MAX_JOURNAL) + if (not isinstance(journal, dict) or set(journal) != {'record', 'sha256'} + or raw != _encoded(journal) + or journal['sha256'] != hashlib.sha256(_encoded(journal['record'])).hexdigest()): + raise ValueError() + if journal is None: + stamp = datetime.now(timezone.utc).isoformat(timespec='seconds') + before = current + record = {'format': FORMAT, 'state': 'PREPARED', 'loss': LOSS, 'binding': binding, + 'prepared_at': stamp, 'before': before, 'preserved': proof, + 'digest_algorithm': 'sha256-concatenated-sorted-pg-row-sha256-hex-v1'} + else: + record = journal['record'] + if (not isinstance(record, dict) or set(record) != {'format', 'state', 'loss', 'binding', + 'prepared_at', 'before', 'after', 'preserved', 'digest_algorithm'} + or record['format'] != FORMAT or record['state'] != 'PREPARED' or record['loss'] != LOSS + or _encoded(record['binding']) != _encoded(binding) + or _encoded(record['preserved']) != _encoded(proof) + or record['digest_algorithm'] != 'sha256-concatenated-sorted-pg-row-sha256-hex-v1'): + raise ValueError() + stamp, before = record['prepared_at'], record['before'] + if (not isinstance(stamp, str) or datetime.fromisoformat(stamp).isoformat(timespec='seconds') != stamp + or not stamp.endswith('+00:00') or set(before) != {'stream', 'cursor'} + or set(before['stream']) != STREAM_COLUMNS or set(before['cursor']) != CURSOR_COLUMNS + or before['stream']['stream_name'] != 'found_secrets' + or before['stream']['base_relative_path'] != 'found_secrets.jsonl' + or before['cursor']['stream_name'] != 'found_secrets' + or _integer(before['stream']['current_generation']) != OLD_GENERATION + or _integer(before['cursor']['generation']) != OLD_GENERATION + or _integer(before['cursor']['committed_offset']) != OLD_OFFSET): + raise ValueError() + _integer(before['cursor']['last_append_id'], 1) + _integer(before['cursor']['last_job_id'], 1) + last = connection.execute("""SELECT id, job_id, stream_name, generation, byte_offset, + byte_length, event_id, event_hash, state FROM public.projection_appends WHERE id = %s""", + (before['cursor']['last_append_id'],)).fetchone() + if (not last or last['id'] != before['cursor']['last_append_id'] + or last['job_id'] != before['cursor']['last_job_id'] or last['stream_name'] != 'found_secrets' + or last['generation'] != OLD_GENERATION or last['state'] != 'appended' + or _integer(last['byte_offset']) + _integer(last['byte_length']) != OLD_OFFSET + or last['event_id'] != before['cursor']['last_event_id'] + or last['event_hash'] != before['cursor']['last_event_hash'] + or not isinstance(last['event_id'], str) or not last['event_id'] + or re.fullmatch(r'[0-9a-f]{64}', last['event_hash'] or '') is None): + raise ValueError() + after = {key: dict(value) for key, value in before.items()} + after['stream'].update(current_generation=NEW_GENERATION, updated_at=stamp) + after['cursor'].update(generation=NEW_GENERATION, committed_offset=0, last_append_id=None, updated_at=stamp) + if journal is not None and _encoded(record['after']) != _encoded(after): + raise ValueError() + already = _encoded(current) == _encoded(after) + if (not already and _encoded(current) != _encoded(before)) or (already and journal is None): + raise ValueError() + if journal is None: + record['after'] = after + journal = {'record': record, 'sha256': hashlib.sha256(_encoded(record)).hexdigest()} + encoded = _encoded(journal) + if len(encoded) > MAX_JOURNAL: + raise ValueError() + _write(runtime, journal_path, encoded) + if read_private(journal_path, MAX_JOURNAL)[0] != encoded: + raise ValueError() + # A prior failure may have left complete bytes without a confirmed fsync. + with _input(journal_path, runtime) as (handle, _): + if handle.read(MAX_JOURNAL + 1) != _encoded(journal): + raise ValueError() + os.fsync(handle.fileno()) + _fsync_dir(config_dir) + if not already: + stage = 'compare-and-swap' + result = connection.execute("""UPDATE public.projection_streams AS s + SET current_generation = 14, updated_at = %s + WHERE s.stream_name = 'found_secrets' AND pg_catalog.to_jsonb(s) = %s::jsonb""", + (stamp, _encoded(before['stream']).decode('ascii'))) + if result.rowcount != 1: + raise ValueError() + result = connection.execute("""UPDATE public.projection_cursors AS c + SET generation = 14, committed_offset = 0, last_append_id = NULL, updated_at = %s + WHERE c.stream_name = 'found_secrets' AND pg_catalog.to_jsonb(c) = %s::jsonb""", + (stamp, _encoded(before['cursor']).decode('ascii'))) + if result.rowcount != 1: + raise ValueError() + if _encoded(metadata()) != _encoded(after) or _encoded(preserved()) != _encoded(proof): + raise ValueError() + stage = 'precommit' + files_stopped() + if (read_private(identity_path, MAX_JOURNAL)[0] != identity_raw + or read_private(manifest_path, MAX_MANIFEST)[0] != manifest_raw + or read_private(journal_path, MAX_JOURNAL)[0] != _encoded(journal) + or _encoded(connection.execute(catalog_sql).fetchall()) != _encoded(catalog)): + raise ValueError() + online() + stage = 'commit' + return {'status': 'already-committed' if already else 'committed', 'journal_path': str(journal_path), + 'journal_sha256': hashlib.sha256(_encoded(journal)).hexdigest(), **binding, 'generation': NEW_GENERATION} + except BaseException: + raise ProjectionRecoveryError('Projection recovery refused at ' + stage + + '; retain caller maintenance authority and verify stop.') from None diff --git a/app/container_runtime.py b/app/container_runtime.py new file mode 100644 index 0000000..e33d4df --- /dev/null +++ b/app/container_runtime.py @@ -0,0 +1,594 @@ +"""Isolated entrypoint for the private, read-only Docker deployment.""" + +import argparse +import http.client +import json +import os +from pathlib import Path +import re +import runpy +import secrets +import signal +import stat +import subprocess +import sys +from urllib.parse import quote + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('container runtime could not disable bytecode writes') + + +APP = Path('/opt/truf/app') +DATA = Path('/data') +RUN = Path('/run/truf') +UID = GID = 10001 +DEFAULT_CONFIG = APP / 'config.linux.yaml' +PROVISIONED = DATA / '.provisioned.json' +INITIALIZED = DATA / 'initialized.json' +INITIALIZE_LOCK = DATA / 'initialize.lock' +PASSWORD = DATA / 'postgres-password' +PROVIDER_SECRETS = DATA / 'config/secrets.yaml' +FORMAT = 'truf-container-data-v1' +WORKER_HEALTH_TOKEN = '0' * 64 +DIRECTORIES = ( + 'home', 'config', 'managed-files', 'runtime-linux', 'runtime-linux/results', + 'runtime-linux/queues', 'runtime-linux/state', + 'runtime-linux/state/gharchive_cache', 'runtime-linux/logs', + 'runtime-linux/keychecks', 'runtime-linux/postman_cache', + 'runtime-linux/result_spool', 'runtime-linux/postgres', + 'runtime-linux/postgres/logs', 'postgres-linux', 'scanner-work', + 'scanner-result-bundles', 'scanner-result-bundles/tmp', + 'scanner-result-bundles/ready', 'scanner-result-bundles/quarantine', +) +_shutdown_requested = False + + +def private_path(path, *, directory=False): + path = Path(path) + if not path.is_absolute() or '..' in path.parts: + raise RuntimeError('private path must be absolute and normalized') + for component in (*reversed(path.parents), path): + if stat.S_ISLNK(component.lstat().st_mode): + raise RuntimeError('private paths must not contain symlinks') + details = path.lstat() + expected_type = stat.S_ISDIR if directory else stat.S_ISREG + if (not expected_type(details.st_mode) or details.st_uid != UID + or details.st_gid != GID + or stat.S_IMODE(details.st_mode) != (0o700 if directory else 0o600)): + raise RuntimeError('private path ownership, type, or mode is invalid: ' + str(path)) + return path + + +def require_container(*, provisioning=False): + if (sys.platform != 'linux' or Path(__file__) != APP / 'container_runtime.py' + or not Path('/.dockerenv').is_file()): + raise RuntimeError('runtime commands are restricted to the prepared Docker image') + if not (sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode): + raise RuntimeError('container entrypoint requires python -I -S -B') + if os.getuid() != os.geteuid() or os.geteuid() != (0 if provisioning else UID): + raise RuntimeError('unexpected container runtime UID') + if not provisioning and os.getgid() != GID: + raise RuntimeError('unexpected container runtime GID') + if not os.statvfs(APP).f_flag & os.ST_RDONLY: + raise RuntimeError('the application image must be mounted read-only') + private_path(APP.parent, directory=True) + private_path(APP, directory=True) + private_path(APP / 'container_runtime.py') + with open('/proc/self/mountinfo', 'rb') as handle: + payload = handle.read(1024 * 1024 + 1) + if len(payload) > 1024 * 1024: + raise RuntimeError('mount inventory exceeds its bound') + mounts = {} + for line in payload.splitlines(): + fields = line.split() + if len(fields) > 6 and b'-' in fields: + mounts[fields[4]] = fields[fields.index(b'-') + 1] + if mounts.get(b'/data') not in (b'ext4', b'xfs', b'btrfs', b'zfs'): + raise RuntimeError('/data must be an independent native Linux data volume') + if mounts.get(b'/run/truf') != b'tmpfs': + raise RuntimeError('/run/truf must be an independent private tmpfs') + private_path(DATA, directory=True) + private_path(RUN, directory=True) + os.umask(0o077) + + +def _read_json(path): + path = private_path(path) + if path.stat().st_size > 4096: + raise RuntimeError('container marker exceeds its bound') + value = json.loads(path.read_text(encoding='utf-8')) + if not isinstance(value, dict) or value.get('format') != FORMAT: + raise RuntimeError('unrecognized container data marker') + return value + + +def _write_new(path, payload, *, provisioning=False): + try: + private_path(path.parent, directory=True) + descriptor = os.open( + path, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, + ) + with os.fdopen(descriptor, 'wb') as handle: + if provisioning: + os.fchown(handle.fileno(), UID, GID) + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + descriptor = os.open(path.parent, os.O_RDONLY | os.O_DIRECTORY) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + private_path(path) + finally: + payload = None + + +def provision(): + import fcntl + + lock = DATA / '.provision.lock' + try: + descriptor = os.open(lock, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600) + os.fchown(descriptor, UID, GID) + except FileExistsError: + private_path(lock) + descriptor = os.open(lock, os.O_WRONLY | os.O_NOFOLLOW) + with os.fdopen(descriptor, 'wb') as handle: + fcntl.flock(handle, fcntl.LOCK_EX | fcntl.LOCK_NB) + if PROVISIONED.exists(): + _read_json(PROVISIONED) + managed_files = DATA / 'managed-files' + try: + os.mkdir(managed_files, 0o700) + os.chown(managed_files, UID, GID) + except FileExistsError: + pass + for name in DIRECTORIES: + private_path(DATA / name, directory=True) + private_path(PASSWORD) + private_path(PROVIDER_SECRETS) + print('Container data layout is already provisioned; nothing was changed.', flush=True) + return + if set(os.listdir(DATA)) - {'home', '.provision.lock'}: + raise RuntimeError('refusing to provision nonempty or partially initialized data') + home = DATA / 'home' + if home.exists() and any(home.iterdir()): + raise RuntimeError('refusing to provision a nonempty home directory') + for name in DIRECTORIES: + path = DATA / name + if not path.exists(): + os.mkdir(path, 0o700) + os.chown(path, UID, GID) + private_path(path, directory=True) + _write_new(PASSWORD, (secrets.token_urlsafe(48) + '\n').encode('ascii'), provisioning=True) + _write_new(PROVIDER_SECRETS, b'{}\n', provisioning=True) + _write_new(DATA / 'runtime-linux/proxy.txt', b'', provisioning=True) + _write_new(PROVISIONED, (json.dumps({'format': FORMAT, 'uid': UID, 'gid': GID}) + '\n').encode('ascii'), provisioning=True) + print('Fresh private container data layout provisioned; PostgreSQL is not initialized yet.', flush=True) + + +def prepare_environment( + config_path, *, validate_documents=True, return_config_sha256=False, +): + _read_json(PROVISIONED) + for name in ('authority', 'control', 'tmp'): + path = RUN / name + try: + path.mkdir(mode=0o700) + except FileExistsError: + pass + private_path(path, directory=True) + config_path = Path(config_path) + if config_path != DEFAULT_CONFIG and config_path.parent != DATA / 'config': + raise RuntimeError('configuration must be image-owned or in /data/config') + private_path(config_path) + password = private_path(PASSWORD).read_text(encoding='ascii').rstrip('\n') + if re.fullmatch(r'[A-Za-z0-9_-]{32,128}', password) is None: + raise RuntimeError('the generated PostgreSQL password is invalid') + for name in tuple(os.environ): + upper = name.upper() + if (upper.startswith(('PG', 'TRUF_', 'SCANNER_', 'SCAN_', 'TRUFFLEHOG_', 'KEYCHECK_')) + or upper in ('DATABASE_URL', 'PYTHONPATH', 'PYTHONHOME')): + del os.environ[name] + url = 'postgresql://truf:' + quote(password, safe='') + '@127.0.0.1:5432/truf' + os.environ.update({ + 'PATH': '/usr/local/bin:/usr/bin:/bin:/usr/lib/postgresql/16/bin', + 'HOME': '/data/home', 'TMPDIR': str(RUN / 'tmp'), + 'TMP': str(RUN / 'tmp'), 'TEMP': str(RUN / 'tmp'), + 'TRUF_CONTAINER_CONFIG': str(config_path), + 'TRUF_POSTGRES_DB': 'truf', 'TRUF_POSTGRES_USER': 'truf', + 'TRUF_POSTGRES_PORT': '5432', 'TRUF_POSTGRES_PASSWORD': password, + 'SCANNER_DB_URL': url, 'DATABASE_URL': url, 'TRUF_MANAGED_POSTGRES_DSN': url, + 'TRUF_DB_CONNECT_TIMEOUT_SEC': '3', 'TRUF_DB_STATEMENT_TIMEOUT_MS': '5000', + 'TRUF_DB_LOCK_TIMEOUT_MS': '2000', 'TRUF_DB_IDLE_TRANSACTION_TIMEOUT_MS': '10000', + }) + bootstrap = runpy.run_path(str(APP / 'child_bootstrap.py')) + bootstrap['_enable_dependency_paths']('supervisor') + sys.path.insert(0, str(APP)) + from paths import apply_path_config + from runtime_document_io import ( + load_managed_runtime_config, + validate_managed_runtime_files, + ) + from runtime_security import preflight_lifecycle_paths + + def check_resolved(candidate): + expected = { + 'root_dir': '/opt/truf', 'project_dir': str(APP), + 'runtime_dir': '/data/runtime-linux', 'postgres_data_dir': '/data/postgres-linux', + 'postgres_bin_dir': '/usr/lib/postgresql/16/bin', + 'result_bundle_dir': '/data/scanner-result-bundles', 'work_dir': '/data/scanner-work', + 'control_dir': str(RUN / 'control'), 'secrets_file': str(PROVIDER_SECRETS), + } + if any(candidate['global'].get(name) != value for name, value in expected.items()): + raise RuntimeError('configuration escapes the fixed container storage contract') + if candidate['supervisor'].get('control_dir') != str(RUN / 'control'): + raise RuntimeError('supervisor control must remain on private ephemeral storage') + preflight_lifecycle_paths( + str(config_path), candidate, authority_profile='server', + ) + + documents = ( + validate_managed_runtime_files(str(config_path)) + if validate_documents + else load_managed_runtime_config(str(config_path)) + ) + config = documents.config + config_sha256 = documents.config_sha256 + documents = None + config = apply_path_config(config, str(config_path)) + check_resolved(config) + if return_config_sha256: + return config, config_sha256 + return config + + +def _bootstrap_command(target, *arguments): + return [sys.executable, '-u', '-I', '-S', '-B', str(APP / 'runtime_bootstrap.py'), target, '--', *arguments] + + +def initialize(config_path, config, *, expected_config_sha256=None): + from postgres_runtime import postgres_runtime_paths + from runtime_document_io import load_managed_runtime_config + from runtime_security import PrivateFileLock, read_private_json, write_private_json_exclusive + + def require_stable_config(): + if expected_config_sha256 is None: + return + current = load_managed_runtime_config(str(config_path)) + current_sha256 = current.config_sha256 + current = None + if current_sha256 != expected_config_sha256: + raise RuntimeError('configuration changed after managed runtime validation') + + require_stable_config() + paths = postgres_runtime_paths(config) + + def migrate(*, initialize_base): + if _shutdown_requested: + return False + try: + # Even an uncertain maintenance start must enter the verified stop path. + require_stable_config() + subprocess.run(_bootstrap_command( + 'postgres-runtime', 'maintenance-start', '--config', str(config_path), + ), check=True) + if _shutdown_requested: + return False + require_stable_config() + arguments = [ + 'migrate-runtime-safety', '--config', str(config_path), + ] + if initialize_base: + arguments.append('--initialize-base') + arguments.extend(('--apply', '--sources-stopped')) + subprocess.run(_bootstrap_command(*arguments), check=True) + return True + finally: + subprocess.run(_bootstrap_command( + 'postgres-runtime', 'maintenance-stop', '--config', str(config_path), + ), check=True) + + with PrivateFileLock(str(INITIALIZE_LOCK)): + if INITIALIZED.exists(): + marker = _read_json(INITIALIZED) + identity = read_private_json(paths['identity_path']) + if (not marker.get('system_identifier') + or marker.get('system_identifier') != identity.get('system_identifier') + or marker.get('pg_major') != 16 or identity.get('pg_major') != 16): + raise RuntimeError('initialization marker does not match the bound cluster') + migrated = migrate(initialize_base=False) + if migrated: + print('Existing PostgreSQL migrated and confirmed stopped.', flush=True) + return + if Path(paths['identity_path']).exists() or any(Path(paths['data_dir']).iterdir()): + raise RuntimeError('partial initialization requires offline inspection; no automatic repair is allowed') + if _shutdown_requested: + return + require_stable_config() + subprocess.run(_bootstrap_command('postgres-runtime', 'initialize-empty', '--config', str(config_path)), check=True) + if _shutdown_requested: + return + if not migrate(initialize_base=True): + return + identity = read_private_json(paths['identity_path']) + require_stable_config() + write_private_json_exclusive(str(INITIALIZED), { + 'format': FORMAT, 'system_identifier': identity['system_identifier'], + 'pg_major': identity['pg_major'], + }) + print('Independent PostgreSQL initialized, migrated, cut over, and confirmed stopped.', flush=True) + + +def _probe_worker_api(config): + worker = config['supervisor']['worker_api'] + connection = None + try: + connection = http.client.HTTPConnection( + worker['address'], int(worker['port']), timeout=2, + ) + connection.request( + 'POST', '/api/v1/worker/claim', body=b'', + headers={ + 'Authorization': 'Bearer ' + WORKER_HEALTH_TOKEN, + 'Content-Length': '0', + }, + ) + response = connection.getresponse() + payload = response.read(4097) + if ( + len(payload) > 4096 + or response.status != 401 + or response.getheader('WWW-Authenticate') != 'Bearer' + or json.loads(payload) != { + 'error': { + 'code': 'unauthorized', + 'message': 'worker credentials are invalid', + }, + } + ): + raise RuntimeError('worker API is not ready') + except Exception: + raise RuntimeError('worker API is not ready') from None + finally: + if connection is not None: + connection.close() + + +def health(config, *, require_worker_api=False, require_discovery_producers=False): + # Never create a competing SQL session while first-install migration is exclusive. + _read_json(INITIALIZED) + from scanner_db import ScannerDB + from lifecycle_authority import DISCOVERY_PRODUCER_SOURCES + from supervisor import get_control_snapshot + from supervisor_instance import load_instance_metadata + + metadata = load_instance_metadata(config['supervisor']['instance_file']) + snapshot = get_control_snapshot(metadata) + if (snapshot.get('activation_state') != 'ACTIVE' + or snapshot.get('postgres', {}).get('state') != 'READY' + or snapshot.get('postgres', {}).get('ready') is not True): + raise RuntimeError('supervisor and PostgreSQL are not ready') + signatures = {row[0]: row for row in snapshot.get('signature', ())} + required = ['result-ingester', 'jsonl-projector'] + if config['supervisor'].get('janitor', {}).get('enabled', True): + required.append('janitor') + worker_api_enabled = config['supervisor'].get('worker_api', {}).get( + 'enabled', False, + ) + if require_worker_api and not worker_api_enabled: + raise RuntimeError('worker API is required but disabled') + if worker_api_enabled: + required.append('worker-api') + for name in required: + row = signatures.get(name) + if not row or tuple(row[1:3]) != ('running', 'running') or not row[3] or row[7]: + raise RuntimeError('a required pipeline worker is not running') + if any(row[1] == 'failed' or row[8] for row in signatures.values()): + raise RuntimeError('a managed source is failed or retains uncertain ownership') + if require_discovery_producers: + source_config = config.get('sources') or {} + supervisor_sources = config['supervisor'].get('sources') or {} + enabled_producers = [ + name for name in DISCOVERY_PRODUCER_SOURCES + if ( + (supervisor_sources.get(name) or {}).get('enabled') + if 'enabled' in (supervisor_sources.get(name) or {}) + else (source_config.get(name) or {}).get('enabled', False) + ) + ] + for name in enabled_producers: + row = signatures.get(name) + ready = bool( + row + and row[2] == 'running' + and not row[7] + and not row[8] + and ( + (row[1] == 'running' and row[3]) + or (row[1] == 'waiting' and row[4] == 0) + ) + ) + if not ready: + raise RuntimeError('a required discovery producer is not ready') + if require_worker_api: + _probe_worker_api(config) + db = ScannerDB(db_url=os.environ['SCANNER_DB_URL'], initialize=False) + try: + if not db.enabled or not db.conn.is_postgres: + raise RuntimeError('PostgreSQL application connection is unavailable') + db.set_application_name('truf-container-health') + db.conn.execute('SET default_transaction_read_only = on') + db.conn.commit() + db.require_runtime_safety_schema() + db.require_final_cutover() + for name in ('result_ingester', 'jsonl_projector'): + if not db.pipeline_worker_health(name, metadata['instance_id'])['healthy']: + raise RuntimeError('a required durable worker lease is not ready') + row = db.conn.execute("SELECT current_setting('data_directory') AS data_directory, current_setting('server_version_num') AS version").fetchone() + db.conn.commit() + if row['data_directory'] != '/data/postgres-linux' or int(row['version']) // 10000 != 16: + raise RuntimeError('PostgreSQL identity does not match the container') + finally: + db.close() + for name in ('work_dir', 'result_bundle_dir', 'results_dir'): + path = config['global'][name] + if not os.access(path, os.W_OK | os.X_OK) or os.statvfs(path).f_bavail == 0: + raise RuntimeError('required persistent storage is not writable or is full') + return {'healthy': True, 'activation_state': 'ACTIVE', 'postgres': 'READY', 'workers': sorted(signatures)} + + +def import_secrets(config_path, config, *, expected_config_sha256=None): + from postgres_runtime import postgres_runtime_paths + from runtime_document_io import ( + load_managed_runtime_config, + validate_managed_runtime_files, + ) + from runtime_security import ClusterAuthorityLock, PrivateFileLock, durable_replace, fsync_directory + import yaml + + with PrivateFileLock(str(INITIALIZE_LOCK)), ClusterAuthorityLock( + config, create_parent=False, endpoint_dsn=os.environ['SCANNER_DB_URL'], + ): + if (Path(config['supervisor']['instance_file']).exists() + or (Path(postgres_runtime_paths(config)['data_dir']) / 'postmaster.pid').exists()): + raise RuntimeError('stop the runtime before replacing provider credentials') + payload = sys.stdin.buffer.read(1024 * 1024 + 1) + try: + if not payload or len(payload) > 1024 * 1024: + raise RuntimeError('credential input must be a nonempty YAML document below 1 MiB') + validated = validate_managed_runtime_files( + str(config_path), secrets_bytes=payload, + ) + except BaseException: + payload = None + raise + payload = None + validated_config_sha256 = validated.config_sha256 + if ( + expected_config_sha256 is not None + and validated_config_sha256 != expected_config_sha256 + ): + validated = None + raise RuntimeError('configuration changed while provider credentials were being validated') + try: + serialized = yaml.safe_dump( + validated.secrets, allow_unicode=True, + ).encode('utf-8') + except BaseException: + validated = None + raise + validated = None + temporary = DATA / ('config/secrets-import-' + secrets.token_hex(16)) + try: + try: + _write_new(temporary, serialized) + finally: + serialized = None + private_path(PROVIDER_SECRETS) + current = load_managed_runtime_config(str(config_path)) + current_sha256 = current.config_sha256 + current = None + if current_sha256 != validated_config_sha256: + raise RuntimeError( + 'configuration changed while provider credentials were being validated' + ) + durable_replace(str(temporary), str(PROVIDER_SECRETS)) + fsync_directory(str(PROVIDER_SECRETS.parent)) + finally: + temporary.unlink(missing_ok=True) + print('Private provider credentials replaced; no credentials were printed.', flush=True) + + +def main(argv=None): + global _shutdown_requested + + parser = argparse.ArgumentParser(description=__doc__, allow_abbrev=False) + parser.add_argument('action', choices=('provision', 'initialize', 'run', 'health', 'status', 'import-secrets', 'import-snapshot')) + parser.add_argument('--config', default=os.environ.get('TRUF_CONTAINER_CONFIG', str(DEFAULT_CONFIG))) + parser.add_argument('--manifest-sha256') + parser.add_argument('--require-worker-api', action='store_true') + parser.add_argument('--require-discovery-producers', action='store_true') + args = parser.parse_args(argv) + require_container(provisioning=args.action == 'provision') + if ( + (args.require_worker_api or args.require_discovery_producers) + and args.action not in ('health', 'status') + ): + raise RuntimeError('strict health requirements are restricted to health and status') + if args.action == 'import-snapshot': + if re.fullmatch(r'[0-9a-f]{64}', args.manifest_sha256 or '') is None: + raise RuntimeError('snapshot import requires --manifest-sha256 with the approved manifest digest') + if Path(args.config) != DEFAULT_CONFIG: + raise RuntimeError('snapshot import must begin with the image-owned default configuration') + elif args.manifest_sha256 is not None: + raise RuntimeError('--manifest-sha256 is restricted to snapshot import') + if args.action == 'provision': + provision() + return 0 + bind_config = args.action in ('initialize', 'run', 'import-secrets') + prepared = prepare_environment( + args.config, + validate_documents=args.action != 'import-secrets', + return_config_sha256=bind_config, + ) + if bind_config: + config, config_sha256 = prepared + else: + config = prepared + if args.action in ('health', 'status'): + print(json.dumps(health( + config, require_worker_api=args.require_worker_api, + require_discovery_producers=args.require_discovery_producers, + ), sort_keys=True), flush=True) + return 0 + if args.action == 'import-secrets': + import_secrets( + args.config, config, + expected_config_sha256=config_sha256, + ) + return 0 + + def request_shutdown(_signum, _frame): + global _shutdown_requested + _shutdown_requested = True + + previous = signal.signal(signal.SIGTERM, request_shutdown) + try: + if args.action == 'import-snapshot': + from container_import import import_snapshot + + return import_snapshot(sys.modules[__name__], args.manifest_sha256) + initialize( + Path(args.config), config, + expected_config_sha256=config_sha256, + ) + if _shutdown_requested or args.action == 'initialize': + return 0 + from runtime_document_io import load_managed_runtime_config + + current = load_managed_runtime_config(str(args.config)) + current_sha256 = current.config_sha256 + current = None + if current_sha256 != config_sha256: + raise RuntimeError('configuration changed after managed runtime validation') + command = _bootstrap_command( + 'supervisor', '--runtime-bootstrap-entrypoint', str(APP / 'supervisor.py'), + '--config', args.config, '--with-postgres', '--non-interactive', + '--autostart', '--no-dashboard', + ) + os.execv(sys.executable, command) + finally: + signal.signal(signal.SIGTERM, previous) + + +if __name__ == '__main__': + try: + raise SystemExit(main()) + except Exception as exc: + print('Container runtime rejected: ' + str(exc), file=sys.stderr, flush=True) + raise SystemExit(1) from None diff --git a/app/dashboard.py b/app/dashboard.py new file mode 100644 index 0000000..47cc820 --- /dev/null +++ b/app/dashboard.py @@ -0,0 +1,2581 @@ +import sys + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('dashboard could not disable bytecode writes') + +import argparse +from collections import Counter +from datetime import datetime, timedelta, timezone +import hashlib +import html +import ipaddress +import json +import os +import re +import shutil +import sqlite3 +import time + +import pandas as pd +import plotly.express as px +import streamlit as st + +from scanner_db import DB_FILENAME, get_database_path, queue_counts, sanitize_endpoint +from paths import apply_path_config, default_project_paths, resolve_optional_path +from db_backend import connect_postgres, connect_sqlite, database_url_from_env, is_postgres_url +from lifecycle_authority import CHILD_KIND_ENV, LifecycleAuthorityError, require_active_supervisor_child + + +DEFAULT_PATHS = default_project_paths() +DEFAULT_RESULTS_DIR = os.getenv('SCAN_RESULTS_DIR', DEFAULT_PATHS['results_dir']) +DEFAULT_QUEUE_DIR = DEFAULT_PATHS['queue_dir'] +DEFAULT_STATE_FILE = DEFAULT_PATHS['state_file'] +DEFAULT_LOG_DIR = DEFAULT_PATHS['log_dir'] +SOURCES = ['github', 'github_archive', 'github_archive_files', 'github_gists', 'gitlab', 'github_actions', 'gitlab_ci', 'docker', 'dockerhub', 'npm', 'pypi', 'package_git', 'huggingface', 'postman'] +RUNTIME_SOURCES = ['github', 'github_archive', 'github_archive_files', 'github_gists', 'gitlab', 'github_actions', 'gitlab_ci', 'huggingface', 'dockerhub', 'npm', 'package_git', 'postman', 'keychecks', 'pypi'] +VALIDATION_SERVICES = ['anthropic', 'aws', 'azure', 'deepseek', 'dockerhub', 'gcp', 'gemini', 'github', 'gitlab', 'groq', 'huggingface', 'kimi', 'openai', 'openrouter', 'provider_resolver', 'qwen', 'replicate', 'xai', 'zai'] +VALIDATION_STATUS_GROUPS = ['alive', 'dead', 'restricted', 'no_balance', 'no_context', 'limited', 'network', 'unknown'] +VALIDATION_STATUSES = [ + 'ALIVE', 'VALID', 'VALID_2FA', 'VALID_RATE_LIMITED', + 'VERTEX', 'BEDROCK', 'FOUNDRY', 'ADMIN', 'CANARY', + 'NO_BALANCE', 'NO_QUOTA', 'LIMITED_OR_NO_BALANCE', 'LIMITED_OR_QUOTA', + 'LIMITED', 'RATE_LIMITED', 'NO_CONTEXT', 'NO_TARGET', 'NO_TARGET_MODELS', 'NO_GENERATION_MODEL', + 'RESTRICTED', 'API_DISABLED', 'ACCESS_DENIED', 'QUARANTINED', 'DISABLED', + 'DEAD', 'INVALID', 'EXPIRED', 'LEAKED_REVOKED', 'INVALID_OR_REVOKED', + 'NETWORK', 'NETWORK_ERROR', 'UNKNOWN', +] +VALIDATION_ACCESS_TIERS = ['usable_llm', 'alive_unproven_llm', 'no_quota', 'quota_limited', 'missing_context', 'dead', 'restricted', 'network', 'unknown'] +VALIDATION_PRESETS = [ + 'Usable LLM keys', + 'Ever usable LLM keys', + 'Ever alive / LLM candidates', + 'Quota / no balance', + 'Alive but not proven LLM', + 'Unattributed alive', + 'All latest validation', + 'Latest rows', +] +UNATTRIBUTED = '(unattributed)' +ARCHIVE_INTERESTING_DETECTORS = { + 'googleai', 'googleaistudio', 'openai', 'anthropic', 'deepseek', 'openrouter', 'groq', + 'replicate', 'xai', 'huggingface', 'qwendashscope', 'kimimoonshot', 'zaiglm', 'github', 'githuboauth2', + 'gitlab', 'aws', 'gcp', 'gcpapplicationdefaultcredentials', 'azure', 'azureopenai', + 'azurecontainerregistry', 'dockerhub', 'npmtoken', 'sentrytoken', 'weightsandbiases', + 'twilio', 'scalewaykey', 'fastlypersonaltoken', 'sonarcloud', 'honeycomb', 'rapidapi', +} +DASHBOARD_DB_DIALECT = 'sqlite' +DASHBOARD_QUERY_TIMEOUT_SEC = 10 +MAX_LOG_TAIL_BYTES = 1024 * 1024 +LOOKUP_MAX_INPUT_CHARS = 65536 +LOOKUP_METADATA_LIMIT = 100000 +REPORTING_PRESETS = { + '1 hour': timedelta(hours=1), + '24 hours': timedelta(hours=24), + '7 days': timedelta(days=7), + '30 days': timedelta(days=30), +} +CORE_RUNTIME_SOURCES = { + 'result-ingester', 'jsonl-projector', 'janitor', 'worker-api', + 'github', 'gitlab', 'huggingface', 'dockerhub', 'package_git', 'keychecks', +} +FORBIDDEN_DASHBOARD_QUERY_COLUMNS = ( + 'raw_secret', 'raw_result_json', 'raw_finding_json', 'config_json', 'evidence_json', +) + + +def parse_args(): + parser = argparse.ArgumentParser(add_help=False) + parser.add_argument('--config') + parser.add_argument('--db') + parser.add_argument('--db-url') + parser.add_argument('--immutable-db', action='store_true') + parser.add_argument('--results-dir', default=DEFAULT_RESULTS_DIR) + parser.add_argument('--queue-dir', default=DEFAULT_QUEUE_DIR) + parser.add_argument('--state-file', default=DEFAULT_STATE_FILE) + parser.add_argument('--log-dir', default=DEFAULT_LOG_DIR) + parser.add_argument('--work-dir', default=DEFAULT_PATHS['work_dir']) + parser.add_argument('--keycheck-dir', default=DEFAULT_PATHS['keycheck_dir']) + parser.add_argument('--scan-limiter-db', default=os.path.join(DEFAULT_PATHS['state_dir'], 'scan_limiter.db')) + parser.add_argument('--max-active-scans', type=int, default=0) + args, _ = parser.parse_known_args(sys.argv[1:]) + if not args.config: + cwd_config = os.path.join(os.getcwd(), 'config.yaml') + if os.path.exists(cwd_config): + args.config = cwd_config + if args.config: + try: + import yaml + with open(args.config, 'r', encoding='utf-8') as f: + config = apply_path_config(yaml.safe_load(f) or {}, args.config) + global_config = config.get('global') or {} + args.results_dir = global_config.get('results_dir', args.results_dir) + args.queue_dir = global_config.get('queue_dir', args.queue_dir) + args.state_file = global_config.get('state_file', args.state_file) + args.log_dir = global_config.get('log_dir', args.log_dir) + args.work_dir = global_config.get('work_dir', args.work_dir) + args.keycheck_dir = global_config.get('keycheck_dir', args.keycheck_dir) + args.scan_limiter_db = global_config.get('scan_limiter_db', args.scan_limiter_db) + base_scans = int(global_config.get('max_active_scans', args.max_active_scans) or 0) + bonus_scans = max(0, min(1, int(global_config.get('opportunistic_scan_slots', 0) or 0))) + args.max_active_scans = base_scans + bonus_scans + if not args.db_url: + args.db_url = global_config.get('dashboard_db_url') or global_config.get('database_url') or database_url_from_env() + if not args.db: + args.db = resolve_optional_path(global_config.get('dashboard_db_path') or global_config.get('database_path'), global_config) + elif args.db: + args.db = resolve_optional_path(args.db, global_config) + args.immutable_db = bool(global_config.get('dashboard_immutable_db', args.immutable_db)) + except Exception as e: + if os.getenv(CHILD_KIND_ENV) == 'dashboard': + raise SystemExit(f'Managed dashboard config failed closed: {type(e).__name__}') from e + print(f'Unable to load dashboard config {args.config}: {e}') + return args + + +def resolve_db_path(args): + if args.db: + return args.db + return get_database_path(args.results_dir) + + +def resolve_db_url(args): + return args.db_url or database_url_from_env() + + +def connect_db(path, db_url=None, immutable=False): + global DASHBOARD_DB_DIALECT + if is_postgres_url(db_url): + conn = connect_postgres( + db_url, + connect_timeout_sec=3, + statement_timeout_ms=10000, + lock_timeout_ms=2000, + idle_in_transaction_timeout_ms=10000, + tcp_user_timeout_ms=5000, + ) + try: + conn.execute("SET application_name = 'truf-dashboard'") + conn.execute('SET default_transaction_read_only = on') + conn.commit() + except BaseException: + conn.close() + raise + DASHBOARD_DB_DIALECT = 'postgres' + return conn + if not path or not os.path.exists(path): + return None + conn = connect_sqlite(path, timeout_sec=5, read_only=True, immutable=immutable, check_same_thread=False) + conn.execute('PRAGMA query_only=ON') + conn.execute('PRAGMA busy_timeout=5000') + DASHBOARD_DB_DIALECT = 'sqlite' + return conn + + +def query_df(conn, sql, params=None): + if conn is None: + return pd.DataFrame() + if any(re.search(rf'\b{re.escape(column)}\b', str(sql), re.IGNORECASE) for column in FORBIDDEN_DASHBOARD_QUERY_COLUMNS): + st.error('Dashboard query refused because it requested a sensitive payload column.') + return pd.DataFrame() + raw_sqlite = getattr(conn, '_conn', None) if getattr(conn, 'is_sqlite', False) is True else None + if raw_sqlite is not None: + deadline = time.monotonic() + DASHBOARD_QUERY_TIMEOUT_SEC + raw_sqlite.set_progress_handler(lambda: 1 if time.monotonic() >= deadline else 0, 10000) + try: + rows = conn.execute(sql, params or []).fetchall() + if getattr(conn, 'is_postgres', False): + conn.commit() + return pd.DataFrame([dict(row) for row in rows]) + except Exception as e: + if getattr(conn, 'is_postgres', False): + try: + conn.rollback() + except Exception: + pass + st.error(f'Database query failed ({type(e).__name__}). Retry after database recovery.') + return pd.DataFrame() + finally: + if raw_sqlite is not None: + raw_sqlite.set_progress_handler(None, 0) + + +def table_columns(conn, table): + if conn is None: + return set() + raw_sqlite = getattr(conn, '_conn', None) if getattr(conn, 'is_sqlite', False) is True else None + if raw_sqlite is not None: + deadline = time.monotonic() + DASHBOARD_QUERY_TIMEOUT_SEC + raw_sqlite.set_progress_handler(lambda: 1 if time.monotonic() >= deadline else 0, 10000) + try: + columns = conn.table_columns(table) + if getattr(conn, 'is_postgres', False): + conn.commit() + return columns + except Exception: + if getattr(conn, 'is_postgres', False): + try: + conn.rollback() + except Exception: + pass + return set() + finally: + if raw_sqlite is not None: + raw_sqlite.set_progress_handler(None, 0) + + +def table_exists(conn, table): + if conn is None: + return False + raw_sqlite = getattr(conn, '_conn', None) if getattr(conn, 'is_sqlite', False) is True else None + if raw_sqlite is not None: + deadline = time.monotonic() + DASHBOARD_QUERY_TIMEOUT_SEC + raw_sqlite.set_progress_handler(lambda: 1 if time.monotonic() >= deadline else 0, 10000) + try: + exists = conn.table_exists(table) + if getattr(conn, 'is_postgres', False): + conn.commit() + return exists + except Exception: + if getattr(conn, 'is_postgres', False): + try: + conn.rollback() + except Exception: + pass + return False + finally: + if raw_sqlite is not None: + raw_sqlite.set_progress_handler(None, 0) + + +def json_extract_sql(column, path): + if DASHBOARD_DB_DIALECT == 'postgres': + parts = str(path or '').lstrip('$.').split('.') + pg_path = ','.join(part for part in parts if part) + return f"(NULLIF({column}, '')::jsonb #>> '{{{pg_path}}}')" + return f"json_extract({column}, '{path}')" + + +def today_sql(): + return "CURRENT_DATE::text" if DASHBOARD_DB_DIALECT == 'postgres' else "date('now')" + + +def finding_classification_sql(alias='f'): + prefix = f'{alias}.' if alias else '' + detector = f"LOWER(COALESCE({prefix}detector_name, ''))" + credential = f"LOWER(COALESCE({prefix}credential_kind, ''))" + return f''' + CASE + WHEN {detector} IN ('googleai', 'googleaistudio', 'openai', 'anthropic', 'deepseek', 'openrouter', 'groq', 'replicate', 'xai', 'huggingface', 'qwendashscope', 'qwen_dashscope', 'qwen', 'dashscope', 'kimimoonshot', 'moonshotai', 'moonshot', 'kimi', 'zaiglm', 'github', 'githuboauth2', 'gitlab', 'aws', 'gcp', 'gcpapplicationdefaultcredentials', 'azure', 'azureopenai', 'azurecontainerregistry') THEN 1 + WHEN {credential} IN ('google_ai_api_key', 'google_ai_studio_api_key', 'qwen_dashscope_api_key', 'kimi_moonshot_api_key', 'glm_api_key', 'dockerhub_pat') THEN 1 + ELSE 0 + END + ''' + + +def finding_service_sql(alias='f'): + prefix = f'{alias}.' if alias else '' + detector = f"LOWER(COALESCE({prefix}detector_name, ''))" + credential = f"LOWER(COALESCE({prefix}credential_kind, ''))" + return f''' + CASE + WHEN {detector} IN ('googleai', 'googleaistudio') THEN 'gemini' + WHEN {credential} IN ('google_ai_api_key', 'google_ai_studio_api_key') THEN 'gemini' + WHEN {detector} = 'openai' THEN 'openai' + WHEN {detector} = 'anthropic' THEN 'anthropic' + WHEN {detector} = 'deepseek' THEN 'deepseek' + WHEN {detector} = 'openrouter' THEN 'openrouter' + WHEN {detector} = 'groq' THEN 'groq' + WHEN {detector} = 'replicate' THEN 'replicate' + WHEN {detector} = 'xai' THEN 'xai' + WHEN {detector} = 'huggingface' THEN 'huggingface' + WHEN {detector} IN ('qwendashscope', 'qwen_dashscope', 'qwen', 'dashscope') THEN 'qwen' + WHEN {credential} = 'qwen_dashscope_api_key' THEN 'qwen' + WHEN {detector} IN ('kimimoonshot', 'moonshotai', 'moonshot', 'kimi') THEN 'kimi' + WHEN {credential} = 'kimi_moonshot_api_key' THEN 'kimi' + WHEN {detector} = 'zaiglm' THEN 'zai' + WHEN {credential} = 'glm_api_key' THEN 'zai' + WHEN {detector} IN ('github', 'githuboauth2') THEN 'github' + WHEN {detector} = 'gitlab' THEN 'gitlab' + WHEN {detector} = 'aws' THEN 'aws' + WHEN {detector} IN ('gcp', 'gcpapplicationdefaultcredentials') THEN 'gcp' + WHEN {detector} IN ('azure', 'azureopenai', 'azurecontainerregistry') THEN 'azure' + WHEN {credential} = 'dockerhub_pat' THEN 'dockerhub' + ELSE 'noise' + END + ''' + + +def finding_noise_reason_sql(alias='f'): + prefix = f'{alias}.' if alias else '' + detector = f"LOWER(COALESCE({prefix}detector_name, ''))" + secret_hash = f"COALESCE({prefix}secret_hash, '')" + return f''' + CASE + WHEN {finding_classification_sql(alias)} = 1 THEN 'keycheckable' + WHEN {detector} = 'dockerhub' THEN 'dockerhub_non_pat_or_missing_username' + WHEN {detector} IN ('uri', 'jdbc', 'postgres', 'mongodb', 'sqlserver') THEN 'connection_string_or_url' + WHEN {detector} IN ('box', 'circle', 'flatio', 'roaring', 'linkpreview') THEN 'generic_detector_noise' + WHEN {secret_hash} = '' THEN 'missing_secret_identity' + ELSE 'no_checker_or_generic_secret' + END + ''' + + +def keycheckable_findings_view_sql(): + return f''' + SELECT + f.id, f.source, f.query, f.detector_name, f.secret_hash, + {finding_classification_sql('f')} AS is_keycheckable, + {finding_service_sql('f')} AS validation_service, + {finding_noise_reason_sql('f')} AS noise_reason + FROM findings f + ''' + + +def keycheckable_backlog_sql(where=''): + where_clause = f'WHERE {where}' if where else '' + latest_keychecks = latest_keycheck_view_sql() + return f''' + WITH classified AS ({keycheckable_findings_view_sql()}), + checked_hashes AS ( + SELECT service, + COALESCE(NULLIF(secret_hash, ''), NULLIF(key_hash, ''), NULLIF(key_masked, '')) AS checked_hash, + MAX(CASE WHEN status_group = 'alive' THEN 1 ELSE 0 END) AS has_alive + FROM ({latest_keychecks}) + WHERE COALESCE(NULLIF(secret_hash, ''), NULLIF(key_hash, ''), NULLIF(key_masked, '')) IS NOT NULL + GROUP BY service, checked_hash + ) + SELECT source, query, validation_service, detector_name, + COUNT(*) AS raw_findings, + COUNT(DISTINCT NULLIF(secret_hash, '')) AS unique_secrets, + COUNT(DISTINCT CASE WHEN is_keycheckable = 1 THEN NULLIF(secret_hash, '') END) AS keycheckable_unique, + COUNT(DISTINCT CASE WHEN is_keycheckable = 0 THEN NULLIF(secret_hash, '') END) AS noise_unique, + COUNT(DISTINCT CASE WHEN is_keycheckable = 1 AND checked_hashes.checked_hash IS NOT NULL THEN NULLIF(secret_hash, '') END) AS checked_unique, + COUNT(DISTINCT CASE WHEN is_keycheckable = 1 AND checked_hashes.has_alive = 1 THEN NULLIF(secret_hash, '') END) AS alive_unique, + COUNT(DISTINCT CASE WHEN is_keycheckable = 1 AND checked_hashes.checked_hash IS NULL THEN NULLIF(secret_hash, '') END) AS pending_unique + FROM classified + LEFT JOIN checked_hashes + ON checked_hashes.service = classified.validation_service + AND checked_hashes.checked_hash = classified.secret_hash + {where_clause} + GROUP BY source, query, validation_service, detector_name + ''' + + +def scalar(conn, sql, params=None, default=0): + df = query_df(conn, sql, params) + if df.empty: + return default + return df.iloc[0, 0] + + +def current_queue_counts(queue_dir): + rows = [] + for source in SOURCES: + platform = 'docker' if source == 'dockerhub' else source + todo_file = os.path.join(queue_dir, f'todo_{platform}.txt') + checked_file = os.path.join(queue_dir, f'checked_{platform}.txt') + counts = queue_counts(todo_file, checked_file) + if counts['todo_count'] or counts['checked_count'] or os.path.exists(todo_file) or os.path.exists(checked_file): + rows.append({ + 'source': source, + 'todo': counts['todo_count'], + 'checked': counts['checked_count'], + 'todo_file': todo_file, + 'checked_file': checked_file, + }) + return pd.DataFrame(rows) + + +def queue_backlog(queue_dir): + df = current_queue_counts(queue_dir) + if df.empty: + return 0 + return int(df['todo'].fillna(0).sum()) + + +def disk_free_gb(path): + target = path if path and os.path.exists(path) else os.path.abspath(os.path.splitdrive(path or os.getcwd())[0] + os.sep) + try: + return round(shutil.disk_usage(target).free / (1024 ** 3), 2) + except OSError: + return 0 + + +def scan_slots_df(scan_limiter_db): + if not scan_limiter_db or not os.path.exists(scan_limiter_db): + return pd.DataFrame() + try: + uri = 'file:' + scan_limiter_db.replace('\\', '/') + '?mode=ro' + conn = sqlite3.connect(uri, uri=True) + conn.row_factory = sqlite3.Row + columns = {row[1] for row in conn.execute('PRAGMA table_info(scan_slots)').fetchall()} + slot_kind = "COALESCE(slot_kind, 'base') AS slot_kind" if 'slot_kind' in columns else "'base' AS slot_kind" + rows = pd.read_sql_query(f''' + SELECT owner_source, owner_pid, owner_thread, {slot_kind}, command, + acquired_at, updated_at + FROM scan_slots + ORDER BY acquired_at + ''', conn) + conn.close() + if not rows.empty: + now = datetime.now().timestamp() + rows['age_sec'] = (now - rows['acquired_at']).round(0).astype(int) + return rows + except Exception: + return pd.DataFrame() + + +def load_runner_state(state_file): + path = state_file + if not os.path.exists(path): + return {}, path + try: + with open(path, 'r', encoding='utf-8') as f: + return json.load(f), path + except (OSError, json.JSONDecodeError): + return {}, path + + +def parse_datetime(value): + if not value: + return None + text = str(value).strip() + for candidate in (text, text.replace('Z', '+00:00')): + try: + return datetime.fromisoformat(candidate) + except ValueError: + continue + return None + + +def human_age(value): + dt = parse_datetime(value) + if not dt: + return '' + now = datetime.now(dt.tzinfo) if dt.tzinfo else datetime.now() + seconds = max(0, int((now - dt).total_seconds())) + if seconds < 60: + return f'{seconds}s ago' + minutes = seconds // 60 + if minutes < 60: + return f'{minutes}m ago' + hours = minutes // 60 + if hours < 48: + return f'{hours}h {minutes % 60}m ago' + days = hours // 24 + return f'{days}d {hours % 24}h ago' + + +def read_tail(path, lines=80): + if not path or not os.path.exists(path): + return [] + max_lines = max(1, int(lines or 80)) + try: + with open(path, 'rb') as f: + f.seek(0, os.SEEK_END) + size = f.tell() + position = max(0, size - MAX_LOG_TAIL_BYTES) + f.seek(position) + data = f.read(MAX_LOG_TAIL_BYTES) + if position and data: + newline = data.find(b'\n') + data = data[newline + 1:] if newline >= 0 else b'' + text = data.decode('utf-8', errors='replace') + return text.splitlines(keepends=True)[-max_lines:] + except OSError: + return [] + + +def parse_supervisor_status(log_dir): + path = os.path.join(log_dir, 'supervisor.status.txt') + rows = [] + if not os.path.exists(path): + return pd.DataFrame(rows), path + headers = None + for line in read_tail(path, 80): + if '|' not in line or line.lstrip().startswith('-'): + continue + parts = [part.strip() for part in line.split('|')] + lowered = [part.lower() for part in parts] + if lowered and lowered[0] == 'source' and 'status' in lowered: + headers = lowered + continue + if headers and len(parts) >= len(headers): + item = dict(zip(headers, parts)) + rows.append({ + 'source': item.get('source', ''), + 'runtime_status': item.get('status', ''), + 'desired': item.get('desired', ''), + 'pid': item.get('pid', ''), + 'mode': item.get('mode', ''), + 'up': item.get('up', ''), + 'exit': item.get('exit', ''), + 'next': item.get('next', ''), + 'restarts': item.get('rs', item.get('restarts', '')), + 'auth': item.get('auth', ''), + }) + continue + if len(parts) < 9 or parts[0] in ('', 'Type `help` for commands. Use `command ` for full log/state paths.'): + continue + source = parts[0] + if source == 'source': + continue + has_auth_column = len(parts) >= 10 + rows.append({ + 'source': source, + 'runtime_status': parts[1], + 'desired': '', + 'pid': parts[2], + 'mode': parts[3], + 'up': parts[4], + 'exit': parts[5], + 'next': parts[6], + 'restarts': parts[7], + 'auth': parts[8] if has_auth_column else '', + }) + return pd.DataFrame(rows), path + + +def load_per_source_states(log_dir): + state_dir = os.path.normpath(os.path.join(log_dir, '..', 'state')) + rows = [] + for source in RUNTIME_SOURCES: + path = os.path.join(state_dir, f'runner_state_{source}.json') + if not os.path.exists(path): + continue + try: + with open(path, 'r', encoding='utf-8') as f: + state = json.load(f) + except (OSError, json.JSONDecodeError): + continue + item = (state.get('sources') or {}).get(source) or {} + rows.append({ + 'source': source, + 'last_query': item.get('last_query'), + 'state_status': item.get('last_status'), + 'last_started_at': item.get('last_started_at'), + 'last_started_age': human_age(item.get('last_started_at')), + 'last_completed_at': item.get('last_completed_at'), + 'last_completed_age': human_age(item.get('last_completed_at')), + 'cycles': item.get('cycles'), + 'last_scanned': item.get('last_scanned'), + 'last_auth': item.get('last_auth'), + }) + return pd.DataFrame(rows) + + +def latest_cycle_df(conn): + return query_df(conn, ''' + SELECT source, status AS db_status, query AS db_query, started_at AS db_started_at, + ended_at AS db_ended_at, scanned_count AS db_scanned, findings_count AS db_findings, + error_count AS db_errors, skipped_count AS db_skipped, duration_sec AS db_duration_sec + FROM source_cycles + WHERE id IN (SELECT MAX(id) FROM source_cycles GROUP BY source) + ''') + + +def runtime_source_health(conn, log_dir): + runtime_df, status_path = parse_supervisor_status(log_dir) + states_df = load_per_source_states(log_dir) + latest_df = latest_cycle_df(conn) + sources = sorted(set(RUNTIME_SOURCES) + | set(runtime_df['source'].tolist() if not runtime_df.empty else []) + | set(states_df['source'].tolist() if not states_df.empty else []) + | set(latest_df['source'].tolist() if not latest_df.empty else [])) + rows = pd.DataFrame({'source': sources}) + for df in (runtime_df, states_df, latest_df): + if not df.empty: + rows = rows.merge(df, on='source', how='left') + log_rows = [] + for source in sources: + log_path = os.path.join(log_dir, f'{source}.log') + log_rows.append({ + 'source': source, + 'log_path': log_path if os.path.exists(log_path) else '', + 'log_updated': datetime.fromtimestamp(os.path.getmtime(log_path)).isoformat(timespec='seconds') if os.path.exists(log_path) else '', + 'log_age': human_age(datetime.fromtimestamp(os.path.getmtime(log_path)).isoformat(timespec='seconds')) if os.path.exists(log_path) else '', + 'log_bytes': os.path.getsize(log_path) if os.path.exists(log_path) else 0, + }) + rows = rows.merge(pd.DataFrame(log_rows), on='source', how='left') + + def classify(row): + runtime = str(row.get('runtime_status') or '').lower() + if runtime in ('running', 'waiting', 'blocked', 'paused', 'backoff', 'done', 'failed', 'stopped', 'disabled'): + return runtime + if row.get('log_path'): + return 'has logs' + return 'no data' + + rows['health'] = rows.apply(classify, axis=1) + preferred = [ + 'source', 'health', 'runtime_status', 'desired', 'pid', 'mode', 'up', 'exit', 'next', 'restarts', 'auth', + 'state_status', 'last_query', 'last_started_age', 'last_completed_age', 'last_scanned', + 'db_status', 'db_query', 'db_scanned', 'db_findings', 'db_errors', 'log_age', 'log_bytes', + ] + existing = [column for column in preferred if column in rows.columns] + return rows[existing], status_path + + +def show_metrics(metrics): + cols = st.columns(len(metrics)) + for col, (label, value) in zip(cols, metrics): + col.metric(label, value) + + +def format_pct(value): + try: + return f'{float(value) * 100:.2f}%' + except (TypeError, ValueError): + return '0.00%' + + +def display_df(df, height=None): + if df.empty: + st.info('No data yet') + return + safe_secret_metadata = { + 'redacted_secret', 'secret_hash', 'detector_secret_hash', 'unique_secrets', + 'unique_secrets_count', 'secrets_found', 'credential_kind', + 'credential_confidence', 'required_context_missing', + } + blocked = [] + for column in df.columns: + name = str(column).lower() + if name in safe_secret_metadata: + continue + if ( + name.startswith('raw') + or name in {'config_json', 'evidence_json', 'credential', 'credential_value'} + or 'password' in name + or name == 'token' + or name.endswith('_token') + ): + blocked.append(column) + df = df.drop(columns=blocked, errors='ignore') + endpoint_columns = [ + column for column in df.columns + if str(column).lower() == 'endpoint' or str(column).lower().endswith('_endpoint') + ] + if endpoint_columns: + df = df.copy() + for column in endpoint_columns: + df[column] = df[column].map( + lambda value: value if pd.isna(value) else sanitize_endpoint(value) + ) + if 'resource' in df.columns: + adc_rows = pd.Series(False, index=df.index) + for detector_column in ('detector_name', 'detector'): + if detector_column in df.columns: + adc_rows |= df[detector_column].fillna('').astype(str).str.lower().eq( + 'gcpapplicationdefaultcredentials' + ) + if 'credential_kind' in df.columns: + adc_rows |= df['credential_kind'].fillna('').astype(str).str.lower().eq( + 'application_default_credentials' + ) + if adc_rows.any(): + df = df.copy() + df.loc[adc_rows, 'resource'] = '' + if 'redacted_secret' in df.columns: + df = df.copy() + df['redacted_secret'] = df['redacted_secret'].map( + lambda value: '***REDACTED***' if pd.notna(value) and str(value) else '' + ) + if height is None: + st.dataframe(df, width='stretch') + else: + st.dataframe(df, width='stretch', height=height) + + +def latest_keycheck_view_sql(): + return ''' + SELECT kr.*, + COALESCE(NULLIF(kr.key_hash, ''), NULLIF(kr.secret_hash, ''), NULLIF(kr.key_masked, '')) AS key_identity, + 1 AS latest_rank + FROM keycheck_current_state state + JOIN keycheck_results kr ON kr.id = state.last_result_id + ''' + + +def validation_access_tier_sql(alias='kr'): + prefix = f'{alias}.' if alias else '' + service = f"LOWER(COALESCE({prefix}service, ''))" + status = f"UPPER(COALESCE({prefix}status, ''))" + group = f"LOWER(COALESCE({prefix}status_group, ''))" + metadata = f"COALESCE({prefix}metadata_json, '{{}}')" + llm_probe_status = json_extract_sql(metadata, '$.llm_probe_status') + probe_status = json_extract_sql(metadata, '$.probe.status') + deployment_count = json_extract_sql(metadata, '$.deployment_count') + route_probe = json_extract_sql(metadata, '$.route_probe') + foundry_route_probe = json_extract_sql(metadata, '$.foundry_route_probe') + return f''' + CASE + WHEN {group} = 'no_balance' THEN 'no_quota' + WHEN {status} IN ('LIMITED', 'RATE_LIMITED', 'VALID_RATE_LIMITED') THEN 'quota_limited' + WHEN {service} = 'openai' AND {status} = 'ALIVE' THEN 'usable_llm' + WHEN {service} IN ('anthropic', 'deepseek', 'kimi', 'openrouter') AND {status} = 'VALID' THEN 'usable_llm' + WHEN {service} IN ('groq', 'qwen', 'xai') AND {status} = 'VALID' + AND {llm_probe_status} = 'GENERATION_OK' THEN 'usable_llm' + WHEN {service} = 'gemini' AND {status} = 'VALID' + AND {probe_status} = 'GENERATION_OK' THEN 'usable_llm' + WHEN {service} = 'gcp' AND {status} = 'VERTEX' THEN 'usable_llm' + WHEN {service} = 'aws' AND {status} = 'BEDROCK' THEN 'usable_llm' + WHEN {service} = 'azure' AND {status} = 'VALID' + AND COALESCE(CAST({deployment_count} AS INTEGER), 0) > 0 + AND {route_probe} = 'accepted_auth_route' THEN 'usable_llm' + WHEN {service} = 'azure' AND {status} = 'FOUNDRY' + AND {foundry_route_probe} = 'accepted' THEN 'usable_llm' + WHEN {group} = 'alive' THEN 'alive_unproven_llm' + WHEN {group} = 'limited' THEN 'quota_limited' + WHEN {group} = 'no_context' THEN 'missing_context' + ELSE {group} + END + ''' + + +def as_utc(value): + if not isinstance(value, datetime): + raise ValueError('Reporting timestamps must be datetime values.') + if value.tzinfo is None: + return value.replace(tzinfo=timezone.utc) + return value.astimezone(timezone.utc) + + +def reporting_window(preset, now=None, custom_start=None, custom_end=None): + end = as_utc(now or datetime.now(timezone.utc)) + if preset in REPORTING_PRESETS: + start = end - REPORTING_PRESETS[preset] + elif preset == 'Custom': + if custom_start is None or custom_end is None: + raise ValueError('Choose both custom UTC timestamps.') + start = as_utc(custom_start) + end = as_utc(custom_end) + else: + raise ValueError('Unknown reporting window.') + if end <= start: + raise ValueError('The end of the reporting window must be later than the start.') + return start, end + + +def utc_parameter(value): + return as_utc(value).isoformat(timespec='seconds') + + +def _secret_from_payload(value, depth=0): + if depth > 4: + return None + if isinstance(value, dict): + lowered = {str(key).lower(): item for key, item in value.items()} + for name in ('rawv2', 'raw_v2', 'raw', 'secret', 'api_key', 'apikey', 'access_token'): + candidate = lowered.get(name) + if isinstance(candidate, str) and candidate.strip(): + return candidate.strip() + for item in value.values(): + candidate = _secret_from_payload(item, depth + 1) + if candidate: + return candidate + elif isinstance(value, list): + for item in value[:100]: + candidate = _secret_from_payload(item, depth + 1) + if candidate: + return candidate + return None + + +def _credential_like(text): + lowered = text.lower() + known_prefixes = ( + 'sk-', 'sk_', 'ghp_', 'github_pat_', 'glpat-', 'hf_', 'gsk_', 'xai-', + 'sk-or-', 'r8_', 'npm_', 'akia', 'asia', 'aiza', 'eyj', + ) + if lowered.startswith(known_prefixes): + return True + if len(text) < 20 or any(character.isspace() for character in text): + return False + if text.startswith(('http://', 'https://', 'git@')) or any(character in text for character in ('/', '\\', '@')): + return False + if re.fullmatch(r'[0-9a-fA-F]{40}', text) or re.fullmatch( + r'[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}', text): + return False + return bool(re.fullmatch(r'[A-Za-z0-9_.:+\-=]+', text)) + + +def normalize_lookup(text): + value = str(text or '').strip() + if not value: + raise ValueError('Paste a key, hash, finding ID/UID, target, path, or commit.') + if len(value) > LOOKUP_MAX_INPUT_CHARS: + raise ValueError('Lookup input is too large.') + + extracted = None + if value[:1] in ('{', '['): + try: + extracted = _secret_from_payload(json.loads(value)) + except (TypeError, ValueError, json.JSONDecodeError): + extracted = None + if not extracted and ('\n' in value or '\r' in value): + match = re.search(r'(?im)^\s*(?:rawv2|raw|secret|api[_ -]?key)\s*[:=]\s*["\']?([^\s"\']+)', value) + extracted = match.group(1) if match else None + if extracted: + return {'kind': 'digest', 'digest': hashlib.sha256(extracted.encode('utf-8')).hexdigest()} + + finding_id = re.fullmatch(r'(?i)(?:finding\s*[:#]?\s*)?(\d+)', value) + if finding_id: + return {'kind': 'finding_id', 'finding_id': int(finding_id.group(1))} + if re.fullmatch(r'[0-9a-fA-F]{64}', value): + return {'kind': 'digest', 'digest': value.lower()} + if re.fullmatch(r'[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}', value): + return {'kind': 'identity', 'identity': value} + if _credential_like(value): + return {'kind': 'digest', 'digest': hashlib.sha256(value.encode('utf-8')).hexdigest()} + if '\n' in value or '\r' in value: + raise ValueError('Could not identify a credential in the pasted finding.') + if len(value) < 3: + raise ValueError('Metadata lookup needs at least three characters.') + if len(value) > 512: + raise ValueError('Metadata lookup is limited to 512 characters.') + return {'kind': 'metadata', 'text': value} + + +def escaped_like(value): + return '%' + str(value).replace('!', '!!').replace('%', '!%').replace('_', '!_') + '%' + + +def period_summary_df(conn, start, end): + access_tier = validation_access_tier_sql('r') + return query_df(conn, f''' + WITH bounds AS ( + SELECT CAST(? AS timestamptz) AS start_at, CAST(? AS timestamptz) AS end_at + ), period_scans AS ( + SELECT COUNT(*) AS scans, + COALESCE(SUM(ts.findings_count), 0) AS findings, + COALESCE(SUM(ts.error_count), 0) AS errors + FROM target_scans ts CROSS JOIN bounds b + WHERE ts.ended_at::timestamptz >= b.start_at + AND ts.ended_at::timestamptz < b.end_at + ), period_checks AS ( + SELECT COUNT(*) AS checks + FROM keycheck_results kr CROSS JOIN bounds b + WHERE kr.checked_at::timestamptz >= b.start_at + AND kr.checked_at::timestamptz < b.end_at + ), period_discovery AS ( + SELECT COALESCE(SUM(sc.queued_new_count), 0) AS queued_new, + COALESCE(SUM(sc.queued_updated_count), 0) AS queued_updated + FROM source_cycles sc CROSS JOIN bounds b + WHERE sc.started_at::timestamptz >= b.start_at + AND sc.started_at::timestamptz < b.end_at + ), ranked_alive AS ( + SELECT r.id, r.credential_id, r.checked_at, + ROW_NUMBER() OVER ( + PARTITION BY r.credential_id + ORDER BY r.checked_at::timestamptz, r.id + ) AS alive_rank + FROM keycheck_results r + WHERE r.credential_id IS NOT NULL + AND r.status_group = 'alive' + ), ranked_usable AS ( + SELECT r.id, r.credential_id, r.checked_at, + ROW_NUMBER() OVER ( + PARTITION BY r.credential_id + ORDER BY r.checked_at::timestamptz, r.id + ) AS usable_rank + FROM keycheck_results r + WHERE r.credential_id IS NOT NULL + AND ({access_tier}) = 'usable_llm' + ), period_new_alive AS ( + SELECT COUNT(*) AS credentials + FROM ranked_alive r CROSS JOIN bounds b + WHERE r.alive_rank = 1 + AND r.checked_at::timestamptz >= b.start_at + AND r.checked_at::timestamptz < b.end_at + ), period_new_usable AS ( + SELECT COUNT(*) AS credentials + FROM ranked_usable r CROSS JOIN bounds b + WHERE r.usable_rank = 1 + AND r.checked_at::timestamptz >= b.start_at + AND r.checked_at::timestamptz < b.end_at + ) + SELECT ps.scans, ps.findings, ps.errors, pc.checks, + pd.queued_new, pd.queued_updated, + pna.credentials AS new_alive, pnu.credentials AS new_usable, + (SELECT COUNT(*) FROM keycheck_current_state WHERE status_group = 'alive') AS alive_now, + (SELECT COUNT(*) + FROM keycheck_current_state s + JOIN keycheck_results r ON r.id = s.last_result_id + WHERE s.status_group = 'alive' AND ({access_tier}) = 'usable_llm') AS usable_llm_now, + (SELECT COUNT(*) FROM target_queue WHERE status IN ('pending', 'deferred')) AS target_backlog, + (SELECT COUNT(*) FROM keycheck_candidates WHERE state IN ('pending', 'deferred', 'leased')) AS keycheck_backlog + FROM period_scans ps CROSS JOIN period_checks pc CROSS JOIN period_discovery pd + CROSS JOIN period_new_alive pna CROSS JOIN period_new_usable pnu + ''', [utc_parameter(start), utc_parameter(end)]) + + +def source_activity_df(conn, start, end): + return query_df(conn, ''' + WITH bounds AS ( + SELECT CAST(? AS timestamptz) AS start_at, CAST(? AS timestamptz) AS end_at + ), scan_activity AS ( + SELECT ts.source, COUNT(*) AS scans, + COALESCE(SUM(ts.findings_count), 0) AS findings, + COALESCE(SUM(ts.error_count), 0) AS errors, + MAX(ts.ended_at) AS latest_scan + FROM target_scans ts CROSS JOIN bounds b + WHERE ts.ended_at::timestamptz >= b.start_at + AND ts.ended_at::timestamptz < b.end_at + GROUP BY ts.source + ), cycle_activity AS ( + SELECT sc.source, + COALESCE(SUM(sc.queued_new_count), 0) AS queued_new, + COALESCE(SUM(sc.queued_updated_count), 0) AS queued_updated, + MAX(sc.started_at) AS latest_cycle + FROM source_cycles sc CROSS JOIN bounds b + WHERE sc.started_at::timestamptz >= b.start_at + AND sc.started_at::timestamptz < b.end_at + GROUP BY sc.source + ), sources AS ( + SELECT source FROM scan_activity + UNION + SELECT source FROM cycle_activity + ) + SELECT s.source, COALESCE(sa.scans, 0) AS scans, + COALESCE(sa.findings, 0) AS findings, + COALESCE(sa.errors, 0) AS errors, + COALESCE(ca.queued_new, 0) AS queued_new, + COALESCE(ca.queued_updated, 0) AS queued_updated, + GREATEST(sa.latest_scan, ca.latest_cycle) AS latest + FROM sources s + LEFT JOIN scan_activity sa ON sa.source = s.source + LEFT JOIN cycle_activity ca ON ca.source = s.source + ORDER BY findings DESC, scans DESC, s.source + LIMIT 50 + ''', [utc_parameter(start), utc_parameter(end)]) + + +def new_alive_breakdown_df(conn, start, end): + access_tier = validation_access_tier_sql('kr') + return query_df(conn, f''' + WITH bounds AS ( + SELECT CAST(? AS timestamptz) AS start_at, CAST(? AS timestamptz) AS end_at + ), ranked_alive AS ( + SELECT kr.credential_id AS id, kr.service, kr.status, kr.checked_at, + ROW_NUMBER() OVER ( + PARTITION BY kr.credential_id + ORDER BY kr.checked_at::timestamptz, kr.id + ) AS alive_rank, + {access_tier} AS access_tier + FROM keycheck_results kr + WHERE kr.credential_id IS NOT NULL + AND kr.status_group = 'alive' + ), recent AS ( + SELECT ra.id, ra.service, ra.status, ra.checked_at, ra.access_tier + FROM ranked_alive ra CROSS JOIN bounds b + WHERE ra.alive_rank = 1 + AND ra.checked_at::timestamptz >= b.start_at + AND ra.checked_at::timestamptz < b.end_at + ) + SELECT r.service, r.status, r.access_tier, + COALESCE(origin.source, '(unknown)') AS source, + COUNT(*) AS credentials, MAX(r.checked_at) AS latest + FROM recent r + LEFT JOIN LATERAL ( + SELECT kc.source + FROM keycheck_candidates kc + WHERE kc.credential_id = r.id + ORDER BY kc.id + LIMIT 1 + ) origin ON TRUE + GROUP BY r.service, r.status, r.access_tier, origin.source + ORDER BY credentials DESC, r.service, r.status, r.access_tier, source + LIMIT 100 + ''', [utc_parameter(start), utc_parameter(end)]) + + +def pipeline_snapshot_df(conn): + return query_df(conn, ''' + SELECT bundle_items, projection_items, keycheck_items, quarantine_items, + bundle_bytes, projection_bytes, keycheck_bytes, quarantine_bytes, updated_at + FROM pipeline_capacity + WHERE id = 1 + ''') + + +def lookup_findings_df(conn, lookup, limit=200): + kind = lookup.get('kind') + params = [] + if kind == 'finding_id': + where = 'f.id = ?' + params.append(int(lookup['finding_id'])) + elif kind == 'digest': + where = '''( + f.secret_hash = ? OR f.detector_secret_hash = ? + OR f.finding_uid = ? OR f.finding_fingerprint = ? + )''' + params.extend([lookup['digest']] * 4) + elif kind == 'identity': + where = '(f.finding_uid = ? OR f.finding_fingerprint = ?)' + params.extend([lookup['identity']] * 2) + elif kind == 'metadata': + pattern = escaped_like(lookup['text']) + where = ''' + f.id >= (SELECT GREATEST(COALESCE(MAX(id), 0) - ?, 0) FROM findings) + AND ( + f.source ILIKE ? ESCAPE '!' OR f.query ILIKE ? ESCAPE '!' + OR f.target ILIKE ? ESCAPE '!' OR f.detector_name ILIKE ? ESCAPE '!' + OR f.redacted_secret ILIKE ? ESCAPE '!' OR f.file_path ILIKE ? ESCAPE '!' + OR f.commit_hash ILIKE ? ESCAPE '!' OR f.provider ILIKE ? ESCAPE '!' + OR f.credential_kind ILIKE ? ESCAPE '!' OR ts.package_name ILIKE ? ESCAPE '!' + OR ts.package_version ILIKE ? ESCAPE '!' OR ts.package_filename ILIKE ? ESCAPE '!' + ) + ''' + params.extend([LOOKUP_METADATA_LIMIT, *([pattern] * 12)]) + else: + return pd.DataFrame() + params.append(max(1, min(int(limit or 200), 500))) + return query_df(conn, f''' + SELECT f.id AS finding_id, f.finding_uid, f.created_at AS found_at, + f.source, f.query, f.target, f.detector_name, f.provider, + f.file_path, f.line_number, f.commit_hash, f.secret_hash, + ts.package_name, ts.package_version, ts.package_filename, ts.package_type + FROM findings f + LEFT JOIN target_scans ts ON ts.id = f.target_scan_id + WHERE {where} + ORDER BY f.id DESC + LIMIT ? + ''', params) + + +def lookup_status_df(conn, lookup, findings=None, limit=100): + findings = findings if isinstance(findings, pd.DataFrame) else pd.DataFrame() + hashes = set() + finding_ids = set() + if not findings.empty: + if 'secret_hash' in findings: + hashes.update(str(value) for value in findings['secret_hash'].dropna() if str(value)) + if 'finding_id' in findings: + finding_ids.update(int(value) for value in findings['finding_id'].dropna()) + if lookup.get('kind') == 'digest': + hashes.add(lookup['digest']) + if lookup.get('kind') == 'finding_id': + finding_ids.add(int(lookup['finding_id'])) + + clauses = [] + params = [] + if hashes: + values = sorted(hashes) + placeholders = ','.join('?' for _ in values) + clauses.append(f"(r.secret_hash IN ({placeholders}) OR r.key_hash IN ({placeholders}))") + params.extend(values) + params.extend(values) + clauses.append(f'''EXISTS ( + SELECT 1 FROM keycheck_candidates kc + WHERE kc.credential_id = s.credential_id AND kc.secret_hash IN ({placeholders}) + )''') + params.extend(values) + if finding_ids: + values = sorted(finding_ids) + placeholders = ','.join('?' for _ in values) + clauses.append(f'r.finding_id IN ({placeholders})') + params.extend(values) + clauses.append(f'''EXISTS ( + SELECT 1 FROM keycheck_candidates kc + WHERE kc.credential_id = s.credential_id AND kc.finding_id IN ({placeholders}) + )''') + params.extend(values) + if lookup.get('kind') == 'metadata': + pattern = escaped_like(lookup['text']) + clauses.append('''( + r.key_masked ILIKE ? ESCAPE '!' OR r.source ILIKE ? ESCAPE '!' + OR r.query ILIKE ? ESCAPE '!' OR r.target ILIKE ? ESCAPE '!' + OR r.detector_name ILIKE ? ESCAPE '!' + )''') + params.extend([pattern] * 5) + if not clauses: + return pd.DataFrame() + + access_tier = validation_access_tier_sql('r') + params.append(max(1, min(int(limit or 100), 200))) + return query_df(conn, f''' + SELECT s.credential_id, s.service, s.status, s.status_group, + {access_tier} AS access_tier, s.checked_at, s.recheck_after, + r.key_masked, + COALESCE(r.source, origin.source) AS source, + COALESCE(r.query, origin.query) AS query, + COALESCE(r.target, origin.target) AS target, + COALESCE(r.finding_id, origin.finding_id) AS finding_id, + COALESCE(r.detector_name, origin.detector_name) AS detector_name, + COALESCE(r.found_at, origin.found_at) AS found_at + FROM keycheck_current_state s + JOIN keycheck_results r ON r.id = s.last_result_id + LEFT JOIN LATERAL ( + SELECT kc.source, kc.query, kc.target, kc.finding_id, kc.detector_name, kc.found_at + FROM keycheck_candidates kc + WHERE kc.credential_id = s.credential_id + ORDER BY (kc.finding_id IS NOT NULL) DESC, kc.id DESC + LIMIT 1 + ) origin ON TRUE + WHERE {' OR '.join(f'({clause})' for clause in clauses)} + ORDER BY s.checked_at DESC, s.credential_id + LIMIT ? + ''', params) + + +def dashboard_search_is_safe(text): + text = str(text or '').strip() + return bool( + not text + or re.fullmatch(r'[0-9a-fA-F]{64}', text) + or len(text) < 20 + or any(character.isspace() for character in text) + or '***' in text + or '...' in text + ) + + +def finding_metadata_search(conn, text, limit=500): + text = str(text or '').strip() + if not text or not dashboard_search_is_safe(text): + return pd.DataFrame() + digest = text.lower() if re.fullmatch(r'[0-9a-fA-F]{64}', text) else '' + like = f'%{text}%' + hash_clause = 'f.secret_hash = ? OR' if digest else '' + params = ([digest] if digest else []) + [like] * 9 + [int(limit)] + return query_df(conn, f''' + SELECT f.id AS finding_id, f.created_at, f.source, f.query, f.target, + f.detector_name, f.redacted_secret, f.secret_hash, + f.file_path, f.line_number, f.commit_hash, f.source_timestamp, + ts.package_name, ts.package_version, ts.package_filename, ts.package_type, + f.target_scan_id, f.cycle_id, f.run_id + FROM ( + SELECT id, created_at, source, query, target, detector_name, redacted_secret, + secret_hash, file_path, line_number, commit_hash, source_timestamp, + target_scan_id, cycle_id, run_id + FROM findings + ORDER BY id DESC + LIMIT 50000 + ) f + LEFT JOIN target_scans ts ON ts.id = f.target_scan_id + WHERE {hash_clause} f.redacted_secret LIKE ? + OR f.secret_hash LIKE ? + OR f.detector_name LIKE ? + OR f.source LIKE ? + OR f.query LIKE ? + OR f.target LIKE ? + OR f.file_path LIKE ? + OR f.commit_hash LIKE ? + OR ts.package_name LIKE ? + ORDER BY f.id DESC + LIMIT ? + ''', params) + + +def github_archive_yield(conn, limit=2000): + rows = query_df(conn, ''' + SELECT id, target, ended_at, findings_count, verified_findings_count + FROM target_scans + WHERE source = 'github_archive' AND status = 'found' + ORDER BY id DESC + LIMIT ? + ''', [int(limit or 2000)]) + if rows.empty: + return {}, pd.DataFrame(), pd.DataFrame() + scan_ids = [int(value) for value in rows['id'].dropna().tolist()] + finding_rows = query_df(conn, f''' + SELECT target_scan_id, detector_name + FROM findings + WHERE target_scan_id IN ({','.join('?' for _ in scan_ids)}) + ''', scan_ids) if scan_ids else pd.DataFrame() + findings_by_scan = {} + if not finding_rows.empty: + for scan_id, group in finding_rows.groupby('target_scan_id'): + findings_by_scan[int(scan_id)] = [str(value or '') for value in group['detector_name'].tolist()] + detectors = Counter() + target_rows = [] + total_findings = 0 + total_verified = 0 + interesting_rows = 0 + for _, row in rows.iterrows(): + findings_count = int(row.get('findings_count') or 0) + verified_count = int(row.get('verified_findings_count') or 0) + total_findings += findings_count + total_verified += verified_count + target_interesting = 0 + for detector in findings_by_scan.get(int(row.get('id') or 0), []): + detectors[detector] += 1 + if detector.lower() in ARCHIVE_INTERESTING_DETECTORS: + interesting_rows += 1 + target_interesting += 1 + target_rows.append({ + 'target_scan_id': int(row.get('id') or 0), + 'target': row.get('target'), + 'ended_at': row.get('ended_at'), + 'findings': findings_count, + 'interesting_findings': target_interesting, + 'verified_findings': verified_count, + }) + detector_df = pd.DataFrame([ + { + 'detector': detector, + 'rows': count, + 'kind': 'interesting' if str(detector).lower() in ARCHIVE_INTERESTING_DETECTORS else 'noise_or_generic', + } + for detector, count in detectors.most_common(50) + ]) + target_df = pd.DataFrame(target_rows).sort_values(['interesting_findings', 'findings'], ascending=False).head(100) + summary = { + 'found_targets': int(len(rows)), + 'raw_findings': int(total_findings), + 'interesting_findings': int(interesting_rows), + 'noise_or_generic_findings': int(max(0, total_findings - interesting_rows)), + 'verified_findings': int(total_verified), + } + return summary, detector_df, target_df + + +def keycheck_summary_df(keycheck_dir): + path = os.path.join(keycheck_dir, 'summary.tsv') + if not os.path.exists(path): + return pd.DataFrame(), path + try: + df = pd.read_csv(path, sep='\t') + except Exception: + return pd.DataFrame(), path + if 'updated_at' in df.columns: + df['file_updated_at'] = df['updated_at'] + return df, path + + +def recent_keycheck_db_summary(conn, limit=10000): + result_source_expr = json_extract_sql('metadata_json', '$.result_source') + rows = query_df(conn, ''' + SELECT id, service, status_group, status, checked_at, created_at, + {result_source_expr} AS result_source + FROM keycheck_results + ORDER BY id DESC + LIMIT ? + '''.format(result_source_expr=result_source_expr), [int(limit or 10000)]) + if rows.empty: + return pd.DataFrame() + rows['result_source'] = rows['result_source'].fillna('api_check') + grouped = rows.groupby('service', dropna=False).agg( + db_recent_rows=('id', 'count'), + db_latest_id=('id', 'max'), + db_latest_checked=('checked_at', 'max'), + db_latest_created=('created_at', 'max'), + db_cached_rows=('result_source', lambda s: int((s == 'cached_status').sum())), + db_alive_rows=('status_group', lambda s: int((s == 'alive').sum())), + ).reset_index() + return grouped + + +def keycheck_db_metric_summary(keycheck_dir): + rows = [] + if not keycheck_dir or not os.path.isdir(keycheck_dir): + return pd.DataFrame(rows) + for service in sorted(os.listdir(keycheck_dir)): + path = os.path.join(keycheck_dir, service, 'db_write_metrics.jsonl') + if not os.path.exists(path): + continue + ok = failed = cached = total = 0 + latest = '' + for line in read_tail(path, 2000): + try: + item = json.loads(line) + except ValueError: + continue + total += 1 + ok += 1 if item.get('ok') else 0 + failed += 0 if item.get('ok') else 1 + cached += 1 if item.get('result_source') == 'cached_status' else 0 + latest = max(latest, str(item.get('created_at') or '')) + rows.append({ + 'service': service, + 'metric_rows_tail': total, + 'db_write_ok_tail': ok, + 'db_write_failed_tail': failed, + 'cached_occurrence_tail': cached, + 'db_metric_latest': latest, + }) + return pd.DataFrame(rows) + + +def keycheck_pipeline_health(log_dir): + path = os.path.join(log_dir, 'keychecks.log') + lines = read_tail(path, 600) + if not lines: + return pd.DataFrame(), path + latest_start = '' + latest_exit = '' + current_service = '' + last_service = '' + db_locks = 0 + processed = 0 + skipped = 0 + for line in lines: + text = line.strip() + if text.startswith('=== supervisor start') and 'source=keychecks' in text: + latest_start = text.replace('=== supervisor start ', '').split(' source=', 1)[0] + current_service = '' + processed = 0 + skipped = 0 + db_locks = 0 + elif text.startswith('=== supervisor exit') and 'source=keychecks' in text: + latest_exit = text.replace('=== supervisor exit ', '').split(' source=', 1)[0] + current_service = '' + elif ': ' in text and 'keycheckers' in text and '.py' in text: + current_service = text.split(':', 1)[0] + last_service = current_service + elif 'Observability DB locked' in text or 'database is locked' in text: + db_locks += 1 + elif text.startswith('Done. Processed=') or text.startswith('Processed='): + numbers = re.findall(r'(?:Processed|skipped)=([0-9]+)', text) + if numbers: + processed += int(numbers[0]) + if len(numbers) > 1: + skipped += int(numbers[1]) + row = { + 'latest_start': latest_start, + 'latest_exit': latest_exit, + 'current_or_last_service': current_service or last_service, + 'db_lock_messages_tail': db_locks, + 'processed_tail': processed, + 'skipped_tail': skipped, + 'log_path': path, + } + return pd.DataFrame([row]), path + + +def _integer(value): + try: + if pd.isna(value): + return 0 + return int(value) + except (TypeError, ValueError): + return 0 + + +def _submit_dashboard_lookup(): + submitted = st.session_state.get('dashboard_lookup_input', '') + try: + st.session_state['dashboard_lookup_request'] = normalize_lookup(submitted) + st.session_state['dashboard_lookup_error'] = '' + except ValueError as exc: + st.session_state.pop('dashboard_lookup_request', None) + st.session_state['dashboard_lookup_error'] = str(exc) + st.session_state['dashboard_lookup_input'] = '' + + +def _clear_dashboard_lookup(): + st.session_state.pop('dashboard_lookup_request', None) + st.session_state.pop('dashboard_lookup_error', None) + st.session_state['dashboard_lookup_input'] = '' + + +def _dashboard_styles(): + st.markdown(''' + + ''', unsafe_allow_html=True) + + +def _metric_grid(metrics): + cards = ''.join( + '
    ' + f'
    {html.escape(str(label))}
    ' + f'
    {html.escape(str(value))}
    ' + '
    ' + for label, value in metrics + ) + st.markdown(f'
    {cards}
    ', unsafe_allow_html=True) + + +def page_simple_dashboard(conn, log_dir, work_dir, scan_limiter_db, max_active_scans): + _dashboard_styles() + title_col, refresh_col = st.columns([8, 1]) + with title_col: + st.title('TRUF') + st.caption('Scanner and credential status / PostgreSQL read-only / UTC') + with refresh_col: + if st.button('Refresh', width='stretch'): + st.rerun() + + st.subheader('Find a credential or finding') + with st.form('dashboard_lookup_form', clear_on_submit=False, border=False): + input_col, submit_col = st.columns([8, 1]) + with input_col: + st.text_input( + 'Lookup', + key='dashboard_lookup_input', + placeholder='Paste a key, SHA-256, finding ID/UID, URL, path, or commit', + label_visibility='collapsed', + ) + with submit_col: + st.form_submit_button('Find', width='stretch', on_click=_submit_dashboard_lookup) + st.caption('Credential-like input is hashed immediately, cleared, and never queried as raw text.') + + lookup_error = st.session_state.get('dashboard_lookup_error') + lookup = st.session_state.get('dashboard_lookup_request') + if lookup_error: + st.warning(lookup_error) + if lookup: + findings = lookup_findings_df(conn, lookup) + statuses = lookup_status_df(conn, lookup, findings) + result_col, clear_col = st.columns([8, 1]) + with result_col: + st.markdown(f'**Lookup result:** {len(statuses)} current status row(s), {len(findings)} origin(s)') + with clear_col: + st.button('Clear', width='stretch', on_click=_clear_dashboard_lookup) + if statuses.empty and findings.empty: + st.info('No current status or finding origin matched this lookup.') + if not statuses.empty: + st.markdown('**Current status**') + status_columns = [ + 'service', 'status', 'status_group', 'access_tier', 'checked_at', + 'source', 'query', 'target', 'detector_name', 'finding_id', 'found_at', + ] + display_df(statuses[[column for column in status_columns if column in statuses]], height=260) + if not findings.empty: + st.markdown('**Origins**') + origins = findings.copy() + origins['identity'] = origins['secret_hash'].map( + lambda value: (str(value)[:12] + '...') if pd.notna(value) and str(value) else '' + ) + origin_columns = [ + 'finding_id', 'identity', 'found_at', 'source', 'query', 'target', + 'detector_name', 'provider', 'file_path', 'line_number', 'commit_hash', + 'package_name', 'package_version', 'package_filename', 'finding_uid', + ] + display_df(origins[[column for column in origin_columns if column in origins]], height=360) + + st.divider() + st.subheader('Activity') + period = st.radio( + 'Reporting window', + [*REPORTING_PRESETS, 'Custom'], + index=1, + horizontal=True, + label_visibility='collapsed', + ) + now = datetime.now(timezone.utc).replace(microsecond=0) + custom_start = None + custom_end = None + if period == 'Custom': + default_start = now - timedelta(hours=24) + start_date_col, start_time_col, end_date_col, end_time_col = st.columns(4) + with start_date_col: + start_date = st.date_input('Start date (UTC)', value=default_start.date()) + with start_time_col: + start_time = st.time_input('Start time (UTC)', value=default_start.time()) + with end_date_col: + end_date = st.date_input('End date (UTC)', value=now.date()) + with end_time_col: + end_time = st.time_input('End time (UTC)', value=now.time()) + custom_start = datetime.combine(start_date, start_time, tzinfo=timezone.utc) + custom_end = datetime.combine(end_date, end_time, tzinfo=timezone.utc) + try: + start, end = reporting_window(period, now=now, custom_start=custom_start, custom_end=custom_end) + except ValueError as exc: + st.error(str(exc)) + return + st.caption(f'{start:%Y-%m-%d %H:%M} to {end:%Y-%m-%d %H:%M} UTC') + + summary = period_summary_df(conn, start, end) + summary_row = summary.iloc[0] if not summary.empty else {} + _metric_grid([ + ('Scans', _integer(summary_row.get('scans', 0))), + ('Findings', _integer(summary_row.get('findings', 0))), + ('Errors', _integer(summary_row.get('errors', 0))), + ('Checks', _integer(summary_row.get('checks', 0))), + ('New targets', _integer(summary_row.get('queued_new', 0))), + ('Updated rescans', _integer(summary_row.get('queued_updated', 0))), + ('New alive', _integer(summary_row.get('new_alive', 0))), + ('New usable', _integer(summary_row.get('new_usable', 0))), + ]) + _metric_grid([ + ('Alive now', _integer(summary_row.get('alive_now', 0))), + ('Usable LLM now', _integer(summary_row.get('usable_llm_now', 0))), + ('Target queue now', _integer(summary_row.get('target_backlog', 0))), + ('Keycheck queue now', _integer(summary_row.get('keycheck_backlog', 0))), + ]) + + source_activity = source_activity_df(conn, start, end) + new_alive = new_alive_breakdown_df(conn, start, end) + source_col, alive_col = st.columns([1.2, 1], gap='large') + with source_col: + st.markdown('**Sources in period**') + if not source_activity.empty: + source_activity = source_activity.copy() + source_activity['findings / scan'] = source_activity.apply( + lambda row: round(_integer(row.get('findings')) / max(1, _integer(row.get('scans'))), 2), + axis=1, + ) + display_df(source_activity, height=340) + with alive_col: + st.markdown('**Alive discovered in period**') + display_df(new_alive, height=340) + + st.divider() + st.subheader('Runtime now') + st.caption('Current state is not restricted by the reporting window.') + slots = scan_slots_df(scan_limiter_db) + pipeline = pipeline_snapshot_df(conn) + pipeline_row = pipeline.iloc[0] if not pipeline.empty else {} + drive = os.path.splitdrive(work_dir or '')[0] or 'Scratch' + _metric_grid([ + ('Scan slots', f"{len(slots)}/{max_active_scans or '?'}"), + (f'{drive} free', f'{disk_free_gb(work_dir):.2f} GiB'), + ('Bundle backlog', _integer(pipeline_row.get('bundle_items', 0))), + ('Projection lag', _integer(pipeline_row.get('projection_items', 0))), + ('Candidate capacity', _integer(pipeline_row.get('keycheck_items', 0))), + ('Quarantine capacity', _integer(pipeline_row.get('quarantine_items', 0))), + ]) + runtime, _ = parse_supervisor_status(log_dir) + if not runtime.empty: + runtime = runtime[runtime['source'].isin(CORE_RUNTIME_SOURCES)].copy() + order = {name: index for index, name in enumerate([ + 'result-ingester', 'jsonl-projector', 'janitor', 'worker-api', 'github', 'gitlab', + 'huggingface', 'dockerhub', 'package_git', 'keychecks', + ])} + runtime['_order'] = runtime['source'].map(order).fillna(len(order)) + runtime = runtime.sort_values('_order').drop(columns=['_order']) + runtime_columns = ['source', 'runtime_status', 'up', 'next', 'restarts'] + display_df(runtime[[column for column in runtime_columns if column in runtime]], height=360) + else: + st.info('Supervisor status is not available yet.') + + +def page_overview(conn, results_dir, queue_dir, log_dir, work_dir, keycheck_dir, scan_limiter_db, max_active_scans): + st.header('Overview') + today_expr = today_sql() + totals = query_df(conn, ''' + SELECT + COUNT(*) AS cycles, + COALESCE(SUM(scanned_count), 0) AS scanned, + COALESCE(SUM(found_count), 0) AS found_targets, + COALESCE(SUM(error_count), 0) AS error_targets, + COALESCE(SUM(skipped_count), 0) AS skipped, + COALESCE(SUM(findings_count), 0) AS findings + FROM source_cycles + ''') + finding_totals = query_df(conn, 'SELECT COUNT(*) AS finding_rows FROM findings') + today = query_df(conn, ''' + SELECT COALESCE(SUM(scanned_count), 0) AS scanned_today, + COALESCE(SUM(findings_count), 0) AS findings_today, + COALESCE(SUM(error_count), 0) AS errors_today + FROM source_cycles + WHERE started_at >= {today_expr} + '''.format(today_expr=today_expr)) + runtime_health, status_path = runtime_source_health(conn, log_dir) + active_sources = int(runtime_health['runtime_status'].astype(str).str.lower().eq('running').sum()) if not runtime_health.empty and 'runtime_status' in runtime_health else 0 + slots = scan_slots_df(scan_limiter_db) + validation_available = table_exists(conn, 'keycheck_current_state') + load_unique_totals = st.checkbox('Load unique scanner totals (slower)', value=False) + load_alive_total = st.checkbox('Load all-time alive key count (slower)', value=False) + if load_unique_totals: + unique_totals = query_df(conn, ''' + SELECT COUNT(DISTINCT NULLIF(secret_hash, '')) AS unique_secrets, + COUNT(DISTINCT NULLIF(finding_fingerprint, '')) AS unique_findings + FROM findings + ''') + urow = unique_totals.iloc[0] if not unique_totals.empty else {} + unique_secrets = int(urow.get('unique_secrets', 0)) + unique_findings = int(urow.get('unique_findings', 0)) + else: + unique_secrets = 'off' + unique_findings = 'off' + if validation_available: + alive_total = scalar(conn, "SELECT COUNT(*) FROM keycheck_current_state WHERE status_group = 'alive'") if load_alive_total else 'off' + alive_today = scalar(conn, f"SELECT COUNT(*) FROM keycheck_current_state WHERE status_group = 'alive' AND checked_at >= {today_expr}") + else: + alive_total = 0 + alive_today = 0 + row = totals.iloc[0] if not totals.empty else {} + frow = finding_totals.iloc[0] if not finding_totals.empty else {} + trow = today.iloc[0] if not today.empty else {} + authoritative_queue_backlog = ( + scalar(conn, "SELECT COUNT(*) FROM target_queue WHERE status IN ('pending','deferred')") + if table_exists(conn, 'target_queue') else queue_backlog(queue_dir) + ) + show_metrics([ + ('Active sources', active_sources), + ('Scan slots', f"{len(slots)}/{max_active_scans or '?'}"), + ('Queue backlog', int(authoritative_queue_backlog or 0)), + ('Scanned today', int(trow.get('scanned_today', 0))), + ('Findings today', int(trow.get('findings_today', 0))), + ('Alive keys total', alive_total if isinstance(alive_total, str) else int(alive_total or 0)), + ('Alive rows today', int(alive_today or 0)), + ('Errors today', int(trow.get('errors_today', 0))), + ]) + if table_exists(conn, 'pipeline_capacity'): + pipeline = query_df(conn, 'SELECT * FROM pipeline_capacity WHERE id = 1') + prow = pipeline.iloc[0] if not pipeline.empty else {} + show_metrics([ + ('Bundle backlog', int(prow.get('bundle_items', 0))), + ('Projection lag', int(prow.get('projection_items', 0))), + ('Keycheck candidates', int(prow.get('keycheck_items', 0))), + ('Pipeline quarantine', int(prow.get('quarantine_items', 0))), + ]) + if table_exists(conn, 'pipeline_leases'): + leases = query_df(conn, ''' + SELECT worker_name, state, heartbeat_at, lease_expires_at, last_error + FROM pipeline_leases ORDER BY worker_name + ''') + st.caption('Pipeline worker leases') + display_df(leases, height=150) + st.caption(f"Scanner totals: cycles={int(row.get('cycles', 0))}, scanned={int(row.get('scanned', 0))}, findings={int(frow.get('finding_rows', 0))}, unique secrets={unique_secrets}, unique findings={unique_findings}. D free: {disk_free_gb(work_dir)} GB") + archive_summary, archive_detectors, archive_targets = github_archive_yield(conn) + if archive_summary: + st.subheader('GitHub Archive Yield') + show_metrics([ + ('Archive found targets', archive_summary['found_targets']), + ('Archive raw findings', archive_summary['raw_findings']), + ('Archive interesting', archive_summary['interesting_findings']), + ('Archive noise/generic', archive_summary['noise_or_generic_findings']), + ('Archive verified', archive_summary['verified_findings']), + ]) + col_archive_1, col_archive_2 = st.columns(2) + with col_archive_1: + st.caption('Detector split from redacted findings metadata; raw result payloads are not loaded.') + display_df(archive_detectors, height=320) + with col_archive_2: + st.caption('Targets ranked by interesting detector rows.') + display_df(archive_targets, height=320) + + st.subheader('Keycheck PostgreSQL Current State And Compatibility Lag') + summary_df, summary_path = keycheck_summary_df(keycheck_dir) + db_summary = recent_keycheck_db_summary(conn) + metric_summary = keycheck_db_metric_summary(keycheck_dir) + if not summary_df.empty: + merged = summary_df.merge(db_summary, on='service', how='left') if not db_summary.empty else summary_df + if not metric_summary.empty: + merged = merged.merge(metric_summary, on='service', how='left') + preferred = [ + 'service', 'alive', 'alive_rate_limited', 'no_balance', 'no_quota', 'limited', 'network', 'dead', 'restricted', + 'file_updated_at', 'db_latest_checked', 'db_latest_created', 'db_recent_rows', 'db_cached_rows', + 'db_write_failed_tail', 'cached_occurrence_tail', 'output_dir', + ] + display_df(merged[[column for column in preferred if column in merged.columns]], height=420) + st.caption(f'Current-state summary: {summary_path}') + else: + st.info('No keycheck summary.tsv found') + + health_df, health_path = keycheck_pipeline_health(log_dir) + st.subheader('Keycheck Pipeline Health') + st.caption(health_path) + display_df(health_df, height=160) + if st.checkbox('Load keycheckable/noise totals (slower)', value=False): + classified_totals = query_df(conn, f''' + SELECT COUNT(*) AS raw_findings, + COUNT(DISTINCT NULLIF(secret_hash, '')) AS raw_unique, + COUNT(DISTINCT CASE WHEN is_keycheckable = 1 THEN NULLIF(secret_hash, '') END) AS keycheckable_unique, + COUNT(DISTINCT CASE WHEN is_keycheckable = 0 THEN NULLIF(secret_hash, '') END) AS noise_unique + FROM ({keycheckable_findings_view_sql()}) + ''') + crow = classified_totals.iloc[0] if not classified_totals.empty else {} + show_metrics([ + ('Raw unique findings', int(crow.get('raw_unique', 0))), + ('Keycheckable unique', int(crow.get('keycheckable_unique', 0))), + ('Noise unique', int(crow.get('noise_unique', 0))), + ('Noise share', format_pct((int(crow.get('noise_unique', 0)) / int(crow.get('raw_unique', 1))) if int(crow.get('raw_unique', 0)) else 0)), + ]) + + st.subheader('Runtime Source Health') + st.caption(f'Runtime status file: {status_path}') + display_df(runtime_health, height=420) + + st.subheader('Active Scan Slots') + st.caption(scan_limiter_db) + display_df(slots[['owner_source', 'owner_pid', 'age_sec', 'command']] if not slots.empty else slots, height=260) + + st.subheader('Per Source DB Aggregates') + source_df = query_df(conn, ''' + SELECT + source, + COUNT(*) AS cycles, + SUM(fetched_count) AS fetched, + SUM(queued_new_count) AS queued_new, + SUM(queued_updated_count) AS queued_updated, + SUM(scanned_count) AS scanned, + SUM(found_count) AS found_targets, + SUM(error_count) AS error_targets, + SUM(skipped_count) AS skipped, + SUM(findings_count) AS findings, + SUM(verified_findings_count) AS verified_findings, + AVG(hit_rate) AS avg_hit_rate, + AVG(error_rate) AS avg_error_rate, + AVG(targets_per_hour) AS avg_targets_per_hour, + MAX(started_at) AS latest_cycle + FROM source_cycles + GROUP BY source + ORDER BY latest_cycle DESC + ''') + if not source_df.empty: + source_df['avg_hit_rate'] = source_df['avg_hit_rate'].map(format_pct) + source_df['avg_error_rate'] = source_df['avg_error_rate'].map(format_pct) + display_df(source_df) + + st.subheader('Current Queues') + display_df(current_queue_counts(queue_dir)) + + colv1, colv2 = st.columns(2) + with colv1: + st.subheader('Top Detectors (Recent)') + display_df(query_df(conn, ''' + SELECT source, detector_name, COUNT(*) AS findings, + COUNT(DISTINCT NULLIF(secret_hash, '')) AS unique_secrets + FROM ( + SELECT source, detector_name, secret_hash + FROM findings + ORDER BY id DESC + LIMIT 20000 + ) + GROUP BY source, detector_name + ORDER BY findings DESC + LIMIT 20 + '''), height=360) + with colv2: + st.subheader('Top Useful Sources By Alive Keys (Recent)') + if validation_available: + display_df(query_df(conn, ''' + SELECT source, service, COUNT(DISTINCT key_hash) AS alive_keys, COUNT(*) AS linked_findings + FROM ( + SELECT source, service, key_hash, status_group + FROM keycheck_results + ORDER BY id DESC + LIMIT 20000 + ) + WHERE status_group = 'alive' AND source IS NOT NULL AND key_hash != '' + GROUP BY source, service + ORDER BY alive_keys DESC, linked_findings DESC + LIMIT 20 + '''), height=360) + else: + st.info('No keycheck_results yet') + + col1, col2 = st.columns(2) + with col1: + st.subheader('Recent Source Cycles') + display_df(query_df(conn, ''' + SELECT started_at, ended_at, status, source, mode, query, fetched_count, + queued_new_count, queued_updated_count, + scanned_count, findings_count, error_count, skipped_count + FROM source_cycles + ORDER BY id DESC + LIMIT 20 + '''), height=420) + with col2: + st.subheader('Top Error Categories') + errors = query_df(conn, ''' + SELECT source, category, COUNT(*) AS count + FROM errors + GROUP BY source, category + ORDER BY count DESC + LIMIT 20 + ''') + display_df(errors, height=420) + + +def page_runtime(conn, queue_dir, log_dir, work_dir, scan_limiter_db, max_active_scans): + st.header('Runtime') + runtime_health, status_path = runtime_source_health(conn, log_dir) + slots = scan_slots_df(scan_limiter_db) + active_sources = int(runtime_health['runtime_status'].astype(str).str.lower().eq('running').sum()) if not runtime_health.empty and 'runtime_status' in runtime_health else 0 + waiting_sources = int(runtime_health['runtime_status'].astype(str).str.lower().isin(('waiting', 'blocked', 'backoff')).sum()) if not runtime_health.empty and 'runtime_status' in runtime_health else 0 + show_metrics([ + ('Running sources', active_sources), + ('Waiting sources', waiting_sources), + ('Active scan slots', f"{len(slots)}/{max_active_scans or '?'}"), + ('Queue backlog', queue_backlog(queue_dir)), + ('D/free GB', disk_free_gb(work_dir)), + ]) + + st.subheader('Source Health') + st.caption(f'Runtime status file: {status_path}') + display_df(runtime_health, height=440) + + col1, col2 = st.columns(2) + with col1: + st.subheader('Active Scan Slots') + st.caption(scan_limiter_db) + display_df(slots[['owner_source', 'owner_pid', 'age_sec', 'command']] if not slots.empty else slots, height=360) + with col2: + st.subheader('Current Queues') + display_df(current_queue_counts(queue_dir), height=360) + + st.subheader('Log Metadata') + rows = [] + if os.path.isdir(log_dir): + for name in sorted(item for item in os.listdir(log_dir) if item.lower().endswith('.log')): + path = os.path.join(log_dir, name) + rows.append({ + 'log': name, + 'age': human_age(datetime.fromtimestamp(os.path.getmtime(path)).isoformat(timespec='seconds')) if os.path.exists(path) else '', + 'bytes': os.path.getsize(path) if os.path.exists(path) else 0, + }) + display_df(pd.DataFrame(rows), height=360) + + +def page_scanner_results(conn): + st.header('Scanner Results') + cycles = query_df(conn, ''' + SELECT id AS cycle_id, started_at, ended_at, source, mode, query, status, + fetched_count, queued_new_count, queued_updated_count, scanned_count, clean_count, found_count, + skipped_count, error_count, findings_count, verified_findings_count, + unique_secrets_count, unique_findings_count, hit_rate, error_rate, duration_sec + FROM source_cycles + ORDER BY id DESC + LIMIT 5000 + ''') + if cycles.empty: + st.info('No source cycle data yet') + return + + col1, col2, col3, col4 = st.columns(4) + with col1: + source_filter = st.multiselect('Sources', sorted(cycles['source'].dropna().unique()), default=sorted(cycles['source'].dropna().unique())) + with col2: + status_filter = st.multiselect('Cycle statuses', sorted(cycles['status'].dropna().unique()), default=sorted(cycles['status'].dropna().unique())) + with col3: + query_contains = st.text_input('Query contains') + with col4: + days = st.number_input('Last N days (0 = all loaded)', min_value=0, max_value=365, value=0, step=1) + + filtered = cycles + if source_filter: + filtered = filtered[filtered['source'].isin(source_filter)] + if status_filter: + filtered = filtered[filtered['status'].isin(status_filter)] + if query_contains: + filtered = filtered[filtered['query'].astype(str).str.contains(query_contains, case=False, na=False)] + if days: + cutoff = pd.Timestamp.utcnow() - pd.Timedelta(days=int(days)) + started = pd.to_datetime(filtered['started_at'], errors='coerce', utc=True) + filtered = filtered[started >= cutoff] + + show_metrics([ + ('Cycles', len(filtered)), + ('Scanned', int(filtered['scanned_count'].fillna(0).sum())), + ('Findings', int(filtered['findings_count'].fillna(0).sum())), + ('Errors', int(filtered['error_count'].fillna(0).sum())), + ('Skipped', int(filtered['skipped_count'].fillna(0).sum())), + ('Hit rate', format_pct((filtered['found_count'].fillna(0).sum() / filtered['scanned_count'].fillna(0).sum()) if filtered['scanned_count'].fillna(0).sum() else 0)), + ]) + + section = st.selectbox('Scanner results section', ['By Source', 'By Query', 'By Cycle', 'Detectors', 'Keycheckable vs Noise', 'Targets']) + if section == 'By Source': + by_source = filtered.groupby('source', dropna=False).agg( + cycles=('cycle_id', 'count'), + scanned=('scanned_count', 'sum'), + findings=('findings_count', 'sum'), + errors=('error_count', 'sum'), + skipped=('skipped_count', 'sum'), + ).reset_index().sort_values(['findings', 'scanned'], ascending=False) + display_df(by_source, height=420) + if not by_source.empty: + st.plotly_chart(px.bar(by_source, x='source', y='findings', title='Findings By Source'), width='stretch') + elif section == 'By Query': + by_query = filtered.groupby(['source', 'query'], dropna=False).agg( + cycles=('cycle_id', 'count'), + scanned=('scanned_count', 'sum'), + findings=('findings_count', 'sum'), + errors=('error_count', 'sum'), + ).reset_index().sort_values(['findings', 'scanned'], ascending=False) + display_df(by_query, height=520) + elif section == 'By Cycle': + display_df(filtered.sort_values('started_at', ascending=False), height=620) + chart_df = filtered.sort_values('started_at') + if not chart_df.empty: + st.plotly_chart(px.line(chart_df, x='started_at', y='findings_count', color='source', markers=True, title='Findings By Cycle'), width='stretch') + elif section == 'Detectors': + clauses = [] + params = [] + if source_filter: + clauses.append('source IN ({})'.format(','.join('?' for _ in source_filter))) + params.extend(source_filter) + if query_contains: + clauses.append('query LIKE ?') + params.append(f'%{query_contains}%') + where = ('WHERE ' + ' AND '.join(clauses)) if clauses else '' + detectors = query_df(conn, f''' + SELECT detector_name, source, validation_service, noise_reason, is_keycheckable, + COUNT(*) AS findings, COUNT(DISTINCT NULLIF(secret_hash, '')) AS unique_secrets + FROM ({keycheckable_findings_view_sql()}) + {where} + GROUP BY detector_name, source, validation_service, noise_reason, is_keycheckable + ORDER BY findings DESC + LIMIT 200 + ''', params) + display_df(detectors, height=620) + elif section == 'Keycheckable vs Noise': + st.warning('This diagnostic query scans findings and can be slow on the active DB.') + if not st.button('Run keycheckable/noise query'): + st.info('Press the button to calculate keycheckable/noise backlog.') + return + clauses = [] + params = [] + if source_filter: + clauses.append('source IN ({})'.format(','.join('?' for _ in source_filter))) + params.extend(source_filter) + if query_contains: + clauses.append('query LIKE ?') + params.append(f'%{query_contains}%') + where = ' AND '.join(clauses) + backlog = query_df(conn, f''' + SELECT source, query, validation_service, detector_name, + SUM(raw_findings) AS raw_findings, + SUM(unique_secrets) AS unique_secrets, + SUM(keycheckable_unique) AS keycheckable_unique, + SUM(noise_unique) AS noise_unique, + MAX(checked_unique) AS checked_unique, + MAX(alive_unique) AS alive_unique, + MAX(pending_unique) AS pending_unique + FROM ({keycheckable_backlog_sql(where)}) + GROUP BY source, query, validation_service, detector_name + ORDER BY keycheckable_unique DESC, noise_unique DESC, raw_findings DESC + LIMIT 500 + ''', params) + display_df(backlog, height=620) + elif section == 'Targets': + targets = query_df(conn, ''' + SELECT ended_at, source, query, status, target, normalized_target, duration_sec, + findings_count, verified_findings_count, error_count, + package_name, package_version, package_filename, package_type, package_size + FROM target_scans + ORDER BY id DESC + LIMIT 5000 + ''') + if not targets.empty: + if source_filter: + targets = targets[targets['source'].isin(source_filter)] + if query_contains: + targets = targets[targets['query'].astype(str).str.contains(query_contains, case=False, na=False)] + display_df(targets, height=620) + + +def page_runs(conn): + st.header('Historical Runs (Legacy / Debug)') + st.caption('Loop-mode sources usually do not finish runs, so these totals are not the source of truth for scanner metrics.') + runs = query_df(conn, ''' + SELECT id, started_at, ended_at, duration_sec, status, invocation_mode, selected_source, + selected_platform, config_path, total_fetched, total_queued_new, total_scanned, + total_clean, total_found, total_skipped, total_errors, total_findings, + total_verified_findings, total_unique_secrets, total_unique_findings + FROM runs + ORDER BY id DESC + ''') + display_df(runs) + if not runs.empty: + run_id = st.selectbox('Inspect run', runs['id'].tolist()) + config = query_df(conn, 'SELECT scope, source, captured_at FROM config_snapshots WHERE run_id = ? ORDER BY id', [int(run_id)]) + display_df(config) + + +def page_sources(conn): + st.header('Sources') + cycles = query_df(conn, ''' + SELECT started_at, source, mode, query, status, fetched_count, queued_new_count, + queued_updated_count, scanned_count, + clean_count, found_count, skipped_count, error_count, findings_count, verified_findings_count, + unique_secrets_count, unique_findings_count, targets_per_hour, hit_rate, error_rate + FROM source_cycles + ORDER BY id DESC + ''') + if cycles.empty: + st.info('No source cycle data yet') + return + source_filter = st.multiselect('Sources', sorted(cycles['source'].dropna().unique()), default=sorted(cycles['source'].dropna().unique())) + filtered = cycles[cycles['source'].isin(source_filter)] if source_filter else cycles + display_df(filtered) + chart_df = filtered.sort_values('started_at') + if not chart_df.empty: + st.plotly_chart(px.line(chart_df, x='started_at', y='scanned_count', color='source', markers=True, title='Scanned targets by cycle'), width='stretch') + st.plotly_chart(px.line(chart_df, x='started_at', y='error_rate', color='source', markers=True, title='Error rate by cycle'), width='stretch') + + +def page_queries(conn): + st.header('Queries') + queries = query_df(conn, ''' + SELECT source, query, COUNT(*) AS cycles, SUM(fetched_count) AS fetched, + SUM(queued_new_count) AS queued_new, + SUM(queued_updated_count) AS queued_updated, SUM(scanned_count) AS scanned, + SUM(findings_count) AS findings, SUM(error_count) AS errors, + AVG(hit_rate) AS avg_hit_rate, AVG(error_rate) AS avg_error_rate, + AVG(duration_sec) AS avg_duration_sec + FROM source_cycles + GROUP BY source, query + ORDER BY findings DESC, scanned DESC + ''') + if not queries.empty: + queries['avg_hit_rate'] = queries['avg_hit_rate'].map(format_pct) + queries['avg_error_rate'] = queries['avg_error_rate'].map(format_pct) + display_df(queries) + + +def page_findings(conn): + st.header('Findings') + st.caption('Redacted scanner metadata. Raw secret columns are never queried by this dashboard.') + desired_columns = [ + 'id', 'created_at', 'source', 'query', 'target', 'detector_name', 'detector_type', 'verified', + 'provider', 'credential_kind', 'credential_confidence', 'required_context_missing', + 'principal', 'username', 'email', 'project_id', 'organization', 'registry', 'endpoint', 'scope', 'resource', + 'file_path', 'line_number', 'commit_hash', 'secret_hash', 'detector_secret_hash', + 'finding_fingerprint', 'redacted_secret', + ] + existing = table_columns(conn, 'findings') + columns = ', '.join(column for column in desired_columns if column in existing) + if not columns: + st.info('No findings columns available') + return + findings = query_df(conn, f'SELECT {columns} FROM findings ORDER BY id DESC LIMIT 1000') + if findings.empty: + st.info('No findings yet') + return + col1, col2, col3, col4 = st.columns(4) + col1.metric('Rows shown', len(findings)) + col2.metric('Unique secrets', findings['secret_hash'].replace('', pd.NA).dropna().nunique()) + col3.metric('Unique findings', findings['finding_fingerprint'].replace('', pd.NA).dropna().nunique()) + col4.metric('Verified', int(findings['verified'].sum())) + detectors = query_df(conn, 'SELECT detector_name, COUNT(*) AS count FROM findings GROUP BY detector_name ORDER BY count DESC LIMIT 30') + if not detectors.empty: + st.plotly_chart(px.bar(detectors, x='detector_name', y='count', title='Findings by detector'), width='stretch') + display_df(findings) + + +def page_validation(conn): + st.header('Validation / Keychecks') + if not table_exists(conn, 'keycheck_results'): + st.info('No keycheck_results table yet. New keycheck runs will create it and populate validation data.') + return + + preset = st.selectbox('Preset', VALIDATION_PRESETS, index=0) + historical_preset = preset in ('Ever usable LLM keys', 'Ever alive / LLM candidates') + preset_uses_latest = preset != 'Latest rows' and not historical_preset + if preset == 'Unattributed alive': + default_status_groups = ['alive'] + default_access_tiers = ['alive_unproven_llm', 'usable_llm'] + elif preset == 'Latest rows': + default_status_groups = ['alive'] + default_access_tiers = VALIDATION_ACCESS_TIERS + elif preset == 'Usable LLM keys': + default_status_groups = VALIDATION_STATUS_GROUPS + default_access_tiers = ['usable_llm'] + elif preset == 'Ever usable LLM keys': + default_status_groups = VALIDATION_STATUS_GROUPS + default_access_tiers = ['usable_llm'] + elif preset == 'Ever alive / LLM candidates': + default_status_groups = ['alive'] + default_access_tiers = ['usable_llm', 'alive_unproven_llm'] + elif preset == 'Quota / no balance': + default_status_groups = VALIDATION_STATUS_GROUPS + default_access_tiers = ['no_quota', 'quota_limited'] + elif preset == 'Alive but not proven LLM': + default_status_groups = VALIDATION_STATUS_GROUPS + default_access_tiers = ['alive_unproven_llm'] + else: + default_status_groups = VALIDATION_STATUS_GROUPS + default_access_tiers = VALIDATION_ACCESS_TIERS + + col1, col2, col3, col4 = st.columns(4) + with col1: + services = st.multiselect('Services', VALIDATION_SERVICES, default=VALIDATION_SERVICES) + with col2: + status_groups = st.multiselect('Status groups', VALIDATION_STATUS_GROUPS, default=default_status_groups) + source_options = [UNATTRIBUTED, *SOURCES] + with col3: + sources = st.multiselect('Sources', source_options, default=source_options) + with col4: + row_limit = st.number_input('Rows limit', min_value=500, max_value=50000, value=5000, step=500) + col5, col6, col7 = st.columns([1, 1, 2]) + with col5: + checked_axis = st.selectbox('Time axis', ['checked_at', 'found_at']) + with col6: + dedupe_latest = st.checkbox('Dedupe latest per key (slower)', value=preset_uses_latest) + with col7: + query_text = st.text_input('Source query contains') + access_tiers = st.multiselect('Access tiers', VALIDATION_ACCESS_TIERS, default=default_access_tiers) + exact_statuses = st.multiselect('Exact provider statuses', VALIDATION_STATUSES, default=[]) + search_text = st.text_input( + 'Search hash / masked key / target', + help='Use a SHA-256 hash, masked value, or non-secret metadata. Raw credentials are not accepted or queried.', + ) + if search_text and not dashboard_search_is_safe(search_text): + st.warning('Search refused: use a SHA-256 hash or masked/non-secret metadata.') + search_text = '' + if historical_preset: + st.caption('Historical preset: uses all matching keycheck rows, not latest current-state. This answers “which source ever produced this alive/usable key”.') + + search_like = f'%{search_text}%' if search_text else '' + search_hash = search_text.lower() if re.fullmatch(r'[0-9a-fA-F]{64}', search_text or '') else '' + + clauses = [] + params = [] + id_window = None + if not dedupe_latest and not search_text and not historical_preset: + max_id_df = query_df(conn, 'SELECT MAX(id) AS max_id FROM keycheck_results') + max_id = int(max_id_df.iloc[0]['max_id'] or 0) if not max_id_df.empty else 0 + id_window = max(0, max_id - int(row_limit) * 20) + clauses.append('kr.id >= ?') + params.append(id_window) + if services and set(services) != set(VALIDATION_SERVICES): + clauses.append('kr.service IN ({})'.format(','.join('?' for _ in services))) + params.extend(services) + if status_groups and set(status_groups) != set(VALIDATION_STATUS_GROUPS): + clauses.append('kr.status_group IN ({})'.format(','.join('?' for _ in status_groups))) + params.extend(status_groups) + if sources and set(sources) != set(source_options): + source_clauses = [] + concrete_sources = [item for item in sources if item != UNATTRIBUTED] + if concrete_sources: + source_clauses.append('COALESCE(kr.source, f.source, fh.source) IN ({})'.format(','.join('?' for _ in concrete_sources))) + params.extend(concrete_sources) + if UNATTRIBUTED in sources: + source_clauses.append('COALESCE(kr.source, f.source, fh.source) IS NULL') + if source_clauses: + clauses.append('(' + ' OR '.join(source_clauses) + ')') + if query_text: + clauses.append("COALESCE(kr.query, f.query, fh.query, '') LIKE ?") + params.append(f'%{query_text}%') + access_tier_sql = validation_access_tier_sql('kr') + if access_tiers and set(access_tiers) != set(VALIDATION_ACCESS_TIERS): + clauses.append(f"({access_tier_sql}) IN ({','.join('?' for _ in access_tiers)})") + params.extend(access_tiers) + if exact_statuses: + clauses.append('UPPER(kr.status) IN ({})'.format(','.join('?' for _ in exact_statuses))) + params.extend(exact_statuses) + if preset == 'Unattributed alive': + clauses.append("kr.status_group = 'alive'") + clauses.append("COALESCE(kr.source, kr.finding_id, kr.target_scan_id) IS NULL") + if search_text: + clauses.append('''( + kr.key_masked LIKE ? + OR kr.key_hash = ? + OR kr.secret_hash = ? + OR kr.detector_name LIKE ? + OR kr.target LIKE ? + OR COALESCE(kr.source, f.source, fh.source, '') LIKE ? + OR COALESCE(kr.query, f.query, fh.query, '') LIKE ? + OR COALESCE(kr.target, f.target, fh.target, ts.target, '') LIKE ? + OR COALESCE(f.redacted_secret, fh.redacted_secret, '') LIKE ? + OR COALESCE(f.detector_name, fh.detector_name, '') LIKE ? + OR COALESCE(f.file_path, fh.file_path, '') LIKE ? + OR COALESCE(f.commit_hash, fh.commit_hash, '') LIKE ? + OR ts.package_name LIKE ? + OR ts.package_version LIKE ? + OR ts.package_filename LIKE ? + )''') + params.extend([ + search_like, search_hash, search_hash, + search_like, search_like, search_like, search_like, + search_like, search_like, search_like, search_like, + search_like, search_like, search_like, search_like, + ]) + where = ('WHERE ' + ' AND '.join(clauses)) if clauses else '' + if dedupe_latest: + source_sql = f'({latest_keycheck_view_sql()})' + else: + source_sql = 'keycheck_results' + meta_source = json_extract_sql('kr.metadata_json', '$.source') + meta_backfill_source = json_extract_sql('kr.metadata_json', '$.backfill_source_file') + meta_result_source = json_extract_sql('kr.metadata_json', '$.result_source') + meta_llm_probe_status = json_extract_sql('kr.metadata_json', '$.llm_probe_status') + meta_probe_status = json_extract_sql('kr.metadata_json', '$.probe.status') + meta_llm_probe_model = json_extract_sql('kr.metadata_json', '$.llm_probe_model') + meta_probe_model = json_extract_sql('kr.metadata_json', '$.probe.model') + meta_remaining_credits = json_extract_sql('kr.metadata_json', '$.remaining_credits') + meta_balance_usd = json_extract_sql('kr.metadata_json', '$.balance_usd') + meta_vertex_enabled = json_extract_sql('kr.metadata_json', '$.vertex_enabled') + meta_bedrock_enabled = json_extract_sql('kr.metadata_json', '$.bedrock_enabled') + meta_route_probe = json_extract_sql('kr.metadata_json', '$.route_probe') + meta_foundry_route_probe = json_extract_sql('kr.metadata_json', '$.foundry_route_probe') + rows = query_df(conn, f''' + SELECT kr.id, kr.checked_at, kr.found_at, kr.service, kr.status_group, kr.status, + {validation_access_tier_sql('kr')} AS access_tier, + kr.key_hash, kr.secret_hash, kr.key_masked, + COALESCE(kr.source, f.source, fh.source) AS source, + COALESCE(kr.query, f.query, fh.query) AS query, + COALESCE(kr.cycle_id, f.cycle_id, fh.cycle_id) AS cycle_id, + COALESCE(kr.target_scan_id, f.target_scan_id, fh.target_scan_id) AS target_scan_id, + COALESCE(kr.finding_id, fh.id) AS finding_id, + COALESCE(kr.detector_name, f.detector_name, fh.detector_name) AS detector_name, + COALESCE(kr.target, f.target, fh.target, ts.target) AS target, + COALESCE(f.file_path, fh.file_path) AS file_path, + COALESCE(f.line_number, fh.line_number) AS line_number, + COALESCE(f.commit_hash, fh.commit_hash) AS commit_hash, + ts.package_name, ts.package_version, ts.package_filename, ts.package_type, + CASE + WHEN f.id IS NOT NULL THEN 'finding_id' + WHEN fh.id IS NOT NULL THEN 'secret_hash' + WHEN COALESCE(kr.source, kr.query, kr.target) IS NOT NULL THEN 'keycheck_metadata' + ELSE 'missing_finding' + END AS attribution_status, + {meta_source} AS keycheck_source_line, + {meta_backfill_source} AS backfill_source_file, + COALESCE({meta_result_source}, 'api_check') AS result_source, + COALESCE({meta_llm_probe_status}, {meta_probe_status}) AS llm_probe_status, + COALESCE({meta_llm_probe_model}, {meta_probe_model}) AS llm_probe_model, + {meta_remaining_credits} AS remaining_credits, + {meta_balance_usd} AS balance_usd, + {meta_vertex_enabled} AS vertex_enabled, + {meta_bedrock_enabled} AS bedrock_enabled, + {meta_route_probe} AS azure_route_probe, + {meta_foundry_route_probe} AS foundry_route_probe + FROM {source_sql} kr + LEFT JOIN findings f ON f.id = kr.finding_id + LEFT JOIN findings fh ON f.id IS NULL AND fh.id = ( + SELECT id + FROM findings + WHERE secret_hash = COALESCE(NULLIF(kr.secret_hash, ''), NULLIF(kr.key_hash, '')) + ORDER BY id DESC + LIMIT 1 + ) + LEFT JOIN target_scans ts ON ts.id = COALESCE(kr.target_scan_id, f.target_scan_id, fh.target_scan_id) + {where} + ORDER BY kr.id DESC + LIMIT ? + ''', [*params, int(row_limit)]) + if rows.empty: + st.info('No validation rows match current filters.') + if search_text: + st.subheader('Finding Metadata Matches') + st.caption('Fallback hash/redacted/metadata search in findings.') + display_df(finding_metadata_search(conn, search_text), height=520) + return + + filtered = rows.copy() + if 'source' in filtered: + filtered['source'] = filtered['source'].fillna(UNATTRIBUTED) + if 'query' in filtered: + filtered['query'] = filtered['query'].fillna(UNATTRIBUTED) + if 'key_hash' in filtered: + filtered['key_identity'] = filtered['key_hash'].where(filtered['key_hash'].astype(str) != '', filtered['secret_hash']).fillna(filtered['key_masked']) + else: + filtered['key_identity'] = filtered.get('key_masked', pd.Series(dtype='object')) + + usable_keys = int(filtered.loc[filtered['access_tier'] == 'usable_llm', 'key_identity'].replace('', pd.NA).dropna().nunique()) if 'access_tier' in filtered else 0 + unproven_keys = int(filtered.loc[filtered['access_tier'] == 'alive_unproven_llm', 'key_identity'].replace('', pd.NA).dropna().nunique()) if 'access_tier' in filtered else 0 + quota_keys = int(filtered.loc[filtered['access_tier'].isin(['no_quota', 'quota_limited']), 'key_identity'].replace('', pd.NA).dropna().nunique()) if 'access_tier' in filtered else 0 + unattributed_rows = int((filtered['attribution_status'] == 'missing_finding').sum()) if 'attribution_status' in filtered else 0 + + show_metrics([ + ('Rows', len(filtered)), + ('Usable LLM keys', usable_keys), + ('Alive unproven keys', unproven_keys), + ('No quota / limited keys', quota_keys), + ('Unattributed rows', unattributed_rows), + ]) + + if not filtered.empty: + status_by_service = filtered.groupby(['service', 'access_tier']).size().reset_index(name='count') + st.plotly_chart(px.bar(status_by_service, x='service', y='count', color='access_tier', title='Validation Access Tier By Service'), width='stretch') + usable_rows = filtered[filtered['access_tier'] == 'usable_llm'].copy() + usable_rows['source'] = usable_rows['source'].fillna(UNATTRIBUTED) + usable_rows['query'] = usable_rows['query'].fillna(UNATTRIBUTED) + usable_rows['target'] = usable_rows['target'].fillna(UNATTRIBUTED) + usable = usable_rows.groupby(['source', 'query', 'target', 'service', 'status', 'attribution_status'], dropna=False).agg( + rows=('id', 'count'), + keys=('key_identity', 'nunique'), + latest_checked=('checked_at', 'max'), + ).reset_index().sort_values(['keys', 'rows'], ascending=False).head(100) + st.subheader('Usable LLM By Source / Query / Target') + display_df(usable, height=420) + + unproven_rows = filtered[filtered['access_tier'] == 'alive_unproven_llm'].copy() + if not unproven_rows.empty: + unproven_rows['source'] = unproven_rows['source'].fillna(UNATTRIBUTED) + unproven = unproven_rows.groupby(['source', 'query', 'service', 'status'], dropna=False).agg( + rows=('id', 'count'), + keys=('key_identity', 'nunique'), + latest_checked=('checked_at', 'max'), + ).reset_index().sort_values(['keys', 'rows'], ascending=False).head(100) + st.subheader('Alive But Not Proven LLM') + display_df(unproven, height=300) + + st.subheader('Usable LLM Origins') + origin_columns = [ + 'checked_at', 'found_at', 'service', 'status', 'access_tier', 'result_source', 'source', 'query', 'target', + 'attribution_status', + 'llm_probe_status', 'llm_probe_model', 'remaining_credits', 'balance_usd', + 'vertex_enabled', 'bedrock_enabled', 'azure_route_probe', 'foundry_route_probe', + 'keycheck_source_line', 'backfill_source_file', + 'detector_name', 'file_path', 'line_number', 'commit_hash', + 'package_name', 'package_version', 'package_filename', 'package_type', + 'finding_id', 'target_scan_id', 'cycle_id', 'id', + ] + display_df( + usable_rows[[column for column in origin_columns if column in usable_rows.columns]].head(int(row_limit)), + height=520, + ) + if search_text: + st.subheader('Finding Metadata Matches') + st.caption('Direct hash/redacted/metadata matches, including rows without keycheck results.') + display_df(finding_metadata_search(conn, search_text), height=420) + cycle = filtered.groupby(['source', 'query', 'cycle_id', 'service', 'status_group']).size().reset_index(name='count').sort_values('count', ascending=False).head(200) + st.subheader('Validation By Found Cycle') + display_df(cycle, height=420) + timeline = filtered.dropna(subset=[checked_axis]).copy() + if not timeline.empty: + timeline['day'] = timeline[checked_axis].astype(str).str.slice(0, 10) + timeline_df = timeline.groupby(['day', 'status_group']).size().reset_index(name='count') + st.plotly_chart(px.line(timeline_df, x='day', y='count', color='status_group', markers=True, title=f'Validation Timeline By {checked_axis}'), width='stretch') + + st.subheader('Keycheckable Backlog By Source / Query') + st.caption('Optional diagnostic. Runs a heavier query over findings and keycheck_results.') + if st.button('Calculate keycheckable backlog'): + clauses = ['is_keycheckable = 1'] + params = [] + if services: + clauses.append('validation_service IN ({})'.format(','.join('?' for _ in services))) + params.extend(services) + concrete_sources = [item for item in sources if item != UNATTRIBUTED] + if concrete_sources: + clauses.append('source IN ({})'.format(','.join('?' for _ in concrete_sources))) + params.extend(concrete_sources) + if query_text: + clauses.append('query LIKE ?') + params.append(f'%{query_text}%') + backlog = query_df(conn, f''' + SELECT source, query, validation_service, + SUM(raw_findings) AS raw_findings, + SUM(unique_secrets) AS unique_secrets, + SUM(keycheckable_unique) AS keycheckable_unique, + MAX(checked_unique) AS checked_unique, + MAX(alive_unique) AS alive_unique, + MAX(pending_unique) AS pending_unique + FROM ({keycheckable_backlog_sql(' AND '.join(clauses))}) + GROUP BY source, query, validation_service + ORDER BY pending_unique DESC, keycheckable_unique DESC, alive_unique DESC + LIMIT 200 + ''', params) + display_df(backlog, height=420) + + st.subheader('Latest Validation Rows') + display_df(filtered, height=620) + + +def page_errors(conn): + st.header('Errors') + grouped = query_df(conn, ''' + SELECT source, category, COUNT(*) AS count, MAX(created_at) AS latest + FROM errors + GROUP BY source, category + ORDER BY count DESC, latest DESC + LIMIT 200 + ''') + display_df(grouped) + recent = query_df(conn, ''' + SELECT created_at, source, query, target, category + FROM errors + ORDER BY id DESC + LIMIT 300 + ''') + st.subheader('Recent Errors') + display_df(recent) + + +def page_targets(conn): + st.header('Targets') + targets = query_df(conn, ''' + SELECT ended_at, source, query, status, target, normalized_target, duration_sec, + findings_count, verified_findings_count, error_count, + package_name, package_version, package_filename, package_type, package_size + FROM target_scans + ORDER BY id DESC + LIMIT 2000 + ''') + if targets.empty: + st.info('No target scan data yet') + return + statuses = st.multiselect('Statuses', sorted(targets['status'].dropna().unique()), default=sorted(targets['status'].dropna().unique())) + source_values = st.multiselect('Sources', sorted(targets['source'].dropna().unique()), default=sorted(targets['source'].dropna().unique())) + text = st.text_input('Target contains') + filtered = targets + if statuses: + filtered = filtered[filtered['status'].isin(statuses)] + if source_values: + filtered = filtered[filtered['source'].isin(source_values)] + if text: + filtered = filtered[filtered['target'].str.contains(text, case=False, na=False)] + display_df(filtered) + + +def page_queues_state(conn, queue_dir, state_file): + st.header('Queues And State') + st.subheader('Current Queue Files') + display_df(current_queue_counts(queue_dir)) + st.subheader('Queue Snapshots') + snapshots = query_df(conn, ''' + SELECT captured_at, source, phase, todo_count, checked_count, todo_file, checked_file + FROM queue_snapshots + ORDER BY captured_at DESC + LIMIT 1000 + ''') + display_df(snapshots) + state, path = load_runner_state(state_file) + st.subheader('Runner State') + st.caption(path) + if state: + state_rows = [] + for source, item in (state.get('sources') or {}).items(): + state_rows.append({ + 'source': source, + 'query_index': item.get('query_index'), + 'last_query': item.get('last_query'), + 'last_auth': item.get('last_auth'), + 'last_status': item.get('last_status'), + 'cycles': item.get('cycles'), + 'last_scanned': item.get('last_scanned'), + }) + display_df(pd.DataFrame(state_rows)) + else: + st.info('runner_state.json not found or empty') + + +def page_config(conn): + st.header('Config Snapshots') + st.caption('Payloads are unavailable because historical snapshots may contain credentials.') + configs = query_df(conn, ''' + SELECT captured_at, run_id, cycle_id, scope, source + FROM config_snapshots + ORDER BY captured_at DESC, id DESC + LIMIT 500 + ''') + display_df(configs) + + +def page_logs(log_dir): + st.header('Logs') + st.caption('Only file metadata is exposed. Log contents are never rendered or downloaded.') + if not os.path.isdir(log_dir): + st.info('Log directory not found') + return + log_files = sorted(name for name in os.listdir(log_dir) if name.lower().endswith('.log')) + if not log_files: + st.info('No .log files found') + return + rows = [] + for name in log_files: + path = os.path.join(log_dir, name) + updated = datetime.fromtimestamp(os.path.getmtime(path)).isoformat(timespec='seconds') + rows.append({ + 'log': name, + 'bytes': os.path.getsize(path), + 'updated_at': updated, + 'age': human_age(updated), + }) + display_df(pd.DataFrame(rows), height=520) + + +def page_package_repos(conn): + st.header('Package Git Candidates') + candidates = query_df(conn, ''' + SELECT last_seen_at, package_source, package_name, package_version, query, + provider, repo_url, confidence + FROM package_repo_candidates + ORDER BY last_seen_at DESC + LIMIT 2000 + ''') + display_df(candidates) + + +def managed_config_argument(argv=None): + values = list(sys.argv[1:] if argv is None else argv) + for index, value in enumerate(values): + if value == '--config' and index + 1 < len(values): + return values[index + 1] + if str(value).startswith('--config='): + return str(value).split('=', 1)[1] + return None + + +def main(): + try: + require_active_supervisor_child( + managed_config_argument(), + child_kind='dashboard', + require_dsn=True, + ) + dashboard_host = str(os.getenv('TRUF_DASHBOARD_HOST') or '') + if os.getenv('TRUF_DASHBOARD_CANONICAL_LAUNCH') != '1' or not ipaddress.ip_address(dashboard_host).is_loopback: + raise LifecycleAuthorityError('dashboard requires canonical loopback supervisor launch authority') + except (LifecycleAuthorityError, ValueError) as exc: + raise SystemExit(str(exc)) from exc + args = parse_args() + db_path = resolve_db_path(args) + db_url = resolve_db_url(args) + st.set_page_config( + page_title='TRUF Status', + page_icon='T', + layout='wide', + initial_sidebar_state='collapsed', + ) + + try: + conn = connect_db(db_path, db_url, args.immutable_db) + except Exception as exc: + _dashboard_styles() + st.title('TRUF') + st.warning(f'PostgreSQL observability is temporarily unavailable ({type(exc).__name__}). The dashboard will retry on refresh.') + runtime_health, _ = parse_supervisor_status(args.log_dir) + display_df(runtime_health, height=360) + return + if conn is None: + _dashboard_styles() + st.title('TRUF') + st.info(f'No observability database found. Start a configured source through supervisor.py to create {DB_FILENAME}.') + return + + try: + page_simple_dashboard( + conn, + args.log_dir, + args.work_dir, + args.scan_limiter_db, + args.max_active_scans, + ) + finally: + try: + conn.close() + except Exception: + pass + + +if __name__ == '__main__': + main() diff --git a/app/db_backend.py b/app/db_backend.py new file mode 100644 index 0000000..4e9d8f8 --- /dev/null +++ b/app/db_backend.py @@ -0,0 +1,836 @@ +import os +import re +import sqlite3 +import time +from urllib.parse import parse_qsl, quote, unquote, urlencode, urlsplit, urlunsplit + + +POSTGRES_SCHEMES = ('postgresql://', 'postgres://') +DEFAULT_POSTGRES_CONNECT_TIMEOUT_SEC = 10 +DEFAULT_POSTGRES_STATEMENT_TIMEOUT_MS = 30000 +DEFAULT_POSTGRES_LOCK_TIMEOUT_MS = 10000 +DEFAULT_POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS = 30000 +DEFAULT_POSTGRES_TCP_USER_TIMEOUT_MS = 30000 +POSTGRES_APPLICATION_SCHEMA = 'public' +POSTGRES_CHILD_START_RETRY_ATTEMPTS = 3 +POSTGRES_CHILD_START_RETRY_MARKERS = ( + 'server closed the connection unexpectedly', + 'connection reset by peer', + 'connection was forcibly closed by the remote host', +) +HOST_AGENT_POSTGRES_SOCKET_DIRECTORY = '/run/truf-postgres' +HOST_AGENT_POSTGRES_DATABASE = 'truf' +HOST_AGENT_POSTGRES_USER = 'truf' +HOST_AGENT_POSTGRES_PORT = 5432 + + +class DatabaseUrlError(ValueError): + pass + + +def is_postgres_url(value): + return str(value or '').strip().lower().startswith(POSTGRES_SCHEMES) + + +def database_url_from_env(): + # Supervised processes receive TRUF_MANAGED_POSTGRES_DSN as the canonical + # connection authority. Use that same precedence everywhere that derives + # an endpoint identity or opens a managed connection. + return ( + os.getenv('TRUF_MANAGED_POSTGRES_DSN') + or os.getenv('SCANNER_DB_URL') + or os.getenv('DATABASE_URL') + ) + + +def parse_postgres_url(value): + """Parse one URL-form libpq DSN without allowing alternate authorities.""" + text = str(value or '').strip() + if not text or any(character in text for character in ('\x00', '\r', '\n')): + raise DatabaseUrlError('invalid PostgreSQL database URL') + try: + parsed = urlsplit(text) + port = parsed.port or 5432 + host = parsed.hostname or '' + username = unquote(parsed.username or '') + password = unquote(parsed.password or '') + database = unquote((parsed.path or '')[1:]) if (parsed.path or '').startswith('/') else '' + except (TypeError, ValueError) as exc: + raise DatabaseUrlError('invalid PostgreSQL database URL') from exc + if parsed.scheme.lower() not in ('postgresql', 'postgres'): + raise DatabaseUrlError('database URL must use the PostgreSQL scheme') + if parsed.query: + raise DatabaseUrlError('PostgreSQL database URL query parameters are forbidden') + if parsed.fragment: + raise DatabaseUrlError('PostgreSQL database URL fragments are forbidden') + if not parsed.netloc or not host or not username or not database: + raise DatabaseUrlError('PostgreSQL database URL must include one host, user, and database') + authority = parsed.netloc.rsplit('@', 1)[-1] + decoded_authority = unquote(authority) + if ( + ',' in decoded_authority + or any(character.isspace() for character in decoded_authority) + or '%' in authority + or any(character in host for character in (',', '/', '\\', '\x00')) + ): + raise DatabaseUrlError('PostgreSQL database URL must contain exactly one literal host authority') + if not 0 < int(port) <= 65535: + raise DatabaseUrlError('PostgreSQL database URL port is invalid') + if any(character in database for character in ('/', '\\', '?', '#', '\x00')): + raise DatabaseUrlError('PostgreSQL database URL database name contains encoded authority syntax') + if parsed.path.count('/') != 1: + raise DatabaseUrlError('PostgreSQL database URL must contain exactly one database path segment') + return { + 'parsed': parsed, + 'host': host.lower(), + 'port': int(port), + 'database': database, + 'user': username, + 'password': password, + } + + +def canonical_postgres_url(value, database, user, port, host='127.0.0.1'): + """Return one libpq URL whose endpoint cannot be redirected by DSN options.""" + values = parse_postgres_url(value) + expected_host = str(host).lower() + if values['host'] != expected_host or values['port'] != int(port): + raise DatabaseUrlError('PostgreSQL database URL endpoint does not match managed cluster authority') + if values['database'] != str(database) or values['user'] != str(user): + raise DatabaseUrlError('PostgreSQL database URL identity does not match managed cluster authority') + credentials = quote(str(user), safe='') + if values['parsed'].password is not None: + credentials += ':' + quote(values['password'], safe='') + netloc = f'{credentials}@{expected_host}:{int(port)}' + return urlunsplit(('postgresql', netloc, '/' + quote(str(database), safe=''), '', '')) + + +def redact_database_url(value): + text = str(value or '') + if not is_postgres_url(text): + if text.strip().lower().startswith(('postgres', 'postgre')): + return 'postgresql://***' + if '://' in text: + return text.split('://', 1)[0] + '://***' + return text + try: + parsed = urlsplit(text) + username = parsed.username or '' + host = parsed.hostname or '' + port = f':{parsed.port}' if parsed.port else '' + netloc = parsed.netloc + if parsed.password: + netloc = f'{username}:***@{host}{port}' if username else f'***@{host}{port}' + query = [] + for key, value in parse_qsl(parsed.query, keep_blank_values=True): + key_lower = key.lower() + sensitive = any(part in key_lower for part in ('password', 'passwd', 'pwd', 'token', 'secret', 'credential')) + query.append((key, '***' if sensitive else value)) + return urlunsplit((parsed.scheme, netloc, parsed.path, urlencode(query), parsed.fragment)) + except Exception: + return 'postgresql://***' + + +def _split_sql_script(script): + statements = [] + current = [] + quote = None + escape = False + for char in str(script or ''): + current.append(char) + if escape: + escape = False + continue + if char == '\\': + escape = True + continue + if quote: + if char == quote: + quote = None + continue + if char in ("'", '"'): + quote = char + continue + if char == ';': + statement = ''.join(current).strip() + if statement: + statements.append(statement[:-1].strip()) + current = [] + tail = ''.join(current).strip() + if tail: + statements.append(tail) + return [statement for statement in statements if statement] + + +def _convert_qmark_to_psycopg(sql): + out = [] + quote = None + escape = False + for char in str(sql or ''): + if escape: + out.append(char) + escape = False + continue + if char == '\\': + out.append(char) + escape = True + continue + if quote: + out.append('%%' if char == '%' else char) + if char == quote: + quote = None + continue + if char in ("'", '"'): + out.append(char) + quote = char + continue + if char == '?': + out.append('%s') + elif char == '%': + out.append('%%') + else: + out.append(char) + return ''.join(out) + + +def _postgres_schema_sql(script): + converted = str(script or '').replace( + 'INTEGER PRIMARY KEY AUTOINCREMENT', + 'BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY', + ) + # PostgreSQL cannot create either side of the target_queue/target_scans + # cycle with both inline FKs. The offline migration adds this edge after + # both tables exist; SQLite can retain it in the base schema. + converted = converted.replace( + ',\n FOREIGN KEY(queue_id) REFERENCES target_queue(id)', + '', + ) + for future_foreign_key in ( + ',\n FOREIGN KEY(result_reservation_id) REFERENCES result_reservations(id)', + ',\n FOREIGN KEY(reservation_id) REFERENCES result_reservations(id)', + ',\n FOREIGN KEY(current_result_reservation_id) REFERENCES result_reservations(id)', + ',\n FOREIGN KEY(candidate_id) REFERENCES keycheck_candidates(id)', + ',\n FOREIGN KEY(credential_id) REFERENCES keycheck_credentials(id)', + ',\n FOREIGN KEY(last_append_id) REFERENCES projection_appends(id)', + ',\n FOREIGN KEY(experiment_id) REFERENCES docker_depth_experiments(id)', + ): + converted = converted.replace(future_foreign_key, '') + for column in ( + 'run_id', 'cycle_id', 'target_scan_id', 'finding_id', 'keycheck_result_id', + 'last_run_id', 'last_cycle_id', 'queue_id', 'byte_offset', 'line_number', + 'current_result_reservation_id', 'result_reservation_id', 'reservation_id', + 'projection_job_id', 'keycheck_candidate_id', 'candidate_id', 'credential_id', + 'keycheck_result_id', 'target_scan_id', 'job_id', 'last_append_id', + 'last_job_id', 'last_result_id', 'object_id', 'lease_reservation_id', + 'covered_reservation_id', 'declared_bytes', 'verified_bytes', + 'experiment_id', 'pass_id', 'page_id', 'source_cycle_id', 'retry_work_id', + 'repository_queue_id', 'first_cycle_id', 'last_cycle_id', 'first_page_id', + 'last_page_id', 'target_queue_id', 'manifest_id', 'manifest_layer_id', + 'eligibility_page_id', 'experiment_repository_id', 'experiment_target_id', + 'scan_binding_id', 'manifest_size_bytes', 'layer_size_bytes', + 'fence_generation', 'resolver_generation', 'dispatch_order', 'total_count', + 'user_id', 'remote_user_id', 'remote_device_id', + 'expected_revision', 'resulting_revision', 'revision', + 'before_bytes', 'after_bytes', 'previous_event_id', + ): + converted = converted.replace(f'{column} INTEGER', f'{column} BIGINT') + converted = converted.replace( + 'CREATE VIEW IF NOT EXISTS keycheck_latest_state AS', + 'CREATE OR REPLACE VIEW keycheck_latest_state AS', + ) + return converted + + +def _sqlite_check_constraints(sql): + text = str(sql or '') + constraints = {} + index = 0 + position = 0 + quote = None + while position < len(text): + char = text[position] + if quote: + if char == quote: + if position + 1 < len(text) and text[position + 1] == quote: + position += 2 + continue + quote = None + position += 1 + continue + if char in ("'", '"', '`'): + quote = char + position += 1 + continue + if ( + text[position:position + 5].lower() != 'check' + or (position and (text[position - 1].isalnum() or text[position - 1] == '_')) + or ( + position + 5 < len(text) + and (text[position + 5].isalnum() or text[position + 5] == '_') + ) + ): + position += 1 + continue + opening = position + 5 + while opening < len(text) and text[opening].isspace(): + opening += 1 + if opening >= len(text) or text[opening] != '(': + position += 5 + continue + depth = 1 + closing = opening + 1 + expression_quote = None + while closing < len(text) and depth: + current = text[closing] + if expression_quote: + if current == expression_quote: + if closing + 1 < len(text) and text[closing + 1] == expression_quote: + closing += 2 + continue + expression_quote = None + elif current in ("'", '"', '`'): + expression_quote = current + elif current == '(': + depth += 1 + elif current == ')': + depth -= 1 + closing += 1 + if depth: + break + prefix = text[:position] + named = re.search( + r'\bCONSTRAINT\s+(?:"([A-Za-z_][A-Za-z0-9_$]*)"|' + r'([A-Za-z_][A-Za-z0-9_$]*))\s*$', + prefix, + re.IGNORECASE, + ) + name = (named.group(1) or named.group(2)) if named else f'__unnamed_check_{index}' + expression = text[opening + 1:closing - 1] + constraint = { + 'expression': expression, + 'definition': f'CHECK ({expression})', + 'valid': True, + } + if name in constraints: + constraints[name]['valid'] = False + name = f'__duplicate_check_{index}_{name}' + constraint['valid'] = False + constraints[name] = constraint + index += 1 + position = closing + return constraints + + +class DatabaseConnection: + def __init__(self, dialect, conn, application_schema=None): + self.dialect = dialect + self._conn = conn + self.application_schema = application_schema if dialect == 'postgres' else None + + @property + def is_postgres(self): + return self.dialect == 'postgres' + + @property + def is_sqlite(self): + return self.dialect == 'sqlite' + + def execute(self, sql, params=None): + params = tuple(params or ()) + if self.is_postgres: + params = tuple(value.replace('\x00', '') if isinstance(value, str) else value for value in params) + cur = self._conn.cursor() + cur.execute(_convert_qmark_to_psycopg(sql), params) + return cur + return self._conn.execute(sql, params) + + def executescript(self, script): + if self.is_sqlite: + return self._conn.executescript(script) + for statement in _split_sql_script(_postgres_schema_sql(script)): + self.execute(statement) + return None + + def commit(self): + return self._conn.commit() + + def rollback(self): + return self._conn.rollback() + + def close(self): + return self._conn.close() + + def table_columns(self, table): + if self.is_postgres: + rows = self.execute( + '''SELECT a.attname AS name + FROM pg_catalog.pg_attribute a + JOIN pg_catalog.pg_class c ON c.oid = a.attrelid + JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = ? + AND c.relname = ? + AND a.attnum > 0 + AND NOT a.attisdropped''', + (self.application_schema, table), + ).fetchall() + return {row['name'] for row in rows} + return {row['name'] for row in self.execute(f'PRAGMA table_info({table})').fetchall()} + + def table_column_details(self, table): + if self.is_postgres: + rows = self.execute( + '''SELECT a.attname AS name, + pg_catalog.format_type(a.atttypid, a.atttypmod) AS type, + a.attnotnull AS not_null, + a.attidentity AS identity_generation, + a.attgenerated AS generated_kind, + pg_catalog.pg_get_expr(d.adbin, d.adrelid) AS default_sql, + (d.oid IS NOT NULL) AS has_default, + pg_catalog.pg_get_serial_sequence( + pg_catalog.quote_ident(n.nspname) || '.' || pg_catalog.quote_ident(c.relname), + a.attname + ) AS sequence_name, + COALESCE(i.indisprimary, false) AS primary_key + FROM pg_catalog.pg_attribute a + JOIN pg_catalog.pg_class c ON c.oid = a.attrelid + JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + LEFT JOIN pg_catalog.pg_attrdef d ON d.adrelid = a.attrelid AND d.adnum = a.attnum + LEFT JOIN pg_catalog.pg_index i ON i.indrelid = a.attrelid + AND i.indisprimary AND a.attnum = ANY(i.indkey) + WHERE n.nspname = ? + AND c.relname = ? + AND a.attnum > 0 + AND NOT a.attisdropped + ORDER BY a.attnum''', + (self.application_schema, table), + ).fetchall() + return { + row['name']: { + 'type': str(row['type'] or '').lower(), + 'not_null': bool(row['not_null']), + 'default': str(row['default_sql'] or ''), + 'has_default': bool(row['has_default']), + 'primary_key': bool(row['primary_key']), + 'identity': str(row['identity_generation'] or ''), + 'generated': str(row['generated_kind'] or ''), + 'sequence': str(row['sequence_name'] or ''), + } + for row in rows + } + rows = self.execute(f'PRAGMA table_info({table})').fetchall() + return { + row['name']: { + 'type': str(row['type'] or '').lower(), + 'not_null': bool(row['notnull']) or bool(row['pk']), + 'default': str(row['dflt_value'] or ''), + 'has_default': row['dflt_value'] is not None, + 'primary_key': bool(row['pk']), + 'identity': '', + 'generated': '', + 'sequence': '', + } + for row in rows + } + + def table_indexes(self, table): + if self.is_postgres: + rows = self.execute( + '''SELECT idx.relname AS name, + i.indisunique AS is_unique, + i.indisprimary AS is_primary, + i.indisvalid AS is_valid, + i.indisready AS is_ready, + i.indislive AS is_live, + pg_catalog.pg_get_expr(i.indpred, i.indrelid) AS predicate, + ARRAY( + SELECT pg_catalog.pg_get_indexdef(i.indexrelid, position, true) + FROM pg_catalog.generate_series(1, i.indnkeyatts) AS position + ORDER BY position + ) AS columns, + pg_catalog.pg_get_indexdef(i.indexrelid) AS sql + FROM pg_catalog.pg_index i + JOIN pg_catalog.pg_class tbl ON tbl.oid = i.indrelid + JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace + JOIN pg_catalog.pg_class idx ON idx.oid = i.indexrelid + WHERE n.nspname = ? AND tbl.relname = ?''', + (self.application_schema, table), + ).fetchall() + return { + row['name']: { + 'unique': bool(row['is_unique']), + 'primary': bool(row['is_primary']), + 'valid': bool(row['is_valid']), + 'ready': bool(row['is_ready']), + 'live': bool(row['is_live']), + 'predicate': str(row['predicate'] or ''), + 'columns': [str(value).strip('"') for value in (row['columns'] or [])], + 'sql': str(row['sql'] or ''), + } + for row in rows + } + output = {} + for row in self.execute(f'PRAGMA index_list({table})').fetchall(): + name = row['name'] + columns = [item['name'] for item in self.execute(f'PRAGMA index_info({name})').fetchall()] + sql_row = self.execute( + "SELECT sql FROM sqlite_master WHERE type = 'index' AND name = ?", + (name,), + ).fetchone() + sql = str(sql_row['sql'] if sql_row else '') + predicate = sql.split(' WHERE ', 1)[1] if ' WHERE ' in sql.upper() else '' + if ' WHERE ' in sql.upper(): + position = sql.upper().index(' WHERE ') + predicate = sql[position + 7:] + output[name] = { + 'unique': bool(row['unique']), + 'primary': str(row['origin'] or '') == 'pk', + 'valid': True, + 'ready': True, + 'live': True, + 'predicate': predicate, + 'columns': columns, + 'sql': sql, + } + return output + + def table_foreign_keys(self, table): + if self.is_postgres: + rows = self.execute( + '''SELECT con.conname AS name, + ARRAY( + SELECT src.attname + FROM pg_catalog.unnest(con.conkey) WITH ORDINALITY AS keys(attnum, position) + JOIN pg_catalog.pg_attribute src + ON src.attrelid = con.conrelid AND src.attnum = keys.attnum + ORDER BY keys.position + ) AS columns, + ref_n.nspname AS referenced_schema, + ref.relname AS referenced_table, + ARRAY( + SELECT dst.attname + FROM pg_catalog.unnest(con.confkey) WITH ORDINALITY AS keys(attnum, position) + JOIN pg_catalog.pg_attribute dst + ON dst.attrelid = con.confrelid AND dst.attnum = keys.attnum + ORDER BY keys.position + ) AS referenced_columns, + con.confupdtype::text AS update_action, + con.confdeltype::text AS delete_action, + con.convalidated AS is_valid + FROM pg_catalog.pg_constraint con + JOIN pg_catalog.pg_class tbl ON tbl.oid = con.conrelid + JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace + JOIN pg_catalog.pg_class ref ON ref.oid = con.confrelid + JOIN pg_catalog.pg_namespace ref_n ON ref_n.oid = ref.relnamespace + WHERE con.contype = 'f' AND n.nspname = ? AND tbl.relname = ?''', + (self.application_schema, table), + ).fetchall() + action_names = { + 'a': 'NO ACTION', 'r': 'RESTRICT', 'c': 'CASCADE', + 'n': 'SET NULL', 'd': 'SET DEFAULT', + } + return { + row['name']: { + 'columns': [str(value) for value in (row['columns'] or [])], + 'referenced_schema': str(row['referenced_schema'] or ''), + 'referenced_table': str(row['referenced_table'] or ''), + 'referenced_columns': [str(value) for value in (row['referenced_columns'] or [])], + 'update_action': action_names.get(str(row['update_action'] or ''), str(row['update_action'] or '')), + 'delete_action': action_names.get(str(row['delete_action'] or ''), str(row['delete_action'] or '')), + 'valid': bool(row['is_valid']), + } + for row in rows + } + + output = {} + for row in self.execute(f'PRAGMA foreign_key_list({table})').fetchall(): + name = f'fk_{row["id"]}' + current = output.setdefault(name, { + 'columns': [], + 'referenced_schema': 'main', + 'referenced_table': str(row['table'] or ''), + 'referenced_columns': [], + 'update_action': str(row['on_update'] or '').upper(), + 'delete_action': str(row['on_delete'] or '').upper(), + 'valid': True, + }) + current['columns'].append(str(row['from'] or '')) + current['referenced_columns'].append(str(row['to'] or '')) + return output + + def table_check_constraints(self, table): + if self.is_postgres: + rows = self.execute( + '''SELECT con.conname AS name, + pg_catalog.pg_get_expr(con.conbin, con.conrelid, true) AS expression, + pg_catalog.pg_get_constraintdef(con.oid, true) AS definition, + con.convalidated AS is_valid + FROM pg_catalog.pg_constraint con + JOIN pg_catalog.pg_class tbl ON tbl.oid = con.conrelid + JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace + WHERE con.contype = 'c' AND n.nspname = ? AND tbl.relname = ?''', + (self.application_schema, table), + ).fetchall() + return { + str(row['name']): { + 'expression': str(row['expression'] or ''), + 'definition': str(row['definition'] or ''), + 'valid': bool(row['is_valid']), + } + for row in rows + } + row = self.execute( + "SELECT sql FROM sqlite_master WHERE type = 'table' AND name = ?", + (table,), + ).fetchone() + return _sqlite_check_constraints(row['sql'] if row else '') + + def table_triggers(self, table): + if self.is_postgres: + rows = self.execute( + '''SELECT trg.tgname AS name, + trg.tgenabled <> 'D' AS enabled, + pg_catalog.pg_get_triggerdef(trg.oid, true) AS sql, + pg_catalog.pg_get_functiondef(trg.tgfoid) AS function_sql + FROM pg_catalog.pg_trigger trg + JOIN pg_catalog.pg_class tbl ON tbl.oid = trg.tgrelid + JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace + WHERE n.nspname = ? AND tbl.relname = ? + AND NOT trg.tgisinternal''', + (self.application_schema, table), + ).fetchall() + return { + str(row['name']): { + 'enabled': bool(row['enabled']), + 'sql': str(row['sql'] or ''), + 'function_sql': str(row['function_sql'] or ''), + } + for row in rows + } + rows = self.execute( + "SELECT name, sql FROM sqlite_master WHERE type = 'trigger' AND tbl_name = ?", + (table,), + ).fetchall() + return { + str(row['name']): { + 'enabled': True, + 'sql': str(row['sql'] or ''), + 'function_sql': '', + } + for row in rows + } + + def table_exists(self, table): + if self.is_postgres: + row = self.execute( + '''SELECT c.oid AS name FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = ? AND c.relname = ? AND c.relkind IN ('r', 'p')''', + (self.application_schema, table), + ).fetchone() + return bool(row and row['name']) + row = self.execute("SELECT name FROM sqlite_master WHERE type = 'table' AND name = ?", (table,)).fetchone() + return bool(row) + + def insert_returning_id(self, sql, params=None): + if self.is_postgres: + cur = self.execute(f'{sql.rstrip()} RETURNING id', params) + row = cur.fetchone() + return row['id'] if row else None + cur = self.execute(sql, params) + return cur.lastrowid if int(getattr(cur, 'rowcount', 0) or 0) != 0 else None + + def json_extract(self, column, path): + if self.is_sqlite: + return f"json_extract({column}, '{path}')" + parts = str(path or '').lstrip('$.').split('.') + pg_path = ','.join(part for part in parts if part) + return f"(NULLIF({column}, '')::jsonb #>> '{{{pg_path}}}')" + + +def connect_sqlite(path, timeout_sec=30, read_only=False, immutable=False, check_same_thread=True): + if read_only: + params = 'mode=ro&immutable=1' if immutable else 'mode=ro' + uri = 'file:' + str(path).replace('\\', '/') + '?' + params + conn = sqlite3.connect(uri, uri=True, timeout=max(1, int(timeout_sec or 30)), check_same_thread=check_same_thread) + else: + conn = sqlite3.connect(path, timeout=max(1, int(timeout_sec or 30)), check_same_thread=check_same_thread) + conn.row_factory = sqlite3.Row + return DatabaseConnection('sqlite', conn) + + +def _bounded_int(value, default, minimum=1): + try: + return max(minimum, int(value)) + except (TypeError, ValueError): + return max(minimum, int(default)) + + +def connect_postgres( + url, + connect_timeout_sec=None, + statement_timeout_ms=None, + lock_timeout_ms=None, + idle_in_transaction_timeout_ms=None, + tcp_user_timeout_ms=None, +): + parse_postgres_url(url) + try: + import psycopg + from psycopg.rows import dict_row + except ImportError as exc: + raise RuntimeError('PostgreSQL backend requires psycopg[binary]. Install app requirements first.') from exc + connect_timeout_sec = _bounded_int( + connect_timeout_sec if connect_timeout_sec is not None else os.getenv('TRUF_DB_CONNECT_TIMEOUT_SEC'), + DEFAULT_POSTGRES_CONNECT_TIMEOUT_SEC, + ) + statement_timeout_ms = _bounded_int( + statement_timeout_ms if statement_timeout_ms is not None else os.getenv('TRUF_DB_STATEMENT_TIMEOUT_MS'), + DEFAULT_POSTGRES_STATEMENT_TIMEOUT_MS, + ) + lock_timeout_ms = _bounded_int( + lock_timeout_ms if lock_timeout_ms is not None else os.getenv('TRUF_DB_LOCK_TIMEOUT_MS'), + DEFAULT_POSTGRES_LOCK_TIMEOUT_MS, + ) + idle_in_transaction_timeout_ms = _bounded_int( + idle_in_transaction_timeout_ms if idle_in_transaction_timeout_ms is not None else os.getenv('TRUF_DB_IDLE_TRANSACTION_TIMEOUT_MS'), + DEFAULT_POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS, + ) + tcp_user_timeout_ms = _bounded_int( + tcp_user_timeout_ms if tcp_user_timeout_ms is not None else os.getenv('TRUF_DB_TCP_USER_TIMEOUT_MS'), + DEFAULT_POSTGRES_TCP_USER_TIMEOUT_MS, + minimum=1000, + ) + options = ' '.join(( + f'-c search_path={POSTGRES_APPLICATION_SCHEMA}', + f'-c statement_timeout={statement_timeout_ms}', + f'-c lock_timeout={lock_timeout_ms}', + f'-c idle_in_transaction_session_timeout={idle_in_transaction_timeout_ms}', + )) + for attempt in range(POSTGRES_CHILD_START_RETRY_ATTEMPTS): + try: + conn = psycopg.connect( + url, + row_factory=dict_row, + connect_timeout=connect_timeout_sec, + options=options, + tcp_user_timeout=tcp_user_timeout_ms, + keepalives=1, + keepalives_idle=5, + keepalives_interval=5, + keepalives_count=2, + ) + break + except Exception as exc: + transient_child_start = any( + marker in str(exc).lower() for marker in POSTGRES_CHILD_START_RETRY_MARKERS + ) + if not transient_child_start or attempt + 1 >= POSTGRES_CHILD_START_RETRY_ATTEMPTS: + raise + time.sleep(0.05 * (attempt + 1)) + try: + cursor = conn.cursor() + cursor.execute( + """SELECT pg_catalog.current_schema() AS schema_name, + pg_catalog.current_setting('search_path') AS search_path, + EXISTS ( + SELECT 1 + FROM pg_catalog.pg_namespace n + CROSS JOIN LATERAL pg_catalog.aclexplode( + COALESCE(n.nspacl, pg_catalog.acldefault('n', n.nspowner)) + ) acl + WHERE n.nspname = 'public' + AND acl.grantee = 0 + AND acl.privilege_type = 'CREATE' + ) AS public_create""" + ) + row = cursor.fetchone() + cursor.close() + schema_name = row.get('schema_name') if isinstance(row, dict) else row[0] if row else None + search_path = row.get('search_path') if isinstance(row, dict) else row[1] if row else None + public_create = row.get('public_create', False) if isinstance(row, dict) else row[2] if row and len(row) > 2 else False + normalized_path = re.sub(r'[\s\"]', '', str(search_path or '').lower()) + if schema_name != POSTGRES_APPLICATION_SCHEMA or normalized_path != 'public' or bool(public_create): + raise RuntimeError('PostgreSQL application schema/search_path validation failed') + conn.rollback() + except Exception: + try: + conn.close() + except Exception: + pass + raise + return DatabaseConnection('postgres', conn, application_schema=POSTGRES_APPLICATION_SCHEMA) + + +def connect_host_agent_postgres(): + """Open the one fixed peer-authenticated host-agent authority.""" + if os.name != 'posix' or not hasattr(os, 'geteuid') or os.geteuid() != 0: + raise RuntimeError('PostgreSQL host-agent authority requires root on POSIX') + try: + import psycopg + from psycopg.rows import dict_row + except ImportError as exc: + raise RuntimeError( + 'PostgreSQL backend requires psycopg[binary]. Install app requirements first.' + ) from exc + options = ' '.join(( + f'-c search_path={POSTGRES_APPLICATION_SCHEMA}', + f'-c statement_timeout={DEFAULT_POSTGRES_STATEMENT_TIMEOUT_MS}', + f'-c lock_timeout={DEFAULT_POSTGRES_LOCK_TIMEOUT_MS}', + f'-c idle_in_transaction_session_timeout={DEFAULT_POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS}', + )) + conn = psycopg.connect( + dbname=HOST_AGENT_POSTGRES_DATABASE, + user=HOST_AGENT_POSTGRES_USER, + host=HOST_AGENT_POSTGRES_SOCKET_DIRECTORY, + port=HOST_AGENT_POSTGRES_PORT, + row_factory=dict_row, + connect_timeout=DEFAULT_POSTGRES_CONNECT_TIMEOUT_SEC, + options=options, + sslmode='disable', + ) + try: + cursor = conn.cursor() + cursor.execute( + """SELECT pg_catalog.current_schema() AS schema_name, + pg_catalog.current_setting('search_path') AS search_path, + CURRENT_USER AS current_user, + current_database() AS database_name, + pg_catalog.inet_server_addr() IS NULL AS unix_socket, + pg_catalog.current_setting('port')::integer AS port, + EXISTS ( + SELECT 1 + FROM pg_catalog.pg_namespace n + CROSS JOIN LATERAL pg_catalog.aclexplode( + COALESCE(n.nspacl, pg_catalog.acldefault('n', n.nspowner)) + ) acl + WHERE n.nspname = 'public' + AND acl.grantee = 0 + AND acl.privilege_type = 'CREATE' + ) AS public_create""" + ) + row = cursor.fetchone() + cursor.close() + normalized_path = re.sub( + r'[\s\"]', '', str((row or {}).get('search_path') or '').lower() + ) + if ( + not isinstance(row, dict) + or row.get('schema_name') != POSTGRES_APPLICATION_SCHEMA + or normalized_path != POSTGRES_APPLICATION_SCHEMA + or row.get('current_user') != HOST_AGENT_POSTGRES_USER + or row.get('database_name') != HOST_AGENT_POSTGRES_DATABASE + or row.get('unix_socket') is not True + or row.get('port') != HOST_AGENT_POSTGRES_PORT + or bool(row.get('public_create')) + ): + raise RuntimeError('PostgreSQL host-agent authority validation failed') + conn.rollback() + except Exception: + try: + conn.close() + except Exception: + pass + raise + return DatabaseConnection( + 'postgres', conn, application_schema=POSTGRES_APPLICATION_SCHEMA, + ) diff --git a/app/docker_depth_experiment.py b/app/docker_depth_experiment.py new file mode 100644 index 0000000..7e67065 --- /dev/null +++ b/app/docker_depth_experiment.py @@ -0,0 +1,3898 @@ +import copy +from datetime import datetime, timezone +import hashlib +import inspect +import io +import json +import math +import os +import re +import textwrap +import tokenize +from dataclasses import dataclass +from typing import Optional + +from target_identity import ( + DOCKER_DIGEST_RE, + DOCKER_IMAGE_RE, + DOCKER_REVISION_RE, + DOCKER_TAG_TARGET_SCHEMA, + normalize_docker_digest, + parse_docker_target, + serialize_docker_tag_target, + validate_docker_image_reference, +) + + +DOCKER_DEPTH_SELECTOR_VERSION = 'docker-layer-graph-v2' +DOCKER_RANK1_BREADTH_SELECTOR_VERSION = 'docker-rank1-breadth-v1' +DOCKER_DEPTH_QUERY_COUNT = 61 +DOCKER_DEPTH_ORDINARY_IMAGES_PER_REPOSITORY = 3 +DOCKER_DEPTH_REPOSITORIES_PER_QUERY = 10 +DOCKER_DEPTH_SHALLOW_IMAGES_PER_REPOSITORY = 1 +DOCKER_DEPTH_DEEP_REPOSITORIES_PER_QUERY = 1 +DOCKER_DEPTH_DEEP_IMAGES_PER_REPOSITORY = 10 +DOCKER_DEPTH_MAX_UNIQUE_TARGETS = 1200 +DOCKER_RANK1_BREADTH_REPOSITORIES_PER_QUERY = 39 +DOCKER_RANK1_BREADTH_MAX_UNIQUE_TARGETS = 2000 +DOCKERHUB_DISCOVERY_MAX_PAGES = 30 +DOCKERHUB_DISCOVERY_MAX_PER_PAGE = 100 +DOCKERHUB_DISCOVERY_ALGORITHM_VERSION = 1 +DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS = 250000 +DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS = 3 +DOCKER_DEPTH_RESOLVER_RETRY_SECONDS = 300 +DOCKER_DEPTH_RESOLVER_RETRY_MAX_SECONDS = 3600 +DOCKER_DEPTH_COLLECTION_GENERATION = 'docker-depth-provenance-v1' +DOCKER_DEPTH_COHORT_PLAN_TYPE = 'truf-docker-depth-cohort-plan-v2' +DOCKER_DEPTH_COHORT_MANIFEST_TYPE = 'truf-docker-depth-cohort-review-v2' +DOCKER_DEPTH_HOLD_MANIFEST_TYPE = 'truf-docker-depth-hold-review-v1' +DOCKER_DEPTH_REACTIVATION_MANIFEST_TYPE = 'truf-docker-depth-reactivation-review-v1' +DOCKER_DEPTH_RESOLVER_REFUND_MANIFEST_TYPE = ( + 'truf-docker-depth-resolver-attempt-refund-v1' +) +DOCKER_DEPTH_RESOLVER_REFUND_KIND = 'zero_graph_limit_v1' +DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS = 2 +DOCKER_DEPTH_RESOLVER_REFUND_LOG_MAX_BYTES = 64 * 1024 * 1024 +DOCKER_DEPTH_RESOLVER_REFUND_OLD_ERROR = ( + 'Docker replacement candidate pool must contain 1 through 100 graphs' +) +DOCKER_DEPTH_HOLD_REASON = 'docker_depth_experiment_hold' +DOCKER_DEPTH_DYNAMIC_HOLD_REASON = 'docker_depth_experiment_dynamic_hold' +DOCKER_DEPTH_RELEASE_REASON = 'docker_depth_experiment_reviewed_release' +DOCKER_DEPTH_REPOSITORY_SKIP_REASON = 'no_eligible_physical_target' +DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON = 'remote_unavailable_after_attempt_limit' +DOCKER_DEPTH_RESOLVER_DISPOSITION_MANIFEST_TYPE = ( + 'truf-docker-depth-resolver-disposition-v1' +) +DOCKER_DEPTH_RESOLVER_DISPOSITION_KIND = ( + 'resolver_attempt_limit_replace_or_skip_v1' +) +DOCKER_DEPTH_HOLD_ACTIVE_STATES = frozenset({ + 'holding', 'resolving', 'active', 'draining', 'completed', 'held', +}) + +_DOCKER_EXPERIMENT_PROFILES = { + DOCKER_DEPTH_SELECTOR_VERSION: { + 'query_count': DOCKER_DEPTH_QUERY_COUNT, + 'repositories_per_query': DOCKER_DEPTH_REPOSITORIES_PER_QUERY, + 'shallow_images_per_repository': DOCKER_DEPTH_SHALLOW_IMAGES_PER_REPOSITORY, + 'deep_repositories_per_query': DOCKER_DEPTH_DEEP_REPOSITORIES_PER_QUERY, + 'images_per_repository': DOCKER_DEPTH_DEEP_IMAGES_PER_REPOSITORY, + 'target_limit': DOCKER_DEPTH_MAX_UNIQUE_TARGETS, + 'theoretical_max_targets': DOCKER_DEPTH_QUERY_COUNT * ( + DOCKER_DEPTH_REPOSITORIES_PER_QUERY + + DOCKER_DEPTH_DEEP_IMAGES_PER_REPOSITORY - 1 + ), + }, + DOCKER_RANK1_BREADTH_SELECTOR_VERSION: { + 'query_count': DOCKER_DEPTH_QUERY_COUNT, + 'repositories_per_query': DOCKER_RANK1_BREADTH_REPOSITORIES_PER_QUERY, + 'shallow_images_per_repository': 1, + 'deep_repositories_per_query': 1, + 'images_per_repository': 1, + 'target_limit': DOCKER_RANK1_BREADTH_MAX_UNIQUE_TARGETS, + 'theoretical_max_targets': DOCKER_RANK1_BREADTH_MAX_UNIQUE_TARGETS, + }, +} + + +def reviewed_docker_experiment_profile(selector_version): + try: + return dict(_DOCKER_EXPERIMENT_PROFILES[selector_version]) + except KeyError: + raise ValueError('Docker experiment selector_version is not reviewed') from None + +DOCKER_DEPTH_EXPERIMENT_KEYS = frozenset({ + 'experiment_key', + 'enabled', + 'queries', + 'repositories_per_query', + 'shallow_images_per_repository', + 'deep_repositories_per_query', + 'deep_images_per_repository', + 'target_limit', + 'selector_version', +}) + +_IDENTIFIER_RE = re.compile(r'^[a-z0-9](?:[a-z0-9._-]{0,126}[a-z0-9])?$') +_PLATFORM_COMPONENT_RE = re.compile(r'^[a-z0-9][a-z0-9._-]{0,63}$') +_POSTGRES_URL_RE = re.compile(r'^postgres(?:ql)?://', re.IGNORECASE) +_SELECTOR_DISTINCT_GRAPH = 'ordered-layer-digests' +_SELECTOR_FIRST_THREE_STEPS = ( + 'newest_distinct_graph', + 'maximum_marginal_layer_novelty', + 'oldest_distinct_graph', +) +_SELECTOR_LATER_SCORE = 'maximum_marginal_layer_novelty' +_SELECTOR_LATER_TIE_BREAKS = ( + 'maximum_minimum_temporal_distance', + 'newest_update', + 'normalized_target', + 'ordered_graph_sha256', +) + + +@dataclass(frozen=True) +class DockerDepthExperimentConfig: + experiment_key: str + enabled: bool + collection_generation: str + queries: tuple + repositories_per_query: int + shallow_images_per_repository: int + deep_repositories_per_query: int + deep_images_per_repository: int + target_limit: int + selector_version: str + theoretical_max_targets: int + ordered_query_hash: str + config_hash: str + selector_hash: str + + @property + def ordered_query_sha256(self): + return self.ordered_query_hash + + @property + def ordered_queries_sha256(self): + return self.ordered_query_hash + + @property + def config_sha256(self): + return self.config_hash + + @property + def selector_sha256(self): + return self.selector_hash + + +@dataclass(frozen=True) +class ValidatedDockerDepthConfig: + normalized_config: dict + docker_images_per_repository: int + experiment: Optional[DockerDepthExperimentConfig] = None + ordered_query_hash: str = '' + config_hash: str = '' + selector_hash: str = '' + + @property + def config(self): + return copy.deepcopy(self.normalized_config) + + @property + def ordered_query_sha256(self): + return self.ordered_query_hash + + @property + def ordered_queries_sha256(self): + return self.ordered_query_hash + + @property + def config_sha256(self): + return self.config_hash + + @property + def selector_sha256(self): + return self.selector_hash + + +def _canonical_sha256(value): + payload = json.dumps( + value, + ensure_ascii=True, + allow_nan=False, + sort_keys=True, + separators=(',', ':'), + ).encode('utf-8') + return hashlib.sha256(payload).hexdigest() + + +def canonical_ordered_query_hash(queries): + return _canonical_sha256(list(queries)) + + +def canonical_dockerhub_discovery_policy(pages, per_page, sort_by, sort_order): + if isinstance(pages, bool) or not isinstance(pages, int) or not 1 <= pages <= DOCKERHUB_DISCOVERY_MAX_PAGES: + raise ValueError( + f'DockerHub discovery pages must be an integer from 1 through {DOCKERHUB_DISCOVERY_MAX_PAGES}' + ) + if ( + isinstance(per_page, bool) + or not isinstance(per_page, int) + or not 1 <= per_page <= DOCKERHUB_DISCOVERY_MAX_PER_PAGE + ): + raise ValueError( + 'DockerHub discovery per_page must be an integer from 1 through ' + f'{DOCKERHUB_DISCOVERY_MAX_PER_PAGE}' + ) + if not isinstance(sort_by, str) or not sort_by or sort_by != sort_by.strip(): + raise ValueError('DockerHub discovery sort_by must be a non-empty canonical string') + if sort_order not in ('asc', 'desc'): + raise ValueError("DockerHub discovery sort_order must be 'asc' or 'desc'") + payload = { + 'algorithm_version': DOCKERHUB_DISCOVERY_ALGORITHM_VERSION, + 'page_hard_cap': DOCKERHUB_DISCOVERY_MAX_PAGES, + 'pages': pages, + 'per_page': per_page, + 'per_page_hard_cap': DOCKERHUB_DISCOVERY_MAX_PER_PAGE, + 'sort_by': sort_by, + 'sort_order': sort_order, + } + return {**payload, 'policy_sha256': _canonical_sha256(payload)} + + +def canonical_selector_hash(selector_version): + return _canonical_sha256({ + 'identity': { + 'dependencies': { + 'docker_digest_pattern': DOCKER_DIGEST_RE.pattern, + 'docker_image_pattern': DOCKER_IMAGE_RE.pattern, + 'docker_revision_pattern': DOCKER_REVISION_RE.pattern, + 'docker_target_schema': DOCKER_TAG_TARGET_SCHEMA, + }, + 'distinct_graph': _SELECTOR_DISTINCT_GRAPH, + 'first_three': _SELECTOR_FIRST_THREE_STEPS, + 'implementation_sha256': _selector_implementation_sha256(), + 'later_score': _SELECTOR_LATER_SCORE, + 'later_tie_breaks': _SELECTOR_LATER_TIE_BREAKS, + }, + 'version': selector_version, + }) + + +def canonical_docker_layer_graph_hash(layers): + return _canonical_sha256(list(layers)) + + +def canonical_docker_descriptor_hash(digest, media_type, size_bytes): + return _canonical_sha256({ + 'digest': str(digest), + 'media_type': str(media_type), + 'size_bytes': int(size_bytes), + }) + + +def canonical_docker_depth_selection_evidence_hash(evidence): + normalized = dict(evidence) + # Repository tag churn may change this audit metric after rank-one selection. + normalized.pop('candidate_distinct_graph_count', None) + return _canonical_sha256(normalized) + + +def validate_docker_images_per_repository(value): + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError('docker_images_per_repository must be an integer from 1 through 10') + if value < 1 or value > 10: + raise ValueError('docker_images_per_repository must be an integer from 1 through 10') + return value + + +def _selector_graph_layers(candidate): + layers = candidate.get('layers') + if not isinstance(layers, (list, tuple)) or not layers: + return None + graph = [] + for layer in layers: + digest = normalize_docker_digest( + layer.get('digest') if isinstance(layer, dict) else layer + ) + if not digest: + return None + graph.append(digest) + return tuple(graph) + + +def _selector_graph_timestamp(value): + if isinstance(value, bool): + return float('-inf') + try: + timestamp = float(value) + return timestamp if math.isfinite(timestamp) else float('-inf') + except (TypeError, ValueError, OverflowError): + pass + if isinstance(value, str): + try: + parsed = datetime.fromisoformat(value.strip().replace('Z', '+00:00')) + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.timestamp() + except (ValueError, OverflowError, OSError): + pass + return float('-inf') + + +def _selector_graph_target(candidate): + target = str(candidate.get('target') or '').strip() + try: + return parse_docker_target(target)['target'].lower() + except (TypeError, ValueError): + return target.lower() + + +def _selector_graph_selection_record(candidate, rank, reason, marginal_layers): + record = { + key: value for key, value in candidate.items() + if not key.startswith('_selector_') and key != 'layer_descriptors' + } + graph = candidate['_selector_graph'] + layer_count = len(graph) + descriptors = candidate.get('layer_descriptors') + if isinstance(descriptors, (tuple, list)) and len(descriptors) == layer_count: + layer_metadata = tuple({ + 'digest': descriptor['digest'], + 'media_type': descriptor['media_type'], + 'size_bytes': descriptor['size'], + 'descriptor_sha256': canonical_docker_descriptor_hash( + descriptor['digest'], descriptor['media_type'], descriptor['size'], + ), + 'position_from_base': position, + 'position_from_top': layer_count - position + 1, + } for position, descriptor in enumerate(descriptors, 1)) + else: + layer_metadata = tuple({ + 'digest': digest, + 'position_from_base': position, + 'position_from_top': layer_count - position + 1, + } for position, digest in enumerate(graph, 1)) + record.update({ + 'rank': rank, + 'image_rank': rank, + 'reason': reason, + 'selection_reason': reason, + 'graph': graph, + 'graph_hash': candidate['_selector_graph_hash'], + 'graph_sha256': candidate['_selector_graph_hash'], + 'layers': graph, + 'layer_count': layer_count, + 'marginal_layer_count': marginal_layers, + 'layer_metadata': layer_metadata, + }) + return record + + +def select_docker_layer_graphs(candidates, limit=1, *, replacement_pool=False): + if replacement_pool: + if isinstance(limit, bool) or not isinstance(limit, int) or not 1 <= limit <= 100: + raise ValueError('Docker replacement candidate pool must contain 1 through 100 graphs') + else: + limit = validate_docker_images_per_repository(limit) + + def source_index(candidate): + value = candidate.get('source_index', 0) + if isinstance(value, bool): + return 0 + try: + return int(value) + except (TypeError, ValueError, OverflowError): + return 0 + + def order_key(candidate): + return ( + -candidate['_selector_updated_at'], + source_index(candidate), + _selector_graph_target(candidate), + ) + + valid = [] + for item in candidates or (): + if not isinstance(item, dict): + continue + candidate = dict(item) + graph = _selector_graph_layers(candidate) + if graph is None: + continue + candidate['_selector_graph'] = graph + candidate['_selector_graph_hash'] = canonical_docker_layer_graph_hash(graph) + candidate['_selector_updated_at'] = _selector_graph_timestamp(candidate.get('updated_at')) + valid.append(candidate) + + distinct = [] + seen_graphs = set() + for candidate in sorted(valid, key=order_key): + graph = candidate['_selector_graph'] + if graph in seen_graphs: + continue + seen_graphs.add(graph) + distinct.append(candidate) + if not distinct: + return [] + + selected = [] + first = distinct.pop(0) + covered_layers = set(first['_selector_graph']) + selected.append((first, _SELECTOR_FIRST_THREE_STEPS[0], len(covered_layers))) + + if limit >= 2 and distinct: + index = max( + range(len(distinct)), + key=lambda position: ( + len(set(distinct[position]['_selector_graph']) - covered_layers), + -position, + ), + ) + candidate = distinct.pop(index) + marginal = len(set(candidate['_selector_graph']) - covered_layers) + selected.append((candidate, _SELECTOR_FIRST_THREE_STEPS[1], marginal)) + covered_layers.update(candidate['_selector_graph']) + + if limit >= 3 and distinct: + candidate = distinct.pop() + marginal = len(set(candidate['_selector_graph']) - covered_layers) + selected.append((candidate, _SELECTOR_FIRST_THREE_STEPS[2], marginal)) + covered_layers.update(candidate['_selector_graph']) + + while len(selected) < limit and distinct: + selected_times = [ + candidate['_selector_updated_at'] for candidate, _reason, _marginal in selected + if math.isfinite(candidate['_selector_updated_at']) + ] + + def later_key(candidate): + updated_at = candidate['_selector_updated_at'] + temporal_distance = ( + min(abs(updated_at - selected_at) for selected_at in selected_times) + if math.isfinite(updated_at) and selected_times + else float('-inf') + ) + return ( + -len(set(candidate['_selector_graph']) - covered_layers), + -temporal_distance, + -updated_at, + _selector_graph_target(candidate), + candidate['_selector_graph_hash'], + ) + + candidate = min(distinct, key=later_key) + distinct.remove(candidate) + marginal = len(set(candidate['_selector_graph']) - covered_layers) + selected.append((candidate, _SELECTOR_LATER_SCORE, marginal)) + covered_layers.update(candidate['_selector_graph']) + + return [ + _selector_graph_selection_record(candidate, rank, reason, marginal) + for rank, (candidate, reason, marginal) in enumerate(selected, 1) + ] + + +def _canonical_selector_source(function): + try: + source = textwrap.dedent(inspect.getsource(function)) + except (OSError, TypeError) as exc: + raise RuntimeError('Docker depth selector source authority is unavailable') from exc + canonical = [] + ignored = { + tokenize.COMMENT, tokenize.ENCODING, tokenize.ENDMARKER, tokenize.NL, + } + structural = {tokenize.INDENT, tokenize.DEDENT, tokenize.NEWLINE} + for token in tokenize.generate_tokens(io.StringIO(source).readline): + if token.type in ignored: + continue + if token.type in structural: + canonical.append(tokenize.tok_name[token.type]) + else: + canonical.append(f'{tokenize.tok_name[token.type]}:{token.string}') + return '\n'.join(canonical) + + +def _selector_implementation_sha256(source_functions=None): + if source_functions is None: + source_functions = ( + _canonical_sha256, + canonical_docker_layer_graph_hash, + canonical_docker_descriptor_hash, + validate_docker_images_per_repository, + normalize_docker_digest, + validate_docker_image_reference, + serialize_docker_tag_target, + parse_docker_target, + _selector_graph_layers, + _selector_graph_timestamp, + _selector_graph_target, + _selector_graph_selection_record, + select_docker_layer_graphs, + ) + payload = '\nFUNCTION\n'.join( + _canonical_selector_source(function) for function in source_functions + ).encode('utf-8') + return hashlib.sha256(payload).hexdigest() + + +def _strict_integer(mapping, key, minimum=1, maximum=None): + value = mapping.get(key) + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f'docker_depth_experiment.{key} must be an integer') + if value < minimum or (maximum is not None and value > maximum): + upper = f' through {maximum}' if maximum is not None else '' + raise ValueError( + f'docker_depth_experiment.{key} must be from {minimum}{upper}' + ) + return value + + +def _strict_identifier(value, name): + if not isinstance(value, str) or not _IDENTIFIER_RE.fullmatch(value): + raise ValueError( + f'docker_depth_experiment.{name} must be a lowercase stable identifier' + ) + return value + + +def _strict_config_integer(value, name, minimum=0, maximum=None): + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f'{name} must be an integer') + if value < minimum or (maximum is not None and value > maximum): + upper = f' through {maximum}' if maximum is not None else ' or greater' + raise ValueError(f'{name} must be from {minimum}{upper}') + return value + + +def validate_dockerhub_discovery_policies(source, queries=None): + """Return strict ordered effective query policies without runtime side effects.""" + if not isinstance(source, dict): + raise ValueError('sources.dockerhub must be a mapping') + configured_queries = source.get('queries') if queries is None else list(queries) + if not isinstance(configured_queries, list): + raise ValueError('sources.dockerhub.queries must be an ordered list') + configured_queries = tuple(configured_queries) + if any( + not isinstance(query, str) + or not query + or query != query.strip() + for query in configured_queries + ): + raise ValueError('sources.dockerhub.queries contains a malformed query') + if len(set(configured_queries)) != len(configured_queries): + raise ValueError('sources.dockerhub.queries must be unique') + + pages = _strict_config_integer( + source.get('pages', 1), 'sources.dockerhub.pages', 1, + DOCKERHUB_DISCOVERY_MAX_PAGES, + ) + per_page = _strict_config_integer( + source.get('per_page', 50), 'sources.dockerhub.per_page', 1, + DOCKERHUB_DISCOVERY_MAX_PER_PAGE, + ) + max_targets = _strict_config_integer( + source.get('max_targets', 0), 'sources.dockerhub.max_targets', 0, + ) + sort_by = source.get('docker_sort_by', source.get('sort_by', 'updated_at')) + sort_order = source.get('sort_order', 'desc') + canonical_dockerhub_discovery_policy(pages, per_page, sort_by, sort_order) + + overrides = source.get('query_overrides', {}) + if not isinstance(overrides, dict): + raise ValueError('sources.dockerhub.query_overrides must be a mapping') + unknown_queries = [query for query in overrides if query not in configured_queries] + if unknown_queries: + names = ', '.join(sorted(repr(query) for query in unknown_queries)) + raise ValueError( + 'sources.dockerhub.query_overrides contains unconfigured queries: ' + names + ) + + effective = [] + for query in configured_queries: + override = overrides.get(query, {}) + if not isinstance(override, dict): + raise ValueError( + f'sources.dockerhub.query override for {query!r} must be a mapping' + ) + unknown = set(override) - {'pages', 'per_page', 'max_targets'} + if unknown: + raise ValueError( + f'sources.dockerhub query override for {query!r} has unsupported keys: ' + + ', '.join(sorted(repr(key) for key in unknown)) + ) + effective_pages = _strict_config_integer( + override.get('pages', pages), + f'sources.dockerhub query override pages for {query!r}', 1, + DOCKERHUB_DISCOVERY_MAX_PAGES, + ) + effective_per_page = _strict_config_integer( + override.get('per_page', per_page), + f'sources.dockerhub query override per_page for {query!r}', 1, + DOCKERHUB_DISCOVERY_MAX_PER_PAGE, + ) + effective_max_targets = _strict_config_integer( + override.get('max_targets', max_targets), + f'sources.dockerhub query override max_targets for {query!r}', 0, + ) + policy = canonical_dockerhub_discovery_policy( + effective_pages, effective_per_page, sort_by, sort_order, + ) + effective.append({ + 'query': query, + **policy, + 'max_targets': effective_max_targets, + }) + return tuple(effective) + + +def _docker_source(config): + sources = config.get('sources') + if sources is None: + sources = {} + if not isinstance(sources, dict): + raise ValueError('sources must be a mapping') + source = sources.get('dockerhub') + if source is None: + source = {} + if not isinstance(source, dict): + raise ValueError('sources.dockerhub must be a mapping') + return source + + +def validate_docker_depth_config( + config, *, managed_postgres=None, final_cutover=None, +): + """Validate and copy Docker depth configuration without runtime side effects.""" + if not isinstance(config, dict): + raise ValueError('configuration root must be a mapping') + normalized = copy.deepcopy(config) + global_config = normalized.get('global') + if global_config is None: + global_config = {} + if not isinstance(global_config, dict): + raise ValueError('global must be a mapping') + source = _docker_source(normalized) + + configured_depths = [] + if 'docker_images_per_repository' in global_config: + configured_depths.append((global_config, 'global')) + if 'docker_images_per_repository' in source: + configured_depths.append((source, 'sources.dockerhub')) + for owner, _name in configured_depths: + validate_docker_images_per_repository(owner['docker_images_per_repository']) + image_depth = source.get( + 'docker_images_per_repository', + global_config.get('docker_images_per_repository', 1), + ) + image_depth = validate_docker_images_per_repository(image_depth) + + experiment_mapping = source.get('docker_depth_experiment') + if experiment_mapping is None: + return ValidatedDockerDepthConfig(normalized, image_depth) + if not isinstance(experiment_mapping, dict): + raise ValueError('docker_depth_experiment must be a mapping') + + unknown = sorted(set(experiment_mapping) - DOCKER_DEPTH_EXPERIMENT_KEYS) + if unknown: + raise ValueError( + 'docker_depth_experiment has unsupported keys: ' + ', '.join(unknown) + ) + missing = sorted(DOCKER_DEPTH_EXPERIMENT_KEYS - set(experiment_mapping)) + if missing: + raise ValueError( + 'docker_depth_experiment is missing required keys: ' + ', '.join(missing) + ) + + experiment_key = _strict_identifier( + experiment_mapping['experiment_key'], 'experiment_key', + ) + enabled = experiment_mapping['enabled'] + if not isinstance(enabled, bool): + raise ValueError('docker_depth_experiment.enabled must be a boolean') + selector_version = _strict_identifier( + experiment_mapping['selector_version'], 'selector_version', + ) + profile = reviewed_docker_experiment_profile(selector_version) + + queries_value = experiment_mapping['queries'] + if not isinstance(queries_value, list): + raise ValueError('docker_depth_experiment.queries must be an ordered list') + queries = tuple(queries_value) + if len(queries) != DOCKER_DEPTH_QUERY_COUNT: + raise ValueError( + f'docker_depth_experiment.queries must contain exactly {DOCKER_DEPTH_QUERY_COUNT} queries' + ) + if any( + not isinstance(query, str) + or not query + or query != query.strip() + for query in queries + ): + raise ValueError('docker_depth_experiment.queries contains a malformed query') + if len(set(queries)) != len(queries): + raise ValueError('docker_depth_experiment.queries must be unique') + source_queries = source.get('queries') + if not isinstance(source_queries, list) or tuple(source_queries) != queries: + raise ValueError( + 'docker_depth_experiment.queries must exactly match ordered sources.dockerhub.queries' + ) + + repositories_per_query = _strict_integer( + experiment_mapping, 'repositories_per_query', maximum=100, + ) + shallow_images = _strict_integer( + experiment_mapping, 'shallow_images_per_repository', maximum=10, + ) + deep_repositories = _strict_integer( + experiment_mapping, 'deep_repositories_per_query', maximum=100, + ) + deep_images = _strict_integer( + experiment_mapping, 'deep_images_per_repository', maximum=10, + ) + target_limit = _strict_integer( + experiment_mapping, 'target_limit', maximum=1000000, + ) + if deep_repositories > repositories_per_query: + raise ValueError( + 'docker_depth_experiment deep repositories exceed repositories_per_query' + ) + if shallow_images > deep_images: + raise ValueError( + 'docker_depth_experiment shallow image depth exceeds deep image depth' + ) + theoretical_max = len(queries) * ( + repositories_per_query * shallow_images + + deep_repositories * (deep_images - shallow_images) + ) + if ( + selector_version == DOCKER_DEPTH_SELECTOR_VERSION + and theoretical_max > target_limit + ): + raise ValueError( + 'docker_depth_experiment theoretical target maximum exceeds target_limit' + ) + + fixed_limits = { + 'repositories_per_query': profile['repositories_per_query'], + 'shallow_images_per_repository': profile['shallow_images_per_repository'], + 'deep_repositories_per_query': profile['deep_repositories_per_query'], + 'deep_images_per_repository': profile['images_per_repository'], + 'target_limit': profile['target_limit'], + } + for key, expected in fixed_limits.items(): + if experiment_mapping[key] != expected: + raise ValueError( + f'docker_depth_experiment.{key} must equal {expected}' + ) + theoretical_max = profile['theoretical_max_targets'] + if image_depth != DOCKER_DEPTH_ORDINARY_IMAGES_PER_REPOSITORY: + raise ValueError( + 'docker_images_per_repository must equal the reviewed ordinary ' + f'resolver depth {DOCKER_DEPTH_ORDINARY_IMAGES_PER_REPOSITORY}' + ) + + effective_policies = validate_dockerhub_discovery_policies(source, queries) + if len({policy['policy_sha256'] for policy in effective_policies}) != 1: + raise ValueError( + 'docker_depth_experiment requires one consistent pages/per_page ' + 'discovery policy across all queries' + ) + + configured_platform_filters = [] + if 'docker_platform_filter_enabled' in global_config: + configured_platform_filters.append(global_config['docker_platform_filter_enabled']) + if 'docker_platform_filter_enabled' in source: + configured_platform_filters.append(source['docker_platform_filter_enabled']) + if any(not isinstance(value, bool) for value in configured_platform_filters): + raise ValueError('docker_platform_filter_enabled must be a boolean') + platform_filter_enabled = source.get( + 'docker_platform_filter_enabled', + global_config.get('docker_platform_filter_enabled', True), + ) + + def platform_component(key, default): + configured = [] + if key in global_config: + configured.append(global_config[key]) + if key in source: + configured.append(source[key]) + for value in configured: + if ( + not isinstance(value, str) + or not _PLATFORM_COMPONENT_RE.fullmatch(value) + or value != value.lower() + ): + raise ValueError(f'{key} must be a lowercase Docker platform identifier') + return source.get(key, global_config.get(key, default)) + + platform_os = platform_component('docker_platform_os', 'linux') + platform_arch = platform_component('docker_platform_arch', 'amd64') + if platform_filter_enabled and (platform_os, platform_arch) != ('linux', 'amd64'): + raise ValueError( + 'docker_depth_experiment platform filtering supports only linux/amd64' + ) + + configured_candidate_counts = [] + if 'docker_platform_candidate_tags' in global_config: + configured_candidate_counts.append(( + global_config['docker_platform_candidate_tags'], + 'global.docker_platform_candidate_tags', + )) + if 'docker_platform_candidate_tags' in source: + configured_candidate_counts.append(( + source['docker_platform_candidate_tags'], + 'sources.dockerhub.docker_platform_candidate_tags', + )) + for value, name in configured_candidate_counts: + _strict_config_integer(value, name, 1, 100) + candidate_tags = source.get( + 'docker_platform_candidate_tags', + global_config.get('docker_platform_candidate_tags', 20), + ) + candidate_tags = _strict_config_integer( + candidate_tags, 'docker_platform_candidate_tags', 1, 100, + ) + if candidate_tags < deep_images: + raise ValueError( + 'docker_platform_candidate_tags must be at least the configured deep image depth' + ) + + for key in ( + 'docker_repository_refresh_interval_sec', + 'docker_repository_refresh_max_per_cycle', + ): + for owner, name in ((global_config, 'global'), (source, 'sources.dockerhub')): + if key in owner: + _strict_config_integer(owner[key], f'{name}.{key}') + refresh_max = source.get( + 'docker_repository_refresh_max_per_cycle', + global_config.get('docker_repository_refresh_max_per_cycle', 0), + ) + if enabled and refresh_max != 0: + raise ValueError( + 'docker_depth_experiment requires ' + 'docker_repository_refresh_max_per_cycle=0' + ) + + database_url = global_config.get('database_url') + if database_url not in (None, ''): + if not isinstance(database_url, str) or not _POSTGRES_URL_RE.match(database_url): + raise ValueError( + 'docker_depth_experiment collection requires a PostgreSQL database_url when configured' + ) + if managed_postgres is None: + managed_postgres = True + if managed_postgres is not None and not isinstance(managed_postgres, bool): + raise ValueError('managed_postgres validation authority must be boolean or None') + configured_final_cutover = source.get( + 'sync_file_queues', global_config.get('sync_file_queues', True), + ) is False + if final_cutover is None: + final_cutover = configured_final_cutover + elif not isinstance(final_cutover, bool): + raise ValueError('final_cutover validation authority must be boolean or None') + if managed_postgres is False: + raise ValueError( + 'docker_depth_experiment collection requires managed PostgreSQL' + ) + if not configured_final_cutover or final_cutover is not True: + raise ValueError( + 'docker_depth_experiment collection requires PostgreSQL final cutover' + ) + if source.get('mode') != 'search': + raise ValueError( + 'docker_depth_experiment collection requires Docker Hub search mode' + ) + if source.get('require_digest') is not True: + raise ValueError( + 'docker_depth_experiment collection requires require_digest=true' + ) + + canonical_experiment = { + 'experiment_key': experiment_key, + 'enabled': enabled, + 'queries': list(queries), + 'repositories_per_query': repositories_per_query, + 'shallow_images_per_repository': shallow_images, + 'deep_repositories_per_query': deep_repositories, + 'deep_images_per_repository': deep_images, + 'target_limit': target_limit, + 'selector_version': selector_version, + } + source['docker_depth_experiment'] = canonical_experiment + ordered_query_hash = canonical_ordered_query_hash(queries) + selector_hash = canonical_selector_hash(selector_version) + semantic_experiment = { + key: value for key, value in canonical_experiment.items() + if key != 'enabled' + } + config_hash = _canonical_sha256({ + 'collection_generation': DOCKER_DEPTH_COLLECTION_GENERATION, + 'discovery_policies': list(effective_policies), + 'experiment': semantic_experiment, + 'mode': source.get('mode'), + 'ordinary_images_per_repository': image_depth, + 'platform': { + 'architecture': platform_arch, + 'candidate_tags': candidate_tags, + 'filter_enabled': platform_filter_enabled, + 'os': platform_os, + }, + 'require_digest': source.get('require_digest'), + 'schema': 'docker-depth-experiment-v3', + }) + experiment = DockerDepthExperimentConfig( + experiment_key=experiment_key, + enabled=enabled, + collection_generation=DOCKER_DEPTH_COLLECTION_GENERATION, + queries=queries, + repositories_per_query=repositories_per_query, + shallow_images_per_repository=shallow_images, + deep_repositories_per_query=deep_repositories, + deep_images_per_repository=deep_images, + target_limit=target_limit, + selector_version=selector_version, + theoretical_max_targets=theoretical_max, + ordered_query_hash=ordered_query_hash, + config_hash=config_hash, + selector_hash=selector_hash, + ) + return ValidatedDockerDepthConfig( + normalized_config=normalized, + docker_images_per_repository=image_depth, + experiment=experiment, + ordered_query_hash=ordered_query_hash, + config_hash=config_hash, + selector_hash=selector_hash, + ) + + +def _strict_positive_int(value, name, maximum=None): + if isinstance(value, bool) or not isinstance(value, int) or value < 1: + raise ValueError(f'{name} must be a positive integer') + if maximum is not None and value > maximum: + raise ValueError(f'{name} exceeds its {maximum} bound') + return value + + +def _strict_nonnegative_int(value, name): + if isinstance(value, bool) or not isinstance(value, int) or value < 0: + raise ValueError(f'{name} must be a non-negative integer') + return value + + +def _valid_sha256(value): + return bool(re.fullmatch(r'[a-f0-9]{64}', str(value or ''))) + + +def _docker_depth_authority(experiment, provenance_policy_sha256): + if experiment is None: + raise ValueError('Docker depth experiment authority is unavailable') + queries = tuple(getattr(experiment, 'queries', ())) + if ( + len(queries) != DOCKER_DEPTH_QUERY_COUNT + or len(set(queries)) != DOCKER_DEPTH_QUERY_COUNT + or any(not isinstance(query, str) or not query or query != query.strip() + for query in queries) + ): + raise ValueError('Docker depth planning requires all 61 unique ordered queries') + ordered_queries_sha256 = str(getattr(experiment, 'ordered_query_hash', '') or '') + if ordered_queries_sha256 != canonical_ordered_query_hash(queries): + raise ValueError('Docker depth ordered-query hash conflicts') + selector_version = str(getattr(experiment, 'selector_version', '') or '') + selector_sha256 = str(getattr(experiment, 'selector_hash', '') or '') + if ( + selector_version not in _DOCKER_EXPERIMENT_PROFILES + or selector_sha256 != canonical_selector_hash(selector_version) + ): + raise ValueError('Docker depth selector authority conflicts') + config_sha256 = str(getattr(experiment, 'config_hash', '') or '') + provenance_policy_sha256 = str(provenance_policy_sha256 or '') + if not _valid_sha256(config_sha256) or not _valid_sha256(provenance_policy_sha256): + raise ValueError('Docker depth configuration or provenance policy hash is invalid') + profile = reviewed_docker_experiment_profile(selector_version) + fixed = ( + (getattr(experiment, 'repositories_per_query', None), + profile['repositories_per_query']), + (getattr(experiment, 'shallow_images_per_repository', None), + profile['shallow_images_per_repository']), + (getattr(experiment, 'deep_repositories_per_query', None), + profile['deep_repositories_per_query']), + (getattr(experiment, 'deep_images_per_repository', None), + profile['images_per_repository']), + (getattr(experiment, 'target_limit', None), profile['target_limit']), + ) + if any(actual != expected for actual, expected in fixed): + raise ValueError('Docker depth planning limits conflict with the reviewed experiment') + theoretical_max = profile['theoretical_max_targets'] + if getattr(experiment, 'theoretical_max_targets', None) != theoretical_max: + raise ValueError('Docker depth theoretical target capacity conflicts') + experiment_key = _strict_identifier( + getattr(experiment, 'experiment_key', None), 'experiment_key', + ) + collection_generation = str( + getattr(experiment, 'collection_generation', '') or '' + ) + if collection_generation != DOCKER_DEPTH_COLLECTION_GENERATION: + raise ValueError('Docker depth collection generation authority conflicts') + return { + 'experiment_key': experiment_key, + 'source': 'dockerhub', + 'collection_generation': collection_generation, + 'queries': queries, + 'config_sha256': config_sha256, + 'ordered_queries_sha256': ordered_queries_sha256, + 'selector_version': selector_version, + 'selector_sha256': selector_sha256, + 'provenance_policy_sha256': provenance_policy_sha256, + 'query_count': DOCKER_DEPTH_QUERY_COUNT, + 'repositories_per_query': profile['repositories_per_query'], + 'images_per_repository': profile['images_per_repository'], + 'target_limit': profile['target_limit'], + 'theoretical_max_targets': theoretical_max, + } + + +def docker_depth_resolver_authority(experiment, provenance_policy_sha256): + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + enabled = getattr(experiment, 'enabled', None) + if not isinstance(enabled, bool): + raise ValueError('Docker depth resolver enabled authority is invalid') + return {**authority, 'enabled': enabled} + + +def _cohort_plan_document(authority, planned_queries): + round_robin = [] + by_ordinal = { + item['query_ordinal']: item for item in planned_queries + } + for repository_rank in range(1, authority['repositories_per_query'] + 1): + for query_ordinal in range(authority['query_count']): + query_plan = by_ordinal[query_ordinal] + if repository_rank > len(query_plan['repositories']): + continue + repository = query_plan['repositories'][repository_rank - 1] + round_robin.append({ + 'query_ordinal': query_ordinal, + 'repository_rank': repository_rank, + 'repository_queue_id': repository['repository_queue_id'], + }) + return { + 'schema': 2, + 'type': DOCKER_DEPTH_COHORT_PLAN_TYPE, + 'experiment_key': authority['experiment_key'], + 'source': authority['source'], + 'collection_generation': authority['collection_generation'], + 'config_sha256': authority['config_sha256'], + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'selector_version': authority['selector_version'], + 'selector_sha256': authority['selector_sha256'], + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'query_count': authority['query_count'], + 'repositories_per_query': authority['repositories_per_query'], + 'images_per_repository': authority['images_per_repository'], + 'target_limit': authority['target_limit'], + 'theoretical_max_targets': authority['theoretical_max_targets'], + 'queries': planned_queries, + 'repository_round_robin': round_robin, + } + + +def build_docker_depth_cohort_plan( + experiment, provenance_policy_sha256, candidates_by_query, +): + """Build the immutable cohort from already fenced fresh-candidate rows.""" + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + if not isinstance(candidates_by_query, dict) or set(candidates_by_query) != set( + authority['queries'] + ): + raise ValueError('Docker depth candidates must cover exactly all 61 ordered queries') + normalized_by_query = [] + for query_ordinal, query in enumerate(authority['queries']): + values = candidates_by_query[query] + if isinstance(values, (str, bytes)): + raise ValueError('Docker depth query candidates are invalid') + try: + values = list(values) + except TypeError: + raise ValueError('Docker depth query candidates are invalid') from None + normalized = [] + seen = set() + for raw in values: + if not isinstance(raw, dict): + raw = dict(raw) + queue_id = _strict_positive_int( + raw.get('repository_queue_id'), 'repository_queue_id', + ) + if queue_id in seen: + raise ValueError('Docker depth query candidates contain a duplicate repository') + seen.add(queue_id) + normalized.append({ + 'repository_queue_id': queue_id, + 'eligibility_page_id': _strict_positive_int( + raw.get('eligibility_page_id'), 'eligibility_page_id', + ), + 'best_search_rank': _strict_positive_int( + raw.get('best_search_rank'), 'best_search_rank', + ), + 'valid_distinct_graph_count': _strict_nonnegative_int( + raw.get('valid_distinct_graph_count', 0), + 'valid_distinct_graph_count', + ), + }) + normalized.sort(key=lambda item: ( + item['best_search_rank'], item['repository_queue_id'], + )) + normalized_by_query.append(normalized) + + if authority['selector_version'] == DOCKER_DEPTH_SELECTOR_VERSION: + selected_by_query = [ + values[:DOCKER_DEPTH_REPOSITORIES_PER_QUERY] + for values in normalized_by_query + ] + else: + selected_by_query = [[] for _query in authority['queries']] + cursors = [0] * len(selected_by_query) + seen_queue_ids = set() + selected_count = 0 + while selected_count < authority['target_limit']: + progressed = False + for ordinal, values in enumerate(normalized_by_query): + if len(selected_by_query[ordinal]) >= authority['repositories_per_query']: + continue + while ( + cursors[ordinal] < len(values) + and values[cursors[ordinal]]['repository_queue_id'] in seen_queue_ids + ): + cursors[ordinal] += 1 + if cursors[ordinal] >= len(values): + continue + candidate = values[cursors[ordinal]] + cursors[ordinal] += 1 + selected_by_query[ordinal].append(candidate) + seen_queue_ids.add(candidate['repository_queue_id']) + selected_count += 1 + progressed = True + if selected_count == authority['target_limit']: + break + if not progressed: + break + if selected_count != authority['target_limit']: + raise ValueError('Docker rank1 breadth cohort cannot fill its exact target limit') + + planned_queries = [] + for query_ordinal, query in enumerate(authority['queries']): + selected = selected_by_query[query_ordinal] + deep = min(selected, key=lambda item: ( + -item['valid_distinct_graph_count'], + item['best_search_rank'], + item['repository_queue_id'], + )) if selected else None + repositories = [] + for repository_rank, item in enumerate(selected, 1): + repositories.append({ + 'repository_queue_id': item['repository_queue_id'], + 'eligibility_page_id': item['eligibility_page_id'], + 'repository_rank': repository_rank, + 'is_deep_probe': bool( + deep and item['repository_queue_id'] == deep['repository_queue_id'] + ), + }) + planned_queries.append({ + 'query_ordinal': query_ordinal, + 'query': query, + 'query_sha256': _canonical_sha256(query), + 'selected_repository_count': len(repositories), + 'repositories': repositories, + }) + if authority['selector_version'] == DOCKER_RANK1_BREADTH_SELECTOR_VERSION: + queue_ids = [ + repository['repository_queue_id'] + for query in planned_queries for repository in query['repositories'] + ] + if len(queue_ids) != len(set(queue_ids)): + raise ValueError( + 'Docker rank1 breadth cohort contains a duplicate physical repository' + ) + return _cohort_plan_document(authority, planned_queries) + + +def canonical_docker_depth_plan_hash(plan): + if not isinstance(plan, dict) or plan.get('type') != DOCKER_DEPTH_COHORT_PLAN_TYPE: + raise ValueError('Docker depth cohort plan type is invalid') + return _canonical_sha256(plan) + + +def _validate_docker_depth_cohort_plan( + manifest, experiment=None, provenance_policy_sha256=None, +): + """Validate the immutable plan carried by a reviewed cohort manifest.""" + expected_keys = { + 'schema', 'type', 'experiment_key', 'source', 'collection_generation', + 'config_sha256', 'ordered_queries_sha256', 'selector_version', + 'selector_sha256', 'provenance_policy_sha256', 'query_count', + 'repositories_per_query', 'images_per_repository', 'target_limit', + 'theoretical_max_targets', 'queries', 'repository_round_robin', + } + if ( + not isinstance(manifest, dict) + or set(manifest) != expected_keys + or type(manifest.get('schema')) is not int + or manifest.get('schema') != 2 + or manifest.get('type') != DOCKER_DEPTH_COHORT_PLAN_TYPE + ): + raise ValueError('Docker depth cohort manifest shape is invalid') + if ( + manifest.get('source') != 'dockerhub' + or manifest.get('collection_generation') != DOCKER_DEPTH_COLLECTION_GENERATION + or manifest.get('selector_version') not in _DOCKER_EXPERIMENT_PROFILES + or manifest.get('selector_sha256') + != canonical_selector_hash(manifest.get('selector_version')) + or any(not _valid_sha256(manifest.get(name)) for name in ( + 'config_sha256', 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', + )) + ): + raise ValueError('Docker depth cohort manifest authority is invalid') + _strict_identifier(manifest.get('experiment_key'), 'experiment_key') + profile = reviewed_docker_experiment_profile(manifest['selector_version']) + fixed = { + 'query_count': profile['query_count'], + 'repositories_per_query': profile['repositories_per_query'], + 'images_per_repository': profile['images_per_repository'], + 'target_limit': profile['target_limit'], + 'theoretical_max_targets': profile['theoretical_max_targets'], + } + if any(type(manifest.get(name)) is not int or manifest[name] != value + for name, value in fixed.items()): + raise ValueError('Docker depth cohort manifest limits conflict') + raw_queries = manifest.get('queries') + if not isinstance(raw_queries, list) or len(raw_queries) != fixed['query_count']: + raise ValueError('Docker depth cohort manifest query coverage is incomplete') + planned_queries = [] + queries = [] + global_queue_ids = set() + for ordinal, raw_query in enumerate(raw_queries): + if not isinstance(raw_query, dict) or set(raw_query) != { + 'query_ordinal', 'query', 'query_sha256', + 'selected_repository_count', 'repositories', + }: + raise ValueError('Docker depth cohort manifest query shape is invalid') + query = raw_query.get('query') + if ( + type(raw_query.get('query_ordinal')) is not int + or raw_query['query_ordinal'] != ordinal + or not isinstance(query, str) + or not query + or query != query.strip() + or raw_query.get('query_sha256') != _canonical_sha256(query) + ): + raise ValueError('Docker depth cohort manifest query identity conflicts') + raw_repositories = raw_query.get('repositories') + selected_repository_count = raw_query.get('selected_repository_count') + if ( + not isinstance(raw_repositories, list) + or type(selected_repository_count) is not int + or not 0 <= selected_repository_count <= fixed['repositories_per_query'] + or len(raw_repositories) != selected_repository_count + ): + raise ValueError('Docker depth cohort manifest membership count conflicts') + repositories = [] + queue_ids = set() + for rank, raw_repository in enumerate(raw_repositories, 1): + if not isinstance(raw_repository, dict) or set(raw_repository) != { + 'repository_queue_id', 'eligibility_page_id', 'repository_rank', + 'is_deep_probe', + }: + raise ValueError('Docker depth cohort manifest membership shape is invalid') + queue_id = _strict_positive_int( + raw_repository.get('repository_queue_id'), 'repository_queue_id', + ) + if queue_id in queue_ids: + raise ValueError('Docker depth cohort manifest membership is duplicated') + queue_ids.add(queue_id) + if ( + manifest['selector_version'] == DOCKER_RANK1_BREADTH_SELECTOR_VERSION + and queue_id in global_queue_ids + ): + raise ValueError('Docker rank1 breadth cohort physical membership is duplicated') + global_queue_ids.add(queue_id) + if ( + type(raw_repository.get('repository_rank')) is not int + or raw_repository['repository_rank'] != rank + or type(raw_repository.get('is_deep_probe')) is not bool + ): + raise ValueError('Docker depth cohort manifest rank identity conflicts') + repositories.append({ + 'repository_queue_id': queue_id, + 'eligibility_page_id': _strict_positive_int( + raw_repository.get('eligibility_page_id'), 'eligibility_page_id', + ), + 'repository_rank': rank, + 'is_deep_probe': raw_repository['is_deep_probe'], + }) + expected_deep_count = 1 if repositories else 0 + if sum(int(item['is_deep_probe']) for item in repositories) != expected_deep_count: + raise ValueError('Docker depth cohort manifest deep membership conflicts') + queries.append(query) + planned_queries.append({ + 'query_ordinal': ordinal, + 'query': query, + 'query_sha256': _canonical_sha256(query), + 'selected_repository_count': selected_repository_count, + 'repositories': repositories, + }) + if len(set(queries)) != fixed['query_count']: + raise ValueError('Docker depth cohort manifest queries are duplicated') + if ( + manifest['selector_version'] == DOCKER_RANK1_BREADTH_SELECTOR_VERSION + and len(global_queue_ids) != fixed['target_limit'] + ): + raise ValueError('Docker rank1 breadth cohort target count conflicts') + if manifest['ordered_queries_sha256'] != canonical_ordered_query_hash(queries): + raise ValueError('Docker depth cohort manifest ordered-query hash conflicts') + authority = { + 'experiment_key': manifest['experiment_key'], + 'source': manifest['source'], + 'collection_generation': manifest['collection_generation'], + 'queries': tuple(queries), + 'config_sha256': manifest['config_sha256'], + 'ordered_queries_sha256': manifest['ordered_queries_sha256'], + 'selector_version': manifest['selector_version'], + 'selector_sha256': manifest['selector_sha256'], + 'provenance_policy_sha256': manifest['provenance_policy_sha256'], + **fixed, + } + if experiment is not None: + expected_authority = _docker_depth_authority( + experiment, + provenance_policy_sha256 or manifest['provenance_policy_sha256'], + ) + if authority != expected_authority: + raise ValueError('Docker depth cohort manifest authority drifted') + canonical = _cohort_plan_document(authority, planned_queries) + if canonical != manifest: + raise ValueError('Docker depth cohort manifest is not canonical') + normalized = copy.deepcopy(canonical) + return normalized, canonical_docker_depth_plan_hash(normalized) + + +def _cohort_provenance_snapshot(authority, candidates_by_query): + if not isinstance(candidates_by_query, dict) or set(candidates_by_query) != set( + authority['queries'] + ): + raise ValueError('Docker depth candidates must cover exactly all 61 ordered queries') + queries = [] + for query_ordinal, query in enumerate(authority['queries']): + values = candidates_by_query[query] + if isinstance(values, (str, bytes)): + raise ValueError('Docker depth query candidates are invalid') + try: + values = list(values) + except TypeError: + raise ValueError('Docker depth query candidates are invalid') from None + normalized = [] + seen = set() + for raw in values: + if not isinstance(raw, dict): + raw = dict(raw) + queue_id = _strict_positive_int( + raw.get('repository_queue_id'), 'repository_queue_id', + ) + if queue_id in seen: + raise ValueError('Docker depth query candidates contain a duplicate repository') + seen.add(queue_id) + normalized.append({ + 'repository_queue_id': queue_id, + 'eligibility_page_id': _strict_positive_int( + raw.get('eligibility_page_id'), 'eligibility_page_id', + ), + 'best_search_rank': _strict_positive_int( + raw.get('best_search_rank'), 'best_search_rank', + ), + 'valid_distinct_graph_count': _strict_nonnegative_int( + raw.get('valid_distinct_graph_count', 0), + 'valid_distinct_graph_count', + ), + }) + normalized.sort(key=lambda item: ( + item['best_search_rank'], item['repository_queue_id'], + )) + selected = normalized[:authority['repositories_per_query']] + queries.append({ + 'query_ordinal': query_ordinal, + 'query_sha256': _canonical_sha256(query), + 'selected_repository_count': len(selected), + 'repositories': selected, + }) + return { + 'schema': 2, + 'type': 'truf-docker-depth-cohort-provenance-snapshot-v2', + 'collection_generation': authority['collection_generation'], + 'config_sha256': authority['config_sha256'], + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'selector_sha256': authority['selector_sha256'], + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'queries': queries, + } + + +def build_docker_depth_cohort_manifest( + experiment, provenance_policy_sha256, candidates_by_query, +): + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + plan = build_docker_depth_cohort_plan( + experiment, provenance_policy_sha256, candidates_by_query, + ) + snapshot = _cohort_provenance_snapshot(authority, candidates_by_query) + manifest = { + 'schema': 2, + 'type': DOCKER_DEPTH_COHORT_MANIFEST_TYPE, + 'version': 2, + 'experiment_key': authority['experiment_key'], + 'source': authority['source'], + 'collection_generation': authority['collection_generation'], + 'collection_generation_sha256': _canonical_sha256( + authority['collection_generation'] + ), + 'config_sha256': authority['config_sha256'], + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'selector_version': authority['selector_version'], + 'selector_sha256': authority['selector_sha256'], + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'provenance_snapshot_sha256': _canonical_sha256(snapshot), + 'query_count': authority['query_count'], + 'repository_count': sum( + item['selected_repository_count'] for item in plan['queries'] + ), + 'plan_sha256': canonical_docker_depth_plan_hash(plan), + 'plan': plan, + } + return validate_docker_depth_cohort_manifest( + manifest, experiment, provenance_policy_sha256, + )[0] + + +def validate_docker_depth_cohort_manifest( + manifest, experiment=None, provenance_policy_sha256=None, +): + expected_keys = { + 'schema', 'type', 'version', 'experiment_key', 'source', + 'collection_generation', 'collection_generation_sha256', + 'config_sha256', 'ordered_queries_sha256', 'selector_version', + 'selector_sha256', 'provenance_policy_sha256', + 'provenance_snapshot_sha256', 'query_count', 'repository_count', + 'plan_sha256', 'plan', + } + if ( + not isinstance(manifest, dict) + or set(manifest) != expected_keys + or type(manifest.get('schema')) is not int + or manifest.get('schema') != 2 + or type(manifest.get('version')) is not int + or manifest.get('version') != 2 + or manifest.get('type') != DOCKER_DEPTH_COHORT_MANIFEST_TYPE + ): + raise ValueError('Docker depth cohort review manifest shape is invalid') + plan, plan_sha256 = _validate_docker_depth_cohort_plan( + manifest.get('plan'), experiment, provenance_policy_sha256, + ) + for name in ( + 'collection_generation_sha256', 'config_sha256', + 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', 'provenance_snapshot_sha256', 'plan_sha256', + ): + if not _valid_sha256(manifest.get(name)): + raise ValueError('Docker depth cohort review manifest contains an invalid hash') + if ( + manifest['experiment_key'] != plan['experiment_key'] + or manifest['source'] != plan['source'] + or manifest['collection_generation'] != plan['collection_generation'] + or manifest['collection_generation_sha256'] + != _canonical_sha256(plan['collection_generation']) + or manifest['config_sha256'] != plan['config_sha256'] + or manifest['ordered_queries_sha256'] != plan['ordered_queries_sha256'] + or manifest['selector_version'] != plan['selector_version'] + or manifest['selector_sha256'] != plan['selector_sha256'] + or manifest['provenance_policy_sha256'] != plan['provenance_policy_sha256'] + or type(manifest.get('query_count')) is not int + or manifest['query_count'] != plan['query_count'] + or type(manifest.get('repository_count')) is not int + or manifest['repository_count'] + != sum(item['selected_repository_count'] for item in plan['queries']) + or manifest['plan_sha256'] != plan_sha256 + ): + raise ValueError('Docker depth cohort review manifest evidence conflicts') + normalized = copy.deepcopy(manifest) + normalized['plan'] = plan + if normalized != manifest: + raise ValueError('Docker depth cohort review manifest is not canonical') + return normalized, _canonical_sha256(normalized) + + +def _require_postgres_experiment_db(db, operation): + conn = getattr(db, 'conn', None) + if not conn or not getattr(conn, 'is_postgres', False): + raise RuntimeError(f'{operation} requires managed PostgreSQL') + return conn + + +def _experiment_identity_matches(row, authority): + expected = { + 'experiment_key': authority['experiment_key'], + 'source': authority['source'], + 'collection_generation': authority['collection_generation'], + 'config_sha256': authority['config_sha256'], + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'selector_version': authority['selector_version'], + 'selector_sha256': authority['selector_sha256'], + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'query_count': authority['query_count'], + 'repositories_per_query': authority['repositories_per_query'], + 'images_per_repository': authority['images_per_repository'], + 'target_limit': authority['target_limit'], + } + return all( + (int(row[name]) if isinstance(value, int) else str(row[name])) == value + for name, value in expected.items() + ) + + +def _stored_cohort_plan(conn, experiment_row, authority): + experiment_id = int(experiment_row['id']) + query_rows = conn.execute( + '''SELECT query_ordinal, query, query_sha256, + required_repository_count, selected_repository_count + FROM docker_depth_experiment_queries + WHERE experiment_id = ? ORDER BY query_ordinal''', + (experiment_id,), + ).fetchall() + repository_rows = conn.execute( + '''SELECT query_ordinal, repository_queue_id, eligibility_page_id, + repository_rank, planned_is_deep_probe + FROM docker_depth_experiment_repositories + WHERE experiment_id = ? ORDER BY query_ordinal, repository_rank''', + (experiment_id,), + ).fetchall() + expected_repository_count = sum( + int(row['selected_repository_count']) for row in query_rows + ) if len(query_rows) == authority['query_count'] else -1 + if ( + len(query_rows) != authority['query_count'] + or len(repository_rows) != expected_repository_count + ): + raise RuntimeError('Docker depth persisted cohort is incomplete') + repositories_by_query = { + ordinal: [] for ordinal in range(authority['query_count']) + } + for row in repository_rows: + ordinal = int(row['query_ordinal']) + if ordinal not in repositories_by_query: + raise RuntimeError('Docker depth persisted repository ordinal is invalid') + repositories_by_query[ordinal].append({ + 'repository_queue_id': int(row['repository_queue_id']), + 'eligibility_page_id': int(row['eligibility_page_id']), + 'repository_rank': int(row['repository_rank']), + 'is_deep_probe': bool(row['planned_is_deep_probe']), + }) + planned_queries = [] + for ordinal, row in enumerate(query_rows): + repositories = repositories_by_query[ordinal] + selected_repository_count = int(row['selected_repository_count']) + if ( + int(row['query_ordinal']) != ordinal + or str(row['query']) != authority['queries'][ordinal] + or str(row['query_sha256']) != _canonical_sha256(authority['queries'][ordinal]) + or int(row['required_repository_count']) != authority['repositories_per_query'] + or not 0 <= selected_repository_count <= authority['repositories_per_query'] + or len(repositories) != selected_repository_count + or [item['repository_rank'] for item in repositories] + != list(range(1, selected_repository_count + 1)) + or sum(int(item['is_deep_probe']) for item in repositories) + != (1 if selected_repository_count else 0) + ): + raise RuntimeError('Docker depth persisted cohort identity conflicts') + planned_queries.append({ + 'query_ordinal': ordinal, + 'query': authority['queries'][ordinal], + 'query_sha256': str(row['query_sha256']), + 'selected_repository_count': selected_repository_count, + 'repositories': repositories, + }) + return _cohort_plan_document(authority, planned_queries) + + +def _fresh_cohort_candidates(conn, authority, include_experiment_id=None): + candidates_by_query = {} + coverage = [] + breadth_profile = ( + authority['selector_version'] == DOCKER_RANK1_BREADTH_SELECTOR_VERSION + ) + candidate_limit = 3000 if breadth_profile else authority['repositories_per_query'] + if breadth_profile: + policy_filter = '''AND NOT EXISTS ( + SELECT 1 + FROM target_queue_policy_events event + LEFT JOIN docker_depth_experiments prior + ON prior.id = event.experiment_id + WHERE event.queue_id = queue.id + AND ( + prior.id IS NULL OR prior.state <> 'released' + OR event.source <> queue.source + OR event.platform <> queue.platform + OR event.query <> queue.query + OR event.config_sha256 <> prior.config_sha256 + OR event.policy_sha256 <> prior.provenance_policy_sha256 + OR (event.action = 'cold' AND NOT EXISTS ( + SELECT 1 FROM target_queue_policy_events reverse_event + WHERE reverse_event.reverses_event_id = event.id + AND reverse_event.action = 'reactivate' + AND reverse_event.experiment_id = event.experiment_id + AND reverse_event.queue_id = event.queue_id + )) + OR (event.action = 'reactivate' AND NOT EXISTS ( + SELECT 1 FROM target_queue_policy_events cold_event + WHERE cold_event.id = event.reverses_event_id + AND cold_event.action = 'cold' + AND cold_event.experiment_id = event.experiment_id + AND cold_event.queue_id = event.queue_id + )) + ) + )''' + else: + policy_filter = '''AND NOT EXISTS ( + SELECT 1 FROM target_queue_policy_events event + WHERE event.queue_id = queue.id + )''' + if include_experiment_id is None: + member_filter = '''AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_repositories member + WHERE member.repository_queue_id = queue.id + )''' + member_params = () + else: + member_filter = '''AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_repositories member + WHERE member.repository_queue_id = queue.id + AND member.experiment_id <> ? + )''' + member_params = (int(include_experiment_id),) + for query_ordinal, query in enumerate(authority['queries']): + candidates = conn.execute( + f'''SELECT provenance.repository_queue_id, + MIN(observation.search_rank) AS best_search_rank, + MIN(observation.page_id) AS eligibility_page_id, + COALESCE(( + SELECT COUNT(DISTINCT manifest.graph_sha256) + FROM docker_image_manifests manifest + WHERE manifest.source = queue.source + AND manifest.repository = queue.normalized_target + ), 0) AS valid_distinct_graph_count + FROM docker_repository_query_provenance provenance + JOIN docker_repository_query_observations observation + ON observation.source = provenance.source + AND observation.query = provenance.query + AND observation.repository_queue_id = provenance.repository_queue_id + JOIN docker_discovery_pages page ON page.id = observation.page_id + JOIN docker_discovery_passes discovery_pass ON discovery_pass.id = page.pass_id + JOIN target_queue queue ON queue.id = provenance.repository_queue_id + WHERE provenance.source = ? AND provenance.query = ? + AND provenance.provenance_kind = 'fresh_page' + AND provenance.fresh_coverage_eligible = 1 + AND provenance.fresh_complete_observation_count > 0 + AND page.query_ordinal = ? AND page.query = ? + AND discovery_pass.source = ? AND discovery_pass.pass_kind = 'deep' + AND discovery_pass.collection_generation = ? + AND discovery_pass.policy_sha256 = ? + AND discovery_pass.ordered_queries_sha256 = ? + AND discovery_pass.expected_query_count = ? + AND discovery_pass.state = 'complete' + AND queue.source = ? AND queue.platform = 'docker' + AND queue.status IN ('pending','deferred') + AND queue.target_scan_id IS NULL + AND queue.target NOT LIKE '%@%' AND queue.normalized_target NOT LIKE '%@%' + AND queue.lease_owner IS NULL AND queue.lease_token IS NULL + AND queue.claim_batch IS NULL AND queue.leased_at IS NULL + AND queue.lease_expires_at IS NULL + AND queue.current_result_reservation_id IS NULL + AND queue.claim_event_id IS NULL AND queue.resolver_token IS NULL + AND COALESCE(queue.resolver_state, '') <> 'resolving' + AND NOT EXISTS (SELECT 1 FROM target_scans scan WHERE scan.queue_id = queue.id) + {policy_filter} + AND NOT EXISTS ( + SELECT 1 FROM result_reservations reservation + WHERE reservation.queue_id = queue.id + AND reservation.state IN ('scanning','ready','ingesting','db_committed') + ) + AND NOT EXISTS ( + SELECT 1 FROM result_reservations reservation + JOIN pipeline_quarantine quarantine ON quarantine.reservation_id = reservation.id + WHERE reservation.queue_id = queue.id + ) + AND NOT EXISTS ( + SELECT 1 FROM docker_image_manifests manifest + WHERE manifest.target_queue_id = queue.id + ) + AND NOT EXISTS ( + SELECT 1 FROM docker_image_blob_coverage coverage + JOIN docker_content_blobs blob + ON blob.digest = coverage.blob_digest + AND blob.coverage_policy_sha256 = coverage.coverage_policy_sha256 + WHERE coverage.queue_id = queue.id + AND blob.state IN ('leased','submitted') + ) + {member_filter} + GROUP BY provenance.repository_queue_id, queue.source, queue.normalized_target + ORDER BY MIN(observation.search_rank), provenance.repository_queue_id + LIMIT ?''', + ( + authority['source'], query, query_ordinal, query, + authority['source'], authority['collection_generation'], + authority['provenance_policy_sha256'], + authority['ordered_queries_sha256'], authority['query_count'], + authority['source'], *member_params, + candidate_limit, + ), + ).fetchall() + normalized = [dict(candidate) for candidate in candidates] + candidates_by_query[query] = normalized + coverage.append({ + 'query_ordinal': query_ordinal, + 'query_sha256': _canonical_sha256(query), + 'eligible_slots': len(normalized), + 'required_slots': authority['repositories_per_query'], + 'eligible_slots_sha256': _canonical_sha256(normalized), + }) + return candidates_by_query, coverage + + +def _require_complete_docker_depth_collection_pass(conn, authority): + row = conn.execute( + '''SELECT id FROM docker_discovery_passes + WHERE source = ? AND pass_kind = 'deep' + AND collection_generation = ? AND policy_sha256 = ? + AND ordered_queries_sha256 = ? AND expected_query_count = ? + AND completed_query_count = expected_query_count + AND state = 'complete' AND completed_at IS NOT NULL + ORDER BY id LIMIT 1''', + ( + authority['source'], authority['collection_generation'], + authority['provenance_policy_sha256'], + authority['ordered_queries_sha256'], authority['query_count'], + ), + ).fetchone() + if not row: + raise RuntimeError( + 'Docker depth cohort requires a complete fresh deep discovery pass' + ) + return int(row['id']) + + +def summarize_docker_depth_fresh_coverage( + db, experiment, provenance_policy_sha256, +): + """Return a query-name-free summary of the current 61-by-10 eligibility gate.""" + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + conn = _require_postgres_experiment_db(db, 'Docker depth coverage status') + candidates_by_query, coverage = _fresh_cohort_candidates(conn, authority) + complete = sum( + len(candidates_by_query[query]) >= authority['repositories_per_query'] + for query in authority['queries'] + ) + eligible_slot_histogram = { + str(slot_count): sum( + len(candidates_by_query[query]) == slot_count + for query in authority['queries'] + ) + for slot_count in range(authority['repositories_per_query'] + 1) + } + return { + 'query_count': authority['query_count'], + 'complete_query_count': complete, + 'incomplete_query_count': authority['query_count'] - complete, + 'eligible_slot_histogram': eligible_slot_histogram, + 'required_repository_count': ( + authority['query_count'] * authority['repositories_per_query'] + ), + 'eligible_repository_slots': sum( + len(candidates_by_query[query]) for query in authority['queries'] + ), + 'coverage_sha256': _canonical_sha256({ + 'collection_generation': authority['collection_generation'], + 'config_sha256': authority['config_sha256'], + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'queries': coverage, + 'schema': 1, + 'selector_sha256': authority['selector_sha256'], + 'type': 'truf-docker-depth-fresh-coverage-v1', + }), + } + + +def _cohort_partial_count(conn, experiment_id): + return int(conn.execute( + '''SELECT + (SELECT COUNT(*) FROM docker_depth_experiment_queries + WHERE experiment_id = ?) + + (SELECT COUNT(*) FROM docker_depth_experiment_repositories + WHERE experiment_id = ?) AS count''', + (experiment_id, experiment_id), + ).fetchone()['count']) + + +def _experiment_fence_active(row): + return any( + row[name] is not None + for name in ('fence_owner', 'fence_token', 'fence_expires_at') + ) + + +def generate_docker_depth_cohort_manifest( + db, experiment, provenance_policy_sha256, +): + """Build a deterministic cohort manifest without changing database state.""" + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + conn = _require_postgres_experiment_db(db, 'Docker depth cohort review') + try: + row = conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE experiment_key = ?', + (authority['experiment_key'],), + ).fetchone() + if row: + if not _experiment_identity_matches(row, authority): + raise RuntimeError('Docker depth experiment authority hash drifted') + if _experiment_fence_active(row): + raise RuntimeError('Docker depth experiment has an active authority fence') + if row['plan_sha256'] is not None: + raise RuntimeError('Docker depth cohort is already persisted') + if str(row['state']) != 'collecting' or _cohort_partial_count(conn, row['id']): + raise RuntimeError('Docker depth collecting experiment is not empty') + _require_complete_docker_depth_collection_pass(conn, authority) + candidates_by_query, _coverage = _fresh_cohort_candidates(conn, authority) + manifest = build_docker_depth_cohort_manifest( + experiment, provenance_policy_sha256, candidates_by_query, + ) + _require_planned_cohort_current(conn, manifest['plan'], False) + manifest, manifest_sha256 = validate_docker_depth_cohort_manifest( + manifest, experiment, provenance_policy_sha256, + ) + conn.commit() + return manifest, manifest_sha256 + except Exception: + conn.rollback() + raise + + +def _lock_cohort_manifest_rows(conn, manifest): + queue_ids = sorted({ + repository['repository_queue_id'] + for query_plan in manifest['plan']['queries'] + for repository in query_plan['repositories'] + }) + if not queue_ids: + return + placeholders = ','.join('?' for _ in queue_ids) + rows = conn.execute( + f'''SELECT id FROM target_queue WHERE id IN ({placeholders}) + ORDER BY id FOR UPDATE''', + tuple(queue_ids), + ).fetchall() + if [int(row['id']) for row in rows] != queue_ids: + raise RuntimeError('Docker depth reviewed cohort contains a missing repository') + + +def _policy_event_audit_sha256(action, manifest_sha256, entry, experiment_id): + return _canonical_sha256({ + 'action': str(action), + 'manifest_sha256': str(manifest_sha256), + 'entry': dict(entry), + 'experiment_id': int(experiment_id), + }) + + +def _require_released_policy_history(conn, queue_ids): + grouped = {int(queue_id): [] for queue_id in queue_ids} + values = sorted(grouped) + for offset in range(0, len(values), 500): + chunk = values[offset:offset + 500] + placeholders = ','.join('?' for _ in chunk) + rows = conn.execute( + f'''SELECT event.*, queue.status AS queue_status, + queue.source AS queue_source, + queue.platform AS queue_platform, + queue.query AS queue_query, + prior.state AS experiment_state, + prior.config_sha256 AS experiment_config_sha256, + prior.provenance_policy_sha256 AS experiment_policy_sha256, + prior.hold_manifest_sha256 AS experiment_hold_manifest_sha256 + FROM target_queue_policy_events event + JOIN target_queue queue ON queue.id = event.queue_id + LEFT JOIN docker_depth_experiments prior + ON prior.id = event.experiment_id + WHERE event.queue_id IN ({placeholders}) + ORDER BY event.queue_id, event.id''', + tuple(chunk), + ).fetchall() + for row in rows: + grouped[int(row['queue_id'])].append(row) + + for queue_id, events in grouped.items(): + if not events: + continue + by_id = {int(event['id']): event for event in events} + cold_events = [event for event in events if event['action'] == 'cold'] + reverse_events = [event for event in events if event['action'] == 'reactivate'] + if len(cold_events) != len(reverse_events): + raise RuntimeError('Docker breadth candidate policy history is incomplete') + used_reverse_ids = set() + for cold in cold_events: + reverses = [ + event for event in reverse_events + if int(event['reverses_event_id'] or 0) == int(cold['id']) + ] + if len(reverses) != 1: + raise RuntimeError('Docker breadth candidate policy history is ambiguous') + reverse = reverses[0] + used_reverse_ids.add(int(reverse['id'])) + experiment_id = int(cold['experiment_id'] or 0) + shared_conflict = any( + event['experiment_id'] is None + or int(event['experiment_id']) != experiment_id + or int(event['queue_id']) != queue_id + or str(event['source']) != str(event['queue_source']) + or str(event['platform']) != str(event['queue_platform']) + or str(event['query']) != str(event['queue_query']) + or str(event['experiment_state']) != 'released' + or str(event['config_sha256']) + != str(event['experiment_config_sha256']) + or str(event['policy_sha256']) + != str(event['experiment_policy_sha256']) + for event in (cold, reverse) + ) + if ( + shared_conflict + or cold['reverses_event_id'] is not None + or str(cold['reason_code']) not in ( + DOCKER_DEPTH_HOLD_REASON, DOCKER_DEPTH_DYNAMIC_HOLD_REASON, + ) + or str(cold['manifest_sha256']) + != str(cold['experiment_hold_manifest_sha256']) + or str(cold['prior_status']) not in ('pending', 'deferred') + or str(cold['next_status']) != 'cold' + or str(reverse['reason_code']) != DOCKER_DEPTH_RELEASE_REASON + or str(reverse['prior_status']) != 'cold' + or str(reverse['next_status']) != str(cold['prior_status']) + or str(reverse['queue_status']) != str(reverse['next_status']) + ): + raise RuntimeError('Docker breadth candidate policy history conflicts') + cold_entry = { + 'queue_id': queue_id, + 'source': str(cold['source']), + 'platform': str(cold['platform']), + 'query': str(cold['query']), + 'prior_status': str(cold['prior_status']), + 'prior_updated_at': str(cold['prior_updated_at']), + } + reverse_entry = { + 'queue_id': queue_id, + 'source': str(reverse['source']), + 'platform': str(reverse['platform']), + 'query': str(reverse['query']), + 'cold_event_id': int(cold['id']), + 'restore_status': str(reverse['next_status']), + 'prior_updated_at': str(reverse['prior_updated_at']), + } + if ( + str(cold['review_audit_sha256']) + != _policy_event_audit_sha256( + 'cold', cold['manifest_sha256'], cold_entry, experiment_id, + ) + or str(reverse['review_audit_sha256']) + != _policy_event_audit_sha256( + 'reactivate', reverse['manifest_sha256'], reverse_entry, + experiment_id, + ) + ): + raise RuntimeError('Docker breadth candidate policy audit conflicts') + if len(used_reverse_ids) != len(reverse_events) or any( + int(event['reverses_event_id'] or 0) not in by_id + for event in reverse_events + ): + raise RuntimeError('Docker breadth candidate policy reversal conflicts') + + +def _require_planned_cohort_current(conn, manifest, lock_rows): + queue_ids = sorted({ + repository['repository_queue_id'] + for query_plan in manifest['queries'] + for repository in query_plan['repositories'] + }) + if not queue_ids: + return + placeholders = ','.join('?' for _ in queue_ids) + lock_suffix = ' FOR UPDATE OF queue' if lock_rows else '' + rows = conn.execute( + f'''SELECT queue.id + FROM target_queue queue + WHERE queue.id IN ({placeholders}) + AND queue.source = 'dockerhub' AND queue.platform = 'docker' + AND queue.status IN ('pending','deferred') + AND queue.target_scan_id IS NULL + AND queue.target NOT LIKE '%@%' AND queue.normalized_target NOT LIKE '%@%' + AND queue.lease_owner IS NULL AND queue.lease_token IS NULL + AND queue.claim_batch IS NULL AND queue.leased_at IS NULL + AND queue.lease_expires_at IS NULL + AND queue.current_result_reservation_id IS NULL + AND queue.claim_event_id IS NULL AND queue.resolver_token IS NULL + AND COALESCE(queue.resolver_state, '') <> 'resolving' + AND NOT EXISTS (SELECT 1 FROM target_scans scan WHERE scan.queue_id = queue.id) + AND NOT EXISTS ( + SELECT 1 FROM result_reservations reservation + WHERE reservation.queue_id = queue.id + AND reservation.state IN ('scanning','ready','ingesting','db_committed') + ) + AND NOT EXISTS ( + SELECT 1 FROM result_reservations reservation + JOIN pipeline_quarantine quarantine + ON quarantine.reservation_id = reservation.id + WHERE reservation.queue_id = queue.id + ) + AND NOT EXISTS ( + SELECT 1 FROM docker_image_manifests image_manifest + WHERE image_manifest.target_queue_id = queue.id + ) + AND NOT EXISTS ( + SELECT 1 FROM docker_image_blob_coverage coverage + JOIN docker_content_blobs blob + ON blob.digest = coverage.blob_digest + AND blob.coverage_policy_sha256 = coverage.coverage_policy_sha256 + WHERE coverage.queue_id = queue.id + AND blob.state IN ('leased','submitted') + ) + ORDER BY queue.id{lock_suffix}''', + tuple(queue_ids), + ).fetchall() + if [int(row['id']) for row in rows] != queue_ids: + raise RuntimeError('Docker depth cohort selection drifted after review') + _require_released_policy_history(conn, queue_ids) + + +def apply_docker_depth_cohort_manifest( + db, experiment, provenance_policy_sha256, manifest, manifest_sha256, +): + """Atomically persist only the exact currently eligible reviewed cohort.""" + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + manifest, expected_sha256 = validate_docker_depth_cohort_manifest( + manifest, experiment, provenance_policy_sha256, + ) + if str(manifest_sha256 or '') != expected_sha256: + raise ValueError('Docker depth cohort manifest hash conflicts') + plan = manifest['plan'] + plan_sha256 = manifest['plan_sha256'] + conn = _require_postgres_experiment_db(db, 'Docker depth cohort application') + now = datetime.now(timezone.utc).isoformat(timespec='seconds') + try: + _require_complete_docker_depth_collection_pass(conn, authority) + conn.execute( + '''INSERT INTO docker_depth_experiments( + experiment_key, source, state, collection_generation, + config_sha256, ordered_queries_sha256, selector_version, + selector_sha256, provenance_policy_sha256, query_count, + repositories_per_query, images_per_repository, target_limit, + created_at, updated_at + ) VALUES (?, ?, 'collecting', ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(experiment_key) DO NOTHING''', + ( + authority['experiment_key'], authority['source'], + authority['collection_generation'], authority['config_sha256'], + authority['ordered_queries_sha256'], authority['selector_version'], + authority['selector_sha256'], authority['provenance_policy_sha256'], + authority['query_count'], authority['repositories_per_query'], + authority['images_per_repository'], authority['target_limit'], now, now, + ), + ) + row = conn.execute( + '''SELECT * FROM docker_depth_experiments + WHERE experiment_key = ? FOR UPDATE''', + (authority['experiment_key'],), + ).fetchone() + if not row or not _experiment_identity_matches(row, authority): + raise RuntimeError('Docker depth experiment authority is absent or drifted') + if _experiment_fence_active(row): + raise RuntimeError('Docker depth experiment has an active authority fence') + if row['plan_sha256'] is not None: + stored = _stored_cohort_plan(conn, row, authority) + stored_sha256 = canonical_docker_depth_plan_hash(stored) + if ( + stored != plan + or stored_sha256 != plan_sha256 + or str(row['plan_sha256']) != plan_sha256 + or str(row['state']) not in ('planned', 'holding') + ): + raise RuntimeError('Docker depth reviewed cohort authority conflicts') + _require_planned_cohort_current(conn, stored, True) + candidates_by_query, _coverage = _fresh_cohort_candidates( + conn, authority, include_experiment_id=int(row['id']), + ) + current = build_docker_depth_cohort_manifest( + experiment, provenance_policy_sha256, candidates_by_query, + ) + if current != manifest or _canonical_sha256(current) != expected_sha256: + raise RuntimeError('Docker depth cohort provenance drifted after review') + conn.commit() + return { + 'experiment_id': int(row['id']), + 'state': str(row['state']), + 'planning_allowed': True, + 'planned': False, + 'plan': stored, + 'plan_sha256': stored_sha256, + } + if str(row['state']) != 'collecting': + raise RuntimeError('Docker depth experiment is not in collecting state') + if _cohort_partial_count(conn, row['id']): + raise RuntimeError('Docker depth collecting experiment has a partial cohort') + _lock_cohort_manifest_rows(conn, manifest) + _require_planned_cohort_current(conn, manifest['plan'], True) + candidates_by_query, _coverage = _fresh_cohort_candidates(conn, authority) + current = build_docker_depth_cohort_manifest( + experiment, provenance_policy_sha256, candidates_by_query, + ) + current, current_sha256 = validate_docker_depth_cohort_manifest( + current, experiment, provenance_policy_sha256, + ) + if current != manifest or current_sha256 != expected_sha256: + raise RuntimeError('Docker depth cohort selection drifted after review') + experiment_id = int(row['id']) + for query_plan in plan['queries']: + conn.execute( + '''INSERT INTO docker_depth_experiment_queries( + experiment_id, source, query_ordinal, query, query_sha256, + required_repository_count, selected_repository_count, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?)''', + ( + experiment_id, authority['source'], query_plan['query_ordinal'], + query_plan['query'], query_plan['query_sha256'], + authority['repositories_per_query'], + query_plan['selected_repository_count'], now, + ), + ) + for repository in query_plan['repositories']: + conn.execute( + '''INSERT INTO docker_depth_experiment_repositories( + experiment_id, query_ordinal, source, query, + repository_queue_id, eligibility_page_id, repository_rank, + planned_is_deep_probe, is_deep_probe, work_state, + created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'pending', ?, ?)''', + ( + experiment_id, query_plan['query_ordinal'], authority['source'], + query_plan['query'], repository['repository_queue_id'], + repository['eligibility_page_id'], repository['repository_rank'], + int(repository['is_deep_probe']), int(repository['is_deep_probe']), + now, now, + ), + ) + cursor = conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'planned', plan_sha256 = ?, planned_at = ?, updated_at = ? + WHERE id = ? AND state = 'collecting' AND plan_sha256 IS NULL + AND target_count = 0 AND selection_count = 0''', + (plan_sha256, now, now, experiment_id), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError('Docker depth cohort plan lost its authority fence') + conn.commit() + return { + 'experiment_id': experiment_id, + 'state': 'planned', + 'planning_allowed': True, + 'planned': True, + 'plan': plan, + 'plan_sha256': plan_sha256, + } + except Exception: + conn.rollback() + raise + + +def plan_docker_depth_experiment( + db, experiment, provenance_policy_sha256, *, manifest=None, + manifest_sha256=None, +): + """Compatibility name that cannot bypass reviewed cohort approval.""" + if manifest is None or manifest_sha256 is None: + raise RuntimeError('Docker depth planning requires a reviewed cohort manifest') + return apply_docker_depth_cohort_manifest( + db, experiment, provenance_policy_sha256, manifest, manifest_sha256, + ) + + +def _experiment_row(conn, authority, lock_rows): + lock_suffix = ' FOR UPDATE' if lock_rows else '' + row = conn.execute( + f'''SELECT * FROM docker_depth_experiments + WHERE experiment_key = ?{lock_suffix}''', + (authority['experiment_key'],), + ).fetchone() + if not row or not _experiment_identity_matches(row, authority): + raise RuntimeError('Docker depth experiment authority is absent or drifted') + return row + + +def _locked_experiment_row(conn, authority): + return _experiment_row(conn, authority, True) + + +def _persisted_experiment_drift_reason(db, row, authority, now): + checker = getattr(db, '_docker_depth_persisted_drift_reason_locked', None) + if not callable(checker): + raise RuntimeError('Docker depth persisted authority validator is unavailable') + return checker(row, authority, now) + + +def _require_persisted_experiment_authority(db, conn, row, authority, now): + holder = getattr(db, '_hold_docker_depth_experiment_locked', None) + if not callable(holder): + raise RuntimeError('Docker depth persisted authority validator is unavailable') + reason = _persisted_experiment_drift_reason(db, row, authority, now) + if reason: + holder(row, reason, now) + conn.commit() + raise RuntimeError(f'Docker depth persisted authority drifted: {reason}') + + +def _noncohort_hold_snapshot(conn, experiment_id, source, maximum, lock_rows): + lock_suffix = ' FOR UPDATE OF queue' if lock_rows else '' + rows = conn.execute( + f'''SELECT queue.id, queue.source, queue.platform, queue.query, + queue.status, queue.updated_at, queue.target_scan_id, + queue.resolver_state, queue.lease_owner, queue.lease_token, + queue.claim_batch, queue.leased_at, queue.lease_expires_at, + queue.current_result_reservation_id, queue.claim_event_id, + queue.resolver_token, + EXISTS ( + SELECT 1 FROM result_reservations reservation + WHERE reservation.queue_id = queue.id + AND reservation.state IN ('scanning','ready','ingesting','db_committed') + ) AS active_reservation, + EXISTS ( + SELECT 1 FROM docker_image_blob_coverage coverage + JOIN docker_content_blobs blob + ON blob.digest = coverage.blob_digest + AND blob.coverage_policy_sha256 = coverage.coverage_policy_sha256 + WHERE coverage.queue_id = queue.id + AND blob.state IN ('leased','submitted') + ) AS active_docker_blob, + EXISTS ( + SELECT 1 FROM result_reservations reservation + JOIN pipeline_quarantine quarantine + ON quarantine.reservation_id = reservation.id + WHERE reservation.queue_id = queue.id + ) AS quarantined_reservation, + EXISTS ( + SELECT 1 FROM target_scans scan WHERE scan.queue_id = queue.id + ) AS prior_scan, + EXISTS ( + SELECT 1 FROM target_queue_policy_events event + WHERE event.queue_id = queue.id + ) AS prior_policy_event + FROM target_queue queue + WHERE queue.source = ? AND queue.platform = 'docker' + AND queue.target NOT LIKE '%@%' AND queue.normalized_target NOT LIKE '%@%' + AND queue.status NOT IN ('done','failed') + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_repositories member + WHERE member.experiment_id = ? + AND member.repository_queue_id = queue.id + ) + ORDER BY queue.id LIMIT ?{lock_suffix}''', + (source, experiment_id, maximum + 1), + ).fetchall() + if len(rows) > maximum: + raise RuntimeError( + f'Docker depth hold selection exceeds its reviewed {maximum}-row bound' + ) + entries = [] + conflicts = [] + fence_fields = ( + 'lease_owner', 'lease_token', 'claim_batch', 'leased_at', 'lease_expires_at', + 'current_result_reservation_id', 'claim_event_id', 'resolver_token', + ) + for row in rows: + status = str(row['status'] or '') + query = str(row['query'] or '') + reason = None + if status == 'cold': + reason = 'independently_cold' + elif status in ('done', 'failed'): + reason = 'terminal_status' + elif status == 'quarantined' or bool(row['quarantined_reservation']): + reason = 'quarantined' + elif row['target_scan_id'] is not None or bool(row['prior_scan']): + reason = 'previously_scanned' + elif status not in ('pending', 'deferred'): + reason = 'ineligible_status' + elif any(row[field] is not None for field in fence_fields): + reason = 'active_claim_fence' + elif str(row['resolver_state'] or '') == 'resolving': + reason = 'active_resolver_fence' + elif bool(row['active_reservation']) or bool(row['active_docker_blob']): + reason = 'active_content_fence' + elif bool(row['prior_policy_event']): + reason = 'unrelated_policy_event' + elif not query: + reason = 'missing_query_identity' + if reason is None: + entries.append({ + 'queue_id': int(row['id']), + 'source': str(row['source']), + 'platform': str(row['platform']), + 'query': query, + 'prior_status': status, + 'prior_updated_at': str(row['updated_at']), + }) + else: + conflicts.append({ + 'queue_id': int(row['id']), + 'source': str(row['source']), + 'platform': str(row['platform']), + 'query': query, + 'status': status, + 'prior_updated_at': str(row['updated_at']), + 'reason': reason, + }) + return entries, conflicts + + +def _hold_manifest_document(authority, experiment_row, entries, conflicts): + return { + 'schema': 1, + 'type': DOCKER_DEPTH_HOLD_MANIFEST_TYPE, + 'experiment_id': int(experiment_row['id']), + 'experiment_key': authority['experiment_key'], + 'source': authority['source'], + 'config_sha256': authority['config_sha256'], + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'selector_sha256': authority['selector_sha256'], + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'plan_sha256': str(experiment_row['plan_sha256']), + 'reason_code': DOCKER_DEPTH_HOLD_REASON, + 'entry_count': len(entries), + 'conflict_count': len(conflicts), + 'selection_sha256': _canonical_sha256(entries), + 'conflicts_sha256': _canonical_sha256(conflicts), + 'entries': entries, + 'conflicts': conflicts, + } + + +def _normalize_hold_entries(entries, maximum): + if not isinstance(entries, list) or len(entries) > maximum: + raise ValueError('Docker depth hold manifest entries exceed their bound') + required = { + 'queue_id', 'source', 'platform', 'query', 'prior_status', 'prior_updated_at', + } + normalized = [] + seen = set() + for raw in entries: + if not isinstance(raw, dict) or set(raw) != required: + raise ValueError('Docker depth hold manifest entry shape is invalid') + queue_id = _strict_positive_int(raw['queue_id'], 'queue_id') + if queue_id in seen: + raise ValueError('Docker depth hold manifest contains a duplicate queue row') + seen.add(queue_id) + item = { + 'queue_id': queue_id, + 'source': str(raw['source'] or ''), + 'platform': str(raw['platform'] or ''), + 'query': str(raw['query'] or ''), + 'prior_status': str(raw['prior_status'] or ''), + 'prior_updated_at': str(raw['prior_updated_at'] or ''), + } + if ( + item['source'] != 'dockerhub' + or item['platform'] != 'docker' + or not item['query'] + or item['prior_status'] not in ('pending', 'deferred') + or not item['prior_updated_at'] + ): + raise ValueError('Docker depth hold manifest entry identity is invalid') + normalized.append(item) + normalized.sort(key=lambda item: item['queue_id']) + return normalized + + +def _normalize_hold_conflicts(conflicts, maximum): + if not isinstance(conflicts, list) or len(conflicts) > maximum: + raise ValueError('Docker depth hold manifest conflicts exceed their bound') + required = { + 'queue_id', 'source', 'platform', 'query', 'status', 'prior_updated_at', 'reason', + } + allowed_reasons = { + 'independently_cold', 'terminal_status', 'quarantined', 'previously_scanned', + 'ineligible_status', 'active_claim_fence', 'active_resolver_fence', + 'active_content_fence', 'unrelated_policy_event', 'missing_query_identity', + } + normalized = [] + seen = set() + for raw in conflicts: + if not isinstance(raw, dict) or set(raw) != required: + raise ValueError('Docker depth hold conflict shape is invalid') + queue_id = _strict_positive_int(raw['queue_id'], 'queue_id') + if queue_id in seen: + raise ValueError('Docker depth hold conflicts contain a duplicate queue row') + seen.add(queue_id) + item = { + 'queue_id': queue_id, + 'source': str(raw['source'] or ''), + 'platform': str(raw['platform'] or ''), + 'query': str(raw['query'] or ''), + 'status': str(raw['status'] or ''), + 'prior_updated_at': str(raw['prior_updated_at'] or ''), + 'reason': str(raw['reason'] or ''), + } + if ( + item['source'] != 'dockerhub' + or item['platform'] != 'docker' + or not item['prior_updated_at'] + or item['reason'] not in allowed_reasons + ): + raise ValueError('Docker depth hold conflict identity is invalid') + normalized.append(item) + normalized.sort(key=lambda item: item['queue_id']) + return normalized + + +def validate_docker_depth_hold_manifest( + manifest, experiment=None, provenance_policy_sha256=None, + max_rows=DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, +): + maximum = min( + DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, + _strict_positive_int(max_rows, 'max_rows'), + ) + expected_keys = { + 'schema', 'type', 'experiment_id', 'experiment_key', 'source', + 'config_sha256', 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', 'plan_sha256', 'reason_code', 'entry_count', + 'conflict_count', 'selection_sha256', 'conflicts_sha256', 'entries', 'conflicts', + } + if ( + not isinstance(manifest, dict) + or set(manifest) != expected_keys + or type(manifest.get('schema')) is not int + or manifest.get('schema') != 1 + or manifest.get('type') != DOCKER_DEPTH_HOLD_MANIFEST_TYPE + ): + raise ValueError('Docker depth hold manifest shape is invalid') + entries = _normalize_hold_entries(manifest['entries'], maximum) + conflicts = _normalize_hold_conflicts(manifest['conflicts'], maximum) + if entries != manifest['entries'] or conflicts != manifest['conflicts']: + raise ValueError('Docker depth hold manifest rows are not canonical') + if len(entries) + len(conflicts) > maximum: + raise ValueError('Docker depth hold manifest selection exceeds its bound') + if set(item['queue_id'] for item in entries).intersection( + item['queue_id'] for item in conflicts + ): + raise ValueError('Docker depth hold manifest entry and conflict sets overlap') + hashes = ( + 'config_sha256', 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', 'plan_sha256', 'selection_sha256', + 'conflicts_sha256', + ) + if any(not _valid_sha256(manifest.get(name)) for name in hashes): + raise ValueError('Docker depth hold manifest contains an invalid hash') + entry_count = _strict_nonnegative_int(manifest['entry_count'], 'entry_count') + conflict_count = _strict_nonnegative_int( + manifest['conflict_count'], 'conflict_count', + ) + _strict_identifier(manifest['experiment_key'], 'experiment_key') + if ( + _strict_positive_int(manifest['experiment_id'], 'experiment_id') <= 0 + or manifest['source'] != 'dockerhub' + or manifest['reason_code'] != DOCKER_DEPTH_HOLD_REASON + or entry_count != len(entries) + or conflict_count != len(conflicts) + or manifest['selection_sha256'] != _canonical_sha256(entries) + or manifest['conflicts_sha256'] != _canonical_sha256(conflicts) + ): + raise ValueError('Docker depth hold manifest evidence conflicts') + if experiment is not None: + authority = _docker_depth_authority( + experiment, + provenance_policy_sha256 or manifest['provenance_policy_sha256'], + ) + for manifest_name, authority_name in ( + ('experiment_key', 'experiment_key'), + ('source', 'source'), + ('config_sha256', 'config_sha256'), + ('ordered_queries_sha256', 'ordered_queries_sha256'), + ('selector_sha256', 'selector_sha256'), + ('provenance_policy_sha256', 'provenance_policy_sha256'), + ): + if manifest[manifest_name] != authority[authority_name]: + raise ValueError('Docker depth hold manifest authority drifted') + normalized = copy.deepcopy(manifest) + return normalized, _canonical_sha256(normalized) + + +def generate_docker_depth_hold_manifest( + db, experiment, provenance_policy_sha256, + max_rows=DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, +): + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + conn = _require_postgres_experiment_db(db, 'Docker depth hold planning') + maximum = min( + DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, + _strict_positive_int(max_rows, 'max_rows'), + ) + try: + row = _experiment_row(conn, authority, False) + if str(row['state']) != 'planned' or not _valid_sha256(row['plan_sha256']): + raise RuntimeError('Docker depth hold planning requires an immutable planned cohort') + if _experiment_fence_active(row): + raise RuntimeError('Docker depth experiment has an active authority fence') + plan = _stored_cohort_plan(conn, row, authority) + if canonical_docker_depth_plan_hash(plan) != str(row['plan_sha256']): + raise RuntimeError('Docker depth hold planning found plan hash drift') + _require_planned_cohort_current(conn, plan, False) + entries, conflicts = _noncohort_hold_snapshot( + conn, int(row['id']), authority['source'], maximum, False, + ) + manifest = _hold_manifest_document(authority, row, entries, conflicts) + manifest, manifest_sha256 = validate_docker_depth_hold_manifest( + manifest, experiment, provenance_policy_sha256, maximum, + ) + conn.commit() + return manifest, manifest_sha256 + except Exception: + conn.rollback() + raise + + +def apply_docker_depth_hold_manifest( + db, experiment, provenance_policy_sha256, manifest, manifest_sha256, + max_rows=DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, +): + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + manifest, expected_sha256 = validate_docker_depth_hold_manifest( + manifest, experiment, provenance_policy_sha256, max_rows, + ) + if str(manifest_sha256 or '') != expected_sha256: + raise ValueError('Docker depth hold manifest hash conflicts') + conn = _require_postgres_experiment_db(db, 'Docker depth hold application') + maximum = min( + DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, + _strict_positive_int(max_rows, 'max_rows'), + ) + now = datetime.now(timezone.utc).isoformat(timespec='seconds') + try: + row = _locked_experiment_row(conn, authority) + if _experiment_fence_active(row): + raise RuntimeError('Docker depth experiment has an active authority fence') + if int(row['id']) != manifest['experiment_id']: + raise RuntimeError('Docker depth hold manifest experiment identity drifted') + if str(row['plan_sha256'] or '') != manifest['plan_sha256']: + raise RuntimeError('Docker depth hold manifest plan identity drifted') + plan = _stored_cohort_plan(conn, row, authority) + if canonical_docker_depth_plan_hash(plan) != manifest['plan_sha256']: + raise RuntimeError('Docker depth hold manifest plan identity drifted') + _require_planned_cohort_current(conn, plan, True) + stored_manifest_sha256 = str(row['hold_manifest_sha256'] or '') + if stored_manifest_sha256: + _require_persisted_experiment_authority( + db, conn, row, authority, now, + ) + if ( + stored_manifest_sha256 != expected_sha256 + or str(row['state']) != 'holding' + ): + raise RuntimeError('Docker depth reviewed hold authority conflicts') + reviewed_rows = conn.execute( + '''SELECT queue_id, reason_code, manifest_sha256 + FROM target_queue_policy_events + WHERE experiment_id = ? AND action = 'cold' + ORDER BY queue_id, id''', + (row['id'],), + ).fetchall() + if ( + [int(item['queue_id']) for item in reviewed_rows] + != [item['queue_id'] for item in manifest['entries']] + or any( + str(item['reason_code']) != DOCKER_DEPTH_HOLD_REASON + or str(item['manifest_sha256']) != expected_sha256 + for item in reviewed_rows + ) + ): + raise RuntimeError('Docker depth reviewed hold set changed') + else: + if str(row['state']) != 'planned': + raise RuntimeError('Docker depth experiment is not ready for reviewed holds') + entries, conflicts = _noncohort_hold_snapshot( + conn, int(row['id']), authority['source'], maximum, True, + ) + current = _hold_manifest_document(authority, row, entries, conflicts) + if current != manifest: + raise RuntimeError('Docker depth hold selection drifted after review') + entries = db._normalize_target_queue_policy_entries( + manifest['entries'], 'cold', maximum, + ) if manifest['entries'] else [] + result = db._cold_target_queue_rows_locked( + entries, + reason_code=DOCKER_DEPTH_HOLD_REASON, + config_sha256=authority['config_sha256'], + policy_sha256=authority['provenance_policy_sha256'], + manifest_sha256=expected_sha256, + experiment_id=int(row['id']), + now=now, + ) + if not stored_manifest_sha256: + cursor = conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'holding', hold_manifest_sha256 = ?, + activated_at = COALESCE(activated_at, ?), updated_at = ? + WHERE id = ? AND state = 'planned' + AND plan_sha256 = ? AND hold_manifest_sha256 IS NULL''', + (expected_sha256, now, now, row['id'], manifest['plan_sha256']), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError('Docker depth hold activation lost its authority fence') + conn.commit() + return { + **result, + 'experiment_id': int(row['id']), + 'state': 'holding' if not stored_manifest_sha256 else str(row['state']), + 'conflicts': manifest['conflict_count'], + } + except Exception: + conn.rollback() + raise + + +def _experiment_reactivation_entries(conn, experiment_id, maximum, lock_rows): + lock_suffix = ' FOR UPDATE OF queue, cold_event' if lock_rows else '' + rows = conn.execute( + f'''SELECT queue.*, cold_event.id AS cold_event_id, + cold_event.prior_status AS restore_status, + reverse_event.id AS reverse_event_id + FROM target_queue_policy_events cold_event + JOIN target_queue queue ON queue.id = cold_event.queue_id + LEFT JOIN target_queue_policy_events reverse_event + ON reverse_event.reverses_event_id = cold_event.id + WHERE cold_event.experiment_id = ? AND cold_event.action = 'cold' + AND reverse_event.id IS NULL + ORDER BY queue.id, cold_event.id LIMIT ?{lock_suffix}''', + (experiment_id, maximum + 1), + ).fetchall() + if len(rows) > maximum: + raise RuntimeError( + f'Docker depth reactivation selection exceeds its reviewed {maximum}-row bound' + ) + entries = [] + seen = set() + for row in rows: + queue_id = int(row['id']) + if queue_id in seen or row['reverse_event_id'] is not None or row['status'] != 'cold': + raise RuntimeError('Docker depth reactivation evidence conflicts') + seen.add(queue_id) + entries.append({ + 'queue_id': queue_id, + 'source': str(row['source']), + 'platform': str(row['platform']), + 'query': str(row['query']), + 'cold_event_id': int(row['cold_event_id']), + 'restore_status': str(row['restore_status']), + 'prior_updated_at': str(row['updated_at']), + }) + return entries + + +def _reactivation_manifest_document(authority, experiment_row, entries): + return { + 'schema': 1, + 'type': DOCKER_DEPTH_REACTIVATION_MANIFEST_TYPE, + 'experiment_id': int(experiment_row['id']), + 'experiment_key': authority['experiment_key'], + 'source': authority['source'], + 'config_sha256': authority['config_sha256'], + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'selector_sha256': authority['selector_sha256'], + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'plan_sha256': str(experiment_row['plan_sha256']), + 'hold_manifest_sha256': str(experiment_row['hold_manifest_sha256']), + 'reason_code': DOCKER_DEPTH_RELEASE_REASON, + 'entry_count': len(entries), + 'selection_sha256': _canonical_sha256(entries), + 'entries': entries, + } + + +def _normalize_reactivation_entries(entries, maximum): + if not isinstance(entries, list) or len(entries) > maximum: + raise ValueError('Docker depth reactivation entries exceed their bound') + required = { + 'queue_id', 'source', 'platform', 'query', 'cold_event_id', + 'restore_status', 'prior_updated_at', + } + normalized = [] + queue_ids = set() + event_ids = set() + for raw in entries: + if not isinstance(raw, dict) or set(raw) != required: + raise ValueError('Docker depth reactivation entry shape is invalid') + queue_id = _strict_positive_int(raw['queue_id'], 'queue_id') + event_id = _strict_positive_int(raw['cold_event_id'], 'cold_event_id') + if queue_id in queue_ids or event_id in event_ids: + raise ValueError('Docker depth reactivation identity is duplicated') + queue_ids.add(queue_id) + event_ids.add(event_id) + item = { + 'queue_id': queue_id, + 'source': str(raw['source'] or ''), + 'platform': str(raw['platform'] or ''), + 'query': str(raw['query'] or ''), + 'cold_event_id': event_id, + 'restore_status': str(raw['restore_status'] or ''), + 'prior_updated_at': str(raw['prior_updated_at'] or ''), + } + if ( + item['source'] != 'dockerhub' + or item['platform'] != 'docker' + or not item['query'] + or item['restore_status'] not in ('pending', 'deferred') + or not item['prior_updated_at'] + ): + raise ValueError('Docker depth reactivation entry identity is invalid') + normalized.append(item) + normalized.sort(key=lambda item: item['queue_id']) + return normalized + + +def validate_docker_depth_reactivation_manifest( + manifest, experiment=None, provenance_policy_sha256=None, + max_rows=DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, +): + maximum = min( + DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, + _strict_positive_int(max_rows, 'max_rows'), + ) + expected_keys = { + 'schema', 'type', 'experiment_id', 'experiment_key', 'source', + 'config_sha256', 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', 'plan_sha256', 'hold_manifest_sha256', + 'reason_code', 'entry_count', 'selection_sha256', 'entries', + } + if ( + not isinstance(manifest, dict) + or set(manifest) != expected_keys + or type(manifest.get('schema')) is not int + or manifest.get('schema') != 1 + or manifest.get('type') != DOCKER_DEPTH_REACTIVATION_MANIFEST_TYPE + ): + raise ValueError('Docker depth reactivation manifest shape is invalid') + entries = _normalize_reactivation_entries(manifest['entries'], maximum) + if entries != manifest['entries']: + raise ValueError('Docker depth reactivation entries are not canonical') + for name in ( + 'config_sha256', 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', 'plan_sha256', 'hold_manifest_sha256', + 'selection_sha256', + ): + if not _valid_sha256(manifest.get(name)): + raise ValueError('Docker depth reactivation manifest contains an invalid hash') + entry_count = _strict_nonnegative_int(manifest['entry_count'], 'entry_count') + _strict_identifier(manifest['experiment_key'], 'experiment_key') + if ( + _strict_positive_int(manifest['experiment_id'], 'experiment_id') <= 0 + or manifest['source'] != 'dockerhub' + or manifest['reason_code'] != DOCKER_DEPTH_RELEASE_REASON + or entry_count != len(entries) + or manifest['selection_sha256'] != _canonical_sha256(entries) + ): + raise ValueError('Docker depth reactivation evidence conflicts') + if experiment is not None: + authority = _docker_depth_authority( + experiment, + provenance_policy_sha256 or manifest['provenance_policy_sha256'], + ) + for manifest_name, authority_name in ( + ('experiment_key', 'experiment_key'), ('source', 'source'), + ('config_sha256', 'config_sha256'), + ('ordered_queries_sha256', 'ordered_queries_sha256'), + ('selector_sha256', 'selector_sha256'), + ('provenance_policy_sha256', 'provenance_policy_sha256'), + ): + if manifest[manifest_name] != authority[authority_name]: + raise ValueError('Docker depth reactivation authority drifted') + normalized = copy.deepcopy(manifest) + return normalized, _canonical_sha256(normalized) + + +def generate_docker_depth_reactivation_manifest( + db, experiment, provenance_policy_sha256, + max_rows=DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, +): + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + conn = _require_postgres_experiment_db(db, 'Docker depth reactivation planning') + maximum = min( + DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, + _strict_positive_int(max_rows, 'max_rows'), + ) + try: + row = _experiment_row(conn, authority, False) + if ( + str(row['state']) != 'completed' + or not _valid_sha256(row['plan_sha256']) + or not _valid_sha256(row['hold_manifest_sha256']) + ): + raise RuntimeError('Docker depth reactivation requires a completed experiment') + now = datetime.now(timezone.utc).isoformat(timespec='seconds') + reason = _persisted_experiment_drift_reason(db, row, authority, now) + if reason: + raise RuntimeError(f'Docker depth persisted authority drifted: {reason}') + entries = _experiment_reactivation_entries( + conn, int(row['id']), maximum, False, + ) + for entry in entries: + queue = conn.execute( + 'SELECT * FROM target_queue WHERE id = ?', + (entry['queue_id'],), + ).fetchone() + db._require_target_queue_policy_unfenced(queue, lock_rows=False) + manifest = _reactivation_manifest_document(authority, row, entries) + manifest, manifest_sha256 = validate_docker_depth_reactivation_manifest( + manifest, experiment, provenance_policy_sha256, maximum, + ) + conn.commit() + return manifest, manifest_sha256 + except Exception: + conn.rollback() + raise + + +def _require_released_reactivation_replay( + conn, experiment_id, manifest, manifest_sha256, +): + rows = conn.execute( + '''SELECT cold_event.id AS cold_event_id, cold_event.queue_id, + cold_event.reason_code AS cold_reason_code, + cold_event.config_sha256 AS cold_config_sha256, + cold_event.policy_sha256 AS cold_policy_sha256, + cold_event.manifest_sha256 AS cold_manifest_sha256, + reverse_event.id AS reverse_event_id, + reverse_event.action AS reverse_action, + reverse_event.experiment_id AS reverse_experiment_id, + reverse_event.reason_code AS reverse_reason_code, + reverse_event.config_sha256 AS reverse_config_sha256, + reverse_event.policy_sha256 AS reverse_policy_sha256, + reverse_event.manifest_sha256 AS reverse_manifest_sha256, + queue.status AS queue_status + FROM target_queue_policy_events cold_event + JOIN target_queue queue ON queue.id = cold_event.queue_id + LEFT JOIN target_queue_policy_events reverse_event + ON reverse_event.reverses_event_id = cold_event.id + WHERE cold_event.experiment_id = ? AND cold_event.action = 'cold' + ORDER BY cold_event.queue_id, cold_event.id''', + (experiment_id,), + ).fetchall() + expected = [ + (entry['queue_id'], entry['cold_event_id']) for entry in manifest['entries'] + ] + actual = [(int(row['queue_id']), int(row['cold_event_id'])) for row in rows] + if actual != expected or any( + str(row['cold_reason_code']) not in ( + DOCKER_DEPTH_HOLD_REASON, DOCKER_DEPTH_DYNAMIC_HOLD_REASON, + ) + or str(row['cold_config_sha256']) != manifest['config_sha256'] + or str(row['cold_policy_sha256']) != manifest['provenance_policy_sha256'] + or str(row['cold_manifest_sha256']) != manifest['hold_manifest_sha256'] + or row['reverse_event_id'] is None + or str(row['reverse_action']) != 'reactivate' + or row['reverse_experiment_id'] is None + or int(row['reverse_experiment_id']) != experiment_id + or str(row['reverse_reason_code']) != DOCKER_DEPTH_RELEASE_REASON + or str(row['reverse_config_sha256']) != manifest['config_sha256'] + or str(row['reverse_policy_sha256']) != manifest['provenance_policy_sha256'] + or str(row['reverse_manifest_sha256']) != manifest_sha256 + or str(row['queue_status']) != manifest['entries'][index]['restore_status'] + for index, row in enumerate(rows) + ): + raise RuntimeError('Docker depth released hold set changed') + + +def apply_docker_depth_reactivation_manifest( + db, experiment, provenance_policy_sha256, manifest, manifest_sha256, + max_rows=DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, +): + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + manifest, expected_sha256 = validate_docker_depth_reactivation_manifest( + manifest, experiment, provenance_policy_sha256, max_rows, + ) + if str(manifest_sha256 or '') != expected_sha256: + raise ValueError('Docker depth reactivation manifest hash conflicts') + conn = _require_postgres_experiment_db(db, 'Docker depth reactivation') + maximum = min( + DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, + _strict_positive_int(max_rows, 'max_rows'), + ) + now = datetime.now(timezone.utc).isoformat(timespec='seconds') + try: + row = _locked_experiment_row(conn, authority) + if ( + int(row['id']) != manifest['experiment_id'] + or str(row['plan_sha256'] or '') != manifest['plan_sha256'] + or str(row['hold_manifest_sha256'] or '') != manifest['hold_manifest_sha256'] + ): + raise RuntimeError('Docker depth reactivation experiment identity drifted') + plan = _stored_cohort_plan(conn, row, authority) + if canonical_docker_depth_plan_hash(plan) != manifest['plan_sha256']: + raise RuntimeError('Docker depth reactivation cohort identity drifted') + released = str(row['state']) == 'released' + if released: + _require_released_reactivation_replay( + conn, int(row['id']), manifest, expected_sha256, + ) + else: + if str(row['state']) != 'completed': + raise RuntimeError('Docker depth reactivation requires a completed experiment') + _require_persisted_experiment_authority( + db, conn, row, authority, now, + ) + current_entries = _experiment_reactivation_entries( + conn, int(row['id']), maximum, True, + ) + current = _reactivation_manifest_document(authority, row, current_entries) + if current != manifest: + raise RuntimeError('Docker depth reactivation selection drifted after review') + entries = db._normalize_target_queue_policy_entries( + manifest['entries'], 'reactivate', maximum, + ) if manifest['entries'] else [] + result = db._reactivate_target_queue_rows_locked( + entries, + reason_code=DOCKER_DEPTH_RELEASE_REASON, + config_sha256=authority['config_sha256'], + policy_sha256=authority['provenance_policy_sha256'], + manifest_sha256=expected_sha256, + experiment_id=int(row['id']), + now=now, + ) + if not released: + remaining = conn.execute( + '''SELECT COUNT(*) AS count + FROM target_queue_policy_events cold_event + LEFT JOIN target_queue_policy_events reverse_event + ON reverse_event.reverses_event_id = cold_event.id + WHERE cold_event.experiment_id = ? AND cold_event.action = 'cold' + AND reverse_event.id IS NULL''', + (row['id'],), + ).fetchone()['count'] + if int(remaining): + raise RuntimeError('Docker depth reactivation left owned cold events unreversed') + cursor = conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'released', released_at = COALESCE(released_at, ?), + updated_at = ? + WHERE id = ? AND state = 'completed' + AND hold_manifest_sha256 = ?''', + (now, now, row['id'], manifest['hold_manifest_sha256']), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError('Docker depth release lost its authority fence') + conn.commit() + return { + **result, + 'experiment_id': int(row['id']), + 'state': 'released', + } + except Exception: + conn.rollback() + raise + + +def _docker_depth_resolver_refund_text_sha256(value): + return hashlib.sha256(str(value).encode('utf-8')).hexdigest() + + +def _docker_depth_resolver_refund_log(path): + absolute = os.path.abspath(os.fspath(path)) + size = os.path.getsize(absolute) + if size < 1 or size > DOCKER_DEPTH_RESOLVER_REFUND_LOG_MAX_BYTES: + raise RuntimeError('Docker depth resolver refund log size is invalid') + with open(absolute, 'rb') as handle: + payload = handle.read(DOCKER_DEPTH_RESOLVER_REFUND_LOG_MAX_BYTES + 1) + if len(payload) != size or len(payload) > DOCKER_DEPTH_RESOLVER_REFUND_LOG_MAX_BYTES: + raise RuntimeError('Docker depth resolver refund log changed while reading') + return payload, hashlib.sha256(payload).hexdigest() + + +def _docker_depth_resolver_refund_entry(row, log_lines): + target = str(row['normalized_target'] or '') + if not target: + raise RuntimeError('Docker depth resolver refund target identity is absent') + exact_error = ( + f'Unable to fetch Docker Hub tags for {target}: ' + f'{DOCKER_DEPTH_RESOLVER_REFUND_OLD_ERROR}' + ) + bug_count = sum(exact_error in line for line in log_lines) + if bug_count != DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS: + return None + work_state = str(row['work_state'] or '') + attempts = int(row['resolver_attempts']) + last_error = str(row['last_error_code'] or '') + expected_shape = ( + work_state == 'pending' + and attempts == DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS + and last_error == DOCKER_DEPTH_RESOLVER_REFUND_OLD_ERROR + ) or ( + work_state == 'held' + and attempts == DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS + 1 + and last_error == 'resolver_attempt_limit' + ) + if not expected_shape: + raise RuntimeError('Docker depth bug-tainted resolver state is not refundable') + entry = { + 'experiment_repository_id': int(row['id']), + 'repository_queue_id': int(row['effective_repository_queue_id']), + 'query_ordinal': int(row['query_ordinal']), + 'repository_rank': int(row['repository_rank']), + 'recovery_kind': DOCKER_DEPTH_RESOLVER_REFUND_KIND, + 'prior_work_state': work_state, + 'next_work_state': 'pending', + 'prior_resolver_attempts': attempts, + 'refund_attempts': DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS, + 'next_resolver_attempts': attempts - DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS, + 'confirmed_bug_event_count': bug_count, + 'prior_error_code_sha256': _docker_depth_resolver_refund_text_sha256(last_error), + 'target_identity_sha256': _docker_depth_resolver_refund_text_sha256(target), + } + entry['entry_evidence_sha256'] = _canonical_sha256(entry) + return entry + + +def validate_docker_depth_resolver_refund_manifest( + manifest, experiment=None, provenance_policy_sha256=None, +): + expected_keys = { + 'schema', 'type', 'version', 'experiment_id', 'experiment_key', 'source', + 'config_sha256', 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', 'plan_sha256', 'hold_reason_code', + 'recovery_kind', 'refund_attempts', 'evidence_log_name', + 'evidence_log_sha256', 'entry_count', 'held_entry_count', + 'selection_sha256', 'entries', + } + if ( + not isinstance(manifest, dict) + or set(manifest) != expected_keys + or type(manifest.get('schema')) is not int + or manifest.get('schema') != 1 + or type(manifest.get('version')) is not int + or manifest.get('version') != 1 + or manifest.get('type') != DOCKER_DEPTH_RESOLVER_REFUND_MANIFEST_TYPE + ): + raise ValueError('Docker depth resolver refund manifest shape is invalid') + _strict_identifier(manifest.get('experiment_key'), 'experiment_key') + entries = manifest.get('entries') + if not isinstance(entries, list) or not entries: + raise ValueError('Docker depth resolver refund manifest entries are invalid') + normalized_entries = [] + member_ids = set() + for raw in entries: + expected_entry_keys = { + 'experiment_repository_id', 'repository_queue_id', 'query_ordinal', + 'repository_rank', 'recovery_kind', 'prior_work_state', + 'next_work_state', 'prior_resolver_attempts', 'refund_attempts', + 'next_resolver_attempts', 'confirmed_bug_event_count', + 'prior_error_code_sha256', 'target_identity_sha256', + 'entry_evidence_sha256', + } + if not isinstance(raw, dict) or set(raw) != expected_entry_keys: + raise ValueError('Docker depth resolver refund entry shape is invalid') + item = copy.deepcopy(raw) + member_id = _strict_positive_int( + item['experiment_repository_id'], 'experiment_repository_id', + ) + if member_id in member_ids: + raise ValueError('Docker depth resolver refund entry is duplicated') + member_ids.add(member_id) + _strict_positive_int(item['repository_queue_id'], 'repository_queue_id') + _strict_nonnegative_int(item['query_ordinal'], 'query_ordinal') + _strict_positive_int(item['repository_rank'], 'repository_rank') + prior_attempts = _strict_positive_int( + item['prior_resolver_attempts'], 'prior_resolver_attempts', + ) + next_attempts = _strict_nonnegative_int( + item['next_resolver_attempts'], 'next_resolver_attempts', + ) + if ( + item['recovery_kind'] != DOCKER_DEPTH_RESOLVER_REFUND_KIND + or item['prior_work_state'] not in ('pending', 'held') + or item['next_work_state'] != 'pending' + or item['refund_attempts'] != DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS + or item['confirmed_bug_event_count'] != DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS + or next_attempts != prior_attempts - DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS + or next_attempts < 0 + or any(not _valid_sha256(item.get(name)) for name in ( + 'prior_error_code_sha256', 'target_identity_sha256', + 'entry_evidence_sha256', + )) + ): + raise ValueError('Docker depth resolver refund entry evidence is invalid') + evidence = dict(item) + evidence.pop('entry_evidence_sha256') + if item['entry_evidence_sha256'] != _canonical_sha256(evidence): + raise ValueError('Docker depth resolver refund entry hash conflicts') + normalized_entries.append(item) + normalized_entries.sort(key=lambda item: item['experiment_repository_id']) + if normalized_entries != entries: + raise ValueError('Docker depth resolver refund entries are not canonical') + hashes = ( + 'config_sha256', 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', 'plan_sha256', 'evidence_log_sha256', + 'selection_sha256', + ) + if any(not _valid_sha256(manifest.get(name)) for name in hashes): + raise ValueError('Docker depth resolver refund manifest hash is invalid') + if ( + _strict_positive_int(manifest.get('experiment_id'), 'experiment_id') <= 0 + or manifest.get('source') != 'dockerhub' + or manifest.get('hold_reason_code') != 'resolver_attempt_limit' + or manifest.get('recovery_kind') != DOCKER_DEPTH_RESOLVER_REFUND_KIND + or manifest.get('refund_attempts') != DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS + or manifest.get('evidence_log_name') != 'dockerhub.log' + or manifest.get('entry_count') != len(normalized_entries) + or manifest.get('held_entry_count') + != sum(item['prior_work_state'] == 'held' for item in normalized_entries) + or manifest.get('held_entry_count') != 1 + or manifest.get('selection_sha256') != _canonical_sha256(normalized_entries) + ): + raise ValueError('Docker depth resolver refund manifest evidence conflicts') + if experiment is not None: + authority = _docker_depth_authority( + experiment, + provenance_policy_sha256 or manifest['provenance_policy_sha256'], + ) + for manifest_name, authority_name in ( + ('experiment_key', 'experiment_key'), ('source', 'source'), + ('config_sha256', 'config_sha256'), + ('ordered_queries_sha256', 'ordered_queries_sha256'), + ('selector_sha256', 'selector_sha256'), + ('provenance_policy_sha256', 'provenance_policy_sha256'), + ): + if manifest[manifest_name] != authority[authority_name]: + raise ValueError('Docker depth resolver refund authority drifted') + normalized = copy.deepcopy(manifest) + normalized['entries'] = normalized_entries + if normalized != manifest: + raise ValueError('Docker depth resolver refund manifest is not canonical') + return normalized, _canonical_sha256(normalized) + + +def generate_docker_depth_resolver_refund_manifest( + db, experiment, provenance_policy_sha256, evidence_log_path, +): + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + conn = _require_postgres_experiment_db(db, 'Docker depth resolver refund planning') + log_payload, log_sha256 = _docker_depth_resolver_refund_log(evidence_log_path) + log_lines = log_payload.decode('utf-8', 'replace').splitlines() + try: + row = _experiment_row(conn, authority, False) + if ( + str(row['state']) != 'held' + or str(row['hold_reason_code'] or '') != 'resolver_attempt_limit' + or not _valid_sha256(row['plan_sha256']) + or _experiment_fence_active(row) + ): + raise RuntimeError('Docker depth resolver refund requires an attempt-limit hold') + members = conn.execute( + '''SELECT member.*, queue.id AS effective_repository_queue_id, + queue.normalized_target + FROM docker_depth_experiment_repositories member + JOIN target_queue queue + ON queue.id = COALESCE( + member.replacement_repository_queue_id, + member.repository_queue_id + ) + WHERE member.experiment_id = ? AND member.resolver_attempts >= ? + ORDER BY member.id''', + (row['id'], DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS), + ).fetchall() + entries = [] + for member in members: + entry = _docker_depth_resolver_refund_entry(member, log_lines) + if entry is not None: + entries.append(entry) + if not entries: + raise RuntimeError('No confirmed bug-tainted resolver attempts were found') + manifest = { + 'schema': 1, + 'type': DOCKER_DEPTH_RESOLVER_REFUND_MANIFEST_TYPE, + 'version': 1, + 'experiment_id': int(row['id']), + 'experiment_key': authority['experiment_key'], + 'source': authority['source'], + 'config_sha256': authority['config_sha256'], + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'selector_sha256': authority['selector_sha256'], + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'plan_sha256': str(row['plan_sha256']), + 'hold_reason_code': 'resolver_attempt_limit', + 'recovery_kind': DOCKER_DEPTH_RESOLVER_REFUND_KIND, + 'refund_attempts': DOCKER_DEPTH_RESOLVER_REFUND_ATTEMPTS, + 'evidence_log_name': 'dockerhub.log', + 'evidence_log_sha256': log_sha256, + 'entry_count': len(entries), + 'held_entry_count': sum( + item['prior_work_state'] == 'held' for item in entries + ), + 'selection_sha256': _canonical_sha256(entries), + 'entries': entries, + } + normalized, manifest_sha256 = validate_docker_depth_resolver_refund_manifest( + manifest, experiment, provenance_policy_sha256, + ) + conn.commit() + return normalized, manifest_sha256 + except Exception: + conn.rollback() + raise + + +def apply_docker_depth_resolver_refund_manifest( + db, experiment, provenance_policy_sha256, manifest, manifest_sha256, + evidence_log_path, +): + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + manifest, expected_sha256 = validate_docker_depth_resolver_refund_manifest( + manifest, experiment, provenance_policy_sha256, + ) + if str(manifest_sha256 or '') != expected_sha256: + raise ValueError('Docker depth resolver refund manifest hash conflicts') + _, current_log_sha256 = _docker_depth_resolver_refund_log(evidence_log_path) + if current_log_sha256 != manifest['evidence_log_sha256']: + raise RuntimeError('Docker depth resolver refund evidence log drifted') + conn = _require_postgres_experiment_db(db, 'Docker depth resolver refund') + now = datetime.now(timezone.utc).isoformat(timespec='seconds') + try: + row = _locked_experiment_row(conn, authority) + if ( + int(row['id']) != manifest['experiment_id'] + or str(row['plan_sha256'] or '') != manifest['plan_sha256'] + or _experiment_fence_active(row) + ): + raise RuntimeError('Docker depth resolver refund experiment identity drifted') + existing = conn.execute( + '''SELECT * FROM docker_depth_resolver_attempt_refunds + WHERE experiment_id = ? AND recovery_kind = ? ORDER BY id FOR UPDATE''', + (row['id'], DOCKER_DEPTH_RESOLVER_REFUND_KIND), + ).fetchall() + if existing: + expected_entries = { + item['experiment_repository_id']: item for item in manifest['entries'] + } + if ( + len(existing) != len(expected_entries) + or any( + int(audit['experiment_repository_id']) not in expected_entries + or str(audit['manifest_sha256']) != expected_sha256 + or str(audit['entry_evidence_sha256']) + != expected_entries[int(audit['experiment_repository_id'])][ + 'entry_evidence_sha256' + ] + for audit in existing + ) + ): + raise RuntimeError('Docker depth resolver refund audit conflicts') + conn.commit() + return { + 'experiment_id': int(row['id']), + 'refunded': 0, + 'duplicates': len(existing), + 'state': str(row['state']), + } + if ( + str(row['state']) != 'held' + or str(row['hold_reason_code'] or '') != 'resolver_attempt_limit' + ): + raise RuntimeError('Docker depth resolver refund requires an attempt-limit hold') + member_ids = [item['experiment_repository_id'] for item in manifest['entries']] + placeholders = ','.join('?' for _ in member_ids) + members = conn.execute( + f'''SELECT member.*, queue.id AS effective_repository_queue_id, + queue.normalized_target + FROM docker_depth_experiment_repositories member + JOIN target_queue queue + ON queue.id = COALESCE( + member.replacement_repository_queue_id, + member.repository_queue_id + ) + WHERE member.experiment_id = ? AND member.id IN ({placeholders}) + ORDER BY member.id FOR UPDATE OF member, queue''', + (row['id'], *member_ids), + ).fetchall() + if len(members) != len(member_ids): + raise RuntimeError('Docker depth resolver refund membership set drifted') + entries_by_id = { + item['experiment_repository_id']: item for item in manifest['entries'] + } + for member in members: + member_id = int(member['id']) + entry = entries_by_id.get(member_id) + last_error = str(member['last_error_code'] or '') + target = str(member['normalized_target'] or '') + if ( + entry is None + or int(member['effective_repository_queue_id']) + != entry['repository_queue_id'] + or int(member['query_ordinal']) != entry['query_ordinal'] + or int(member['repository_rank']) != entry['repository_rank'] + or str(member['work_state']) != entry['prior_work_state'] + or int(member['resolver_attempts']) != entry['prior_resolver_attempts'] + or _docker_depth_resolver_refund_text_sha256(last_error) + != entry['prior_error_code_sha256'] + or _docker_depth_resolver_refund_text_sha256(target) + != entry['target_identity_sha256'] + or any(member[name] is not None for name in ( + 'resolver_owner', 'resolver_token', 'resolver_expires_at', + )) + ): + raise RuntimeError('Docker depth resolver refund membership evidence drifted') + cursor = conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = 'pending', resolver_attempts = ?, + resolver_due_at = NULL, last_error_code = NULL, + updated_at = ? + WHERE id = ? AND experiment_id = ? AND work_state = ? + AND resolver_attempts = ? AND resolver_owner IS NULL + AND resolver_token IS NULL AND resolver_expires_at IS NULL''', + ( + entry['next_resolver_attempts'], now, member_id, row['id'], + entry['prior_work_state'], entry['prior_resolver_attempts'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError('Docker depth resolver refund lost its member fence') + conn.execute( + '''INSERT INTO docker_depth_resolver_attempt_refunds( + experiment_id, experiment_repository_id, repository_queue_id, + query_ordinal, repository_rank, recovery_kind, manifest_sha256, + entry_evidence_sha256, log_sha256, target_identity_sha256, + prior_error_code_sha256, prior_work_state, next_work_state, + prior_resolver_attempts, refund_attempts, next_resolver_attempts, + confirmed_bug_event_count, applied_at, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + row['id'], member_id, entry['repository_queue_id'], + entry['query_ordinal'], entry['repository_rank'], + DOCKER_DEPTH_RESOLVER_REFUND_KIND, expected_sha256, + entry['entry_evidence_sha256'], manifest['evidence_log_sha256'], + entry['target_identity_sha256'], entry['prior_error_code_sha256'], + entry['prior_work_state'], entry['next_work_state'], + entry['prior_resolver_attempts'], entry['refund_attempts'], + entry['next_resolver_attempts'], entry['confirmed_bug_event_count'], + now, now, + ), + ) + cursor = conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'resolving', hold_reason_code = NULL, held_at = NULL, + updated_at = ? + WHERE id = ? AND state = 'held' + AND hold_reason_code = 'resolver_attempt_limit' + AND fence_owner IS NULL AND fence_token IS NULL + AND fence_expires_at IS NULL''', + (now, row['id']), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError('Docker depth resolver refund lost its experiment fence') + refreshed = conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ? FOR UPDATE', + (row['id'],), + ).fetchone() + drift = _persisted_experiment_drift_reason(db, refreshed, authority, now) + if drift: + raise RuntimeError(f'Docker depth resolver refund left authority drift: {drift}') + conn.commit() + return { + 'experiment_id': int(row['id']), + 'refunded': len(manifest['entries']), + 'duplicates': 0, + 'state': 'resolving', + } + except Exception: + conn.rollback() + raise + + +def _docker_depth_resolver_disposition_entry(member, candidates): + target = str(member['normalized_target'] or '') + if not target: + raise RuntimeError('Docker depth resolver disposition target identity is absent') + if ( + str(member['work_state'] or '') != 'held' + or str(member['last_error_code'] or '') != 'resolver_attempt_limit' + or int(member['resolver_attempts']) < DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS + or int(member['selected_image_count']) != 0 + or any(member[name] is not None for name in ( + 'resolver_owner', 'resolver_token', 'resolver_expires_at', + 'resolver_due_at', + )) + ): + raise RuntimeError('Docker depth attempt-limit disposition state is invalid') + replacement = next( + (candidate for candidate in candidates if not candidate['conflict_reason']), + None, + ) + outcome = 'replaced' if replacement else 'skipped' + attempts = int(member['resolver_attempts']) + entry = { + 'experiment_repository_id': int(member['id']), + 'repository_queue_id': int(member['effective_repository_queue_id']), + 'query_ordinal': int(member['query_ordinal']), + 'repository_rank': int(member['repository_rank']), + 'disposition_kind': DOCKER_DEPTH_RESOLVER_DISPOSITION_KIND, + 'outcome': outcome, + 'prior_work_state': 'held', + 'next_work_state': 'pending' if replacement else 'skipped', + 'prior_resolver_attempts': attempts, + 'next_resolver_attempts': 0 if replacement else attempts, + 'prior_error_code_sha256': _docker_depth_resolver_refund_text_sha256( + member['last_error_code'] + ), + 'target_identity_sha256': _docker_depth_resolver_refund_text_sha256(target), + 'candidate_snapshot_sha256': _canonical_sha256(candidates), + 'inspected_candidate_count': len(candidates), + 'replacement_repository_queue_id': ( + replacement['repository_queue_id'] if replacement else None + ), + 'replacement_eligibility_page_id': ( + replacement['eligibility_page_id'] if replacement else None + ), + 'replacement_best_search_rank': ( + replacement['best_search_rank'] if replacement else None + ), + 'replacement_target_identity_sha256': ( + replacement['target_identity_sha256'] if replacement else None + ), + 'terminal_reason': ( + '' if replacement else DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON + ), + } + entry['entry_evidence_sha256'] = _canonical_sha256(entry) + return entry + + +def validate_docker_depth_resolver_disposition_manifest( + manifest, experiment=None, provenance_policy_sha256=None, +): + expected_keys = { + 'schema', 'type', 'version', 'experiment_id', 'experiment_key', 'source', + 'config_sha256', 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', 'plan_sha256', 'hold_reason_code', + 'disposition_kind', 'entry_count', 'selection_sha256', 'entries', + } + if ( + not isinstance(manifest, dict) + or set(manifest) != expected_keys + or type(manifest.get('schema')) is not int + or manifest.get('schema') != 1 + or type(manifest.get('version')) is not int + or manifest.get('version') != 1 + or manifest.get('type') != DOCKER_DEPTH_RESOLVER_DISPOSITION_MANIFEST_TYPE + ): + raise ValueError('Docker depth resolver disposition manifest shape is invalid') + _strict_identifier(manifest.get('experiment_key'), 'experiment_key') + entries = manifest.get('entries') + if not isinstance(entries, list) or len(entries) != 1: + raise ValueError('Docker depth resolver disposition requires one reviewed entry') + raw = entries[0] + expected_entry_keys = { + 'experiment_repository_id', 'repository_queue_id', 'query_ordinal', + 'repository_rank', 'disposition_kind', 'outcome', 'prior_work_state', + 'next_work_state', 'prior_resolver_attempts', 'next_resolver_attempts', + 'prior_error_code_sha256', 'target_identity_sha256', + 'candidate_snapshot_sha256', 'inspected_candidate_count', + 'replacement_repository_queue_id', 'replacement_eligibility_page_id', + 'replacement_best_search_rank', 'replacement_target_identity_sha256', + 'terminal_reason', 'entry_evidence_sha256', + } + if not isinstance(raw, dict) or set(raw) != expected_entry_keys: + raise ValueError('Docker depth resolver disposition entry shape is invalid') + entry = copy.deepcopy(raw) + _strict_positive_int(entry['experiment_repository_id'], 'experiment_repository_id') + _strict_positive_int(entry['repository_queue_id'], 'repository_queue_id') + _strict_nonnegative_int(entry['query_ordinal'], 'query_ordinal') + _strict_positive_int(entry['repository_rank'], 'repository_rank') + attempts = _strict_positive_int( + entry['prior_resolver_attempts'], 'prior_resolver_attempts', + ) + next_attempts = _strict_nonnegative_int( + entry['next_resolver_attempts'], 'next_resolver_attempts', + ) + inspected = _strict_nonnegative_int( + entry['inspected_candidate_count'], 'inspected_candidate_count', + ) + if inspected > 3000 or attempts < DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS: + raise ValueError('Docker depth resolver disposition range is invalid') + outcome = entry['outcome'] + replacement_fields = ( + 'replacement_repository_queue_id', 'replacement_eligibility_page_id', + 'replacement_best_search_rank', 'replacement_target_identity_sha256', + ) + if outcome == 'replaced': + for name in replacement_fields[:3]: + _strict_positive_int(entry[name], name) + if ( + entry['next_work_state'] != 'pending' + or next_attempts != 0 + or entry['terminal_reason'] != '' + or not _valid_sha256(entry['replacement_target_identity_sha256']) + or inspected < 1 + ): + raise ValueError('Docker depth replacement disposition is invalid') + elif outcome == 'skipped': + if ( + any(entry[name] is not None for name in replacement_fields) + or entry['next_work_state'] != 'skipped' + or next_attempts != attempts + or entry['terminal_reason'] != DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON + ): + raise ValueError('Docker depth skip disposition is invalid') + else: + raise ValueError('Docker depth resolver disposition outcome is invalid') + if ( + entry['disposition_kind'] != DOCKER_DEPTH_RESOLVER_DISPOSITION_KIND + or entry['prior_work_state'] != 'held' + or any(not _valid_sha256(entry.get(name)) for name in ( + 'prior_error_code_sha256', 'target_identity_sha256', + 'candidate_snapshot_sha256', 'entry_evidence_sha256', + )) + ): + raise ValueError('Docker depth resolver disposition evidence is invalid') + evidence = dict(entry) + evidence.pop('entry_evidence_sha256') + if entry['entry_evidence_sha256'] != _canonical_sha256(evidence): + raise ValueError('Docker depth resolver disposition entry hash conflicts') + hashes = ( + 'config_sha256', 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', 'plan_sha256', 'selection_sha256', + ) + if any(not _valid_sha256(manifest.get(name)) for name in hashes): + raise ValueError('Docker depth resolver disposition manifest hash is invalid') + if ( + _strict_positive_int(manifest.get('experiment_id'), 'experiment_id') <= 0 + or manifest.get('source') != 'dockerhub' + or manifest.get('hold_reason_code') != 'resolver_attempt_limit' + or manifest.get('disposition_kind') != DOCKER_DEPTH_RESOLVER_DISPOSITION_KIND + or manifest.get('entry_count') != 1 + or manifest.get('selection_sha256') != _canonical_sha256(entries) + ): + raise ValueError('Docker depth resolver disposition authority is invalid') + if experiment is not None: + authority = _docker_depth_authority( + experiment, + provenance_policy_sha256 or manifest['provenance_policy_sha256'], + ) + for manifest_name, authority_name in ( + ('experiment_key', 'experiment_key'), ('source', 'source'), + ('config_sha256', 'config_sha256'), + ('ordered_queries_sha256', 'ordered_queries_sha256'), + ('selector_sha256', 'selector_sha256'), + ('provenance_policy_sha256', 'provenance_policy_sha256'), + ): + if manifest[manifest_name] != authority[authority_name]: + raise ValueError('Docker depth resolver disposition authority drifted') + normalized = copy.deepcopy(manifest) + normalized['entries'] = [entry] + if normalized != manifest: + raise ValueError('Docker depth resolver disposition manifest is not canonical') + return normalized, _canonical_sha256(normalized) + + +def generate_docker_depth_resolver_disposition_manifest( + db, experiment, provenance_policy_sha256, +): + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + conn = _require_postgres_experiment_db( + db, 'Docker depth resolver disposition planning', + ) + try: + row = _experiment_row(conn, authority, False) + if ( + str(row['state']) != 'held' + or str(row['hold_reason_code'] or '') != 'resolver_attempt_limit' + or not _valid_sha256(row['plan_sha256']) + or _experiment_fence_active(row) + ): + raise RuntimeError( + 'Docker depth resolver disposition requires an attempt-limit hold' + ) + members = conn.execute( + '''SELECT member.*, queue.id AS effective_repository_queue_id, + queue.normalized_target + FROM docker_depth_experiment_repositories member + JOIN target_queue queue ON queue.id = COALESCE( + member.replacement_repository_queue_id, + member.repository_queue_id + ) + WHERE member.experiment_id = ? AND member.work_state = 'held' + AND member.last_error_code = 'resolver_attempt_limit' + ORDER BY member.id''', + (row['id'],), + ).fetchall() + if len(members) != 1: + raise RuntimeError( + 'Docker depth resolver disposition requires one held membership' + ) + member = members[0] + candidates = db._docker_depth_repository_replacement_candidates( + row, member, authority, lock=False, + ) + entry = _docker_depth_resolver_disposition_entry(member, candidates) + manifest = { + 'schema': 1, + 'type': DOCKER_DEPTH_RESOLVER_DISPOSITION_MANIFEST_TYPE, + 'version': 1, + 'experiment_id': int(row['id']), + 'experiment_key': authority['experiment_key'], + 'source': authority['source'], + 'config_sha256': authority['config_sha256'], + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'selector_sha256': authority['selector_sha256'], + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'plan_sha256': str(row['plan_sha256']), + 'hold_reason_code': 'resolver_attempt_limit', + 'disposition_kind': DOCKER_DEPTH_RESOLVER_DISPOSITION_KIND, + 'entry_count': 1, + 'selection_sha256': _canonical_sha256([entry]), + 'entries': [entry], + } + normalized, manifest_sha256 = ( + validate_docker_depth_resolver_disposition_manifest( + manifest, experiment, provenance_policy_sha256, + ) + ) + conn.commit() + return normalized, manifest_sha256 + except Exception: + conn.rollback() + raise + + +def apply_docker_depth_resolver_disposition_manifest( + db, experiment, provenance_policy_sha256, manifest, manifest_sha256, +): + authority = _docker_depth_authority(experiment, provenance_policy_sha256) + manifest, expected_sha256 = validate_docker_depth_resolver_disposition_manifest( + manifest, experiment, provenance_policy_sha256, + ) + if str(manifest_sha256 or '') != expected_sha256: + raise ValueError('Docker depth resolver disposition manifest hash conflicts') + conn = _require_postgres_experiment_db(db, 'Docker depth resolver disposition') + now = datetime.now(timezone.utc).isoformat(timespec='seconds') + entry = manifest['entries'][0] + try: + row = _locked_experiment_row(conn, authority) + if ( + int(row['id']) != manifest['experiment_id'] + or str(row['plan_sha256'] or '') != manifest['plan_sha256'] + or _experiment_fence_active(row) + ): + raise RuntimeError('Docker depth resolver disposition identity drifted') + existing = conn.execute( + '''SELECT * FROM docker_depth_resolver_dispositions + WHERE experiment_repository_id = ? AND disposition_kind = ? + FOR UPDATE''', + ( + entry['experiment_repository_id'], + DOCKER_DEPTH_RESOLVER_DISPOSITION_KIND, + ), + ).fetchone() + if existing: + if ( + int(existing['experiment_id']) != int(row['id']) + or str(existing['manifest_sha256']) != expected_sha256 + or str(existing['entry_evidence_sha256']) + != entry['entry_evidence_sha256'] + or str(existing['outcome']) != entry['outcome'] + ): + raise RuntimeError('Docker depth resolver disposition audit conflicts') + conn.commit() + return { + 'experiment_id': int(row['id']), 'applied': 0, 'duplicates': 1, + 'outcome': str(existing['outcome']), 'state': str(row['state']), + } + if ( + str(row['state']) != 'held' + or str(row['hold_reason_code'] or '') != 'resolver_attempt_limit' + ): + raise RuntimeError( + 'Docker depth resolver disposition requires an attempt-limit hold' + ) + member = conn.execute( + '''SELECT member.*, queue.id AS effective_repository_queue_id, + queue.normalized_target + FROM docker_depth_experiment_repositories member + JOIN target_queue queue ON queue.id = COALESCE( + member.replacement_repository_queue_id, + member.repository_queue_id + ) + WHERE member.id = ? AND member.experiment_id = ? + FOR UPDATE OF member, queue''', + (entry['experiment_repository_id'], row['id']), + ).fetchone() + if not member: + raise RuntimeError('Docker depth resolver disposition membership is absent') + target = str(member['normalized_target'] or '') + if ( + int(member['effective_repository_queue_id']) != entry['repository_queue_id'] + or int(member['query_ordinal']) != entry['query_ordinal'] + or int(member['repository_rank']) != entry['repository_rank'] + or str(member['work_state']) != entry['prior_work_state'] + or int(member['resolver_attempts']) != entry['prior_resolver_attempts'] + or str(member['last_error_code'] or '') != 'resolver_attempt_limit' + or int(member['selected_image_count']) != 0 + or _docker_depth_resolver_refund_text_sha256(target) + != entry['target_identity_sha256'] + or _docker_depth_resolver_refund_text_sha256(member['last_error_code']) + != entry['prior_error_code_sha256'] + or any(member[name] is not None for name in ( + 'resolver_owner', 'resolver_token', 'resolver_expires_at', + 'resolver_due_at', + )) + ): + raise RuntimeError('Docker depth resolver disposition evidence drifted') + if conn.execute( + '''SELECT 1 FROM docker_depth_experiment_selections + WHERE experiment_repository_id = ? LIMIT 1''', + (member['id'],), + ).fetchone(): + raise RuntimeError('Docker depth resolver disposition cannot replace selected work') + candidates = db._docker_depth_repository_replacement_candidates( + row, member, authority, lock=True, + ) + current_entry = _docker_depth_resolver_disposition_entry(member, candidates) + if current_entry != entry: + raise RuntimeError('Docker depth resolver disposition selection drifted after review') + for candidate in candidates: + if not candidate['conflict_reason']: + break + db._record_docker_depth_candidate_skip_locked( + row, member, candidate['repository_queue_id'], 'repository', + candidate['best_search_rank'], + {'repository_queue_id': candidate['repository_queue_id']}, + candidate['conflict_reason'], now, + ) + current_queue_id = int(member['effective_repository_queue_id']) + db._record_docker_depth_candidate_skip_locked( + row, member, current_queue_id, 'repository', + int(member['replacement_count']) + 1, + {'repository_queue_id': current_queue_id}, + DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON, now, + ) + replacement = next( + (candidate for candidate in candidates if not candidate['conflict_reason']), + None, + ) + if replacement: + previous_hash = str(member['replacement_evidence_sha256'] or '') + replacement_document = { + 'schema': 1, + 'type': 'docker-depth-repository-replacement-v1', + 'experiment_id': int(row['id']), + 'experiment_repository_id': int(member['id']), + 'replacement_number': int(member['replacement_count']) + 1, + 'from_repository_queue_id': current_queue_id, + 'to_repository_queue_id': replacement['repository_queue_id'], + 'eligibility_page_id': replacement['eligibility_page_id'], + 'best_search_rank': replacement['best_search_rank'], + 'previous_evidence_sha256': previous_hash, + 'reason_code': DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON, + } + replacement_sha256 = _canonical_sha256(replacement_document) + cursor = conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET replacement_repository_queue_id = ?, + replacement_eligibility_page_id = ?, + replacement_count = replacement_count + 1, + replacement_evidence_sha256 = ?, work_state = 'pending', + resolver_attempts = 0, resolver_due_at = NULL, + last_error_code = 'repository_candidate_replaced', + resolved_at = NULL, updated_at = ? + WHERE id = ? AND experiment_id = ? AND work_state = 'held' + AND resolver_attempts = ? AND resolver_owner IS NULL + AND resolver_token IS NULL AND resolver_expires_at IS NULL''', + ( + replacement['repository_queue_id'], + replacement['eligibility_page_id'], replacement_sha256, now, + member['id'], row['id'], entry['prior_resolver_attempts'], + ), + ) + else: + cursor = conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = 'skipped', resolver_due_at = NULL, + last_error_code = ?, selected_image_count = 0, + resolved_at = ?, updated_at = ? + WHERE id = ? AND experiment_id = ? AND work_state = 'held' + AND resolver_attempts = ? AND resolver_owner IS NULL + AND resolver_token IS NULL AND resolver_expires_at IS NULL''', + ( + DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON, now, now, + member['id'], row['id'], entry['prior_resolver_attempts'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError('Docker depth resolver disposition lost its member fence') + resume = conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'resolving', hold_reason_code = NULL, held_at = NULL, + updated_at = ? + WHERE id = ? AND state = 'held' + AND hold_reason_code = 'resolver_attempt_limit' + AND fence_owner IS NULL AND fence_token IS NULL + AND fence_expires_at IS NULL''', + (now, row['id']), + ) + if int(resume.rowcount or 0) != 1: + raise RuntimeError('Docker depth resolver disposition lost its experiment fence') + if not replacement: + refreshed_member = conn.execute( + 'SELECT * FROM docker_depth_experiment_repositories WHERE id = ?', + (member['id'],), + ).fetchone() + db._finalize_docker_depth_query_breadth_locked(row, refreshed_member, now) + refreshed = conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ? FOR UPDATE', + (row['id'],), + ).fetchone() + refreshed, selection_reason = db._freeze_docker_depth_runtime_selection_locked( + refreshed, authority, now, + ) + if selection_reason: + raise RuntimeError( + f'Docker depth resolver disposition selection drift: {selection_reason}' + ) + drift = _persisted_experiment_drift_reason(db, refreshed, authority, now) + if drift: + raise RuntimeError( + f'Docker depth resolver disposition left authority drift: {drift}' + ) + refreshed = db._advance_docker_depth_experiment_state_locked(refreshed, now) + conn.execute( + '''INSERT INTO docker_depth_resolver_dispositions( + experiment_id, experiment_repository_id, + prior_repository_queue_id, replacement_repository_queue_id, + replacement_eligibility_page_id, replacement_best_search_rank, + query_ordinal, repository_rank, disposition_kind, outcome, + manifest_sha256, entry_evidence_sha256, + candidate_snapshot_sha256, prior_target_identity_sha256, + replacement_target_identity_sha256, prior_error_code_sha256, + prior_work_state, next_work_state, prior_resolver_attempts, + next_resolver_attempts, applied_at, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + row['id'], member['id'], entry['repository_queue_id'], + entry['replacement_repository_queue_id'], + entry['replacement_eligibility_page_id'], + entry['replacement_best_search_rank'], entry['query_ordinal'], + entry['repository_rank'], DOCKER_DEPTH_RESOLVER_DISPOSITION_KIND, + entry['outcome'], expected_sha256, entry['entry_evidence_sha256'], + entry['candidate_snapshot_sha256'], entry['target_identity_sha256'], + entry['replacement_target_identity_sha256'], + entry['prior_error_code_sha256'], entry['prior_work_state'], + entry['next_work_state'], entry['prior_resolver_attempts'], + entry['next_resolver_attempts'], now, now, + ), + ) + conn.commit() + return { + 'experiment_id': int(row['id']), 'applied': 1, 'duplicates': 0, + 'outcome': entry['outcome'], 'state': str(refreshed['state']), + } + except Exception: + conn.rollback() + raise diff --git a/app/docker_depth_operator.py b/app/docker_depth_operator.py new file mode 100644 index 0000000..1355423 --- /dev/null +++ b/app/docker_depth_operator.py @@ -0,0 +1,546 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import contextlib +import hmac +import json +import os +import re + +from db_backend import canonical_postgres_url, is_postgres_url +from docker_depth_experiment import ( + DOCKER_DEPTH_QUERY_COUNT, + _docker_depth_authority, + _experiment_identity_matches, + _stored_cohort_plan, + apply_docker_depth_cohort_manifest, + apply_docker_depth_hold_manifest, + apply_docker_depth_reactivation_manifest, + apply_docker_depth_resolver_disposition_manifest, + apply_docker_depth_resolver_refund_manifest, + canonical_docker_depth_plan_hash, + generate_docker_depth_cohort_manifest, + generate_docker_depth_hold_manifest, + generate_docker_depth_reactivation_manifest, + generate_docker_depth_resolver_disposition_manifest, + generate_docker_depth_resolver_refund_manifest, + summarize_docker_depth_fresh_coverage, + validate_docker_depth_cohort_manifest, + validate_docker_depth_config, + validate_docker_depth_hold_manifest, + validate_docker_depth_reactivation_manifest, + validate_docker_depth_resolver_disposition_manifest, + validate_docker_depth_resolver_refund_manifest, + validate_dockerhub_discovery_policies, +) +from migrate_runtime_safety import ( + postgres_migration_guard, + require_local_sources_stopped, +) +from paths import apply_path_config +from postgres_runtime import ( + canonical_database_url, + load_postgres_environment, + verify_cluster_identity, +) +from runtime_security import ( + ClusterAuthorityLock, + MAX_EXTENDED_PRIVATE_JSON_BYTES, + read_private_json, + reject_reparse_components, + require_private_directory, + require_private_file, + preflight_lifecycle_paths, + write_private_json_exclusive, +) +from scanner_db import ScannerDB + + +APPLICATION_NAME = 'truf-docker-depth-operator' +MANIFEST_MAX_BYTES = MAX_EXTENDED_PRIVATE_JSON_BYTES +_SHA256_RE = re.compile(r'^[a-f0-9]{64}$') +_DISABLED_ACTIONS = frozenset({ + 'generate-cohort', 'apply-cohort', 'generate-hold', 'apply-hold', +}) +_ENABLED_ACTIONS = frozenset({ + 'generate-reactivation', 'apply-reactivation', + 'generate-resolver-disposition', 'apply-resolver-disposition', + 'generate-resolver-refund', 'apply-resolver-refund', +}) + + +def load_config(path): + """Load path-expanded YAML only; validation and runtime access are separate.""" + try: + import yaml + except ImportError as exc: + raise RuntimeError('PyYAML is required') from exc + with open(path, 'r', encoding='utf-8') as handle: + return apply_path_config(yaml.safe_load(handle) or {}, path) + + +def provenance_policy_sha256(validated): + source = validated.normalized_config['sources']['dockerhub'] + policies = validate_dockerhub_discovery_policies( + source, validated.experiment.queries, + ) + hashes = {policy['policy_sha256'] for policy in policies} + if len(policies) != DOCKER_DEPTH_QUERY_COUNT or len(hashes) != 1: + raise RuntimeError('Docker depth provenance policy authority is ambiguous') + return next(iter(hashes)) + + +def _action_name(args): + for attribute, name in ( + ('status', 'status'), + ('generate_cohort_manifest', 'generate-cohort'), + ('apply_cohort_manifest', 'apply-cohort'), + ('generate_hold_manifest', 'generate-hold'), + ('apply_hold_manifest', 'apply-hold'), + ('generate_reactivation_manifest', 'generate-reactivation'), + ('apply_reactivation_manifest', 'apply-reactivation'), + ('generate_resolver_disposition_manifest', 'generate-resolver-disposition'), + ('apply_resolver_disposition_manifest', 'apply-resolver-disposition'), + ('generate_resolver_refund_manifest', 'generate-resolver-refund'), + ('apply_resolver_refund_manifest', 'apply-resolver-refund'), + ): + if getattr(args, attribute, None): + return name + raise RuntimeError('Docker depth operator action is unavailable') + + +def _require_action_arguments(parser, args, action): + applying = action.startswith('apply-') + supplied_apply_option = bool( + args.confirm_apply or args.sources_stopped or args.approve_sha256 + ) + if applying: + if not args.confirm_apply or not args.sources_stopped or not args.approve_sha256: + parser.error( + 'apply actions require --approve-sha256, --apply, and --sources-stopped' + ) + if not _SHA256_RE.fullmatch(args.approve_sha256): + parser.error('--approve-sha256 must be one lowercase SHA-256 value') + elif supplied_apply_option: + parser.error('approval options are valid only for apply actions') + + +def _require_action_config_state(experiment, action): + if action in _DISABLED_ACTIONS and experiment.enabled: + raise RuntimeError('Docker depth reviewed preparation requires disabled config') + if action in _ENABLED_ACTIONS and not experiment.enabled: + raise RuntimeError('Docker depth reviewed release requires enabled config') + + +def _manifest_path(path, *, existing): + absolute = reject_reparse_components(os.path.abspath(os.fspath(path))) + require_private_directory(os.path.dirname(absolute), create=False) + if existing: + require_private_file(absolute) + elif os.path.lexists(absolute): + require_private_file(absolute) + return absolute + + +def _publish_manifest(path, manifest): + absolute = _manifest_path(path, existing=False) + if os.path.lexists(absolute): + if read_private_json(absolute, max_bytes=MANIFEST_MAX_BYTES) != manifest: + raise RuntimeError('A different reviewed manifest already exists') + return absolute, False + try: + write_private_json_exclusive( + absolute, manifest, max_bytes=MANIFEST_MAX_BYTES, + ) + except FileExistsError: + if read_private_json(absolute, max_bytes=MANIFEST_MAX_BYTES) != manifest: + raise RuntimeError('Reviewed manifest publication raced a different file') + return absolute, False + require_private_file(absolute) + return absolute, True + + +def _read_approved_manifest(path, validator, experiment, policy_sha256, approved): + absolute = _manifest_path(path, existing=True) + manifest = read_private_json(absolute, max_bytes=MANIFEST_MAX_BYTES) + normalized, manifest_sha256 = validator( + manifest, experiment, policy_sha256, + ) + if not hmac.compare_digest(manifest_sha256, approved): + raise ValueError('Reviewed manifest approval hash conflicts') + return absolute, normalized, manifest_sha256 + + +def _prepare_action(args, action, experiment, policy_sha256): + if action.startswith('generate-'): + attribute = action.replace('-', '_') + '_manifest' + path = _manifest_path(getattr(args, attribute), existing=False) + return {'path': path} + if not action.startswith('apply-'): + return {} + kind = action.removeprefix('apply-') + validator = { + 'cohort': validate_docker_depth_cohort_manifest, + 'hold': validate_docker_depth_hold_manifest, + 'reactivation': validate_docker_depth_reactivation_manifest, + 'resolver-disposition': validate_docker_depth_resolver_disposition_manifest, + 'resolver-refund': validate_docker_depth_resolver_refund_manifest, + }[kind] + path = getattr(args, f'apply_{kind.replace("-", "_")}_manifest') + absolute, manifest, manifest_sha256 = _read_approved_manifest( + path, validator, experiment, policy_sha256, args.approve_sha256, + ) + return { + 'path': absolute, + 'manifest': manifest, + 'manifest_sha256': manifest_sha256, + } + + +def _verify_online_cluster_identity(db, dsn, identity): + canonical = canonical_postgres_url( + dsn, identity['database'], identity['user'], identity['port'], + ) + if canonical != dsn: + raise RuntimeError('Managed PostgreSQL DSN is not canonical') + row = db.conn.execute( + '''SELECT pg_catalog.current_database() AS database, + CURRENT_USER AS user_name, + pg_catalog.current_setting('data_directory') AS data_directory, + pg_catalog.current_setting('port')::integer AS port, + (SELECT system_identifier::text + FROM pg_catalog.pg_control_system()) AS system_identifier''' + ).fetchone() + checks = { + 'database': (str(row['database']), str(identity['database'])), + 'user': (str(row['user_name']), str(identity['user'])), + 'data_directory': ( + os.path.normcase(os.path.realpath(os.path.abspath(row['data_directory']))), + os.path.normcase(os.path.realpath(os.path.abspath(identity['data_directory']))), + ), + 'port': (int(row['port']), int(identity['port'])), + 'system_identifier': ( + str(row['system_identifier']), str(identity['system_identifier']), + ), + } + if any(actual != expected for actual, expected in checks.values()): + raise RuntimeError('Online PostgreSQL identity does not match private authority') + db.conn.commit() + + +@contextlib.contextmanager +def operator_database(config_path, config, *, read_only): + preflight_lifecycle_paths(config_path, config) + load_postgres_environment(config_path, config) + dsn = canonical_database_url() + if not dsn or not is_postgres_url(dsn): + raise RuntimeError('Canonical managed PostgreSQL DSN is unavailable') + with ClusterAuthorityLock(config, endpoint_dsn=dsn): + require_local_sources_stopped(config) + identity = verify_cluster_identity(config) + dsn = canonical_postgres_url( + dsn, identity['database'], identity['user'], identity['port'], + ) + db = ScannerDB(db_url=dsn, initialize=False) + try: + if not db.enabled: + raise RuntimeError('Managed PostgreSQL connection is unavailable') + _verify_online_cluster_identity(db, dsn, identity) + db.set_application_name(APPLICATION_NAME) + with postgres_migration_guard(db): + db.require_runtime_safety_schema() + db.require_final_cutover() + if read_only: + db.conn.execute('SET default_transaction_read_only = on') + db.conn.commit() + yield db + finally: + db.close() + + +def _status(db, experiment, policy_sha256): + authority = _docker_depth_authority(experiment, policy_sha256) + coverage = summarize_docker_depth_fresh_coverage( + db, experiment, policy_sha256, + ) + row = db.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE experiment_key = ?', + (authority['experiment_key'],), + ).fetchone() + state = 'absent' + plan_sha256 = '' + hold_manifest_sha256 = '' + fence_active = 0 + counts = { + 'owned_policy_events': 0, + 'planned_queries': 0, + 'planned_repositories': 0, + 'targets': 0, + 'unreleased_holds': 0, + } + if row: + if not _experiment_identity_matches(row, authority): + raise RuntimeError('Docker depth persisted authority drifted') + state = str(row['state']) + plan_sha256 = str(row['plan_sha256'] or '') + hold_manifest_sha256 = str(row['hold_manifest_sha256'] or '') + fence_active = int(any( + row[name] is not None + for name in ('fence_owner', 'fence_token', 'fence_expires_at') + )) + if plan_sha256: + stored = _stored_cohort_plan(db.conn, row, authority) + if canonical_docker_depth_plan_hash(stored) != plan_sha256: + raise RuntimeError('Docker depth persisted cohort hash drifted') + count_row = db.conn.execute( + '''SELECT + (SELECT COUNT(*) FROM docker_depth_experiment_queries + WHERE experiment_id = ?) AS planned_queries, + (SELECT COUNT(*) FROM docker_depth_experiment_repositories + WHERE experiment_id = ?) AS planned_repositories, + (SELECT COUNT(*) FROM docker_depth_experiment_targets + WHERE experiment_id = ?) AS targets, + (SELECT COUNT(*) FROM target_queue_policy_events + WHERE experiment_id = ?) AS owned_policy_events, + (SELECT COUNT(*) + FROM target_queue_policy_events cold_event + LEFT JOIN target_queue_policy_events reverse_event + ON reverse_event.reverses_event_id = cold_event.id + WHERE cold_event.experiment_id = ? + AND cold_event.action = 'cold' + AND reverse_event.id IS NULL) AS unreleased_holds''', + (row['id'], row['id'], row['id'], row['id'], row['id']), + ).fetchone() + counts = {name: int(count_row[name]) for name in counts} + db.conn.commit() + return { + 'action': 'status', + 'config_enabled': bool(experiment.enabled), + 'config_sha256': authority['config_sha256'], + 'counts': counts, + 'experiment_present': bool(row), + 'fence_active': fence_active, + 'fresh_coverage': coverage, + 'hold_manifest_sha256': hold_manifest_sha256, + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'plan_sha256': plan_sha256, + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'selector_sha256': authority['selector_sha256'], + 'state': state, + } + + +def _execute_action(db, action, prepared, experiment, policy_sha256): + if action == 'status': + return _status(db, experiment, policy_sha256) + if action == 'generate-cohort': + manifest, manifest_sha256 = generate_docker_depth_cohort_manifest( + db, experiment, policy_sha256, + ) + path, created = _publish_manifest(prepared['path'], manifest) + return { + 'action': action, + 'files_created': int(created), + 'manifest_sha256': manifest_sha256, + 'path': os.path.basename(path), + 'plan_sha256': manifest['plan_sha256'], + 'queries': len(manifest['plan']['queries']), + 'repositories': sum( + len(item['repositories']) for item in manifest['plan']['queries'] + ), + } + if action == 'apply-cohort': + result = apply_docker_depth_cohort_manifest( + db, experiment, policy_sha256, prepared['manifest'], + prepared['manifest_sha256'], + ) + return { + 'action': action, + 'manifest_sha256': prepared['manifest_sha256'], + 'path': os.path.basename(prepared['path']), + 'plan_sha256': result['plan_sha256'], + 'plans_persisted': int(bool(result['planned'])), + 'queries': len(result['plan']['queries']), + 'repositories': sum( + len(item['repositories']) for item in result['plan']['queries'] + ), + } + if action == 'generate-hold': + manifest, manifest_sha256 = generate_docker_depth_hold_manifest( + db, experiment, policy_sha256, + ) + path, created = _publish_manifest(prepared['path'], manifest) + return { + 'action': action, + 'conflicts': manifest['conflict_count'], + 'entries': manifest['entry_count'], + 'files_created': int(created), + 'manifest_sha256': manifest_sha256, + 'path': os.path.basename(path), + 'plan_sha256': manifest['plan_sha256'], + } + if action == 'apply-hold': + result = apply_docker_depth_hold_manifest( + db, experiment, policy_sha256, prepared['manifest'], + prepared['manifest_sha256'], + ) + return { + 'action': action, + 'conflicts': int(result['conflicts']), + 'duplicates': int(result['duplicates']), + 'manifest_sha256': prepared['manifest_sha256'], + 'path': os.path.basename(prepared['path']), + 'transitioned': int(result['transitioned']), + } + if action == 'generate-reactivation': + manifest, manifest_sha256 = generate_docker_depth_reactivation_manifest( + db, experiment, policy_sha256, + ) + path, created = _publish_manifest(prepared['path'], manifest) + return { + 'action': action, + 'entries': manifest['entry_count'], + 'files_created': int(created), + 'hold_manifest_sha256': manifest['hold_manifest_sha256'], + 'manifest_sha256': manifest_sha256, + 'path': os.path.basename(path), + } + if action == 'apply-reactivation': + result = apply_docker_depth_reactivation_manifest( + db, experiment, policy_sha256, prepared['manifest'], + prepared['manifest_sha256'], + ) + return { + 'action': action, + 'duplicates': int(result['duplicates']), + 'manifest_sha256': prepared['manifest_sha256'], + 'path': os.path.basename(prepared['path']), + 'transitioned': int(result['transitioned']), + } + if action == 'generate-resolver-disposition': + manifest, manifest_sha256 = ( + generate_docker_depth_resolver_disposition_manifest( + db, experiment, policy_sha256, + ) + ) + path, created = _publish_manifest(prepared['path'], manifest) + return { + 'action': action, + 'files_created': int(created), + 'manifest_sha256': manifest_sha256, + 'outcome': manifest['entries'][0]['outcome'], + 'path': os.path.basename(path), + } + if action == 'apply-resolver-disposition': + result = apply_docker_depth_resolver_disposition_manifest( + db, experiment, policy_sha256, prepared['manifest'], + prepared['manifest_sha256'], + ) + return { + 'action': action, + 'applied': int(result['applied']), + 'duplicates': int(result['duplicates']), + 'manifest_sha256': prepared['manifest_sha256'], + 'outcome': result['outcome'], + 'path': os.path.basename(prepared['path']), + 'state': result['state'], + } + if action == 'generate-resolver-refund': + manifest, manifest_sha256 = generate_docker_depth_resolver_refund_manifest( + db, experiment, policy_sha256, prepared['evidence_log_path'], + ) + path, created = _publish_manifest(prepared['path'], manifest) + return { + 'action': action, + 'entries': manifest['entry_count'], + 'files_created': int(created), + 'held_entries': manifest['held_entry_count'], + 'manifest_sha256': manifest_sha256, + 'path': os.path.basename(path), + 'refund_attempts': manifest['refund_attempts'], + } + if action == 'apply-resolver-refund': + result = apply_docker_depth_resolver_refund_manifest( + db, experiment, policy_sha256, prepared['manifest'], + prepared['manifest_sha256'], prepared['evidence_log_path'], + ) + return { + 'action': action, + 'duplicates': int(result['duplicates']), + 'manifest_sha256': prepared['manifest_sha256'], + 'path': os.path.basename(prepared['path']), + 'refunded': int(result['refunded']), + 'state': result['state'], + } + raise RuntimeError('Docker depth operator action is unsupported') + + +def parse_args(argv=None): + parser = argparse.ArgumentParser( + description='Offline reviewed operator for the bounded Docker depth experiment.', + ) + parser.add_argument( + '--config', default=os.path.join(os.path.dirname(__file__), 'config.yaml'), + ) + actions = parser.add_mutually_exclusive_group(required=True) + actions.add_argument('--status', action='store_true') + actions.add_argument('--generate-cohort-manifest') + actions.add_argument('--apply-cohort-manifest') + actions.add_argument('--generate-hold-manifest') + actions.add_argument('--apply-hold-manifest') + actions.add_argument('--generate-reactivation-manifest') + actions.add_argument('--apply-reactivation-manifest') + actions.add_argument('--generate-resolver-disposition-manifest') + actions.add_argument('--apply-resolver-disposition-manifest') + actions.add_argument('--generate-resolver-refund-manifest') + actions.add_argument('--apply-resolver-refund-manifest') + parser.add_argument('--approve-sha256') + parser.add_argument('--apply', dest='confirm_apply', action='store_true') + parser.add_argument('--sources-stopped', action='store_true') + args = parser.parse_args(argv) + action = _action_name(args) + _require_action_arguments(parser, args, action) + return args, action + + +def main(argv=None): + try: + args, action = parse_args(argv) + config_path = os.path.abspath(args.config) + config = load_config(config_path) + validated = validate_docker_depth_config( + config, managed_postgres=True, final_cutover=True, + ) + if validated.experiment is None: + raise RuntimeError('Docker depth experiment configuration is unavailable') + experiment = validated.experiment + _require_action_config_state(experiment, action) + policy_sha256 = provenance_policy_sha256(validated) + prepared = _prepare_action( + args, action, experiment, policy_sha256, + ) + if action in ('generate-resolver-refund', 'apply-resolver-refund'): + log_path = reject_reparse_components(os.path.abspath(os.path.join( + validated.normalized_config['global']['log_dir'], 'dockerhub.log', + ))) + require_private_file(log_path) + prepared['evidence_log_path'] = log_path + with operator_database( + config_path, validated.normalized_config, + read_only=not action.startswith('apply-'), + ) as db: + report = _execute_action( + db, action, prepared, experiment, policy_sha256, + ) + print(json.dumps(report, ensure_ascii=True, sort_keys=True)) + return 0 + except Exception as exc: + raise SystemExit( + f'Docker depth operator failed closed: {type(exc).__name__}' + ) from None + + +if __name__ == '__main__': + main() diff --git a/app/docker_depth_report.py b/app/docker_depth_report.py new file mode 100644 index 0000000..67f9364 --- /dev/null +++ b/app/docker_depth_report.py @@ -0,0 +1,1164 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import json +import math +import os +from collections import Counter, defaultdict + + +RANK_BUCKETS = ('ranks_1_3', 'ranks_4_10') +INVALID_OR_UNKNOWN = 'invalid_or_unknown' +DOCKER_DEPTH_SOURCES = frozenset(('dockerhub',)) +EXPERIMENT_STATES = frozenset(( + 'collecting', 'planned', 'holding', 'resolving', 'active', 'draining', + 'completed', 'released', 'held', +)) +REPOSITORY_STATES = frozenset(( + 'pending', 'resolving', 'resolved', 'held', 'failed', 'skipped', +)) +TARGET_STATES = frozenset(( + 'pending', 'reserved', 'scanning', 'done', 'failed', 'held', + 'quarantined', 'skipped', +)) +BINDING_STATES = frozenset(( + 'reserved', 'scanning', 'completed', 'failed', 'quarantined', 'released', +)) +RESERVATION_STATES = frozenset(( + 'scanning', 'ready', 'ingesting', 'db_committed', 'acknowledged', + 'refunded', 'quarantined', +)) +SCAN_STATUSES = frozenset(('clean', 'found', 'skipped', 'error', 'degraded')) +CANDIDATE_STATES = frozenset(( + 'pending', 'leased', 'deferred', 'completed', 'quarantined', +)) +KEYCHECK_STATUSES = frozenset(( + 'ALIVE', 'VALID', 'VALID_2FA', 'VALID_RATE_LIMITED', 'VERTEX', 'BEDROCK', + 'FOUNDRY', 'ADMIN', 'CANARY', 'NO_BALANCE', 'NO_QUOTA', + 'LIMITED_OR_NO_BALANCE', 'LIMITED_OR_QUOTA', 'LIMITED', 'RATE_LIMITED', + 'NO_CONTEXT', 'NO_USERNAME', 'NO_TARGET', 'NO_TARGET_MODELS', + 'NO_GENERATION_MODEL', 'RESTRICTED', 'API_DISABLED', 'ACCESS_DENIED', + 'QUARANTINED', 'DISABLED', 'DEAD', 'INVALID', 'EXPIRED', + 'LEAKED_REVOKED', 'INVALID_OR_REVOKED', 'NETWORK', 'NETWORK_ERROR', + 'FOUNDRY_UNRESOLVED', 'FOUNDRY_BAD_ENDPOINT', 'GENERATION_OK', 'OK', + 'OPENAI_BAD_ENDPOINT', 'OPENAI_UNRESOLVED', 'REFRESH_TOKEN', 'UNKNOWN', +)) +KEYCHECK_STATUS_GROUPS = frozenset(( + 'alive', 'dead', 'restricted', 'no_balance', 'no_context', 'limited', + 'network', 'unknown', +)) +ERROR_CATEGORIES = frozenset(( + 'secondary_rate_limit', 'rate_limit', 'auth_invalid', 'auth_forbidden', + 'query_invalid', 'not_found', 'server_error', 'disk_space', 'timeout', + 'extract', 'download', 'network', 'api', 'trufflehog', 'unknown', +)) +REPORT_MAX_MATERIALIZED_ROWS = 250000 + + +_RESERVATION_BOUND_BINDINGS_CTE = '''reservation_bound_bindings AS ( + SELECT t.id AS target_id, t.target_queue_id, t.manifest_id, + b.id AS binding_id, b.reservation_id, b.attempt, + b.state AS binding_state, reservation.state AS reservation_state, + ts.id AS scan_id, + ts.status AS scan_status, ts.duration_sec, + ts.error_count AS scan_error_count + FROM bounded_targets t + LEFT JOIN target_queue queue ON queue.id = t.target_queue_id + LEFT JOIN docker_depth_experiment_scan_bindings b + ON b.experiment_target_id = t.id + AND EXISTS ( + SELECT 1 FROM result_reservations exact_reservation + WHERE exact_reservation.id = b.reservation_id + AND exact_reservation.queue_id = t.target_queue_id + AND exact_reservation.source = queue.source + AND exact_reservation.platform = queue.platform + AND COALESCE(exact_reservation.query, '') = COALESCE(queue.query, '') + AND exact_reservation.target = queue.target + AND exact_reservation.normalized_target = queue.normalized_target + ) + LEFT JOIN result_reservations reservation + ON reservation.id = b.reservation_id + AND reservation.queue_id = t.target_queue_id + LEFT JOIN target_scans ts + ON ts.id = b.target_scan_id + AND ts.queue_id = t.target_queue_id + AND ts.queue_id = reservation.queue_id + AND ts.result_reservation_id = reservation.id + AND ts.result_reservation_id = b.reservation_id + AND ts.scan_event_id = reservation.scan_event_id + AND ts.claim_lease_token = reservation.claim_lease_token + AND ts.source = reservation.source + AND COALESCE(ts.query, '') = COALESCE(reservation.query, '') + AND ts.target = reservation.target + AND ts.normalized_target = reservation.normalized_target + AND ts.scan_type = reservation.platform +)''' + + +def _rows(connection, sql, params, *, max_rows=REPORT_MAX_MATERIALIZED_ROWS): + if isinstance(max_rows, bool) or not isinstance(max_rows, int) or max_rows < 1: + raise ValueError('report row bound must be a positive integer') + cursor = connection.execute(sql, params) + columns = [item[0] for item in cursor.description] + result = [] + while True: + batch = cursor.fetchmany(512) + if not batch: + break + for row in batch: + if len(result) >= max_rows: + raise RuntimeError('Docker depth report row bound is exceeded') + if hasattr(row, 'keys'): + result.append({key: row[key] for key in row.keys()}) + else: + result.append(dict(zip(columns, row))) + return result + + +def _row(connection, sql, params): + rows = _rows(connection, sql, params) + return rows[0] if rows else None + + +def _integer(value, default=0): + try: + return int(value) + except (TypeError, ValueError, OverflowError): + return default + + +def _duration(value): + try: + result = float(value) + except (TypeError, ValueError, OverflowError): + return None + return result if math.isfinite(result) and result >= 0 else None + + +def _category(value, allowed): + return value if isinstance(value, str) and value in allowed else INVALID_OR_UNKNOWN + + +def _experiment_key(value): + if not isinstance(value, str) or not 1 <= len(value) <= 128: + return INVALID_OR_UNKNOWN + allowed = frozenset('abcdefghijklmnopqrstuvwxyz0123456789._-') + if value[0].isalnum() and value[-1].isalnum() and all(char in allowed for char in value): + return value + return INVALID_OR_UNKNOWN + + +def _sha256(value): + if ( + isinstance(value, str) and len(value) == 64 + and all(char in '0123456789abcdef' for char in value)): + return value + return INVALID_OR_UNKNOWN + + +def _counts(values, allowed): + counts = Counter(_category(value, allowed) for value in values) + return {key: counts[key] for key in sorted(counts)} + + +def _rank_bucket(rank): + rank = _integer(rank, -1) + if 1 <= rank <= 3: + return 'ranks_1_3' + if 4 <= rank <= 10: + return 'ranks_4_10' + return None + + +def _finding_identity(row): + fingerprint = str(row.get('finding_fingerprint') or '') + detector_hash = str(row.get('detector_secret_hash') or '') + if fingerprint or detector_hash: + return ('identity', fingerprint, detector_hash) + return ('finding_id', _integer(row.get('finding_id'))) + + +def _finding_metrics(rows, layer_rows): + finding_ids = set() + identities = set() + identity_by_finding = {} + detector_hashes = set() + missing_attribution_findings = set() + exact_attribution_count = 0 + exact_findings = set() + explicit_unattributed_count = 0 + explicit_unattributed_findings = set() + invalid_exact_count = 0 + invalid_attribution_count = 0 + + for row in rows: + finding_id = _integer(row.get('finding_id')) + identity = _finding_identity(row) + finding_ids.add(finding_id) + identities.add(identity) + identity_by_finding[finding_id] = identity + detector_hash = str(row.get('detector_secret_hash') or '') + if detector_hash: + detector_hashes.add(detector_hash) + attribution_count = max(0, _integer(row.get('attribution_count'))) + if attribution_count == 0: + missing_attribution_findings.add(finding_id) + exact_count = max(0, _integer(row.get('exact_attribution_count'))) + exact_attribution_count += exact_count + if exact_count: + exact_findings.add(finding_id) + unattributed_count = max( + 0, _integer(row.get('unattributed_attribution_count')), + ) + explicit_unattributed_count += unattributed_count + if unattributed_count: + explicit_unattributed_findings.add(finding_id) + invalid_exact_count += max( + 0, _integer(row.get('invalid_exact_attribution_count')), + ) + invalid_attribution_count += max( + 0, _integer(row.get('invalid_attribution_count')), + ) + + positions_from_base = Counter() + positions_from_top = Counter() + manifest_layers = set() + for row in layer_rows: + count = max(0, _integer(row.get('attribution_count'))) + manifest_layers.add(_integer(row.get('matched_manifest_layer_id'))) + positions_from_base[_integer(row.get('position_from_base'))] += count + positions_from_top[_integer(row.get('position_from_top'))] += count + + unattributed_findings = finding_ids - exact_findings + exact_identities = {identity_by_finding[item] for item in exact_findings} + unattributed_identities = { + identity_by_finding[item] for item in unattributed_findings + } + return { + 'findings': { + 'occurrence_count': len(finding_ids), + 'deduplicated_count': len(identities), + 'detector_secret_identity_count': len(detector_hashes), + 'without_detector_secret_hash_count': sum( + 1 for identity in identities + if identity[0] == 'finding_id' or not identity[2] + ), + }, + 'layers': { + 'exact_attribution_count': exact_attribution_count, + 'exact_finding_occurrence_count': len(exact_findings), + 'exact_deduplicated_finding_count': len(exact_identities), + 'exact_manifest_layer_count': len(manifest_layers), + 'unattributed_attribution_count': explicit_unattributed_count, + 'unattributed_finding_occurrence_count': len(unattributed_findings), + 'unattributed_deduplicated_finding_count': len(unattributed_identities), + 'missing_attribution_finding_count': len(missing_attribution_findings), + 'invalid_exact_attribution_count': invalid_exact_count, + 'invalid_or_unknown_attribution_count': invalid_attribution_count, + 'positions_from_base': { + str(key): positions_from_base[key] for key in sorted(positions_from_base) + }, + 'positions_from_top': { + str(key): positions_from_top[key] for key in sorted(positions_from_top) + }, + }, + } + + +def _keycheck_metrics(rows): + candidates = {} + for row in rows: + candidates.setdefault(_integer(row.get('candidate_id')), row) + + credentials = { + _integer(row.get('credential_id')) for row in candidates.values() + if row.get('credential_id') is not None + } + frozen_results = set() + frozen_by_credential = {} + current_by_credential = {} + + for row in candidates.values(): + credential_id = _integer(row.get('credential_id')) + frozen_result_id = row.get('frozen_result_id') + if frozen_result_id is not None: + frozen_result_id = _integer(frozen_result_id) + frozen_results.add(frozen_result_id) + previous = frozen_by_credential.get(credential_id) + if previous is None or frozen_result_id > previous[0]: + frozen_by_credential[credential_id] = ( + frozen_result_id, + row.get('frozen_status'), + row.get('frozen_status_group'), + ) + + if row.get('current_state_credential_id') is not None: + current_by_credential[credential_id] = ( + row.get('current_status'), + row.get('current_status_group'), + row.get('matched_current_result_id') is not None, + ) + + candidate_attempts = sum( + max(0, _integer(row.get('candidate_attempts'))) for row in candidates.values() + ) + return { + 'credentials': { + 'deduplicated_count': len(credentials), + }, + 'keychecks': { + 'candidate_count': len(candidates), + 'candidate_state_counts': _counts( + (row.get('candidate_state') for row in candidates.values()), + CANDIDATE_STATES, + ), + 'attempt_count': candidate_attempts, + 'retry_count': sum( + max(0, _integer(row.get('candidate_attempts')) - 1) + for row in candidates.values() + ), + 'pending_candidate_count': sum( + str(row.get('candidate_state') or '').lower() == 'pending' + for row in candidates.values() + ), + 'missing_frozen_result_candidate_count': sum( + row.get('frozen_result_id') is None for row in candidates.values() + ), + 'frozen': { + 'result_count': len(frozen_results), + 'credential_count': len(frozen_by_credential), + 'missing_credential_count': len(credentials - set(frozen_by_credential)), + 'status_counts': _counts( + (value[1] for value in frozen_by_credential.values()), + KEYCHECK_STATUSES, + ), + 'status_group_counts': _counts( + (value[2] for value in frozen_by_credential.values()), + KEYCHECK_STATUS_GROUPS, + ), + }, + 'current': { + 'credential_count': len(current_by_credential), + 'missing_credential_count': len(credentials - set(current_by_credential)), + 'missing_backing_result_count': sum( + not value[2] for value in current_by_credential.values() + ), + 'status_counts': _counts( + (value[0] for value in current_by_credential.values()), + KEYCHECK_STATUSES, + ), + 'status_group_counts': _counts( + (value[1] for value in current_by_credential.values()), + KEYCHECK_STATUS_GROUPS, + ), + }, + }, + } + + +def _aggregate( + target_ids, physical_rows, finding_rows, layer_rows, + candidate_rows, error_rows): + target_ids = set(target_ids) + target_records = {} + bindings = {} + scans = {} + for row in physical_rows: + target_id = _integer(row.get('target_id')) + if target_id not in target_ids: + continue + target_records.setdefault(target_id, row) + if row.get('binding_id') is not None: + bindings.setdefault(_integer(row.get('binding_id')), row) + if row.get('scan_id') is not None: + scans.setdefault(_integer(row.get('scan_id')), row) + + selected_findings = [ + row for row in finding_rows if _integer(row.get('target_id')) in target_ids + ] + selected_candidates = [ + row for row in candidate_rows if _integer(row.get('target_id')) in target_ids + ] + selected_errors = [ + row for row in error_rows if _integer(row.get('target_id')) in target_ids + ] + selected_layers = [ + row for row in layer_rows if _integer(row.get('target_id')) in target_ids + ] + + durations = { + scan_id: value for scan_id, row in scans.items() + if (value := _duration(row.get('duration_sec'))) is not None + } + errors = { + _integer(row.get('error_id')): row for row in selected_errors + if row.get('error_id') is not None + } + error_ids = set(errors) + scans_with_error_rows = { + _integer(row.get('scan_id')) for row in selected_errors + if row.get('error_id') is not None + } + scans_with_reported_errors = { + scan_id for scan_id, row in scans.items() + if _integer(row.get('scan_error_count')) > 0 + } + finding_metrics = _finding_metrics(selected_findings, selected_layers) + keycheck_metrics = _keycheck_metrics(selected_candidates) + + duration_values = list(durations.values()) + duration_total = sum(duration_values) + targets_with_bindings = { + _integer(row.get('target_id')) for row in bindings.values() + } + targets_with_scans = { + _integer(row.get('target_id')) for row in scans.values() + } + return { + 'target_count': len(target_ids), + 'manifest_count': len({ + _integer(row.get('matched_manifest_id')) for row in target_records.values() + if row.get('matched_manifest_id') is not None + }), + 'declared_manifest_layer_count': sum( + max(0, _integer(row.get('manifest_layer_count'))) + for row in target_records.values() + if row.get('matched_manifest_id') is not None + ), + 'target_state_counts': _counts( + (row.get('target_state') for row in target_records.values()), + TARGET_STATES, + ), + 'targets_with_bindings': len(targets_with_bindings), + 'targets_without_bindings': len(target_ids - targets_with_bindings), + 'targets_with_scans': len(targets_with_scans), + 'targets_without_scans': len(target_ids - targets_with_scans), + 'scan_bindings': { + 'attempt_count': len(bindings), + 'retry_count': sum( + _integer(row.get('attempt')) > 1 for row in bindings.values() + ), + 'refunded_attempt_count': sum( + str(row.get('binding_state') or '') == 'released' + for row in bindings.values() + ), + 'maximum_attempt': max( + (_integer(row.get('attempt')) for row in bindings.values()), + default=0, + ), + 'state_counts': _counts( + (row.get('binding_state') for row in bindings.values()), + BINDING_STATES, + ), + 'reservation_state_counts': _counts( + (row.get('reservation_state') for row in bindings.values()), + RESERVATION_STATES, + ), + }, + 'scans': { + 'count': len(scans), + 'status_counts': _counts( + (row.get('scan_status') for row in scans.values()), + SCAN_STATUSES, + ), + 'duration': { + 'count': len(duration_values), + 'missing_count': len(scans) - len(duration_values), + 'total_seconds': duration_total, + 'average_seconds': ( + duration_total / len(duration_values) if duration_values else 0.0 + ), + 'minimum_seconds': min(duration_values, default=0.0), + 'maximum_seconds': max(duration_values, default=0.0), + }, + 'errors': { + 'row_count': len(error_ids), + 'category_counts': _counts( + (row.get('error_category') for row in errors.values()), + ERROR_CATEGORIES, + ), + 'reported_count': sum( + max(0, _integer(row.get('scan_error_count'))) + for row in scans.values() + ), + 'scan_count': len( + scans_with_error_rows | scans_with_reported_errors + ), + }, + }, + **finding_metrics, + **keycheck_metrics, + } + + +def _marginal_minimum_rank_yield(target_ranks, finding_rows, candidate_rows): + minimum_finding_rank = {} + minimum_detector_rank = {} + minimum_credential_rank = {} + + def record_minimum(mapping, identity, rank): + if rank is None: + mapping.setdefault(identity, None) + elif mapping.get(identity) is None or rank < mapping[identity]: + mapping[identity] = rank + + for row in finding_rows: + target_id = _integer(row.get('target_id')) + if target_id not in target_ranks: + continue + rank = target_ranks[target_id] + record_minimum(minimum_finding_rank, _finding_identity(row), rank) + detector_hash = str(row.get('detector_secret_hash') or '') + if detector_hash: + record_minimum(minimum_detector_rank, detector_hash, rank) + + for row in candidate_rows: + target_id = _integer(row.get('target_id')) + if target_id not in target_ranks or row.get('credential_id') is None: + continue + record_minimum( + minimum_credential_rank, + _integer(row.get('credential_id')), + target_ranks[target_id], + ) + + def distribution(mapping): + buckets = Counter(_rank_bucket(rank) or 'unranked' for rank in mapping.values()) + return { + 'deduplicated_total': len(mapping), + 'ranks_1_3': buckets['ranks_1_3'], + 'ranks_4_10': buckets['ranks_4_10'], + 'unranked': buckets['unranked'], + } + + return { + 'findings': distribution(minimum_finding_rank), + 'detector_secret_identities': distribution(minimum_detector_rank), + 'credentials': distribution(minimum_credential_rank), + } + + +def _scope_report( + target_ranks, physical_rows, finding_rows, layer_rows, + candidate_rows, error_rows): + target_ids = set(target_ranks) + bucket_targets = { + bucket: { + target_id for target_id, rank in target_ranks.items() + if _rank_bucket(rank) == bucket + } + for bucket in RANK_BUCKETS + } + return { + 'totals': _aggregate( + target_ids, physical_rows, finding_rows, layer_rows, + candidate_rows, error_rows, + ), + 'rank_buckets': { + bucket: _aggregate( + bucket_targets[bucket], physical_rows, finding_rows, + layer_rows, candidate_rows, error_rows, + ) + for bucket in RANK_BUCKETS + }, + 'unranked_target_count': sum( + _rank_bucket(rank) is None for rank in target_ranks.values() + ), + 'marginal_minimum_rank_yield': _marginal_minimum_rank_yield( + target_ranks, finding_rows, candidate_rows, + ), + } + + +def _overlap_summary(item_queries): + credits = sum(len(query_ids) for query_ids in item_queries.values()) + attributed = sum(bool(query_ids) for query_ids in item_queries.values()) + return { + 'physical_count': len(item_queries), + 'attributed_physical_count': attributed, + 'unattributed_physical_count': len(item_queries) - attributed, + 'attribution_credit_count': credits, + 'overlap_credit_count': credits - attributed, + 'shared_physical_count': sum( + len(query_ids) > 1 for query_ids in item_queries.values() + ), + } + + +def _repository_overlap_summary(repository_queries): + summary = _overlap_summary(repository_queries) + return { + 'physical_count': summary['physical_count'], + 'query_membership_credit_count': summary['attribution_credit_count'], + 'overlap_credit_count': summary['overlap_credit_count'], + 'shared_physical_count': summary['shared_physical_count'], + } + + +def _report_connection(connection): + if not hasattr(connection, 'execute'): + connection = getattr(connection, 'conn', None) + if connection is None or not hasattr(connection, 'execute'): + raise ValueError('a database connection is required') + module = str(type(connection).__module__ or '').lower() + if module.startswith(('psycopg', 'psycopg2')) and not hasattr( + connection, 'is_postgres', + ): + raise ValueError( + 'raw PostgreSQL connections are unsupported; use DatabaseConnection' + ) + return connection + + +def _begin_report_snapshot(connection): + is_postgres = bool(getattr(connection, 'is_postgres', False)) + raw_connection = getattr(connection, '_conn', connection) + if is_postgres: + transaction_status = getattr( + getattr(raw_connection, 'info', None), 'transaction_status', None, + ) + if transaction_status is not None and int(transaction_status) != 0: + raise RuntimeError('Docker depth report requires an idle database connection') + statement = ( + 'BEGIN TRANSACTION ISOLATION LEVEL REPEATABLE READ READ ONLY' + ) + else: + if bool(getattr(raw_connection, 'in_transaction', False)): + raise RuntimeError('Docker depth report requires an idle database connection') + statement = 'BEGIN' + try: + connection.execute(statement) + except Exception: + connection.rollback() + raise + return 'repeatable_read' if is_postgres else 'sqlite_transaction' + + +def _build_docker_depth_report_snapshot( + connection, *, experiment_id=None, experiment_key=None): + """Return secret-safe aggregates for one durable Docker depth experiment.""" + connection = _report_connection(connection) + if (experiment_id is None) == (experiment_key is None): + raise ValueError('provide exactly one of experiment_id or experiment_key') + + if experiment_id is not None: + if isinstance(experiment_id, bool) or not isinstance(experiment_id, int): + raise ValueError('experiment_id must be a positive integer') + if experiment_id < 1: + raise ValueError('experiment_id must be a positive integer') + where = 'id = ?' + selector = experiment_id + else: + experiment_key = str(experiment_key or '').strip() + if not experiment_key: + raise ValueError('experiment_key must be non-empty') + where = 'experiment_key = ?' + selector = experiment_key + + experiment = _row( + connection, + f'''SELECT id, experiment_key, source, state, query_count, + target_count, selection_count + FROM docker_depth_experiments + WHERE {where}''', + (selector,), + ) + if experiment is None: + raise LookupError('Docker depth experiment was not found') + experiment_id = _integer(experiment['id']) + + query_rows = _rows( + connection, + '''WITH bounded_queries AS ( + SELECT id, experiment_id, source, query_ordinal, query, + query_sha256, required_repository_count, + selected_repository_count + FROM docker_depth_experiment_queries + WHERE experiment_id = ? + ) + SELECT q.id AS query_id, q.query_ordinal, q.query_sha256, q.source, + q.required_repository_count, q.selected_repository_count, + r.id AS repository_id, + COALESCE(r.replacement_repository_queue_id, + r.repository_queue_id) AS repository_queue_id, + r.repository_queue_id AS planned_repository_queue_id, + r.repository_rank, + r.work_state AS repository_state, r.resolver_attempts, + s.id AS selection_id, s.image_rank, + t.id AS target_id + FROM bounded_queries q + LEFT JOIN docker_depth_experiment_repositories r + ON r.experiment_id = q.experiment_id + AND r.query_ordinal = q.query_ordinal + AND r.source = q.source AND r.query = q.query + LEFT JOIN docker_depth_experiment_selections s + ON s.experiment_id = q.experiment_id + AND s.query_ordinal = q.query_ordinal + AND s.experiment_repository_id = r.id + LEFT JOIN docker_depth_experiment_targets t + ON t.experiment_id = q.experiment_id + AND t.id = s.experiment_target_id + ORDER BY q.query_ordinal, q.id, r.id, s.id''', + (experiment_id,), + ) + + physical_rows = _rows( + connection, + f'''WITH bounded_targets AS ( + SELECT id, experiment_id, target_queue_id, manifest_id, state + FROM docker_depth_experiment_targets + WHERE experiment_id = ? + ), target_ranks AS ( + SELECT experiment_target_id, MIN(image_rank) AS image_rank + FROM docker_depth_experiment_selections + WHERE experiment_id = ? + GROUP BY experiment_target_id + ), {_RESERVATION_BOUND_BINDINGS_CTE} + SELECT t.id AS target_id, t.state AS target_state, + tr.image_rank, + m.id AS matched_manifest_id, + m.layer_count AS manifest_layer_count, + rb.binding_id, rb.attempt, rb.binding_state, + rb.reservation_state, + rb.scan_id, rb.scan_status, rb.duration_sec, + rb.scan_error_count + FROM bounded_targets t + LEFT JOIN target_ranks tr ON tr.experiment_target_id = t.id + LEFT JOIN docker_image_manifests m + ON m.id = t.manifest_id AND m.target_queue_id = t.target_queue_id + LEFT JOIN reservation_bound_bindings rb ON rb.target_id = t.id + ORDER BY t.id, rb.binding_id''', + (experiment_id, experiment_id), + ) + + finding_rows = _rows( + connection, + f'''WITH bounded_targets AS ( + SELECT id, target_queue_id, manifest_id + FROM docker_depth_experiment_targets + WHERE experiment_id = ? + ), {_RESERVATION_BOUND_BINDINGS_CTE} + SELECT rb.target_id, rb.binding_id, rb.scan_id, + f.id AS finding_id, f.finding_fingerprint, + f.detector_secret_hash, + COUNT(a.id) AS attribution_count, + SUM(CASE WHEN a.attribution_state = 'exact' + AND ml.id IS NOT NULL THEN 1 ELSE 0 END) + AS exact_attribution_count, + SUM(CASE WHEN a.attribution_state = 'exact' + AND ml.id IS NULL THEN 1 ELSE 0 END) + AS invalid_exact_attribution_count, + SUM(CASE WHEN a.attribution_state = 'unattributed' + THEN 1 ELSE 0 END) + AS unattributed_attribution_count, + SUM(CASE WHEN a.id IS NOT NULL + AND a.attribution_state NOT IN ('exact','unattributed') + THEN 1 ELSE 0 END) + AS invalid_attribution_count + FROM reservation_bound_bindings rb + JOIN findings f ON f.target_scan_id = rb.scan_id + LEFT JOIN docker_finding_layer_attributions a + ON a.scan_binding_id = rb.binding_id AND a.finding_id = f.id + LEFT JOIN docker_manifest_layers ml + ON ml.id = a.manifest_layer_id + AND ml.manifest_id = rb.manifest_id + AND ml.position_from_base = a.position_from_base + AND ml.position_from_top = a.position_from_top + AND ml.layer_digest = a.reported_layer_digest + GROUP BY rb.target_id, rb.binding_id, rb.scan_id, f.id, + f.finding_fingerprint, f.detector_secret_hash + ORDER BY rb.target_id, f.id''', + (experiment_id,), + ) + + layer_rows = _rows( + connection, + f'''WITH bounded_targets AS ( + SELECT id, target_queue_id, manifest_id + FROM docker_depth_experiment_targets + WHERE experiment_id = ? + ), {_RESERVATION_BOUND_BINDINGS_CTE} + SELECT rb.target_id, ml.id AS matched_manifest_layer_id, + ml.position_from_base, ml.position_from_top, + COUNT(a.id) AS attribution_count + FROM reservation_bound_bindings rb + JOIN findings f ON f.target_scan_id = rb.scan_id + JOIN docker_finding_layer_attributions a + ON a.scan_binding_id = rb.binding_id AND a.finding_id = f.id + AND a.attribution_state = 'exact' + JOIN docker_manifest_layers ml + ON ml.id = a.manifest_layer_id + AND ml.manifest_id = rb.manifest_id + AND ml.position_from_base = a.position_from_base + AND ml.position_from_top = a.position_from_top + AND ml.layer_digest = a.reported_layer_digest + GROUP BY rb.target_id, ml.id, ml.position_from_base, + ml.position_from_top + ORDER BY rb.target_id, ml.id''', + (experiment_id,), + ) + + candidate_rows = _rows( + connection, + f'''WITH bounded_targets AS ( + SELECT id, target_queue_id, manifest_id + FROM docker_depth_experiment_targets + WHERE experiment_id = ? + ), {_RESERVATION_BOUND_BINDINGS_CTE}, bounded_findings AS ( + SELECT rb.target_id, rb.scan_id, f.id AS finding_id + FROM reservation_bound_bindings rb + JOIN findings f ON f.target_scan_id = rb.scan_id + ) + SELECT bf.target_id, bf.scan_id, bf.finding_id, + c.id AS candidate_id, kc.id AS credential_id, + c.state AS candidate_state, c.attempts AS candidate_attempts, + frozen.id AS frozen_result_id, + frozen.status AS frozen_status, + frozen.status_group AS frozen_status_group, + current_state.credential_id AS current_state_credential_id, + current_result.status AS current_status, + current_result.status_group AS current_status_group, + current_result.id AS matched_current_result_id + FROM bounded_findings bf + JOIN keycheck_candidates c + ON c.finding_id = bf.finding_id + AND c.target_scan_id = bf.scan_id + JOIN keycheck_credentials kc ON kc.id = c.credential_id + LEFT JOIN keycheck_results frozen + ON frozen.id = c.keycheck_result_id + AND frozen.candidate_id = c.id + AND frozen.credential_id = kc.id + LEFT JOIN keycheck_current_state current_state + ON current_state.credential_id = kc.id + LEFT JOIN keycheck_results current_result + ON current_result.id = current_state.last_result_id + AND current_result.credential_id = current_state.credential_id + AND current_result.status = current_state.status + AND current_result.status_group = current_state.status_group + ORDER BY bf.target_id, c.id''', + (experiment_id,), + ) + + error_rows = _rows( + connection, + f'''WITH bounded_targets AS ( + SELECT id, target_queue_id, manifest_id + FROM docker_depth_experiment_targets + WHERE experiment_id = ? + ), {_RESERVATION_BOUND_BINDINGS_CTE} + SELECT rb.target_id, rb.binding_id, rb.scan_id, + e.id AS error_id, e.category AS error_category + FROM reservation_bound_bindings rb + JOIN errors e ON e.target_scan_id = rb.scan_id + ORDER BY rb.target_id, rb.scan_id, e.id''', + (experiment_id,), + ) + + query_contexts = {} + for row in query_rows: + query_id = _integer(row.get('query_id')) + context = query_contexts.setdefault(query_id, { + 'query_id': query_id, + 'query_ordinal': _integer(row.get('query_ordinal')), + 'query_sha256': _sha256(row.get('query_sha256')), + 'source': str(row.get('source') or ''), + 'required_repository_count': _integer( + row.get('required_repository_count') + ), + 'cohort_repository_count': _integer( + row.get('selected_repository_count') + ), + 'repository_attempts': {}, + 'repository_states': {}, + 'repository_queue_ids': {}, + 'selected_repositories': set(), + 'selection_ids': set(), + 'target_ranks': {}, + }) + repository_id = row.get('repository_id') + if repository_id is not None: + repository_id = _integer(repository_id) + context['repository_attempts'].setdefault( + repository_id, max(0, _integer(row.get('resolver_attempts'))), + ) + context['repository_states'].setdefault( + repository_id, row.get('repository_state'), + ) + if row.get('repository_queue_id') is not None: + context['repository_queue_ids'].setdefault( + repository_id, _integer(row.get('repository_queue_id')), + ) + if row.get('selection_id') is not None and row.get('target_id') is not None: + context['selection_ids'].add(_integer(row.get('selection_id'))) + if row.get('repository_queue_id') is not None: + context['selected_repositories'].add( + _integer(row.get('repository_queue_id')) + ) + target_id = _integer(row.get('target_id')) + rank = _integer(row.get('image_rank')) + previous = context['target_ranks'].get(target_id) + if previous is None or rank < previous: + context['target_ranks'][target_id] = rank + + physical_target_ranks = {} + for row in physical_rows: + target_id = _integer(row.get('target_id')) + rank = row.get('image_rank') + physical_target_ranks.setdefault( + target_id, _integer(rank) if rank is not None else None, + ) + + physical = _scope_report( + physical_target_ranks, physical_rows, finding_rows, layer_rows, + candidate_rows, error_rows, + ) + repository_queries = defaultdict(set) + selected_repository_queries = defaultdict(set) + all_selections = set() + for query_id, context in query_contexts.items(): + for repository_queue_id in set(context['repository_queue_ids'].values()): + repository_queries[repository_queue_id].add(query_id) + for repository_queue_id in context['selected_repositories']: + selected_repository_queries[repository_queue_id].add(query_id) + all_selections.update(context['selection_ids']) + repository_coverage = _repository_overlap_summary(repository_queries) + selected_repository_coverage = _repository_overlap_summary( + selected_repository_queries + ) + skipped_repository_queue_ids = { + context['repository_queue_ids'][repository_id] + for context in query_contexts.values() + for repository_id, state in context['repository_states'].items() + if state == 'skipped' and repository_id in context['repository_queue_ids'] + } + skipped_repository_memberships = sum( + state == 'skipped' + for context in query_contexts.values() + for state in context['repository_states'].values() + ) + physical['coverage'] = { + 'query_count': len(query_contexts), + 'queries_with_repositories': sum( + bool(item['repository_attempts']) for item in query_contexts.values() + ), + 'queries_with_selections': sum( + bool(item['selection_ids']) for item in query_contexts.values() + ), + 'repository_count': repository_coverage['physical_count'], + 'repository_query_membership_credit_count': ( + repository_coverage['query_membership_credit_count'] + ), + 'repository_overlap_credit_count': repository_coverage['overlap_credit_count'], + 'shared_repository_count': repository_coverage['shared_physical_count'], + 'selected_repository_count': selected_repository_coverage['physical_count'], + 'selected_repository_query_membership_credit_count': ( + selected_repository_coverage['query_membership_credit_count'] + ), + 'selected_repository_overlap_credit_count': ( + selected_repository_coverage['overlap_credit_count'] + ), + 'shared_selected_repository_count': ( + selected_repository_coverage['shared_physical_count'] + ), + 'image_unavailable_repository_count': len(skipped_repository_queue_ids), + 'image_unavailable_repository_membership_count': ( + skipped_repository_memberships + ), + 'queries_with_image_unavailable_repositories': sum( + any(state == 'skipped' for state in item['repository_states'].values()) + for item in query_contexts.values() + ), + 'selection_count': len(all_selections), + 'target_attribution_credit_count': sum( + len(item['target_ranks']) for item in query_contexts.values() + ), + } + + per_query = {} + for context in sorted( + query_contexts.values(), + key=lambda item: (item['query_ordinal'], item['query_id'])): + scope = _scope_report( + context['target_ranks'], physical_rows, finding_rows, layer_rows, + candidate_rows, error_rows, + ) + resolver_attempts = context['repository_attempts'].values() + scope['query'] = { + 'id': context['query_id'], + 'ordinal': context['query_ordinal'], + 'sha256': context['query_sha256'], + 'source': _category(context['source'], DOCKER_DEPTH_SOURCES), + } + scope['coverage'] = { + 'required_repository_count': context['required_repository_count'], + 'cohort_repository_count': context['cohort_repository_count'], + 'unavailable_repository_count': max( + 0, + context['required_repository_count'] + - context['cohort_repository_count'], + ), + 'repository_count': len(context['repository_attempts']), + 'missing_cohort_repository_count': max( + 0, + context['cohort_repository_count'] + - len(context['repository_attempts']), + ), + 'selected_repository_count': len(context['selected_repositories']), + 'image_unavailable_repository_count': sum( + state == 'skipped' + for state in context['repository_states'].values() + ), + 'selection_count': len(context['selection_ids']), + 'target_count': len(context['target_ranks']), + 'repository_state_counts': _counts( + context['repository_states'].values(), REPOSITORY_STATES, + ), + 'resolver_attempt_count': sum(resolver_attempts), + 'resolver_retry_count': sum( + max(0, attempts - 1) + for attempts in context['repository_attempts'].values() + ), + } + per_query[str(context['query_id'])] = scope + + target_queries = {target_id: set() for target_id in physical_target_ranks} + for query_id, context in query_contexts.items(): + for target_id in context['target_ranks']: + target_queries.setdefault(target_id, set()).add(query_id) + + scan_queries = {} + for row in physical_rows: + if row.get('scan_id') is not None: + scan_queries.setdefault( + _integer(row.get('scan_id')), + set(target_queries.get(_integer(row.get('target_id')), set())), + ) + finding_queries = defaultdict(set) + for row in finding_rows: + finding_queries[_finding_identity(row)].update( + target_queries.get(_integer(row.get('target_id')), set()) + ) + credential_queries = defaultdict(set) + for row in candidate_rows: + if row.get('credential_id') is not None: + credential_queries[_integer(row.get('credential_id'))].update( + target_queries.get(_integer(row.get('target_id')), set()) + ) + + return { + 'experiment': { + 'id': experiment_id, + 'key': _experiment_key(experiment.get('experiment_key')), + 'source': _category(experiment.get('source'), DOCKER_DEPTH_SOURCES), + 'state': _category(experiment.get('state'), EXPERIMENT_STATES), + 'persisted_counts': { + 'queries': _integer(experiment.get('query_count')), + 'targets': _integer(experiment.get('target_count')), + 'selections': _integer(experiment.get('selection_count')), + }, + }, + 'physical': physical, + 'per_query': per_query, + 'counting_semantics': { + 'physical_totals_are_deduplicated': True, + 'per_query_totals_are_attribution_credits': True, + 'per_query_totals_are_summable_to_physical': False, + }, + 'overlap': { + 'repositories': repository_coverage, + 'selected_repositories': selected_repository_coverage, + 'targets': _overlap_summary(target_queries), + 'scans': _overlap_summary(scan_queries), + 'findings': _overlap_summary(finding_queries), + 'credentials': _overlap_summary(credential_queries), + }, + } + + +def build_docker_depth_report( + connection, *, experiment_id=None, experiment_key=None, + allow_live=True): + """Build one aggregate report from a clean, read-only database snapshot.""" + connection = _report_connection(connection) + consistency = _begin_report_snapshot(connection) + try: + report = _build_docker_depth_report_snapshot( + connection, experiment_id=experiment_id, + experiment_key=experiment_key, + ) + terminal = report['experiment']['state'] in ('completed', 'released') + if not allow_live and not terminal: + raise RuntimeError( + 'Docker depth report is live; pass allow_live=True explicitly' + ) + report['snapshot'] = { + 'read_only': True, + 'consistency': consistency, + 'experiment_terminal': terminal, + } + return report + finally: + connection.rollback() + + +def main(argv=None): + parser = argparse.ArgumentParser( + description='Build a secret-safe read-only Docker depth experiment report.', + ) + parser.add_argument('--config', required=True) + selector = parser.add_mutually_exclusive_group(required=True) + selector.add_argument('--experiment-id', type=int) + selector.add_argument('--experiment-key') + parser.add_argument( + '--allow-live', action='store_true', + help='Allow a coherent snapshot before the experiment is completed.', + ) + args = parser.parse_args(argv) + connection = None + try: + import yaml + + from db_backend import ( + connect_postgres, database_url_from_env, is_postgres_url, + ) + from paths import apply_path_config + + with open(os.path.abspath(args.config), 'r', encoding='utf-8') as handle: + config = apply_path_config(yaml.safe_load(handle) or {}, args.config) + global_config = config.get('global') or {} + database_url = ( + global_config.get('dashboard_db_url') + or global_config.get('database_url') + or database_url_from_env() + ) + if not is_postgres_url(database_url): + raise RuntimeError('Docker depth report requires PostgreSQL') + connection = connect_postgres( + database_url, connect_timeout_sec=5, statement_timeout_ms=120000, + lock_timeout_ms=5000, idle_in_transaction_timeout_ms=120000, + tcp_user_timeout_ms=10000, + ) + connection.execute("SET application_name = 'truf-docker-depth-report'") + connection.execute('SET default_transaction_read_only = on') + connection.commit() + report = build_docker_depth_report( + connection, + experiment_id=args.experiment_id, + experiment_key=args.experiment_key, + allow_live=args.allow_live, + ) + print(json.dumps(report, ensure_ascii=True, sort_keys=True)) + return 0 + except Exception as exc: + raise SystemExit( + f'Docker depth report failed closed: {type(exc).__name__}' + ) from None + finally: + if connection is not None: + connection.close() + + +if __name__ == '__main__': + main() diff --git a/app/docker_shadow.py b/app/docker_shadow.py new file mode 100644 index 0000000..d374138 --- /dev/null +++ b/app/docker_shadow.py @@ -0,0 +1,849 @@ +import sys + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('Docker shadow runner could not disable bytecode writes') + +import argparse +import copy +import hashlib +import math +import os +import re +import secrets +import socket +import time + +import scanner +import console_runner +from keycheck_candidates import extract_candidates +from lifecycle_authority import require_active_supervisor_child +from scanner_db import ( + DOCKER_ADAPTIVE_GATE_MAX_CONTROLS, + DOCKER_ADAPTIVE_GATE_MIN_CONTROLS, + DOCKER_ADAPTIVE_LAYER_CLASS_ORDER, + DOCKER_ADAPTIVE_SELECTOR_VERSION, + DOCKER_ADAPTIVE_SHADOW_SELECTION_METRIC_KEYS, + ScannerDB, + canonical_docker_layer_plan_bytes, + docker_content_media_class, + docker_layer_execution_policy_sha256, + docker_layer_selection_policy_sha256, + finding_identity, + select_docker_adaptive_payload, + validate_docker_adaptive_checkpoint, + validate_docker_adaptive_shadow_selection_metrics, + validate_docker_layer_execution, + validate_docker_layer_limits, + validate_docker_layer_plan, + validate_docker_layer_resolution, +) + + +_SHA256_RE = re.compile(r'[a-f0-9]{64}') +_ROUTED_SERVICE_RE = re.compile(r'[a-z0-9][a-z0-9_.-]{0,63}') +SHADOW_FAILURE_METRIC_KEYS = ( + 'diagnostic_full_incomplete', + 'diagnostic_adaptive_incomplete', + 'diagnostic_blob_chunk_processing', + 'diagnostic_blob_detector_timeout', + 'diagnostic_blob_network', + 'diagnostic_blob_timeout', + 'diagnostic_blob_mixed', + 'diagnostic_blob_other', +) +_SHADOW_BLOB_FAILURE_METRICS = { + 'chunk_processing': 'diagnostic_blob_chunk_processing', + 'detector_timeout': 'diagnostic_blob_detector_timeout', + 'network': 'diagnostic_blob_network', + 'transfer_timeout': 'diagnostic_blob_timeout', + 'timeout': 'diagnostic_blob_timeout', + 'mixed': 'diagnostic_blob_mixed', +} + + +class DockerShadowPrivacyError(RuntimeError): + pass + + +def empty_selection_metrics(): + return {name: 0 for name in DOCKER_ADAPTIVE_SHADOW_SELECTION_METRIC_KEYS} + + +def empty_failure_metrics(): + return {name: 0 for name in SHADOW_FAILURE_METRIC_KEYS} + + +def _blob_failure_metric(error_code): + return _SHADOW_BLOB_FAILURE_METRICS.get( + str(error_code or ''), 'diagnostic_blob_other', + ) + + +def _canonical_private_plan(plan): + plan = validate_docker_layer_plan(plan) + payload = canonical_docker_layer_plan_bytes(plan) + return plan, hashlib.sha256(payload).hexdigest() + + +def _validate_duplicate_descriptors(descriptors): + identities = {} + for descriptor in descriptors: + identity = ( + descriptor['kind'], descriptor['size'], + docker_content_media_class(descriptor['kind'], descriptor['media_type']), + ) + previous = identities.setdefault(descriptor['digest'], identity) + if previous != identity: + raise ValueError('Docker shadow duplicate digest metadata conflicts') + + +def build_private_adaptive_plan( + resolved, payload_classes, limits, checkpoint, scan_policy_sha256, +): + resolved = validate_docker_layer_resolution(resolved) + limits = validate_docker_layer_limits(limits) + checkpoint = validate_docker_adaptive_checkpoint(checkpoint) + scan_policy_sha256 = str(scan_policy_sha256 or '').lower() + if not _SHA256_RE.fullmatch(scan_policy_sha256): + raise ValueError('Docker shadow scan policy must be a lowercase SHA-256') + + descriptors = [resolved['config'], *resolved['layers']] + _validate_duplicate_descriptors(descriptors) + decisions = select_docker_adaptive_payload( + descriptors, payload_classes, limits, covered_digests=(), + ) + metrics = empty_selection_metrics() + planned = [] + omitted = 0 + for descriptor, selected, reason, payload_class in decisions: + if reason == 'config_selected': + metrics['selected_config'] += 1 + elif reason.startswith('selected_'): + metrics[reason] += 1 + elif reason == 'already_covered': + metrics['reuse_already_covered'] += 1 + elif reason == 'duplicate_digest': + metrics['reuse_duplicate_digest'] += 1 + elif not selected: + metric_name = f'omitted_{reason}' + if metric_name not in metrics: + raise ValueError('Docker shadow selection reason is not aggregate-safe') + metrics[metric_name] += 1 + omitted += 1 + planned.append({ + **descriptor, + 'payload_class': payload_class, + 'selected': bool(selected), + 'selection_reason': reason, + 'coverage_state': 'selected' if selected else 'skipped', + 'lease_token': None, + 'attempt': 0, + 'max_attempts': limits['blob_max_attempts'], + }) + + plan, plan_sha256 = _canonical_private_plan({ + 'version': 2, + 'image': resolved['image'], + 'repository': resolved['repository'], + 'manifest_digest': resolved['manifest_digest'], + 'platform_os': resolved['platform_os'], + 'platform_arch': resolved['platform_arch'], + 'manifest_media_type': resolved['manifest_media_type'], + 'limits': limits, + 'selector_version': DOCKER_ADAPTIVE_SELECTOR_VERSION, + 'selection_policy_sha256': docker_layer_selection_policy_sha256(limits), + 'scan_policy_sha256': scan_policy_sha256, + 'execution_policy_sha256': docker_layer_execution_policy_sha256( + scan_policy_sha256, limits, + ), + 'checkpoint': checkpoint, + 'descriptors': planned, + }) + return { + 'plan': plan, + 'plan_sha256': plan_sha256, + 'selection_metrics': validate_docker_adaptive_shadow_selection_metrics(metrics), + 'omitted_descriptor_count': omitted, + } + + +def _checkpoint_order(descriptor): + if descriptor['kind'] == 'config': + return (0, 0, -descriptor['position'], descriptor['size'], descriptor['digest']) + try: + class_rank = DOCKER_ADAPTIVE_LAYER_CLASS_ORDER.index(descriptor['payload_class']) + except ValueError as exc: + raise ValueError('Docker shadow descriptor class is not schedulable') from exc + return (1, class_rank, -descriptor['position'], descriptor['size'], descriptor['digest']) + + +def lease_private_adaptive_checkpoint(plan, token_factory=None): + plan = validate_docker_layer_plan(plan) + if plan['version'] != 2: + raise ValueError('Docker shadow checkpoints require a version-two plan') + if any( + descriptor['coverage_state'] in ('leased', 'shared_pending') + for descriptor in plan['descriptors'] + ): + raise ValueError('Docker shadow plan already contains active leases') + + pending = {} + for descriptor in plan['descriptors']: + if descriptor['coverage_state'] == 'selected': + pending.setdefault(descriptor['digest'], []).append(descriptor) + if not pending: + return None + + groups = [] + for digest, descriptors in pending.items(): + attempts = {item['attempt'] for item in descriptors} + maximums = {item['max_attempts'] for item in descriptors} + if len(attempts) != 1 or len(maximums) != 1: + raise ValueError('Docker shadow duplicate attempts conflict') + if next(iter(attempts)) >= next(iter(maximums)): + raise ValueError('Docker shadow plan exceeded its private attempt budget') + representative = min(descriptors, key=_checkpoint_order) + groups.append((representative, digest)) + groups.sort(key=lambda item: _checkpoint_order(item[0])) + + leased_digests = set() + leased_bytes = 0 + max_blobs = plan['checkpoint']['max_blobs'] + max_bytes = plan['checkpoint']['max_bytes'] + for descriptor, digest in groups: + if len(leased_digests) >= max_blobs: + continue + if leased_digests and leased_bytes + descriptor['size'] > max_bytes: + continue + leased_digests.add(digest) + leased_bytes += descriptor['size'] + if not leased_digests: + raise ValueError('Docker shadow checkpoint made no bounded progress') + + token_factory = token_factory or (lambda: secrets.token_urlsafe(32)) + tokens = {digest: str(token_factory()) for digest in leased_digests} + leased_plan = copy.deepcopy(plan) + for descriptor in leased_plan['descriptors']: + if descriptor['digest'] not in leased_digests: + continue + descriptor['coverage_state'] = 'leased' + descriptor['lease_token'] = tokens[descriptor['digest']] + descriptor['attempt'] += 1 + leased_plan, plan_sha256 = _canonical_private_plan(leased_plan) + return {'plan': leased_plan, 'plan_sha256': plan_sha256} + + +def apply_private_adaptive_execution(plan, execution): + plan, plan_sha256 = _canonical_private_plan(plan) + execution = validate_docker_layer_execution(execution, plan, plan_sha256) + records = {record['digest']: record for record in execution['blobs']} + next_plan = copy.deepcopy(plan) + for descriptor in next_plan['descriptors']: + if descriptor['coverage_state'] != 'leased': + continue + status = records[descriptor['digest']]['status'] + if status == 'covered': + next_state = 'covered' + elif status == 'retryable_failed' and descriptor['attempt'] < descriptor['max_attempts']: + next_state = 'selected' + else: + next_state = 'terminal_failed' + descriptor['coverage_state'] = next_state + descriptor['lease_token'] = None + return _canonical_private_plan(next_plan)[0] + + +def private_result_identities(result, normalized_target): + if not isinstance(result, dict): + raise ValueError('Docker shadow private result must be an object') + routed = set() + detectors = set() + try: + scanner.strip_nearby_context_for_persistence(result) + findings = result.get('findings') or [] + if not isinstance(findings, list): + raise ValueError('Docker shadow private findings must be a list') + for finding in findings: + if not isinstance(finding, dict): + raise ValueError('Docker shadow private finding must be an object') + detector_hash = str( + finding_identity('dockerhub', normalized_target, finding)[2] or '' + ).lower() + if not _SHA256_RE.fullmatch(detector_hash): + raise DockerShadowPrivacyError('Docker shadow detector identity is invalid') + detectors.add(detector_hash) + for candidate in extract_candidates(finding): + service = str(candidate.service or '') + provider_key_hash = str(candidate.provider_key_hash or '').lower() + if ( + not _ROUTED_SERVICE_RE.fullmatch(service) + or not _SHA256_RE.fullmatch(provider_key_hash) + ): + raise DockerShadowPrivacyError('Docker shadow routed identity is invalid') + routed.add((service, provider_key_hash)) + finally: + findings = result.get('findings') if isinstance(result, dict) else None + if isinstance(findings, list): + for finding in findings: + if isinstance(finding, dict): + finding.clear() + findings.clear() + result.clear() + return frozenset(routed), frozenset(detectors) + + +def run_timed_private_side( + side, operation, durable_checkpoint, *, timeout_sec, + monotonic_ns=time.monotonic_ns, +): + if side not in ('full', 'adaptive'): + raise ValueError('Docker shadow side is invalid') + with scanner.scan_slot_scope(['docker-shadow', side], timeout_sec=timeout_sec): + started_ns = monotonic_ns() + metrics = operation() + if not isinstance(metrics, dict) or any( + not isinstance(name, str) + or isinstance(value, bool) + or not isinstance(value, int) + or value < 0 + for name, value in metrics.items() + ): + raise ValueError('Docker shadow private sink accepts only non-negative aggregates') + durable_checkpoint() + elapsed_ns = max(1, monotonic_ns() - started_ns) + return metrics, max(1, math.ceil(elapsed_ns / 1_000_000)) + + +def _clear_private_result(result): + if not isinstance(result, dict): + return + findings = result.get('findings') + if isinstance(findings, list): + for finding in findings: + if isinstance(finding, dict): + finding.clear() + findings.clear() + result.clear() + + +def _full_scan_incomplete_reason(result): + if result.get('source_failure'): + return 'full_scan_source_failure' + if result.get('skipped'): + return 'full_scan_skipped' + return 'full_scan_error' + + +def _adaptive_scan_incomplete_reason(exc): + if isinstance(exc, scanner.DockerLayerInfrastructureError): + return 'adaptive_scan_infrastructure' + if isinstance(exc, scanner.DockerContentTransferError): + return 'adaptive_scan_transfer' + if isinstance(exc, TimeoutError): + return 'adaptive_scan_timeout' + if isinstance(exc, scanner.DockerRemoteAccessError): + return 'adaptive_scan_remote_access' + if isinstance(exc, ValueError): + return 'adaptive_scan_contract' + return 'adaptive_scan_error' + + +def shadow_failure_reason_code(exc): + if isinstance(exc, DockerShadowPrivacyError): + return 'privacy_violation' + if isinstance(exc, (KeyboardInterrupt, SystemExit)): + return 'operator_interrupt' + if exc.__class__.__name__ == 'ScanSlotFatalError': + return 'scan_slot_fatal' + if isinstance(exc, TimeoutError): + return 'timeout' + if isinstance(exc, ValueError): + return 'invalid_contract' + if isinstance(exc, RuntimeError): + message = str(exc).lower() + if 'checkpoint' in message: + return 'checkpoint_failure' + if 'fence' in message or 'lease' in message: + return 'report_fence_failure' + if 'cohort' in message or 'control' in message: + return 'control_failure' + return 'runtime_failure' + return 'unexpected_failure' + + +def execute_private_adaptive_scan( + normalized_target, *, limits, checkpoint, scan_policy_sha256, timeout_sec, + platform_os='linux', platform_arch='amd64', min_free_bytes=0, + scan_kwargs=None, +): + timeout_sec = max(1, int(timeout_sec)) + deadline = time.monotonic() + timeout_sec + scan_kwargs = dict(scan_kwargs or {}) + resolved, bearer_auth = scanner.resolve_docker_content_manifest( + normalized_target, + platform_os=str(platform_os or 'linux'), + platform_arch=str(platform_arch or 'amd64'), + deadline=deadline, + ) + payload_classes, bearer_auth = scanner.fetch_docker_config_payload_classes( + resolved, bearer_auth, deadline=deadline, + min_free_bytes=max(0, int(min_free_bytes or 0)), + ) + built = build_private_adaptive_plan( + resolved, payload_classes, limits, checkpoint, scan_policy_sha256, + ) + plan = built['plan'] + selection_metrics = dict(built['selection_metrics']) + failure_metrics = empty_failure_metrics() + routed = set() + detectors = set() + selected_unique = { + item['digest']: item['max_attempts'] + for item in plan['descriptors'] if item['selected'] + } + max_checkpoints = max(1, sum(selected_unique.values())) + checkpoint_count = 0 + while True: + leased = lease_private_adaptive_checkpoint(plan) + if leased is None: + break + checkpoint_count += 1 + if checkpoint_count > max_checkpoints or time.monotonic() >= deadline: + raise RuntimeError('Docker shadow adaptive checkpoint budget was exhausted') + work = { + 'plan': leased['plan'], + 'plan_sha256': leased['plan_sha256'], + 'bearer_auth': bearer_auth, + 'min_free_bytes': max(0, int(min_free_bytes or 0)), + 'deadline': deadline, + } + result = scanner.scan_docker_layer_plan( + normalized_target, work, + timeout_sec=max(1, math.ceil(deadline - time.monotonic())), + detectors=scan_kwargs.get('detectors'), + exclude_detectors=scan_kwargs.get('exclude_detectors'), + no_verification=bool(scan_kwargs.get('no_verification', False)), + trufflehog_config=scan_kwargs.get('trufflehog_config'), + log_target=False, + ) + try: + execution = result.get('docker_layer_execution') + next_plan = apply_private_adaptive_execution( + leased['plan'], result.get('docker_layer_execution'), + ) + records = { + record['digest']: record for record in execution['blobs'] + } + newly_terminal = { + item['digest'] for item in next_plan['descriptors'] + if item['coverage_state'] == 'terminal_failed' + and item['digest'] in records + } + for digest in newly_terminal: + failure_metrics[ + _blob_failure_metric(records[digest]['error_code']) + ] += 1 + plan = next_plan + checkpoint_routed, checkpoint_detectors = private_result_identities( + result, normalized_target, + ) + except Exception: + _clear_private_result(result) + raise + routed.update(checkpoint_routed) + detectors.update(checkpoint_detectors) + del checkpoint_routed, checkpoint_detectors + + if any( + item['coverage_state'] in ('selected', 'leased', 'shared_pending') + for item in plan['descriptors'] + ): + raise RuntimeError('Docker shadow adaptive plan did not reach a terminal state') + selection_metrics['adaptive_checkpoints'] += checkpoint_count + terminal_digests = { + item['digest'] for item in plan['descriptors'] + if item['coverage_state'] == 'terminal_failed' + } + if sum(failure_metrics.values()) != len(terminal_digests): + raise RuntimeError('Docker shadow terminal failure accounting is inconsistent') + return { + 'routed': frozenset(routed), + 'detectors': frozenset(detectors), + 'selection_metrics': validate_docker_adaptive_shadow_selection_metrics( + selection_metrics, + ), + 'omitted_descriptor_count': int(built['omitted_descriptor_count']), + 'failure_count': len(terminal_digests), + 'failure_metrics': failure_metrics, + } + + +def private_full_side_metrics(db, control, scan_kwargs): + reference_routed, reference_detectors = db.docker_adaptive_shadow_control_identities( + control['target_scan_id'], + ) + reference_routed = set(reference_routed) + reference_detectors = set(reference_detectors) + rerun_routed = set() + rerun_detectors = set() + result = None + try: + max_attempts = min(10, max(1, int(scan_kwargs.get('max_attempts', 3) or 3))) + incomplete_reason = 'full_scan_error' + attempts = 0 + for attempts in range(1, max_attempts + 1): + result = scanner.scan_docker_image( + control['normalized_target'], + timeout_sec=int(scan_kwargs['timeout_sec']), + detectors=scan_kwargs.get('detectors'), + exclude_detectors=scan_kwargs.get('exclude_detectors'), + no_verification=bool(scan_kwargs.get('no_verification', False)), + trufflehog_config=scan_kwargs.get('trufflehog_config'), + config_dir=scanner.docker_token_manager.get_next_config(), + trufflehog_concurrency=int(scan_kwargs.get('trufflehog_concurrency', 0) or 0), + docker_recovery_limits=scan_kwargs.get('docker_recovery_limits'), + docker_recovery_min_free_bytes=int(scan_kwargs.get('docker_recovery_min_free_bytes', 20 << 30)), + log_target=False, + ) + if not isinstance(result, dict): + raise ValueError('Docker shadow full result must be an object') + if not ( + result.get('errors') or result.get('skipped') + or result.get('source_failure') + or result.get('warnings') or result.get('degraded') + or ((result.get('scan_meta') or {}).get('docker_layer_scope') or {}).get('coverage_complete') is False + or ((result.get('scan_meta') or {}).get('docker_full_recovery') or {}).get('coverage_complete') is False + ): + rerun_routed, rerun_detectors = map( + set, private_result_identities(result, control['normalized_target']), + ) + result = None + return { + 'full_routed_count': len(reference_routed), + 'full_detector_count': len(reference_detectors), + 'failure_count': 0, + 'safety_regression_count': int( + rerun_routed != reference_routed + or rerun_detectors != reference_detectors + ), + **empty_failure_metrics(), + } + incomplete_reason = _full_scan_incomplete_reason(result) + retryable = bool(result.get('retryable', True)) + _clear_private_result(result) + result = None + if not retryable: + break + print( + 'Docker adaptive shadow full side incomplete: ' + f'reason_code={incomplete_reason} attempts={attempts}' + ) + metrics = empty_failure_metrics() + metrics['diagnostic_full_incomplete'] = 1 + metrics.update({ + 'full_routed_count': len(reference_routed), + 'full_detector_count': len(reference_detectors), + 'failure_count': 1, + 'safety_regression_count': 0, + }) + return metrics + finally: + _clear_private_result(result) + reference_routed.clear() + reference_detectors.clear() + rerun_routed.clear() + rerun_detectors.clear() + + +def private_adaptive_side_metrics( + db, control, *, limits, checkpoint, scan_policy_sha256, scan_kwargs, + platform_os='linux', platform_arch='amd64', min_free_bytes=0, +): + reference_routed, reference_detectors = db.docker_adaptive_shadow_control_identities( + control['target_scan_id'], + ) + reference_routed = set(reference_routed) + reference_detectors = set(reference_detectors) + adaptive_routed = set() + adaptive_detectors = set() + outcome = None + try: + max_attempts = min(10, max(1, int(scan_kwargs.get('max_attempts', 3) or 3))) + incomplete_reason = 'adaptive_scan_error' + attempts = 0 + for attempts in range(1, max_attempts + 1): + try: + outcome = execute_private_adaptive_scan( + control['normalized_target'], limits=limits, checkpoint=checkpoint, + scan_policy_sha256=scan_policy_sha256, + timeout_sec=int(scan_kwargs['timeout_sec']), + platform_os=platform_os, platform_arch=platform_arch, + min_free_bytes=min_free_bytes, scan_kwargs=scan_kwargs, + ) + adaptive_routed = set(outcome.pop('routed')) + adaptive_detectors = set(outcome.pop('detectors')) + metrics = { + 'adaptive_routed_count': len(adaptive_routed), + 'routed_intersection_count': len(reference_routed & adaptive_routed), + 'adaptive_detector_count': len(adaptive_detectors), + 'detector_intersection_count': len(reference_detectors & adaptive_detectors), + 'omitted_descriptor_count': int(outcome['omitted_descriptor_count']), + 'failure_count': int(outcome['failure_count']), + } + metrics.update(outcome['selection_metrics']) + metrics.update(outcome['failure_metrics']) + return metrics + except DockerShadowPrivacyError: + raise + except scanner.ScanSlotFatalError: + raise + except MemoryError: + raise + except Exception as exc: + incomplete_reason = _adaptive_scan_incomplete_reason(exc) + retryable = bool(getattr(exc, 'retryable', not isinstance(exc, ValueError))) + if not retryable: + break + finally: + if isinstance(outcome, dict): + outcome.clear() + outcome = None + adaptive_routed.clear() + adaptive_detectors.clear() + print( + 'Docker adaptive shadow adaptive side incomplete: ' + f'reason_code={incomplete_reason} attempts={attempts}' + ) + metrics = empty_selection_metrics() + metrics.update(empty_failure_metrics()) + metrics['diagnostic_adaptive_incomplete'] = 1 + metrics.update({ + 'adaptive_routed_count': 0, + 'routed_intersection_count': 0, + 'adaptive_detector_count': 0, + 'detector_intersection_count': 0, + 'omitted_descriptor_count': 0, + 'failure_count': 1, + }) + return metrics + finally: + if isinstance(outcome, dict): + outcome.clear() + reference_routed.clear() + reference_detectors.clear() + adaptive_routed.clear() + adaptive_detectors.clear() + + +def _shadow_config(config): + supervisor = config.get('supervisor') if isinstance(config, dict) else None + supervisor = supervisor if isinstance(supervisor, dict) else {} + value = supervisor.get('docker_shadow') + if not isinstance(value, dict): + raise ValueError('Docker shadow supervisor configuration is required') + allowed = {'enabled', 'cohort_size', 'lease_seconds'} + if set(value) - allowed: + raise ValueError('Docker shadow supervisor configuration has unsupported fields') + if not console_runner.bool_config(value.get('enabled'), False): + raise ValueError('Docker shadow operator command is disabled') + cohort_size = int(value.get('cohort_size', DOCKER_ADAPTIVE_GATE_MIN_CONTROLS)) + lease_seconds = int(value.get('lease_seconds', 3600)) + if not DOCKER_ADAPTIVE_GATE_MIN_CONTROLS <= cohort_size <= DOCKER_ADAPTIVE_GATE_MAX_CONTROLS: + raise ValueError('Docker shadow cohort size must be between 50 and 100') + if not 60 <= lease_seconds <= 86400: + raise ValueError('Docker shadow lease duration is outside the supported range') + return {'cohort_size': cohort_size, 'lease_seconds': lease_seconds} + + +def _scan_kwargs(source_args): + return { + 'timeout_sec': max(1, int(source_args.timeout)), + 'detectors': source_args.detectors, + 'exclude_detectors': source_args.exclude_detectors, + 'no_verification': bool(source_args.no_verification), + 'trufflehog_config': source_args.trufflehog_config, + 'trufflehog_concurrency': max(0, int(source_args.trufflehog_concurrency or 0)), + 'max_attempts': max( + 1, int(getattr(source_args, 'target_retry_max_attempts', 3) or 3), + ), + } + + +def run_shadow(config_path): + require_active_supervisor_child( + config_path, child_kind='docker-shadow', require_dsn=True, + ) + config = console_runner.load_config(config_path) + settings = _shadow_config(config) + source_config = (config.get('sources') or {}).get('dockerhub') + if not isinstance(source_config, dict): + raise ValueError('Docker shadow requires the DockerHub source configuration') + global_config = config.get('global') or {} + if not isinstance(global_config, dict): + raise ValueError('Docker shadow global configuration must be a mapping') + + console_runner.apply_global_config(global_config) + scanner.initialize_scanner_runtime(preflight_complete=True) + secrets_config = console_runner.load_secrets(config, config_path) + state = { + 'version': 1, + 'sources': {'dockerhub': console_runner.default_source_state()}, + } + console_runner.configure_source_auth( + 'dockerhub', source_config, state=state, secrets=secrets_config, + ) + source_args = console_runner.build_args_from_source_config( + 'dockerhub', source_config, global_config, '', auth_entry=None, + ) + scanner.scan_config.drop_detectors = scanner.csv_items(source_args.drop_detectors) + scanner.scan_config.trufflehog_job_memory_limit_bytes = int( + source_args.trufflehog_job_memory_limit_bytes + ) + scan_kwargs = _scan_kwargs(source_args) + limits = console_runner.docker_layer_limits(source_args) + scan_kwargs['docker_recovery_limits'] = limits + scan_kwargs['docker_recovery_min_free_bytes'] = int(getattr(source_args, 'docker_layer_min_free_bytes', 20 << 30)) + checkpoint = console_runner.docker_adaptive_checkpoint(source_args) + scan_policy_sha256 = console_runner.docker_layer_scan_policy_sha256( + source_args, scan_kwargs, + ) + execution_policy_sha256 = docker_layer_execution_policy_sha256( + scan_policy_sha256, limits, + ) + selection_policy_sha256 = docker_layer_selection_policy_sha256(limits) + selection_salt = hashlib.sha256( + ('docker-shadow-controls-v1:' + scan_policy_sha256 + ':' + + execution_policy_sha256 + ':' + selection_policy_sha256).encode('ascii') + ).hexdigest() + + db_url = str(os.getenv('TRUF_MANAGED_POSTGRES_DSN') or '') + if not db_url: + raise RuntimeError('Docker shadow canonical PostgreSQL authority is unavailable') + db = ScannerDB(db_url=db_url, initialize=False) + controls = [] + report = None + owner = 'docker-shadow:{}:{}'.format( + os.getpid(), hashlib.sha256(socket.gethostname().encode('utf-8')).hexdigest()[:16], + ) + try: + if not db.enabled: + raise RuntimeError('Docker shadow PostgreSQL connection is unavailable') + db.set_application_name('truf-docker-adaptive-shadow') + controls = db.docker_adaptive_shadow_controls( + scan_policy_sha256, settings['cohort_size'], selection_salt, + ) + report = db.start_docker_adaptive_shadow_report( + scan_policy_sha256, execution_policy_sha256, selection_policy_sha256, + settings['cohort_size'], owner, lease_seconds=settings['lease_seconds'], + ) + aggregate = { + 'completed_pairs': 0, + 'full_routed_count': 0, + 'adaptive_routed_count': 0, + 'routed_intersection_count': 0, + 'full_detector_count': 0, + 'adaptive_detector_count': 0, + 'detector_intersection_count': 0, + 'full_slot_ms': 0, + 'adaptive_slot_ms': 0, + 'omitted_descriptor_count': 0, + 'failure_count': 0, + 'privacy_violation_count': 0, + 'safety_regression_count': 0, + } + selection_metrics = empty_selection_metrics() + failure_metrics = empty_failure_metrics() + + def durable_checkpoint(): + db.checkpoint_docker_adaptive_shadow_report( + report['report_token'], owner, report['lease_token'], + lease_seconds=settings['lease_seconds'], + ) + + for index, control in enumerate(controls): + sides = ('full', 'adaptive') if index % 2 == 0 else ('adaptive', 'full') + for side in sides: + if side == 'full': + operation = lambda control=control: private_full_side_metrics( + db, control, scan_kwargs, + ) + else: + operation = lambda control=control: private_adaptive_side_metrics( + db, control, limits=limits, checkpoint=checkpoint, + scan_policy_sha256=scan_policy_sha256, + scan_kwargs=scan_kwargs, + platform_os=source_args.docker_platform_os, + platform_arch=source_args.docker_platform_arch, + min_free_bytes=source_args.docker_layer_min_free_bytes, + ) + side_metrics, elapsed_ms = run_timed_private_side( + side, operation, durable_checkpoint, + timeout_sec=scan_kwargs['timeout_sec'], + ) + aggregate[f'{side}_slot_ms'] += elapsed_ms + for name, value in side_metrics.items(): + if name in selection_metrics: + selection_metrics[name] += value + elif name in failure_metrics: + failure_metrics[name] += value + else: + aggregate[name] += value + side_metrics.clear() + aggregate['completed_pairs'] += 1 + control.clear() + + completed = db.finish_docker_adaptive_shadow_report( + report['report_token'], owner, report['lease_token'], + selection_metrics=selection_metrics, **aggregate, + ) + print( + 'Docker adaptive shadow report: ' + f'id={completed["report_id"]} passed={str(completed["passed"]).lower()} ' + f'controls={completed["completed_pairs"]} ' + f'routed_recall_ppm={completed["routed_recall_ppm"]} ' + f'slot_ratio_ppm={completed["slot_ratio_ppm"]} ' + 'failure_categories=' + ','.join( + f'{name.removeprefix("diagnostic_")}:{failure_metrics[name]}' + for name in SHADOW_FAILURE_METRIC_KEYS + if failure_metrics[name] + ) + ) + return 0 if completed['passed'] else 2 + except BaseException as exc: + if report is not None: + try: + db.fail_docker_adaptive_shadow_report( + report['report_token'], owner, report['lease_token'], + privacy_violation_count=int(isinstance(exc, DockerShadowPrivacyError)), + safety_regression_count=0, + ) + except Exception: + pass + print( + 'Docker adaptive shadow report failed safely: ' + f'reason_code={shadow_failure_reason_code(exc)}' + ) + return 1 + finally: + for control in controls: + if isinstance(control, dict): + control.clear() + controls.clear() + db.close() + scanner.docker_token_manager.cleanup() + + +def parse_args(argv=None): + parser = argparse.ArgumentParser(description='Run one private Docker adaptive shadow report') + parser.add_argument('--config', required=True) + return parser.parse_args(argv) + + +def main(argv=None): + args = parse_args(argv) + return run_shadow(args.config) + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/app/host_agent_apply.py b/app/host_agent_apply.py new file mode 100644 index 0000000..5fda06f --- /dev/null +++ b/app/host_agent_apply.py @@ -0,0 +1,1119 @@ +from dataclasses import dataclass, field +import hashlib +import hmac +import os +from pathlib import Path +import stat + +from host_agent_protocol import ( + HostAgentAction, + HostAgentProtocolError, + HostAgentRequest, + decode_request_payload, + encode_request_payload, +) +from host_agent_state import HostStateError, failed_hold_operation +from runtime_document import ( + MAX_CONFIG_DOCUMENT_BYTES, + MAX_SECRETS_DOCUMENT_BYTES, + RuntimeDocumentError, + preview_runtime_documents, +) +from runtime_security import ( + PrivateFileLock, + durable_replace, + fsync_directory, + reject_reparse_components, +) + + +HOST_ROOT_UID = 0 +HOST_ROOT_GID = 0 +HOST_RUNTIME_UID = 10001 +HOST_RUNTIME_GID = 10001 + +HOST_ACTIVE_DIRECTORY = Path('/etc/truf/runtime') +HOST_ACTIVE_DIRECTORY_MODE = 0o755 +HOST_ACTIVE_CONFIG_PATH = HOST_ACTIVE_DIRECTORY / 'config.yaml' +HOST_ACTIVE_SECRETS_PATH = HOST_ACTIVE_DIRECTORY / 'secrets.yaml' + +HOST_CANDIDATE_DIRECTORY = Path('/var/lib/truf/runtime-document-candidates') +HOST_CANDIDATE_DIRECTORY_MODE = 0o700 +HOST_CONFIG_CANDIDATE_PATH = HOST_CANDIDATE_DIRECTORY / 'config.yaml' +HOST_SECRETS_CANDIDATE_PATH = HOST_CANDIDATE_DIRECTORY / 'secrets.yaml' + +HOST_APPLY_DIRECTORY = Path('/run/truf-host-agent') +HOST_APPLY_LOCK_PATH = HOST_APPLY_DIRECTORY / 'apply.lock' +HOST_BACKUP_ROOT = Path('/var/lib/truf/host-agent/backups') +HOST_BACKUP_DIRECTORY_MODE = 0o700 +HOST_TEMPLATE_PATH = Path(__file__).with_name('config.linux.yaml') + + +class HostApplyError(RuntimeError): + def __init__(self, category): + self.category = category + super().__init__('host runtime apply failed') + + +_STOPPED_RUNTIME_AUTHORITY = object() + + +@dataclass(frozen=True, slots=True) +class _StoppedRuntimeProof: + operation_id: str + purpose: str + nonce: object = field(repr=False) + authority: object = field(repr=False) + + +def _new_stopped_runtime_proof(operation_id, *, purpose='forward'): + if purpose not in ('forward', 'rollback'): + raise HostApplyError('state') + return _StoppedRuntimeProof( + str(operation_id), purpose, object(), _STOPPED_RUNTIME_AUTHORITY, + ) + + +@dataclass(frozen=True, slots=True) +class _FileSnapshot: + payload: bytes = field(repr=False) + sha256: str + byte_count: int + identity: tuple + + +def _expected_identity(request): + return { + 'active_config_sha256': request.active_config_sha256, + 'active_secrets_sha256': request.active_secrets_sha256, + 'candidate_config_sha256': request.candidate_config_sha256, + 'candidate_secrets_sha256': request.candidate_secrets_sha256, + } + + +def _require_directory(path, *, uid, gid, mode): + try: + reject_reparse_components(path) + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISDIR(details.st_mode): + raise OSError('not a directory') + if os.name != 'nt' and ( + details.st_uid != uid + or details.st_gid != gid + or stat.S_IMODE(details.st_mode) != mode + ): + raise OSError('directory metadata') + return details + except HostApplyError: + raise + except Exception: + raise HostApplyError('filesystem') from None + + +def _file_identity(details): + return ( + details.st_dev, + details.st_ino, + details.st_size, + getattr(details, 'st_mtime_ns', None), + None if os.name == 'nt' else getattr(details, 'st_ctime_ns', None), + ) + + +def _require_file_metadata(details, *, uid, gid, mode=None, immutable=False): + if not stat.S_ISREG(details.st_mode) or details.st_nlink != 1: + raise OSError('file type') + if os.name == 'nt': + return + if details.st_uid != uid or details.st_gid != gid: + raise OSError('file owner') + permissions = stat.S_IMODE(details.st_mode) + if mode is not None and permissions != mode: + raise OSError('file mode') + if immutable and permissions & 0o022: + raise OSError('mutable trusted file') + + +def _snapshot_file( + path, + maximum, + *, + uid, + gid, + mode=None, + immutable=False, + expected_sha256=None, +): + descriptor = None + payload = None + try: + path = reject_reparse_components(path) + flags = os.O_RDONLY + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + if hasattr(os, 'O_CLOEXEC'): + flags |= os.O_CLOEXEC + if hasattr(os, 'O_NOFOLLOW'): + flags |= os.O_NOFOLLOW + descriptor = os.open(path, flags) + before = os.fstat(descriptor) + _require_file_metadata( + before, uid=uid, gid=gid, mode=mode, immutable=immutable, + ) + with os.fdopen(descriptor, 'rb') as handle: + descriptor = None + payload = handle.read(maximum + 1) + after = os.fstat(handle.fileno()) + current = os.stat(path, follow_symlinks=False) + _require_file_metadata( + after, uid=uid, gid=gid, mode=mode, immutable=immutable, + ) + _require_file_metadata( + current, uid=uid, gid=gid, mode=mode, immutable=immutable, + ) + if _file_identity(before) != _file_identity(after): + raise OSError('file changed during read') + if _file_identity(after) != _file_identity(current): + raise OSError('file path changed during read') + if len(payload) > maximum: + raise HostApplyError('size') + digest = hashlib.sha256(payload).hexdigest() + if expected_sha256 is not None and not hmac.compare_digest( + digest, expected_sha256, + ): + raise HostApplyError('identity') + return _FileSnapshot( + payload=payload, + sha256=digest, + byte_count=len(payload), + identity=_file_identity(current), + ) + except HostApplyError: + raise + except Exception: + payload = None + raise HostApplyError('filesystem') from None + finally: + if descriptor is not None: + os.close(descriptor) + + +def _create_owned_file(path, payload, *, uid, gid, mode, expected_sha256): + descriptor = None + created = False + try: + reject_reparse_components(Path(path).parent) + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + if hasattr(os, 'O_CLOEXEC'): + flags |= os.O_CLOEXEC + if hasattr(os, 'O_NOFOLLOW'): + flags |= os.O_NOFOLLOW + descriptor = os.open(path, flags, mode) + created = True + if os.name != 'nt': + os.fchmod(descriptor, mode) + details = os.fstat(descriptor) + if details.st_uid != uid or details.st_gid != gid: + os.fchown(descriptor, uid, gid) + view = memoryview(payload) + written = 0 + while written < len(view): + count = os.write(descriptor, view[written:]) + if count <= 0: + raise OSError('short write') + written += count + os.fsync(descriptor) + os.close(descriptor) + descriptor = None + return _snapshot_file( + path, + len(payload), + uid=uid, + gid=gid, + mode=mode, + expected_sha256=expected_sha256, + ) + except BaseException: + if descriptor is not None: + os.close(descriptor) + if created: + try: + os.unlink(path) + fsync_directory(Path(path).parent) + except OSError: + pass + raise + + +def _remove_stale_stage(path): + try: + reject_reparse_components(Path(path).parent) + details = os.stat(path, follow_symlinks=False) + if stat.S_ISDIR(details.st_mode): + raise OSError('stage is a directory') + os.unlink(path) + fsync_directory(Path(path).parent) + except FileNotFoundError: + return + except Exception: + raise HostApplyError('filesystem') from None + + +def _read_descriptor(descriptor, maximum): + os.lseek(descriptor, 0, os.SEEK_SET) + chunks = [] + remaining = maximum + 1 + while remaining: + chunk = os.read(descriptor, min(remaining, 64 * 1024)) + if not chunk: + break + chunks.append(chunk) + remaining -= len(chunk) + return b''.join(chunks) + + +def _open_displaced_guard(path, original, maximum, *, allow_root=False): + if os.name == 'nt': + return None + descriptor = None + try: + flags = os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) | getattr(os, 'O_NOFOLLOW', 0) + descriptor = os.open(path, flags) + details = os.fstat(descriptor) + try: + _require_file_metadata( + details, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + ) + except OSError: + if not allow_root: + raise + _require_file_metadata( + details, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=0o600, + ) + payload = _read_descriptor(descriptor, maximum) + if ( + _file_identity(details) != original.identity + or len(payload) > maximum + or not hmac.compare_digest(payload, original.payload) + ): + raise HostApplyError('identity') + return descriptor + except HostApplyError: + if descriptor is not None: + os.close(descriptor) + raise + except Exception: + if descriptor is not None: + os.close(descriptor) + raise HostApplyError('filesystem') from None + + +def _verify_displaced_guard(descriptor, original, maximum): + if descriptor is None: + return + try: + payload = _read_descriptor(descriptor, maximum) + details = os.fstat(descriptor) + if ( + not stat.S_ISREG(details.st_mode) + or details.st_dev != original.identity[0] + or details.st_ino != original.identity[1] + or len(payload) > maximum + or not hmac.compare_digest(payload, original.payload) + ): + raise HostApplyError('identity') + except HostApplyError: + raise + except Exception: + raise HostApplyError('filesystem') from None + + +def _adopt_published_file(path, source): + descriptor = None + try: + staged = _snapshot_file( + path, + source.byte_count, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=0o600, + expected_sha256=source.sha256, + ) + if not hmac.compare_digest(staged.payload, source.payload): + raise HostApplyError('identity') + flags = os.O_RDWR | getattr(os, 'O_CLOEXEC', 0) | getattr(os, 'O_NOFOLLOW', 0) + descriptor = os.open(path, flags) + details = os.fstat(descriptor) + _require_file_metadata( + details, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=0o600, + ) + if _file_identity(details) != staged.identity: + raise HostApplyError('identity') + if os.name != 'nt': + os.fchown(descriptor, HOST_RUNTIME_UID, HOST_RUNTIME_GID) + os.fchmod(descriptor, 0o600) + os.fsync(descriptor) + os.close(descriptor) + descriptor = None + fsync_directory(Path(path).parent) + final = _snapshot_file( + path, + source.byte_count, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + expected_sha256=source.sha256, + ) + if not hmac.compare_digest(final.payload, source.payload): + raise HostApplyError('identity') + except HostApplyError: + raise + except Exception: + raise HostApplyError('filesystem') from None + finally: + if descriptor is not None: + os.close(descriptor) + + +def _snapshot_active(path, maximum, *, allow_root=False, expected_sha256=None): + if os.name == 'nt': + return _snapshot_file( + path, + maximum, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + expected_sha256=expected_sha256, + ), False + try: + reject_reparse_components(path) + details = os.stat(path, follow_symlinks=False) + runtime_owned = ( + details.st_uid == HOST_RUNTIME_UID + and details.st_gid == HOST_RUNTIME_GID + ) + root_owned = ( + details.st_uid == HOST_ROOT_UID + and details.st_gid == HOST_ROOT_GID + and not runtime_owned + ) + if not runtime_owned and not (allow_root and root_owned): + raise OSError('active file owner') + uid = HOST_ROOT_UID if root_owned else HOST_RUNTIME_UID + gid = HOST_ROOT_GID if root_owned else HOST_RUNTIME_GID + return _snapshot_file( + path, + maximum, + uid=uid, + gid=gid, + mode=0o600, + expected_sha256=expected_sha256, + ), root_owned + except HostApplyError: + raise + except Exception: + raise HostApplyError('filesystem') from None + + +def _same_snapshot(left, right): + return ( + left.sha256 == right.sha256 + and left.byte_count == right.byte_count + and left.identity == right.identity + ) + + +class HostApplySession: + """One fixed-path host apply claim held under the global apply lock.""" + + def __init__( + self, request, database, *, package_capabilities=None, + package_capability_provider=None, + ): + try: + request = decode_request_payload(encode_request_payload(request)) + except HostAgentProtocolError: + raise HostApplyError('request') + self.request = request + self.database = database + self.package_capabilities = package_capabilities + self.package_capability_provider = package_capability_provider + self.claim = None + self._lock = None + self._active = None + self._candidates = None + self._selected = None + self._backup_complete = False + self._publication_complete = False + self._publication_state = 'original' + self._pending_adoption = set() + self._used_proof_nonces = set() + self._failed_hold_replay = False + self._entered = False + + def __enter__(self): + if self._entered: + raise HostApplyError('state') + try: + self._lock = PrivateFileLock(HOST_APPLY_LOCK_PATH).acquire() + except BlockingIOError: + raise HostApplyError('busy') from None + except Exception: + raise HostApplyError('filesystem') from None + self._entered = True + try: + try: + held_operation = failed_hold_operation() + if ( + held_operation is not None + and held_operation != self.request.operation_id + ): + raise HostApplyError('failed_hold') + self._failed_hold_replay = ( + held_operation == self.request.operation_id + ) + except HostStateError: + raise HostApplyError('failed_hold') from None + connection = getattr(self.database, 'conn', None) + if connection is None or not connection.is_postgres: + raise HostApplyError('authority') + try: + self.claim = self.database.claim_runtime_operation_execution( + operation_id=self.request.operation_id, + action=self.request.action.value, + expected_identity=_expected_identity(self.request), + ) + except Exception: + raise HostApplyError('authority') from None + if self._failed_hold_replay: + return self + self._prepare() + return self + except BaseException: + self.close() + raise + + def _prepare(self): + _require_directory( + HOST_ACTIVE_DIRECTORY, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=HOST_ACTIVE_DIRECTORY_MODE, + ) + candidate_names = { + HostAgentAction.APPLY_CONFIG: ('config',), + HostAgentAction.APPLY_SECRETS: ('secrets',), + HostAgentAction.APPLY_BOTH: ('config', 'secrets'), + HostAgentAction.RESTART: (), + }[self.request.action] + active = {} + root_owned = set() + root_original = set() + active_paths = { + 'config': (HOST_ACTIVE_CONFIG_PATH, MAX_CONFIG_DOCUMENT_BYTES), + 'secrets': (HOST_ACTIVE_SECRETS_PATH, MAX_SECRETS_DOCUMENT_BYTES), + } + for name, (path, maximum) in active_paths.items(): + active[name], is_root_owned = _snapshot_active( + path, + maximum, + allow_root=( + bool(self.claim.get('replayed')) and name in candidate_names + ), + ) + if is_root_owned: + root_owned.add(name) + candidates = {} + if candidate_names: + _require_directory( + HOST_CANDIDATE_DIRECTORY, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=HOST_CANDIDATE_DIRECTORY_MODE, + ) + if 'config' in candidate_names: + candidates['config'] = _snapshot_file( + HOST_CONFIG_CANDIDATE_PATH, + MAX_CONFIG_DOCUMENT_BYTES, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + expected_sha256=self.request.candidate_config_sha256, + ) + if 'secrets' in candidate_names: + candidates['secrets'] = _snapshot_file( + HOST_SECRETS_CANDIDATE_PATH, + MAX_SECRETS_DOCUMENT_BYTES, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + expected_sha256=self.request.candidate_secrets_sha256, + ) + original_hashes = { + 'config': self.request.active_config_sha256, + 'secrets': self.request.active_secrets_sha256, + } + published = [] + for name, snapshot in active.items(): + if hmac.compare_digest(snapshot.sha256, original_hashes[name]): + if name in root_owned: + if not self.claim.get('replayed') or name not in candidate_names: + raise HostApplyError('identity') + root_original.add(name) + continue + candidate = candidates.get(name) + if ( + self.claim.get('replayed') + and candidate is not None + and hmac.compare_digest(snapshot.sha256, candidate.sha256) + and hmac.compare_digest(snapshot.payload, candidate.payload) + ): + published.append(name) + continue + raise HostApplyError('identity') + if published or root_original: + _require_directory( + HOST_BACKUP_ROOT, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=HOST_BACKUP_DIRECTORY_MODE, + ) + operation_directory = HOST_BACKUP_ROOT / self.request.operation_id + _require_directory( + operation_directory, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=HOST_BACKUP_DIRECTORY_MODE, + ) + maximums = { + 'config': MAX_CONFIG_DOCUMENT_BYTES, + 'secrets': MAX_SECRETS_DOCUMENT_BYTES, + } + for name in candidate_names: + active[name] = _snapshot_file( + operation_directory / f'{name}.yaml', + maximums[name], + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=0o600, + expected_sha256=original_hashes[name], + ) + template = _snapshot_file( + HOST_TEMPLATE_PATH, + MAX_CONFIG_DOCUMENT_BYTES, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + immutable=True, + ) + try: + capability_provider = self.package_capability_provider + active_capabilities = ( + capability_provider(active['config'].payload) + if capability_provider is not None else self.package_capabilities + ) + effective_config = candidates.get('config', active['config']).payload + candidate_capabilities = ( + capability_provider(effective_config) + if capability_provider is not None else self.package_capabilities + ) + preview_runtime_documents( + active['config'].payload, + active['secrets'].payload, + config_template_payload=template.payload, + package_capabilities=active_capabilities, + ) + preview_runtime_documents( + effective_config, + candidates.get('secrets', active['secrets']).payload, + config_template_payload=template.payload, + package_capabilities=candidate_capabilities, + ) + except (RuntimeDocumentError, OSError, ValueError): + raise HostApplyError('validation') from None + finally: + active_capabilities = candidate_capabilities = effective_config = None + self._active = active + self._candidates = candidates + self._selected = candidate_names + if published or root_original: + self._backup_complete = True + self._publication_complete = len(published) == len(candidate_names) + self._publication_state = ( + 'candidate' + if self._publication_complete else ( + 'partial' if published else 'original' + ) + ) + self._pending_adoption = root_owned.intersection(published) + + @property + def publication_state(self): + if not self._entered or ( + self._active is None and not self._failed_hold_replay + ): + raise HostApplyError('state') + return self._publication_state + + def original_identity(self): + if not self._entered or self._active is None: + raise HostApplyError('state') + return { + 'active_config_sha256': self._active['config'].sha256, + 'active_secrets_sha256': self._active['secrets'].sha256, + } + + def _consume_stopped_proof(self, stopped_proof, purpose): + if ( + not isinstance(stopped_proof, _StoppedRuntimeProof) + or stopped_proof.authority is not _STOPPED_RUNTIME_AUTHORITY + or stopped_proof.operation_id != self.request.operation_id + or stopped_proof.purpose != purpose + or stopped_proof.nonce in self._used_proof_nonces + ): + raise HostApplyError('state') + self._used_proof_nonces.add(stopped_proof.nonce) + + def backup(self): + if not self._entered or self._active is None: + raise HostApplyError('state') + if self._backup_complete: + return + if not self._selected: + self._backup_complete = True + return + _require_directory( + HOST_BACKUP_ROOT, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=HOST_BACKUP_DIRECTORY_MODE, + ) + operation_directory = HOST_BACKUP_ROOT / self.request.operation_id + try: + os.mkdir(operation_directory, HOST_BACKUP_DIRECTORY_MODE) + fsync_directory(HOST_BACKUP_ROOT) + except FileExistsError: + pass + except Exception: + raise HostApplyError('filesystem') from None + _require_directory( + operation_directory, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=HOST_BACKUP_DIRECTORY_MODE, + ) + for name in self._selected: + source = self._active[name] + destination = operation_directory / f'{name}.yaml' + try: + stored = _create_owned_file( + destination, + source.payload, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=0o600, + expected_sha256=source.sha256, + ) + except FileExistsError: + stored = _snapshot_file( + destination, + source.byte_count, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=0o600, + expected_sha256=source.sha256, + ) + except HostApplyError: + raise + except Exception: + raise HostApplyError('filesystem') from None + if not hmac.compare_digest(stored.payload, source.payload): + raise HostApplyError('identity') + fsync_directory(operation_directory) + self._backup_complete = True + + def _revalidate(self): + current_active = { + 'config': _snapshot_file( + HOST_ACTIVE_CONFIG_PATH, + MAX_CONFIG_DOCUMENT_BYTES, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + expected_sha256=self._active['config'].sha256, + ), + 'secrets': _snapshot_file( + HOST_ACTIVE_SECRETS_PATH, + MAX_SECRETS_DOCUMENT_BYTES, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + expected_sha256=self._active['secrets'].sha256, + ), + } + for name, original in self._active.items(): + if not _same_snapshot(current_active[name], original): + raise HostApplyError('identity') + candidate_paths = { + 'config': (HOST_CONFIG_CANDIDATE_PATH, MAX_CONFIG_DOCUMENT_BYTES), + 'secrets': (HOST_SECRETS_CANDIDATE_PATH, MAX_SECRETS_DOCUMENT_BYTES), + } + for name, original in self._candidates.items(): + path, maximum = candidate_paths[name] + current = _snapshot_file( + path, + maximum, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + expected_sha256=original.sha256, + ) + if not _same_snapshot(current, original): + raise HostApplyError('identity') + + def revalidate_for_stop(self): + if not self._entered or self._active is None or not self._backup_complete: + raise HostApplyError('state') + if self._publication_state == 'partial': + raise HostApplyError('partial') + if self._publication_complete: + self._revalidate_published() + else: + self._revalidate() + + def _revalidate_published(self): + active_paths = { + 'config': (HOST_ACTIVE_CONFIG_PATH, MAX_CONFIG_DOCUMENT_BYTES), + 'secrets': (HOST_ACTIVE_SECRETS_PATH, MAX_SECRETS_DOCUMENT_BYTES), + } + still_root_owned = set() + for name, (path, maximum) in active_paths.items(): + expected = self._candidates.get(name, self._active[name]) + current, root_owned = _snapshot_active( + path, + maximum, + allow_root=name in self._pending_adoption, + expected_sha256=expected.sha256, + ) + if not hmac.compare_digest(current.payload, expected.payload): + raise HostApplyError('identity') + if root_owned: + still_root_owned.add(name) + candidate_paths = { + 'config': (HOST_CONFIG_CANDIDATE_PATH, MAX_CONFIG_DOCUMENT_BYTES), + 'secrets': (HOST_SECRETS_CANDIDATE_PATH, MAX_SECRETS_DOCUMENT_BYTES), + } + for name, original in self._candidates.items(): + path, maximum = candidate_paths[name] + current = _snapshot_file( + path, + maximum, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + expected_sha256=original.sha256, + ) + if not _same_snapshot(current, original): + raise HostApplyError('identity') + operation_directory = HOST_BACKUP_ROOT / self.request.operation_id + maximums = { + 'config': MAX_CONFIG_DOCUMENT_BYTES, + 'secrets': MAX_SECRETS_DOCUMENT_BYTES, + } + for name in self._selected: + current = _snapshot_file( + operation_directory / f'{name}.yaml', + maximums[name], + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=0o600, + expected_sha256=self._active[name].sha256, + ) + if not _same_snapshot(current, self._active[name]): + raise HostApplyError('identity') + self._pending_adoption = still_root_owned + + def replace(self, stopped_proof): + if not self._entered or self._active is None or not self._backup_complete: + raise HostApplyError('state') + self._consume_stopped_proof(stopped_proof, 'forward') + if self._publication_state == 'partial': + raise HostApplyError('partial') + if not self._selected: + return { + 'active_config_sha256': self._active['config'].sha256, + 'active_secrets_sha256': self._active['secrets'].sha256, + } + if self._publication_complete: + self._revalidate_published() + for name in tuple(self._pending_adoption): + path = ( + HOST_ACTIVE_CONFIG_PATH + if name == 'config' else HOST_ACTIVE_SECRETS_PATH + ) + _adopt_published_file(path, self._candidates[name]) + self._pending_adoption.clear() + self._revalidate_published() + return { + 'active_config_sha256': ( + self._candidates.get('config', self._active['config']).sha256 + ), + 'active_secrets_sha256': ( + self._candidates.get('secrets', self._active['secrets']).sha256 + ), + } + active_paths = { + 'config': HOST_ACTIVE_CONFIG_PATH, + 'secrets': HOST_ACTIVE_SECRETS_PATH, + } + staged = {} + guards = {} + publication_started = False + try: + for name in self._selected: + source = self._candidates[name] + stage = HOST_ACTIVE_DIRECTORY / ( + f'.{name}.yaml.{self.request.operation_id}.stage' + ) + _remove_stale_stage(stage) + _create_owned_file( + stage, + source.payload, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=0o600, + expected_sha256=source.sha256, + ) + staged[name] = stage + fsync_directory(HOST_ACTIVE_DIRECTORY) + self._revalidate() + maximums = { + 'config': MAX_CONFIG_DOCUMENT_BYTES, + 'secrets': MAX_SECRETS_DOCUMENT_BYTES, + } + for name in self._selected: + guards[name] = _open_displaced_guard( + active_paths[name], self._active[name], maximums[name], + ) + for name in self._selected: + publication_started = True + durable_replace(staged[name], active_paths[name]) + staged.pop(name) + source = self._candidates[name] + _verify_displaced_guard( + guards[name], self._active[name], maximums[name], + ) + descriptor = guards.pop(name) + if descriptor is not None: + os.close(descriptor) + _adopt_published_file(active_paths[name], source) + except BaseException as exc: + for stage in staged.values(): + try: + os.unlink(stage) + fsync_directory(HOST_ACTIVE_DIRECTORY) + except OSError: + pass + if publication_started: + self._publication_state = 'partial' + raise HostApplyError('partial') from None + if isinstance(exc, HostApplyError): + raise + raise HostApplyError('filesystem') from None + finally: + for descriptor in guards.values(): + if descriptor is not None: + os.close(descriptor) + self._publication_complete = True + self._publication_state = 'candidate' + return { + 'active_config_sha256': ( + self._candidates.get('config', self._active['config']).sha256 + ), + 'active_secrets_sha256': ( + self._candidates.get('secrets', self._active['secrets']).sha256 + ), + } + + def restore_backups(self, stopped_proof): + if not self._entered or self._active is None or not self._backup_complete: + raise HostApplyError('state') + self._consume_stopped_proof(stopped_proof, 'rollback') + if not self._selected: + return self.original_identity() + + active_paths = { + 'config': (HOST_ACTIVE_CONFIG_PATH, MAX_CONFIG_DOCUMENT_BYTES), + 'secrets': (HOST_ACTIVE_SECRETS_PATH, MAX_SECRETS_DOCUMENT_BYTES), + } + candidate_paths = { + 'config': (HOST_CONFIG_CANDIDATE_PATH, MAX_CONFIG_DOCUMENT_BYTES), + 'secrets': (HOST_SECRETS_CANDIDATE_PATH, MAX_SECRETS_DOCUMENT_BYTES), + } + operation_directory = HOST_BACKUP_ROOT / self.request.operation_id + _require_directory( + operation_directory, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=HOST_BACKUP_DIRECTORY_MODE, + ) + for name in self._selected: + maximum = active_paths[name][1] + backup = _snapshot_file( + operation_directory / f'{name}.yaml', + maximum, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=0o600, + expected_sha256=self._active[name].sha256, + ) + if not hmac.compare_digest(backup.payload, self._active[name].payload): + raise HostApplyError('identity') + for name, expected in self._candidates.items(): + path, maximum = candidate_paths[name] + current = _snapshot_file( + path, + maximum, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + expected_sha256=expected.sha256, + ) + if not _same_snapshot(current, expected): + raise HostApplyError('identity') + + observed = {} + observed_root = set() + states = {} + for name, (path, maximum) in active_paths.items(): + current, root_owned = _snapshot_active( + path, + maximum, + allow_root=name in self._selected, + ) + observed[name] = current + if root_owned: + observed_root.add(name) + if hmac.compare_digest(current.payload, self._active[name].payload): + states[name] = 'original' + elif ( + name in self._selected + and hmac.compare_digest( + current.payload, self._candidates[name].payload, + ) + ): + states[name] = 'candidate' + else: + raise HostApplyError('identity') + + staged = {} + guards = {} + restoration_started = False + try: + for name in self._selected: + if states[name] == 'original': + continue + source = self._active[name] + stage = HOST_ACTIVE_DIRECTORY / ( + f'.{name}.yaml.{self.request.operation_id}.rollback' + ) + _remove_stale_stage(stage) + _create_owned_file( + stage, + source.payload, + uid=HOST_ROOT_UID, + gid=HOST_ROOT_GID, + mode=0o600, + expected_sha256=source.sha256, + ) + staged[name] = stage + fsync_directory(HOST_ACTIVE_DIRECTORY) + for name, (path, maximum) in active_paths.items(): + current, _ = _snapshot_active( + path, + maximum, + allow_root=name in self._selected, + ) + if not _same_snapshot(current, observed[name]): + raise HostApplyError('identity') + for name in staged: + path, maximum = active_paths[name] + guards[name] = _open_displaced_guard( + path, observed[name], maximum, allow_root=True, + ) + for name in self._selected: + if name not in staged and name not in observed_root: + continue + path, maximum = active_paths[name] + restoration_started = True + self._publication_state = 'partial' + if name in staged: + durable_replace(staged[name], path) + staged.pop(name) + _verify_displaced_guard(guards[name], observed[name], maximum) + descriptor = guards.pop(name) + if descriptor is not None: + os.close(descriptor) + _adopt_published_file(path, self._active[name]) + except BaseException as exc: + for stage in staged.values(): + try: + os.unlink(stage) + fsync_directory(HOST_ACTIVE_DIRECTORY) + except OSError: + pass + if restoration_started: + raise HostApplyError('rollback') from None + if isinstance(exc, HostApplyError): + raise + raise HostApplyError('filesystem') from None + finally: + for descriptor in guards.values(): + if descriptor is not None: + os.close(descriptor) + + for name, (path, maximum) in active_paths.items(): + current = _snapshot_file( + path, + maximum, + uid=HOST_RUNTIME_UID, + gid=HOST_RUNTIME_GID, + mode=0o600, + expected_sha256=self._active[name].sha256, + ) + if not hmac.compare_digest(current.payload, self._active[name].payload): + raise HostApplyError('rollback') + self._publication_complete = False + self._publication_state = 'original' + self._pending_adoption.clear() + return self.original_identity() + + def close(self): + lock = self._lock + self._lock = None + self._entered = False + self.package_capabilities = None + self.package_capability_provider = None + self._active = None + self._candidates = None + self._selected = None + self._publication_complete = False + self._publication_state = 'original' + self._pending_adoption.clear() + self._used_proof_nonces.clear() + if lock is not None: + lock.release() + + def __exit__(self, exc_type, value, traceback): + self.close() + return False diff --git a/app/host_agent_client.py b/app/host_agent_client.py new file mode 100644 index 0000000..d7328e9 --- /dev/null +++ b/app/host_agent_client.py @@ -0,0 +1,131 @@ +import os +import socket +import stat +import struct +import time + +from host_agent_protocol import ( + CLIENT_CONNECT_TIMEOUT_SECONDS, + EXCHANGE_TIMEOUT_SECONDS, + HOST_AGENT_SOCKET_PATH, + MAX_RESPONSE_PAYLOAD_BYTES, + HostAgentAction, + HostAgentProtocolError, + HostAgentRequest, + HostAgentStatus, + decode_response_frame, + encode_request_frame, + receive_frame, + require_eof, + send_frame, +) + + +class HostAgentClientError(RuntimeError): + def __init__(self, category): + self.category = category + super().__init__('host operations agent request failed') + + +class HostAgentUnavailableError(HostAgentClientError): + pass + + +class HostAgentRejectedError(HostAgentClientError): + pass + + +def _fixed_socket_is_safe(): + try: + details = os.lstat(HOST_AGENT_SOCKET_PATH) + except OSError: + return False + return stat.S_ISSOCK(details.st_mode) and details.st_uid == 0 + + +def _peer_credentials(sock): + if not hasattr(socket, 'SO_PEERCRED'): + raise HostAgentUnavailableError('peer_credentials_unavailable') + try: + raw = sock.getsockopt( + socket.SOL_SOCKET, socket.SO_PEERCRED, struct.calcsize('3i'), + ) + pid, uid, gid = struct.unpack('3i', raw) + except (OSError, struct.error) as exc: + raise HostAgentUnavailableError('peer_credentials_unavailable') from exc + # A peer outside the client's PID namespace is reported as PID 0 even + # though its UID/GID remain authoritative through SO_PEERCRED. + if pid < 0: + raise HostAgentUnavailableError('peer_identity_invalid') + return pid, uid, gid + + +class HostAgentClient: + def __init__(self): + pass + + @staticmethod + def is_available(): + return _fixed_socket_is_safe() + + def dispatch( + self, *, operation_id, action, active_config_sha256, + active_secrets_sha256, candidate_config_sha256, + candidate_secrets_sha256, + ): + try: + request = HostAgentRequest( + operation_id=operation_id, + action=HostAgentAction(action), + active_config_sha256=active_config_sha256, + active_secrets_sha256=active_secrets_sha256, + candidate_config_sha256=candidate_config_sha256, + candidate_secrets_sha256=candidate_secrets_sha256, + ) + frame = encode_request_frame(request) + except (HostAgentProtocolError, TypeError, ValueError) as exc: + raise HostAgentRejectedError('request_invalid') from exc + if not _fixed_socket_is_safe(): + raise HostAgentUnavailableError('socket_unavailable') + deadline = time.monotonic() + EXCHANGE_TIMEOUT_SECONDS + connection = None + response = None + try: + family = getattr(socket, 'AF_UNIX', None) + if family is None: + raise HostAgentUnavailableError('unix_socket_unavailable') + connection = socket.socket(family, socket.SOCK_STREAM) + connection.settimeout(min( + CLIENT_CONNECT_TIMEOUT_SECONDS, + max(0.001, deadline - time.monotonic()), + )) + connection.connect(HOST_AGENT_SOCKET_PATH) + _pid, peer_uid, _gid = _peer_credentials(connection) + if peer_uid != 0: + raise HostAgentUnavailableError('server_identity_invalid') + send_frame(connection, frame, deadline=deadline) + connection.shutdown(socket.SHUT_WR) + response = decode_response_frame(receive_frame( + connection, maximum=MAX_RESPONSE_PAYLOAD_BYTES, deadline=deadline, + )) + require_eof(connection, deadline=deadline) + except HostAgentClientError: + raise + except HostAgentProtocolError as exc: + raise HostAgentUnavailableError('response_invalid') from exc + except (OSError, TimeoutError) as exc: + raise HostAgentUnavailableError('transport_unavailable') from exc + finally: + if connection is not None: + try: + connection.close() + except OSError: + pass + connection = frame = request = family = None + if response.operation_id != operation_id: + raise HostAgentUnavailableError('response_identity_invalid') + if response.status is HostAgentStatus.ACCEPTED: + return response + if response.status is HostAgentStatus.REJECTED: + raise HostAgentRejectedError('request_rejected') + raise HostAgentUnavailableError('request_unavailable') diff --git a/app/host_agent_lifecycle.py b/app/host_agent_lifecycle.py new file mode 100644 index 0000000..594b5d1 --- /dev/null +++ b/app/host_agent_lifecycle.py @@ -0,0 +1,1477 @@ +"""Fixed privileged lifecycle for the managed runtime deployment.""" + +from dataclasses import dataclass +import json +import os +import re +import signal +import stat +import subprocess +import threading +import time + +from host_agent_apply import _new_stopped_runtime_proof +from host_agent_state import HostOperationState, HostStateError + + +DOCKER = '/usr/bin/docker' +PROJECT = 'truf-docker' +PROJECT_DIRECTORY = '/opt/truf' +COMPOSE_FILES = ('/opt/truf/compose.yaml', '/opt/truf/compose.edge.yaml') +DEPLOYMENT_PROFILE_FILE = '/etc/truf/deployment-profile' +STANDALONE_PROFILE_NAME = 'standalone-edge-v1' +SHARED_HOST_PROFILE_NAME = 'shared-host-edge-v1' +EDGE_ENV_FILE = '/etc/truf-edge/edge.env' +RUNTIME_IMAGE = 'truf-local:runtime' +EDGE_IMAGE = 'truf-local:edge' +RUNTIME_SERVICE = 'runtime' +EDGE_SERVICE = 'edge' + +COMMAND_OUTPUT_LIMIT = 16 * 1024 +INSPECT_TIMEOUT = 15.0 +EDGE_STOP_TIMEOUT = 45.0 +RUNTIME_STOP_TIMEOUT = 660.0 +REMOVE_TIMEOUT = 45.0 +RECREATE_TIMEOUT = 120.0 +RUNTIME_HEALTH_TIMEOUT = 240.0 +HEALTH_COMMAND_TIMEOUT = 30.0 +EDGE_START_TIMEOUT = 45.0 +EDGE_STABILITY_SECONDS = 10.0 +EDGE_VERIFY_TIMEOUT = 100.0 +POLL_SECONDS = 2.0 + +_HEX_ID = re.compile(r'[0-9a-f]{64}') +_IMAGE_ID = re.compile(r'sha256:[0-9a-f]{64}') +_RUNTIME_ENTRYPOINT = ( + '/usr/bin/tini', '--', '/usr/local/bin/python3', '-u', '-I', '-S', '-B', + '/opt/truf/app/container_runtime.py', +) +_RUNTIME_HEALTH_TEST = ( + 'CMD', '/usr/local/bin/python3', '-I', '-S', '-B', + '/opt/truf/app/container_runtime.py', 'health', '--config', + '/data/config/config.yaml', +) +_RUNTIME_TMPFS = { + '/run/truf': 'rw,nosuid,nodev,noexec,size=64m,mode=0700,uid=10001,gid=10001', + '/tmp': 'rw,nosuid,nodev,noexec,size=128m,mode=1777', +} +_EDGE_TMPFS = { + '/tmp': 'rw,nosuid,nodev,noexec,size=16m,mode=1777', + '/run': 'rw,nosuid,nodev,noexec,size=4m,mode=0700,uid=10001,gid=10001', +} +_INSPECT_FORMAT = ( + '{"id":{{json .Id}},"image":{{json .Image}},' + '"status":{{json .State.Status}},"running":{{json .State.Running}},' + '"paused":{{json .State.Paused}},' + '"restarting":{{json .State.Restarting}},"dead":{{json .State.Dead}},' + '"pid":{{json .State.Pid}},"exit_code":{{json .State.ExitCode}},' + '"oom_killed":{{json .State.OOMKilled}},' + '"restarts":{{json .RestartCount}},"user":{{json .Config.User}},' + '"entrypoint":{{json .Config.Entrypoint}},' + '"command":{{json .Config.Cmd}},' + '"stop_timeout":{{json .Config.StopTimeout}},' + '"stop_signal":{{with index .Config "StopSignal"}}{{json .}}' + '{{else}}""{{end}},' + '"mounts":"{{range .Mounts}}{{.Type}}|{{if eq .Type "volume"}}' + '{{.Name}}{{end}}|{{.Source}}|' + '{{.Destination}}|{{.RW}};{{end}}",' + '"readonly":{{json .HostConfig.ReadonlyRootfs}},' + '"privileged":{{json .HostConfig.Privileged}},' + '"network":{{json .HostConfig.NetworkMode}},' + '"pid_mode":{{json .HostConfig.PidMode}},' + '"ipc_mode":{{json .HostConfig.IpcMode}},' + '"userns_mode":{{json .HostConfig.UsernsMode}},' + '"cgroupns_mode":{{json .HostConfig.CgroupnsMode}},' + '"uts_mode":{{json .HostConfig.UTSMode}},' + '"group_add":{{json .HostConfig.GroupAdd}},' + '"oci_runtime":{{json .HostConfig.Runtime}},' + '"devices":{{if .HostConfig.Devices}}{{len .HostConfig.Devices}}' + '{{else}}0{{end}},' + '"device_requests":{{if .HostConfig.DeviceRequests}}' + '{{len .HostConfig.DeviceRequests}}{{else}}0{{end}},' + '"device_cgroup_rules":{{if .HostConfig.DeviceCgroupRules}}' + '{{len .HostConfig.DeviceCgroupRules}}{{else}}0{{end}},' + '"ports":{{json .HostConfig.PortBindings}},' + '"tmpfs":{{json .HostConfig.Tmpfs}},' + '"cpus":{{json .HostConfig.NanoCpus}},' + '"memory":{{json .HostConfig.Memory}},' + '"pids_limit":{{json .HostConfig.PidsLimit}},' + '"shm_size":{{json .HostConfig.ShmSize}},' + '"log_config":{{json .HostConfig.LogConfig}},' + '"cap_drop":{{json .HostConfig.CapDrop}},' + '"cap_add":{{json .HostConfig.CapAdd}},' + '"security_opt":{{json .HostConfig.SecurityOpt}},' + '"restart_policy":{{json .HostConfig.RestartPolicy}},' + '"project":{{json (index .Config.Labels "com.docker.compose.project")}},' + '"service":{{json (index .Config.Labels "com.docker.compose.service")}},' + '"oneoff":{{json (index .Config.Labels "com.docker.compose.oneoff")}},' + '"config_hash":{{json (index .Config.Labels ' + '"com.docker.compose.config-hash")}},' + '"config_files":{{json (index .Config.Labels ' + '"com.docker.compose.project.config_files")}},' + '"working_dir":{{json (index .Config.Labels ' + '"com.docker.compose.project.working_dir")}},' + '"health_test":{{with index .Config "Healthcheck"}}{{json .Test}}' + '{{else}}null{{end}},' + '"health":{{with index .State "Health"}}{{json .Status}}' + '{{else}}null{{end}}}' +) +_INSPECT_KEYS = { + 'id', 'image', 'status', 'running', 'paused', 'restarting', 'dead', + 'pid', 'exit_code', 'oom_killed', 'restarts', 'user', 'entrypoint', + 'command', 'stop_timeout', 'stop_signal', 'mounts', 'readonly', + 'privileged', 'network', 'pid_mode', 'ipc_mode', 'userns_mode', + 'cgroupns_mode', 'uts_mode', 'group_add', 'oci_runtime', 'devices', + 'device_requests', 'device_cgroup_rules', 'ports', 'tmpfs', 'cpus', + 'memory', 'pids_limit', 'shm_size', 'log_config', + 'cap_drop', 'cap_add', 'security_opt', 'restart_policy', 'project', + 'service', 'oneoff', 'config_hash', 'config_files', 'working_dir', + 'health_test', 'health', +} + + +class HostLifecycleError(RuntimeError): + def __init__(self, category): + super().__init__('host runtime lifecycle failed') + self.category = str(category) + + +@dataclass(frozen=True, slots=True) +class _DeploymentProfile: + name: str + compose_files: tuple + runtime_network: str + runtime_ports: dict + runtime_data_volume: str + runtime_cpus: int + runtime_memory: int + edge_cap_add: tuple + edge_caddyfile: str + + +STANDALONE_PROFILE = _DeploymentProfile( + name=STANDALONE_PROFILE_NAME, + compose_files=COMPOSE_FILES, + runtime_network=f'{PROJECT}_default', + runtime_ports={'443/tcp': [{'HostIp': '', 'HostPort': '443'}]}, + runtime_data_volume='truf-docker_data', + runtime_cpus=2_000_000_000, + runtime_memory=6 * 1024 ** 3, + edge_cap_add=('NET_BIND_SERVICE',), + edge_caddyfile='/etc/caddy/Caddyfile', +) +SHARED_HOST_PROFILE = _DeploymentProfile( + name=SHARED_HOST_PROFILE_NAME, + compose_files=( + '/opt/truf/compose.yaml', '/opt/truf/compose.shared-host.yaml', + ), + runtime_network='host', + runtime_ports={}, + runtime_data_volume='truf-remote-server-data', + runtime_cpus=900_000_000, + runtime_memory=720 * 1024 ** 2, + edge_cap_add=(), + edge_caddyfile='/etc/caddy/Caddyfile.shared-host', +) + + +def _compose_command(profile): + return ( + DOCKER, 'compose', '--ansi', 'never', '--project-name', PROJECT, + '--env-file', EDGE_ENV_FILE, '--project-directory', PROJECT_DIRECTORY, + *(item for path in profile.compose_files for item in ('--file', path)), + ) + + +_COMPOSE = _compose_command(STANDALONE_PROFILE) +_CONFIG_FILES_LABEL = ','.join(COMPOSE_FILES) + + +def _deployment_profile(path=DEPLOYMENT_PROFILE_FILE): + descriptor = None + try: + descriptor = os.open( + path, os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) + | getattr(os, 'O_NOFOLLOW', 0), + ) + before = os.fstat(descriptor) + if ( + not stat.S_ISREG(before.st_mode) or before.st_uid != 0 + or before.st_gid != 0 or stat.S_IMODE(before.st_mode) != 0o444 + or before.st_nlink != 1 or before.st_size > 64 + ): + raise HostLifecycleError('profile') + payload = os.read(descriptor, 65) + after = os.fstat(descriptor) + if ( + len(payload) > 64 or before.st_dev != after.st_dev + or before.st_ino != after.st_ino or before.st_mode != after.st_mode + or before.st_uid != after.st_uid or before.st_gid != after.st_gid + or before.st_nlink != after.st_nlink or before.st_size != after.st_size + or before.st_mtime_ns != after.st_mtime_ns + ): + raise HostLifecycleError('profile') + except FileNotFoundError: + return STANDALONE_PROFILE + except HostLifecycleError: + raise + except OSError: + raise HostLifecycleError('profile') from None + finally: + if descriptor is not None: + os.close(descriptor) + try: + name = payload.decode('ascii').strip() + except UnicodeDecodeError: + raise HostLifecycleError('profile') from None + profiles = { + STANDALONE_PROFILE_NAME: STANDALONE_PROFILE, + SHARED_HOST_PROFILE_NAME: SHARED_HOST_PROFILE, + } + if name not in profiles: + raise HostLifecycleError('profile') + return profiles[name] + + +@dataclass(frozen=True, slots=True) +class _ContainerState: + container_id: str + image_id: str + status: str + running: bool + paused: bool + restarting: bool + dead: bool + pid: int + exit_code: int + oom_killed: bool + restarts: int + user: str + entrypoint: object + command: object + stop_timeout: int + stop_signal: str + mounts: tuple + readonly: bool + privileged: bool + network: str + pid_mode: str + ipc_mode: str + userns_mode: str + cgroupns_mode: str + uts_mode: str + group_add: tuple + oci_runtime: str + devices: int + device_requests: int + device_cgroup_rules: int + ports: dict + tmpfs: dict + cpus: int + memory: int + pids_limit: int + shm_size: int + log_config: dict + cap_drop: tuple + cap_add: tuple + security_opt: tuple + restart_policy: dict + project: str + service: str + oneoff: str + config_hash: str + config_files: str + working_dir: str + health_test: object + health: object + + +@dataclass(frozen=True, slots=True) +class DeploymentSnapshot: + runtime: _ContainerState + edge: _ContainerState + runtime_image_id: str + edge_image_id: str + + +@dataclass(slots=True) +class _DeploymentAttempt: + snapshot: DeploymentSnapshot + token: object + phase: str = 'preflight' + mutation_started: bool = False + edge_stop_issued: bool = False + runtime_stop_issued: bool = False + forward_runtime_id: str = None + forward_edge_id: str = None + forward_edge_config_hash: str = None + + +def _subprocess_runner(command, timeout): + environment = { + 'HOME': '/root', + 'LANG': 'C.UTF-8', + 'LC_ALL': 'C.UTF-8', + 'PATH': '/usr/bin:/bin', + } + process = None + reader = None + output_fd = None + output_lock = threading.Lock() + stop_output = threading.Event() + captured = bytearray() + overflow = threading.Event() + read_failed = threading.Event() + + def close_output(): + nonlocal output_fd + with output_lock: + descriptor = output_fd + output_fd = None + if descriptor is not None: + try: + os.close(descriptor) + except OSError: + pass + + def terminate(): + if process is None: + return + try: + os.killpg(process.pid, signal.SIGKILL) + except OSError: + if process.poll() is None: + try: + process.kill() + except OSError: + pass + if process.poll() is None: + try: + process.wait(timeout=1) + except subprocess.TimeoutExpired: + pass + + def read_output(): + try: + while not stop_output.is_set(): + chunk = os.read(output_fd, 4096) + if not chunk: + return + if len(captured) <= COMMAND_OUTPUT_LIMIT: + captured.extend(chunk[:COMMAND_OUTPUT_LIMIT + 1 - len(captured)]) + if len(captured) > COMMAND_OUTPUT_LIMIT: + overflow.set() + terminate() + except Exception: + if not stop_output.is_set(): + read_failed.set() + terminate() + finally: + close_output() + + try: + process = subprocess.Popen( + command, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + cwd='/', + env=environment, + close_fds=True, + start_new_session=True, + ) + output_fd = os.dup(process.stdout.fileno()) + process.stdout.close() + reader = threading.Thread(target=read_output, daemon=True) + reader.start() + try: + return_code = process.wait(timeout=max(0.001, float(timeout))) + except subprocess.TimeoutExpired: + terminate() + raise HostLifecycleError('timeout') from None + reader.join(timeout=1) + if ( + reader.is_alive() + or read_failed.is_set() + or overflow.is_set() + or return_code != 0 + ): + terminate() + raise HostLifecycleError('command') + return bytes(captured) + except HostLifecycleError: + raise + except Exception: + terminate() + raise HostLifecycleError('command') from None + finally: + if process is not None and ( + process.poll() is None or (reader is not None and reader.is_alive()) + ): + terminate() + if reader is not None and reader.is_alive(): + stop_output.set() + close_output() + reader.join(timeout=1) + close_output() + if ( + process is not None and process.stdout is not None + and not process.stdout.closed + ): + process.stdout.close() + + +def _json_payload(payload): + try: + if not payload or len(payload) > COMMAND_OUTPUT_LIMIT: + raise ValueError('bounded JSON required') + return json.loads(payload.decode('utf-8')) + except (UnicodeDecodeError, ValueError, TypeError): + raise HostLifecycleError('evidence') from None + + +def _string_tuple(value): + if value is None: + return () + if not isinstance(value, list) or any(type(item) is not str for item in value): + raise HostLifecycleError('evidence') + return tuple(value) + + +def _optional_string_tuple(value): + return None if value is None else _string_tuple(value) + + +def _mount_tuple(value): + if type(value) is not str or len(value) > 4096 or not value.endswith(';'): + raise HostLifecycleError('evidence') + mounts = [] + entries = value.split(';') + if entries[-1] != '' or len(entries) > 17: + raise HostLifecycleError('evidence') + for entry in entries[:-1]: + fields = entry.split('|') + if len(fields) != 5 or fields[4] not in ('true', 'false'): + raise HostLifecycleError('evidence') + mount_type, name, source, destination, writable = fields + identity = name if mount_type == 'volume' else source + if not mount_type or not identity or not destination or any( + character in mount_type + identity + destination for character in '\r\n"' + ): + raise HostLifecycleError('evidence') + mounts.append((mount_type, identity, destination, writable == 'true')) + return tuple(sorted(mounts)) + + +def _string_map(value): + if value is None: + return {} + if ( + not isinstance(value, dict) + or len(value) > 16 + or any(type(key) is not str or type(item) is not str + for key, item in value.items()) + ): + raise HostLifecycleError('evidence') + return dict(value) + + +def _port_map(value): + if value is None: + return {} + if not isinstance(value, dict) or len(value) > 8: + raise HostLifecycleError('evidence') + result = {} + for port, bindings in value.items(): + if type(port) is not str or not isinstance(bindings, list) or len(bindings) > 8: + raise HostLifecycleError('evidence') + normalized = [] + for binding in bindings: + if ( + not isinstance(binding, dict) + or set(binding) != {'HostIp', 'HostPort'} + or type(binding['HostIp']) is not str + or type(binding['HostPort']) is not str + ): + raise HostLifecycleError('evidence') + normalized.append(dict(binding)) + result[port] = normalized + return result + + +def _container_state(payload): + value = _json_payload(payload) + try: + if not isinstance(value, dict) or set(value) != _INSPECT_KEYS: + raise ValueError('inspect shape') + for name in ('running', 'paused', 'restarting', 'dead', 'oom_killed', + 'readonly', 'privileged'): + if type(value[name]) is not bool: + raise ValueError('inspect boolean') + for name in ( + 'pid', 'exit_code', 'restarts', 'stop_timeout', 'devices', + 'device_requests', 'device_cgroup_rules', 'cpus', 'memory', + 'pids_limit', 'shm_size', + ): + if type(value[name]) is not int or value[name] < 0: + raise ValueError('inspect integer') + for name in ( + 'id', 'image', 'status', 'user', 'stop_signal', 'network', + 'pid_mode', 'ipc_mode', 'userns_mode', 'cgroupns_mode', 'uts_mode', + 'oci_runtime', 'project', 'service', 'oneoff', 'config_hash', + 'config_files', 'working_dir', + ): + if type(value[name]) is not str: + raise ValueError('inspect string') + if value['health'] is not None and type(value['health']) is not str: + raise ValueError('inspect health') + if not isinstance(value['restart_policy'], dict): + raise ValueError('inspect restart policy') + if not isinstance(value['log_config'], dict): + raise ValueError('inspect log config') + return _ContainerState( + container_id=value['id'], image_id=value['image'], + status=value['status'], running=value['running'], + paused=value['paused'], restarting=value['restarting'], + dead=value['dead'], pid=value['pid'], + exit_code=value['exit_code'], oom_killed=value['oom_killed'], + restarts=value['restarts'], user=value['user'], + entrypoint=_optional_string_tuple(value['entrypoint']), + command=_optional_string_tuple(value['command']), + stop_timeout=value['stop_timeout'], stop_signal=value['stop_signal'], + mounts=_mount_tuple(value['mounts']), + readonly=value['readonly'], privileged=value['privileged'], + network=value['network'], pid_mode=value['pid_mode'], + ipc_mode=value['ipc_mode'], userns_mode=value['userns_mode'], + cgroupns_mode=value['cgroupns_mode'], uts_mode=value['uts_mode'], + group_add=_string_tuple(value['group_add']), + oci_runtime=value['oci_runtime'], devices=value['devices'], + device_requests=value['device_requests'], + device_cgroup_rules=value['device_cgroup_rules'], + ports=_port_map(value['ports']), + tmpfs=_string_map(value['tmpfs']), cpus=value['cpus'], + memory=value['memory'], pids_limit=value['pids_limit'], + shm_size=value['shm_size'], log_config=dict(value['log_config']), + cap_drop=_string_tuple(value['cap_drop']), + cap_add=_string_tuple(value['cap_add']), + security_opt=_string_tuple(value['security_opt']), + restart_policy=dict(value['restart_policy']), + project=value['project'], service=value['service'], + oneoff=value['oneoff'], config_hash=value['config_hash'], + config_files=value['config_files'], working_dir=value['working_dir'], + health_test=_optional_string_tuple(value['health_test']), + health=value['health'], + ) + except (KeyError, TypeError, ValueError): + raise HostLifecycleError('evidence') from None + + +class FixedDeploymentLifecycle: + def __init__(self, *, _runner=None, _clock=None, _sleep=None, _profile=None): + self._runner = _runner or _subprocess_runner + self._clock = _clock or time.monotonic + self._sleep = _sleep or time.sleep + self._profile = _profile or _deployment_profile() + self._compose_command = _compose_command(self._profile) + self._token = object() + + def _run(self, command, timeout): + try: + payload = self._runner(tuple(command), float(timeout)) + if not isinstance(payload, bytes) or len(payload) > COMMAND_OUTPUT_LIMIT: + raise HostLifecycleError('command') + return payload + except HostLifecycleError: + raise + except Exception: + raise HostLifecycleError('command') from None + + def _compose(self, *arguments, timeout=INSPECT_TIMEOUT): + return self._run((*self._compose_command, *arguments), timeout) + + def _image_id(self, image): + payload = self._run( + (DOCKER, 'image', 'inspect', '--format', '{{json .Id}}', image), + INSPECT_TIMEOUT, + ) + value = _json_payload(payload) + if type(value) is not str or _IMAGE_ID.fullmatch(value) is None: + raise HostLifecycleError('evidence') + return value + + def _resolve(self, service, timeout=INSPECT_TIMEOUT): + payload = self._compose( + 'ps', '--all', '--quiet', service, timeout=timeout, + ) + try: + lines = payload.decode('ascii').splitlines() + except UnicodeDecodeError: + raise HostLifecycleError('evidence') from None + if len(lines) != 1 or _HEX_ID.fullmatch(lines[0]) is None: + raise HostLifecycleError('evidence') + return lines[0] + + def _resolve_optional(self, service, timeout=INSPECT_TIMEOUT): + payload = self._compose( + 'ps', '--all', '--quiet', service, timeout=timeout, + ) + try: + lines = payload.decode('ascii').splitlines() + except UnicodeDecodeError: + raise HostLifecycleError('evidence') from None + if not lines: + return None + if len(lines) != 1 or _HEX_ID.fullmatch(lines[0]) is None: + raise HostLifecycleError('evidence') + return lines[0] + + def _inspect(self, container_id, timeout=INSPECT_TIMEOUT): + if _HEX_ID.fullmatch(container_id) is None: + raise HostLifecycleError('evidence') + state = _container_state(self._run( + (DOCKER, 'container', 'inspect', '--format', _INSPECT_FORMAT, + container_id), + timeout, + )) + if state.container_id != container_id: + raise HostLifecycleError('identity') + return state + + def _require_metadata( + self, state, service, image_id, runtime_id, *, config_hash=None, + ): + expected_policy = ( + {'Name': 'on-failure', 'MaximumRetryCount': 3} + if service == RUNTIME_SERVICE + else {'Name': 'unless-stopped', 'MaximumRetryCount': 0} + ) + expected_cap_add = ( + () if service == RUNTIME_SERVICE else self._profile.edge_cap_add + ) + expected_network = ( + self._profile.runtime_network + if service == RUNTIME_SERVICE else f'container:{runtime_id}' + ) + runtime = service == RUNTIME_SERVICE + expected_mounts = ( + { + ('volume', self._profile.runtime_data_volume, '/data', True), + ('bind', '/etc/truf/runtime', '/data/config', False), + ('bind', '/etc/truf/worker-packages', '/data/worker-packages', False), + ('bind', '/var/lib/truf/runtime-document-candidates', '/data/runtime-document-candidates', True), + ('bind', '/run/truf/host-agent.sock', '/run/truf/host-agent.sock', False), + ('bind', '/var/lib/truf/host-agent/results', '/data/host-agent-results', False), + ('bind', '/run/truf-postgres', '/run/truf-postgres', True), + } if runtime else { + ('volume', '/data', True), ('volume', '/config', True), + ('bind', '/var/log/caddy', True), + ('bind', '/etc/caddy/denylist', False), + } + ) + actual_mounts = ( + set(state.mounts) + if runtime else { + (mount_type, destination, writable) + for mount_type, _identity, destination, writable in state.mounts + } + ) + expected_ports = self._profile.runtime_ports if runtime else {} + expected_log = { + 'Type': 'json-file', + 'Config': {'max-size': '16m' if runtime else '8m', 'max-file': '4'}, + } + if ( + _HEX_ID.fullmatch(state.container_id) is None + or state.image_id != image_id + or _IMAGE_ID.fullmatch(state.image_id) is None + or state.user != '10001:10001' + or state.readonly is not True + or state.privileged is not False + or state.project != PROJECT + or state.service != service + or state.oneoff != 'False' + or _HEX_ID.fullmatch(state.config_hash) is None + or (config_hash is not None and state.config_hash != config_hash) + or state.config_files != ','.join(self._profile.compose_files) + or state.working_dir != PROJECT_DIRECTORY + or state.network != expected_network + or set(state.cap_drop) != {'ALL'} + or set(state.cap_add) != set(expected_cap_add) + or set(state.security_opt) != {'no-new-privileges:true'} + or state.restart_policy != expected_policy + or state.entrypoint != ( + _RUNTIME_ENTRYPOINT if runtime + else ('/usr/local/bin/truf-edge-entrypoint',) + ) + or state.command != ( + ('run', '--config', '/data/config/config.yaml') if runtime else None + ) + or state.health_test != (_RUNTIME_HEALTH_TEST if runtime else None) + or state.stop_timeout != (600 if runtime else 30) + or state.stop_signal != ('SIGTERM' if runtime else '') + or len(state.mounts) != len(expected_mounts) + or actual_mounts != expected_mounts + or state.pid_mode != '' or state.ipc_mode != 'private' + or state.userns_mode != '' or state.cgroupns_mode != 'private' + or state.uts_mode != '' or state.group_add + or state.oci_runtime != 'runc' or state.devices != 0 + or state.device_requests != 0 or state.device_cgroup_rules != 0 + or state.ports != expected_ports + or state.tmpfs != (_RUNTIME_TMPFS if runtime else _EDGE_TMPFS) + or state.cpus != ( + self._profile.runtime_cpus if runtime else 1_000_000_000 + ) + or state.memory != ( + self._profile.runtime_memory if runtime else 256 * 1024 ** 2 + ) + or state.pids_limit != (512 if runtime else 128) + or state.shm_size != (256 * 1024 ** 2 if runtime else 64 * 1024 ** 2) + or state.log_config != expected_log + ): + raise HostLifecycleError('identity') + + def _require_running( + self, state, service, image_id, runtime_id, *, fresh, config_hash=None, + ): + self._require_metadata( + state, service, image_id, runtime_id, config_hash=config_hash, + ) + if ( + state.status != 'running' + or state.running is not True + or state.paused or state.restarting or state.dead or state.oom_killed + or state.pid <= 0 + or (fresh and state.restarts != 0) + ): + raise HostLifecycleError('health') + + def _require_stopped(self, current, previous, service, runtime_id): + self._require_metadata( + current, service, previous.image_id, runtime_id, + config_hash=previous.config_hash, + ) + if ( + current.status != 'exited' + or current.running or current.paused or current.restarting + or current.dead or current.oom_killed + or current.pid != 0 or current.exit_code != 0 + or current.restarts != previous.restarts + ): + raise HostLifecycleError('stop') + + def preflight(self): + self._compose('config', '--quiet') + runtime_image = self._image_id(RUNTIME_IMAGE) + edge_image = self._image_id(EDGE_IMAGE) + runtime_id = self._resolve(RUNTIME_SERVICE) + edge_id = self._resolve(EDGE_SERVICE) + runtime = self._inspect(runtime_id) + edge = self._inspect(edge_id) + self._require_running( + runtime, RUNTIME_SERVICE, runtime_image, runtime_id, fresh=False, + ) + self._require_running( + edge, EDGE_SERVICE, edge_image, runtime_id, fresh=False, + ) + return DeploymentSnapshot(runtime, edge, runtime_image, edge_image) + + def begin_attempt(self, snapshot): + if not isinstance(snapshot, DeploymentSnapshot): + raise HostLifecycleError('state') + return _DeploymentAttempt(snapshot, self._token) + + def _require_attempt(self, attempt): + if ( + not isinstance(attempt, _DeploymentAttempt) + or attempt.token is not self._token + ): + raise HostLifecycleError('state') + return attempt + + def stop_cleanly(self, attempt): + attempt = self._require_attempt(attempt) + snapshot = attempt.snapshot + attempt.mutation_started = True + attempt.edge_stop_issued = True + attempt.phase = 'stopping-edge' + self._compose( + 'stop', '--timeout', '30', EDGE_SERVICE, + timeout=EDGE_STOP_TIMEOUT, + ) + stopped_edge = self._inspect(snapshot.edge.container_id) + self._require_stopped( + stopped_edge, snapshot.edge, EDGE_SERVICE, + snapshot.runtime.container_id, + ) + attempt.runtime_stop_issued = True + attempt.phase = 'stopping-runtime' + self._compose( + 'stop', '--timeout', '600', RUNTIME_SERVICE, + timeout=RUNTIME_STOP_TIMEOUT, + ) + stopped_runtime = self._inspect(snapshot.runtime.container_id) + self._require_stopped( + stopped_runtime, snapshot.runtime, RUNTIME_SERVICE, + snapshot.runtime.container_id, + ) + attempt.phase = 'stopped' + return attempt + + def authorize_replacement(self, attempt, operation_id): + attempt = self._require_attempt(attempt) + if attempt.phase != 'stopped': + raise HostLifecycleError('state') + snapshot = attempt.snapshot + runtime_id = snapshot.runtime.container_id + if ( + self._resolve(RUNTIME_SERVICE) != runtime_id + or self._resolve(EDGE_SERVICE) != snapshot.edge.container_id + ): + raise HostLifecycleError('identity') + self._require_stopped( + self._inspect(snapshot.edge.container_id), snapshot.edge, + EDGE_SERVICE, runtime_id, + ) + self._require_stopped( + self._inspect(runtime_id), snapshot.runtime, + RUNTIME_SERVICE, runtime_id, + ) + if ( + self._image_id(RUNTIME_IMAGE) != snapshot.runtime_image_id + or self._image_id(EDGE_IMAGE) != snapshot.edge_image_id + ): + raise HostLifecycleError('identity') + return _new_stopped_runtime_proof(operation_id, purpose='forward') + + def _remaining(self, deadline, maximum): + remaining = deadline - self._clock() + if remaining <= 0: + raise HostLifecycleError('health') + return min(float(maximum), remaining) + + def _strict_runtime_health(self, timeout): + try: + payload = self._compose( + 'exec', '-T', '--user', '10001:10001', RUNTIME_SERVICE, + '/usr/local/bin/python3', '-I', '-S', '-B', + '/opt/truf/app/container_runtime.py', 'health', + '--config', '/data/config/config.yaml', '--require-worker-api', + '--require-discovery-producers', + timeout=min(HEALTH_COMMAND_TIMEOUT, timeout), + ) + value = _json_payload(payload) + if ( + not isinstance(value, dict) + or set(value) != { + 'healthy', 'activation_state', 'postgres', 'workers', + } + or value.get('healthy') is not True + or value.get('activation_state') != 'ACTIVE' + or value.get('postgres') != 'READY' + or not isinstance(value.get('workers'), list) + or any(type(name) is not str for name in value['workers']) + or not { + 'worker-api', 'result-ingester', 'jsonl-projector', + }.issubset(value['workers']) + ): + raise HostLifecycleError('health') + except HostLifecycleError as error: + if error.category == 'health': + raise + raise HostLifecycleError('health') from None + + def _wait_runtime(self, runtime_id, image_id, config_hash): + deadline = self._clock() + RUNTIME_HEALTH_TIMEOUT + while True: + if self._resolve( + RUNTIME_SERVICE, self._remaining(deadline, INSPECT_TIMEOUT), + ) != runtime_id: + raise HostLifecycleError('identity') + current = self._inspect( + runtime_id, self._remaining(deadline, INSPECT_TIMEOUT), + ) + self._require_running( + current, RUNTIME_SERVICE, image_id, runtime_id, fresh=True, + config_hash=config_hash, + ) + if current.health == 'healthy': + try: + self._strict_runtime_health( + self._remaining(deadline, HEALTH_COMMAND_TIMEOUT), + ) + except HostLifecycleError as error: + if error.category != 'health': + raise + self._sleep(min( + POLL_SECONDS, + self._remaining(deadline, POLL_SECONDS), + )) + continue + else: + return + if current.health not in ('starting', 'unhealthy'): + raise HostLifecycleError('health') + self._sleep(min(POLL_SECONDS, self._remaining(deadline, POLL_SECONDS))) + + def _wait_edge(self, edge_id, edge_image_id, runtime_id, config_hash): + deadline = self._clock() + EDGE_VERIFY_TIMEOUT + running_deadline = self._clock() + EDGE_START_TIMEOUT + while True: + if self._resolve( + EDGE_SERVICE, self._remaining(running_deadline, INSPECT_TIMEOUT), + ) != edge_id: + raise HostLifecycleError('identity') + current = self._inspect( + edge_id, self._remaining(running_deadline, INSPECT_TIMEOUT), + ) + self._require_metadata( + current, EDGE_SERVICE, edge_image_id, runtime_id, + config_hash=config_hash, + ) + if ( + current.status == 'running' + and current.running is True + and current.paused is False + and current.restarting is False + and current.dead is False + and current.oom_killed is False + and current.pid > 0 + and current.restarts == 0 + ): + break + if ( + current.restarting or current.dead or current.oom_killed + or current.status == 'exited' + ): + raise HostLifecycleError('health') + self._sleep(min( + POLL_SECONDS, self._remaining(running_deadline, POLL_SECONDS), + )) + self._compose( + 'exec', '-T', '--user', '10001:10001', EDGE_SERVICE, + 'caddy', 'validate', '--config', self._profile.edge_caddyfile, + '--adapter', 'caddyfile', + timeout=self._remaining(deadline, EDGE_START_TIMEOUT), + ) + stable_until = self._clock() + EDGE_STABILITY_SECONDS + if stable_until > deadline: + raise HostLifecycleError('health') + while self._clock() < stable_until: + self._sleep(min(POLL_SECONDS, stable_until - self._clock())) + if self._resolve( + EDGE_SERVICE, self._remaining(deadline, INSPECT_TIMEOUT), + ) != edge_id: + raise HostLifecycleError('identity') + current = self._inspect( + edge_id, self._remaining(deadline, INSPECT_TIMEOUT), + ) + self._require_running( + current, EDGE_SERVICE, edge_image_id, runtime_id, fresh=True, + config_hash=config_hash, + ) + + def recreate_and_verify(self, attempt): + attempt = self._require_attempt(attempt) + snapshot = attempt.snapshot + runtime_id = snapshot.runtime.container_id + self._require_stopped( + self._inspect(snapshot.edge.container_id), snapshot.edge, + EDGE_SERVICE, runtime_id, + ) + self._require_stopped( + self._inspect(runtime_id), snapshot.runtime, + RUNTIME_SERVICE, runtime_id, + ) + if ( + self._image_id(RUNTIME_IMAGE) != snapshot.runtime_image_id + or self._image_id(EDGE_IMAGE) != snapshot.edge_image_id + ): + raise HostLifecycleError('identity') + self._compose('rm', '--force', EDGE_SERVICE, timeout=REMOVE_TIMEOUT) + self._compose('rm', '--force', RUNTIME_SERVICE, timeout=REMOVE_TIMEOUT) + attempt.phase = 'forward-removed' + if self._image_id(RUNTIME_IMAGE) != snapshot.runtime_image_id: + raise HostLifecycleError('identity') + self._compose( + 'up', '--detach', '--no-deps', '--no-build', '--pull', 'never', + '--force-recreate', RUNTIME_SERVICE, timeout=RECREATE_TIMEOUT, + ) + new_runtime_id = self._resolve(RUNTIME_SERVICE) + attempt.forward_runtime_id = new_runtime_id + if new_runtime_id == runtime_id: + raise HostLifecycleError('identity') + new_runtime = self._inspect(new_runtime_id) + self._require_running( + new_runtime, RUNTIME_SERVICE, snapshot.runtime_image_id, + new_runtime_id, fresh=True, config_hash=snapshot.runtime.config_hash, + ) + self._wait_runtime( + new_runtime_id, snapshot.runtime_image_id, + snapshot.runtime.config_hash, + ) + if self._image_id(EDGE_IMAGE) != snapshot.edge_image_id: + raise HostLifecycleError('identity') + self._compose( + 'up', '--detach', '--no-deps', '--no-build', '--pull', 'never', + '--force-recreate', EDGE_SERVICE, timeout=RECREATE_TIMEOUT, + ) + new_edge_id = self._resolve(EDGE_SERVICE) + attempt.forward_edge_id = new_edge_id + if new_edge_id == snapshot.edge.container_id: + raise HostLifecycleError('identity') + new_edge = self._inspect(new_edge_id) + self._require_metadata( + new_edge, EDGE_SERVICE, snapshot.edge_image_id, new_runtime_id, + ) + attempt.forward_edge_config_hash = new_edge.config_hash + self._wait_edge( + new_edge_id, snapshot.edge_image_id, new_runtime_id, + attempt.forward_edge_config_hash, + ) + attempt.phase = 'forward-healthy' + + @staticmethod + def _attempt_edge_config_hash(attempt, edge_id): + if edge_id == attempt.snapshot.edge.container_id: + return attempt.snapshot.edge.config_hash + if ( + edge_id == attempt.forward_edge_id + and _HEX_ID.fullmatch(attempt.forward_edge_config_hash or '') + ): + return attempt.forward_edge_config_hash + raise HostLifecycleError('rollback') + + def _require_quiescent(self, current, service, image_id, runtime_id, config_hash): + self._require_metadata( + current, service, image_id, runtime_id, config_hash=config_hash, + ) + if ( + current.status != 'exited' or current.running or current.paused + or current.restarting or current.dead or current.pid != 0 + ): + raise HostLifecycleError('rollback') + + def _quiesce_service( + self, service, current_id, image_id, runtime_id, config_hash, timeout, grace, + ): + if current_id is None: + return + current = self._inspect(current_id) + self._require_metadata( + current, service, image_id, runtime_id, config_hash=config_hash, + ) + if current.running: + try: + self._compose( + 'stop', '--timeout', str(grace), service, timeout=timeout, + ) + except HostLifecycleError: + pass + if self._resolve_optional(service) != current_id: + raise HostLifecycleError('rollback') + current = self._inspect(current_id) + self._require_quiescent( + current, service, image_id, runtime_id, config_hash, + ) + + def quiesce_for_rollback(self, attempt, operation_id): + attempt = self._require_attempt(attempt) + snapshot = attempt.snapshot + runtime_id = self._resolve_optional(RUNTIME_SERVICE) + edge_id = self._resolve_optional(EDGE_SERVICE) + if edge_id is not None and runtime_id is None: + raise HostLifecycleError('rollback') + edge_config_hash = ( + self._attempt_edge_config_hash(attempt, edge_id) + if edge_id is not None else None + ) + self._quiesce_service( + EDGE_SERVICE, edge_id, snapshot.edge_image_id, runtime_id, + edge_config_hash, EDGE_STOP_TIMEOUT, 30, + ) + self._quiesce_service( + RUNTIME_SERVICE, runtime_id, snapshot.runtime_image_id, runtime_id, + snapshot.runtime.config_hash, RUNTIME_STOP_TIMEOUT, 600, + ) + if ( + self._image_id(RUNTIME_IMAGE) != snapshot.runtime_image_id + or self._image_id(EDGE_IMAGE) != snapshot.edge_image_id + ): + raise HostLifecycleError('rollback') + attempt.phase = 'rollback-quiescent' + return _new_stopped_runtime_proof(operation_id, purpose='rollback') + + def recreate_restored_and_verify(self, attempt): + attempt = self._require_attempt(attempt) + if attempt.phase != 'rollback-quiescent': + raise HostLifecycleError('state') + snapshot = attempt.snapshot + edge_id = self._resolve_optional(EDGE_SERVICE) + runtime_id = self._resolve_optional(RUNTIME_SERVICE) + if edge_id is not None: + self._require_quiescent( + self._inspect(edge_id), EDGE_SERVICE, snapshot.edge_image_id, + runtime_id, self._attempt_edge_config_hash(attempt, edge_id), + ) + self._compose('rm', '--force', EDGE_SERVICE, timeout=REMOVE_TIMEOUT) + if runtime_id is not None: + self._require_quiescent( + self._inspect(runtime_id), RUNTIME_SERVICE, + snapshot.runtime_image_id, runtime_id, + snapshot.runtime.config_hash, + ) + self._compose('rm', '--force', RUNTIME_SERVICE, timeout=REMOVE_TIMEOUT) + if self._image_id(RUNTIME_IMAGE) != snapshot.runtime_image_id: + raise HostLifecycleError('rollback') + self._compose( + 'up', '--detach', '--no-deps', '--no-build', '--pull', 'never', + '--force-recreate', RUNTIME_SERVICE, timeout=RECREATE_TIMEOUT, + ) + restored_runtime_id = self._resolve(RUNTIME_SERVICE) + if restored_runtime_id in { + snapshot.runtime.container_id, attempt.forward_runtime_id, + }: + raise HostLifecycleError('rollback') + restored_runtime = self._inspect(restored_runtime_id) + self._require_running( + restored_runtime, RUNTIME_SERVICE, snapshot.runtime_image_id, + restored_runtime_id, fresh=True, + config_hash=snapshot.runtime.config_hash, + ) + self._wait_runtime( + restored_runtime_id, snapshot.runtime_image_id, + snapshot.runtime.config_hash, + ) + if self._image_id(EDGE_IMAGE) != snapshot.edge_image_id: + raise HostLifecycleError('rollback') + self._compose( + 'up', '--detach', '--no-deps', '--no-build', '--pull', 'never', + '--force-recreate', EDGE_SERVICE, timeout=RECREATE_TIMEOUT, + ) + restored_edge_id = self._resolve(EDGE_SERVICE) + if restored_edge_id in {snapshot.edge.container_id, attempt.forward_edge_id}: + raise HostLifecycleError('rollback') + restored_edge = self._inspect(restored_edge_id) + self._require_metadata( + restored_edge, EDGE_SERVICE, snapshot.edge_image_id, + restored_runtime_id, + ) + self._wait_edge( + restored_edge_id, snapshot.edge_image_id, restored_runtime_id, + restored_edge.config_hash, + ) + attempt.phase = 'rollback-healthy' + + def contain_for_failed_hold(self, attempt, operation_id): + try: + self.quiesce_for_rollback(attempt, operation_id) + attempt.phase = 'failed-hold-contained' + return True + except Exception: + attempt.phase = 'failed-hold-uncontained' + return False + + +def execute_fixed_forward(session, lifecycle=None): + if getattr(session, '_entered', False) is not True: + raise HostLifecycleError('state') + lifecycle = lifecycle or FixedDeploymentLifecycle() + session.backup() + snapshot = lifecycle.preflight() + attempt = lifecycle.begin_attempt(snapshot) + session.revalidate_for_stop() + lifecycle.stop_cleanly(attempt) + proof = lifecycle.authorize_replacement( + attempt, session.request.operation_id, + ) + resulting_identity = session.replace(proof) + lifecycle.recreate_and_verify(attempt) + return resulting_identity + + +def _forward_failure(request, error): + category = getattr(error, 'category', '') + if category == 'health': + return 'health_check_failed', 'health_check_failed' + if request.action.value == 'restart': + return 'restart_failed', 'restart_failed' + return 'apply_failed', 'apply_failed' + + +def _preflight_failure(error): + category = getattr(error, 'category', '') + if category == 'validation': + return 'validation_failed', 'validation_failed' + if category in ('identity', 'partial'): + return 'operation_conflict', 'bounded_result' + return 'apply_failed', 'apply_failed' + + +def _terminal_result_for_phase(session, phase): + name = phase['phase'] + if name == 'succeeded': + identity = { + 'active_config_sha256': ( + session.request.candidate_config_sha256 + or session.request.active_config_sha256 + ), + 'active_secrets_sha256': ( + session.request.candidate_secrets_sha256 + or session.request.active_secrets_sha256 + ), + } + return name, None, None, identity + if name in {'failed', 'rolled_back'}: + category = phase.get('forward_category') or 'apply_failed' + detail = phase.get('safe_detail') or category + return name, category, detail, session.original_identity() + if name == 'failed_hold': + return name, 'rollback_failed', 'rollback_failed', None + raise HostLifecycleError('state') + + +def execute_fixed_operation(session, lifecycle=None, state=None): + """Run one forward attempt and at most one fixed rollback attempt.""" + if getattr(session, '_entered', False) is not True: + raise HostLifecycleError('state') + lifecycle = lifecycle or FixedDeploymentLifecycle() + state = state or HostOperationState(session.request) + try: + terminal = state.terminal_result() + if terminal is not None: + return terminal + phase = state.initialize(session.publication_state) + except HostStateError: + raise HostLifecycleError('state') from None + if getattr(session, '_failed_hold_replay', False): + if phase['phase'] == 'forward_started': + try: + phase = state.advance( + 'forward_started', 'rollback_started', + phase['publication_state'], + forward_category=phase.get('forward_category'), + safe_detail=phase.get('safe_detail'), + ) + except HostStateError: + raise HostLifecycleError('state') from None + if phase['phase'] == 'rollback_started': + try: + phase = state.advance( + 'rollback_started', 'failed_hold', + phase['publication_state'], + forward_category=phase.get('forward_category'), + safe_detail='rollback_failed', + containment_confirmed=False, + ) + except HostStateError: + raise HostLifecycleError('state') from None + elif phase['phase'] != 'failed_hold': + raise HostLifecycleError('state') + if phase['phase'] in {'succeeded', 'failed', 'rolled_back', 'failed_hold'}: + result, category, detail, identity = _terminal_result_for_phase( + session, phase, + ) + try: + return state.publish_result( + result, safe_category=category, safe_detail=detail, + resulting_identity=identity, + ) + except HostStateError: + raise HostLifecycleError('state') from None + + current_phase = phase['phase'] + attempt = None + cancellation = None + rollback_required = current_phase in {'forward_started', 'rollback_started'} + rollback_required = rollback_required or session.publication_state != 'original' + forward_category = phase.get('forward_category') + if rollback_required and forward_category is None: + forward_category = ( + 'restart_failed' + if session.request.action.value == 'restart' else 'apply_failed' + ) + + if not rollback_required: + try: + session.backup() + snapshot = lifecycle.preflight() + attempt = lifecycle.begin_attempt(snapshot) + session.revalidate_for_stop() + state.advance( + 'prepared', 'forward_started', session.publication_state, + ) + current_phase = 'forward_started' + lifecycle.stop_cleanly(attempt) + proof = lifecycle.authorize_replacement( + attempt, session.request.operation_id, + ) + resulting_identity = session.replace(proof) + lifecycle.recreate_and_verify(attempt) + state.advance( + 'forward_started', 'succeeded', session.publication_state, + ) + current_phase = 'succeeded' + result = state.publish_result( + 'succeeded', safe_category=None, safe_detail=None, + resulting_identity=resulting_identity, + ) + return result + except BaseException as error: + is_cancellation = not isinstance(error, Exception) + if current_phase == 'succeeded': + if is_cancellation: + raise + raise HostLifecycleError('state') from None + if ( + isinstance(error, HostStateError) + and error.category == 'uncertain' + ): + if error.cancellation is not None: + raise error.cancellation + raise HostLifecycleError('state') from None + if attempt is None or not attempt.mutation_started: + if is_cancellation: + raise + category, detail = _preflight_failure(error) + try: + state.advance( + 'prepared', 'failed', session.publication_state, + forward_category=category, safe_detail=detail, + ) + current_phase = 'failed' + result = state.publish_result( + 'failed', safe_category=category, safe_detail=detail, + resulting_identity=session.original_identity(), + ) + return result + except HostStateError: + raise HostLifecycleError('state') from None + rollback_required = True + if is_cancellation: + cancellation = error + forward_category, _ = _forward_failure(session.request, error) + + if attempt is None: + try: + session.backup() + snapshot = lifecycle.preflight() + attempt = lifecycle.begin_attempt(snapshot) + except BaseException as error: + if not isinstance(error, Exception) and not rollback_required: + raise + category, detail = _preflight_failure(error) + try: + if current_phase != 'rollback_started': + state.advance( + current_phase, 'rollback_started', + session.publication_state, + forward_category=forward_category or category, + safe_detail=detail, + ) + current_phase = 'rollback_started' + state.publish_failed_hold( + forward_category=forward_category or category, + publication_state=session.publication_state, + containment_confirmed=False, + ) + state.advance( + 'rollback_started', 'failed_hold', + session.publication_state, + forward_category=forward_category or category, + safe_detail='rollback_failed', + containment_confirmed=False, + ) + current_phase = 'failed_hold' + result = state.publish_result( + 'failed_hold', safe_category='rollback_failed', + safe_detail='rollback_failed', resulting_identity=None, + ) + if not isinstance(error, Exception): + raise error + return result + except HostStateError: + if not isinstance(error, Exception): + raise error + raise HostLifecycleError('state') from None + + rollback_terminal_transition = False + try: + if current_phase != 'rollback_started': + state.advance( + current_phase, 'rollback_started', session.publication_state, + forward_category=forward_category, safe_detail=forward_category, + ) + current_phase = 'rollback_started' + proof = lifecycle.quiesce_for_rollback( + attempt, session.request.operation_id, + ) + original_identity = session.restore_backups(proof) + lifecycle.recreate_restored_and_verify(attempt) + detail = forward_category + rollback_terminal_transition = True + state.advance( + 'rollback_started', 'rolled_back', session.publication_state, + forward_category=forward_category, safe_detail=detail, + ) + current_phase = 'rolled_back' + result = state.publish_result( + 'rolled_back', safe_category=forward_category, safe_detail=detail, + resulting_identity=original_identity, + ) + if cancellation is not None: + raise cancellation + return result + except BaseException as error: + if current_phase == 'rolled_back': + if cancellation is not None: + raise cancellation + if not isinstance(error, Exception): + raise + raise HostLifecycleError('state') from None + if ( + isinstance(error, HostStateError) + and error.category == 'uncertain' + and rollback_terminal_transition + ): + if error.cancellation is not None: + raise error.cancellation + raise HostLifecycleError('state') from None + if ( + isinstance(error, HostStateError) + and error.category == 'uncertain' + and error.cancellation is not None + ): + cancellation = error.cancellation + if not isinstance(error, Exception): + cancellation = error + try: + state.publish_failed_hold( + forward_category=forward_category, + publication_state=session.publication_state, + containment_confirmed=False, + ) + except HostStateError: + if cancellation is not None: + raise cancellation + raise HostLifecycleError('state') from None + contained = lifecycle.contain_for_failed_hold( + attempt, session.request.operation_id, + ) + try: + state.advance( + 'rollback_started', 'failed_hold', session.publication_state, + forward_category=forward_category, safe_detail='rollback_failed', + containment_confirmed=contained, + ) + current_phase = 'failed_hold' + result = state.publish_result( + 'failed_hold', safe_category='rollback_failed', + safe_detail='rollback_failed', resulting_identity=None, + ) + if cancellation is not None: + raise cancellation + return result + except HostStateError: + if cancellation is not None: + raise cancellation + raise HostLifecycleError('state') from None diff --git a/app/host_agent_protocol.py b/app/host_agent_protocol.py new file mode 100644 index 0000000..771e88c --- /dev/null +++ b/app/host_agent_protocol.py @@ -0,0 +1,315 @@ +import hmac +import json +import re +import struct +import time +import uuid +from dataclasses import dataclass +from enum import Enum + + +HOST_AGENT_SOCKET_PATH = '/run/truf/host-agent.sock' +HOST_AGENT_RUNTIME_UID = 10001 +MAX_REQUEST_PAYLOAD_BYTES = 1024 +MAX_RESPONSE_PAYLOAD_BYTES = 256 +CLIENT_CONNECT_TIMEOUT_SECONDS = 1.0 +SERVER_READ_TIMEOUT_SECONDS = 2.0 +EXCHANGE_TIMEOUT_SECONDS = 5.0 + +_FRAME_HEADER_BYTES = 4 +_SHA256_RE = re.compile(r'^[0-9a-f]{64}$') +_REQUEST_FIELDS = frozenset(( + 'operation_id', 'action', + 'active_config_sha256', 'active_secrets_sha256', + 'candidate_config_sha256', 'candidate_secrets_sha256', +)) +_RESPONSE_FIELDS = frozenset(('operation_id', 'status')) + + +class HostAgentProtocolError(ValueError): + def __init__(self, category): + self.category = category + super().__init__('host agent protocol message is invalid') + + +class HostAgentAction(str, Enum): + APPLY_CONFIG = 'apply-config' + APPLY_SECRETS = 'apply-secrets' + APPLY_BOTH = 'apply-both' + RESTART = 'restart' + + +class HostAgentStatus(str, Enum): + ACCEPTED = 'accepted' + UNAVAILABLE = 'unavailable' + REJECTED = 'rejected' + INVALID = 'invalid' + + +@dataclass(frozen=True, slots=True) +class HostAgentRequest: + operation_id: str + action: HostAgentAction + active_config_sha256: str + active_secrets_sha256: str + candidate_config_sha256: str | None + candidate_secrets_sha256: str | None + + +@dataclass(frozen=True, slots=True) +class HostAgentResponse: + operation_id: str | None + status: HostAgentStatus + + +def _canonical_uuid(value): + if not isinstance(value, str) or not value: + raise HostAgentProtocolError('operation_id') + try: + parsed = uuid.UUID(value) + except (ValueError, AttributeError) as exc: + raise HostAgentProtocolError('operation_id') from exc + if parsed.int == 0 or str(parsed) != value: + raise HostAgentProtocolError('operation_id') + return value + + +def _sha256(value, field, *, optional=False): + if optional and value is None: + return None + if not isinstance(value, str) or _SHA256_RE.fullmatch(value) is None: + raise HostAgentProtocolError(field) + return value + + +def _canonical_json(value): + try: + return json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + allow_nan=False, + ).encode('ascii') + except (TypeError, ValueError, UnicodeError) as exc: + raise HostAgentProtocolError('json') from exc + + +def _strict_json(payload, *, maximum): + if type(payload) is not bytes or not 1 <= len(payload) <= maximum: + raise HostAgentProtocolError('bounds') + + def reject_duplicate(pairs): + result = {} + for key, value in pairs: + if key in result: + raise HostAgentProtocolError('duplicate_field') + result[key] = value + return result + + try: + text = payload.decode('utf-8', errors='strict') + value = json.loads( + text, object_pairs_hook=reject_duplicate, + parse_constant=lambda _value: (_ for _ in ()).throw( + HostAgentProtocolError('constant') + ), + ) + except HostAgentProtocolError: + raise + except (UnicodeDecodeError, json.JSONDecodeError, TypeError, ValueError) as exc: + raise HostAgentProtocolError('json') from exc + finally: + text = None + if not isinstance(value, dict): + raise HostAgentProtocolError('shape') + if not hmac.compare_digest(_canonical_json(value), payload): + raise HostAgentProtocolError('canonical') + return value + + +def _normalize_request(value): + if not isinstance(value, dict) or set(value) != _REQUEST_FIELDS: + raise HostAgentProtocolError('shape') + try: + action = HostAgentAction(value.get('action')) + except (TypeError, ValueError) as exc: + raise HostAgentProtocolError('action') from exc + active_config = _sha256(value.get('active_config_sha256'), 'active_config_sha256') + active_secrets = _sha256(value.get('active_secrets_sha256'), 'active_secrets_sha256') + candidate_config = _sha256( + value.get('candidate_config_sha256'), 'candidate_config_sha256', optional=True, + ) + candidate_secrets = _sha256( + value.get('candidate_secrets_sha256'), 'candidate_secrets_sha256', optional=True, + ) + required = { + HostAgentAction.APPLY_CONFIG: (True, False), + HostAgentAction.APPLY_SECRETS: (False, True), + HostAgentAction.APPLY_BOTH: (True, True), + HostAgentAction.RESTART: (False, False), + }[action] + if (candidate_config is not None, candidate_secrets is not None) != required: + raise HostAgentProtocolError('candidate_identity') + return HostAgentRequest( + operation_id=_canonical_uuid(value.get('operation_id')), + action=action, + active_config_sha256=active_config, + active_secrets_sha256=active_secrets, + candidate_config_sha256=candidate_config, + candidate_secrets_sha256=candidate_secrets, + ) + + +def _request_value(request): + if not isinstance(request, HostAgentRequest): + raise HostAgentProtocolError('request_type') + return { + 'operation_id': request.operation_id, + 'action': request.action.value if isinstance(request.action, HostAgentAction) else request.action, + 'active_config_sha256': request.active_config_sha256, + 'active_secrets_sha256': request.active_secrets_sha256, + 'candidate_config_sha256': request.candidate_config_sha256, + 'candidate_secrets_sha256': request.candidate_secrets_sha256, + } + + +def encode_request_payload(request): + normalized = _normalize_request(_request_value(request)) + payload = _canonical_json(_request_value(normalized)) + if len(payload) > MAX_REQUEST_PAYLOAD_BYTES: + raise HostAgentProtocolError('bounds') + return payload + + +def decode_request_payload(payload): + return _normalize_request(_strict_json(payload, maximum=MAX_REQUEST_PAYLOAD_BYTES)) + + +def _normalize_response(value): + if not isinstance(value, dict) or set(value) != _RESPONSE_FIELDS: + raise HostAgentProtocolError('shape') + try: + status = HostAgentStatus(value.get('status')) + except (TypeError, ValueError) as exc: + raise HostAgentProtocolError('status') from exc + operation_id = value.get('operation_id') + if status is HostAgentStatus.INVALID: + if operation_id is not None: + raise HostAgentProtocolError('operation_id') + else: + operation_id = _canonical_uuid(operation_id) + return HostAgentResponse(operation_id=operation_id, status=status) + + +def _response_value(response): + if not isinstance(response, HostAgentResponse): + raise HostAgentProtocolError('response_type') + return { + 'operation_id': response.operation_id, + 'status': response.status.value if isinstance(response.status, HostAgentStatus) else response.status, + } + + +def encode_response_payload(response): + normalized = _normalize_response(_response_value(response)) + payload = _canonical_json(_response_value(normalized)) + if len(payload) > MAX_RESPONSE_PAYLOAD_BYTES: + raise HostAgentProtocolError('bounds') + return payload + + +def decode_response_payload(payload): + return _normalize_response(_strict_json(payload, maximum=MAX_RESPONSE_PAYLOAD_BYTES)) + + +def _encode_frame(payload, maximum): + if type(payload) is not bytes or not 1 <= len(payload) <= maximum: + raise HostAgentProtocolError('bounds') + return struct.pack('!I', len(payload)) + payload + + +def _decode_frame(frame, maximum): + if type(frame) is not bytes or len(frame) < _FRAME_HEADER_BYTES: + raise HostAgentProtocolError('frame') + length = struct.unpack('!I', frame[:_FRAME_HEADER_BYTES])[0] + if not 1 <= length <= maximum or len(frame) != _FRAME_HEADER_BYTES + length: + raise HostAgentProtocolError('frame') + return frame[_FRAME_HEADER_BYTES:] + + +def encode_request_frame(request): + return _encode_frame(encode_request_payload(request), MAX_REQUEST_PAYLOAD_BYTES) + + +def decode_request_frame(frame): + return decode_request_payload(_decode_frame(frame, MAX_REQUEST_PAYLOAD_BYTES)) + + +def encode_response_frame(response): + return _encode_frame(encode_response_payload(response), MAX_RESPONSE_PAYLOAD_BYTES) + + +def decode_response_frame(frame): + return decode_response_payload(_decode_frame(frame, MAX_RESPONSE_PAYLOAD_BYTES)) + + +def _remaining(deadline): + remaining = deadline - time.monotonic() + if remaining <= 0: + raise HostAgentProtocolError('timeout') + return remaining + + +def receive_frame(sock, *, maximum, deadline): + header = _receive_exact(sock, _FRAME_HEADER_BYTES, deadline) + length = struct.unpack('!I', header)[0] + if not 1 <= length <= maximum: + raise HostAgentProtocolError('bounds') + return header + _receive_exact(sock, length, deadline) + + +def _receive_exact(sock, length, deadline): + chunks = bytearray() + try: + while len(chunks) < length: + sock.settimeout(_remaining(deadline)) + chunk = sock.recv(length - len(chunks)) + if not chunk: + raise HostAgentProtocolError('truncated') + chunks.extend(chunk) + return bytes(chunks) + except HostAgentProtocolError: + raise + except (OSError, TimeoutError) as exc: + raise HostAgentProtocolError('transport') from exc + finally: + chunks.clear() + chunk = None + + +def require_eof(sock, *, deadline): + try: + sock.settimeout(_remaining(deadline)) + if sock.recv(1): + raise HostAgentProtocolError('trailing_data') + except HostAgentProtocolError: + raise + except (OSError, TimeoutError) as exc: + raise HostAgentProtocolError('transport') from exc + + +def send_frame(sock, frame, *, deadline): + if type(frame) is not bytes: + raise HostAgentProtocolError('frame') + view = memoryview(frame) + try: + while view: + sock.settimeout(_remaining(deadline)) + sent = sock.send(view) + if sent <= 0: + raise HostAgentProtocolError('transport') + view = view[sent:] + except HostAgentProtocolError: + raise + except (OSError, TimeoutError) as exc: + raise HostAgentProtocolError('transport') from exc + finally: + view.release() diff --git a/app/host_agent_reconcile.py b/app/host_agent_reconcile.py new file mode 100644 index 0000000..7ba2d84 --- /dev/null +++ b/app/host_agent_reconcile.py @@ -0,0 +1,94 @@ +"""Bounded runtime reconciliation of fixed host-agent result evidence.""" + +import json +import os +from pathlib import Path +import stat + +from host_agent_state import MAX_STATE_BYTES +from runtime_security import reject_reparse_components + + +HOST_RESULT_DIRECTORY = Path('/data/host-agent-results') +HOST_ROOT_UID = 0 +HOST_RUNTIME_GID = 10001 + + +class HostResultError(RuntimeError): + pass + + +def fixed_result_directory_is_safe(): + try: + reject_reparse_components(HOST_RESULT_DIRECTORY) + details = os.stat(HOST_RESULT_DIRECTORY, follow_symlinks=False) + return ( + stat.S_ISDIR(details.st_mode) + and details.st_uid == HOST_ROOT_UID + and details.st_gid == HOST_RUNTIME_GID + and stat.S_IMODE(details.st_mode) == 0o750 + ) + except Exception: + return False + + +def _read_result(operation_id): + path = HOST_RESULT_DIRECTORY / f'{operation_id}.json' + descriptor = None + try: + flags = os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) | getattr(os, 'O_NOFOLLOW', 0) + descriptor = os.open(path, flags) + before = os.fstat(descriptor) + if ( + not stat.S_ISREG(before.st_mode) + or before.st_nlink != 1 + or before.st_uid != HOST_ROOT_UID + or before.st_gid != HOST_RUNTIME_GID + or stat.S_IMODE(before.st_mode) != 0o640 + ): + raise HostResultError('host result metadata is invalid') + with os.fdopen(descriptor, 'rb') as handle: + descriptor = None + payload = handle.read(MAX_STATE_BYTES + 1) + after = os.fstat(handle.fileno()) + current = os.stat(path, follow_symlinks=False) + identity = lambda item: ( + item.st_dev, item.st_ino, item.st_size, + getattr(item, 'st_mtime_ns', None), getattr(item, 'st_ctime_ns', None), + ) + if ( + len(payload) > MAX_STATE_BYTES + or identity(before) != identity(after) + or identity(after) != identity(current) + ): + raise HostResultError('host result changed during read') + value = json.loads(payload.decode('ascii')) + canonical = json.dumps( + value, sort_keys=True, separators=(',', ':'), ensure_ascii=True, + allow_nan=False, + ).encode('ascii') + if canonical != payload: + raise HostResultError('host result is not canonical') + return payload + except FileNotFoundError: + return None + except HostResultError: + raise + except Exception: + raise HostResultError('host result is invalid') from None + finally: + if descriptor is not None: + os.close(descriptor) + + +def reconcile_pending_host_results(database, *, limit=32): + if not fixed_result_directory_is_safe(): + return 0 + reconciled = 0 + for operation in database.pending_runtime_agent_operations(limit=limit): + envelope = _read_result(operation['operation_id']) + if envelope is None: + continue + if database.reconcile_runtime_operation_result(envelope) is not None: + reconciled += 1 + return reconciled diff --git a/app/host_agent_runtime.py b/app/host_agent_runtime.py new file mode 100644 index 0000000..8e108b7 --- /dev/null +++ b/app/host_agent_runtime.py @@ -0,0 +1,214 @@ +"""Fixed production authority and asynchronous host-operation dispatch.""" + +import hmac +from pathlib import Path, PurePosixPath +import threading + +from host_agent_apply import HostApplyError, HostApplySession +from host_agent_lifecycle import execute_fixed_operation +from host_agent_protocol import HostAgentStatus, encode_request_payload +from host_agent_state import HostOperationState +from runtime_document import ( + MAX_CONFIG_DOCUMENT_BYTES, + _resolve_package_manifest_path, + load_yaml_document, +) +from runtime_security import read_stable_root_file +from scanner_db import ScannerDB +from worker_package import ( + MAX_WORKER_PACKAGE_MANIFEST_BYTES, + load_worker_package_manifest_bytes, +) + + +HOST_RUNTIME_ACTIVE_ROOT = Path('/etc/truf/runtime') +HOST_WORKER_PACKAGE_ROOT = Path('/etc/truf/worker-packages') + + +class HostRuntimeError(RuntimeError): + def __init__(self, category): + self.category = str(category) + super().__init__('host operation runtime failed') + + +def _stable_root_file(path, maximum): + try: + return read_stable_root_file(path, maximum, HOST_WORKER_PACKAGE_ROOT) + except Exception: + raise HostRuntimeError('package_evidence') from None + + +def _host_manifest_path(resolved): + value = PurePosixPath(resolved) + try: + relative = value.relative_to(PurePosixPath('/data/worker-packages')) + except ValueError: + raise HostRuntimeError('package_evidence') from None + if not relative.parts or any(part in ('', '.', '..') for part in relative.parts): + raise HostRuntimeError('package_evidence') + return HOST_WORKER_PACKAGE_ROOT.joinpath(*relative.parts) + + +def load_fixed_package_capabilities(config_payload): + config = load_yaml_document( + config_payload, max_bytes=MAX_CONFIG_DOCUMENT_BYTES, + ) + try: + profiles = config['supervisor']['worker_api']['compatibility_profiles'] + except (KeyError, TypeError): + raise HostRuntimeError('package_evidence') from None + if type(profiles) is not dict: + raise HostRuntimeError('package_evidence') + evidence = {} + try: + for profile_name, profile in profiles.items(): + if type(profile_name) is not str or type(profile) is not dict: + raise HostRuntimeError('package_evidence') + reference = profile.get('package_manifest') + resolved = _resolve_package_manifest_path(config, reference) + if resolved is None: + raise HostRuntimeError('package_evidence') + payload = _stable_root_file( + _host_manifest_path(resolved), MAX_WORKER_PACKAGE_MANIFEST_BYTES, + ) + manifest = load_worker_package_manifest_bytes(payload) + evidence[profile_name] = { + 'package_manifest': reference, + 'capabilities': manifest['capabilities'], + } + return evidence + except HostRuntimeError: + raise + except Exception: + raise HostRuntimeError('package_evidence') from None + finally: + config = profiles = profile_name = profile = reference = None + resolved = payload = manifest = None + + +class FixedHostOperationDispatcher: + def __init__(self): + self._guard = threading.Lock() + self._active_request = None + self._worker = None + self._closing = False + + def _execute(self, session, database, state): + try: + try: + execute_fixed_operation(session, state=state) + except BaseException: + # The durable executor owns safety/result handling. Do not let + # thread tracebacks disclose host details at this outer boundary. + pass + finally: + try: + session.close() + except BaseException: + pass + try: + database.close() + except BaseException: + pass + finally: + with self._guard: + self._active_request = None + self._worker = None + + @staticmethod + def _record_validation_failure(state, request): + try: + phase = state.initialize('original') + if phase.get('phase') == 'prepared': + if phase.get('publication_state') != 'original': + return False + phase = state.advance( + 'prepared', 'failed', 'original', + forward_category='validation_failed', + safe_detail='validation_failed', + ) + if ( + phase.get('phase') != 'failed' + or phase.get('publication_state') != 'original' + or phase.get('forward_category') != 'validation_failed' + or phase.get('safe_detail') != 'validation_failed' + ): + return False + state.publish_result( + 'failed', + safe_category='validation_failed', + safe_detail='validation_failed', + resulting_identity={ + 'active_config_sha256': request.active_config_sha256, + 'active_secrets_sha256': request.active_secrets_sha256, + }, + ) + return True + except Exception: + return False + + def handle(self, request): + encoded = encode_request_payload(request) + with self._guard: + if self._closing: + return HostAgentStatus.UNAVAILABLE + if self._active_request is not None: + return ( + HostAgentStatus.ACCEPTED + if hmac.compare_digest(encoded, self._active_request) + else HostAgentStatus.REJECTED + ) + state = HostOperationState(request) + if state.terminal_result() is not None: + return HostAgentStatus.ACCEPTED + database = None + session = None + try: + database = ScannerDB.host_agent_authority() + session = HostApplySession( + request, database, + package_capability_provider=load_fixed_package_capabilities, + ) + session.__enter__() + state.initialize(session.publication_state) + worker = threading.Thread( + target=self._execute, + args=(session, database, state), + name='truf-host-operation', + daemon=False, + ) + self._active_request = encoded + self._worker = worker + worker.start() + except Exception as error: + accepted = ( + isinstance(error, HostApplyError) + and error.category == 'validation' + and session is not None + and session.claim is not None + and self._record_validation_failure(state, request) + ) + self._active_request = None + self._worker = None + if session is not None: + try: + session.close() + except BaseException: + pass + if database is not None: + try: + database.close() + except BaseException: + pass + return ( + HostAgentStatus.ACCEPTED + if accepted else HostAgentStatus.UNAVAILABLE + ) + return HostAgentStatus.ACCEPTED + + def close(self): + with self._guard: + self._closing = True + worker = self._worker + if worker is not None: + worker.join() diff --git a/app/host_agent_server.py b/app/host_agent_server.py new file mode 100644 index 0000000..59b5243 --- /dev/null +++ b/app/host_agent_server.py @@ -0,0 +1,167 @@ +import os +import socket +import stat +import struct +import time + +from host_agent_protocol import ( + EXCHANGE_TIMEOUT_SECONDS, + HOST_AGENT_RUNTIME_UID, + HOST_AGENT_SOCKET_PATH, + MAX_REQUEST_PAYLOAD_BYTES, + SERVER_READ_TIMEOUT_SECONDS, + HostAgentProtocolError, + HostAgentResponse, + HostAgentStatus, + decode_request_frame, + encode_response_frame, + receive_frame, + require_eof, + send_frame, +) + + +SYSTEMD_LISTEN_FD = 3 +ACCEPT_POLL_SECONDS = 1.0 + + +class HostAgentServerError(RuntimeError): + def __init__(self, category): + self.category = category + super().__init__('host operations agent server failed') + + +def _peer_credentials(connection): + if not hasattr(socket, 'SO_PEERCRED'): + raise HostAgentServerError('peer_credentials_unavailable') + try: + raw = connection.getsockopt( + socket.SOL_SOCKET, socket.SO_PEERCRED, struct.calcsize('3i'), + ) + pid, uid, gid = struct.unpack('3i', raw) + except (OSError, struct.error) as exc: + raise HostAgentServerError('peer_credentials_unavailable') from exc + return pid, uid, gid + + +def unavailable_handler(_request): + return HostAgentStatus.UNAVAILABLE + + +def serve_connection(connection, *, handler=unavailable_handler, accepted_at=None): + if not isinstance(connection, socket.socket): + raise HostAgentServerError('connection_invalid') + started = time.monotonic() if accepted_at is None else accepted_at + try: + peer_pid, peer_uid, _peer_gid = _peer_credentials(connection) + except HostAgentServerError: + return False + if peer_pid <= 0 or peer_uid != HOST_AGENT_RUNTIME_UID: + return False + + request = None + response = None + try: + read_deadline = min( + started + SERVER_READ_TIMEOUT_SECONDS, + started + EXCHANGE_TIMEOUT_SECONDS, + ) + frame = receive_frame( + connection, maximum=MAX_REQUEST_PAYLOAD_BYTES, + deadline=read_deadline, + ) + require_eof(connection, deadline=read_deadline) + request = decode_request_frame(frame) + except HostAgentProtocolError: + response = HostAgentResponse(None, HostAgentStatus.INVALID) + else: + try: + outcome = handler(request) + if isinstance(outcome, HostAgentResponse): + response = outcome + else: + response = HostAgentResponse( + request.operation_id, HostAgentStatus(outcome), + ) + if response.operation_id != request.operation_id: + raise HostAgentServerError('handler_identity_invalid') + except BaseException as exc: + if not isinstance(exc, Exception): + raise + response = HostAgentResponse( + request.operation_id, HostAgentStatus.UNAVAILABLE, + ) + try: + send_frame( + connection, encode_response_frame(response), + deadline=started + EXCHANGE_TIMEOUT_SECONDS, + ) + except HostAgentProtocolError: + return False + finally: + frame = request = response = outcome = None + return True + + +def _validate_listener(listener): + if not isinstance(listener, socket.socket): + raise HostAgentServerError('listener_invalid') + unix_family = getattr(socket, 'AF_UNIX', None) + if unix_family is None: + raise HostAgentServerError('listener_invalid') + try: + socket_type = listener.getsockopt(socket.SOL_SOCKET, socket.SO_TYPE) + accepting = listener.getsockopt(socket.SOL_SOCKET, socket.SO_ACCEPTCONN) + except OSError as exc: + raise HostAgentServerError('listener_invalid') from exc + if ( + listener.family != unix_family + or socket_type != socket.SOCK_STREAM + or accepting != 1 + ): + raise HostAgentServerError('listener_invalid') + try: + if listener.getsockname() != HOST_AGENT_SOCKET_PATH: + raise HostAgentServerError('listener_path_invalid') + details = os.lstat(HOST_AGENT_SOCKET_PATH) + except HostAgentServerError: + raise + except OSError as exc: + raise HostAgentServerError('listener_unavailable') from exc + if not stat.S_ISSOCK(details.st_mode) or details.st_uid != 0: + raise HostAgentServerError('listener_owner_invalid') + + +def inherited_systemd_listener(): + geteuid = getattr(os, 'geteuid', None) + if geteuid is None or geteuid() != 0: + raise HostAgentServerError('root_required') + if os.environ.get('LISTEN_PID') != str(os.getpid()): + raise HostAgentServerError('socket_activation_invalid') + if os.environ.get('LISTEN_FDS') != '1': + raise HostAgentServerError('socket_activation_invalid') + try: + listener = socket.socket(fileno=SYSTEMD_LISTEN_FD) + _validate_listener(listener) + except BaseException: + try: + listener.close() + except (OSError, UnboundLocalError): + pass + raise + return listener + + +def serve_forever(listener, *, handler=unavailable_handler, stop_event=None): + _validate_listener(listener) + if stop_event is not None: + listener.settimeout(ACCEPT_POLL_SECONDS) + while stop_event is None or not stop_event.is_set(): + try: + connection, _address = listener.accept() + except InterruptedError: + continue + except TimeoutError: + continue + with connection: + serve_connection(connection, handler=handler) diff --git a/app/host_agent_state.py b/app/host_agent_state.py new file mode 100644 index 0000000..32e7bd8 --- /dev/null +++ b/app/host_agent_state.py @@ -0,0 +1,412 @@ +"""Durable fixed-path evidence for privileged runtime operations.""" + +import hashlib +import json +import os +from pathlib import Path +import stat + +from host_agent_protocol import decode_request_payload, encode_request_payload +from runtime_security import fsync_directory, reject_reparse_components + + +HOST_ROOT_UID = 0 +HOST_ROOT_GID = 0 +HOST_RUNTIME_GID = 10001 + +HOST_STATE_ROOT = Path('/var/lib/truf/host-agent') +HOST_STATE_ROOT_MODE = 0o700 +HOST_OPERATION_DIRECTORY = HOST_STATE_ROOT / 'operations' +HOST_OPERATION_DIRECTORY_MODE = 0o700 +HOST_RESULT_DIRECTORY = HOST_STATE_ROOT / 'results' +HOST_RESULT_DIRECTORY_MODE = 0o750 +HOST_FAILED_HOLD_PATH = HOST_STATE_ROOT / 'failed-hold.json' + +MAX_STATE_BYTES = 16 * 1024 +_PHASES = { + 'prepared', 'forward_started', 'rollback_started', + 'succeeded', 'failed', 'rolled_back', 'failed_hold', +} +_PUBLICATION_STATES = {'original', 'partial', 'candidate'} +_TERMINAL_RESULTS = {'succeeded', 'failed', 'rolled_back', 'failed_hold'} +_TRANSITIONS = { + 'prepared': {'forward_started', 'rollback_started', 'failed'}, + 'forward_started': {'rollback_started', 'succeeded'}, + 'rollback_started': {'rolled_back', 'failed_hold'}, +} + + +class HostStateError(RuntimeError): + def __init__(self, category, *, cancellation=None): + self.category = str(category) + self.cancellation = cancellation + super().__init__('host runtime state failed') + + +def _canonical(value): + try: + payload = json.dumps( + value, sort_keys=True, separators=(',', ':'), ensure_ascii=True, + allow_nan=False, + ).encode('ascii') + except (TypeError, ValueError): + raise HostStateError('evidence') from None + if not payload or len(payload) > MAX_STATE_BYTES: + raise HostStateError('evidence') + return payload + + +def _require_directory(path, *, gid, mode): + try: + reject_reparse_components(path) + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISDIR(details.st_mode): + raise OSError('not a directory') + if os.name != 'nt' and ( + details.st_uid != HOST_ROOT_UID + or details.st_gid != gid + or stat.S_IMODE(details.st_mode) != mode + ): + raise OSError('directory metadata') + except Exception: + raise HostStateError('filesystem') from None + + +def _read_file(path, *, gid, mode): + descriptor = None + try: + reject_reparse_components(Path(path).parent) + flags = os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + if hasattr(os, 'O_NOFOLLOW'): + flags |= os.O_NOFOLLOW + descriptor = os.open(path, flags) + before = os.fstat(descriptor) + if ( + not stat.S_ISREG(before.st_mode) or before.st_nlink != 1 + or ( + os.name != 'nt' + and ( + before.st_uid != HOST_ROOT_UID or before.st_gid != gid + or stat.S_IMODE(before.st_mode) != mode + ) + ) + ): + raise OSError('file metadata') + with os.fdopen(descriptor, 'rb') as handle: + descriptor = None + payload = handle.read(MAX_STATE_BYTES + 1) + after = os.fstat(handle.fileno()) + current = os.stat(path, follow_symlinks=False) + identity = lambda item: ( + item.st_dev, item.st_ino, item.st_size, + getattr(item, 'st_mtime_ns', None), + None if os.name == 'nt' else getattr(item, 'st_ctime_ns', None), + ) + if ( + identity(before) != identity(after) + or identity(after) != identity(current) + or len(payload) > MAX_STATE_BYTES + ): + raise OSError('file changed') + return payload + except FileNotFoundError: + raise + except Exception: + raise HostStateError('filesystem') from None + finally: + if descriptor is not None: + os.close(descriptor) + + +def _decode_canonical(payload): + try: + value = json.loads(payload.decode('ascii')) + except (UnicodeDecodeError, json.JSONDecodeError): + raise HostStateError('evidence') from None + if not isinstance(value, dict) or _canonical(value) != payload: + raise HostStateError('evidence') + return value + + +def _write_stage(path, payload, *, gid, mode): + stage = Path(path).parent / f'.{Path(path).name}.stage' + descriptor = None + created = False + published = False + try: + try: + details = os.stat(stage, follow_symlinks=False) + if ( + not stat.S_ISREG(details.st_mode) or details.st_nlink != 1 + or (os.name != 'nt' and details.st_uid != HOST_ROOT_UID) + ): + raise OSError('unsafe stage') + os.unlink(stage) + fsync_directory(stage.parent) + except FileNotFoundError: + pass + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_CLOEXEC', 0) + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + if hasattr(os, 'O_NOFOLLOW'): + flags |= os.O_NOFOLLOW + descriptor = os.open(stage, flags, mode) + created = True + if os.name != 'nt': + os.fchmod(descriptor, mode) + details = os.fstat(descriptor) + if details.st_uid != HOST_ROOT_UID or details.st_gid != gid: + os.fchown(descriptor, HOST_ROOT_UID, gid) + view = memoryview(payload) + written = 0 + while written < len(view): + count = os.write(descriptor, view[written:]) + if count <= 0: + raise OSError('short write') + written += count + os.fsync(descriptor) + os.close(descriptor) + descriptor = None + os.replace(stage, path) + created = False + published = True + fsync_directory(Path(path).parent) + stored = _read_file(path, gid=gid, mode=mode) + if not hashlib.sha256(stored).digest() == hashlib.sha256(payload).digest(): + raise HostStateError('evidence') + except HostStateError: + if published: + raise HostStateError('uncertain') from None + raise + except BaseException as error: + if published: + cancellation = error if not isinstance(error, Exception) else None + raise HostStateError( + 'uncertain', cancellation=cancellation, + ) from None + if not isinstance(error, Exception): + raise + raise HostStateError('filesystem') from None + finally: + if descriptor is not None: + os.close(descriptor) + if created: + try: + os.unlink(stage) + fsync_directory(stage.parent) + except OSError: + pass + + +def _publish_exact(path, payload, *, gid, mode): + try: + existing = _read_file(path, gid=gid, mode=mode) + except FileNotFoundError: + _write_stage(path, payload, gid=gid, mode=mode) + return payload + if existing != payload: + raise HostStateError('conflict') + return existing + + +def failed_hold_operation(): + try: + payload = _read_file( + HOST_FAILED_HOLD_PATH, gid=HOST_ROOT_GID, mode=0o600, + ) + except FileNotFoundError: + return None + value = _decode_canonical(payload) + if ( + set(value) != { + 'schema', 'operation_id', 'action', 'forward_category', + 'publication_state', 'containment_confirmed', + } + or value.get('schema') != 1 + or not isinstance(value.get('operation_id'), str) + or value.get('publication_state') not in _PUBLICATION_STATES + or type(value.get('containment_confirmed')) is not bool + or not isinstance(value.get('forward_category'), str) + ): + raise HostStateError('evidence') + return value['operation_id'] + + +class HostOperationState: + def __init__(self, request): + self.request = decode_request_payload(encode_request_payload(request)) + self.operation_path = ( + HOST_OPERATION_DIRECTORY / f'{self.request.operation_id}.json' + ) + self.result_path = HOST_RESULT_DIRECTORY / f'{self.request.operation_id}.json' + + def _phase_record( + self, phase, publication_state, *, forward_category=None, + safe_detail=None, containment_confirmed=None, + ): + if ( + phase not in _PHASES + or publication_state not in _PUBLICATION_STATES + or forward_category is not None + and not isinstance(forward_category, str) + or safe_detail is not None + and not isinstance(safe_detail, str) + or containment_confirmed is not None + and type(containment_confirmed) is not bool + ): + raise HostStateError('evidence') + return { + 'schema': 1, + 'operation_id': self.request.operation_id, + 'action': self.request.action.value, + 'active_config_sha256': self.request.active_config_sha256, + 'active_secrets_sha256': self.request.active_secrets_sha256, + 'candidate_config_sha256': self.request.candidate_config_sha256, + 'candidate_secrets_sha256': self.request.candidate_secrets_sha256, + 'phase': phase, + 'publication_state': publication_state, + 'forward_category': forward_category, + 'safe_detail': safe_detail, + 'containment_confirmed': containment_confirmed, + } + + def _read_phase(self): + payload = _read_file(self.operation_path, gid=HOST_ROOT_GID, mode=0o600) + value = _decode_canonical(payload) + if set(value) != set(self._phase_record('prepared', 'original')): + raise HostStateError('evidence') + expected = self._phase_record( + value.get('phase'), value.get('publication_state'), + forward_category=value.get('forward_category'), + safe_detail=value.get('safe_detail'), + containment_confirmed=value.get('containment_confirmed'), + ) + if value != expected: + raise HostStateError('evidence') + return value + + def initialize(self, publication_state='original'): + _require_directory( + HOST_STATE_ROOT, gid=HOST_ROOT_GID, mode=HOST_STATE_ROOT_MODE, + ) + _require_directory( + HOST_OPERATION_DIRECTORY, + gid=HOST_ROOT_GID, + mode=HOST_OPERATION_DIRECTORY_MODE, + ) + hold = failed_hold_operation() + if hold is not None and hold != self.request.operation_id: + raise HostStateError('failed_hold') + try: + return self._read_phase() + except FileNotFoundError: + value = self._phase_record('prepared', publication_state) + _write_stage( + self.operation_path, _canonical(value), + gid=HOST_ROOT_GID, mode=0o600, + ) + return value + + def advance( + self, expected_phase, next_phase, publication_state, *, + forward_category=None, safe_detail=None, containment_confirmed=None, + ): + current = self._read_phase() + value = self._phase_record( + next_phase, publication_state, + forward_category=forward_category, + safe_detail=safe_detail, + containment_confirmed=containment_confirmed, + ) + if current == value: + return value + if ( + current['phase'] != expected_phase + or next_phase not in _TRANSITIONS.get(expected_phase, set()) + ): + raise HostStateError('state') + _write_stage( + self.operation_path, _canonical(value), + gid=HOST_ROOT_GID, mode=0o600, + ) + return value + + def terminal_result(self): + try: + payload = _read_file( + self.result_path, gid=HOST_RUNTIME_GID, mode=0o640, + ) + except FileNotFoundError: + return None + value = _decode_canonical(payload) + if ( + set(value) != { + 'schema', 'operation_id', 'action', 'result', 'safe_category', + 'safe_detail', 'resulting_identity', + } + or value.get('schema') != 1 + or value.get('operation_id') != self.request.operation_id + or value.get('action') != self.request.action.value + or value.get('result') not in _TERMINAL_RESULTS + ): + raise HostStateError('evidence') + return value + + def publish_result( + self, result, *, safe_category, safe_detail, resulting_identity, + ): + if result not in _TERMINAL_RESULTS: + raise HostStateError('evidence') + if result == 'succeeded': + if safe_category is not None or safe_detail is not None: + raise HostStateError('evidence') + elif not isinstance(safe_category, str) or not isinstance(safe_detail, str): + raise HostStateError('evidence') + if resulting_identity is not None and ( + not isinstance(resulting_identity, dict) + or set(resulting_identity) != { + 'active_config_sha256', 'active_secrets_sha256', + } + or any( + not isinstance(value, str) or len(value) != 64 + for value in resulting_identity.values() + ) + ): + raise HostStateError('evidence') + value = { + 'schema': 1, + 'operation_id': self.request.operation_id, + 'action': self.request.action.value, + 'result': result, + 'safe_category': safe_category, + 'safe_detail': safe_detail, + 'resulting_identity': resulting_identity, + } + _require_directory( + HOST_RESULT_DIRECTORY, + gid=HOST_RUNTIME_GID, + mode=HOST_RESULT_DIRECTORY_MODE, + ) + _publish_exact( + self.result_path, _canonical(value), gid=HOST_RUNTIME_GID, mode=0o640, + ) + return value + + def publish_failed_hold( + self, *, forward_category, publication_state, containment_confirmed, + ): + marker = { + 'schema': 1, + 'operation_id': self.request.operation_id, + 'action': self.request.action.value, + 'forward_category': str(forward_category), + 'publication_state': publication_state, + 'containment_confirmed': bool(containment_confirmed), + } + _publish_exact( + HOST_FAILED_HOLD_PATH, _canonical(marker), + gid=HOST_ROOT_GID, mode=0o600, + ) + return marker diff --git a/app/janitor.py b/app/janitor.py new file mode 100644 index 0000000..e28f2ac --- /dev/null +++ b/app/janitor.py @@ -0,0 +1,455 @@ +import sys + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('janitor could not disable bytecode writes') + +import argparse +import hashlib +import json +import os +import stat +import time +from dataclasses import dataclass +from datetime import datetime, timezone + +from lifecycle_authority import require_active_supervisor_child +from paths import apply_path_config +from process_identity import exact_process_identity_state +from runtime_security import ( + atomic_write_private_json, + canonical_path, + fsync_directory, + is_reparse_point, + private_directory_ready, + private_file_ready, + read_private_json, + reject_reparse_components, + require_private_directory, +) + + +MARKER_NAME = '.scanner-owner.json' +MARKER_SCHEMA = 2 +APPROVED_LAYOUTS = ( + ('work', '', ('trufflehog-', 'trufflehog-run-', 'trufflehog-probe-', 'hg-run-', 'docker-config-', 'docker-layer-', 'tmp-', 'worker-assignment-')), + ('work', 'docker-config', ('docker-config-',)), + ('work', 'hg', ('hg-run-',)), + ('work', 'tmp', ('trufflehog-', 'trufflehog-run-', 'hg-run-', 'docker-config-', 'docker-layer-', 'tmp-')), + ('work', os.path.join('tmp', 'docker-config'), ('docker-config-',)), + ('work', 'abandoned', ('worker-assignment-',)), +) + + +@dataclass +class JanitorBudget: + max_candidates: int = 50 + max_entries: int = 10000 + max_bytes: int = 1024 * 1024 * 1024 + max_seconds: float = 30.0 + max_depth: int = 64 + max_enumerated: int = 1000 + candidates: int = 0 + entries: int = 0 + bytes: int = 0 + started_at: float = 0.0 + exhausted: bool = False + enumerated: int = 0 + + def __post_init__(self): + self.started_at = self.started_at or time.monotonic() + + def consume(self, size=0, candidate=False, depth=0, allow_oversized=False): + if candidate: + self.candidates += 1 + else: + self.entries += 1 + self.bytes += max(0, int(size or 0)) + candidate_limit = self.candidates > self.max_candidates + entry_limit = self.entries > self.max_entries + byte_limit = self.bytes > self.max_bytes + depth_limit = depth > self.max_depth + time_limit = time.monotonic() - self.started_at >= self.max_seconds + self.exhausted = bool( + candidate_limit or entry_limit or byte_limit or depth_limit or time_limit + ) + # Unlinking one regular file is bounded metadata work regardless of its + # payload size. Directory traversal remains bounded by the other limits. + oversized_progress = bool( + allow_oversized and byte_limit + and not (candidate_limit or entry_limit or depth_limit or time_limit) + ) + return not self.exhausted or oversized_progress + + def consume_enumerated(self): + self.enumerated += 1 + self.exhausted = bool( + self.enumerated > self.max_enumerated + or time.monotonic() - self.started_at >= self.max_seconds + ) + return not self.exhausted + + +def _marker_relative_path(root, path): + relative = os.path.relpath(path, root) + if relative == '.' or relative.startswith('..' + os.sep) or os.path.isabs(relative): + raise ValueError('candidate escapes the approved janitor root') + return relative.replace(os.sep, '/') + + +def validate_marker(root, path, marker, allowed_executables, minimum_age_sec, now=None): + if marker.get('schema') != MARKER_SCHEMA or marker.get('root_kind') != 'work': + return False, 'unsupported_marker' + try: + if marker.get('relative_path') != _marker_relative_path(root, path): + return False, 'path_mismatch' + created = datetime.fromisoformat(str(marker.get('created_at') or '').replace('Z', '+00:00')) + if created.tzinfo is None: + created = created.replace(tzinfo=timezone.utc) + now_value = now or datetime.now(timezone.utc) + if (now_value - created).total_seconds() < max(0, float(minimum_age_sec)): + return False, 'too_young' + except (TypeError, ValueError): + return False, 'invalid_time_or_path' + + allowed = {canonical_path(value) for value in allowed_executables if value} + identities = {} + states = {} + prefixes = ('owner', 'parent') + if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')): + prefixes += ('child',) + for prefix in prefixes: + if prefix == 'child' and any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')): + return False, 'child_identity_invalid' + identity = { + 'pid': marker.get(f'{prefix}_pid'), + 'creation_time': marker.get(f'{prefix}_creation_time'), + 'executable': marker.get(f'{prefix}_executable'), + } + try: + executable = canonical_path(identity['executable']) + except (OSError, TypeError, ValueError): + return False, f'{prefix}_identity_invalid' + if executable not in allowed: + return False, f'{prefix}_executable_unapproved' + identity['executable'] = executable + identities[prefix] = identity + states[prefix] = exact_process_identity_state( + identity['pid'], identity['creation_time'], identity['executable'], + ) + if 'child' in states and states['child'] != 'dead': + return False, 'child_live_or_unknown' + if states['owner'] != 'dead': + return False, 'owner_live_or_unknown' + same_identity = all( + identities['owner'].get(field) == identities['parent'].get(field) + for field in ('pid', 'creation_time', 'executable') + ) + if same_identity and states['parent'] != 'dead': + return False, 'parent_owned_live_or_unknown' + return True, 'eligible' + + +def bounded_remove_tree(path, budget, marker_name=MARKER_NAME): + """Delete without recursion or reparse traversal; leave the root marker last.""" + path = os.path.abspath(path) + reject_reparse_components(path) + if is_reparse_point(path) or not private_directory_ready(path): + raise OSError(f'janitor candidate is not an exact private directory: {path}') + marker_path = os.path.join(path, marker_name) + stack = [] + root_iterator = os.scandir(path) + stack.append((path, root_iterator, 0)) + try: + while stack: + if time.monotonic() - budget.started_at >= budget.max_seconds: + budget.exhausted = True + return False + directory, iterator, depth = stack[-1] + try: + entry = next(iterator) + except StopIteration: + iterator.close() + stack.pop() + if directory == path: + if os.path.lexists(marker_path): + details = os.stat(marker_path, follow_symlinks=False) + # The verified owner marker is removed only after every payload + # entry is gone, so finishing the empty root must make progress + # even when one oversized payload exhausted this pass's budget. + budget.consume(details.st_size, depth=depth + 1) + if not private_file_ready(marker_path): + raise OSError('janitor owner marker lost its private identity') + os.remove(marker_path) + os.rmdir(directory) + fsync_directory(os.path.dirname(directory)) + return True + os.rmdir(directory) + continue + if entry.path == marker_path: + continue + details = entry.stat(follow_symlinks=False) + is_regular = stat.S_ISREG(details.st_mode) + if not budget.consume( + details.st_size, depth=depth + 1, allow_oversized=is_regular, + ): + return False + if entry.is_symlink() or is_reparse_point(entry.path): + raise OSError(f'janitor candidate contains a link or reparse point: {entry.path}') + if stat.S_ISDIR(details.st_mode): + child_iterator = os.scandir(entry.path) + stack.append((entry.path, child_iterator, depth + 1)) + elif is_regular: + os.chmod(entry.path, stat.S_IWRITE | stat.S_IREAD) + os.remove(entry.path) + else: + raise OSError(f'janitor candidate contains an unsupported entry: {entry.path}') + finally: + for _, iterator, _ in stack: + iterator.close() + return False + + +def _layout_cursor_name(root, relative_parent, prefixes): + identity = '|'.join(( + canonical_path(root), str(relative_parent).replace(os.sep, '/'), ','.join(prefixes), + )) + return 'layout:' + hashlib.sha256(identity.encode('utf-8')).hexdigest() + + +class JanitorCursorStore: + SCHEMA = 1 + + def __init__(self, path, root): + self.path = os.path.abspath(path) + self.root_hash = hashlib.sha256(canonical_path(root).encode('utf-8')).hexdigest() + self.dirty = False + self.state = { + 'schema': self.SCHEMA, + 'root_sha256': self.root_hash, + 'next_layout': 0, + 'layouts': {}, + } + self.iterators = {} + self.seeking = {} + if os.path.lexists(self.path): + loaded = read_private_json(self.path, max_bytes=256 * 1024) + if ( + not isinstance(loaded, dict) + or loaded.get('schema') != self.SCHEMA + or loaded.get('root_sha256') != self.root_hash + or not isinstance(loaded.get('layouts'), dict) + ): + raise RuntimeError('janitor cursor authority is invalid') + self.state = loaded + else: + require_private_directory(os.path.dirname(self.path), create=True) + atomic_write_private_json(self.path, self.state) + + def _save(self): + atomic_write_private_json(self.path, self.state) + self.dirty = False + + def _mark_dirty(self): + self.dirty = True + + def flush(self): + if self.dirty: + self._save() + + def next_layout(self, count): + index = int(self.state.get('next_layout') or 0) % max(1, int(count)) + self.state['next_layout'] = (index + 1) % max(1, int(count)) + self._mark_dirty() + return index + + def next_entry(self, layout_name, parent): + iterator = self.iterators.get(layout_name) + if iterator is None: + iterator = os.scandir(parent) + self.iterators[layout_name] = iterator + last_name = str((self.state['layouts'].get(layout_name) or {}).get('last_name') or '') + self.seeking[layout_name] = bool(last_name) + try: + entry = next(iterator) + except StopIteration: + iterator.close() + self.iterators.pop(layout_name, None) + self.seeking.pop(layout_name, None) + current = self.state['layouts'].setdefault(layout_name, {}) + current['last_name'] = '' + current['wrap_count'] = int(current.get('wrap_count') or 0) + 1 + self._mark_dirty() + return None, False + + current = self.state['layouts'].setdefault(layout_name, {'last_name': '', 'wrap_count': 0}) + target = str(current.get('last_name') or '') + if self.seeking.get(layout_name): + if entry.name == target: + self.seeking[layout_name] = False + return entry, True + current['last_name'] = entry.name + self._mark_dirty() + return entry, False + + def close(self): + for iterator in self.iterators.values(): + iterator.close() + self.iterators.clear() + + +class _MemoryCursorStore(JanitorCursorStore): + def __init__(self): + self.path = '' + self.root_hash = '' + self.dirty = False + self.state = {'schema': 1, 'root_sha256': '', 'next_layout': 0, 'layouts': {}} + self.iterators = {} + self.seeking = {} + + def _save(self): + self.dirty = False + return None + + +def iter_candidates(root, budget, cursor_store): + layouts = list(APPROVED_LAYOUTS) + completed_layouts = set() + while not budget.exhausted and len(completed_layouts) < len(layouts): + if ( + budget.enumerated >= budget.max_enumerated + or time.monotonic() - budget.started_at >= budget.max_seconds + ): + budget.exhausted = True + return + index = cursor_store.next_layout(len(layouts)) + root_kind, relative_parent, prefixes = layouts[index] + layout_name = _layout_cursor_name(root, relative_parent, prefixes) + if layout_name in completed_layouts: + continue + parent = os.path.join(root, relative_parent) if relative_parent else root + try: + if not os.path.isdir(parent) or is_reparse_point(parent): + completed_layouts.add(layout_name) + continue + entry, seeking = cursor_store.next_entry(layout_name, parent) + except OSError: + completed_layouts.add(layout_name) + continue + if entry is None: + completed_layouts.add(layout_name) + continue + if not budget.consume_enumerated(): + return + if seeking: + continue + if not entry.name.startswith(prefixes) or not entry.is_dir(follow_symlinks=False): + continue + if not budget.consume(candidate=True): + return + yield root_kind, layout_name, entry.name, entry.path + + +def run_janitor_pass( + root, allowed_executables, minimum_age_sec=7200, budget=None, + cursor_store=None, excluded_relative_paths=(), +): + budget = budget or JanitorBudget() + root = require_private_directory(root, create=False) + cursor_store = cursor_store or _MemoryCursorStore() + excluded = set() + for value in excluded_relative_paths: + relative = str(value or '').replace('\\', '/') + if ( + not relative or relative.startswith('/') or relative.endswith('/') + or any(part in ('', '.', '..') for part in relative.split('/')) + ): + raise ValueError('janitor exclusion path is invalid') + excluded.add(relative) + if len(excluded) > 4096: + raise ValueError('janitor exclusion set exceeds its bound') + report = {'considered': 0, 'removed': 0, 'retained': 0, 'errors': 0, 'exhausted': False} + try: + for _, _, _, path in iter_candidates(root, budget, cursor_store): + report['considered'] += 1 + if _marker_relative_path(root, path) in excluded: + report['retained'] += 1 + continue + marker_path = os.path.join(path, MARKER_NAME) + try: + if not private_file_ready(marker_path): + report['retained'] += 1 + continue + marker = read_private_json(marker_path) + eligible, _ = validate_marker( + root, path, marker, allowed_executables, minimum_age_sec, + ) + if not eligible: + report['retained'] += 1 + continue + if bounded_remove_tree(path, budget): + report['removed'] += 1 + else: + report['retained'] += 1 + except (OSError, ValueError): + report['errors'] += 1 + if budget.exhausted: + break + finally: + cursor_store.flush() + report['exhausted'] = budget.exhausted + report['entries'] = budget.entries + report['bytes'] = budget.bytes + report['enumerated'] = budget.enumerated + return report + + +def parse_args(): + parser = argparse.ArgumentParser(description='Bounded scanner work-directory janitor') + parser.add_argument('--config', required=True) + return parser.parse_args() + + +def main(): + metadata = require_active_supervisor_child(child_kind='janitor', require_dsn=False) + args = parse_args() + import yaml + + with open(args.config, 'r', encoding='utf-8') as handle: + config = apply_path_config(yaml.safe_load(handle) or {}, args.config) + global_config = config.get('global') or {} + janitor_config = ((config.get('supervisor') or {}).get('janitor') or {}) + manifest = metadata.get('code_manifest') or {} + allowed = [sys.executable] + allowed.extend( + item.get('path') for item in (manifest.get('executables') or {}).values() + if isinstance(item, dict) and item.get('path') + ) + interval = max(5.0, float(janitor_config.get('interval_sec', 60) or 60)) + cursor_store = JanitorCursorStore( + os.path.join(global_config['state_dir'], 'janitor.cursor.json'), + global_config['work_dir'], + ) + try: + while True: + budget = JanitorBudget( + max_candidates=max(1, int(janitor_config.get('max_candidates', 50) or 50)), + max_entries=max(1, int(janitor_config.get('max_entries', 10000) or 10000)), + max_bytes=max(1, int(janitor_config.get('max_bytes', 1024 * 1024 * 1024) or 1)), + max_seconds=max(0.1, float(janitor_config.get('max_seconds', 30) or 30)), + max_depth=max(1, int(janitor_config.get('max_depth', 64) or 64)), + max_enumerated=max(1, int(janitor_config.get('max_enumerated', 1000) or 1000)), + ) + report = run_janitor_pass( + global_config['work_dir'], allowed, + minimum_age_sec=max(0, int(janitor_config.get('minimum_age_sec', 7200) or 0)), + budget=budget, cursor_store=cursor_store, + ) + print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True) + time.sleep(interval) + finally: + cursor_store.close() + + +if __name__ == '__main__': + main() diff --git a/app/jsonl_projector.py b/app/jsonl_projector.py new file mode 100644 index 0000000..b137a5b --- /dev/null +++ b/app/jsonl_projector.py @@ -0,0 +1,736 @@ +import sys + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('JSONL projector could not disable bytecode writes') + +import argparse +import hashlib +import json +import os +import re +import time +from dataclasses import dataclass + +from lifecycle_authority import require_active_supervisor_child +from paths import apply_path_config +from process_identity import current_process_identity +from runtime_security import ( + PrivateFileLock, + PrivatePathState, + durable_publish, + durable_unlink, + harden_private_file, + inspect_private_relative_path, + private_file_ready, + reject_reparse_components, + require_private_directory, +) +from scanner_db import ScannerDB + + +STREAM_MASKS = {'scan_results': 1, 'found_secrets': 2, 'scan_errors': 4} +MAX_SERIALIZED_EVENT_BYTES = 192 * 1024 * 1024 +MAX_TAIL_QUARANTINE_BYTES = MAX_SERIALIZED_EVENT_BYTES + + +@dataclass(frozen=True) +class SerializedStream: + stream_name: str + path: str + byte_length: int + payload_sha256: str + record_count: int + artifact_id: int = 0 + + +class _HashedWriter: + def __init__(self, handle, max_bytes): + self.handle = handle + self.max_bytes = int(max_bytes) + self.digest = hashlib.sha256() + self.bytes_written = 0 + + def write(self, payload): + payload = payload.encode('utf-8') if isinstance(payload, str) else bytes(payload) + if self.bytes_written + len(payload) > self.max_bytes: + raise ValueError('projection serialization exceeds its event byte bound') + self.handle.write(payload) + self.digest.update(payload) + self.bytes_written += len(payload) + + +def _write_json_line(writer, value): + encoder = json.JSONEncoder( + ensure_ascii=False, sort_keys=True, separators=(',', ':'), default=str, + ) + for chunk in encoder.iterencode(value): + writer.write(chunk) + writer.write(b'\n') + + +def _write_json_value(writer, value): + encoder = json.JSONEncoder( + ensure_ascii=False, sort_keys=True, separators=(',', ':'), default=str, + ) + for chunk in encoder.iterencode(value): + writer.write(chunk) + + +def _write_json_array(writer, values): + writer.write(b'[') + first = True + for value in values: + if not first: + writer.write(b',') + _write_json_value(writer, value) + first = False + writer.write(b']') + + +def _write_scan_result(writer, header, findings, errors): + values = dict(header) + keys = sorted(set(values) | {'findings', 'errors'}) + writer.write(b'{') + for index, key in enumerate(keys): + if index: + writer.write(b',') + _write_json_value(writer, key) + writer.write(b':') + if key == 'findings': + _write_json_array(writer, findings) + elif key == 'errors': + _write_json_array(writer, errors) + else: + _write_json_value(writer, values[key]) + writer.write(b'}\n') + + +class JsonlProjector: + def __init__( + self, db, results_dir, supervisor_instance_id, lease_seconds=300, + fault=None, keycheck_dir=None, quarantine_max_items=10000, + quarantine_max_bytes=1024 * 1024 * 1024, artifact_tracking=True, + projection_max_bytes=2 * 1024 * 1024 * 1024, + ): + self.db = db + self.results_dir = require_private_directory(results_dir, create=False) + self.keycheck_dir = require_private_directory( + keycheck_dir or os.path.join(os.path.dirname(self.results_dir), 'keychecks'), + create=False, + ) + self.supervisor_instance_id = str(supervisor_instance_id) + self.lease_seconds = max(30, int(lease_seconds)) + self.fault = fault + self.lease = None + self.file_lock = None + self.temp_dir = require_private_directory( + os.path.join(self.results_dir, '.projection-tmp'), create=True, + ) + self.quarantine_dir = require_private_directory( + os.path.join(self.results_dir, '.projection-quarantine'), create=True, + ) + self.quarantine_max_items = max(0, int(quarantine_max_items)) + self.quarantine_max_bytes = max(0, int(quarantine_max_bytes)) + self.projection_max_bytes = max(0, int(projection_max_bytes)) + self.artifact_tracking = bool(artifact_tracking) + + def _inject(self, stage, value=None): + if self.fault is not None: + self.fault(stage, value) + + def start(self): + self.db.require_runtime_safety_schema() + self.db.require_final_cutover() + self.file_lock = PrivateFileLock( + os.path.join(self.results_dir, '.jsonl-projector.lock') + ).acquire() + self.lease = self.db.acquire_pipeline_lease( + 'jsonl_projector', self.supervisor_instance_id, current_process_identity(), + lease_seconds=self.lease_seconds, initial_state='starting', + ) + if not self.lease: + self.file_lock.release() + self.file_lock = None + raise RuntimeError('another JSONL projector owns the singleton advisory lock') + if self.artifact_tracking: + self.reconcile_terminal_temps() + self.recover_rotations() + if not self.heartbeat(): + raise RuntimeError('JSONL projector ready lease publication failed') + return self + + def heartbeat(self, error=''): + return self.db.heartbeat_pipeline_lease( + 'jsonl_projector', self.lease['generation'], self.lease['lease_token'], + lease_seconds=self.lease_seconds, state='ready', error=error, + ) + + def _rollback_database(self): + connection = getattr(self.db, 'conn', None) + if connection is not None: + try: + connection.rollback() + except BaseException: + pass + + def stop(self, error=''): + try: + if self.lease: + self._rollback_database() + try: + self.db.release_pipeline_lease( + 'jsonl_projector', self.lease['generation'], self.lease['lease_token'], + state='failed' if error else 'released', error=error, + ) + except BaseException: + self._rollback_database() + raise + self.lease = None + finally: + if self.file_lock: + self.file_lock.release() + self.file_lock = None + + def _results_path(self, relative, stream_name=''): + root = self.keycheck_dir if str(stream_name).startswith('keycheck:') else self.results_dir + path = os.path.abspath(os.path.join(root, str(relative).replace('/', os.sep))) + if os.path.commonpath((root, path)) != root or path == root: + raise ValueError('projection path escapes results_dir') + reject_reparse_components(os.path.dirname(path)) + return path + + @staticmethod + def _segment_relative(base_relative, generation): + base, extension = os.path.splitext(base_relative) + return f'{base}.g{int(generation):06d}{extension}' + + def _create_private_empty(self, path): + if os.path.exists(path): + if not private_file_ready(path): + raise OSError(f'projection active file is not private: {path}') + return + descriptor = os.open( + path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), 0o600, + ) + os.close(descriptor) + harden_private_file(path) + + def _prepared_path(self, job, stream_name): + safe_stream_name = re.sub(r'[^A-Za-z0-9_.-]+', '_', stream_name) + event_hash = str(job['event_hash'] or '').lower() + if not re.fullmatch(r'[a-f0-9]{64}', event_hash): + raise ValueError('projection job event hash is not a canonical SHA-256 identity') + path = os.path.join( + self.temp_dir, + f'job-{int(job["id"])}-{safe_stream_name}-{event_hash}.prepared', + ) + relative = os.path.relpath(path, self.results_dir).replace(os.sep, '/') + artifact_id = 0 + if self.artifact_tracking: + artifact_id = self.db.register_pipeline_artifact( + 'jsonl_projector', 'prepared_stream', job['id'], stream_name, + relative, state='expected', byte_count=int(job['capacity_bytes']), + ) + if os.path.lexists(path): + if not private_file_ready(path): + raise OSError(f'projection prepared file is not private: {path}') + else: + descriptor = os.open( + path, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), + 0o600, + ) + os.close(descriptor) + harden_private_file(path) + return path, artifact_id + + def _delete_registered_artifact(self, path, artifact_id): + relative = os.path.relpath(path, self.results_dir).replace(os.sep, '/') + inspection = inspect_private_relative_path(self.results_dir, relative) + if inspection.state == PrivatePathState.UNKNOWN: + raise OSError('projection artifact state is unknown during cleanup') + if inspection.state == PrivatePathState.PRESENT: + durable_unlink(inspection.path) + inspection = inspect_private_relative_path(self.results_dir, relative) + if inspection.state != PrivatePathState.ABSENT: + raise OSError('projection artifact unlink was not confirmed') + if artifact_id: + self.db.mark_pipeline_artifact_deleted(artifact_id) + + def reconcile_terminal_temps(self, max_pages=100): + if not self.artifact_tracking or not hasattr(self.db, 'projection_terminal_temp_artifacts'): + return + for _ in range(max(1, int(max_pages))): + rows = self.db.projection_terminal_temp_artifacts(100) + if not rows: + return + for row in rows: + path = self._results_path(row['relative_path']) + self._delete_registered_artifact(path, row['id']) + if len(rows) < 100: + return + return + + def recover_rotations(self): + for rotation in self.db.pending_projection_rotations(100): + active = self._results_path(rotation['base_relative_path'], rotation['stream_name']) + segment = self._results_path(rotation['segment_relative_path'], rotation['stream_name']) + require_private_directory(os.path.dirname(active), create=True) + active_exists = os.path.isfile(active) + segment_exists = os.path.isfile(segment) + if active_exists and segment_exists: + if ( + os.path.getsize(active) != 0 + or os.path.getsize(segment) != int(rotation['source_bytes']) + ): + raise RuntimeError('projection rotation has conflicting active and immutable names') + elif active_exists: + if os.path.getsize(active) != int(rotation['source_bytes']): + raise RuntimeError('projection rotation active size changed') + durable_publish(active, segment) + elif not segment_exists: + raise RuntimeError('projection rotation lost both exact source names') + self._create_private_empty(active) + if not self.db.complete_projection_rotation(rotation['id']): + raise RuntimeError('projection rotation completion fence failed') + + def _serialize(self, job): + if job['job_kind'] == 'scan_event': + scan = self.db.projection_scan_header( + job['target_scan_id'], max_bytes=self.projection_max_bytes, + ) + if not isinstance(scan, dict) or not isinstance(scan.get('result'), dict): + raise ValueError('authoritative scan result cannot be reconstructed') + result = scan['result'] + normalized = scan['storage'] == 'normalized_v2' + stream_specs = [ + (name, mask) for name, mask in STREAM_MASKS.items() + if int(job['required_stream_mask']) & mask + ] + elif job['job_kind'] == 'keycheck_event': + result = self.db.keycheck_result_for_projection(job['keycheck_result_id']) + if not isinstance(result, dict): + raise ValueError('authoritative keycheck result cannot be reconstructed') + service = str(result['service']) + stream_specs = [ + (name, mask) for name, mask in ( + (f'keycheck:{service}:results', 8), + (f'keycheck:{service}:status', 16), + ) if int(job['required_stream_mask']) & mask + ] + else: + raise ValueError(f'unsupported projection job kind: {job["job_kind"]}') + streams = [] + prepared_paths = [] + try: + for stream_name, mask in stream_specs: + path, artifact_id = self._prepared_path(job, stream_name) + prepared_paths.append((path, artifact_id)) + count = 0 + with open(path, 'wb', buffering=0) as handle: + writer = _HashedWriter(handle, self.projection_max_bytes) + if stream_name == 'scan_results': + findings = ( + self.db.iter_projection_findings(job['target_scan_id']) + if normalized else iter(result.get('findings') or ()) + ) + errors = ( + self.db.iter_projection_errors(job['target_scan_id']) + if normalized else iter(result.get('errors') or ()) + ) + _write_scan_result(writer, result, findings, errors) + count = 1 + elif stream_name == 'found_secrets': + findings = ( + self.db.iter_projection_findings(job['target_scan_id']) + if normalized else iter(result.get('findings') or ()) + ) + for finding in findings: + _write_json_line(writer, finding) + count += 1 + elif stream_name == 'scan_errors': + timestamp = result.get('timestamp') or '' + scan_type = result.get('scan_type') or '' + target = result.get('target') or '' + event_id = result.get('scan_event_id') or job['event_id'] + errors = ( + self.db.iter_projection_errors(job['target_scan_id']) + if normalized else iter(result.get('errors') or ()) + ) + for index, error in enumerate(errors, 1): + row_id = hashlib.sha256( + f'{event_id}|{index}'.encode('utf-8') + ).hexdigest() + writer.write( + f'{row_id}\t{timestamp}\t{scan_type}\t{target}\t{error}\n' + ) + count += 1 + elif stream_name.endswith(':results'): + payload = { + 'event_id': result['event_id'], + 'service': result['service'], + 'status': result['status'], + 'status_group': result['status_group'], + 'checked_at': result['checked_at'], + 'key_hash': result['key_hash'], + 'secret_hash': result['secret_hash'], + 'key_masked': result['key_masked'], + 'finding_uid': result['finding_uid'], + 'detector': result['detector_name'], + 'source': result['source'], + 'message': result['message'], + 'metadata': json.loads(result['metadata_json'] or '{}'), + 'result_source': result['result_source'], + } + _write_json_line(writer, payload) + count = 1 + else: + secret = result.get('credential_secret_text') or result.get('credential_secret_json') or '' + message = str(result.get('message') or '').replace('\r', ' ').replace('\n', ' ')[:1000] + writer.write( + f'{secret}\t{result["status"]}\t{result["checked_at"]}\t{message}\n' + ) + count = 1 + handle.flush() + os.fsync(handle.fileno()) + streams.append(SerializedStream( + stream_name, path, writer.bytes_written, writer.digest.hexdigest(), count, + artifact_id, + )) + if self.artifact_tracking: + self.db.register_pipeline_artifact( + 'jsonl_projector', 'prepared_stream', job['id'], stream_name, + os.path.relpath(path, self.results_dir).replace(os.sep, '/'), + state='present', payload_sha256=writer.digest.hexdigest(), + byte_count=writer.bytes_written, + ) + return streams + except BaseException: + self._rollback_database() + for path, artifact_id in prepared_paths: + try: + self._delete_registered_artifact(path, artifact_id) + except BaseException: + pass + raise + + def _hash_region(self, path, offset, length): + digest = hashlib.sha256() + remaining = int(length) + with open(path, 'rb', buffering=0) as handle: + handle.seek(int(offset)) + while remaining: + block = handle.read(min(1024 * 1024, remaining)) + if not block: + raise OSError('projection append proof is truncated') + digest.update(block) + remaining -= len(block) + return digest.hexdigest() + + def _quarantine_tail(self, active, offset, job, stream_name, append): + size = os.path.getsize(active) + length = max(0, size - int(offset)) + safe_stream_name = re.sub(r'[^A-Za-z0-9_.-]+', '_', stream_name) + tail_hash = self._hash_region(active, offset, length) if length else hashlib.sha256(b'').hexdigest() + path = os.path.join( + self.quarantine_dir, + f'append-{append["id"]}-o{int(offset)}-l{length}-{tail_hash}.tail', + ) + evidence_error = None + evidence_registered = False + try: + if length: + if length > MAX_TAIL_QUARANTINE_BYTES: + raise ValueError('projection partial tail exceeds quarantine byte bound') + relative = os.path.relpath(path, self.results_dir).replace(os.sep, '/') + registration = self.db.register_projection_tail_quarantine( + job['id'], append['id'], stream_name, relative, tail_hash, length, + self.quarantine_max_items, self.quarantine_max_bytes, + ) + evidence_registered = True + if not os.path.lexists(path): + temporary = path + '.partial' + temp_relative = os.path.relpath(temporary, self.results_dir).replace(os.sep, '/') + temp_artifact = self.db.register_pipeline_artifact( + 'jsonl_projector', 'projection_tail_temp', job['id'], + f'{append["id"]}:{int(offset)}:{length}:{tail_hash}', + temp_relative, state='expected', byte_count=length, + ) + if os.path.lexists(temporary): + if not private_file_ready(temporary): + raise OSError('existing projection tail temporary is not private') + self._delete_registered_artifact(temporary, temp_artifact) + temp_artifact = self.db.register_pipeline_artifact( + 'jsonl_projector', 'projection_tail_temp', job['id'], + f'{append["id"]}:{int(offset)}:{length}:{tail_hash}', + temp_relative, state='expected', byte_count=length, + ) + descriptor = os.open( + temporary, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), + 0o600, + ) + try: + os.close(descriptor) + descriptor = None + harden_private_file(temporary) + with open(active, 'rb', buffering=0) as source, open(temporary, 'wb', buffering=0) as target: + source.seek(int(offset)) + remaining = length + while remaining: + block = source.read(min(1024 * 1024, remaining)) + if not block: + raise OSError('projection partial tail changed while quarantining') + target.write(block) + remaining -= len(block) + target.flush() + os.fsync(target.fileno()) + if ( + os.path.getsize(temporary) != length + or self._hash_region(temporary, 0, length) != tail_hash + ): + raise OSError('projection tail quarantine proof failed before publication') + self.db.register_pipeline_artifact( + 'jsonl_projector', 'projection_tail_temp', job['id'], + f'{append["id"]}:{int(offset)}:{length}:{tail_hash}', + temp_relative, state='present', payload_sha256=tail_hash, + byte_count=length, + ) + durable_publish(temporary, path) + temporary_state = inspect_private_relative_path( + self.results_dir, temp_relative, + ) + if temporary_state.state != PrivatePathState.ABSENT: + raise OSError('projection tail temporary retirement was not confirmed') + self.db.mark_pipeline_artifact_deleted(temp_artifact) + finally: + if descriptor is not None: + os.close(descriptor) + elif ( + not private_file_ready(path) + or os.path.getsize(path) != length + or self._hash_region(path, 0, length) != tail_hash + ): + raise OSError('immutable projection tail evidence is invalid') + if not self.db.confirm_projection_tail_artifact( + registration['artifact_id'], tail_hash, length, + ): + raise RuntimeError('projection tail artifact confirmation lost its fence') + except BaseException as exc: + evidence_error = exc + finally: + with open(active, 'r+b', buffering=0) as handle: + handle.truncate(int(offset)) + handle.flush() + os.fsync(handle.fileno()) + if os.path.getsize(active) != int(offset): + raise OSError('projection partial tail corrective truncation was not confirmed') + if evidence_error is not None: + raise evidence_error + + def _rotate_if_needed(self, state, payload_bytes): + active = self._results_path(state['base_relative_path'], state['stream_name']) + require_private_directory(os.path.dirname(active), create=True) + self._create_private_empty(active) + current_size = os.path.getsize(active) + if current_size != int(state['committed_offset']): + state = self.db.initialize_projection_stream_offset( + state['stream_name'], current_size, + ) + if current_size != int(state['committed_offset']): + raise RuntimeError('projection active size does not match its committed cursor') + if not current_size or current_size + payload_bytes <= int(state['rotation_bytes']): + return state + segment_relative = self._segment_relative( + state['base_relative_path'], state['current_generation'], + ) + rotation = self.db.prepare_projection_rotation( + state['stream_name'], current_size, segment_relative, + ) + segment = self._results_path(segment_relative, state['stream_name']) + self._inject('before_rotation_rename', rotation) + durable_publish(active, segment) + self._inject('after_rotation_rename', rotation) + self._create_private_empty(active) + if not self.db.complete_projection_rotation(rotation['id']): + raise RuntimeError('projection rotation completion fence failed') + updated = self.db.projection_stream_state(state['stream_name']) + oldest = int(updated['current_generation']) - int(updated['max_generations']) + if oldest >= 0: + old_relative = self._segment_relative(updated['base_relative_path'], oldest) + old_path = self._results_path(old_relative, updated['stream_name']) + if os.path.isfile(old_path): + durable_unlink(old_path) + return updated + + def _append_stream(self, job, serialized): + state = self.db.projection_stream_state(serialized.stream_name) + if not state: + raise ValueError(f'projection stream is absent: {serialized.stream_name}') + existing = self.db.projection_append_for_job(job['id'], serialized.stream_name) + if existing is None: + state = self._rotate_if_needed(state, serialized.byte_length) + append = self.db.prepare_projection_append( + job['id'], job['lease_token'], serialized.stream_name, + state['generation'], serialized.byte_length, + serialized.payload_sha256, serialized.record_count, + ) + else: + append = existing + if ( + int(append['byte_length']) != serialized.byte_length + or str(append['payload_sha256']) != serialized.payload_sha256 + or int(append['record_count']) != serialized.record_count + ): + raise ValueError('prepared projection append conflicts with deterministic serialization') + active = self._results_path(state['base_relative_path'], serialized.stream_name) + require_private_directory(os.path.dirname(active), create=True) + self._create_private_empty(active) + offset = int(append['byte_offset']) + end = offset + int(append['byte_length']) + size = os.path.getsize(active) + if append['state'] == 'appended': + if size < end or self._hash_region(active, offset, append['byte_length']) != append['payload_sha256']: + raise ValueError('acknowledged projection append proof is invalid') + return + if size >= end: + if size == end and self._hash_region(active, offset, append['byte_length']) == append['payload_sha256']: + if not self.db.complete_projection_append(append['id'], job['id'], job['lease_token']): + raise RuntimeError('projection append replay acknowledgement failed') + return + self._quarantine_tail(active, offset, job, serialized.stream_name, append) + self._inject('after_tail_recovery', append) + elif size > offset: + self._quarantine_tail(active, offset, job, serialized.stream_name, append) + self._inject('after_tail_recovery', append) + elif size < offset: + raise ValueError('projection active file is shorter than its prepared offset') + self._inject('before_append', append) + with open(active, 'ab', buffering=0) as target, open(serialized.path, 'rb', buffering=0) as source: + while True: + block = source.read(1024 * 1024) + if not block: + break + target.write(block) + target.flush() + os.fsync(target.fileno()) + self._inject('after_append_fsync', append) + if os.path.getsize(active) != end or self._hash_region(active, offset, append['byte_length']) != append['payload_sha256']: + raise RuntimeError('projection append proof failed after fsync') + if not self.db.complete_projection_append(append['id'], job['id'], job['lease_token']): + raise RuntimeError('projection append completion fence failed') + + def process_one(self): + self.reconcile_terminal_temps(max_pages=1) + job = self.db.claim_projection_job( + self.lease['generation'], self.lease['lease_token'], self.lease_seconds, + ) + if not job: + return False + streams = [] + primary_failure = False + try: + streams = self._serialize(job) + actual_bytes = sum(item.byte_length for item in streams) + expanded = self.db.expand_projection_job_capacity( + job['id'], job['lease_token'], actual_bytes, + self.projection_max_bytes, + ) + if expanded is False: + return True + if expanded is None: + raise RuntimeError('projection capacity expansion lost its lease fence') + job = expanded + for serialized in streams: + self._append_stream(job, serialized) + if not self.db.complete_projection_job(job['id'], job['lease_token']): + raise RuntimeError('projection job completion fence failed') + return True + except (ValueError, TypeError, UnicodeError, json.JSONDecodeError) as exc: + self._rollback_database() + try: + self.db.quarantine_projection_job( + job['id'], job['lease_token'], 'deterministic_projection_error', str(exc), + quarantine_max_items=self.quarantine_max_items, + quarantine_max_bytes=self.quarantine_max_bytes, + ) + except BaseException: + primary_failure = True + self._rollback_database() + raise + return True + except BaseException: + primary_failure = True + self._rollback_database() + raise + finally: + for serialized in streams: + try: + self._delete_registered_artifact( + serialized.path, serialized.artifact_id, + ) + except OSError: + pass + except BaseException: + self._rollback_database() + if not primary_failure: + raise + + +def parse_args(): + parser = argparse.ArgumentParser(description='Singleton PostgreSQL-backed JSONL projector') + parser.add_argument('--config', required=True) + return parser.parse_args() + + +def main(): + metadata = require_active_supervisor_child(child_kind='jsonl-projector', require_dsn=True) + args = parse_args() + import yaml + + with open(args.config, 'r', encoding='utf-8') as handle: + config = apply_path_config(yaml.safe_load(handle) or {}, args.config) + global_config = config.get('global') or {} + settings = ((config.get('supervisor') or {}).get('jsonl_projector') or {}) + db = ScannerDB(db_url=global_config['database_url'], initialize=False) + if not db.enabled: + raise SystemExit('JSONL projector PostgreSQL connection is unavailable') + db.set_application_name('truf-jsonl-projector') + worker = JsonlProjector( + db, global_config['results_dir'], metadata['instance_id'], + lease_seconds=int(settings.get('lease_seconds', 300)), + keycheck_dir=global_config['keycheck_dir'], + quarantine_max_items=int(global_config.get('pipeline_quarantine_max_items', 10000)), + quarantine_max_bytes=int(global_config.get('pipeline_quarantine_max_bytes', 1024 * 1024 * 1024)), + projection_max_bytes=int(global_config.get( + 'projection_backlog_max_bytes', 2 * 1024 * 1024 * 1024, + )), + ) + error = '' + try: + worker.start() + idle = max(0.05, float(settings.get('poll_sec', 0.2))) + next_heartbeat = time.monotonic() + worker.lease_seconds / 3 + while True: + processed = worker.process_one() + if time.monotonic() >= next_heartbeat: + if not worker.heartbeat(): + raise RuntimeError('JSONL projector heartbeat fence was lost') + next_heartbeat = time.monotonic() + worker.lease_seconds / 3 + if not processed: + time.sleep(idle) + except KeyboardInterrupt: + pass + except BaseException as exc: + error = f'{type(exc).__name__}: {exc}' + raise + finally: + try: + worker.stop(error) + finally: + db.close() + + +if __name__ == '__main__': + main() diff --git a/app/keycheck_accounting_smoke.py b/app/keycheck_accounting_smoke.py new file mode 100644 index 0000000..7975cf9 --- /dev/null +++ b/app/keycheck_accounting_smoke.py @@ -0,0 +1,71 @@ +import sys + +sys.dont_write_bytecode = True + +import os +import sqlite3 +import tempfile + +from keycheckers.keycheck_common import ( + append_checked, + append_status, + record_cached_keycheck_occurrence, +) +from keycheck_runner import ingest_keycheck_results_to_db +from scanner_db import ScannerDB +from runtime_security import ensure_private_directory + + +def main(): + for key in ('SCANNER_DB_URL', 'DATABASE_URL', 'TRUF_MANAGED_POSTGRES_DSN', 'KEYCHECK_DB_URL'): + os.environ.pop(key, None) + with tempfile.TemporaryDirectory(prefix="keycheck-accounting-") as tmp: + db_path = os.path.join(tmp, "scanner.db") + output_dir = os.path.join(tmp, "keychecks", "openai") + ensure_private_directory(output_dir, reject_reparse=True) + + db = ScannerDB(db_path=db_path) + db.close() + + alive_file = os.path.join(output_dir, "openaiAlive.txt") + checked_file = os.path.join(output_dir, "openaiChecked.txt") + key = "sk-test-keycheck-accounting-1234567890" + append_status(alive_file, key, "ALIVE", "fixture", "smoke") + append_checked(checked_file, key, "ALIVE") + + os.environ["KEYCHECK_DB_PATH"] = db_path + os.environ["KEYCHECK_OUTPUT_DIR"] = output_dir + os.environ["KEYCHECK_SERVICE"] = "openai" + + finding = {"DetectorName": "OpenAI", "Raw": key} + ok = record_cached_keycheck_occurrence("openai", key, "ALIVE", "fixture.jsonl:1", finding, "OpenAI") + if not ok: + raise SystemExit("cached occurrence write returned false") + + inserted = ingest_keycheck_results_to_db( + {"database_path": db_path, "keycheck_dir": os.path.join(tmp, "keychecks")}, + ["openai"], + max_rows=10, + ) + if inserted != 1: + raise SystemExit(f"unexpected ingest count: {inserted}") + + conn = sqlite3.connect(db_path) + try: + row = conn.execute( + "SELECT service, status, status_group, metadata_json FROM keycheck_results" + ).fetchone() + finally: + conn.close() + if not row: + raise SystemExit("missing keycheck_results row") + service, status, status_group, metadata_json = row + if (service, status, status_group) != ("openai", "ALIVE", "alive"): + raise SystemExit(f"unexpected row status: {(service, status, status_group)}") + if "cached_status" not in metadata_json: + raise SystemExit("missing cached_status metadata") + print("keycheck accounting smoke ok") + + +if __name__ == "__main__": + main() diff --git a/app/keycheck_candidates.py b/app/keycheck_candidates.py new file mode 100644 index 0000000..e666cd0 --- /dev/null +++ b/app/keycheck_candidates.py @@ -0,0 +1,524 @@ +import hashlib +import json +import re +from dataclasses import dataclass, field + + +MAX_CANDIDATE_SECRET_BYTES = 1024 * 1024 +MAX_CANDIDATE_METADATA_BYTES = 64 * 1024 + + +@dataclass(frozen=True) +class CandidateSpec: + service: str + candidate_kind: str + credential_hash: str + provider_key_hash: str + secret_hash: str + secret_text: str | None = None + secret_json: str | None = None + key_masked: str = '' + endpoint: str = '' + principal: str = '' + metadata: dict = field(default_factory=dict) + + def as_frame(self, attribution=None): + return { + 'service': self.service, + 'candidate_kind': self.candidate_kind, + 'credential_hash': self.credential_hash, + 'provider_key_hash': self.provider_key_hash, + 'secret_hash': self.secret_hash, + 'secret_text': self.secret_text, + 'secret_json': self.secret_json, + 'key_masked': self.key_masked, + 'endpoint': self.endpoint, + 'principal': self.principal, + 'metadata': self.metadata, + 'attribution': dict(attribution or {}), + } + + +DETECTOR_SERVICES = { + 'openai': 'openai', + 'anthropic': 'anthropic', + 'qwendashscope': 'qwen', + 'qwen_dashscope': 'qwen', + 'qwen': 'qwen', + 'dashscope': 'qwen', + 'deepseek': 'deepseek', + 'deepseekapikey': 'deepseek', + 'deepseek_api_key': 'deepseek', + 'zaiglm': 'zai', + 'kimimoonshot': 'kimi', + 'moonshotai': 'kimi', + 'moonshot': 'kimi', + 'kimi': 'kimi', + 'openrouter': 'openrouter', + 'groq': 'groq', + 'replicate': 'replicate', + 'xai': 'xai', + 'huggingface': 'huggingface', + 'github': 'github', + 'githuboauth2': 'github', + 'gitlab': 'gitlab', + 'aws': 'aws', + 'gcp': 'gcp', + 'gcpapplicationdefaultcredentials': 'gcp', + 'googleai': 'gemini', + 'googleaistudio': 'gemini', + 'azure': 'azure', + 'azureopenai': 'azure', + 'azurecontainerregistry': 'azure', + 'azurefoundryendpointbeforekey': 'azure', + 'azurefoundrykeybeforeendpoint': 'azure', + 'dockerhub': 'dockerhub', +} +GEMINI_KEY_RE = re.compile( + r'AIza[0-9A-Za-z_-]{20,}|(? max_bytes: + raise ValueError('keycheck candidate field exceeds its byte bound') + return text + + +def _mask(value): + text = str(value or '') + if len(text) <= 8: + return '*' * len(text) + return text[:4] + ('*' * min(24, len(text) - 8)) + text[-4:] + + +def _service_for_finding(finding): + context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {} + hint = str(context.get('provider_hint') or '').lower() + if hint in AMBIGUOUS_PROVIDER_HINTS: + return 'provider_resolver' + if hint in GENERIC_SK_PROVIDERS: + return hint + detector = re.sub(r'[^a-z0-9_]', '', str( + finding.get('DetectorName') or finding.get('DetectorType') or '' + ).lower()) + service = DETECTOR_SERVICES.get(detector, '') + if service: + return service + extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {} + name = re.sub(r'[^a-z0-9_]', '', str(extra.get('name') or '').lower()) + return DETECTOR_SERVICES.get(name, '') + + +def _make_candidate( + service, candidate_kind, probe_material, raw_material, *, secret_text=None, + secret_json=None, endpoint='', principal='', metadata=None, +): + probe_material = _bounded(probe_material, MAX_CANDIDATE_SECRET_BYTES) + raw_material = _bounded(raw_material, MAX_CANDIDATE_SECRET_BYTES) + provider_key_hash = hashlib.sha256(probe_material.encode('utf-8')).hexdigest() + secret_hash = hashlib.sha256(raw_material.encode('utf-8')).hexdigest() + credential_hash = hashlib.sha256('|'.join(( + 'truf-credential-v2', service, probe_material, + )).encode('utf-8')).hexdigest() + metadata = dict(metadata or {}) + if len(_json(metadata).encode('utf-8')) > MAX_CANDIDATE_METADATA_BYTES: + raise ValueError('keycheck candidate metadata exceeds its byte bound') + return CandidateSpec( + service=service, + candidate_kind=candidate_kind, + credential_hash=credential_hash, + provider_key_hash=provider_key_hash, + secret_hash=secret_hash, + secret_text=secret_text, + secret_json=secret_json, + key_masked=_mask(probe_material), + endpoint=endpoint, + principal=principal, + metadata=metadata, + ) + + +def stored_provider_key_hash(service, candidate_kind, secret_text, secret_json, endpoint='', principal=''): + service = str(service or '').lower() + candidate_kind = str(candidate_kind or '') + secret_text = str(secret_text or '') + secret_json = str(secret_json or '') + endpoint = str(endpoint or '').lower() + principal = str(principal or '') + if service == 'gcp': + parsed = _json_object(secret_json) + probe = _provider_json(parsed) if parsed is not None else secret_json + elif service == 'azure' and candidate_kind == 'azure_service_principal': + parsed = _json_object(secret_json) or {} + probe = ':'.join(str(parsed.get(name) or '') for name in ('tenantId', 'clientId', 'clientSecret')) + if probe == '::': + probe = ':'.join(str(parsed.get(name) or '') for name in ('tenant_id', 'client_id', 'client_secret')) + elif service == 'azure' and candidate_kind == 'azure_container_registry': + parsed = _json_object(secret_json) or {} + probe = f"{parsed.get('username') or principal}:{parsed.get('password') or ''}" + elif service == 'azure' and endpoint: + probe = f'{endpoint}:{secret_text}' + elif service == 'dockerhub' and principal: + probe = f'{principal}:{secret_text}' + else: + probe = secret_text or secret_json + probe = _bounded(probe, MAX_CANDIDATE_SECRET_BYTES) + if not probe: + raise ValueError('stored keycheck credential has no provider probe material') + return hashlib.sha256(probe.encode('utf-8')).hexdigest() + + +def _json_object(value): + try: + parsed = json.loads(str(value or '')) + except (TypeError, ValueError): + return None + return parsed if isinstance(parsed, dict) else None + + +def _raw_material(finding): + value = finding.get('RawV2') or finding.get('Raw') + if value: + return str(value) + structured = finding.get('StructuredData') + return _json(structured) if isinstance(structured, dict) and structured else '' + + +def _azure_foundry_keyish(value): + text = re.sub(r'(?i)^bearer\s+', '', str(value or '').strip().strip('"\'`,;')).strip() + lowered = text.lower() + if not (20 <= len(text) <= 512) or any(character.isspace() for character in text): + return False + if any(marker in lowered for marker in ('http://', 'https://', '{{', '${', '<', 'azure.com')): + return False + if text.startswith(NON_FOUNDRY_KEY_PREFIXES): + return False + return bool(re.search(r'[A-Za-z]', text) and re.search(r'[0-9]', text)) + + +def extract_azure_foundry_parts(raw, raw_v2=''): + materials = [str(value or '') for value in (raw_v2, raw) if value] + endpoint = next(( + match.group(1).lower() + for value in materials + for match in [AZURE_FOUNDRY_ENDPOINT_RE.search(value)] + if match + ), '') + if not endpoint: + return None + for value in materials: + for key in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(value): + key = str(key).strip().strip('"\'`,;') + if _azure_foundry_keyish(key): + return {'key': key, 'endpoint': endpoint} + match = AZURE_FOUNDRY_ENDPOINT_RE.search(value) + if not match: + continue + before = value[:match.start()].strip(' \t\r\n:=,;\'"/') + after = value[match.end():].strip(' \t\r\n:=,;\'"/') + for key in (after, before): + if _azure_foundry_keyish(key): + return {'key': key, 'endpoint': endpoint} + return None + + +def extract_candidates(finding, attribution=None): + if not isinstance(finding, dict): + return + service = _service_for_finding(finding) + if not service: + return + detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '') + detector_key = re.sub(r'[^a-z0-9_]', '', detector.lower()) + postman = finding.get('PostmanContext') if isinstance(finding.get('PostmanContext'), dict) else {} + raw = str(finding.get('Raw') or '') + raw_v2 = str(finding.get('RawV2') or '') + raw_material = _raw_material(finding) + base_metadata = { + 'detector_name': detector, + 'finding_uid': str(finding.get('finding_uid') or ''), + } + context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {} + provider_candidates = [ + str(provider).lower() for provider in context.get('provider_candidates') or () + if str(provider).lower() in GENERIC_SK_PROVIDERS + ] + provider_hint = str(context.get('provider_hint') or '') + if provider_hint: + base_metadata['provider_hint'] = provider_hint + if provider_candidates: + base_metadata['provider_candidates'] = list(dict.fromkeys(provider_candidates)) + endpoint = str(postman.get('endpoint') or '') + principal = str(postman.get('principal') or postman.get('username') or '') + + if service == 'aws': + probe = raw_v2 or raw + if ':' in probe: + yield _make_candidate( + service, 'aws_access_key_pair', probe, raw_material, + secret_text=probe, metadata={**base_metadata, 'raw_v2': probe}, + ) + return + if service == 'gcp': + parsed = _json_object(raw_v2) + if parsed is None: + context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {} + parsed = _json_object(context.get('nearby')) + if parsed is not None: + probe = _provider_json(parsed) + yield _make_candidate( + service, 'gcp_json', probe, raw_material or probe, + secret_json=probe, metadata={**base_metadata, 'raw_v2': probe}, + ) + return + if service == 'azure': + parsed = _json_object(raw_v2) + if detector_key == 'azure': + if parsed: + tenant = parsed.get('tenantId') or parsed.get('tenant_id') + client = parsed.get('clientId') or parsed.get('client_id') + secret = parsed.get('clientSecret') or parsed.get('client_secret') + if tenant and client and secret: + probe = f'{tenant}:{client}:{secret}' + canonical_json = _json(parsed) + yield _make_candidate( + service, 'azure_service_principal', probe, raw_material, + secret_json=canonical_json, + metadata={**base_metadata, 'raw_v2': canonical_json}, + ) + return + if detector_key == 'azurecontainerregistry': + if parsed: + username = parsed.get('username') + password = parsed.get('password') + if username and password: + probe = f'{username}:{password}' + canonical_json = _json(parsed) + yield _make_candidate( + service, 'azure_container_registry', probe, raw_material, + secret_json=canonical_json, principal=str(username), + metadata={**base_metadata, 'raw_v2': canonical_json}, + ) + return + if detector_key == 'azureopenai': + match = re.match(r'^([a-fA-F0-9]{32}):(.+\.openai\.azure\.com)$', raw_v2) + key = match.group(1) if match else raw + found_endpoint = match.group(2).lower() if match else endpoint.lower() + if key: + probe = f'{found_endpoint}:{key}' if found_endpoint else key + provider_raw_v2 = f'{key}:{found_endpoint}' if found_endpoint else raw_v2 + yield _make_candidate( + service, 'azure_openai', probe, raw_material, + secret_text=key, endpoint=found_endpoint, + metadata={**base_metadata, 'raw_v2': provider_raw_v2}, + ) + return + foundry = extract_azure_foundry_parts(raw, raw_v2) + if foundry: + foundry_endpoint = foundry['endpoint'] + foundry_key = foundry['key'] + probe = f'{foundry_endpoint}:{foundry_key}' + yield _make_candidate( + service, 'azure_foundry', probe, raw_material, + secret_text=foundry_key, endpoint=foundry_endpoint, + metadata={**base_metadata, 'raw_v2': probe}, + ) + return + if service == 'dockerhub': + token_match = DOCKER_PAT_RE.search(raw_v2) or DOCKER_PAT_RE.search(raw) + if not token_match: + return + token = token_match.group(0) + username = '' + if ':' in raw_v2 and raw_v2.rsplit(':', 1)[-1] == token: + username = raw_v2.rsplit(':', 1)[0] + extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {} + analysis = finding.get('AnalysisInfo') if isinstance(finding.get('AnalysisInfo'), dict) else {} + username = username or str(extra.get('hub_username') or analysis.get('username') or '') + probe = f'{username}:{token}' if username else token + yield _make_candidate( + service, 'dockerhub_pat', probe, raw_material, + secret_text=token, principal=username, + metadata={**base_metadata, 'raw_v2': probe}, + ) + return + patterns = { + 'github': GITHUB_TOKEN_RE, + 'gitlab': GITLAB_TOKEN_RE, + 'gemini': GEMINI_KEY_RE, + 'qwen': QWEN_KEY_RE, + 'kimi': KIMI_KEY_RE, + 'zai': ZAI_KEY_RE, + 'provider_resolver': KIMI_KEY_RE, + } + pattern = patterns.get(service) + values = pattern.findall(raw + '\n' + raw_v2) if pattern else [raw or raw_v2] + seen = set() + for value in values: + value = str(value or '').strip() + if not value or value in seen: + continue + seen.add(value) + yield _make_candidate( + service, 'provider_key', value, raw_material or value, + secret_text=value, endpoint=endpoint, principal=principal, + metadata={**base_metadata, 'raw_v2': raw_v2}, + ) + + +def extract_structured_candidates(artifact_context, attribution=None): + if not isinstance(artifact_context, dict): + return + for finding in artifact_context.get('findings') or (): + yield from extract_candidates(finding, attribution) + contexts = artifact_context.get('contexts') or () + origin = str(artifact_context.get('origin') or 'structured-artifact') + endpoint_values = [] + for context in contexts: + if not isinstance(context, dict): + continue + text = ' '.join(str(context.get(key) or '') for key in ('value', 'endpoint', 'host', 'key')) + endpoint_values.extend(AZURE_OPENAI_ENDPOINT_RE.findall(text)) + endpoint_values.extend(AZURE_FOUNDRY_ENDPOINT_RE.findall(text)) + seen = set() + for index, context in enumerate(contexts): + if not isinstance(context, dict): + continue + value = str(context.get('value') or '').strip() + if not value or '{{' in value or '${' in value: + continue + path = str(context.get('path') or '') + text = ' '.join(( + value, str(context.get('endpoint') or ''), str(context.get('host') or ''), + str(context.get('key') or ''), + )) + for key in GEMINI_KEY_RE.findall(value): + identity = ('gemini', key) + if identity in seen: + continue + seen.add(identity) + yield _make_candidate( + 'gemini', 'structured_postman', key, key, secret_text=key, + metadata={ + 'detector_name': 'GoogleAIStudio', 'origin': origin, + 'structured_origin': f'{origin}:{path or index}', + }, + ) + azure_key = AZURE_OPENAI_KEY_RE.search(value) + if azure_key: + endpoints = AZURE_OPENAI_ENDPOINT_RE.findall(text) or [ + endpoint for endpoint in endpoint_values + if AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint) + ] + for endpoint in endpoints[:5]: + key = azure_key.group(0) + secret = f'{key}:{str(endpoint).lower()}' + identity = ('azure-openai', secret) + if identity in seen: + continue + seen.add(identity) + yield _make_candidate( + 'azure', 'structured_postman', f'{str(endpoint).lower()}:{key}', + secret, secret_text=key, endpoint=str(endpoint).lower(), metadata={ + 'detector_name': 'AzureOpenAI', 'origin': origin, + 'raw_v2': secret, + 'structured_origin': f'{origin}:{path or index}', + }, + ) + key_label = str(context.get('key') or '').lower() + normalized_key_label = re.sub(r'[^a-z0-9]+', '_', key_label).strip('_') + structured_provider = None + structured_detector = '' + structured_pattern = None + if normalized_key_label in ('dashscope_api_key', 'qwen_api_key'): + structured_provider = 'qwen' + structured_detector = 'QwenDashScope' + structured_pattern = QWEN_KEY_RE + elif normalized_key_label in ('moonshot_api_key', 'kimi_api_key'): + structured_provider = 'kimi' + structured_detector = 'KimiMoonshot' + structured_pattern = KIMI_KEY_RE + elif normalized_key_label in ( + 'zai_api_key', 'z_ai_api_key', 'glm_api_key', + 'zhipuai_api_key', 'bigmodel_api_key', + ): + structured_provider = 'zai' + structured_detector = 'ZaiGLM' + structured_pattern = ZAI_KEY_RE + if structured_provider and structured_pattern.fullmatch(value): + identity = (structured_provider, value) + if identity not in seen: + seen.add(identity) + yield _make_candidate( + structured_provider, 'structured_postman', value, value, + secret_text=value, metadata={ + 'detector_name': structured_detector, 'origin': origin, + 'structured_origin': f'{origin}:{path or index}', + }, + ) + foundry_endpoints = AZURE_FOUNDRY_ENDPOINT_RE.findall(text) + if ( + foundry_endpoints and 20 <= len(value) <= 512 + and not any(character.isspace() for character in value) + and any(token in key_label for token in ('key', 'token', 'secret', 'authorization')) + ): + for endpoint in foundry_endpoints[:5]: + secret = f'{str(endpoint).lower()}:{value}' + identity = ('azure-foundry', secret) + if identity in seen: + continue + seen.add(identity) + yield _make_candidate( + 'azure', 'structured_postman', secret, secret, + secret_text=value, endpoint=str(endpoint).lower(), metadata={ + 'detector_name': 'AzureFoundryEndpointBeforeKey', 'origin': origin, + 'raw_v2': secret, + 'structured_origin': f'{origin}:{path or index}', + }, + ) + + +def candidate_uid(scan_event_id, finding_uid_or_origin, service, credential_hash): + return hashlib.sha256('|'.join(( + 'truf-keycheck-candidate-v1', str(scan_event_id), str(finding_uid_or_origin), + str(service), str(credential_hash), + )).encode('utf-8')).hexdigest() diff --git a/app/keycheck_runner.py b/app/keycheck_runner.py new file mode 100644 index 0000000..54ad831 --- /dev/null +++ b/app/keycheck_runner.py @@ -0,0 +1,2842 @@ +import sys + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('keycheck runner could not disable bytecode writes') + +import argparse +import copy +import json +import math +import os +import re +import stat +import subprocess +import threading +import time +from collections import deque +from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait +from datetime import datetime, timedelta, timezone + +from paths import apply_path_config, default_project_paths +from owned_process import OwnedProcess +from scanner_db import ScannerDB, extract_finding_location, json_dumps, extract_raw_secret, redact_argv, sha256_text +from keycheckers.keycheck_common import ( + STATUS_TRANSACTION_JOURNAL_FILENAME, + checkpoint_matches_signature, + keycheck_status_group, + mask_secret, + private_atomic_writer, +) +from postgres_runtime import load_postgres_environment +from lifecycle_authority import ( + LifecycleAuthorityError, + require_active_supervisor_child, + supervised_child_environment, + strip_supervisor_credentials, +) +from runtime_security import ( + ensure_private_directory, + preflight_lifecycle_paths, + require_private_file, + require_sensitive_runtime_paths, +) + + +SERVICES = { + "anthropic": os.path.join("keycheckers", "anthropic", "anthropicKeycheck.py"), + "aws": os.path.join("keycheckers", "aws", "awsKeycheck.py"), + "azure": os.path.join("keycheckers", "azure", "azureKeycheck.py"), + "deepseek": os.path.join("keycheckers", "deepseek", "deepseekKeycheck.py"), + "dockerhub": os.path.join("keycheckers", "dockerhub", "dockerhubKeycheck.py"), + "gcp": os.path.join("keycheckers", "gcp", "gcpKeycheck.py"), + "gemini": os.path.join("keycheckers", "gemini", "geminiKeycheck.py"), + "groq": os.path.join("keycheckers", "groq", "groqKeycheck.py"), + "github": os.path.join("keycheckers", "github", "githubKeycheck.py"), + "gitlab": os.path.join("keycheckers", "gitlab", "gitlabKeycheck.py"), + "kimi": os.path.join("keycheckers", "kimi", "kimiKeycheck.py"), + "openai": os.path.join("keycheckers", "openai", "Keycheck.py"), + "openrouter": os.path.join("keycheckers", "openrouter", "OpenrouterKeycheck.py"), + "provider_resolver": os.path.join("keycheckers", "provider_resolver", "providerResolverKeycheck.py"), + "qwen": os.path.join("keycheckers", "qwen", "qwenKeycheck.py"), + "replicate": os.path.join("keycheckers", "replicate", "replicateKeycheck.py"), + "xai": os.path.join("keycheckers", "xai", "xaiKeycheck.py"), + "huggingface": os.path.join("keycheckers", "huggingface", "huggingfaceKeycheck.py"), + "zai": os.path.join("keycheckers", "zai", "zaiKeycheck.py"), +} +KEYCHECK_CAPACITY_BLOCKED_EXIT = 76 +_unconfirmed_provider_processes = [] + +SERVICE_CAPABILITIES = { + "anthropic": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_no_balance", "recheck_all"}}, + "aws": {"proxy": True, "flags": {"retry_network", "retry_unknown", "retry_valid", "recheck_all"}}, + "azure": {"proxy": True, "flags": {"retry_network", "retry_unknown", "retry_valid", "recheck_all"}}, + "deepseek": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_no_balance", "retry_valid", "recheck_all"}}, + "dockerhub": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "recheck_all"}}, + "gcp": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_valid", "recheck_all"}}, + "gemini": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_valid", "recheck_all"}}, + "groq": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "retry_no_balance", "retry_valid", "recheck_all"}}, + "github": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "recheck_all"}}, + "gitlab": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "recheck_all"}}, + "kimi": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "retry_no_balance", "retry_valid", "recheck_all"}}, + "openai": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "retry_no_balance", "recheck_all"}}, + "openrouter": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_no_balance", "retry_valid", "recheck_all"}}, + "provider_resolver": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "retry_no_balance", "retry_valid", "recheck_all"}}, + "qwen": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "retry_no_balance", "retry_valid", "recheck_all"}}, + "replicate": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "retry_no_balance", "retry_valid", "no_resource_probe", "recheck_all"}}, + "xai": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "retry_no_balance", "retry_valid", "recheck_all"}}, + "huggingface": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "recheck_all"}}, + "zai": {"proxy": True, "flags": {"retry_network", "retry_limited", "retry_unknown", "retry_restricted", "retry_no_balance", "retry_valid", "recheck_all"}}, +} + +RETRY_STATUS_DEFAULT_SUFFIXES = { + "retry_network": ("Network",), + "retry_limited": ("Limited",), + "retry_unknown": ("Unknown",), + "retry_restricted": ("Restricted",), + "retry_no_balance": ("NoBalance",), + "retry_valid": ("Alive",), +} +RETRY_STATUS_FILE_OVERRIDES = { + ("anthropic", "retry_no_balance"): ("anthropicNoQuota.txt",), + ("aws", "retry_valid"): ("awsAlive.txt", "awsBedrock.txt", "awsAdmin.txt"), + ("azure", "retry_network"): ( + "azureNetwork.txt", "azureOpenAIBadEndpoint.txt", "azureFoundryBadEndpoint.txt", + ), + ("azure", "retry_unknown"): ( + "azureUnknown.txt", "azureOpenAIUnresolved.txt", "azureFoundryUnresolved.txt", + ), + ("azure", "retry_valid"): ("azureAlive.txt", "azureFoundryLLM.txt"), + ("dockerhub", "retry_limited"): ("dockerhubRateLimited.txt",), + ("deepseek", "retry_unknown"): ("deepseekUnknown.txt", "deepseekNoContext.txt"), + ("gcp", "retry_limited"): ("gcpUnknown.txt",), + ("gcp", "retry_valid"): ("gcpAlive.txt", "gcpVertex.txt"), + ("gemini", "retry_limited"): ("geminiRateLimited.txt", "geminiAliveRateLimited.txt"), + ("gemini", "retry_valid"): ("geminiAlive.txt", "geminiAliveRateLimited.txt"), + ("huggingface", "retry_unknown"): ("huggingfaceUnknown.txt", "huggingfaceNoContext.txt"), + ("github", "retry_limited"): ("githubRateLimited.txt",), + ("gitlab", "retry_limited"): ("gitlabRateLimited.txt",), + ("openai", "retry_no_balance"): ("openaiLimited.txt",), + ("qwen", "retry_unknown"): ("qwenUnknown.txt", "qwenNoContext.txt"), + ("replicate", "retry_unknown"): ("replicateUnknown.txt", "replicateNoContext.txt"), + ("xai", "retry_unknown"): ("xaiUnknown.txt", "xaiNoContext.txt"), +} + +POSTGRES_STATUS_FILE_NAMES = { + "anthropic": { + "VALID": "anthropicAlive.txt", "NO_QUOTA": "anthropicNoQuota.txt", + "DEAD": "anthropicDead.txt", "LIMITED": "anthropicLimited.txt", + "RESTRICTED": "anthropicRestricted.txt", "NETWORK": "anthropicNetwork.txt", + "UNKNOWN": "anthropicUnknown.txt", + }, + "aws": { + "VALID": "awsAlive.txt", "BEDROCK": "awsBedrock.txt", "ADMIN": "awsAdmin.txt", + "CANARY": "awsCanary.txt", "QUARANTINED": "awsQuarantined.txt", + "ACCESS_DENIED": "awsAccessDenied.txt", "DEAD": "awsDead.txt", + "NETWORK": "awsNetwork.txt", "UNKNOWN": "awsUnknown.txt", + }, + "azure": { + "VALID": "azureAlive.txt", "DEAD": "azureDead.txt", + "RESTRICTED": "azureRestricted.txt", "NETWORK": "azureNetwork.txt", + "UNKNOWN": "azureUnknown.txt", "OPENAI_UNRESOLVED": "azureOpenAIUnresolved.txt", + "OPENAI_BAD_ENDPOINT": "azureOpenAIBadEndpoint.txt", "FOUNDRY": "azureFoundryLLM.txt", + "FOUNDRY_UNRESOLVED": "azureFoundryUnresolved.txt", + "FOUNDRY_BAD_ENDPOINT": "azureFoundryBadEndpoint.txt", + }, + "deepseek": { + "VALID": "deepseekAlive.txt", "NO_BALANCE": "deepseekNoBalance.txt", + "DEAD": "deepseekDead.txt", "LIMITED": "deepseekLimited.txt", + "NETWORK": "deepseekNetwork.txt", "NO_CONTEXT": "deepseekNoContext.txt", + "UNKNOWN": "deepseekUnknown.txt", + }, + "dockerhub": { + "VALID": "dockerhubAlive.txt", "VALID_2FA": "dockerhubAlive.txt", + "DEAD": "dockerhubDead.txt", "NO_USERNAME": "dockerhubNoUsername.txt", + "RATE_LIMITED": "dockerhubRateLimited.txt", "NETWORK": "dockerhubNetwork.txt", + "UNKNOWN": "dockerhubUnknown.txt", + }, + "gcp": { + "VALID": "gcpAlive.txt", "VERTEX": "gcpVertex.txt", "DEAD": "gcpDead.txt", + "RESTRICTED": "gcpRestricted.txt", "NETWORK": "gcpNetwork.txt", + "RATE_LIMITED": "gcpUnknown.txt", "UNKNOWN": "gcpUnknown.txt", + "NO_CONTEXT": "gcpNoContext.txt", + }, + "gemini": { + "VALID": "geminiAlive.txt", "VALID_RATE_LIMITED": "geminiAliveRateLimited.txt", + "INVALID": "geminiDead.txt", "EXPIRED": "geminiExpired.txt", + "LEAKED_REVOKED": "geminiLeaked.txt", "API_DISABLED": "geminiDisabled.txt", + "RESTRICTED": "geminiRestricted.txt", "RATE_LIMITED": "geminiRateLimited.txt", + "NETWORK_ERROR": "geminiNetwork.txt", "UNKNOWN": "geminiUnknown.txt", + }, + "groq": { + "VALID": "groqAlive.txt", "NO_BALANCE": "groqNoBalance.txt", + "DEAD": "groqDead.txt", "RESTRICTED": "groqRestricted.txt", + "LIMITED": "groqLimited.txt", "NO_CONTEXT": "groqNoContext.txt", + "NETWORK": "groqNetwork.txt", "UNKNOWN": "groqUnknown.txt", + }, + "github": { + "VALID": "githubAlive.txt", "DEAD": "githubDead.txt", + "RESTRICTED": "githubRestricted.txt", "RATE_LIMITED": "githubRateLimited.txt", + "NETWORK": "githubNetwork.txt", "UNKNOWN": "githubUnknown.txt", + "REFRESH_TOKEN": "githubRefreshToken.txt", + }, + "gitlab": { + "VALID": "gitlabAlive.txt", "DEAD": "gitlabDead.txt", + "RESTRICTED": "gitlabRestricted.txt", "RATE_LIMITED": "gitlabRateLimited.txt", + "NETWORK": "gitlabNetwork.txt", "UNKNOWN": "gitlabUnknown.txt", + }, + "kimi": { + "VALID": "kimiAlive.txt", "NO_BALANCE": "kimiNoBalance.txt", + "DEAD": "kimiDead.txt", "RESTRICTED": "kimiRestricted.txt", + "LIMITED": "kimiLimited.txt", "NETWORK": "kimiNetwork.txt", + "UNKNOWN": "kimiUnknown.txt", + }, + "openai": { + "ALIVE": "openaiAlive.txt", "DEAD": "openaiDead.txt", + "INVALID_OR_REVOKED": "openaiDead.txt", "NETWORK_ERROR": "openaiNetwork.txt", + "LIMITED": "openaiLimited.txt", "LIMITED_OR_NO_BALANCE": "openaiLimited.txt", + "RESTRICTED": "openaiRestricted.txt", "UNKNOWN": "openaiUnknown.txt", + "NO_TARGET_MODELS": "openaiNoTarget.txt", + }, + "openrouter": { + "VALID": "openrouterAlive.txt", "NO_BALANCE": "openrouterNoBalance.txt", + "DEAD": "openrouterDead.txt", "LIMITED": "openrouterLimited.txt", + "NETWORK": "openrouterNetwork.txt", "UNKNOWN": "openrouterUnknown.txt", + }, + "provider_resolver": { + "VALID": "providerResolverAlive.txt", "NO_BALANCE": "providerResolverNoBalance.txt", + "DEAD": "providerResolverDead.txt", "RESTRICTED": "providerResolverRestricted.txt", + "LIMITED": "providerResolverLimited.txt", "NETWORK": "providerResolverNetwork.txt", + "NO_CONTEXT": "providerResolverNoContext.txt", "UNKNOWN": "providerResolverUnknown.txt", + }, + "qwen": { + "VALID": "qwenAlive.txt", "DEAD": "qwenDead.txt", + "RESTRICTED": "qwenRestricted.txt", "LIMITED": "qwenLimited.txt", + "NO_BALANCE": "qwenNoBalance.txt", "NO_CONTEXT": "qwenNoContext.txt", + "NETWORK": "qwenNetwork.txt", "UNKNOWN": "qwenUnknown.txt", + }, + "replicate": { + "VALID": "replicateAlive.txt", "DEAD": "replicateDead.txt", + "RESTRICTED": "replicateRestricted.txt", "LIMITED": "replicateLimited.txt", + "NO_BALANCE": "replicateNoBalance.txt", "NETWORK": "replicateNetwork.txt", + "NO_CONTEXT": "replicateNoContext.txt", + "UNKNOWN": "replicateUnknown.txt", + }, + "xai": { + "VALID": "xaiAlive.txt", "NO_BALANCE": "xaiNoBalance.txt", + "DEAD": "xaiDead.txt", "RESTRICTED": "xaiRestricted.txt", + "LIMITED": "xaiLimited.txt", "NO_CONTEXT": "xaiNoContext.txt", + "NETWORK": "xaiNetwork.txt", "UNKNOWN": "xaiUnknown.txt", + }, + "huggingface": { + "VALID": "huggingfaceAlive.txt", "DEAD": "huggingfaceDead.txt", + "RESTRICTED": "huggingfaceRestricted.txt", "LIMITED": "huggingfaceLimited.txt", + "NETWORK": "huggingfaceNetwork.txt", "NO_CONTEXT": "huggingfaceNoContext.txt", + "UNKNOWN": "huggingfaceUnknown.txt", + }, + "zai": { + "VALID": "zaiAlive.txt", "NO_BALANCE": "zaiNoBalance.txt", + "DEAD": "zaiDead.txt", "RESTRICTED": "zaiRestricted.txt", + "LIMITED": "zaiLimited.txt", "NETWORK": "zaiNetwork.txt", + "UNKNOWN": "zaiUnknown.txt", + }, +} + +POSTGRES_AUXILIARY_STATUS_FILE_NAMES = { + "gcp": ("gcpVertexGemini.txt", "gcpVertexAnthropic.txt"), + "azure": ("azureOpenAILLM.txt",), +} +LEGACY_GCP_VERTEX_STATUS_FILES = ("gcpVertexGemini.txt", "gcpVertexAnthropic.txt") + + +def load_config(config_path): + if not config_path: + return {"global": default_project_paths(), "keychecks": {}} + try: + import yaml + except ImportError as e: + raise SystemExit("PyYAML is required for --config") from e + with open(config_path, "r", encoding="utf-8") as f: + return apply_path_config(yaml.safe_load(f) or {}, config_path) + + +def load_layout(config_path): + return (load_config(config_path).get("global") or {}) + + +def load_postgres_env(config_path, layout): + return load_postgres_environment(config_path, {'global': layout or {}}) + + +def service_list(value): + if isinstance(value, (list, tuple)): + value = ",".join(str(item) for item in value) + if value == "all": + return list(SERVICES) + return [item.strip().lower() for item in value.split(",") if item.strip()] + + +def count_nonempty_lines(path): + if not os.path.exists(path): + return 0 + with open(path, "r", encoding="utf-8", errors="replace") as f: + return sum(1 for line in f if line.strip()) + + +def file_line_stats(path, unique_keys=False): + # Operational summary should be fast even with large historical files. + if not unique_keys: + count = 0 + tail = b'' + with open(path, "rb") as f: + while True: + chunk = f.read(1024 * 1024) + if not chunk: + break + count += chunk.count(b"\n") + tail = chunk + try: + if os.path.getsize(path) > 0 and not tail.endswith(b"\n"): + count += 1 + except OSError: + pass + return count, None + # Avoid expensive unique parsing on very large checked files; line count is enough for dashboard health. + if os.path.getsize(path) > 5 * 1024 * 1024: + count, _ = file_line_stats(path, unique_keys=False) + return count, count + line_count = 0 + keys = set() + with open(path, "r", encoding="utf-8", errors="replace") as f: + for line in f: + if not line.strip(): + continue + line_count += 1 + key = status_key(line) + if key: + keys.add(key) + return line_count, len(keys) + + +def status_key(line): + line = line.strip() + if not line: + return None + if "\t" in line: + return line.split("\t", 1)[0].strip() + return line.split(":", 1)[0].strip() + + +def _row_value(row, name, default=None): + try: + value = row[name] + except (KeyError, TypeError): + value = getattr(row, name, default) + return default if value is None else value + + +def _projection_metadata(row): + value = _row_value(row, "metadata_json", "") + if isinstance(value, dict): + return value + try: + parsed = json.loads(str(value or "{}")) + except (TypeError, ValueError, json.JSONDecodeError): + return {} + return parsed if isinstance(parsed, dict) else {} + + +def _bounded_projection_text(value, max_bytes): + text = str(value or "").replace("\x00", "").replace("\t", " ").replace("\r", " ").replace("\n", " ") + encoded = text.encode("utf-8", errors="strict") + if len(encoded) <= max_bytes: + return text + return encoded[:max_bytes].decode("utf-8", errors="ignore") + + +def postgres_status_projection_detail(metadata, max_bytes=16000): + metadata = metadata if isinstance(metadata, dict) else {} + probe = metadata.get("probe") if isinstance(metadata.get("probe"), dict) else {} + probe_model = metadata.get("llm_probe_model") or probe.get("model") or "" + probe_status = metadata.get("llm_probe_status") or probe.get("status") or "" + inventory = metadata.get("model_inventory") + if not isinstance(inventory, list): + inventory = metadata.get("models") if isinstance(metadata.get("models"), list) else [] + models = [] + seen = set() + for value in inventory[:500]: + model = str(value or "").replace("\x00", "").replace("\t", " ").replace("\r", " ").replace("\n", " ")[:200] + if model and model not in seen: + seen.add(model) + models.append(model) + + parts = [] + message = str(metadata.get("message") or "").strip() + if message: + parts.append(message[:1000]) + if probe_model: + parts.append(f"probe_model={probe_model}") + if probe_status: + parts.append(f"probe_status={probe_status}") + model_count = metadata.get("model_count") + if model_count is not None: + parts.append(f"model_count={model_count}") + if models: + parts.append(f"models={','.join(models)}") + return _bounded_projection_text("; ".join(parts), max(0, int(max_bytes))) + + +def _metadata_enabled(metadata, name): + value = metadata.get(name) + if isinstance(value, bool): + return value + return str(value or "").strip().lower() in ("1", "true", "yes", "on") + + +def postgres_status_projection_targets(service, status, metadata=None): + service = str(service or "").strip().lower() + status = str(status or "UNKNOWN").strip().upper() + filename = (POSTGRES_STATUS_FILE_NAMES.get(service) or {}).get(status) + if not filename: + return [] + targets = [filename] + metadata = metadata or {} + if service == "gcp" and status == "VERTEX": + if _metadata_enabled(metadata, "vertex_google_enabled"): + targets.append("gcpVertexGemini.txt") + if _metadata_enabled(metadata, "vertex_anthropic_enabled"): + targets.append("gcpVertexAnthropic.txt") + if ( + service == "azure" and status == "VALID" + and int(metadata.get("deployment_count") or 0) > 0 + and (not metadata.get("route_probe") or metadata.get("route_probe") == "accepted_auth_route") + ): + targets.append("azureOpenAILLM.txt") + return list(dict.fromkeys(targets)) + + +def _read_status_projection(path, max_items, max_bytes, max_line_bytes): + if not os.path.exists(path): + return [] + require_private_file(path) + if os.path.getsize(path) > max_bytes: + raise RuntimeError(f"keycheck status projection exceeds its byte bound: {path}") + lines = [] + total = 0 + with open(path, "rb") as handle: + for raw_line in handle: + if len(raw_line) > max_line_bytes: + raise RuntimeError(f"keycheck status projection line exceeds its byte bound: {path}") + if not raw_line.strip(): + continue + total += len(raw_line) + if total > max_bytes or len(lines) >= max_items: + raise RuntimeError(f"keycheck status projection exceeds its aggregate bound: {path}") + lines.append(raw_line.decode("utf-8", errors="replace").rstrip("\r\n") + "\n") + return lines + + +def _write_status_projection(path, lines, max_items, max_bytes): + if len(lines) > max_items: + raise RuntimeError(f"keycheck status projection exceeds its item bound: {path}") + encoded_bytes = sum(len(line.encode("utf-8", errors="strict")) for line in lines) + if encoded_bytes > max_bytes: + raise RuntimeError(f"keycheck status projection exceeds its byte bound: {path}") + with private_atomic_writer(path, suffix=".status-projection.tmp") as handle: + handle.writelines(lines) + + +def project_postgres_status_files(layout, services, rows, managed_rows=None): + keycheck_dir = layout["keycheck_dir"] + ensure_private_directory(keycheck_dir, reject_reparse=True) + selected = {str(service or "").strip().lower() for service in services} + max_items = max(1, env_int("KEYCHECK_STATUS_PROJECTION_MAX_ITEMS", 100000)) + max_bytes = max(1024, env_int("KEYCHECK_STATUS_PROJECTION_MAX_BYTES", 32 * 1024 * 1024)) + max_line_bytes = max(1024, env_int("KEYCHECK_INPUT_MAX_LINE_BYTES", 16 * 1024 * 1024)) + rows = list(rows or []) + if len(rows) > max_items: + raise RuntimeError("PostgreSQL keycheck status projection exceeds its row bound") + + grouped = {service: [] for service in selected if service in POSTGRES_STATUS_FILE_NAMES} + managed_by_service = {service: set() for service in grouped} + for row in managed_rows or (): + service = str(_row_value(row, "service", "") or "").strip().lower() + if service not in managed_by_service: + continue + key = str( + _row_value(row, "secret_text", "") + or _row_value(row, "secret_json", "") + or "" + ) + if ( + key and not any(character in key for character in ("\x00", "\t", "\r", "\n")) + and len(key.encode("utf-8", errors="strict")) <= max_line_bytes + ): + managed_by_service[service].add(key) + skipped = 0 + for row in rows: + service = str(_row_value(row, "service", "") or "").strip().lower() + if service not in grouped: + continue + key = str( + _row_value(row, "secret_text", "") + or _row_value(row, "secret_json", "") + or "" + ) + status = str(_row_value(row, "status", "UNKNOWN") or "UNKNOWN").strip().upper() + metadata = _projection_metadata(row) + targets = postgres_status_projection_targets(service, status, metadata) + if ( + not key or not targets + or any(character in key for character in ("\x00", "\t", "\r", "\n")) + or len(key.encode("utf-8", errors="strict")) > max_line_bytes + ): + skipped += 1 + continue + grouped[service].append({ + "key": key, + "status": status, + "checked_at": str(_row_value(row, "checked_at", "") or ""), + "message": postgres_status_projection_detail( + metadata, + min(16000, max(0, max_line_bytes - len(key.encode("utf-8", errors="strict")) - 128)), + ), + "targets": targets, + }) + + changed_files = 0 + projected_rows = 0 + for service, service_rows in grouped.items(): + service_dir = os.path.join(keycheck_dir, service) + ensure_private_directory(service_dir, reject_reparse=True) + filenames = list(dict.fromkeys([ + *(POSTGRES_STATUS_FILE_NAMES[service].values()), + *POSTGRES_AUXILIARY_STATUS_FILE_NAMES.get(service, ()), + ])) + paths = {name: os.path.join(service_dir, name) for name in filenames} + snapshots = { + name: _read_status_projection(path, max_items, max_bytes, max_line_bytes) + for name, path in paths.items() + } + managed_keys = managed_by_service[service] | {row["key"] for row in service_rows} + rewritten = { + name: [line for line in lines if status_key(line) not in managed_keys] + for name, lines in snapshots.items() + } + for row in service_rows: + status_line = f'{row["key"]}\t{row["status"]}\t{row["message"]}\tpostgres-authoritative\n' + for filename in row["targets"]: + rewritten[filename].append(status_line) + + checked_name = f"{service}Checked.txt" + checked_path = os.path.join(service_dir, checked_name) + checked_snapshot = _read_status_projection( + checked_path, max_items, max_bytes, max_line_bytes, + ) + checked_rewritten = [ + line for line in checked_snapshot if status_key(line) not in managed_keys + ] + checked_rewritten.extend( + f'{row["key"]}\t{row["status"]}\t{row["checked_at"]}\n' + for row in service_rows + ) + + for name, lines in rewritten.items(): + if lines == snapshots[name]: + continue + _write_status_projection(paths[name], lines, max_items, max_bytes) + changed_files += 1 + if checked_rewritten != checked_snapshot: + _write_status_projection(checked_path, checked_rewritten, max_items, max_bytes) + changed_files += 1 + projected_rows += len(service_rows) + return { + "projected_rows": projected_rows, + "changed_files": changed_files, + "skipped_rows": skipped, + } + + +def count_unique_status_keys(path): + if not os.path.exists(path): + return 0 + with open(path, "r", encoding="utf-8", errors="replace") as f: + return len({key for key in (status_key(line) for line in f) if key}) + + +def utc_now_iso(): + return datetime.now(timezone.utc).isoformat(timespec="seconds") + + +def env_int(name, default): + try: + return int(os.getenv(name, default)) + except (TypeError, ValueError): + return default + + +def truthy_env(name, default=False): + value = os.getenv(name) + if value is None: + return default + return str(value).strip().lower() in ("1", "true", "yes", "on") + + +def file_mtime_iso(path): + try: + return datetime.fromtimestamp(os.path.getmtime(path), timezone.utc).isoformat(timespec="seconds") + except OSError: + return utc_now_iso() + + +def status_from_payload(service, payload): + status = str(payload.get("status") or "UNKNOWN").upper() + if service == "gemini" and status == "VALID": + probe = payload.get("probe") if isinstance(payload.get("probe"), dict) else {} + if str(probe.get("status") or "").upper() == "RATE_LIMITED": + return "VALID_RATE_LIMITED" + return status + + +def parse_source_line(source_line): + text = str(source_line or "") + if ":plain:" in text: + return None, None + marker = ":byte:" + if marker in text: + path, byte_text = text.rsplit(marker, 1) + if path and byte_text.isdigit(): + return path, ("byte", int(byte_text)) + path, sep, line_text = text.rpartition(":") + if not sep or not path or not line_text.isdigit(): + return None, None + return path, int(line_text) + + +def load_jsonl_line(path, line_number, cache): + if not path or not line_number or not os.path.exists(path): + return None + path = os.path.normpath(path) + max_line_bytes = max(1024, env_int("KEYCHECK_INPUT_MAX_LINE_BYTES", 16 * 1024 * 1024)) + if isinstance(line_number, tuple) and line_number[0] == "byte": + try: + with open(path, "rb") as f: + f.seek(int(line_number[1])) + raw_line = f.readline(max_line_bytes + 1) + if len(raw_line) > max_line_bytes or not raw_line.endswith(b"\n"): + return None + line = raw_line.decode("utf-8", errors="replace") + return json.loads(line) + except (OSError, ValueError, json.JSONDecodeError): + return None + try: + wanted = int(line_number) + except (TypeError, ValueError): + return None + max_line_number = max(1, env_int("KEYCHECK_FINDING_LINE_MAX_NUMBER", 10000000)) + if wanted <= 0 or wanted > max_line_number: + return None + cache_key = (path, wanted) + line = cache.get(cache_key) + if line is None: + try: + with open(path, "rb") as handle: + raw_line = b'' + for _ in range(wanted): + raw_line = handle.readline(max_line_bytes + 1) + if not raw_line or len(raw_line) > max_line_bytes: + return None + if not raw_line.endswith(b"\n"): + return None + line = raw_line.decode("utf-8", errors="replace") + except OSError: + return None + max_cache_rows = max(1, env_int("KEYCHECK_FINDING_LINE_CACHE_ITEMS", 128)) + max_cache_bytes = max(1024, env_int("KEYCHECK_FINDING_LINE_CACHE_BYTES", 16 * 1024 * 1024)) + line_bytes = len(line.encode("utf-8", errors="replace")) + while cache and ( + len(cache) >= max_cache_rows + or sum(len(value.encode("utf-8", errors="replace")) for value in cache.values()) + line_bytes > max_cache_bytes + ): + cache.pop(next(iter(cache))) + if line_bytes <= max_cache_bytes: + cache[cache_key] = line + try: + return json.loads(line) + except json.JSONDecodeError: + return None + + +def finding_from_payload(payload, line_cache): + finding = payload.get("finding") + if isinstance(finding, dict) and finding: + return finding + path, line_number = parse_source_line(payload.get("source")) + return load_jsonl_line(path, line_number, line_cache) + + +def safe_metadata(payload): + out = {} + for key, value in payload.items(): + lowered = str(key).lower() + if lowered in ("finding", "raw", "raw_v2", "rawsecret", "raw_secret"): + continue + out[key] = value + return out + + +def infer_finding_attribution(finding): + if not isinstance(finding, dict): + return {} + metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {} + data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {} + detector = finding.get("DetectorName") or finding.get("DetectorType") or "" + for source_type, details in data.items(): + if not isinstance(details, dict): + continue + lowered = str(source_type or "").lower() + repo = str(details.get("repository") or details.get("repo") or "") + target = repo or details.get("image") or details.get("link") or "" + source = "" + query = None + if lowered == "docker" or details.get("image"): + source = "dockerhub" + target = details.get("image") or target + elif lowered == "huggingface" or "huggingface.co/" in repo: + source = "huggingface" + query = "spaces" if "/spaces/" in repo or details.get("resource_type") == "space" else None + elif "gitlab.com" in repo: + source = "gitlab" + elif "github.com" in repo: + source = "github" + elif lowered == "git": + source = "git" + if source: + return { + "source": source, + "query": query, + "target": target, + "detector_name": detector, + "found_at": details.get("timestamp") or "", + } + return {"detector_name": detector} + + +def sync_keycheck_results_to_db(layout, services, reset=False): + if reset: + raise SystemExit('--reset-keycheck-results is retired; online keycheck result deletion is not supported') + db_path = layout.get("database_path") + db_url = layout.get("database_url") or os.getenv("SCANNER_DB_URL") or os.getenv("DATABASE_URL") + keycheck_dir = layout.get("keycheck_dir") + if not db_path and not db_url: + raise SystemExit("global.database_path or global.database_url is required for keycheck DB sync") + if not keycheck_dir or not os.path.isdir(keycheck_dir): + raise SystemExit(f"keycheck_dir not found: {keycheck_dir}") + + db = ScannerDB(db_path=db_path, db_url=db_url, initialize=False) + if not db.enabled: + raise SystemExit(f"Unable to open scanner DB: {db.db_display or db_path or 'configured database'}") + try: + db.require_runtime_safety_schema() + except Exception as exc: + db.close() + raise SystemExit(f'Keycheck DB schema is incomplete; offline migration required: {exc}') from exc + line_cache = {} + inserted = 0 + skipped = 0 + missing_finding = 0 + service_names = services if services and services != ["all"] else sorted(name for name in os.listdir(keycheck_dir) if os.path.isdir(os.path.join(keycheck_dir, name))) + + for service in service_names: + service_dir = os.path.join(keycheck_dir, service) + if not os.path.isdir(service_dir): + continue + result_files = [name for name in os.listdir(service_dir) if name.lower().endswith("results.jsonl")] + for name in result_files: + path = os.path.join(service_dir, name) + checked_at = file_mtime_iso(path) + with open(path, "r", encoding="utf-8", errors="replace") as f: + for line in f: + if not line.strip(): + continue + try: + payload = json.loads(line) + except json.JSONDecodeError: + skipped += 1 + continue + status = status_from_payload(service, payload) + finding = finding_from_payload(payload, line_cache) + if not finding: + missing_finding += 1 + raw_secret = extract_raw_secret(finding or {}) + finding_hash = sha256_text(raw_secret) if raw_secret else "" + key_hash = payload.get("key_hash") or finding_hash + secret_hash = finding_hash or payload.get("secret_hash") or key_hash + key_masked = payload.get("key_masked") or mask_secret(raw_secret) + if not key_hash and not key_masked: + skipped += 1 + continue + detector = payload.get("detector") or (finding or {}).get("DetectorName") or "" + detector_secret_hash = payload.get("detector_secret_hash") or (sha256_text("|".join([str(detector or ""), secret_hash])) if secret_hash or detector else "") + finding_uid = payload.get("finding_uid") or (finding or {}).get("finding_uid") or "" + event_id = payload.get("event_id") or "" + message = payload.get("message") or "" + error = payload.get("error") + if not message and isinstance(error, dict): + message = error.get("message") or json.dumps(error, ensure_ascii=False)[:1000] + metadata = safe_metadata(payload) + metadata["backfill_source_file"] = path + metadata["finding_uid"] = finding_uid + metadata["event_id"] = event_id + matches = [None] + now = utc_now_iso() + if event_id: + duplicate_event = db.conn.execute( + "SELECT id FROM keycheck_results WHERE event_id = ? LIMIT 1", + (event_id,), + ).fetchone() + if duplicate_event: + skipped += 1 + continue + duplicate = db.conn.execute(''' + SELECT id FROM keycheck_results + WHERE service = ? + AND status = ? + AND checked_at = ? + AND COALESCE(source_line, '') = ? + AND ( + (key_hash != '' AND key_hash = ?) + OR (key_hash = '' AND key_masked = ?) + ) + LIMIT 1 + ''', ( + service, + status, + payload.get("checked_at") or checked_at, + payload.get("source") or "", + key_hash, + key_masked, + )).fetchone() + if duplicate: + skipped += 1 + continue + for row in matches: + keycheck_result_id = db.conn.insert_returning_id( + '''INSERT INTO keycheck_results ( + service, status, status_group, checked_at, key_hash, secret_hash, key_masked, + finding_id, target_scan_id, cycle_id, run_id, source, query, target, detector_name, + found_at, message, metadata_json, source_line, detector_secret_hash, + event_id, finding_uid, link_status, linked_at, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + service, + status, + keycheck_status_group(status), + payload.get("checked_at") or checked_at, + key_hash, + secret_hash, + key_masked, + row["id"] if row else None, + row["target_scan_id"] if row else None, + row["cycle_id"] if row else None, + row["run_id"] if row else None, + row["source"] if row else None, + row["query"] if row else None, + row["target"] if row else None, + row["detector_name"] if row else detector, + row["created_at"] if row else None, + str(message or "")[:1000].replace("\n", " "), + json.dumps(metadata, ensure_ascii=False, default=str, sort_keys=True), + payload.get("source") or "", + detector_secret_hash, + event_id, + finding_uid, + "linked" if row else "pending", + now if row else None, + now, + ), + ) + if event_id and keycheck_result_id: + db.conn.execute( + '''INSERT INTO keycheck_event_map (event_id, keycheck_result_id, created_at) + VALUES (?, ?, ?) + ON CONFLICT(event_id) DO NOTHING''', + (event_id, keycheck_result_id, now), + ) + inserted += 1 + if inserted % 1000 == 0: + db.conn.commit() + db.conn.commit() + db_display = db.db_display + db.close() + print(f"keycheck_results synced: inserted={inserted} skipped={skipped} missing_finding={missing_finding} db={db_display}") + return inserted + + +def set_linker_timeouts(db): + if getattr(db.conn, "is_postgres", False): + db.conn.execute("SELECT set_config('statement_timeout', ?, false)", (f"{int(os.getenv('KEYCHECK_LINK_STATEMENT_TIMEOUT_MS', '10000'))}ms",)) + db.conn.execute("SELECT set_config('lock_timeout', ?, false)", (f"{int(os.getenv('KEYCHECK_LINK_LOCK_TIMEOUT_MS', '2000'))}ms",)) + + +def update_link_status(db, row_id, status, error=""): + db.conn.execute( + '''UPDATE keycheck_results + SET link_status = ?, link_attempts = COALESCE(link_attempts, 0) + 1, + linked_at = CASE WHEN ? IN ('linked', 'source_only') THEN ? ELSE linked_at END, + link_error = ? + WHERE id = ?''', + (status, status, utc_now_iso(), str(error or '')[:1000], row_id), + ) + + +def backfill_keycheck_event_map(db, limit=5000): + limit = max(0, int(limit or 0)) + if limit <= 0: + return 0 + if not db.conn.table_exists("keycheck_event_map"): + return 0 + rows = db.conn.execute( + '''SELECT id, event_id, created_at + FROM ( + SELECT id, event_id, created_at + FROM keycheck_results + ORDER BY id DESC + LIMIT ? + ) recent + WHERE event_id IS NOT NULL AND event_id != '' + ORDER BY id DESC''', + (limit,), + ).fetchall() + inserted = 0 + for row in rows: + cur = db.conn.execute( + '''INSERT INTO keycheck_event_map (event_id, keycheck_result_id, created_at) + VALUES (?, ?, ?) + ON CONFLICT(event_id) DO NOTHING''', + (row["event_id"], row["id"], row["created_at"] or utc_now_iso()), + ) + if getattr(cur, "rowcount", 0) > 0: + inserted += 1 + db.conn.commit() + return inserted + + +def backfill_finding_uid_map(db, limit=5000): + limit = max(0, int(limit or 0)) + if limit <= 0: + return 0 + if not db.conn.table_exists("finding_uid_map"): + return 0 + rows = db.conn.execute( + '''SELECT id, finding_uid, created_at + FROM ( + SELECT id, finding_uid, created_at + FROM findings + ORDER BY id DESC + LIMIT ? + ) recent + WHERE finding_uid IS NOT NULL AND finding_uid != '' + ORDER BY id DESC''', + (limit,), + ).fetchall() + inserted = 0 + for row in rows: + cur = db.conn.execute( + '''INSERT INTO finding_uid_map (finding_uid, finding_id, created_at) + VALUES (?, ?, ?) + ON CONFLICT(finding_uid) DO NOTHING''', + (row["finding_uid"], row["id"], row["created_at"] or utc_now_iso()), + ) + if getattr(cur, "rowcount", 0) > 0: + inserted += 1 + db.conn.commit() + return inserted + + +def read_json_file(path, default=None): + try: + with open(path, "r", encoding="utf-8") as f: + return json.load(f) + except (OSError, ValueError): + return default + + +def write_json_file(path, data): + with private_atomic_writer(path) as f: + json.dump(data, f, ensure_ascii=False, indent=2, sort_keys=True) + + +def jsonl_manifest_path(path): + base, _ = os.path.splitext(path) + return f"{base}.manifest.json" + + +def keycheck_result_paths(current_path): + manifest_path = jsonl_manifest_path(current_path) + if os.path.exists(manifest_path): + if os.path.getsize(manifest_path) > 1024 * 1024: + raise RuntimeError(f"keycheck result manifest exceeds bounded size: {manifest_path}") + with open(manifest_path, "r", encoding="utf-8") as handle: + manifest = json.load(handle) + if not isinstance(manifest, dict): + raise RuntimeError(f"invalid keycheck result manifest: {manifest_path}") + else: + manifest = {} + root = os.path.dirname(os.path.abspath(current_path)) + base, extension = os.path.splitext(os.path.basename(current_path)) + pattern = re.compile(rf"^{re.escape(base)}\.(\d{{6}}){re.escape(extension)}$") + physical = [] + with os.scandir(root) as entries: + for entry in entries: + match = pattern.fullmatch(entry.name) + if not match: + continue + if entry.is_symlink() or not entry.is_file(follow_symlinks=False): + raise RuntimeError(f"unsafe keycheck result segment: {entry.path}") + physical.append((int(match.group(1)), os.path.abspath(entry.path))) + physical.sort() + paths = [path for _, path in physical] + listed = { + os.path.basename(str(segment.get("path") or segment.get("name") or "")) + for segment in (manifest.get("segments") or []) if isinstance(segment, dict) + } + present = {os.path.basename(path) for path in paths} + missing = sorted(name for name in listed if name and name not in present) + if missing: + raise RuntimeError(f"manifest-listed keycheck result segment is missing: {missing[0]}") + declared_current = os.path.abspath(manifest.get("current_path") or current_path) + if declared_current != os.path.abspath(current_path): + raise RuntimeError("keycheck result manifest current path mismatch") + if os.path.exists(current_path): + paths.append(os.path.abspath(current_path)) + if not paths and os.path.exists(current_path): + paths.append(os.path.abspath(current_path)) + seen = set() + output = [] + for path in paths: + if path not in seen: + seen.add(path) + output.append(path) + return output + + +def keycheck_result_current_skip(current_path): + # A manifest skip can become stale between validation and file read. Event + # IDs make duplicate replay safe, while skipping a new row is not safe. + return 0 + + +def service_result_paths(service_dir): + output = [] + for name in sorted(os.listdir(service_dir)): + if not name.lower().endswith("results.jsonl"): + continue + path = os.path.join(service_dir, name) + if os.path.isfile(path): + output.extend(keycheck_result_paths(path)) + seen = set() + deduped = [] + for path in output: + if path not in seen: + seen.add(path) + deduped.append(path) + return deduped + + +def keycheck_ingest_tail_bytes(): + value = os.getenv("KEYCHECK_DB_INGEST_TAIL_MB", "0") + try: + return max(0, int(float(value) * 1024 * 1024)) + except (TypeError, ValueError): + return 64 * 1024 * 1024 + + +def file_signature(path, handle=None): + details = os.fstat(handle.fileno()) if handle is not None else os.stat(path) + return { + "path": os.path.abspath(path), + "device": int(getattr(details, "st_dev", 0) or 0), + "size": int(details.st_size), + "mtime_ns": int(getattr(details, "st_mtime_ns", int(details.st_mtime * 1_000_000_000))), + "inode": int(getattr(details, "st_ino", 0) or 0), + } + + +def ingest_start_offset(path, file_state, signature): + file_state = file_state or {} + if checkpoint_matches_signature(path, file_state, signature): + offset = int(file_state.get("offset", 0) or 0) + if 0 <= offset <= signature["size"]: + return offset, "offset" + return 0, "full_replay" if file_state else "full_initial" + + +def finding_uid_lookup_available(db): + if str(os.getenv("KEYCHECK_FINDING_UID_LOOKUP", "1")).strip().lower() in ("0", "false", "no", "off"): + return False + if str(os.getenv("KEYCHECK_ASSUME_FINDING_UID_INDEX", "")).strip().lower() in ("1", "true", "yes", "on"): + return True + try: + if db.conn.table_exists("finding_uid_map"): + return True + if getattr(db.conn, "is_postgres", False): + row = db.conn.execute( + '''SELECT 1 AS ok + FROM pg_catalog.pg_indexes + WHERE schemaname = 'public' + AND tablename = 'findings' + AND indexname = 'idx_findings_finding_uid' + LIMIT 1''' + ).fetchone() + return bool(row) + rows = db.conn.execute("PRAGMA index_list(findings)").fetchall() + return any(str(row["name"] or "") == "idx_findings_finding_uid" for row in rows) + except Exception: + return False + + +def finding_uid_index_available(db): + if str(os.getenv("KEYCHECK_ASSUME_FINDING_UID_INDEX", "")).strip().lower() in ("1", "true", "yes", "on"): + return True + try: + if getattr(db.conn, "is_postgres", False): + row = db.conn.execute( + '''SELECT 1 AS ok + FROM pg_catalog.pg_indexes + WHERE schemaname = 'public' + AND tablename = 'findings' + AND indexname = 'idx_findings_finding_uid' + LIMIT 1''' + ).fetchone() + return bool(row) + rows = db.conn.execute("PRAGMA index_list(findings)").fetchall() + return any(str(row["name"] or "") == "idx_findings_finding_uid" for row in rows) + except Exception: + return False + + +def select_finding_by_uid(db, finding_uid, lookup_available=True): + if not lookup_available: + return None + if not finding_uid: + return None + if db.conn.table_exists("finding_uid_map"): + rows = db.conn.execute( + '''SELECT f.id, f.run_id, f.cycle_id, f.target_scan_id, f.source, f.query, f.target, f.detector_name, f.created_at + FROM finding_uid_map m + JOIN findings f ON f.id = m.finding_id + WHERE m.finding_uid = ? + LIMIT 2''', + (finding_uid,), + ).fetchall() + if len(rows) == 1: + return rows[0] + if len(rows) > 1 or not finding_uid_index_available(db): + return None + rows = db.conn.execute( + '''SELECT id, run_id, cycle_id, target_scan_id, source, query, target, detector_name, created_at + FROM findings + WHERE finding_uid = ? + ORDER BY id DESC + LIMIT 2''', + (finding_uid,), + ).fetchall() + else: + rows = db.conn.execute( + '''SELECT id, run_id, cycle_id, target_scan_id, source, query, target, detector_name, created_at + FROM findings + WHERE finding_uid = ? + ORDER BY id DESC + LIMIT 2''', + (finding_uid,), + ).fetchall() + return rows[0] if len(rows) == 1 else None + + +def insert_keycheck_payload(db, service, path, line_offset, payload, line_cache, uid_lookup_available=True): + status = status_from_payload(service, payload) + finding = finding_from_payload(payload, line_cache) + raw_secret = extract_raw_secret(finding or {}) + finding_hash = sha256_text(raw_secret) if raw_secret else "" + key_hash = payload.get("key_hash") or finding_hash + secret_hash = payload.get("secret_hash") or finding_hash or key_hash + key_masked = payload.get("key_masked") or mask_secret(raw_secret) + if not key_hash and not key_masked: + return "skipped" + detector = payload.get("detector") or (finding or {}).get("DetectorName") or "" + detector_secret_hash = payload.get("detector_secret_hash") or (sha256_text("|".join([str(detector or ""), secret_hash])) if secret_hash or detector else "") + checked_at = payload.get("checked_at") or file_mtime_iso(path) + source_line = payload.get("source") or payload.get("source_line") or "" + finding_uid = payload.get("finding_uid") or (finding or {}).get("finding_uid") or "" + event_id = payload.get("event_id") or sha256_text("|".join([ + str(service or ""), + os.path.abspath(path), + str(line_offset), + str(key_hash or key_masked or ""), + str(status or ""), + str(source_line or ""), + ])) + duplicate = db.conn.execute( + "SELECT event_id FROM keycheck_event_map WHERE event_id = ? LIMIT 1", + (event_id,), + ).fetchone() + if duplicate: + return "duplicate" + reservation = db.conn.execute( + '''INSERT INTO keycheck_event_map (event_id, keycheck_result_id, created_at) + VALUES (?, NULL, ?) + ON CONFLICT(event_id) DO NOTHING''', + (event_id, utc_now_iso()), + ) + if getattr(reservation, "rowcount", -1) == 0: + return "duplicate" + metadata = safe_metadata(payload) + metadata["event_id"] = event_id + metadata["finding_uid"] = finding_uid + metadata["ingest_source_file"] = path + metadata["ingest_source_offset"] = int(line_offset) + message = payload.get("message") or "" + error = payload.get("error") + if not message and isinstance(error, dict): + message = error.get("message") or json.dumps(error, ensure_ascii=False)[:1000] + now = utc_now_iso() + match = select_finding_by_uid(db, finding_uid, uid_lookup_available) + keycheck_result_id = db.conn.insert_returning_id( + '''INSERT INTO keycheck_results ( + service, status, status_group, checked_at, key_hash, secret_hash, key_masked, + finding_id, target_scan_id, cycle_id, run_id, source, query, target, detector_name, + found_at, message, metadata_json, source_line, detector_secret_hash, + event_id, finding_uid, link_status, link_attempts, linked_at, link_error, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + service, + status, + keycheck_status_group(status), + checked_at, + key_hash, + secret_hash, + key_masked, + match["id"] if match else None, + match["target_scan_id"] if match else None, + match["cycle_id"] if match else None, + match["run_id"] if match else None, + match["source"] if match else None, + match["query"] if match else None, + match["target"] if match else None, + match["detector_name"] if match else detector, + match["created_at"] if match else None, + str(message or "")[:1000].replace("\n", " "), + json.dumps(metadata, ensure_ascii=False, default=str, sort_keys=True), + source_line, + detector_secret_hash, + event_id, + finding_uid, + "linked" if match else "pending", + 0, + now if match else None, + "", + now, + ), + ) + if keycheck_result_id: + db.conn.execute( + "UPDATE keycheck_event_map SET keycheck_result_id = ? WHERE event_id = ?", + (keycheck_result_id, event_id), + ) + return "inserted" + + +def ingest_keycheck_results_to_db(layout, services=None, max_rows=1000): + db_path = layout.get("database_path") + db_url = layout.get("database_url") or os.getenv("SCANNER_DB_URL") or os.getenv("DATABASE_URL") + keycheck_dir = layout.get("keycheck_dir") + if not keycheck_dir or not os.path.isdir(keycheck_dir): + return 0 + if not db_path and not db_url: + return 0 + db = ScannerDB(db_path=db_path, db_url=db_url, initialize=False) + if not db.enabled: + print(f"keycheck_results ingest skipped: unable to open DB: {db.db_display or db_path or 'configured database'}") + return 0 + set_linker_timeouts(db) + try: + db.require_runtime_safety_schema() + except Exception as exc: + db.close() + raise RuntimeError(f'Keycheck ingest schema is incomplete; offline migration required: {exc}') from exc + backfill_keycheck_event_map(db, env_int("KEYCHECK_EVENT_MAP_BACKFILL_ROWS", 0)) + backfill_finding_uid_map(db, env_int("KEYCHECK_UID_MAP_BACKFILL_ROWS", 0)) + uid_lookup_available = finding_uid_lookup_available(db) + requested = [service for service in (services or []) if service and service != "all"] + if set(requested) == set(SERVICES): + requested = [] + service_names = requested or sorted(name for name in os.listdir(keycheck_dir) if os.path.isdir(os.path.join(keycheck_dir, name))) + state_identity = db_url or os.path.abspath(db_path or "") + state_suffix = sha256_text(state_identity)[:12] if state_identity else "default" + state_path = os.path.join(keycheck_dir, f"db_ingest_state_{state_suffix}.json") + state = read_json_file(state_path, {}) or {} + files_state = state.get("files") if isinstance(state.get("files"), dict) else {} + inserted = 0 + duplicates = 0 + skipped = 0 + errors = 0 + rows_seen = 0 + line_cache = {} + max_rows = max(0, int(max_rows or 0)) + try: + for service in service_names: + if max_rows and rows_seen >= max_rows: + break + service_dir = os.path.join(keycheck_dir, service) + if not os.path.isdir(service_dir): + continue + service_paths = service_result_paths(service_dir) + current_skips = { + os.path.abspath(candidate): keycheck_result_current_skip(candidate) + for candidate in service_paths + if os.path.basename(candidate).lower().endswith("results.jsonl") + } + for path in service_paths: + if max_rows and rows_seen >= max_rows: + break + if not os.path.isfile(path): + continue + key = os.path.abspath(path) + processed = 0 + max_line_bytes = max(1024, env_int("KEYCHECK_DB_INGEST_MAX_LINE_BYTES", 16 * 1024 * 1024)) + with open(path, "rb") as f: + signature = file_signature(path, f) + offset, mode = ingest_start_offset(path, files_state.get(key), signature) + offset = max(offset, min(int(current_skips.get(key, 0)), signature["size"])) + final_offset = offset + f.seek(offset) + if mode.startswith("tail"): + skipped_line = f.readline(max_line_bytes + 1) + if skipped_line and not skipped_line.endswith(b"\n"): + raise RuntimeError(f"oversized keycheck ingest tail boundary in {path}") + final_offset = f.tell() + while True: + if max_rows and rows_seen >= max_rows: + break + line_offset = f.tell() + raw_line = f.readline(max_line_bytes + 1) + if not raw_line: + break + final_offset = f.tell() + if len(raw_line) > max_line_bytes: + final_offset = line_offset + raise RuntimeError(f"oversized committed keycheck result at {path}:{line_offset}") + if not raw_line.endswith(b"\n"): + final_offset = line_offset + raise RuntimeError(f"torn committed keycheck result at {path}:{line_offset}") + try: + payload = json.loads(raw_line.decode("utf-8", errors="replace")) + except ValueError as exc: + final_offset = line_offset + kind = "torn" if not raw_line.endswith(b"\n") else "invalid" + raise RuntimeError( + f"{kind} committed keycheck result at {path}:{line_offset}" + ) from exc + if not isinstance(payload, dict): + skipped += 1 + continue + rows_seen += 1 + try: + outcome = insert_keycheck_payload(db, service, path, line_offset, payload, line_cache, uid_lookup_available) + if outcome == "inserted": + inserted += 1 + elif outcome == "duplicate": + duplicates += 1 + else: + skipped += 1 + db.conn.commit() + processed += 1 + except Exception as exc: + errors += 1 + try: + db.conn.rollback() + set_linker_timeouts(db) + except Exception: + pass + final_offset = line_offset + print(f"keycheck_results ingest row failed: service={service} file={path} offset={line_offset} error={str(exc)[:300]}") + break + checkpoint_signature = file_signature(path, f) + files_state[key] = { + **checkpoint_signature, + "offset": int(final_offset), + "mode": mode, + "processed": int(processed), + "updated_at": utc_now_iso(), + } + db.conn.commit() + finally: + write_json_file(state_path, {"files": files_state, "updated_at": utc_now_iso()}) + db_display = db.db_display + db.close() + print(f"keycheck_results ingest: scanned={rows_seen} inserted={inserted} duplicates={duplicates} skipped={skipped} errors={errors} db={db_display}") + return inserted + + +def repair_keycheck_links(layout, services=None, batch_size=500, max_rows=0, max_attempts=3, since_days=14): + db_path = layout.get("database_path") + db_url = layout.get("database_url") or os.getenv("SCANNER_DB_URL") or os.getenv("DATABASE_URL") + if not db_path and not db_url: + raise SystemExit("global.database_path or global.database_url is required for keycheck link repair") + db = ScannerDB(db_path=db_path, db_url=db_url, initialize=False) + if not db.enabled: + raise SystemExit(f"Unable to open scanner DB: {db.db_display or db_path or 'configured database'}") + service_detectors = { + "anthropic": ("anthropic",), + "aws": ("aws",), + "azure": ( + "azure", "azureopenai", "azurecontainerregistry", "azurefoundry", + "azurefoundryendpointbeforekey", "azurefoundrykeybeforeendpoint", + ), + "deepseek": ("deepseek", "deepseekapikey", "deepseek_api_key"), + "dockerhub": ("dockerhub",), + "gcp": ("gcp", "gcpapplicationdefaultcredentials"), + "gemini": ("googleai", "googleaistudio"), + "groq": ("groq",), + "github": ("github", "githuboauth2"), + "gitlab": ("gitlab",), + "kimi": ("kimimoonshot", "moonshotai", "moonshot", "kimi"), + "openai": ("openai",), + "openrouter": ("openrouter",), + "provider_resolver": ( + "qwendashscope", "qwen_dashscope", "qwen", "dashscope", + "deepseek", "deepseekapikey", "deepseek_api_key", + "kimimoonshot", "moonshotai", "moonshot", "kimi", "zaiglm", + ), + "qwen": ("qwendashscope", "qwen_dashscope"), + "replicate": ("replicate",), + "xai": ("xai",), + "huggingface": ("huggingface",), + "zai": ("zaiglm",), + } + set_linker_timeouts(db) + try: + db.require_runtime_safety_schema() + except Exception as exc: + db.close() + raise RuntimeError(f'Keycheck repair schema is incomplete; offline migration required: {exc}') from exc + backfill_finding_uid_map(db, env_int("KEYCHECK_UID_MAP_BACKFILL_ROWS", 5000)) + requested = [service for service in (services or []) if service and service != "all"] + if set(requested) == set(SERVICES): + requested = [] + repaired = 0 + repaired_metadata = 0 + not_found = 0 + errors = 0 + scanned = 0 + line_cache = {} + uid_lookup_available = finding_uid_lookup_available(db) + window_size = max(batch_size * 5, 100) + max_scanned_candidates = max(max_rows * 100, window_size) if max_rows else env_int("KEYCHECK_REPAIR_MAX_SCAN_CANDIDATES", 10000) + cutoff = (datetime.now(timezone.utc) - timedelta(days=max(0, int(since_days or 0)))).isoformat(timespec="seconds") + last_id = 9223372036854775807 + scanned_candidates = 0 + while True: + if max_rows and scanned >= max_rows: + break + if max_scanned_candidates and scanned_candidates >= max_scanned_candidates: + break + params = [last_id, window_size] + try: + candidates = db.conn.execute(f''' + SELECT id, service, checked_at, metadata_json, source_line, detector_name, detector_secret_hash, + key_hash, secret_hash, + COALESCE(NULLIF(secret_hash, ''), NULLIF(key_hash, '')) AS hash_value, + COALESCE(link_status, 'pending') AS link_status, + COALESCE(link_attempts, 0) AS link_attempts, + finding_id, finding_uid + FROM keycheck_results + WHERE id < ? + ORDER BY id DESC + LIMIT ? + ''', params).fetchall() + except Exception as exc: + try: + db.conn.rollback() + except Exception: + pass + print(f"link repair stopped: candidate fetch failed after scanned={scanned}: {str(exc)[:300]}") + break + if not candidates: + break + scanned_candidates += len(candidates) + last_id = candidates[-1]['id'] + rows = [] + for row in candidates: + if row['checked_at'] and row['checked_at'] < cutoff: + continue + if requested and row['service'] not in requested: + continue + if row['finding_id'] is not None: + continue + if row['link_status'] not in ('pending', 'error', 'not_found', 'source_only'): + continue + if int(row['link_attempts'] or 0) >= max_attempts: + continue + rows.append(row) + if not rows: + continue + for row in rows: + if max_rows and scanned >= max_rows: + break + scanned += 1 + try: + detectors = service_detectors.get(row["service"], ()) + try: + metadata = json.loads(row["metadata_json"] or "{}") + except json.JSONDecodeError: + metadata = {} + row_finding_uid = row["finding_uid"] or metadata.get("finding_uid") or "" + if row_finding_uid: + uid_match = select_finding_by_uid(db, row_finding_uid, uid_lookup_available) + if uid_match: + db.conn.execute(''' + UPDATE keycheck_results + SET finding_id = ?, run_id = ?, cycle_id = ?, target_scan_id = ?, source = ?, query = ?, + target = ?, detector_name = ?, found_at = ?, link_status = 'linked', finding_uid = ?, + link_attempts = COALESCE(link_attempts, 0) + 1, linked_at = ?, link_error = '' + WHERE id = ? + ''', ( + uid_match["id"], uid_match["run_id"], uid_match["cycle_id"], uid_match["target_scan_id"], uid_match["source"], uid_match["query"], + uid_match["target"], uid_match["detector_name"], uid_match["created_at"], row_finding_uid, utc_now_iso(), row["id"], + )) + db.conn.commit() + repaired += 1 + continue + source_ref = row["source_line"] or metadata.get("source_line") or metadata.get("source") + finding = None + inferred = {} + stored_hashes = [value for value in (row["secret_hash"], row["key_hash"]) if value] + source_ref_is_finding = False + source_hash_mismatch = False + if source_ref: + path, line_number = parse_source_line(source_ref) + source_ref_is_finding = bool(path and line_number) + finding = load_jsonl_line(path, line_number, line_cache) + if isinstance(finding, dict) and isinstance(finding.get("finding"), dict): + finding = finding["finding"] + source_hash = row["secret_hash"] or row["key_hash"] or "" + if finding and stored_hashes: + raw_secret = extract_raw_secret(finding) + raw_hash = sha256_text(raw_secret) if raw_secret else "" + if raw_hash and raw_hash not in stored_hashes: + finding = None + source_hash_mismatch = True + elif raw_hash: + source_hash = raw_hash + inferred = infer_finding_attribution(finding) + source_match = None + if finding and source_hash: + location = extract_finding_location(finding) + detector = str(finding.get("DetectorName") or finding.get("DetectorType") or row["detector_name"] or "") + clauses = ["secret_hash = ?"] + params = [source_hash] + if detector: + clauses.append("LOWER(detector_name) = LOWER(?)") + params.append(detector) + exact_clauses = list(clauses) + exact_params = list(params) + exact_clauses.append("raw_finding_json = ?") + exact_params.append(json_dumps(finding)) + source_matches = db.conn.execute(f''' + SELECT id, run_id, cycle_id, target_scan_id, source, query, target, detector_name, created_at + FROM findings + WHERE {' AND '.join(exact_clauses)} + ORDER BY id DESC + LIMIT 2 + ''', exact_params).fetchall() + if len(source_matches) == 1: + source_match = source_matches[0] + if location.get("file_path"): + clauses.append("file_path = ?") + params.append(location.get("file_path")) + if location.get("line_number"): + clauses.append("line_number = ?") + params.append(str(location.get("line_number"))) + if location.get("commit_hash"): + clauses.append("commit_hash = ?") + params.append(location.get("commit_hash")) + if not source_match and (location.get("commit_hash") or inferred.get("source") or inferred.get("target")) and len(clauses) > 2: + if inferred.get("source"): + clauses.append("source = ?") + params.append(inferred.get("source")) + if inferred.get("target"): + clauses.append("target = ?") + params.append(inferred.get("target")) + source_matches = db.conn.execute(f''' + SELECT id, run_id, cycle_id, target_scan_id, source, query, target, detector_name, created_at + FROM findings + WHERE {' AND '.join(clauses)} + ORDER BY id DESC + LIMIT 2 + ''', params).fetchall() + if len(source_matches) == 1: + source_match = source_matches[0] + if source_match: + db.conn.execute(''' + UPDATE keycheck_results + SET finding_id = ?, run_id = ?, cycle_id = ?, target_scan_id = ?, source = ?, query = ?, + target = ?, detector_name = ?, found_at = ?, link_status = 'linked', + link_attempts = COALESCE(link_attempts, 0) + 1, linked_at = ?, link_error = '' + WHERE id = ? + ''', ( + source_match["id"], source_match["run_id"], source_match["cycle_id"], source_match["target_scan_id"], source_match["source"], source_match["query"], + source_match["target"], source_match["detector_name"], source_match["created_at"], utc_now_iso(), row["id"], + )) + db.conn.commit() + repaired += 1 + continue + if row_finding_uid or source_ref_is_finding: + if inferred.get("source") or inferred.get("target"): + db.conn.execute(''' + UPDATE keycheck_results + SET source = COALESCE(source, ?), query = COALESCE(query, ?), target = COALESCE(target, ?), + detector_name = COALESCE(NULLIF(detector_name, ''), ?), found_at = COALESCE(NULLIF(found_at, ''), ?), + link_status = 'source_only', link_attempts = COALESCE(link_attempts, 0) + 1, + linked_at = ?, link_error = '' + WHERE id = ? + ''', ( + inferred.get("source"), inferred.get("query"), inferred.get("target"), + inferred.get("detector_name"), inferred.get("found_at"), utc_now_iso(), row["id"], + )) + db.conn.commit() + repaired_metadata += 1 + else: + update_link_status(db, row["id"], "not_found", "exact finding_uid/source finding not present in DB") + db.conn.commit() + not_found += 1 + continue + hashes = [] + if row["secret_hash"]: + hashes.append(("secret_hash", row["secret_hash"])) + if not row["secret_hash"] and row["key_hash"]: + hashes.append(("secret_hash", row["key_hash"])) + if row["detector_secret_hash"] and row["service"] not in ("gemini", "qwen"): + hashes.insert(0, ("detector_secret_hash", row["detector_secret_hash"])) + match = None + for column, hash_value in hashes: + params = [hash_value] + detector_clause = "" + extra_name = db.conn.json_extract('raw_finding_json', '$.ExtraData.name') + if column == "secret_hash" and row["service"] == "gemini": + detector_clause = f""" + AND ( + LOWER(detector_name) IN (?, ?) + OR ( + LOWER(detector_name) = 'customregex' + AND LOWER(COALESCE({extra_name}, '')) = 'googleaistudio' + ) + ) + """ + params.extend(detectors) + elif column == "secret_hash" and row["service"] == "qwen": + detector_clause = f""" + AND ( + LOWER(detector_name) IN ({','.join('?' for _ in detectors)}) + OR ( + LOWER(detector_name) = 'customregex' + AND LOWER(COALESCE({extra_name}, '')) IN ({','.join('?' for _ in detectors)}) + ) + ) + """ + params.extend((*detectors, *detectors)) + elif column == "secret_hash" and detectors: + detector_clause = " AND LOWER(detector_name) IN ({})".format(','.join('?' for _ in detectors)) + params.extend(detectors) + match_rows = db.conn.execute(f''' + SELECT id, run_id, cycle_id, target_scan_id, source, query, target, detector_name, created_at + FROM findings + WHERE {column} = ? {detector_clause} + ORDER BY id DESC + LIMIT 2 + ''', params).fetchall() + if len(match_rows) == 1: + match = match_rows[0] + break + if match: + db.conn.execute(''' + UPDATE keycheck_results + SET finding_id = ?, run_id = ?, cycle_id = ?, target_scan_id = ?, source = ?, query = ?, + target = ?, detector_name = ?, found_at = ?, link_status = 'linked', + link_attempts = COALESCE(link_attempts, 0) + 1, linked_at = ?, link_error = '' + WHERE id = ? + ''', ( + match["id"], match["run_id"], match["cycle_id"], match["target_scan_id"], match["source"], match["query"], + match["target"], match["detector_name"], match["created_at"], utc_now_iso(), row["id"], + )) + db.conn.commit() + repaired += 1 + continue + if inferred.get("source") or inferred.get("target"): + db.conn.execute(''' + UPDATE keycheck_results + SET source = COALESCE(source, ?), query = COALESCE(query, ?), target = COALESCE(target, ?), + detector_name = COALESCE(NULLIF(detector_name, ''), ?), found_at = COALESCE(NULLIF(found_at, ''), ?), + link_status = 'source_only', link_attempts = COALESCE(link_attempts, 0) + 1, + linked_at = ?, link_error = '' + WHERE id = ? + ''', ( + inferred.get("source"), inferred.get("query"), inferred.get("target"), + inferred.get("detector_name"), inferred.get("found_at"), utc_now_iso(), row["id"], + )) + db.conn.commit() + repaired_metadata += 1 + continue + update_link_status(db, row["id"], "not_found", "no matching finding") + db.conn.commit() + not_found += 1 + except Exception as exc: + try: + db.conn.rollback() + set_linker_timeouts(db) + update_link_status(db, row["id"], "error", str(exc)) + db.conn.commit() + errors += 1 + except Exception as mark_exc: + try: + db.conn.rollback() + except Exception: + pass + print(f"link repair error: row={row['id']} error={str(exc)[:200]} mark_error_failed={str(mark_exc)[:200]}") + errors += 1 + db.conn.commit() + print(f"link repair progress: scanned={scanned} linked={repaired} source_only={repaired_metadata} not_found={not_found} errors={errors}") + db.conn.commit() + db_display = db.db_display + db.close() + print(f"keycheck_results link repair: scanned={scanned} linked={repaired} source_only={repaired_metadata} not_found={not_found} errors={errors} db={db_display}") + return repaired + repaired_metadata + + +def service_summary(service, layout): + output_dir = os.path.join(layout["keycheck_dir"], service) + counts = { + "service": service, + "alive": 0, + "alive_rate_limited": 0, + "checked": 0, + "dead": 0, + "limited": 0, + "restricted": 0, + "network": 0, + "unknown": 0, + "no_quota": 0, + "no_balance": 0, + "no_username": 0, + "no_target": 0, + "results": 0, + "total_status": 0, + "output_dir": output_dir, + "updated_at": utc_now_iso(), + } + file_counts = {} + if not os.path.isdir(output_dir): + counts["file_counts"] = file_counts + return counts + for name in sorted(os.listdir(output_dir)): + path = os.path.join(output_dir, name) + if not os.path.isfile(path) or not name.lower().endswith((".txt", ".jsonl")): + continue + lower = name.lower() + if lower.endswith("results.jsonl") and os.path.getsize(path) > 5 * 1024 * 1024: + file_counts[name] = None + continue + is_checked = lower.endswith("checked.txt") + line_count, unique_checked = file_line_stats(path, unique_keys=is_checked) + file_counts[name] = line_count + if lower.endswith("results.jsonl"): + counts["results"] += line_count + if is_checked: + counts["checked"] += int(unique_checked or 0) + + status_bucket = None + if "aliveratelimited" in lower: + status_bucket = "alive_rate_limited" + elif lower.endswith("alive.txt") or any(item in lower for item in ("bedrock", "admin", "canary", "vertex", "foundryllm")): + status_bucket = "alive" + elif "noquota" in lower: + status_bucket = "no_quota" + elif "nobalance" in lower: + status_bucket = "no_balance" + elif "nousername" in lower: + status_bucket = "no_username" + elif "notarget" in lower: + status_bucket = "no_target" + elif any(item in lower for item in ("restricted", "disabled", "accessdenied", "quarantined")): + status_bucket = "restricted" + elif any(item in lower for item in ("expired", "leaked", "revoked", "dead")): + status_bucket = "dead" + elif "ratelimited" in lower or "limited" in lower: + status_bucket = "limited" + elif any(item in lower for item in ("network", "badendpoint", "unresolved", "nocontext", "unknown")): + status_bucket = "unknown" if "network" not in lower else "network" + + if status_bucket == "alive_rate_limited": + counts["alive_rate_limited"] += line_count + counts["total_status"] += line_count + elif status_bucket == "alive": + counts["alive"] += line_count + counts["total_status"] += line_count + elif status_bucket == "dead": + counts["dead"] += line_count + counts["total_status"] += line_count + elif status_bucket == "limited": + counts["limited"] += line_count + counts["total_status"] += line_count + elif status_bucket == "restricted": + counts["restricted"] += line_count + counts["total_status"] += line_count + elif status_bucket == "network": + counts["network"] += line_count + counts["total_status"] += line_count + elif status_bucket == "unknown": + counts["unknown"] += line_count + counts["total_status"] += line_count + elif status_bucket == "no_quota": + counts["no_quota"] += line_count + counts["total_status"] += line_count + elif status_bucket == "no_balance": + counts["no_balance"] += line_count + counts["total_status"] += line_count + elif status_bucket == "no_username": + counts["no_username"] += line_count + counts["total_status"] += line_count + elif status_bucket == "no_target": + counts["no_target"] += line_count + counts["total_status"] += line_count + counts["file_counts"] = file_counts + return counts + + +def write_summary( + layout, services, summary_tsv=None, summary_json=None, alive_summary_tsv=None, + input_mode=None, +): + ensure_private_directory(layout["keycheck_dir"], reject_reparse=True) + summary_tsv = summary_tsv or os.path.join(layout["keycheck_dir"], "summary.tsv") + summary_json = summary_json or os.path.join(layout["keycheck_dir"], "summary.json") + alive_summary_tsv = alive_summary_tsv or os.path.join(layout["keycheck_dir"], "alive_summary.tsv") + if str(input_mode or os.getenv('KEYCHECK_INPUT_MODE') or 'jsonl').lower() == 'postgres': + db = ScannerDB(db_url=layout.get('database_url') or os.getenv('KEYCHECK_DB_URL'), initialize=False) + if not db.enabled: + raise RuntimeError('PostgreSQL keycheck summary database is unavailable') + try: + placeholders = ','.join('?' for _ in services) + state_rows = db.conn.execute( + f'''SELECT service, status, status_group, COUNT(*) AS count, MAX(updated_at) AS updated_at + FROM keycheck_current_state WHERE service IN ({placeholders}) + GROUP BY service, status, status_group''', + tuple(services), + ).fetchall() + projection_rows = db.conn.execute( + f'''SELECT c.service, c.secret_text, c.secret_json, s.status, s.status_group, + s.checked_at, s.metadata_json + FROM keycheck_current_state s + JOIN keycheck_credentials c ON c.id = s.credential_id + WHERE c.service IN ({placeholders}) + ORDER BY c.service, c.id''', + tuple(services), + ).fetchall() + managed_projection_rows = db.conn.execute( + f'''SELECT service, secret_text, secret_json + FROM keycheck_credentials WHERE service IN ({placeholders})''', + tuple(services), + ).fetchall() + history_rows = db.conn.execute( + f'''SELECT service, COUNT(*) AS count FROM keycheck_results + WHERE service IN ({placeholders}) GROUP BY service''', + tuple(services), + ).fetchall() + db.conn.commit() + finally: + db.close() + grouped = {service: {} for service in services} + rate_limited_alive = {service: 0 for service in services} + updated = {service: '' for service in services} + for row in state_rows: + service_counts = grouped[row['service']] + service_counts[row['status_group']] = ( + int(service_counts.get(row['status_group'], 0)) + int(row['count'] or 0) + ) + if row['status'] == 'VALID_RATE_LIMITED': + rate_limited_alive[row['service']] += int(row['count'] or 0) + updated[row['service']] = max(updated[row['service']], str(row['updated_at'] or '')) + history = {row['service']: int(row['count'] or 0) for row in history_rows} + rows = [] + for service in services: + counts = grouped.get(service) or {} + alive_rate_limited = int(rate_limited_alive.get(service, 0)) + row = { + 'service': service, + 'alive': max(0, int(counts.get('alive', 0)) - alive_rate_limited), + 'alive_rate_limited': alive_rate_limited, + 'checked': sum(counts.values()), + 'dead': int(counts.get('dead', 0)), + 'limited': int(counts.get('limited', 0)), + 'restricted': int(counts.get('restricted', 0)), + 'network': int(counts.get('network', 0)), + 'unknown': int(counts.get('unknown', 0)), + 'no_quota': int(counts.get('no_quota', 0)), + 'no_balance': int(counts.get('no_balance', 0)), + 'no_username': int(counts.get('no_username', 0)), + 'no_target': int(counts.get('no_context', 0)), + 'results': history.get(service, 0), + 'total_status': sum(counts.values()), + 'updated_at': updated.get(service) or utc_now_iso(), + 'output_dir': os.path.join(layout['keycheck_dir'], service), + } + rows.append(row) + projection = project_postgres_status_files( + layout, services, projection_rows, managed_rows=managed_projection_rows, + ) + print( + 'status_projection: ' + f"rows={projection['projected_rows']} files={projection['changed_files']} " + f"skipped={projection['skipped_rows']}" + ) + else: + rows = [service_summary(service, layout) for service in services] + columns = [ + "service", "alive", "alive_rate_limited", "checked", "dead", "limited", + "restricted", "network", "unknown", "no_quota", "no_balance", "no_username", "no_target", + "results", "total_status", "updated_at", "output_dir", + ] + with private_atomic_writer(summary_tsv) as f: + f.write("\t".join(columns) + "\n") + for row in rows: + f.write("\t".join(str(row.get(column, "")) for column in columns) + "\n") + with private_atomic_writer(summary_json) as f: + json.dump({"updated_at": utc_now_iso(), "services": rows}, f, ensure_ascii=False, indent=2) + with private_atomic_writer(alive_summary_tsv) as f: + f.write("service\talive\talive_rate_limited\talive_total\tupdated_at\toutput_dir\n") + for row in rows: + alive_total = int(row.get("alive", 0)) + int(row.get("alive_rate_limited", 0)) + f.write("\t".join([ + str(row.get("service", "")), + str(row.get("alive", 0)), + str(row.get("alive_rate_limited", 0)), + str(alive_total), + str(row.get("updated_at", "")), + str(row.get("output_dir", "")), + ]) + "\n") + print(f"summary_tsv: {summary_tsv}") + print(f"summary_json: {summary_json}") + print(f"alive_summary_tsv: {alive_summary_tsv}") + return rows + + +def maybe_add(command, flag, value): + if value: + command.extend([flag, value]) + + +def list_value(value): + if not value: + return [] + if isinstance(value, str): + return [item for item in value.split() if item] + return [str(item) for item in value] + + +def bool_config(value, default=False): + if value is None: + return default + if isinstance(value, bool): + return value + return str(value).strip().lower() in ("1", "true", "yes", "on") + + +def apply_keycheck_config_defaults(args, keychecks_config): + for flag in ( + "retry_network", "retry_limited", "retry_unknown", "retry_restricted", + "retry_no_balance", "retry_valid", "no_resource_probe", "recheck_all", "summary_only", "no_summary", + ): + if not getattr(args, flag, False) and bool_config(keychecks_config.get(flag), False): + setattr(args, flag, True) + if not args.max_keys and keychecks_config.get("max_keys"): + args.max_keys = int(keychecks_config.get("max_keys") or 0) + + +def service_extra_args(service, keychecks_config): + output = [] + output.extend(list_value(keychecks_config.get("common_args"))) + per_service = keychecks_config.get("service_args") or keychecks_config.get("per_service_args") or {} + if isinstance(per_service, dict): + output.extend(list_value(per_service.get(service))) + return output + + +def service_supports_proxy(service): + return bool((SERVICE_CAPABILITIES.get(service) or {}).get("proxy", True)) + + +def service_supports_flag(service, flag): + capabilities = SERVICE_CAPABILITIES.get(service) or {} + return flag in capabilities.get("flags", set()) + + +def nonempty_private_status_file(output_dir, filename): + output_dir = os.path.abspath(output_dir) + path = os.path.abspath(os.path.join(output_dir, filename)) + if os.path.commonpath((output_dir, path)) != output_dir: + raise RuntimeError(f"keycheck status path escapes provider output directory: {filename}") + if not os.path.lexists(path): + return False + require_private_file(path) + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode): + raise RuntimeError(f"keycheck status path is not a regular file: {path}") + return details.st_size > 0 + + +def retry_status_file_names(service, flag): + override = RETRY_STATUS_FILE_OVERRIDES.get((service, flag)) + if override is not None: + return override + return tuple( + f"{service}{suffix}.txt" + for suffix in RETRY_STATUS_DEFAULT_SUFFIXES.get(flag, ()) + ) + + +def provider_replay_required(service, args, output_dir, extra_args=None): + extra_flags = {str(value) for value in (extra_args or []) if str(value).startswith("--")} + if service == "qwen" and nonempty_private_status_file( + output_dir, + STATUS_TRANSACTION_JOURNAL_FILENAME, + ): + return True + if service_supports_flag(service, "recheck_all") and ( + getattr(args, "recheck_all", False) or "--recheck-all" in extra_flags + ): + return True + + for flag in RETRY_STATUS_DEFAULT_SUFFIXES: + requested = getattr(args, flag, False) or "--" + flag.replace("_", "-") in extra_flags + if not requested or not service_supports_flag(service, flag): + continue + filenames = retry_status_file_names(service, flag) + if any(nonempty_private_status_file(output_dir, filename) for filename in filenames): + return True + return False + + +def auto_repair_config(keychecks_config): + value = keychecks_config.get("repair_links") or keychecks_config.get("auto_repair_links") + if isinstance(value, bool): + value = {"enabled": value} + if not isinstance(value, dict): + value = {} + return { + "enabled": bool_config(value.get("enabled"), False), + "services": value.get("services"), + "batch_size": max(1, int(value.get("batch_size", 25) or 25)), + "max_rows": max(0, int(value.get("max_rows", 250) or 250)), + "max_attempts": max(1, int(value.get("max_attempts", 3) or 3)), + "since_days": max(0, int(value.get("since_days", 2) or 2)), + } + + +def db_ingest_config(keychecks_config): + value = keychecks_config.get("db_ingest") or keychecks_config.get("ingest_results") + if isinstance(value, bool): + value = {"enabled": value} + if not isinstance(value, dict): + value = {} + return { + "enabled": bool_config(value.get("enabled"), True), + "services": value.get("services"), + "max_rows": max(0, int(value.get("max_rows", 1000) or 1000)), + } + + +def maybe_ingest_keycheck_results(layout, requested_services, keychecks_config): + cfg = db_ingest_config(keychecks_config) + if not cfg["enabled"]: + return 0 + services = service_list(str(cfg["services"])) if cfg.get("services") else requested_services + print(f"db_ingest_keycheck_results: services={','.join(services)} max_rows={cfg['max_rows']}") + return ingest_keycheck_results_to_db(layout, services, max_rows=cfg["max_rows"]) + + +def maybe_auto_repair_links(layout, requested_services, keychecks_config): + cfg = auto_repair_config(keychecks_config) + if not cfg["enabled"]: + return + services = service_list(str(cfg["services"])) if cfg.get("services") else requested_services + print( + "auto_repair_links: " + f"services={','.join(services)} batch_size={cfg['batch_size']} " + f"max_rows={cfg['max_rows']} since_days={cfg['since_days']}" + ) + repair_keycheck_links( + layout, + services, + batch_size=cfg["batch_size"], + max_rows=cfg["max_rows"], + max_attempts=cfg["max_attempts"], + since_days=cfg["since_days"], + ) + + +def sync_service_outputs(script_dir, output_dir): + ensure_private_directory(output_dir, reject_reparse=True) + for name in os.listdir(script_dir): + if not name.lower().endswith((".txt", ".jsonl")): + continue + src = os.path.join(script_dir, name) + dst = os.path.join(output_dir, name) + if os.path.isfile(src): + with open(src, "rb") as fsrc, private_atomic_writer(dst, binary=True) as fdst: + while True: + chunk = fsrc.read(1024 * 1024) + if not chunk: + break + fdst.write(chunk) + + +def _provider_diagnostic_path(output_dir): + return os.path.join(output_dir, '.provider-last-run.log') + + +def _record_provider_startup_failure(service, output_dir, exc): + diagnostic_path = _provider_diagnostic_path(output_dir) + payload = f'provider startup failed: {type(exc).__name__}\n'.encode('ascii', errors='replace') + with private_atomic_writer(diagnostic_path, binary=True, suffix='.provider.tmp') as diagnostic: + diagnostic.write(payload) + print( + f'provider_exit: service={service} code=1 outcome=infrastructure_failure ' + f'diagnostic={diagnostic_path}', + flush=True, + ) + return 1 + + +def _stop_failed_provider_process(process): + if process is None: + return True + try: + process.terminate() + except Exception: + pass + try: + if process.wait(timeout=5) is not None: + return True + except subprocess.TimeoutExpired: + pass + except Exception: + pass + try: + process.kill() + if process.wait(timeout=5) is not None: + return True + except Exception: + pass + # Keep the exact owner reachable and fail closed until this runner exits. + _unconfirmed_provider_processes.append(process) + return False + + +def run_service(service, script_path, args, layout, extra_args=None): + if _unconfirmed_provider_processes: + raise RuntimeError('provider containment cleanup was not confirmed') + deadline = float(getattr(args, '_provider_deadline', time.monotonic() + 1800)) + if not math.isfinite(deadline): + raise ValueError('provider deadline must be finite') + project_dir = os.path.dirname(os.path.abspath(__file__)) + script = os.path.join(project_dir, script_path) + output_dir = os.path.join(layout["keycheck_dir"], service) + ensure_private_directory(output_dir, reject_reparse=True) + startup_errors = (LifecycleAuthorityError, OSError, subprocess.SubprocessError) + try: + metadata = require_active_supervisor_child(args.config, child_kind='keycheck', require_dsn=True) + except startup_errors as exc: + return _record_provider_startup_failure(service, output_dir, exc) + canonical_dsn = os.getenv('TRUF_MANAGED_POSTGRES_DSN') or '' + extra_args = list(extra_args or []) + + input_mode = str(getattr(args, 'input_mode', 'postgres') or 'postgres').lower() + input_file = args.input or os.path.join(layout["results_dir"], "found_secrets.jsonl") + proxy_file = args.proxy_file or layout["proxy_file"] + if input_mode == 'postgres' and any( + str(value).lower() in ('--input', '--plain') + or str(value).lower().startswith(('--input=', '--plain=')) + for value in extra_args + ): + raise ValueError('PostgreSQL provider configuration forbids compatibility input/plain arguments') + + if not os.path.exists(script): + print(f"{service}: script not found yet: {script}") + return 2 + + command = [ + sys.executable, + '-I', + '-S', + '-B', + os.path.join(project_dir, 'child_bootstrap.py'), + 'keycheck-provider', + str(script_path).replace('\\', '/'), + '--', + ] + if input_mode == 'jsonl': + maybe_add(command, "--input", input_file) + if service_supports_proxy(service): + maybe_add(command, "--proxy-file", proxy_file) + if args.max_keys and input_mode == 'jsonl': + command.extend(["--max-keys", str(args.max_keys)]) + for flag in ("retry_network", "retry_limited", "retry_unknown", "retry_restricted", "retry_no_balance", "retry_valid", "no_resource_probe", "recheck_all"): + if ( + getattr(args, flag, False) + and service_supports_flag(service, flag) + and (input_mode == 'jsonl' or flag == 'no_resource_probe') + ): + command.append("--" + flag.replace("_", "-")) + command.extend(extra_args) + + env = os.environ.copy() + strip_supervisor_credentials(env) + env.update(supervised_child_environment(metadata, canonical_dsn, 'keycheck-provider')) + env['SCANNER_DB_URL'] = canonical_dsn + env['DATABASE_URL'] = canonical_dsn + env["KEYCHECK_INPUT_FILE"] = input_file + env["KEYCHECK_INPUT_MODE"] = input_mode + env["KEYCHECK_PROXY_FILE"] = proxy_file + env["KEYCHECK_OUTPUT_DIR"] = output_dir + env["KEYCHECK_SERVICE"] = service + env["KEYCHECK_STATE_DIR"] = output_dir + env["KEYCHECK_DB_PATH"] = layout.get("database_path", "") + env["KEYCHECK_DB_URL"] = canonical_dsn + env["KEYCHECK_DB_INLINE"] = "0" + if input_mode == 'postgres': + env['KEYCHECK_PROVIDER_SLICE_KEYS'] = str(max(1, int(args.max_keys or 1))) + env['KEYCHECK_RESULT_PROJECTION_RESERVE_BYTES'] = str(int( + layout.get('keycheck_result_projection_reserve_bytes', 3 * 1024 * 1024) + )) + env['KEYCHECK_PROJECTION_MAX_ITEMS'] = str(int( + layout.get('projection_backlog_max_items', 10000) + )) + env['KEYCHECK_PROJECTION_MAX_BYTES'] = str(int( + layout.get('projection_backlog_max_bytes', 2 * 1024 * 1024 * 1024) + )) + env["PYTHONIOENCODING"] = "utf-8" + env["PYTHONPATH"] = os.pathsep.join([project_dir, os.path.join(project_dir, "keycheckers"), env.get("PYTHONPATH", "")]) + env.pop("KEYCHECK_DISABLE_HIGH_WATERMARK", None) + if input_mode == 'jsonl' and provider_replay_required(service, args, output_dir, extra_args): + env["KEYCHECK_DISABLE_HIGH_WATERMARK"] = "1" + + if time.monotonic() >= deadline: + print(f'provider_exit: service={service} code=124 outcome=deadline_exceeded', flush=True) + return 124 + print(f"{service}: {' '.join(redact_argv(command))}") + try: + completed = OwnedProcess( + command, + cwd=os.path.dirname(script), + env=env, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + creationflags=subprocess.CREATE_NO_WINDOW if os.name == "nt" else 0, + ) + except startup_errors as exc: + return _record_provider_startup_failure(service, output_dir, exc) + stream = getattr(completed, 'stdout', None) + output_tail = bytearray() + + def copy_provider_output(): + read = getattr(stream, 'read1', stream.read) + while True: + chunk = read(64 * 1024) + if not chunk: + return + output_tail.extend(chunk) + if len(output_tail) > 64 * 1024: + del output_tail[:-64 * 1024] + try: + binary_output = getattr(sys.stdout, 'buffer', None) + if binary_output is not None: + binary_output.write(chunk) + binary_output.flush() + else: + sys.stdout.write(chunk.decode('utf-8', errors='replace')) + sys.stdout.flush() + except (OSError, ValueError): + pass + + output_thread = None + exited = False + cleanup_confirmed = True + failure = None + timed_out = False + try: + if stream is not None: + output_thread = threading.Thread(target=copy_provider_output, name=f'{service}-provider-output', daemon=True) + output_thread.start() + code = completed.wait(timeout=max(0.0, deadline - time.monotonic())) + exited = True + except startup_errors as exc: + timed_out = isinstance(exc, subprocess.TimeoutExpired) + failure = type(exc).__name__ + code = 124 if timed_out else 1 + finally: + if not exited: + cleanup_confirmed = _stop_failed_provider_process(completed) + if output_thread is not None: + output_thread.join(timeout=5) + if output_thread.is_alive(): + # Closing a buffered pipe here can block on the reader's lock. + failure = failure or 'OutputDrainTimeout' + code = code or 1 + if output_thread is not None or failure: + diagnostic_path = _provider_diagnostic_path(output_dir) + with private_atomic_writer(diagnostic_path, binary=True, suffix='.provider.tmp') as diagnostic: + diagnostic.write(bytes(output_tail[-64 * 1024:])) + if failure: + diagnostic.write(f'\nprovider runtime failed: {failure}\n'.encode('ascii')) + if not cleanup_confirmed: + raise RuntimeError('provider containment cleanup was not confirmed') + outcome = ( + 'deadline_exceeded' if timed_out + else 'ok' if code == 0 + else 'capacity_blocked' if code == KEYCHECK_CAPACITY_BLOCKED_EXIT + else 'infrastructure_failure' + ) + detail = f" diagnostic={diagnostic_path}" if output_thread is not None and code != 0 else "" + print(f"provider_exit: service={service} code={code} outcome={outcome}{detail}", flush=True) + return code + + +def run_provider_services( + requested, args, layout, keychecks_config, work_probe=None, + monotonic=time.monotonic, +): + exit_code = 0 + provider_failures = [] + if not requested: + return exit_code, provider_failures + configured_workers = min(4, max(1, int( + keychecks_config.get('scheduler_workers', 4) or 4 + ))) + worker_count = min(configured_workers, len(requested)) + batch_keys = max(1, int(keychecks_config.get('scheduler_batch_keys', 1000) or 1000)) + explicit_limit = max(0, int(getattr(args, 'max_keys', 0) or 0)) + deadline = monotonic() + max( + 1.0, float(keychecks_config.get('scheduler_deadline_sec', 1800) or 1800), + ) + if not math.isfinite(deadline): + raise ValueError('provider deadline must be finite') + probe_db = None + if work_probe is None and layout.get('database_url'): + probe_db = ScannerDB(db_url=layout['database_url'], initialize=False) + if not probe_db.enabled: + raise RuntimeError('keycheck scheduler work database is unavailable') + probe_db.set_application_name('truf-keycheck-scheduler') + probe_db.require_runtime_safety_schema() + probe_db.require_final_cutover() + work_probe = probe_db.keycheck_service_has_work + + def run_bounded(service): + service_args = copy.copy(args) + service_args.max_keys = explicit_limit or batch_keys + service_args._provider_deadline = deadline + return run_service( + service, SERVICES[service], service_args, layout, + service_extra_args(service, keychecks_config), + ) + + outcomes = {service: 0 for service in requested} + queue = deque( + service for service in requested + if work_probe is None or work_probe(service) + ) + active = {} + try: + with ThreadPoolExecutor(max_workers=worker_count, thread_name_prefix='keycheck-provider') as executor: + while queue or active: + while queue and len(active) < worker_count and monotonic() < deadline: + service = queue.popleft() + active[executor.submit(run_bounded, service)] = service + if not active: + break + done, _ = wait(tuple(active), return_when=FIRST_COMPLETED) + for future in done: + service = active.pop(future) + try: + code = future.result() + except Exception as exc: + print( + f'provider_exit: service={service} code=1 ' + f'outcome=infrastructure_failure error={type(exc).__name__}' + ) + code = 1 + capacity_blocked = code == KEYCHECK_CAPACITY_BLOCKED_EXIT + if capacity_blocked: + print( + f'provider_backpressure: service={service} ' + 'reason=projection_capacity_saturated', + flush=True, + ) + code = 0 + if code and not outcomes[service]: + outcomes[service] = code + if ( + not code and not capacity_blocked + and not explicit_limit and monotonic() < deadline + and work_probe is not None and work_probe(service) + ): + queue.append(service) + finally: + if probe_db is not None: + probe_db.close() + for service in requested: + code = outcomes[service] + if code: + if not exit_code: + exit_code = code + provider_failures.append((service, code)) + return exit_code, provider_failures + + +def enqueue_requested_postgres_rechecks(layout, requested, args): + if str(getattr(args, 'input_mode', 'postgres')).lower() != 'postgres': + return 0 + groups = set() + mapping = { + 'retry_network': {'network'}, + 'retry_limited': {'limited', 'no_balance'}, + 'retry_unknown': {'unknown', 'no_context'}, + 'retry_restricted': {'restricted'}, + 'retry_no_balance': {'no_balance'}, + 'retry_valid': {'alive'}, + } + for flag, values in mapping.items(): + if getattr(args, flag, False): + groups.update(values) + if not groups and not getattr(args, 'recheck_all', False): + return 0 + database_url = layout.get('database_url') + if not database_url: + return 0 + db = ScannerDB(db_url=database_url, initialize=False) + if not db.enabled: + raise RuntimeError('PostgreSQL recheck candidate database is unavailable') + total = 0 + try: + db.set_application_name('truf-keycheck-recheck-generator') + db.require_runtime_safety_schema() + db.require_final_cutover() + for service in requested: + total += db.enqueue_keycheck_rechecks( + service, + None if getattr(args, 'recheck_all', False) else groups, + max_items=max(1, int( + getattr(args, 'max_keys', 0) + or ((layout.get('keycheck_recheck_batch_items') or 10000)) + )), + queue_max_items=int(layout.get('keycheck_queue_max_items', 100000)), + queue_max_bytes=int(layout.get('keycheck_queue_max_bytes', 512 * 1024 * 1024)), + ) + finally: + db.close() + return total + + +def collect_legacy_gcp_vertex_credentials(layout, max_items=100): + service_dir = os.path.join(layout["keycheck_dir"], "gcp") + max_items = max(1, int(max_items)) + max_bytes = max(1024, env_int("KEYCHECK_STATUS_PROJECTION_MAX_BYTES", 32 * 1024 * 1024)) + max_line_bytes = max(1024, env_int("KEYCHECK_INPUT_MAX_LINE_BYTES", 16 * 1024 * 1024)) + entries = {} + skipped = 0 + for filename in LEGACY_GCP_VERTEX_STATUS_FILES: + path = os.path.join(service_dir, filename) + if not os.path.exists(path): + continue + for line in _read_status_projection(path, max_items, max_bytes, max_line_bytes): + raw = status_key(line) + try: + parsed = json.loads(str(raw or "")) + except (TypeError, ValueError, json.JSONDecodeError): + skipped += 1 + continue + private_key = str(parsed.get("private_key") or "") if isinstance(parsed, dict) else "" + if ( + not isinstance(parsed, dict) + or not parsed.get("client_email") + or not parsed.get("private_key_id") + or "-----BEGIN" not in private_key + or "-----END" not in private_key + ): + skipped += 1 + continue + canonical = json.dumps(parsed, ensure_ascii=False, sort_keys=True, separators=(",", ":")) + if len(canonical.encode("utf-8")) > max_line_bytes: + skipped += 1 + continue + entries.setdefault(canonical, set()).add(filename) + if len(entries) > max_items: + raise RuntimeError("legacy GCP Vertex credential import exceeds its item bound") + return entries, skipped + + +def import_legacy_gcp_vertex_credentials(layout, max_items=100): + entries, skipped = collect_legacy_gcp_vertex_credentials(layout, max_items=max_items) + db = ScannerDB(db_url=layout.get("database_url") or os.getenv("SCANNER_DB_URL"), initialize=False) + if not db.enabled: + raise RuntimeError("PostgreSQL legacy GCP credential import database is unavailable") + db.set_application_name("truf-keycheck-legacy-gcp-import") + db.require_runtime_safety_schema() + db.require_final_cutover() + now = utc_now_iso() + queue_max_items = int(layout.get("keycheck_queue_max_items", 100000)) + queue_max_bytes = int(layout.get("keycheck_queue_max_bytes", 512 * 1024 * 1024)) + imported = 0 + existing = 0 + queued = 0 + linked_results = 0 + missing_history = 0 + queued_bytes = 0 + try: + capacity = db.conn.execute("SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE").fetchone() + if not capacity: + raise RuntimeError("pipeline capacity row is unavailable") + for secret_json, source_files in entries.items(): + provider_key_hash = sha256_text(secret_json) + present = db.conn.execute( + "SELECT id FROM keycheck_credentials WHERE service = ? AND provider_key_hash = ?", + ("gcp", provider_key_hash), + ).fetchone() + if present: + existing += 1 + continue + latest = db.conn.execute( + '''SELECT * FROM keycheck_results + WHERE service = ? AND (key_hash = ? OR secret_hash = ?) + ORDER BY checked_at DESC, id DESC LIMIT 1''', + ("gcp", provider_key_hash, provider_key_hash), + ).fetchone() + if not latest: + missing_history += 1 + continue + conflicts = db.conn.execute( + '''SELECT DISTINCT credential_id FROM keycheck_results + WHERE service = ? AND (key_hash = ? OR secret_hash = ?) + AND credential_id IS NOT NULL''', + ("gcp", provider_key_hash, provider_key_hash), + ).fetchall() + if conflicts: + raise RuntimeError("legacy GCP history already belongs to a different credential") + credential_hash = sha256_text("|".join(("truf-credential-v2", "gcp", secret_json))) + credential_id = db.conn.insert_returning_id( + '''INSERT INTO keycheck_credentials( + service, credential_hash, provider_key_hash, candidate_kind, + secret_text, secret_json, key_masked, endpoint, principal, + metadata_json, created_at, updated_at + ) VALUES (?, ?, ?, 'gcp_json', NULL, ?, ?, '', '', ?, ?, ?) + ON CONFLICT(service, provider_key_hash) DO NOTHING''', + ( + "gcp", credential_hash, provider_key_hash, secret_json, mask_secret(secret_json), + json_dumps({ + "legacy_status_import": True, + "source_files": sorted(source_files), + }), + now, now, + ), + ) + if credential_id is None: + raise RuntimeError("legacy GCP credential insert lost its uniqueness race") + updated = db.conn.execute( + '''UPDATE keycheck_results SET credential_id = ? + WHERE service = ? AND (key_hash = ? OR secret_hash = ?) + AND credential_id IS NULL''', + (credential_id, "gcp", provider_key_hash, provider_key_hash), + ) + linked_results += max(0, int(getattr(updated, "rowcount", 0) or 0)) + db.conn.execute( + '''INSERT INTO keycheck_current_state( + credential_id, service, status, status_group, last_result_id, + result_source, checked_at, recheck_after, state_version, + metadata_json, updated_at + ) VALUES (?, 'gcp', ?, ?, ?, ?, ?, NULL, 1, ?, ?)''', + ( + credential_id, latest["status"], latest["status_group"], latest["id"], + latest["result_source"] or "legacy_status_import", latest["checked_at"], + latest["metadata_json"] or "{}", now, + ), + ) + capacity_bytes = len(secret_json.encode("utf-8")) + 512 + if ( + int(capacity["keycheck_items"]) + queued + 1 > queue_max_items + or int(capacity["keycheck_bytes"]) + queued_bytes + capacity_bytes > queue_max_bytes + ): + raise RuntimeError("legacy GCP credential import would exceed keycheck queue capacity") + candidate_uid = sha256_text("|".join(( + "truf-keycheck-legacy-status-v1", "gcp", provider_key_hash, + ))) + candidate_id = db.conn.insert_returning_id( + '''INSERT INTO keycheck_candidates( + candidate_uid, credential_id, service, routed_service, secret_hash, + source, query, target, detector_name, found_at, finding_uid, + metadata_json, state, capacity_bytes, created_at, updated_at + ) VALUES (?, ?, 'gcp', 'gcp', ?, ?, ?, ?, ?, ?, ?, ?, 'pending', ?, ?, ?) + ON CONFLICT(candidate_uid) DO NOTHING''', + ( + candidate_uid, credential_id, latest["secret_hash"] or provider_key_hash, + latest["source"] or "legacy_keycheck_status", latest["query"] or "", + latest["target"] or "", latest["detector_name"] or "GCP", + latest["found_at"] or latest["checked_at"], latest["finding_uid"] or "", + json_dumps({ + "legacy_status_import": True, + "recheck_of_status": latest["status"], + "source_files": sorted(source_files), + }), + capacity_bytes, now, now, + ), + ) + imported += 1 + if candidate_id is not None: + queued += 1 + queued_bytes += capacity_bytes + if queued: + db.conn.execute( + '''UPDATE pipeline_capacity + SET keycheck_items = keycheck_items + ?, keycheck_bytes = keycheck_bytes + ?, + updated_at = ? WHERE id = 1''', + (queued, queued_bytes, now), + ) + db.conn.commit() + except Exception: + db.conn.rollback() + raise + finally: + db.close() + return { + "file_credentials": len(entries), + "imported": imported, + "existing": existing, + "queued": queued, + "linked_results": linked_results, + "missing_history": missing_history, + "skipped": skipped, + } + + +def parse_args(): + parser = argparse.ArgumentParser(description="Run normalized keycheckers from the unified project layout.") + parser.add_argument("--config", default="config.yaml") + parser.add_argument("--service", default="all", help="Service name, comma list, or all") + parser.add_argument("--input") + parser.add_argument("--input-mode", choices=('postgres', 'jsonl'), default='postgres') + parser.add_argument("--proxy-file") + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--retry-no-balance", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--no-resource-probe", action="store_true", help="Pass through to services that support read-only resource probes, currently Replicate.") + parser.add_argument("--recheck-all", action="store_true") + parser.add_argument("--import-legacy-gcp-vertex", action="store_true") + parser.add_argument("--print-plan", action="store_true") + parser.add_argument("--summary-only", action="store_true") + parser.add_argument("--no-summary", action="store_true") + parser.add_argument("--summary-tsv") + parser.add_argument("--summary-json") + parser.add_argument("--alive-summary-tsv") + parser.add_argument("--no-db-ingest", action="store_true", help="Do not import new *Results.jsonl events into keycheck_results") + parser.add_argument("--ingest-keychecks-to-db", action="store_true", help="Import durable *Results.jsonl events into keycheck_results without running provider checks") + parser.add_argument("--ingest-max-rows", type=int, default=None, help="Max result events to ingest for --ingest-keychecks-to-db; defaults to keychecks.db_ingest.max_rows") + parser.add_argument("--no-auto-repair-links", action="store_true") + parser.add_argument("--sync-keychecks-to-db", action="store_true", help="Backfill keycheck_results from existing *Results.jsonl files") + parser.add_argument("--reset-keycheck-results", action="store_true", help="Retired; online keycheck result deletion is not supported") + parser.add_argument("--repair-keycheck-links", action="store_true", help="Link unattributed keycheck_results to matching findings now present in scanner DB") + parser.add_argument("--repair-batch-size", type=int, default=500) + parser.add_argument("--repair-max-rows", type=int, default=0) + parser.add_argument("--repair-max-attempts", type=int, default=3) + parser.add_argument("--repair-since-days", type=int, default=14) + return parser.parse_args() + + +PLAN_MUTATING_MODES = ( + "sync_keychecks_to_db", + "reset_keycheck_results", + "repair_keycheck_links", + "ingest_keychecks_to_db", + "import_legacy_gcp_vertex", + "summary_only", +) + + +def plan_requires_authority(args): + return any(bool(getattr(args, name, False)) for name in PLAN_MUTATING_MODES) + + +def print_keycheck_plan(args, layout, requested, keychecks_config): + """Describe requested work without opening databases or changing canonical files.""" + if args.sync_keychecks_to_db or args.reset_keycheck_results: + print("mode: reset-keycheck-results-retired" if args.reset_keycheck_results else "mode: sync-keychecks-to-db") + print(f"reset_keycheck_results: {bool(args.reset_keycheck_results)}") + print("services: " + ",".join(requested)) + return + if args.repair_keycheck_links: + print("mode: repair-keycheck-links") + print("services: " + ",".join(requested)) + print( + "repair_limits: " + f"batch_size={max(1, args.repair_batch_size)} " + f"max_rows={max(0, args.repair_max_rows)} " + f"max_attempts={max(1, args.repair_max_attempts)} " + f"since_days={max(0, args.repair_since_days)}" + ) + return + if args.ingest_keychecks_to_db: + cfg = db_ingest_config(keychecks_config) + max_rows = cfg["max_rows"] if args.ingest_max_rows is None else args.ingest_max_rows + print("mode: ingest-keychecks-to-db") + print("services: " + ",".join(requested)) + print(f"max_rows: {max_rows}") + return + if args.summary_only: + print("mode: summary-only") + else: + print("mode: run-keychecks") + active_flags = [ + flag for flag in ( + "retry_network", "retry_limited", "retry_unknown", "retry_restricted", + "retry_no_balance", "retry_valid", "recheck_all", + ) if getattr(args, flag, False) + ] + if active_flags: + print("active_flags: " + " ".join("--" + flag.replace("_", "-") for flag in active_flags)) + for service in requested: + extra = service_extra_args(service, keychecks_config) + suffix = f" args={' '.join(extra)}" if extra else "" + print(f"{service}: {os.path.join(layout['project_dir'], SERVICES[service])} -> {os.path.join(layout['keycheck_dir'], service)}{suffix}") + if not args.no_summary: + print(f"summary_tsv: {args.summary_tsv or os.path.join(layout['keycheck_dir'], 'summary.tsv')}") + print(f"summary_json: {args.summary_json or os.path.join(layout['keycheck_dir'], 'summary.json')}") + print(f"alive_summary_tsv: {args.alive_summary_tsv or os.path.join(layout['keycheck_dir'], 'alive_summary.tsv')}") + + +def main(): + args = parse_args() + if getattr(args, 'reset_keycheck_results', False) and not args.print_plan: + raise SystemExit('--reset-keycheck-results is retired; online keycheck result deletion is not supported') + if not args.print_plan or plan_requires_authority(args): + try: + require_active_supervisor_child(args.config, child_kind='keycheck', require_dsn=True) + except LifecycleAuthorityError as exc: + raise SystemExit(str(exc)) from exc + config = load_config(args.config) + layout = config.get("global") or {} + if not args.print_plan: + preflight_lifecycle_paths(args.config, config, authority_profile='server') + require_sensitive_runtime_paths(layout, create=False) + load_postgres_env(args.config, layout) + keychecks_config = config.get("keychecks") or {} + apply_keycheck_config_defaults(args, keychecks_config) + configured_input_mode = str(keychecks_config.get('input_mode') or '').strip().lower() + args.input_mode = str(getattr(args, 'input_mode', 'postgres') or 'postgres').lower() + if args.input_mode == 'postgres' and configured_input_mode in ('postgres', 'jsonl'): + args.input_mode = configured_input_mode + if args.input and args.input_mode != 'jsonl': + raise SystemExit('--input is valid only with explicit --input-mode jsonl') + requested = service_list(args.service) + unknown = [service for service in requested if service not in SERVICES] + if unknown: + raise SystemExit(f"Unknown service(s): {', '.join(unknown)}") + summary_services = list(SERVICES) if args.input_mode == 'postgres' else requested + + input_file = args.input or os.path.join(layout["results_dir"], "found_secrets.jsonl") + proxy_file = args.proxy_file or layout["proxy_file"] + print(f"input_mode: {args.input_mode}") + print(f"input: {input_file if args.input_mode == 'jsonl' else 'postgres:keycheck_candidates'}") + print(f"proxy: {proxy_file}") + print(f"keycheck_dir: {layout['keycheck_dir']}") + + if args.print_plan: + print_keycheck_plan(args, layout, requested, keychecks_config) + return + + if args.sync_keychecks_to_db: + if args.input_mode != 'jsonl': + raise SystemExit('--sync-keychecks-to-db requires explicit --input-mode jsonl') + sync_keycheck_results_to_db(layout, requested) + return + + if args.repair_keycheck_links: + repair_keycheck_links( + layout, + requested, + batch_size=max(1, args.repair_batch_size), + max_rows=max(0, args.repair_max_rows), + max_attempts=max(1, args.repair_max_attempts), + since_days=max(0, args.repair_since_days), + ) + return + + if args.ingest_keychecks_to_db: + if args.input_mode != 'jsonl': + raise SystemExit('--ingest-keychecks-to-db requires explicit --input-mode jsonl') + cfg = db_ingest_config(keychecks_config) + if args.ingest_max_rows is not None and args.ingest_max_rows <= 0: + raise SystemExit("--ingest-max-rows must be > 0; omit it to use keychecks.db_ingest.max_rows") + max_rows = cfg["max_rows"] if args.ingest_max_rows is None else args.ingest_max_rows + ingest_keycheck_results_to_db(layout, requested, max_rows=max_rows) + return + + if args.summary_only: + write_summary( + layout, summary_services, args.summary_tsv, args.summary_json, + args.alive_summary_tsv, input_mode=args.input_mode, + ) + return + + if args.import_legacy_gcp_vertex: + if requested != ["gcp"]: + raise SystemExit("--import-legacy-gcp-vertex requires --service gcp") + imported = import_legacy_gcp_vertex_credentials(layout) + print( + "legacy_gcp_vertex_import: " + + " ".join(f"{name}={value}" for name, value in imported.items()), + flush=True, + ) + + if args.input_mode != 'postgres': + raise SystemExit( + 'Normal provider execution requires PostgreSQL candidate leases; ' + 'legacy JSONL compatibility work is offline-only.' + ) + + queued_rechecks = enqueue_requested_postgres_rechecks(layout, requested, args) + if queued_rechecks: + print(f'postgres_recheck_candidates: {queued_rechecks}', flush=True) + exit_code, provider_failures = run_provider_services( + requested, args, layout, keychecks_config, + ) + if not args.no_summary: + write_summary( + layout, summary_services, args.summary_tsv, args.summary_json, + args.alive_summary_tsv, input_mode=args.input_mode, + ) + if args.input_mode == 'jsonl' and not args.no_db_ingest: + maybe_ingest_keycheck_results(layout, requested, keychecks_config) + if args.input_mode == 'jsonl' and not args.no_auto_repair_links: + maybe_auto_repair_links(layout, requested, keychecks_config) + if provider_failures: + detail = ",".join(f"{service}={code}" for service, code in provider_failures) + print(f"provider_failures: {detail}", flush=True) + else: + print("provider_failures: none", flush=True) + raise SystemExit(exit_code) + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/__init__.py b/app/keycheckers/__init__.py new file mode 100644 index 0000000..623be04 --- /dev/null +++ b/app/keycheckers/__init__.py @@ -0,0 +1 @@ +"""Keychecker package for the unified scanner layout.""" diff --git a/app/keycheckers/anthropic/anthropicKeycheck.py b/app/keycheckers/anthropic/anthropicKeycheck.py new file mode 100644 index 0000000..2303c29 --- /dev/null +++ b/app/keycheckers/anthropic/anthropicKeycheck.py @@ -0,0 +1,274 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os +import re + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + read_plain_keys, + recover_status_transaction, + request_error_message, + record_validation_result, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) + + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +PARENT_DIR = os.path.dirname(SCRIPT_DIR) +SERVICE = "anthropic" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) + +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +CHECKED_FILE = os.path.join(OUTPUT_DIR, "anthropicChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "anthropicResults.jsonl") + +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "anthropicAlive.txt"), + "NO_QUOTA": os.path.join(OUTPUT_DIR, "anthropicNoQuota.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "anthropicDead.txt"), + "LIMITED": os.path.join(OUTPUT_DIR, "anthropicLimited.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "anthropicRestricted.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "anthropicNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "anthropicUnknown.txt"), +} + +ANTHROPIC_REGEX = re.compile(r"sk-ant-(?:api03|admin01)-[A-Za-z0-9\-_]{93}AA|sk-ant-[A-Za-z0-9\-_]{86}") +ANTHROPIC_ADMIN_PREFIX = "sk-ant-admin01-" +ANTHROPIC_ADMIN_API_KEYS_URL = "https://api.anthropic.com/v1/organizations/api_keys" + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def extract_candidates(input_file, plain_files): + seen_plain = set() + for item in iter_findings(input_file, ["Anthropic"]): + key = item["raw"] + if key and ANTHROPIC_REGEX.fullmatch(key): + yield key, item["source"], item["finding"] + for item in read_plain_keys(plain_files, ANTHROPIC_REGEX): + key = item["key"] + if key not in seen_plain: + seen_plain.add(key) + yield key, item["source"], {} + + +def tier_from_rpm(rpm): + mapping = {5: "Free Tier", 50: "Tier 1", 1000: "Tier 2", 2000: "Tier 3", 4000: "Tier 4"} + return mapping.get(rpm, "Scale/Unknown") + + +def anthropic_rate_headers(response): + output = {} + for name, value in response.headers.items(): + lowered = name.lower() + if lowered.startswith("anthropic-ratelimit-"): + output[lowered.replace("anthropic-ratelimit-", "rate_").replace("-", "_")] = value + return output + + +def list_models(key, proxy, timeout): + headers = { + "anthropic-version": "2023-06-01", + "x-api-key": key, + } + try: + response = requests.get("https://api.anthropic.com/v1/models", headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"models_error": str(exc)[:500]} + if response.status_code != 200: + return {"models_status": response.status_code, "models_error": request_error_message(response)} + try: + data = response.json() + except ValueError: + return {"models_status": response.status_code, "models_error": "invalid JSON response"} + models = [] + for item in data.get("data") or []: + if isinstance(item, dict) and item.get("id"): + models.append(item["id"]) + return { + "models_status": response.status_code, + "models_count": len(models), + "models": models[:50], + } + + +def check_admin_key(key, proxy, timeout): + headers = { + "anthropic-version": "2023-06-01", + "x-api-key": key, + } + try: + response = requests.get( + ANTHROPIC_ADMIN_API_KEYS_URL, + headers=headers, + params={"limit": 1}, + proxies=proxy, + timeout=timeout, + ) + except requests.RequestException as exc: + return {"status": "NETWORK", "admin": True, "message": str(exc)[:500]} + + if response.status_code == 200: + return { + "status": "VALID", + "admin": True, + "http_status": 200, + "message": "Anthropic Admin API access confirmed", + } + + status = { + 401: "DEAD", + 403: "RESTRICTED", + 429: "LIMITED", + }.get(response.status_code, "UNKNOWN") + message = request_error_message(response).replace(key, "***REDACTED***") + return { + "status": status, + "admin": True, + "http_status": response.status_code, + "message": message, + } + + +def check_key(key, proxy, timeout, model="claude-opus-4-6", include_models=False): + if key.startswith(ANTHROPIC_ADMIN_PREFIX): + return check_admin_key(key, proxy, timeout) + + url = "https://api.anthropic.com/v1/messages" + headers = { + "content-type": "application/json", + "anthropic-version": "2023-06-01", + "x-api-key": key, + } + payload = { + "model": model, + "messages": [{"role": "user", "content": "ping"}], + "max_tokens": 1, + } + try: + response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc)} + + if response.status_code == 200: + rpm = 0 + try: + rpm = int(response.headers.get("anthropic-ratelimit-requests-limit", "0")) + except ValueError: + rpm = 0 + rate_data = anthropic_rate_headers(response) + result = { + "status": "VALID", + "model": model, + "rpm": rpm, + "tier": tier_from_rpm(rpm), + **rate_data, + } + if include_models: + result.update(list_models(key, proxy, timeout)) + token_limit = rate_data.get("rate_tokens_limit") or "" + token_remaining = rate_data.get("rate_tokens_remaining") or "" + token_part = f" tokens={token_remaining}/{token_limit}" if token_limit or token_remaining else "" + result["message"] = f"model={model}; rpm={rpm}; tier={result['tier']}{token_part}" + return result + + if response.status_code == 429: + return {"status": "LIMITED", "http_status": 429, "message": request_error_message(response)} + + message = request_error_message(response) + lower = message.lower() + if "credit balance is too low" in lower or "usage limits" in lower: + return {"status": "NO_QUOTA", "http_status": response.status_code, "message": message} + if response.status_code in (401, 403): + status = "RESTRICTED" if "disabled" in lower or response.status_code == 403 else "DEAD" + return {"status": status, "http_status": response.status_code, "message": message} + return {"status": "UNKNOWN", "http_status": response.status_code, "message": message} + + +def write_result(key, result, source, finding): + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "Anthropic") + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), source, + ) + record_validation_result(SERVICE, key, result, source, finding, "Anthropic") + + +def parse_args(): + parser = argparse.ArgumentParser(description="Anthropic key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=20) + parser.add_argument("--model", default=os.getenv("ANTHROPIC_CHECK_MODEL", "claude-opus-4-6")) + parser.add_argument("--list-models", action="store_true") + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-no-balance", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = set() + if args.retry_network: + retry_statuses.add("NETWORK") + if args.retry_limited: + retry_statuses.add("LIMITED") + if args.retry_unknown: + retry_statuses.add("UNKNOWN") + if args.retry_no_balance: + retry_statuses.add("NO_QUOTA") + + processed = 0 + skipped = 0 + for key, source, finding in extract_candidates(args.input, args.plain): + if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="Anthropic"): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] Anthropic candidate {mask_secret(key)} from {source}") + proxy = next(proxy_cycler) if proxy_cycler else None + result = check_key(key, proxy, args.timeout, args.model, args.list_models) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/aws/awsKeycheck.py b/app/keycheckers/aws/awsKeycheck.py new file mode 100644 index 0000000..6a3125d --- /dev/null +++ b/app/keycheckers/aws/awsKeycheck.py @@ -0,0 +1,607 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + load_checked_statuses, + load_known_keys, + load_known_statuses, + load_proxies, + mask_secret, + read_plain_keys, + recover_status_transaction, + record_cached_keycheck_occurrence, + record_validation_result, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) + + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +PARENT_DIR = os.path.dirname(SCRIPT_DIR) +SERVICE = "aws" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) + +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +CHECKED_FILE = os.path.join(OUTPUT_DIR, "awsChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "awsResults.jsonl") + +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "awsAlive.txt"), + "BEDROCK": os.path.join(OUTPUT_DIR, "awsBedrock.txt"), + "ADMIN": os.path.join(OUTPUT_DIR, "awsAdmin.txt"), + "CANARY": os.path.join(OUTPUT_DIR, "awsCanary.txt"), + "QUARANTINED": os.path.join(OUTPUT_DIR, "awsQuarantined.txt"), + "ACCESS_DENIED": os.path.join(OUTPUT_DIR, "awsAccessDenied.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "awsDead.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "awsNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "awsUnknown.txt"), +} + +BEDROCK_REGIONS = ["us-east-1", "us-west-2", "eu-west-1", "eu-north-1", "ap-northeast-1", "ap-southeast-4"] +ANTHROPIC_MESSAGES_PROBE = { + "anthropic_version": "bedrock-2023-05-31", + "messages": [{"role": "user", "content": "ping"}], + "max_tokens": -1, +} +ANTHROPIC_MESSAGES_LIVE_PING = { + "anthropic_version": "bedrock-2023-05-31", + "messages": [{"role": "user", "content": "ping"}], + "max_tokens": 1, +} +BEDROCK_MODEL_TESTS = { + # Current Anthropic Bedrock runtime IDs. The default probe intentionally uses + # invalid max_tokens to validate auth/model access without generating tokens. + "anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE, + "us.anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE, + "global.anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE, + "anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE, + "us.anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE, + "global.anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE, + "anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE, + "us.anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE, + "global.anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE, + "anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE, + "us.anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE, + "global.anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE, + "anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE, + "us.anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE, + "global.anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE, + "anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE, + "us.anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE, + "global.anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE, + "anthropic.claude-3-5-sonnet-20241022-v2:0": ANTHROPIC_MESSAGES_PROBE, + "anthropic.claude-3-5-haiku-20241022-v1:0": ANTHROPIC_MESSAGES_PROBE, + "anthropic.claude-3-haiku-20240307-v1:0": ANTHROPIC_MESSAGES_PROBE, + "anthropic.claude-v2": {"prompt": "\n\nHuman:\n\nAssistant:", "max_tokens_to_sample": -1}, + "anthropic.claude-instant-v1": {"prompt": "\n\nHuman:\n\nAssistant:", "max_tokens_to_sample": -1}, +} + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def extract_candidates(input_file, plain_files): + seen_plain = set() + for item in iter_findings(input_file, ["AWS"]): + key = item["raw_v2"] or item["raw"] + if key and ":" in key: + yield key, item["source"], item["finding"] + import re + regex = re.compile(r"AKIA[0-9A-Z]{16}:[A-Za-z0-9+/]{40}") + for item in read_plain_keys(plain_files, regex): + key = item["key"] + if key not in seen_plain: + seen_plain.add(key) + yield key, item["source"], {} + + +def is_dead_aws_error(code): + return code in {"InvalidClientTokenId", "SignatureDoesNotMatch", "AuthFailure", "UnrecognizedClientException"} + + +def is_canary_text(value): + value = str(value or "").lower() + return "canarytokens" in value or "canary token" in value or "is_canary" in value + + +def is_canary_finding(finding): + if not isinstance(finding, dict): + return False + extra = finding.get("ExtraData") or {} + if isinstance(extra, dict): + if str(extra.get("is_canary", "")).lower() == "true": + return True + if any(is_canary_text(value) for value in extra.values()): + return True + return is_canary_text(finding.get("Raw")) or is_canary_text(finding.get("RawV2")) + + +def is_canary_arn(arn): + return is_canary_text(arn) + + +def aws_client(session, service, proxy=None, region_name=None, timeout=20): + kwargs = {} + if region_name: + kwargs["region_name"] = region_name + from botocore.config import Config + kwargs["config"] = Config( + proxies=proxy or None, + connect_timeout=timeout, + read_timeout=timeout, + retries={"max_attempts": 1}, + ) + return session.client(service, **kwargs) + + +def bedrock_validation_allows_invoke(exc): + text = str(exc or "").lower() + if any(item in text for item in ("operation not allowed", "not authorized", "access denied")): + return False + # The default probe sends deliberately invalid token limits. If Bedrock only + # rejects the payload shape after auth, InvokeModel reached the model path. + return any(item in text for item in ("max_tokens", "max_tokens_to_sample", "malformed input", "schema")) + + +def client_error_code(exc): + try: + return exc.response.get("Error", {}).get("Code", "ClientError") + except Exception: + return "ClientError" + + +def client_error_message(exc): + try: + return exc.response.get("Error", {}).get("Message", str(exc)) + except Exception: + return str(exc) + + +def model_arn(region, model_id): + # Cross-region inference profile IDs are not foundation-model ARNs. + if model_id.startswith(("us.", "eu.", "jp.", "au.", "global.")): + return "*" + return f"arn:aws:bedrock:{region}::foundation-model/{model_id}" + + +def iam_policy_source_arn(sts_arn, account): + arn = str(sts_arn or "") + if ":assumed-role/" in arn: + role_part = arn.split(":assumed-role/", 1)[1].split("/", 1)[0] + return f"arn:aws:iam::{account}:role/{role_part}" + return arn if ":iam::" in arn else "" + + +def simulate_bedrock_activation(session, arn, account, region, model_id, proxy=None, timeout=20): + import botocore.exceptions + + source_arn = iam_policy_source_arn(arn, account) + if not source_arn: + return {"status": "not_available", "message": "unsupported principal arn for IAM simulation"} + actions = [ + "bedrock:GetFoundationModelAvailability", + "bedrock:ListFoundationModelAgreementOffers", + "bedrock:GetUseCaseForModelAccess", + "bedrock:PutUseCaseForModelAccess", + "bedrock:CreateFoundationModelAgreement", + "bedrock:GetInferenceProfile", + "bedrock:InvokeModel", + ] + try: + iam = aws_client(session, "iam", proxy, timeout=timeout) + response = iam.simulate_principal_policy( + PolicySourceArn=source_arn, + ActionNames=actions, + ResourceArns=[model_arn(region, model_id)], + ) + except botocore.exceptions.ClientError as exc: + return { + "status": "access_denied" if client_error_code(exc) == "AccessDenied" else "error", + "code": client_error_code(exc), + "message": client_error_message(exc)[:500], + } + decisions = {} + for item in response.get("EvaluationResults", []): + action = str(item.get("EvalActionName") or "") + decisions[action] = str(item.get("EvalDecision") or "") + activation_actions = ["bedrock:PutUseCaseForModelAccess", "bedrock:CreateFoundationModelAgreement"] + can_activate = all(decisions.get(action) == "allowed" for action in activation_actions) + return {"status": "ok", "source_arn": source_arn, "can_activate": can_activate, "decisions": decisions} + + +def check_bedrock_management(session, arn, account, proxy=None, timeout=20, regions=None, models=None, max_attempts=12, debug=False): + import botocore.exceptions + + attempts = [] + findings = [] + tried = 0 + for region in (regions or BEDROCK_REGIONS): + bedrock = aws_client(session, "bedrock", proxy, region, timeout) + use_case = None + try: + use_case = bedrock.get_use_case_for_model_access() + except botocore.exceptions.ClientError as exc: + use_case = {"error_code": client_error_code(exc), "message": client_error_message(exc)[:300]} + try: + profiles = bedrock.list_inference_profiles(typeEquals="SYSTEM_DEFINED", maxResults=20).get("inferenceProfileSummaries", []) + except botocore.exceptions.ClientError as exc: + profiles = {"error_code": client_error_code(exc), "message": client_error_message(exc)[:300]} + for model_id in (models or list(BEDROCK_MODEL_TESTS.keys())): + if max_attempts and tried >= max_attempts: + return {"enabled": bool(findings), "findings": findings, "message": "; ".join(attempts[:10])} + tried += 1 + if debug: + print(f" BEDROCK MGMT TRY: region={region}, model={model_id}") + item = {"region": region, "model": model_id, "use_case": use_case, "profiles": profiles} + try: + item["foundation_model"] = bedrock.get_foundation_model(modelIdentifier=model_id).get("modelDetails", {}) + except botocore.exceptions.ClientError as exc: + item["foundation_model_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]} + try: + item["availability"] = bedrock.get_foundation_model_availability(modelId=model_id) + except botocore.exceptions.ClientError as exc: + item["availability_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]} + try: + item["agreement_offers"] = bedrock.list_foundation_model_agreement_offers(modelId=model_id, offerType="ALL") + except botocore.exceptions.ClientError as exc: + item["agreement_offers_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]} + item["iam_simulation"] = simulate_bedrock_activation(session, arn, account, region, model_id, proxy, timeout) + availability = item.get("availability") or {} + simulation = item.get("iam_simulation") or {} + can_activate = bool(simulation.get("can_activate")) + authorized = str(availability.get("authorizationStatus") or "").lower() in ("authorized", "available") + if can_activate or authorized: + findings.append(item) + else: + code = (item.get("availability_error") or item.get("foundation_model_error") or {}).get("code") or "checked" + attempts.append(f"{region}:{model_id}:can_activate={can_activate}:authorization={availability.get('authorizationStatus') or code}") + return {"enabled": bool(findings), "findings": findings, "message": "; ".join(attempts[:10])} + + +def check_bedrock(session, proxy=None, timeout=20, debug=False, regions=None, models=None, max_attempts=12, live_invoke=False): + import json + import botocore.exceptions + + attempts = [] + accepted = [] + tried = 0 + model_ids = models or list(BEDROCK_MODEL_TESTS.keys()) + for region in (regions or BEDROCK_REGIONS): + for model_id in model_ids: + if max_attempts and tried >= max_attempts: + if accepted: + first = accepted[0] + return { + "enabled": True, + "region": first.get("region", ""), + "model": first.get("model", ""), + "available_models": [f"{item['region']}/{item['model']}" for item in accepted], + "message": "Bedrock InvokeModel accepted", + } + return {"enabled": False, "region": "", "model": "", "available_models": [], "message": "; ".join(attempts[:10]) or "Bedrock probe attempt limit reached"} + tried += 1 + data = BEDROCK_MODEL_TESTS.get(model_id) + if data is None: + data = ANTHROPIC_MESSAGES_PROBE + if live_invoke and data is ANTHROPIC_MESSAGES_PROBE: + data = ANTHROPIC_MESSAGES_LIVE_PING + client = aws_client(session, "bedrock-runtime", proxy, region, timeout) + if debug: + print(f" BEDROCK TRY: region={region}, model={model_id}") + try: + client.invoke_model(body=json.dumps(data), modelId=model_id) + if debug: + print(" BEDROCK RESULT: invoke_model succeeded") + accepted.append({"region": region, "model": model_id, "message": "invoke_model succeeded"}) + continue + except client.exceptions.ValidationException as exc: + message = str(exc) + if bedrock_validation_allows_invoke(exc): + # ValidationException for the intentional bad payload means auth/model access passed. + if debug: + print(f" BEDROCK RESULT: validation_exception_after_auth: {message[:200]}") + accepted.append({"region": region, "model": model_id, "message": message[:300]}) + else: + if debug: + print(f" BEDROCK RESULT: validation_rejected: {message[:200]}") + attempts.append(f"{region}:{model_id}:validation:{message[:120]}") + continue + except client.exceptions.AccessDeniedException: + if debug: + print(" BEDROCK RESULT: access_denied") + attempts.append(f"{region}:{model_id}:access_denied") + continue + except client.exceptions.ResourceNotFoundException: + if debug: + print(" BEDROCK RESULT: model_not_found") + attempts.append(f"{region}:{model_id}:not_found") + continue + except botocore.exceptions.EndpointConnectionError as exc: + if debug: + print(f" BEDROCK RESULT: network_error: {str(exc)[:120]}") + attempts.append(f"{region}:{model_id}:network:{str(exc)[:80]}") + continue + except botocore.exceptions.ClientError as exc: + code = exc.response.get("Error", {}).get("Code", "ClientError") + if debug: + print(f" BEDROCK RESULT: {code}: {str(exc)[:160]}") + attempts.append(f"{region}:{model_id}:{code}") + continue + if accepted: + first = accepted[0] + return { + "enabled": True, + "region": first.get("region", ""), + "model": first.get("model", ""), + "available_models": [f"{item['region']}/{item['model']}" for item in accepted], + "message": "Bedrock InvokeModel accepted", + } + return {"enabled": False, "region": "", "model": "", "available_models": [], "message": "; ".join(attempts[:10])} + + +def inspect_iam(session, arn, proxy=None, timeout=20): + import botocore.exceptions + + output = {"admin": False, "quarantined": False, "policy_check": "not_checked", "message": ""} + if ":user/" not in arn: + output["policy_check"] = "not_user_arn" + return output + username = arn.rsplit("/", 1)[1] + try: + iam = aws_client(session, "iam", proxy, timeout=timeout) + policies = iam.list_attached_user_policies(UserName=username).get("AttachedPolicies", []) + except botocore.exceptions.ClientError as exc: + code = exc.response.get("Error", {}).get("Code", "") + output["policy_check"] = "access_denied" if code == "AccessDenied" else "error" + output["message"] = str(exc) + return output + output["policy_check"] = "ok" + for policy in policies: + name = policy.get("PolicyName", "") + if "AWSCompromisedKeyQuarantine" in name: + output["quarantined"] = True + if name == "AdministratorAccess": + output["admin"] = True + return output + + +def check_key( + key, + probe_bedrock=False, + bedrock_debug=False, + proxy=None, + timeout=20, + bedrock_regions=None, + bedrock_models=None, + bedrock_max_attempts=12, + bedrock_live_invoke=False, + probe_bedrock_management=False, +): + try: + import boto3 + import botocore.exceptions + except ImportError as exc: + return {"status": "UNKNOWN", "message": f"boto3/botocore missing: {exc}"} + + access_key, secret = key.split(":", 1) + session = boto3.Session(aws_access_key_id=access_key, aws_secret_access_key=secret) + try: + identity = aws_client(session, "sts", proxy, timeout=timeout).get_caller_identity() + except botocore.exceptions.EndpointConnectionError as exc: + return {"status": "NETWORK", "message": str(exc)} + except botocore.exceptions.ClientError as exc: + code = exc.response.get("Error", {}).get("Code", "") + status = "DEAD" if is_dead_aws_error(code) else "ACCESS_DENIED" + return {"status": status, "code": code, "message": str(exc)} + except Exception as exc: + return {"status": "UNKNOWN", "message": str(exc)} + + arn = identity.get("Arn", "") + if is_canary_arn(arn): + return { + "status": "CANARY", + "account": identity.get("Account", ""), + "arn": arn, + "admin": False, + "quarantined": False, + "iam_policy_check": "skipped_canary", + "bedrock_enabled": False, + "bedrock_region": "", + "bedrock_model": "", + "bedrock_message": "skipped_canary", + "bedrock_management_enabled": False, + "bedrock_management_message": "skipped_canary", + "message": "canary credential detected from STS arn; skipped IAM/Bedrock probes", + } + + iam_info = inspect_iam(session, arn, proxy, timeout) + bedrock_info = {"enabled": False, "region": "", "model": "", "message": "not_checked"} + bedrock_management_info = {"enabled": False, "findings": [], "message": "not_checked"} + if probe_bedrock: + bedrock_info = check_bedrock(session, proxy, timeout, bedrock_debug, bedrock_regions, bedrock_models, bedrock_max_attempts, bedrock_live_invoke) + if probe_bedrock_management: + bedrock_management_info = check_bedrock_management( + session, + arn, + identity.get("Account", ""), + proxy, + timeout, + bedrock_regions, + bedrock_models, + bedrock_max_attempts, + bedrock_debug, + ) + + if iam_info.get("quarantined"): + status = "QUARANTINED" + elif bedrock_info.get("enabled"): + status = "BEDROCK" + else: + status = "ADMIN" if iam_info.get("admin") else "VALID" + + message_parts = [ + "sts_ok", + f"iam_policy_check={iam_info.get('policy_check')}", + ] + if probe_bedrock: + message_parts.append(f"bedrock_enabled={bedrock_info.get('enabled')}") + if bedrock_info.get("region"): + message_parts.append(f"bedrock_region={bedrock_info.get('region')}") + if bedrock_info.get("model"): + message_parts.append(f"bedrock_model={bedrock_info.get('model')}") + if probe_bedrock_management: + findings = bedrock_management_info.get("findings") or [] + can_activate = any((item.get("iam_simulation") or {}).get("can_activate") for item in findings) + message_parts.append(f"bedrock_mgmt_enabled={bedrock_management_info.get('enabled')}") + message_parts.append(f"bedrock_can_activate={can_activate}") + if iam_info.get("message") and iam_info.get("policy_check") != "access_denied": + message_parts.append(iam_info.get("message")[:300]) + + return { + "status": status, + "account": identity.get("Account", ""), + "arn": arn, + "admin": iam_info.get("admin", False), + "quarantined": iam_info.get("quarantined", False), + "iam_policy_check": iam_info.get("policy_check"), + "bedrock_enabled": bedrock_info.get("enabled"), + "bedrock_region": bedrock_info.get("region"), + "bedrock_model": bedrock_info.get("model"), + "bedrock_available_models": bedrock_info.get("available_models") or [], + "bedrock_message": bedrock_info.get("message"), + "bedrock_management_enabled": bedrock_management_info.get("enabled"), + "bedrock_management_findings": bedrock_management_info.get("findings") or [], + "bedrock_management_message": bedrock_management_info.get("message"), + "message": "; ".join(message_parts), + } + + +def write_result(key, result, source, finding): + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "AWS") + message = result.get("message", "") + if result.get("status") == "BEDROCK": + models = result.get("bedrock_available_models") or [] + model_text = ",".join(str(item) for item in models) or f"{result.get('bedrock_region', '')}/{result.get('bedrock_model', '')}".strip("/") + message = f"{message}; models={model_text}" + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, result["status"], message, result.get("arn", source), + ) + record_validation_result(SERVICE, key, result, source, finding, "AWS") + + +def parse_args(): + parser = argparse.ArgumentParser(description="AWS key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--timeout", type=int, default=20) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--probe-bedrock", action="store_true") + parser.add_argument("--bedrock-debug", action="store_true") + parser.add_argument("--bedrock-regions", default=",".join(BEDROCK_REGIONS)) + parser.add_argument("--bedrock-models", default=",".join(BEDROCK_MODEL_TESTS)) + parser.add_argument("--bedrock-max-attempts", type=int, default=12) + parser.add_argument("--bedrock-live-invoke", action="store_true") + parser.add_argument("--probe-bedrock-management", action="store_true") + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES) + known = set(known_statuses) + retry_statuses = set() + if args.retry_network: + retry_statuses.add("NETWORK") + if args.retry_unknown: + retry_statuses.add("UNKNOWN") + if args.retry_valid: + retry_statuses.update({"VALID", "BEDROCK", "ADMIN"}) + processed = 0 + skipped = 0 + bedrock_regions = [item.strip() for item in str(args.bedrock_regions or "").split(",") if item.strip()] + bedrock_models = [item.strip() for item in str(args.bedrock_models or "").split(",") if item.strip()] + for key, source, finding in extract_candidates(args.input, args.plain): + if should_skip_key( + key, checked, known, args, retry_statuses, + service=SERVICE, source=source, finding=finding, detector="AWS", known_statuses=known_statuses, + ): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] AWS candidate {mask_secret(key)} from {source}") + if is_canary_finding(finding): + result = { + "status": "CANARY", + "message": "canary credential detected in TruffleHog ExtraData; skipped AWS API probes", + } + else: + proxy = next(proxy_cycler) if proxy_cycler else None + result = check_key( + key, + args.probe_bedrock, + args.bedrock_debug, + proxy, + args.timeout, + bedrock_regions, + bedrock_models, + args.bedrock_max_attempts, + args.bedrock_live_invoke, + args.probe_bedrock_management, + ) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + if args.probe_bedrock: + print( + " BEDROCK PING: " + f"enabled={result.get('bedrock_enabled')}, " + f"region={result.get('bedrock_region') or '-'}, " + f"model={result.get('bedrock_model') or '-'}" + ) + if result.get('bedrock_message'): + print(f" BEDROCK RESPONSE: {str(result.get('bedrock_message'))[:300]}") + if args.probe_bedrock_management: + findings = result.get("bedrock_management_findings") or [] + can_activate = any((item.get("iam_simulation") or {}).get("can_activate") for item in findings) + print( + " BEDROCK MGMT: " + f"enabled={result.get('bedrock_management_enabled')}, " + f"can_activate={can_activate}, " + f"findings={len(findings)}" + ) + if result.get("bedrock_management_message"): + print(f" BEDROCK MGMT RESPONSE: {str(result.get('bedrock_management_message'))[:300]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/azure/azureKeycheck.py b/app/keycheckers/azure/azureKeycheck.py new file mode 100644 index 0000000..9d96eef --- /dev/null +++ b/app/keycheckers/azure/azureKeycheck.py @@ -0,0 +1,946 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import json +import os +import re +from urllib.parse import urlsplit + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) +sys.path.append(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))) + +from keycheck_candidates import extract_azure_foundry_parts +from keycheck_common import ( + append_jsonl, + append_status, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + keycheck_input_mode, + iter_findings, + iter_bounded_text_lines, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + recover_status_transaction, + request_error_message, + record_validation_result, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) + + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +PARENT_DIR = os.path.dirname(SCRIPT_DIR) +SERVICE = "azure" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) + +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +CHECKED_FILE = os.path.join(OUTPUT_DIR, "azureChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "azureResults.jsonl") +AZURE_OPENAI_LLM_FILE = os.path.join(OUTPUT_DIR, "azureOpenAILLM.txt") +AZURE_OPENAI_PLAIN_FILE = os.path.join(OUTPUT_DIR, "azureOpenAI.txt") +AZURE_FOUNDRY_PLAIN_FILE = os.path.join(OUTPUT_DIR, "azureFoundry.txt") + +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "azureAlive.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "azureDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "azureRestricted.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "azureNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "azureUnknown.txt"), + "OPENAI_UNRESOLVED": os.path.join(OUTPUT_DIR, "azureOpenAIUnresolved.txt"), + "OPENAI_BAD_ENDPOINT": os.path.join(OUTPUT_DIR, "azureOpenAIBadEndpoint.txt"), + "FOUNDRY": os.path.join(OUTPUT_DIR, "azureFoundryLLM.txt"), + "FOUNDRY_UNRESOLVED": os.path.join(OUTPUT_DIR, "azureFoundryUnresolved.txt"), + "FOUNDRY_BAD_ENDPOINT": os.path.join(OUTPUT_DIR, "azureFoundryBadEndpoint.txt"), +} + +AZURE_OPENAI_ENDPOINT_RE = re.compile(r"([a-z0-9-]+\.openai\.azure\.com)", re.IGNORECASE) +AZURE_FOUNDRY_HOST_RE = r"[a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models\.ai\.azure\.com|services\.ai\.azure\.com|inference\.ai\.azure\.com)" +AZURE_FOUNDRY_ENDPOINT_RE = re.compile(r"((?:https?://)?" + AZURE_FOUNDRY_HOST_RE + r"(?:/[^\s:\"'<>\\]*)?)", re.IGNORECASE) +AZURE_OPENAI_DEPLOYMENTS_API_VERSION = "2023-03-15-preview" +AZURE_OPENAI_CHAT_API_VERSION = "2024-02-15-preview" +AZURE_FOUNDRY_API_VERSION = "2024-05-01-preview" +AZURE_FOUNDRY_KEY_ASSIGNMENT_RE = re.compile( + r"(?is)(?:authorization|api[_-]?key|key|token|secret|credential|bearer)[^\n:=]{0,80}[:=]\s*[\"']?(?:bearer\s+)?([A-Za-z0-9_./+=\-]{20,512})" +) +NON_FOUNDRY_KEY_PREFIXES = ( + "sk-", "sk_", "sk-or-", "xai-", "ghp_", "gho_", "ghu_", "ghs_", "ghr_", "github_pat_", + "glpat-", "glrt-", "hf_", "AIza", "AQ.", "zai-", "gsk_", "r8_", "nvapi-", +) + + +def transaction_status_files(): + return {**STATUS_FILES, "AUX_OPENAI_LLM": AZURE_OPENAI_LLM_FILE} + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, AZURE_OPENAI_LLM_FILE, AZURE_OPENAI_PLAIN_FILE, AZURE_FOUNDRY_PLAIN_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, transaction_status_files()) + + +def parse_azure_sp(raw_v2): + try: + data = json.loads(raw_v2) + except (TypeError, ValueError): + return None + client_secret = data.get("clientSecret") or data.get("client_secret") + client_id = data.get("clientId") or data.get("client_id") + tenant_id = data.get("tenantId") or data.get("tenant_id") + if not all([client_secret, client_id, tenant_id]): + return None + return {"client_secret": client_secret, "client_id": client_id, "tenant_id": tenant_id} + + +def scanner_context_text(finding): + context = finding.get("ScannerContext") if isinstance(finding, dict) else None + if isinstance(context, dict): + return str(context.get("nearby") or "") + return "" + + +def foundry_keyish(value): + text = re.sub(r"(?i)^bearer\s+", "", str(value or "").strip().strip('"\'`,;')).strip() + lower = text.lower() + if not (20 <= len(text) <= 512): + return False + if any(marker in lower for marker in ("http://", "https://", "{{", "${", "<", "azure.com")): + return False + if any(ch.isspace() for ch in text): + return False + if re.match(r"(?i)^(?:authorization|api[_-]?key|key|token|secret|credential|bearer)\s*[:=]", text): + return False + if text.startswith(NON_FOUNDRY_KEY_PREFIXES): + return False + return bool(re.search(r"[A-Za-z]", text) and re.search(r"[0-9]", text)) + + +def normalize_foundry_endpoint(value): + text = str(value or "").strip().strip('"\'`,;') + if not text: + return "" + split_text = text if re.match(r"(?i)^https?://", text) else "https://" + text + try: + parsed = urlsplit(split_text) + host = parsed.netloc or parsed.path.split("/", 1)[0] + path = parsed.path if parsed.netloc else ("/" + parsed.path.split("/", 1)[1] if "/" in parsed.path else "") + except Exception: + host, path = re.sub(r"(?i)^https?://", "", text).split("/", 1)[0], "" + path = path.rstrip(".,;:)]}/") + terminal_routes = ( + ("/models/chat/completions", ""), + ("/openai/v1/chat/completions", "/openai/v1"), + ("/v1/chat/completions", "/v1"), + ("/chat/completions", ""), + ("/v1/models", "/v1"), + ("/models", ""), + ) + lower_path = path.lower() + for suffix, replacement in terminal_routes: + if lower_path.endswith(suffix): + path = path[:-len(suffix)] + replacement + break + return (host + path).strip("/").lower() + + +def split_foundry_endpoint_key(text): + candidate = str(text or "").strip().split("\t", 1)[0].strip() + if not candidate: + return None + endpoint_match = AZURE_FOUNDRY_ENDPOINT_RE.search(candidate) + if not endpoint_match: + return None + endpoint = normalize_foundry_endpoint(endpoint_match.group(1)) + before = candidate[:endpoint_match.start()].replace("https://", " ").replace("http://", " ").strip(" \t:=,;'\"/") + after = candidate[endpoint_match.end():].replace("https://", " ").replace("http://", " ").strip(" \t:=,;'\"/") + for key in (after, before): + if foundry_keyish(key): + return {"key": key, "endpoint": endpoint} + return None + + +def foundry_context_values(finding): + values = [] + for value in (finding.get("Raw"), finding.get("RawV2")) if isinstance(finding, dict) else (): + if value: + text = str(value) + values.append(text) + values.extend(str(item) for item in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(text)) + context = scanner_context_text(finding) + if context: + values.append(context) + values.extend(str(item) for item in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(context or "")) + extra = finding.get("ExtraData") if isinstance(finding, dict) else None + if isinstance(extra, dict): + values.extend(str(value) for value in extra.values() if isinstance(value, str)) + return values + + +def parse_azure_openai(raw, raw_v2, finding): + key = raw or "" + endpoint = "" + raw_v2 = raw_v2 or "" + match = re.match(r"^([a-f0-9]{32}):(.+\.openai\.azure\.com)$", raw_v2, re.IGNORECASE) + if match: + key = match.group(1) + endpoint = match.group(2) + if not endpoint: + context_match = AZURE_OPENAI_ENDPOINT_RE.search(scanner_context_text(finding)) + if context_match: + endpoint = context_match.group(1) + if not key: + return None + return {"key": key, "endpoint": endpoint} + + +def parse_azure_openai_line(line): + text = str(line or "").strip() + if not text: + return None + text = text.split("\t", 1)[0].strip() + if ":" in text: + endpoint, key = text.split(":", 1) + if AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint.strip()) and key.strip(): + return {"endpoint": endpoint.strip(), "key": key.strip()} + endpoint_match = AZURE_OPENAI_ENDPOINT_RE.search(text) + key_match = re.search(r"\b[a-f0-9]{32}\b", text, re.IGNORECASE) + if endpoint_match and key_match: + return {"endpoint": endpoint_match.group(1), "key": key_match.group(0)} + if key_match: + return {"endpoint": "", "key": key_match.group(0)} + return None + + +def parse_azure_foundry(raw, raw_v2, finding): + key = raw or "" + endpoint = "" + raw_v2 = raw_v2 or "" + split = split_foundry_endpoint_key(raw_v2) or split_foundry_endpoint_key(raw) + if not split: + split = extract_azure_foundry_parts(raw, raw_v2) + if split: + key = split["key"] + endpoint = split["endpoint"] + if not endpoint: + for endpoint_text in (raw, raw_v2, scanner_context_text(finding)): + context_match = AZURE_FOUNDRY_ENDPOINT_RE.search(str(endpoint_text or "")) + if context_match: + endpoint = normalize_foundry_endpoint(context_match.group(1)) + break + if not foundry_keyish(key): + for value in foundry_context_values(finding): + if foundry_keyish(value): + key = value.strip().strip('"\'`,;') + break + if not key or not foundry_keyish(key): + return None + return {"key": key.strip().strip('"\'`,;'), "endpoint": normalize_foundry_endpoint(endpoint)} + + +def parse_azure_foundry_line(line): + parts = str(line or "").strip().split("\t", 1) + text = parts[0].strip() + if not text: + return None + split = split_foundry_endpoint_key(text) + parsed = split if split else ({"key": text, "endpoint": ""} if foundry_keyish(text) else None) + if not parsed: + return None + if len(parts) > 1: + try: + metadata = json.loads(parts[1]) + except ValueError: + metadata = {} + if isinstance(metadata, dict): + parsed["finding_uid"] = metadata.get("finding_uid") or "" + parsed["origin"] = metadata.get("origin") or "" + return parsed + + +def parse_azure_acr(raw_v2): + try: + data = json.loads(raw_v2) + except (TypeError, ValueError): + return None + username = data.get("username") + password = data.get("password") + if not username or not password: + return None + return {"username": username, "password": password} + + +def azure_openai_key(parsed): + endpoint = parsed.get("endpoint") or "" + return f"{endpoint}:{parsed['key']}" if endpoint else parsed["key"] + + +def azure_acr_key(parsed): + return f"{parsed['username']}:{parsed['password']}" + + +def azure_sp_key(parsed): + return f"{parsed['tenant_id']}:{parsed['client_id']}:{parsed['client_secret']}" + + +def azure_foundry_key(parsed): + endpoint = parsed.get("endpoint") or "" + return f"{endpoint}:{parsed['key']}" if endpoint else parsed["key"] + + +def extract_candidates(input_file): + seen_plain = set() + foundry_detectors = {"AzureFoundryEndpointBeforeKey", "AzureFoundryKeyBeforeEndpoint"} + detector_names = ["AzureOpenAI", "AzureContainerRegistry", "Azure", *sorted(foundry_detectors)] + for item in iter_findings(input_file, detector_names): + candidate_kind = item.get('candidate_kind') or '' + if keycheck_input_mode() == 'postgres' and candidate_kind: + secret_text = item.get('credential_secret_text') or '' + secret_json = item.get('credential_secret_json') or '' + endpoint = item.get('credential_endpoint') or '' + parsed = None + detector = '' + if candidate_kind == 'azure_service_principal': + parsed = parse_azure_sp(secret_json) + detector = 'Azure' + key = azure_sp_key(parsed) if parsed else '' + elif candidate_kind == 'azure_container_registry': + parsed = parse_azure_acr(secret_json) + detector = 'AzureContainerRegistry' + key = azure_acr_key(parsed) if parsed else '' + elif candidate_kind == 'azure_openai': + parsed = {'key': secret_text, 'endpoint': endpoint} if secret_text else None + detector = 'AzureOpenAI' + key = azure_openai_key(parsed) if parsed else '' + elif candidate_kind == 'azure_foundry': + parsed = ( + {'key': secret_text, 'endpoint': normalize_foundry_endpoint(endpoint)} + if foundry_keyish(secret_text) and endpoint else + parse_azure_foundry(secret_text, '', item.get('finding') or {}) + ) + if parsed and endpoint: + parsed['endpoint'] = normalize_foundry_endpoint(endpoint) + detector = 'AzureFoundry' + key = azure_foundry_key(parsed) if parsed else '' + else: + key = '' + if parsed and key: + yield key, detector, item['source'], item['finding'], parsed + continue + unresolved_detector = { + 'azure_openai': 'AzureOpenAI', + 'azure_foundry': 'AzureFoundry', + 'azure_container_registry': 'AzureContainerRegistry', + 'azure_service_principal': 'Azure', + }.get(candidate_kind, 'Azure') + unresolved_key = secret_text or secret_json or candidate_kind + if unresolved_key: + yield unresolved_key, unresolved_detector, item['source'], item['finding'], { + '_unresolved_candidate': True, + 'candidate_kind': candidate_kind, + } + continue + if item["detector"] == "AzureOpenAI": + foundry = parse_azure_foundry(item["raw"], item["raw_v2"], item["finding"]) + if foundry and foundry.get("endpoint"): + key = azure_foundry_key(foundry) + yield key, "AzureFoundry", item["source"], item["finding"], foundry + parsed = parse_azure_openai(item["raw"], item["raw_v2"], item["finding"]) + if not parsed: + continue + key = azure_openai_key(parsed) + yield key, "AzureOpenAI", item["source"], item["finding"], parsed + continue + if item["detector"] == "AzureContainerRegistry": + parsed = parse_azure_acr(item["raw_v2"]) + if not parsed: + continue + key = azure_acr_key(parsed) + yield key, "AzureContainerRegistry", item["source"], item["finding"], parsed + continue + if item["detector"] == "Azure": + parsed = parse_azure_sp(item["raw_v2"]) + if not parsed: + continue + key = azure_sp_key(parsed) + yield key, "Azure", item["source"], item["finding"], parsed + continue + + if item["detector"] in foundry_detectors: + foundry = parse_azure_foundry(item.get("raw"), item.get("raw_v2"), item.get("finding")) + if not foundry or not foundry.get("endpoint"): + continue + key = azure_foundry_key(foundry) + yield key, "AzureFoundry", item["source"], item["finding"], foundry + + if keycheck_input_mode() == 'jsonl' and os.path.exists(AZURE_FOUNDRY_PLAIN_FILE): + for line_num, line in enumerate(iter_bounded_text_lines(AZURE_FOUNDRY_PLAIN_FILE), 1): + parsed = parse_azure_foundry_line(line) + if not parsed: + continue + key = azure_foundry_key(parsed) + if key not in seen_plain: + seen_plain.add(key) + yield key, "AzureFoundry", f"{AZURE_FOUNDRY_PLAIN_FILE}:{line_num}", {}, parsed + + if keycheck_input_mode() == 'jsonl' and os.path.exists(AZURE_OPENAI_PLAIN_FILE): + for line_num, line in enumerate(iter_bounded_text_lines(AZURE_OPENAI_PLAIN_FILE), 1): + parsed = parse_azure_openai_line(line) + if not parsed: + continue + key = azure_openai_key(parsed) + if key not in seen_plain: + seen_plain.add(key) + yield key, "AzureOpenAI", f"{AZURE_OPENAI_PLAIN_FILE}:{line_num}", {}, parsed + + +def permission_matches(action, pattern): + action = str(action or "").lower() + pattern = str(pattern or "").lower() + if pattern == "*": + return True + if pattern.endswith("/*"): + return action.startswith(pattern[:-1]) + return action == pattern + + +def has_action(actions, wanted): + return any(permission_matches(wanted, action) for action in actions) + + +def probe_azure_rbac(access_token, proxy, timeout, max_subscriptions=3): + if not access_token: + return {"azure_rbac_level": "unknown", "message": "rbac_probe=no_access_token"} + headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"} + try: + response = requests.get( + "https://management.azure.com/subscriptions?api-version=2020-01-01", + headers=headers, + proxies=proxy, + timeout=timeout, + ) + except requests.RequestException as exc: + return {"azure_rbac_level": "network", "message": f"rbac_probe_network={str(exc)[:200]}"} + if response.status_code == 403: + return {"azure_rbac_level": "token_only", "azure_subscription_count": 0, "message": "rbac_probe=subscriptions_forbidden"} + if response.status_code >= 400: + return {"azure_rbac_level": "unknown", "azure_rbac_http_status": response.status_code, "message": f"rbac_probe_http={response.status_code}:{request_error_message(response)[:200]}"} + payload = response.json() + subscriptions = payload.get("value") if isinstance(payload, dict) else [] + subscriptions = subscriptions or [] + sub_ids = [item.get("subscriptionId") for item in subscriptions if isinstance(item, dict) and item.get("subscriptionId")] + if not sub_ids: + return {"azure_rbac_level": "token_only", "azure_subscription_count": 0, "message": "rbac_probe=no_subscriptions"} + + all_actions = set() + all_not_actions = set() + permission_errors = [] + for sub_id in sub_ids[:max_subscriptions]: + url = f"https://management.azure.com/subscriptions/{sub_id}/providers/Microsoft.Authorization/permissions?api-version=2022-04-01" + try: + perms_response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + permission_errors.append(f"{sub_id}:network:{str(exc)[:120]}") + continue + if perms_response.status_code >= 400: + permission_errors.append(f"{sub_id}:http_{perms_response.status_code}:{request_error_message(perms_response)[:120]}") + continue + data = perms_response.json().get("value") or [] + for item in data: + for action in item.get("actions") or []: + all_actions.add(str(action)) + for action in item.get("notActions") or []: + all_not_actions.add(str(action)) + + can_all = has_action(all_actions, "*") + can_assign_roles = has_action(all_actions, "Microsoft.Authorization/roleAssignments/write") and not has_action(all_not_actions, "Microsoft.Authorization/roleAssignments/write") + can_manage_cognitive = any( + has_action(all_actions, action) for action in ( + "Microsoft.CognitiveServices/accounts/write", + "Microsoft.CognitiveServices/accounts/deployments/write", + "Microsoft.CognitiveServices/*", + ) + ) or can_all + can_manage_ml = any( + has_action(all_actions, action) for action in ( + "Microsoft.MachineLearningServices/workspaces/write", + "Microsoft.MachineLearningServices/*", + ) + ) or can_all + can_deploy_resources = has_action(all_actions, "Microsoft.Resources/deployments/write") or can_all + can_manage_ai = can_manage_cognitive or can_manage_ml + + if can_all and can_assign_roles: + level = "owner_like" + elif can_all: + level = "contributor_like" + elif can_manage_ai: + level = "ai_manager" + elif all_actions: + level = "limited" + else: + level = "subscriptions_visible" + + message = ( + f"rbac_probe={level}; subscriptions={len(sub_ids)}; " + f"can_manage_ai={can_manage_ai}; can_assign_roles={can_assign_roles}; can_deploy_resources={can_deploy_resources}" + ) + if permission_errors and not all_actions: + message += "; permission_errors=" + " | ".join(permission_errors[:3]) + return { + "azure_rbac_level": level, + "azure_subscription_count": len(sub_ids), + "azure_subscription_ids": sub_ids[:10], + "azure_can_manage_ai": can_manage_ai, + "azure_can_manage_cognitive": can_manage_cognitive, + "azure_can_manage_ml": can_manage_ml, + "azure_can_assign_roles": can_assign_roles, + "azure_can_deploy_resources": can_deploy_resources, + "azure_permission_actions_sample": sorted(all_actions)[:40], + "message": message, + } + + +def check_service_principal(parsed, proxy, timeout): + url = f"https://login.microsoftonline.com/{parsed['tenant_id']}/oauth2/v2.0/token" + payload = { + "client_id": parsed["client_id"], + "client_secret": parsed["client_secret"], + "scope": "https://management.azure.com/.default", + "grant_type": "client_credentials", + } + try: + response = requests.post(url, data=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc)} + if response.status_code == 200: + data = response.json() + rbac = probe_azure_rbac(data.get("access_token"), proxy, timeout) + message = "token issued" + if rbac.get("message"): + message = f"{message}; {rbac.get('message')}" + return { + "status": "VALID", + "tenant_id": parsed["tenant_id"], + "client_id": parsed["client_id"], + "expires_in": data.get("expires_in"), + "message": message, + **rbac, + } + message = request_error_message(response) + lower = message.lower() + if response.status_code in (400, 401) and ("invalid_client" in lower or "invalid_grant" in lower): + return {"status": "DEAD", "http_status": response.status_code, "message": message} + if response.status_code in (401, 403): + return {"status": "RESTRICTED", "http_status": response.status_code, "message": message} + return {"status": "UNKNOWN", "http_status": response.status_code, "message": message} + + +def azure_openai_deployment_ids(payload): + data = payload.get("data") if isinstance(payload, dict) else None + if data is None and isinstance(payload, dict): + data = payload.get("value") + deployments = [] + for item in data or []: + if not isinstance(item, dict): + continue + deployment_id = item.get("id") or item.get("name") + model = item.get("model") or item.get("modelName") or "" + if deployment_id: + deployments.append({"id": deployment_id, "model": model}) + return deployments + + +def endpoint_url(endpoint, path): + endpoint = str(endpoint or "").strip().rstrip("/") + if not endpoint.startswith("http://") and not endpoint.startswith("https://"): + endpoint = "https://" + endpoint + path = "/" + str(path or "").lstrip("/") + parsed = urlsplit(endpoint) + endpoint_path = parsed.path.rstrip("/") + if endpoint_path and path.lower().startswith(endpoint_path.lower() + "/"): + path = path[len(endpoint_path):] + return endpoint + path + + +def azure_model_ids(payload): + data = payload.get("data") if isinstance(payload, dict) else payload if isinstance(payload, list) else [] + if data is None and isinstance(payload, dict): + data = payload.get("value") or payload.get("models") + models = [] + for item in data or []: + if isinstance(item, str): + models.append(item) + elif isinstance(item, dict): + model_id = item.get("id") or item.get("name") or item.get("model") or item.get("modelName") + if model_id: + models.append(str(model_id)) + return models + + +def endpoint_failure_status(error_text, status): + lower = str(error_text or "").lower() + if any(item in lower for item in ( + "name resolution", "no such host", "failed to resolve", "getaddrinfo", + "unexpected_eof", "eof occurred in violation of protocol", "ssleoferror", + )): + return status + return "NETWORK" + + +def probe_azure_openai_chat_route(endpoint, key, deployments, proxy, timeout): + if not deployments: + return {"route_probe": "no_deployments"} + preferred = None + for item in deployments: + text = f"{item.get('id', '')} {item.get('model', '')}".lower() + if any(marker in text for marker in ("gpt", "chat", "turbo", "4o")): + preferred = item + break + deployment = preferred or deployments[0] + deployment_id = deployment["id"] + url = f"https://{endpoint}/openai/deployments/{deployment_id}/chat/completions?api-version={AZURE_OPENAI_CHAT_API_VERSION}" + headers = {"api-key": key, "Content-Type": "application/json"} + # Empty messages should fail validation after auth/deployment routing, without generating content. + payload = {"messages": [], "max_tokens": 1} + try: + response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"route_probe": "network", "route_deployment": deployment_id, "route_message": str(exc)[:500]} + message = request_error_message(response) + if response.status_code in (200, 400): + return { + "route_probe": "accepted_auth_route", + "route_deployment": deployment_id, + "route_model": deployment.get("model", ""), + "route_http_status": response.status_code, + "route_message": message, + } + if response.status_code in (401, 403): + return {"route_probe": "auth_failed", "route_deployment": deployment_id, "route_http_status": response.status_code, "route_message": message} + if response.status_code == 404: + return {"route_probe": "not_found", "route_deployment": deployment_id, "route_http_status": 404, "route_message": message} + return {"route_probe": "unknown", "route_deployment": deployment_id, "route_http_status": response.status_code, "route_message": message} + + +def check_azure_openai(parsed, proxy, timeout, probe_openai_route=False): + endpoint = (parsed.get("endpoint") or "").strip().strip("/") + key = parsed.get("key") + if not endpoint: + return {"status": "OPENAI_UNRESOLVED", "message": "AzureOpenAI key found without endpoint/resource name"} + url = f"https://{endpoint}/openai/deployments?api-version={AZURE_OPENAI_DEPLOYMENTS_API_VERSION}" + headers = {"api-key": key, "Content-Type": "application/json"} + try: + response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + first_error = str(exc) + def endpoint_failure_status(error_text): + lower = str(error_text or "").lower() + if any(item in lower for item in ( + "name resolution", "no such host", "failed to resolve", "getaddrinfo", + "unexpected_eof", "eof occurred in violation of protocol", "ssleoferror", + )): + return {"status": "OPENAI_BAD_ENDPOINT", "endpoint": endpoint, "message": error_text} + return None + endpoint_status = endpoint_failure_status(first_error) + if endpoint_status: + return endpoint_status + return {"status": "NETWORK", "endpoint": endpoint, "message": first_error} + if response.status_code == 200: + deployments = azure_openai_deployment_ids(response.json()) + result = { + "status": "VALID", + "endpoint": endpoint, + "deployment_count": len(deployments), + "deployments": [item.get("id") for item in deployments[:20]], + "message": f"deployments endpoint accepted key; deployments={len(deployments)}", + } + if probe_openai_route: + result.update(probe_azure_openai_chat_route(endpoint, key, deployments, proxy, timeout)) + return result + message = request_error_message(response) + if response.status_code in (401, 403): + return {"status": "DEAD", "endpoint": endpoint, "http_status": response.status_code, "message": message} + if response.status_code == 404: + return {"status": "UNKNOWN", "endpoint": endpoint, "http_status": 404, "message": message} + if response.status_code >= 500: + return {"status": "NETWORK", "endpoint": endpoint, "http_status": response.status_code, "message": message} + return {"status": "UNKNOWN", "endpoint": endpoint, "http_status": response.status_code, "message": message} + + +def foundry_model_routes(endpoint): + return [ + endpoint_url(endpoint, f"/models?api-version={AZURE_FOUNDRY_API_VERSION}"), + endpoint_url(endpoint, "/models"), + endpoint_url(endpoint, "/v1/models"), + ] + + +def foundry_auth_headers(key): + return [ + {"api-key": key, "Content-Type": "application/json"}, + {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}, + ] + + +def check_foundry_models(endpoint, key, proxy, timeout): + attempts = [] + auth_failures = 0 + attempted = 0 + for url in foundry_model_routes(endpoint): + for headers in foundry_auth_headers(key): + attempted += 1 + auth_kind = "bearer" if "Authorization" in headers else "api-key" + try: + response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + status = endpoint_failure_status(str(exc), "FOUNDRY_BAD_ENDPOINT") + attempts.append(f"{url}:{auth_kind}:network:{str(exc)[:180]}") + if status == "FOUNDRY_BAD_ENDPOINT": + return {"status": status, "endpoint": endpoint, "message": str(exc)} + continue + message = request_error_message(response) + if response.status_code == 200: + models = azure_model_ids(response.json()) + return { + "status": "FOUNDRY", + "endpoint": endpoint, + "auth_scheme": auth_kind, + "model_count": len(models), + "models": models[:50], + "message": f"models endpoint accepted key; auth={auth_kind}; models={len(models)}", + } + if response.status_code == 429: + return { + "status": "FOUNDRY", + "endpoint": endpoint, + "auth_scheme": auth_kind, + "model_count": 0, + "models": [], + "message": f"models endpoint rate limited after auth; auth={auth_kind}; {message[:200]}", + } + if response.status_code in (401, 403): + auth_failures += 1 + attempts.append(f"{url}:{auth_kind}:auth_{response.status_code}:{message[:160]}") + continue + if response.status_code == 404: + attempts.append(f"{url}:{auth_kind}:http_404:{message[:160]}") + continue + if response.status_code >= 500: + attempts.append(f"{url}:{auth_kind}:server_{response.status_code}:{message[:160]}") + continue + attempts.append(f"{url}:{auth_kind}:http_{response.status_code}:{message[:160]}") + if attempted and auth_failures == attempted: + return {"status": "DEAD", "endpoint": endpoint, "message": "; ".join(attempts[:4])} + return {"status": "UNKNOWN", "endpoint": endpoint, "message": "; ".join(attempts[:4])} + + +def probe_foundry_route(endpoint, key, models, proxy, timeout): + configured = [item.strip() for item in (models or []) if item.strip()] + if not configured: + return {"foundry_route_probe": "not_configured"} + attempts = [] + accepted = [] + for model in configured: + route_specs = [ + (endpoint_url(endpoint, f"/models/chat/completions?api-version={AZURE_FOUNDRY_API_VERSION}"), {"model": model, "messages": [], "max_tokens": 1}), + (endpoint_url(endpoint, "/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}), + (endpoint_url(endpoint, "/v1/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}), + (endpoint_url(endpoint, "/openai/v1/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}), + ] + for url, payload in route_specs: + for headers in foundry_auth_headers(key): + auth_kind = "bearer" if "Authorization" in headers else "api-key" + try: + response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + attempts.append(f"{model}:{auth_kind}:network:{str(exc)[:120]}") + continue + message = request_error_message(response) + if response.status_code == 200: + accepted.append(model) + break + if response.status_code == 400 and any(item in message.lower() for item in ("messages", "content", "validation", "empty")): + accepted.append(model) + break + if response.status_code == 429: + accepted.append(model) + break + if response.status_code in (401, 403, 404): + attempts.append(f"{model}:{auth_kind}:http_{response.status_code}:{message[:160]}") + continue + attempts.append(f"{model}:{auth_kind}:http_{response.status_code}:{message[:160]}") + if model in accepted: + break + if accepted: + return {"foundry_route_probe": "accepted", "foundry_route_models": accepted, "foundry_route_message": "route accepted"} + return {"foundry_route_probe": "not_accepted", "foundry_route_models": [], "foundry_route_message": "; ".join(attempts[:8])} + + +def check_azure_foundry(parsed, proxy, timeout, probe_foundry_route_enabled=False, foundry_models=None): + endpoint = (parsed.get("endpoint") or "").strip().strip("/") + key = parsed.get("key") + if not endpoint: + return {"status": "FOUNDRY_UNRESOLVED", "message": "Azure Foundry key found without endpoint"} + result = check_foundry_models(endpoint, key, proxy, timeout) + if probe_foundry_route_enabled: + route = probe_foundry_route(endpoint, key, foundry_models or [], proxy, timeout) + if result.get("status") != "FOUNDRY" and route.get("foundry_route_probe") == "accepted": + result = {"status": "FOUNDRY", "endpoint": endpoint, "model_count": 0, "models": [], "message": "route accepted without model-list support"} + result.update(route) + return result + + +def is_azure_openai_llm(result): + if result.get("status") != "VALID": + return False + if int(result.get("deployment_count") or 0) <= 0: + return False + route_probe = result.get("route_probe") + if route_probe and route_probe != "accepted_auth_route": + return False + return True + + +def append_azure_openai_llm(key, result, source): + deployments = result.get("deployments") or [] + deployment_text = ",".join(str(item) for item in deployments[:20]) + details = " ".join(part for part in [ + f"deployments={int(result.get('deployment_count') or 0)}", + f"route_probe={result.get('route_probe') or ''}" if result.get("route_probe") else "", + f"route_deployment={result.get('route_deployment') or ''}" if result.get("route_deployment") else "", + f"route_model={result.get('route_model') or ''}" if result.get("route_model") else "", + f"deployment_ids={deployment_text}" if deployment_text else "", + ] if part) + append_status(AZURE_OPENAI_LLM_FILE, key, result.get("status", "VALID"), details, source) + + +def check_azure_acr(parsed, proxy, timeout): + username = parsed["username"] + password = parsed["password"] + url = f"https://{username}.azurecr.io/v2/" + try: + response = requests.get(url, auth=(username, password), proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + text = str(exc) + if "no such host" in text.lower(): + return {"status": "DEAD", "registry": username, "message": text} + return {"status": "NETWORK", "registry": username, "message": text} + if response.status_code == 200: + return {"status": "VALID", "registry": username, "message": "ACR /v2 accepted basic auth"} + message = request_error_message(response) + if response.status_code == 401: + return {"status": "DEAD", "registry": username, "http_status": 401, "message": message} + if response.status_code == 403: + return {"status": "RESTRICTED", "registry": username, "http_status": 403, "message": message} + if response.status_code >= 500: + return {"status": "NETWORK", "registry": username, "http_status": response.status_code, "message": message} + return {"status": "UNKNOWN", "registry": username, "http_status": response.status_code, "message": message} + + +def write_result(key, detector, result, source, finding): + safe_finding = strip_finding_nearby_context(finding) + write_keycheck_event(SERVICE, RESULTS_FILE, key, {"detector": detector, **result}, source, safe_finding, detector) + commit_status_transaction( + CHECKED_FILE, + transaction_status_files(), + key, + result["status"], + result.get("message", ""), + source, + ) + if detector == "AzureOpenAI" and is_azure_openai_llm(result): + append_azure_openai_llm(key, result, source) + record_validation_result(SERVICE, key, {"detector": detector, **result}, source, safe_finding, detector) + + +def strip_finding_nearby_context(finding): + if not isinstance(finding, dict): + return finding + output = dict(finding) + context = output.get("ScannerContext") + if isinstance(context, dict) and "nearby" in context: + output["ScannerContext"] = {key: value for key, value in context.items() if key != "nearby"} + return output + + +def parse_args(): + parser = argparse.ArgumentParser(description="Azure key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=20) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + parser.add_argument("--probe-openai-route", action="store_true", help="Probe Azure OpenAI chat route with an invalid no-generation request after deployment listing succeeds") + parser.add_argument("--probe-foundry-route", action="store_true", help="Probe Azure Foundry/MaaS chat route for configured models") + parser.add_argument("--foundry-models", default="", help="Comma-separated Azure Foundry model IDs to route-probe") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = set() + if args.retry_network: + retry_statuses.update({"NETWORK", "FOUNDRY_BAD_ENDPOINT", "OPENAI_BAD_ENDPOINT"}) + if args.retry_unknown: + retry_statuses.update({"UNKNOWN", "FOUNDRY_UNRESOLVED", "OPENAI_UNRESOLVED"}) + if args.retry_valid: + retry_statuses.update({"VALID", "FOUNDRY"}) + foundry_models = [item.strip() for item in str(args.foundry_models or "").split(",") if item.strip()] + processed = 0 + skipped = 0 + for key, detector, source, finding, parsed in extract_candidates(args.input): + if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=detector): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] {detector} candidate {mask_secret(key)} from {source}") + proxy = next(proxy_cycler) if proxy_cycler else None + if parsed.get('_unresolved_candidate'): + result = { + 'status': ( + 'FOUNDRY_UNRESOLVED' + if detector == 'AzureFoundry' else + 'OPENAI_UNRESOLVED' + if detector == 'AzureOpenAI' else + 'UNKNOWN' + ), + 'message': f"normalized {parsed.get('candidate_kind') or 'azure'} candidate is incomplete", + } + elif detector == "AzureOpenAI": + result = check_azure_openai(parsed, proxy, args.timeout, args.probe_openai_route) + elif detector == "AzureFoundry": + result = check_azure_foundry(parsed, proxy, args.timeout, args.probe_foundry_route, foundry_models) + if parsed.get("finding_uid"): + result["finding_uid"] = parsed.get("finding_uid") + elif detector == "AzureContainerRegistry": + result = check_azure_acr(parsed, proxy, args.timeout) + else: + result = check_service_principal(parsed, proxy, args.timeout) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + write_result(key, detector, result, source, finding) + known.add(key) + checked[key] = result["status"] + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/deepseek/deepseekKeycheck.py b/app/keycheckers/deepseek/deepseekKeycheck.py new file mode 100644 index 0000000..2080826 --- /dev/null +++ b/app/keycheckers/deepseek/deepseekKeycheck.py @@ -0,0 +1,292 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os +import re + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, + classify_common_http_status, + combined_provider_routing_hint, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + keycheck_input_mode, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + provider_routing_database_failed, + recover_status_transaction, + request_error_message, + record_validation_result, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) +from keycheckers.provider_resolution import resolve_provider_key + + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +PARENT_DIR = os.path.dirname(SCRIPT_DIR) +SERVICE = "deepseek" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) + +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +CHECKED_FILE = os.path.join(OUTPUT_DIR, "deepseekChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "deepseekResults.jsonl") + +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "deepseekAlive.txt"), + "NO_BALANCE": os.path.join(OUTPUT_DIR, "deepseekNoBalance.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "deepseekDead.txt"), + "LIMITED": os.path.join(OUTPUT_DIR, "deepseekLimited.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "deepseekNetwork.txt"), + "NO_CONTEXT": os.path.join(OUTPUT_DIR, "deepseekNoContext.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "deepseekUnknown.txt"), +} + +DEEPSEEK_REGEX = re.compile(r"sk-[a-z0-9]{32}") +DEEPSEEK_DETECTOR_NAMES = {"deepseek", "deepseekapikey", "deepseek_api_key"} +DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"} +QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"} +KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"} +QWEN_CONTEXT_REGEX = re.compile( + r"(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)", + re.IGNORECASE, +) +DEEPSEEK_CONTEXT_REGEX = re.compile(r"(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)", re.IGNORECASE) +KIMI_CONTEXT_REGEX = re.compile( + r"(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))", + re.IGNORECASE, +) +AMBIGUOUS_PROVIDER_HINT = "ambiguous_qwen_deepseek" +AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk" +GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"} +EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment" + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def iter_candidate_decisions(input_file, plain_files): + detector_names = ["DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key", "CustomRegex"] + for item in iter_findings(input_file, detector_names): + finding = item.get("finding") or {} + if not finding_has_deepseek_detector(finding): + continue + key = item.get("credential_secret_text") or item["raw"] + if key and DEEPSEEK_REGEX.fullmatch(key): + hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding)) + if provider_routing_database_failed(): + raise RuntimeError("provider routing evidence lookup failed closed") + yield key, item["source"], finding, hint + + +def extract_candidates(input_file, plain_files): + for key, source, finding, hint in iter_candidate_decisions(input_file, plain_files): + if hint == "deepseek": + yield key, source, finding + + +def route_rejection_result(hint): + normalized = str(hint or "missing").strip().lower() + return { + "status": "NO_CONTEXT", + "routing_hint": normalized, + "message": f"candidate is not safely attributable to DeepSeek; routing_hint={normalized}", + } + + +def finding_detector_names(finding): + if not isinstance(finding, dict): + return set() + extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {} + names = { + str(finding.get("DetectorName") or finding.get("detector") or "").strip().lower(), + str(extra.get("name") or "").strip().lower(), + } + return {name for name in names if name} + + +def finding_has_deepseek_detector(finding): + return bool(finding_detector_names(finding) & DEEPSEEK_DETECTOR_NAMES) + + +def finding_has_explicit_detector(finding, detector_names): + return bool(finding_detector_names(finding) & set(detector_names)) + + +def finding_provider_routing_hint(finding): + if not isinstance(finding, dict): + return "" + context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {} + persisted_hint = context.get("provider_hint") + if ( + context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE + and persisted_hint in (*GENERIC_SK_PROVIDERS, AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT) + ): + return persisted_hint + parts = [] + for key in ("nearby", "file"): + if context.get(key): + parts.append(str(context.get(key))) + metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {} + data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {} + for details in data.values(): + if not isinstance(details, dict): + continue + for key in ("file", "repository", "repo", "link", "image"): + if details.get(key): + parts.append(str(details.get(key))) + text = "\n".join(parts) + evidence = set() + if QWEN_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES): + evidence.add("qwen") + if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES): + evidence.add("deepseek") + if KIMI_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES): + evidence.add("kimi") + if persisted_hint == AMBIGUOUS_PROVIDER_HINT: + evidence.update(("qwen", "deepseek")) + elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT: + evidence.update(GENERIC_SK_PROVIDERS) + elif persisted_hint in GENERIC_SK_PROVIDERS: + evidence.add(persisted_hint) + if len(evidence) > 1: + return AMBIGUOUS_PROVIDER_HINT if evidence == {"qwen", "deepseek"} else AMBIGUOUS_GENERIC_SK_HINT + return next(iter(evidence)) if evidence else "" + + +def finding_has_ambiguous_provider_hint(finding): + return finding_provider_routing_hint(finding) in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT) + + +def finding_looks_like_qwen_context(finding): + return finding_provider_routing_hint(finding) == "qwen" + + +def check_key(key, proxy, timeout): + url = "https://api.deepseek.com/user/balance" + headers = {"Authorization": f"Bearer {key}"} + try: + response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc)} + + if response.status_code == 200: + data = response.json() + balance_infos = data.get("balance_infos", []) + total_usd = 0.0 + for balance in balance_infos: + amount = float(balance.get("total_balance", "0") or 0) + currency = balance.get("currency", "USD") + if currency == "CNY": + amount *= 0.14 + total_usd += amount + available = bool(data.get("is_available", False)) + status = "VALID" if available and total_usd > 0 else "NO_BALANCE" + return { + "status": status, + "authenticated": True, + "available": available, + "balance_usd": round(total_usd, 4), + "message": f"available={available}; balance=${total_usd:.4f}", + } + + status = classify_common_http_status(response.status_code) + return {"status": status, "http_status": response.status_code, "message": request_error_message(response)} + + +def write_result(key, result, source, finding): + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "DeepSeek") + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), source, + ) + record_validation_result(SERVICE, key, result, source, finding, "DeepSeek") + + +def parse_args(): + parser = argparse.ArgumentParser(description="DeepSeek key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=20) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-no-balance", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = set() + if args.retry_network: + retry_statuses.add("NETWORK") + if args.retry_limited: + retry_statuses.add("LIMITED") + if args.retry_unknown: + retry_statuses.update({"UNKNOWN", "NO_CONTEXT"}) + if args.retry_no_balance: + retry_statuses.add("NO_BALANCE") + if args.retry_valid: + retry_statuses.add("VALID") + + processed = 0 + skipped = 0 + postgres_mode = keycheck_input_mode() == "postgres" + for key, source, finding, routing_hint in iter_candidate_decisions(args.input, args.plain): + route_rejected = routing_hint != "deepseek" + ambiguous_route = routing_hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT) + if route_rejected and not postgres_mode: + skipped += 1 + continue + if not route_rejected and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="DeepSeek"): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] DeepSeek candidate {mask_secret(key)} from {source}") + if route_rejected and ambiguous_route: + proxy = next(proxy_cycler) if proxy_cycler else None + result = resolve_provider_key( + key, finding, proxy, args.timeout, + hint=routing_hint, origin_service=SERVICE, + ) + elif route_rejected: + result = route_rejection_result(routing_hint) + else: + proxy = next(proxy_cycler) if proxy_cycler else None + result = check_key(key, proxy, args.timeout) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/dockerhub/dockerhubKeycheck.py b/app/keycheckers/dockerhub/dockerhubKeycheck.py new file mode 100644 index 0000000..8182f89 --- /dev/null +++ b/app/keycheckers/dockerhub/dockerhubKeycheck.py @@ -0,0 +1,240 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import base64 +import json +import os +import re +import time + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + keycheck_input_mode, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + recover_status_transaction, + request_error_message, + record_validation_result, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) + + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +PARENT_DIR = os.path.dirname(SCRIPT_DIR) +SERVICE = "dockerhub" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) + +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +PLAIN_FILE = os.path.join(OUTPUT_DIR, "dockerhub.txt") +CHECKED_FILE = os.path.join(OUTPUT_DIR, "dockerhubChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "dockerhubResults.jsonl") + +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "dockerhubAlive.txt"), + "VALID_2FA": os.path.join(OUTPUT_DIR, "dockerhubAlive.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "dockerhubDead.txt"), + "NO_USERNAME": os.path.join(OUTPUT_DIR, "dockerhubNoUsername.txt"), + "RATE_LIMITED": os.path.join(OUTPUT_DIR, "dockerhubRateLimited.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "dockerhubNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "dockerhubUnknown.txt"), +} + +DOCKER_PAT_RE = re.compile(r"\bdckr_pat_[A-Za-z0-9_-]{27}\b") + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *set(STATUS_FILES.values()), PLAIN_FILE]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def candidate_key(username, token): + return f"{username}:{token}" if username else token + + +def parse_username_token(raw, raw_v2, finding): + raw = raw or "" + raw_v2 = raw_v2 or "" + token = "" + username = "" + + if ":" in raw_v2: + maybe_user, maybe_token = raw_v2.split(":", 1) + if DOCKER_PAT_RE.fullmatch(maybe_token): + username, token = maybe_user.strip(), maybe_token.strip() + if not token: + match = DOCKER_PAT_RE.search(raw_v2) or DOCKER_PAT_RE.search(raw) + if match: + token = match.group(0) + + extra = finding.get("ExtraData") if isinstance(finding, dict) else {} + analysis = finding.get("AnalysisInfo") if isinstance(finding, dict) else {} + if isinstance(extra, dict): + username = username or extra.get("hub_username") or "" + if isinstance(analysis, dict): + username = username or analysis.get("username") or "" + return username, token + + +def extract_candidates(input_file, plain_file): + seen_plain = set() + for item in iter_findings(input_file, ["Dockerhub"]): + username, token = parse_username_token(item.get("raw"), item.get("raw_v2"), item.get("finding") or {}) + if not token: + continue + key = candidate_key(username, token) + yield key, username, token, item["source"], item["finding"] + + if keycheck_input_mode() == 'jsonl' and os.path.exists(plain_file): + with open(plain_file, "r", encoding="utf-8") as f: + for line_num, line in enumerate(f, 1): + line = line.strip() + if not line: + continue + username = "" + token = "" + if ":" in line: + maybe_user, rest = line.split(":", 1) + match = DOCKER_PAT_RE.search(rest) + if match: + username, token = maybe_user.strip(), match.group(0) + else: + match = DOCKER_PAT_RE.search(line) + if match: + token = match.group(0) + if not token: + continue + key = candidate_key(username, token) + if key not in seen_plain: + seen_plain.add(key) + yield key, username, token, f"{plain_file}:{line_num}", {} + + +def decode_jwt_payload(jwt_token): + try: + payload = jwt_token.split(".")[1] + payload += "=" * (-len(payload) % 4) + return json.loads(base64.urlsafe_b64decode(payload.encode()).decode()) + except Exception: + return {} + + +def check_token(username, token, proxy, timeout): + if not username: + return {"status": "NO_USERNAME", "message": "DockerHub PAT requires username/email for login check"} + + url = "https://hub.docker.com/v2/users/login" + payload = {"username": username, "password": token} + try: + response = requests.post(url, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc), "username": username} + + message = request_error_message(response) + if response.status_code == 200: + data = response.json() + hub_token = data.get("token", "") + claims = decode_jwt_payload(hub_token) if hub_token else {} + hub_claims = claims.get("https://hub.docker.com", {}) if isinstance(claims, dict) else {} + return { + "status": "VALID", + "message": "login accepted", + "username": username, + "hub_username": hub_claims.get("username", username), + "hub_email": hub_claims.get("email", ""), + "scope": claims.get("scope", "") if isinstance(claims, dict) else "", + } + if response.status_code == 401: + try: + data = response.json() + except ValueError: + data = {} + if data.get("login_2fa_token"): + return {"status": "VALID_2FA", "message": "credentials accepted; 2FA required", "username": username} + return {"status": "DEAD", "http_status": 401, "message": message, "username": username} + if response.status_code == 429: + return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "username": username} + if response.status_code >= 500: + return {"status": "NETWORK", "http_status": response.status_code, "message": message, "username": username} + return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "username": username} + + +def write_result(key, result, source, finding): + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "Dockerhub") + extra = result.get("hub_username") or result.get("username") or source + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra, + ) + record_validation_result(SERVICE, key, result, source, finding, "Dockerhub") + + +def parse_args(): + parser = argparse.ArgumentParser(description="DockerHub PAT checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", default=PLAIN_FILE) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=20) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-no-username", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = set() + if args.retry_network: + retry_statuses.add("NETWORK") + if args.retry_limited: + retry_statuses.add("RATE_LIMITED") + if args.retry_unknown: + retry_statuses.add("UNKNOWN") + if args.retry_no_username: + retry_statuses.add("NO_USERNAME") + + processed = 0 + skipped = 0 + for key, username, token, source, finding in extract_candidates(args.input, args.plain): + if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="Dockerhub"): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] DockerHub candidate {mask_secret(key)} from {source}") + proxy = next(proxy_cycler) if proxy_cycler else None + result = check_token(username, token, proxy, args.timeout) + print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + time.sleep(0.1) + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/gcp/gcpKeycheck.py b/app/keycheckers/gcp/gcpKeycheck.py new file mode 100644 index 0000000..353c726 --- /dev/null +++ b/app/keycheckers/gcp/gcpKeycheck.py @@ -0,0 +1,789 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import base64 +import binascii +import hashlib +import json +import os +import re +import time +from urllib.parse import urlsplit + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, + append_status, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + keycheck_input_mode, + load_checked_statuses, + load_known_keys, + load_known_statuses, + load_proxies, + mask_secret, + recover_status_transaction, + request_error_message, + record_cached_keycheck_occurrence, + record_validation_result, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) + + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +PARENT_DIR = os.path.dirname(SCRIPT_DIR) +SERVICE = "gcp" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) + +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +PLAIN_FILE = os.path.join(OUTPUT_DIR, "gcp.txt") +CHECKED_FILE = os.path.join(OUTPUT_DIR, "gcpChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "gcpResults.jsonl") + +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "gcpAlive.txt"), + "VERTEX": os.path.join(OUTPUT_DIR, "gcpVertex.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "gcpDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "gcpRestricted.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "gcpNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "gcpUnknown.txt"), + "NO_CONTEXT": os.path.join(OUTPUT_DIR, "gcpNoContext.txt"), +} +VERTEX_GEMINI_FILE = os.path.join(OUTPUT_DIR, "gcpVertexGemini.txt") +VERTEX_ANTHROPIC_FILE = os.path.join(OUTPUT_DIR, "gcpVertexAnthropic.txt") + +GOOGLE_TOKEN_URL = "https://oauth2.googleapis.com/token" +TRUSTED_GOOGLE_TOKEN_ENDPOINTS = frozenset({ + GOOGLE_TOKEN_URL, + "https://accounts.google.com/o/oauth2/token", +}) +TOKEN_REDIRECT_STATUSES = {301, 302, 303, 307, 308} +SA_SCOPE = "https://www.googleapis.com/auth/cloud-platform" +MAX_PEM_BYTES = 24 * 1024 +MAX_DER_BYTES = 16 * 1024 +MAX_DER_LENGTH_BYTES = 2 +MIN_RSA_BITS = 2048 +MAX_RSA_BITS = 8192 +MAX_RSA_INTEGER_BYTES = MAX_RSA_BITS // 8 +RSA_ENCRYPTION_OID = bytes.fromhex("2a864886f70d010101") +VERTEX_LOCATIONS = ["global", "us", "eu"] +VERTEX_MODELS = ["gemini-3.6-flash", "gemini-3.1-pro-preview"] +VERTEX_ANTHROPIC_LOCATIONS = ["global", "us", "eu", "us-east5", "europe-west1"] +VERTEX_ANTHROPIC_MODELS = ["claude-opus-5", "claude-opus-4-7", "claude-opus-4-6", "claude-fable-5"] + + +def transaction_status_files(): + return { + **STATUS_FILES, + "RATE_LIMITED": STATUS_FILES["UNKNOWN"], + "AUX_VERTEX_GEMINI": VERTEX_GEMINI_FILE, + "AUX_VERTEX_ANTHROPIC": VERTEX_ANTHROPIC_FILE, + } + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), VERTEX_GEMINI_FILE, VERTEX_ANTHROPIC_FILE, PLAIN_FILE]) + recover_status_transaction(CHECKED_FILE, transaction_status_files()) + + +def b64url(data): + return base64.urlsafe_b64encode(data).rstrip(b"=").decode() + + +def validate_google_token_uri(value): + token_uri = str(value or GOOGLE_TOKEN_URL).strip() + try: + parsed = urlsplit(token_uri) + port = parsed.port + except (TypeError, ValueError) as exc: + raise ValueError("invalid Google OAuth token_uri") from exc + if ( + parsed.scheme != "https" + or parsed.username is not None + or parsed.password is not None + or port not in (None, 443) + or parsed.query + or parsed.fragment + or token_uri not in TRUSTED_GOOGLE_TOKEN_ENDPOINTS + ): + raise ValueError("untrusted Google OAuth token_uri") + return token_uri + + +class InvalidRSAPrivateKey(ValueError): + pass + + +class DERReader: + def __init__(self, data): + if not isinstance(data, (bytes, bytearray, memoryview)): + raise InvalidRSAPrivateKey("DER value is not binary") + if len(data) > MAX_DER_BYTES: + raise InvalidRSAPrivateKey("DER value exceeds size limit") + self.data = data + self.pos = 0 + + def read_tlv(self): + if len(self.data) - self.pos < 2: + raise InvalidRSAPrivateKey("truncated DER tag or length") + tag = self.data[self.pos] + self.pos += 1 + first_len = self.data[self.pos] + self.pos += 1 + if first_len & 0x80: + length_len = first_len & 0x7F + if length_len == 0: + raise InvalidRSAPrivateKey("indefinite DER length is not allowed") + if length_len > MAX_DER_LENGTH_BYTES: + raise InvalidRSAPrivateKey("DER length-of-length exceeds limit") + if len(self.data) - self.pos < length_len: + raise InvalidRSAPrivateKey("truncated DER length") + length_bytes = self.data[self.pos:self.pos + length_len] + if length_bytes[0] == 0: + raise InvalidRSAPrivateKey("non-minimal DER length") + length = int.from_bytes(length_bytes, "big") + self.pos += length_len + if length < 0x80: + raise InvalidRSAPrivateKey("non-minimal DER length") + else: + length = first_len + if length > MAX_DER_BYTES: + raise InvalidRSAPrivateKey("DER value length exceeds limit") + if length > len(self.data) - self.pos: + raise InvalidRSAPrivateKey("truncated DER value") + value = self.data[self.pos:self.pos + length] + self.pos += length + return tag, value + + def expect(self, tag): + actual, value = self.read_tlv() + if actual != tag: + raise InvalidRSAPrivateKey(f"expected DER tag {tag:#x}, got {actual:#x}") + return value + + def at_end(self): + return self.pos == len(self.data) + + def require_eof(self, context="DER structure"): + if not self.at_end(): + raise InvalidRSAPrivateKey(f"trailing data in {context}") + + +def der_int(value, name="integer", max_bytes=MAX_RSA_INTEGER_BYTES): + if not value: + raise InvalidRSAPrivateKey(f"empty RSA {name}") + if len(value) > max_bytes + 1: + raise InvalidRSAPrivateKey(f"RSA {name} exceeds size limit") + if value[0] & 0x80: + raise InvalidRSAPrivateKey(f"negative RSA {name}") + if value[0] == 0: + if len(value) > 1 and not value[1] & 0x80: + raise InvalidRSAPrivateKey(f"non-minimal RSA {name}") + unsigned = value[1:] + else: + unsigned = value + if len(unsigned) > max_bytes: + raise InvalidRSAPrivateKey(f"RSA {name} exceeds size limit") + return int.from_bytes(unsigned, "big") if unsigned else 0 + + +def parse_pkcs1_rsa_private_key(data): + rsa = DERReader(data) + version = der_int(rsa.expect(0x02), "version", 1) + if version != 0: + raise InvalidRSAPrivateKey("unsupported RSA private key version") + n = der_int(rsa.expect(0x02), "modulus") + public_exponent = der_int(rsa.expect(0x02), "public exponent") + d = der_int(rsa.expect(0x02), "private exponent") + for name in ("prime1", "prime2", "exponent1", "exponent2", "coefficient"): + der_int(rsa.expect(0x02), name) + rsa.require_eof("RSA private key") + + modulus_bits = n.bit_length() + if not MIN_RSA_BITS <= modulus_bits <= MAX_RSA_BITS: + raise InvalidRSAPrivateKey( + f"RSA modulus must be between {MIN_RSA_BITS} and {MAX_RSA_BITS} bits" + ) + if public_exponent == 0: + raise InvalidRSAPrivateKey("RSA public exponent is zero") + if d == 0 or d >= n: + raise InvalidRSAPrivateKey("RSA private exponent is out of range") + return n, d + + +def validate_rsa_algorithm_identifier(data): + algorithm = DERReader(data) + if algorithm.expect(0x06) != RSA_ENCRYPTION_OID: + raise InvalidRSAPrivateKey("PKCS#8 key does not use rsaEncryption") + if not algorithm.at_end() and algorithm.expect(0x05): + raise InvalidRSAPrivateKey("invalid rsaEncryption parameters") + algorithm.require_eof("PKCS#8 algorithm identifier") + + +def parse_rsa_private_key_from_pem(pem): + pem = str(pem or "") + if len(pem) > MAX_PEM_BYTES: + raise InvalidRSAPrivateKey("PEM private key exceeds size limit") + try: + pem_bytes = pem.encode("utf-8") + except UnicodeEncodeError as exc: + raise InvalidRSAPrivateKey("PEM private key is not valid UTF-8") from exc + if len(pem_bytes) > MAX_PEM_BYTES: + raise InvalidRSAPrivateKey("PEM private key exceeds size limit") + pem = pem.replace("\\n", "\n") + match = re.fullmatch( + r"\s*-----BEGIN (RSA PRIVATE KEY|PRIVATE KEY)-----\s*(.*?)\s*-----END \1-----\s*", + pem, + re.DOTALL, + ) + if not match: + raise InvalidRSAPrivateKey("missing complete PEM private key block") + body = re.sub(r"\s+", "", match.group(2)) + if len(body) < 256: + raise InvalidRSAPrivateKey("PEM private key body is too short") + try: + der = base64.b64decode(body + ("=" * (-len(body) % 4)), validate=True) + except (binascii.Error, ValueError) as exc: + raise InvalidRSAPrivateKey("invalid PEM base64") from exc + if len(der) > MAX_DER_BYTES: + raise InvalidRSAPrivateKey("DER private key exceeds size limit") + + reader = DERReader(der) + top_bytes = reader.expect(0x30) + reader.require_eof("DER private key") + top = DERReader(top_bytes) + + # PKCS#8 PrivateKeyInfo: SEQUENCE(version, alg, OCTET STRING(RSAPrivateKey)) + first_tag, first_val = top.read_tlv() + if first_tag != 0x02: + raise InvalidRSAPrivateKey("unexpected private key structure") + second_tag, second_val = top.read_tlv() + if second_tag == 0x30: + if der_int(first_val, "PKCS#8 version", 1) != 0: + raise InvalidRSAPrivateKey("unsupported PKCS#8 version") + validate_rsa_algorithm_identifier(second_val) + private_octet = top.expect(0x04) + top.require_eof("PKCS#8 private key") + wrapped = DERReader(private_octet) + rsa_bytes = wrapped.expect(0x30) + wrapped.require_eof("PKCS#8 private key octets") + return parse_pkcs1_rsa_private_key(rsa_bytes) + if second_tag != 0x02: + raise InvalidRSAPrivateKey("unexpected private key structure") + return parse_pkcs1_rsa_private_key(top_bytes) + + +def rsa_pkcs1v15_sha256_sign(message, pem): + n, d = parse_rsa_private_key_from_pem(pem) + digest = hashlib.sha256(message).digest() + digest_info = bytes.fromhex("3031300d060960864801650304020105000420") + digest + key_len = (n.bit_length() + 7) // 8 + if key_len < len(digest_info) + 11: + raise InvalidRSAPrivateKey("RSA key too small") + encoded = b"\x00\x01" + b"\xff" * (key_len - len(digest_info) - 3) + b"\x00" + digest_info + sig = pow(int.from_bytes(encoded, "big"), d, n).to_bytes(key_len, "big") + return sig + + +def make_service_account_assertion(creds): + now = int(time.time()) + token_uri = validate_google_token_uri(creds.get("token_uri")) + header = {"alg": "RS256", "typ": "JWT", "kid": creds.get("private_key_id")} + payload = { + "iss": creds["client_email"], + "scope": SA_SCOPE, + "aud": token_uri, + "iat": now, + "exp": now + 3600, + } + signing_input = (b64url(json.dumps(header, separators=(",", ":")).encode()) + "." + b64url(json.dumps(payload, separators=(",", ":")).encode())).encode() + signature = rsa_pkcs1v15_sha256_sign(signing_input, creds["private_key"]) + return signing_input.decode() + "." + b64url(signature), token_uri + + +def compact_json(data): + return json.dumps(data, ensure_ascii=False, separators=(",", ":"), sort_keys=True) + + +def scanner_context_text(finding): + context = finding.get("ScannerContext") if isinstance(finding, dict) else None + if isinstance(context, dict): + return str(context.get("nearby") or "") + return "" + + +def parse_json_object(text): + try: + return json.loads(text) + except (TypeError, ValueError): + pass + match = re.search(r"\{.*\}", str(text or ""), re.DOTALL) + if not match: + return None + try: + return json.loads(match.group(0)) + except ValueError: + return None + + +def parse_service_account(raw_v2): + data = parse_json_object(raw_v2) + if not isinstance(data, dict): + return None + required = ["client_email", "private_key", "private_key_id"] + if not all(data.get(item) for item in required): + return None + private_key = str(data.get("private_key") or "") + if "-----BEGIN" not in private_key or "-----END" not in private_key or len(private_key) < 800: + return None + return data + + +def parse_adc(raw_v2, finding): + data = parse_json_object(scanner_context_text(finding)) or parse_json_object(raw_v2) + if not isinstance(data, dict): + return None + required = ["client_id", "client_secret", "refresh_token"] + if not all(data.get(item) for item in required): + return None + return data + + +def extract_candidates(input_file, plain_file): + seen_plain = set() + for item in iter_findings(input_file, ["GCP", "GCPApplicationDefaultCredentials"]): + if item["detector"] == "GCP": + parsed = parse_service_account(item["raw_v2"]) + detector = "GCP" + else: + parsed = parse_adc(item["raw_v2"], item["finding"]) + detector = "GCPApplicationDefaultCredentials" + if not parsed: + key = f"{detector}:no_context:{item['source']}" + yield key, detector, item["source"], item["finding"], None + continue + key = compact_json(parsed) + yield key, detector, item["source"], item["finding"], parsed + + if keycheck_input_mode() == 'jsonl' and os.path.exists(plain_file): + with open(plain_file, "r", encoding="utf-8", errors="replace") as f: + content = f.read() + candidates = [] + whole_file = parse_json_object(content) + if whole_file: + candidates.append((plain_file, whole_file)) + for line_num, line in enumerate(content.splitlines(), 1): + key_text = line.split("\t", 1)[0].strip() + parsed = parse_json_object(key_text) or parse_json_object(line) + if parsed: + candidates.append((f"{plain_file}:{line_num}", parsed)) + for source, parsed in candidates: + detector = "GCP" if parsed.get("private_key") else "GCPApplicationDefaultCredentials" + key = compact_json(parsed) + if key not in seen_plain: + seen_plain.add(key) + yield key, detector, source, {}, parsed + + +def vertex_api_host(location): + location = str(location or "").strip().lower() + if location == "global": + return "aiplatform.googleapis.com" + if location in ("us", "eu"): + return f"aiplatform.{location}.rep.googleapis.com" + return f"{location}-aiplatform.googleapis.com" + + +def probe_vertex_llm(access_token, project_id, proxy, timeout, locations=None, models=None, max_attempts=2): + if not project_id: + return {"enabled": False, "message": "project_id unavailable"} + headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"} + payload = {"contents": [{"role": "user", "parts": [{"text": "ping"}]}]} + attempts = [] + accepted = [] + tried = 0 + for location in (locations or VERTEX_LOCATIONS): + for model in (models or VERTEX_MODELS): + if max_attempts and tried >= max_attempts: + if accepted: + first = accepted[0] + return { + "enabled": True, + "location": first.get("location", ""), + "model": first.get("model", ""), + "total_tokens": first.get("total_tokens"), + "available_models": [f"google/{item['location']}/{item['model']}" for item in accepted], + "message": "Vertex countTokens accepted", + } + return {"enabled": False, "message": "; ".join(attempts[:8]) or "Vertex probe attempt limit reached"} + tried += 1 + url = f"https://{vertex_api_host(location)}/v1/projects/{project_id}/locations/{location}/publishers/google/models/{model}:countTokens" + try: + response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + attempts.append(f"{location}:{model}:network:{str(exc)[:120]}") + continue + if response.status_code == 200: + data = response.json() + accepted.append({ + "location": location, + "model": model, + "total_tokens": data.get("totalTokens") or data.get("total_tokens"), + }) + continue + message = request_error_message(response) + if response.status_code in (400, 401, 403, 404, 429): + attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}") + continue + if response.status_code >= 500: + attempts.append(f"{location}:{model}:server_{response.status_code}:{message[:160]}") + continue + attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}") + if accepted: + first = accepted[0] + return { + "enabled": True, + "location": first.get("location", ""), + "model": first.get("model", ""), + "total_tokens": first.get("total_tokens"), + "available_models": [f"google/{item['location']}/{item['model']}" for item in accepted], + "message": "Vertex countTokens accepted", + } + return {"enabled": False, "message": "; ".join(attempts[:8])} + + +def probe_vertex_anthropic(access_token, project_id, proxy, timeout, locations=None, models=None, max_attempts=2): + if not project_id: + return {"enabled": False, "message": "project_id unavailable"} + models = models or [] + if not models: + return {"enabled": False, "message": "no Anthropic models configured"} + headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"} + payload = { + "anthropic_version": "vertex-2023-10-16", + "messages": [{"role": "user", "content": "ping"}], + "max_tokens": 1, + } + attempts = [] + accepted = [] + tried = 0 + for location in (locations or VERTEX_ANTHROPIC_LOCATIONS): + for model in models: + if max_attempts and tried >= max_attempts: + if accepted: + first = accepted[0] + return { + "enabled": True, + "location": first.get("location", ""), + "model": first.get("model", ""), + "available_models": [f"anthropic/{item['location']}/{item['model']}" for item in accepted], + "message": "Vertex Anthropic rawPredict accepted", + } + return {"enabled": False, "message": "; ".join(attempts[:8]) or "Vertex Anthropic probe attempt limit reached"} + tried += 1 + url = f"https://{vertex_api_host(location)}/v1/projects/{project_id}/locations/{location}/publishers/anthropic/models/{model}:rawPredict" + try: + response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + attempts.append(f"{location}:{model}:network:{str(exc)[:120]}") + continue + if response.status_code == 200: + accepted.append({"location": location, "model": model}) + continue + message = request_error_message(response) + if response.status_code in (400, 401, 403, 404, 429): + attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}") + continue + if response.status_code >= 500: + attempts.append(f"{location}:{model}:server_{response.status_code}:{message[:160]}") + continue + attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}") + if accepted: + first = accepted[0] + return { + "enabled": True, + "location": first.get("location", ""), + "model": first.get("model", ""), + "available_models": [f"anthropic/{item['location']}/{item['model']}" for item in accepted], + "message": "Vertex Anthropic rawPredict accepted", + } + return {"enabled": False, "message": "; ".join(attempts[:8])} + + +def merge_vertex_results(google_vertex, anthropic_vertex): + google_vertex = google_vertex or {"enabled": False, "message": ""} + anthropic_vertex = anthropic_vertex or {"enabled": False, "message": ""} + available = [] + available.extend(google_vertex.get("available_models") or []) + available.extend(anthropic_vertex.get("available_models") or []) + first = google_vertex if google_vertex.get("enabled") else anthropic_vertex if anthropic_vertex.get("enabled") else {} + messages = [] + if google_vertex.get("message"): + messages.append(f"google: {google_vertex.get('message')}") + if anthropic_vertex.get("message"): + messages.append(f"anthropic: {anthropic_vertex.get('message')}") + return { + "enabled": bool(available), + "location": first.get("location", ""), + "model": first.get("model", ""), + "total_tokens": first.get("total_tokens"), + "available_models": available, + "google_enabled": bool(google_vertex.get("enabled")), + "anthropic_enabled": bool(anthropic_vertex.get("enabled")), + "message": "; ".join(messages), + } + + +def check_service_account(creds, proxy, timeout, probe_vertex=False, vertex_timeout=6, vertex_locations=None, vertex_models=None, vertex_max_attempts=2, vertex_anthropic_locations=None, vertex_anthropic_models=None, vertex_anthropic_max_attempts=2): + try: + assertion, token_uri = make_service_account_assertion(creds) + except InvalidRSAPrivateKey as exc: + return { + "status": "DEAD", + "classification": "invalid_private_key", + "message": f"invalid RSA private key: {exc}", + "project_id": creds.get("project_id"), + "client_email": creds.get("client_email"), + } + except Exception as exc: + return {"status": "UNKNOWN", "message": f"failed to build JWT assertion: {exc}"} + data = {"grant_type": "urn:ietf:params:oauth:grant-type:jwt-bearer", "assertion": assertion} + try: + response = requests.post( + token_uri, data=data, proxies=proxy, timeout=timeout, allow_redirects=False, + ) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc), "project_id": creds.get("project_id"), "client_email": creds.get("client_email")} + if response.status_code in TOKEN_REDIRECT_STATUSES: + return { + "status": "UNKNOWN", + "http_status": response.status_code, + "message": "Google OAuth token endpoint redirect refused", + "project_id": creds.get("project_id"), + "client_email": creds.get("client_email"), + } + if response.status_code == 200: + payload = response.json() + result = { + "status": "VALID", + "message": "OAuth token issued", + "project_id": creds.get("project_id"), + "client_email": creds.get("client_email"), + "private_key_id": creds.get("private_key_id"), + "expires_in": payload.get("expires_in"), + } + if probe_vertex: + google_vertex = probe_vertex_llm( + payload.get("access_token"), creds.get("project_id"), proxy, + vertex_timeout, vertex_locations, vertex_models, vertex_max_attempts, + ) + anthropic_vertex = probe_vertex_anthropic( + payload.get("access_token"), creds.get("project_id"), proxy, + vertex_timeout, vertex_anthropic_locations, vertex_anthropic_models, vertex_anthropic_max_attempts, + ) if vertex_anthropic_models else {"enabled": False, "message": ""} + vertex = merge_vertex_results(google_vertex, anthropic_vertex) + result.update({ + "vertex_enabled": vertex.get("enabled"), + "vertex_location": vertex.get("location", ""), + "vertex_model": vertex.get("model", ""), + "vertex_available_models": vertex.get("available_models") or [], + "vertex_google_enabled": vertex.get("google_enabled"), + "vertex_anthropic_enabled": vertex.get("anthropic_enabled"), + "vertex_total_tokens": vertex.get("total_tokens"), + "vertex_message": vertex.get("message", ""), + }) + if vertex.get("enabled"): + result["status"] = "VERTEX" + result["message"] = "OAuth token issued; Vertex countTokens accepted" + return result + message = request_error_message(response) + lower = message.lower() + if response.status_code in (400, 401) and any(item in lower for item in ("invalid_grant", "invalid_client", "invalid jwt", "invalid signature")): + return {"status": "DEAD", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")} + if response.status_code in (401, 403): + return {"status": "RESTRICTED", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")} + if response.status_code == 429: + return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")} + if response.status_code >= 500: + return {"status": "NETWORK", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")} + return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")} + + +def check_adc(creds, proxy, timeout, probe_vertex=False, vertex_timeout=6, vertex_locations=None, vertex_models=None, vertex_max_attempts=2, vertex_anthropic_locations=None, vertex_anthropic_models=None, vertex_anthropic_max_attempts=2): + try: + token_uri = validate_google_token_uri(creds.get("token_uri")) + except ValueError as exc: + return {"status": "UNKNOWN", "message": str(exc), "client_id": creds.get("client_id")} + data = { + "client_id": creds["client_id"], + "client_secret": creds["client_secret"], + "refresh_token": creds["refresh_token"], + "grant_type": "refresh_token", + } + try: + response = requests.post( + token_uri, data=data, proxies=proxy, timeout=timeout, allow_redirects=False, + ) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc), "client_id": creds.get("client_id")} + if response.status_code in TOKEN_REDIRECT_STATUSES: + return { + "status": "UNKNOWN", + "http_status": response.status_code, + "message": "Google OAuth token endpoint redirect refused", + "client_id": creds.get("client_id"), + } + if response.status_code == 200: + payload = response.json() + project_id = creds.get("quota_project_id") or creds.get("project_id") + result = {"status": "VALID", "message": "refresh token accepted", "client_id": creds.get("client_id"), "project_id": project_id, "expires_in": payload.get("expires_in")} + if probe_vertex: + google_vertex = probe_vertex_llm( + payload.get("access_token"), project_id, proxy, + vertex_timeout, vertex_locations, vertex_models, vertex_max_attempts, + ) + anthropic_vertex = probe_vertex_anthropic( + payload.get("access_token"), project_id, proxy, + vertex_timeout, vertex_anthropic_locations, vertex_anthropic_models, vertex_anthropic_max_attempts, + ) if vertex_anthropic_models else {"enabled": False, "message": ""} + vertex = merge_vertex_results(google_vertex, anthropic_vertex) + result.update({ + "vertex_enabled": vertex.get("enabled"), + "vertex_location": vertex.get("location", ""), + "vertex_model": vertex.get("model", ""), + "vertex_available_models": vertex.get("available_models") or [], + "vertex_google_enabled": vertex.get("google_enabled"), + "vertex_anthropic_enabled": vertex.get("anthropic_enabled"), + "vertex_total_tokens": vertex.get("total_tokens"), + "vertex_message": vertex.get("message", ""), + }) + if vertex.get("enabled"): + result["status"] = "VERTEX" + result["message"] = "refresh token accepted; Vertex countTokens accepted" + return result + message = request_error_message(response) + lower = message.lower() + if response.status_code in (400, 401) and any(item in lower for item in ("invalid_grant", "invalid_client", "unauthorized_client")): + return {"status": "DEAD", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")} + if response.status_code in (401, 403): + return {"status": "RESTRICTED", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")} + if response.status_code == 429: + return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "client_id": creds.get("client_id")} + if response.status_code >= 500: + return {"status": "NETWORK", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")} + return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")} + + +def write_result(key, detector, result, source, finding): + write_keycheck_event(SERVICE, RESULTS_FILE, key, {"detector": detector, **result}, source, finding, detector) + extra = result.get("client_email") or result.get("client_id") or result.get("project_id") or source + message = result.get("message", "") + if result.get("status") == "VERTEX": + models = result.get("vertex_available_models") or [] + model_text = ",".join(str(item) for item in models) or f"{result.get('vertex_location', '')}/{result.get('vertex_model', '')}".strip("/") + message = f"{message}; models={model_text}" + commit_status_transaction( + CHECKED_FILE, transaction_status_files(), key, result["status"], message, extra, + ) + if result.get("status") == "VERTEX" and result.get("vertex_google_enabled"): + append_status(VERTEX_GEMINI_FILE, key, result["status"], message, extra) + if result.get("status") == "VERTEX" and result.get("vertex_anthropic_enabled"): + append_status(VERTEX_ANTHROPIC_FILE, key, result["status"], message, extra) + record_validation_result(SERVICE, key, {"detector": detector, **result}, source, finding, detector) + + +def parse_args(): + parser = argparse.ArgumentParser(description="GCP credential checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", default=PLAIN_FILE) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=25) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + parser.add_argument("--probe-vertex", action="store_true", help="After OAuth succeeds, probe Vertex AI Gemini with countTokens through the configured proxy") + parser.add_argument("--vertex-timeout", type=int, default=6, help="Seconds per Vertex countTokens request") + parser.add_argument("--vertex-max-attempts", type=int, default=6, help="Maximum location/model countTokens attempts per credential") + parser.add_argument("--vertex-locations", default=",".join(VERTEX_LOCATIONS), help="Comma-separated Vertex locations to probe") + parser.add_argument("--vertex-models", default=",".join(VERTEX_MODELS), help="Comma-separated Vertex models to probe") + parser.add_argument("--vertex-anthropic-locations", default=",".join(VERTEX_ANTHROPIC_LOCATIONS), help="Comma-separated Vertex Anthropic locations to probe") + parser.add_argument("--vertex-anthropic-models", default=",".join(VERTEX_ANTHROPIC_MODELS), help="Comma-separated Vertex Anthropic model IDs to probe with rawPredict") + parser.add_argument("--vertex-anthropic-max-attempts", type=int, default=20, help="Maximum Anthropic location/model attempts per credential") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES) + known = set(known_statuses) + retry_statuses = set() + if args.retry_network: + retry_statuses.add("NETWORK") + if args.retry_limited: + retry_statuses.add("RATE_LIMITED") + if args.retry_unknown: + retry_statuses.add("UNKNOWN") + if args.retry_valid: + retry_statuses.update({"VALID", "VERTEX"}) + vertex_locations = [item.strip() for item in str(args.vertex_locations or "").split(",") if item.strip()] + vertex_models = [item.strip() for item in str(args.vertex_models or "").split(",") if item.strip()] + vertex_anthropic_locations = [item.strip() for item in str(args.vertex_anthropic_locations or "").split(",") if item.strip()] + vertex_anthropic_models = [item.strip() for item in str(args.vertex_anthropic_models or "").split(",") if item.strip()] + + processed = 0 + skipped = 0 + for key, detector, source, finding, parsed in extract_candidates(args.input, args.plain): + if should_skip_key( + key, checked, known, args, retry_statuses, + service=SERVICE, source=source, finding=finding, detector=detector, known_statuses=known_statuses, + ): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] {detector} candidate {mask_secret(key)} from {source}", flush=True) + proxy = next(proxy_cycler) if proxy_cycler else None + if not parsed: + result = {"status": "NO_CONTEXT", "message": "credential JSON is incomplete or unavailable"} + elif detector == "GCP": + result = check_service_account( + parsed, proxy, args.timeout, args.probe_vertex, + args.vertex_timeout, vertex_locations, vertex_models, args.vertex_max_attempts, + vertex_anthropic_locations, vertex_anthropic_models, args.vertex_anthropic_max_attempts, + ) + else: + result = check_adc( + parsed, proxy, args.timeout, args.probe_vertex, + args.vertex_timeout, vertex_locations, vertex_models, args.vertex_max_attempts, + vertex_anthropic_locations, vertex_anthropic_models, args.vertex_anthropic_max_attempts, + ) + print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}", flush=True) + write_result(key, detector, result, source, finding) + known.add(key) + checked[key] = result["status"] + time.sleep(0.1) + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/gemini/geminiKeycheck.py b/app/keycheckers/gemini/geminiKeycheck.py new file mode 100644 index 0000000..39055fc --- /dev/null +++ b/app/keycheckers/gemini/geminiKeycheck.py @@ -0,0 +1,738 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import json +import os +import re +import time +from datetime import datetime, timezone +from itertools import cycle + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + acquire_file_lock, + append_checked, + default_input_file, + default_proxy_file, + env_int, + ensure_output_files as ensure_private_output_files, + iter_findings, + iter_bounded_text_lines, + keycheck_input_mode, + load_known_statuses, + private_atomic_writer, + record_cached_keycheck_occurrence, + record_validation_result, + release_file_lock, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) +from runtime_security import durable_replace, reject_reparse_components, require_private_directory, require_private_file + + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +PARENT_DIR = os.path.dirname(SCRIPT_DIR) +SERVICE = "gemini" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) + + +def here(*parts): + return os.path.join(SCRIPT_DIR, *parts) + + +def out(*parts): + return os.path.join(OUTPUT_DIR, *parts) + + +def parent(*parts): + return os.path.join(PARENT_DIR, *parts) + + +# --- Configuration --- +DEFAULT_INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +DEFAULT_PLAIN_INPUT_FILES = [out("gem.txt")] +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() + +CHECKED_FILE = out("geminiChecked.txt") +RESULTS_FILE = out("geminiResults.jsonl") + +STATUS_FILES = { + "VALID": out("geminiAlive.txt"), + "VALID_RATE_LIMITED": out("geminiAliveRateLimited.txt"), + "INVALID": out("geminiDead.txt"), + "EXPIRED": out("geminiExpired.txt"), + "LEAKED_REVOKED": out("geminiLeaked.txt"), + "API_DISABLED": out("geminiDisabled.txt"), + "RESTRICTED": out("geminiRestricted.txt"), + "RATE_LIMITED": out("geminiRateLimited.txt"), + "NETWORK_ERROR": out("geminiNetwork.txt"), + "UNKNOWN": out("geminiUnknown.txt"), +} + +GEMINI_KEY_REGEX = re.compile(r"(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})") +GEMINI_DETECTOR_NAMES = {"googleai", "googleaistudio"} +MODELS_URL = "https://generativelanguage.googleapis.com/v1beta/models" + +PROBE_MODEL_PRIORITY = [ + "gemini-3.1-pro-preview", + "gemini-3.7-flash", +] + +MODEL_PRIORITY = [ + "gemini-3", + "gemini-2.5-pro", + "gemini-2.5-flash", + "gemini-2.0-flash", + "gemini-1.5-pro", + "gemini-1.5-flash", + "imagen", + "embedding", +] + + +def now_iso(): + return datetime.now(timezone.utc).isoformat(timespec="seconds") + + +def mask_key(key): + if not key or len(key) < 12: + return key + return f"{key[:8]}...{key[-4:]}" + + +def redact_key_text(text, key): + if not isinstance(text, str): + return text + redacted = text.replace(key, "***REDACTED***") if key else text + return GEMINI_KEY_REGEX.sub("***REDACTED***", redacted) + + +def redact_result_text(result, key): + if isinstance(result, dict): + return {k: redact_result_text(v, key) for k, v in result.items()} + if isinstance(result, list): + return [redact_result_text(v, key) for v in result] + return redact_key_text(result, key) + + +def key_from_line(line): + line = line.strip() + if not line: + return None + if "\t" in line: + return line.split("\t", 1)[0].strip() + return line.split(":", 1)[0].strip() + + +def load_keys_from_file(filepath): + if keycheck_input_mode() == 'postgres': + return set() + if not os.path.exists(filepath): + return set() + keys = set() + for line in iter_bounded_text_lines(filepath): + key = key_from_line(line) + if key: + keys.add(key) + return keys + + +def load_checked_statuses(filepath=CHECKED_FILE): + statuses = {} + if keycheck_input_mode() == 'postgres': + return statuses + if not os.path.exists(filepath): + return statuses + for line in iter_bounded_text_lines(filepath): + parts = line.rstrip("\n").split("\t") + if not parts or not parts[0]: + continue + key = parts[0] + status = parts[1] if len(parts) > 1 else "UNKNOWN" + statuses[key] = status + return statuses + + +def load_all_known_keys(): + known = set(load_checked_statuses().keys()) + for path in STATUS_FILES.values(): + known.update(load_keys_from_file(path)) + return known + + +def ensure_output_files(): + if keycheck_input_mode() == 'postgres': + return + require_private_directory(OUTPUT_DIR, create=True) + + legacy_rate_limited = out("geminiLimited.txt") + rate_limited = STATUS_FILES["RATE_LIMITED"] + if os.path.exists(legacy_rate_limited) and not os.path.exists(rate_limited): + require_private_file(legacy_rate_limited) + durable_replace(legacy_rate_limited, rate_limited) + require_private_file(rate_limited) + + paths = {CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()} + ensure_private_output_files(paths) + migrate_legacy_alive_rate_limited() + + +def effective_status(result): + status = result.get("status") + probe_status = (result.get("probe") or {}).get("status") + if status == "VALID" and probe_status == "RATE_LIMITED": + return "VALID_RATE_LIMITED" + return status + + +def _gemini_status_layout(): + paths_by_status = { + status: os.path.abspath(os.fspath(path)) + for status, path in STATUS_FILES.items() + } + paths = list(dict.fromkeys(paths_by_status.values())) + directories = {os.path.normcase(os.path.dirname(path)) for path in paths} + if len(paths) != len(paths_by_status) or len(directories) != 1: + raise RuntimeError("Gemini status files must be unique files in one directory") + directory = os.path.dirname(paths[0]) + require_private_directory(directory, create=True) + return paths_by_status, paths, os.path.join(directory, "geminiStatus.lock") + + +def migrate_legacy_alive_rate_limited(): + paths_by_status, _, lock_path = _gemini_status_layout() + alive_path = paths_by_status["VALID"] + limited_path = paths_by_status["VALID_RATE_LIMITED"] + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + if not os.path.lexists(alive_path): + return + require_private_file(alive_path) + + keep = [] + moved = {} + for line in iter_bounded_text_lines(alive_path): + key = key_from_line(line) + if key and line.rstrip("\r\n").endswith(":RATE_LIMITED"): + moved.setdefault(key, line if line.endswith("\n") else f"{line}\n") + else: + keep.append(line) + if not moved: + return + + if os.path.lexists(limited_path): + require_private_file(limited_path) + existing = _normalized_status_lines(list(iter_bounded_text_lines(limited_path))) + limited = [] + published = set() + for line in existing: + key = key_from_line(line) + if key in moved: + if key in published: + continue + published.add(key) + limited.append(line) + for key, line in moved.items(): + if key not in published: + limited.append(line) + published.add(key) + + _validate_status_snapshot(limited_path, limited) + _validate_status_snapshot(alive_path, keep) + + # Make every moved key durable before publishing the source snapshot + # that removes it. An interruption can therefore only leave duplicates. + _replace_status_snapshot(limited_path, limited) + confirmed = {key: 0 for key in moved} + for line in iter_bounded_text_lines(limited_path): + key = key_from_line(line) + if key in confirmed: + confirmed[key] += 1 + if any(count != 1 for count in confirmed.values()): + raise RuntimeError("Gemini legacy rate-limited status publication was incomplete") + _replace_status_snapshot(alive_path, keep) + finally: + release_file_lock(lock, lock_path) + + +def load_proxies(proxy_file): + if not os.path.exists(proxy_file): + print(f"Info: {proxy_file} not found. Requests will go directly.") + return None + + proxies = [] + with open(proxy_file, "r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + ip, port, login, password = line.split(":") + proxy_url = f"http://{login}:{password}@{ip}:{port}" + proxies.append({"http": proxy_url, "https": proxy_url}) + except ValueError: + print(f"Warning: bad proxy format: {line}. Skipping.") + + if not proxies: + print(f"Warning: {proxy_file} is empty. Requests will go directly.") + return None + + print(f"Loaded proxies: {len(proxies)}") + return cycle(proxies) + + +def status_file_line(key, result, status): + if status in ("VALID", "VALID_RATE_LIMITED"): + models_str = ",".join(result.get("notable_models", [])) or "models-only" + probe_status = result.get("probe", {}).get("status", "not_probed") + return f"{key}:[{models_str}]:{result.get('model_class', 'unknown')}:{probe_status}\n" + message = (result.get("error", {}).get("message") or "").replace("\n", " ")[:300] + return f"{key}\t{status}\t{message}\n" + + +def _normalized_status_lines(lines): + return [line if line.endswith("\n") else f"{line}\n" for line in lines] + + +def _validate_status_snapshot(path, lines): + max_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_BYTES", 32 * 1024 * 1024)) + max_items = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_ITEMS", 100000)) + max_line_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_LINE_BYTES", 8192)) + if len(lines) > max_items: + raise RuntimeError(f"Gemini status file exceeds its item bound: {path}") + total = 0 + for index, line in enumerate(lines, 1): + encoded = line.encode("utf-8") + if len(encoded) > max_line_bytes: + raise RuntimeError(f"Gemini status line exceeds its byte bound: {path}:{index}") + total += len(encoded) + if total > max_bytes: + raise RuntimeError(f"Gemini status file exceeds its aggregate byte bound: {path}") + + +def _replace_status_snapshot(path, lines): + with private_atomic_writer(path, binary=True, suffix=".status.tmp") as handle: + for line in lines: + handle.write(line.encode("utf-8")) + + +def append_status_file(key, result): + if keycheck_input_mode() == 'postgres': + return + paths_by_status, paths, lock_path = _gemini_status_layout() + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + status = effective_status(result) + target_path = paths_by_status.get(status, paths_by_status["UNKNOWN"]) + new_line = status_file_line(key, result, status) + snapshots = {} + for path in paths: + if os.path.lexists(path): + reject_reparse_components(path) + snapshots[path] = _normalized_status_lines(list(iter_bounded_text_lines(path))) + + rewritten = { + path: [line for line in lines if key_from_line(line) != key] + for path, lines in snapshots.items() + } + rewritten[target_path].insert(0, new_line) + for path, lines in rewritten.items(): + _validate_status_snapshot(path, lines) + + # Publish the new classification before removing any old copies. A + # failure after this point can leave duplicates, but never no status. + _replace_status_snapshot(target_path, rewritten[target_path]) + for path in paths: + if path == target_path or rewritten[path] == snapshots[path]: + continue + _replace_status_snapshot(path, rewritten[path]) + finally: + release_file_lock(lock, lock_path) + + +def append_checked_file(key, result): + if keycheck_input_mode() == 'postgres': + return + append_checked(CHECKED_FILE, key, effective_status(result)) + + +def is_gemini_detector(detector): + return str(detector or "").lower() in GEMINI_DETECTOR_NAMES + + +def custom_detector_name(data): + if not isinstance(data, dict): + return "" + extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {} + name = extra.get("name") or "" + if str(data.get("DetectorName") or "").lower() == "customregex" and is_gemini_detector(name): + return name + return "" + + +def detector_name_from_finding(data): + if not isinstance(data, dict): + return "" + if is_gemini_detector(data.get("DetectorName")): + return data.get("DetectorName") + custom_name = custom_detector_name(data) + if custom_name: + return custom_name + + # Old wrapped format from earlier scanner versions. + if is_gemini_detector(data.get("detector")): + return data.get("detector") + + finding = data.get("finding") + if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")): + return finding.get("DetectorName") + custom_name = custom_detector_name(finding) + if custom_name: + return custom_name + + return "" + + +def extract_key_from_finding(data): + if is_gemini_detector(data.get("DetectorName")): + return data.get("Raw") or data.get("RawV2") + if custom_detector_name(data): + return data.get("Raw") or data.get("RawV2") + + # Old wrapped format from earlier scanner versions. + if is_gemini_detector(data.get("detector")): + return data.get("raw") or data.get("raw_v2") + + finding = data.get("finding") + if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")): + return finding.get("Raw") or finding.get("RawV2") + if custom_detector_name(finding): + return finding.get("Raw") or finding.get("RawV2") + + return None + + +def iter_candidate_keys(input_file, plain_files): + for item in iter_findings(input_file, ["GoogleAI", "GoogleAIStudio", "CustomRegex"]): + finding = item.get("finding") or {} + key = item.get("raw") or extract_key_from_finding(finding) + if key and GEMINI_KEY_REGEX.fullmatch(key) and detector_name_from_finding(finding): + yield item.get("source") or input_file, key, finding + + if keycheck_input_mode() == 'postgres': + return + for path in plain_files: + if not os.path.exists(path): + print(f"Info: plain input {path} not found. Skipping.") + continue + try: + keys = set() + for line in iter_bounded_text_lines(path): + keys.update(GEMINI_KEY_REGEX.findall(line)) + except (OSError, RuntimeError) as e: + print(f"Warning: cannot read {path}: {e}") + continue + for idx, key in enumerate(sorted(keys), 1): + yield f"{path}:plain:{idx}", key, {} + + +def parse_error_response(response): + try: + payload = response.json() + except json.JSONDecodeError: + payload = {} + + error = payload.get("error", {}) if isinstance(payload, dict) else {} + return { + "http_status": response.status_code, + "code": error.get("code", response.status_code), + "status": error.get("status", ""), + "message": error.get("message", response.text[:500]), + } + + +def classify_error(error): + http_status = int(error.get("http_status") or 0) + status = str(error.get("status") or "").lower() + message = str(error.get("message") or "").lower() + + if "reported as leaked" in message or "leaked" in message: + return "LEAKED_REVOKED" + if "api key expired" in message or "expired" in message: + return "EXPIRED" + if "api key not valid" in message or "invalid api key" in message: + return "INVALID" + if "has not been used" in message or "it is disabled" in message or "api is disabled" in message: + return "API_DISABLED" + if "requests to this api" in message and "blocked" in message: + return "RESTRICTED" + if "api key restrictions" in message or "permission_denied" in status: + return "RESTRICTED" + if http_status == 429 or "resource_exhausted" in status or "quota" in message: + return "RATE_LIMITED" + if http_status in (400, 401): + return "INVALID" + if http_status == 403: + return "RESTRICTED" + return "UNKNOWN" + + +def fetch_models(key, proxy, timeout, debug=False): + try: + response = requests.get(MODELS_URL, params={"key": key}, proxies=proxy, timeout=timeout) + except requests.RequestException as e: + return { + "status": "NETWORK_ERROR", + "error": {"message": str(e)}, + "models": [], + "model_infos": [], + } + + if debug: + print(f" DEBUG /models: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}") + + if response.status_code != 200: + error = parse_error_response(response) + return { + "status": classify_error(error), + "error": error, + "models": [], + "model_infos": [], + } + + payload = response.json() + model_infos = payload.get("models", []) + models = sorted({m.get("name", "").replace("models/", "") for m in model_infos if m.get("name")}) + return { + "status": "VALID", + "error": {}, + "models": models, + "model_infos": model_infos, + } + + +def supported_methods_by_model(model_infos): + output = {} + for model in model_infos: + name = model.get("name", "").replace("models/", "") + if not name: + continue + output[name] = sorted(model.get("supportedGenerationMethods", [])) + return output + + +def classify_models(models, methods_by_model): + notable = [] + lower_models = {m.lower(): m for m in models} + for marker in MODEL_PRIORITY: + for lower, original in lower_models.items(): + if marker in lower and original not in notable: + notable.append(original) + + generation_models = sorted([ + model for model, methods in methods_by_model.items() + if "generateContent" in methods + ]) + + if any("gemini-2.5-pro" in m.lower() for m in generation_models): + model_class = "pro_generation" + elif generation_models: + model_class = "generation" + elif models: + model_class = "models_only" + else: + model_class = "no_models" + + return notable[:20], generation_models, model_class + + +def choose_probe_model(generation_models): + available = set(generation_models) + for model in PROBE_MODEL_PRIORITY: + if model in available: + return model + return generation_models[0] if generation_models else None + + +def probe_generation(key, model, proxy, timeout, debug=False): + if not model: + return {"status": "NO_GENERATION_MODEL", "model": None} + + url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent" + headers = {"x-goog-api-key": key, "Content-Type": "application/json"} + payload = { + "contents": [{"parts": [{"text": "ping"}]}], + "generationConfig": {"maxOutputTokens": 1}, + } + + try: + response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as e: + return {"status": "NETWORK_ERROR", "model": model, "error": {"message": str(e)}} + + if debug: + print(f" DEBUG probe {model}: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}") + + if response.status_code == 200: + return {"status": "GENERATION_OK", "model": model} + + error = parse_error_response(response) + return {"status": classify_error(error), "model": model, "error": error} + + +def check_key(key, proxy, args): + result = fetch_models(key, proxy, args.timeout, args.debug) + result = redact_result_text(result, key) + result.update({ + "checked_at": now_iso(), + "key_masked": mask_key(key), + "model_count": len(result.get("models", [])), + }) + + if result["status"] != "VALID": + result["notable_models"] = [] + result["generation_models"] = [] + result["model_class"] = "none" + return result + + methods_by_model = supported_methods_by_model(result.get("model_infos", [])) + notable, generation_models, model_class = classify_models(result["models"], methods_by_model) + + result["methods_by_model"] = methods_by_model + result["notable_models"] = notable + result["generation_models"] = generation_models[:50] + result["model_class"] = model_class + result["billing_status"] = "unknown" + + if args.probe_generation: + probe_model = choose_probe_model(generation_models) + result["probe"] = redact_result_text(probe_generation(key, probe_model, proxy, args.timeout, args.debug), key) + else: + result["probe"] = {"status": "not_probed", "model": None} + + return result + + +def print_result(index, source, key, result): + print(f"\n[{index}] Candidate {mask_key(key)} from {source}") + print(f" STATUS: {result['status']}") + + if result["status"] == "VALID": + print(f" MODELS: {result.get('model_count', 0)} total; class={result.get('model_class')}") + notable = result.get("notable_models", [])[:8] + if notable: + print(f" NOTABLE: {', '.join(notable)}") + probe = result.get("probe", {}) + print(f" PROBE: {probe.get('status')} ({probe.get('model')})") + if effective_status(result) == "VALID_RATE_LIMITED": + print(f" OUT: {STATUS_FILES['VALID_RATE_LIMITED']}") + else: + error = result.get("error", {}) + message = (error.get("message") or "").replace("\n", " ")[:300] + if message: + print(f" MESSAGE: {message}") + + print(f" OUT: {STATUS_FILES.get(result['status'], STATUS_FILES['UNKNOWN'])}") + + +def parse_args(): + parser = argparse.ArgumentParser(description="Gemini / Google AI API key classifier") + parser.add_argument("--input", default=DEFAULT_INPUT_FILE) + parser.add_argument("--plain", action="append", default=None, help="Plain text file with Gemini keys. Can be repeated.") + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=20) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--probe-generation", action="store_true", help="Optionally call generateContent, preferring gemini-3.1-pro-preview when available.") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + parser.add_argument("--debug", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + plain_files = args.plain if args.plain is not None else DEFAULT_PLAIN_INPUT_FILES + ensure_output_files() + + print("--- Gemini key checker ---") + print("Default mode: /models only. Use --probe-generation for runtime/billing probe.") + print(f"Workspace: {SCRIPT_DIR}") + + proxy_cycler = load_proxies(args.proxy_file) + checked_statuses = load_checked_statuses() + known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES) + known_keys = set(known_statuses) + retry_statuses = set() + if args.retry_limited: + retry_statuses.update(("RATE_LIMITED", "VALID_RATE_LIMITED")) + if args.retry_unknown: + retry_statuses.add("UNKNOWN") + if args.retry_network: + retry_statuses.add("NETWORK_ERROR") + if args.retry_valid: + retry_statuses.update(("VALID", "VALID_RATE_LIMITED")) + print(f"Loaded known keys: {len(known_keys)}; checked records: {len(checked_statuses)}") + + seen_this_run = set() + processed = 0 + skipped = 0 + + for source, key, finding in iter_candidate_keys(args.input, plain_files): + if keycheck_input_mode() != 'postgres' and key in seen_this_run: + cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN' + detector = detector_name_from_finding(finding) or "GoogleAI" + record_cached_keycheck_occurrence(SERVICE, key, cached_status, source, finding, detector) + skipped += 1 + continue + seen_this_run.add(key) + + detector = detector_name_from_finding(finding) or "GoogleAI" + if should_skip_key( + key, checked_statuses, known_keys, args, retry_statuses, + service=SERVICE, source=source, finding=finding, detector=detector, + known_statuses=known_statuses, + ): + skipped += 1 + continue + + if args.max_keys and processed >= args.max_keys: + break + + processed += 1 + + proxy = next(proxy_cycler) if proxy_cycler else None + result = check_key(key, proxy, args) + result["source"] = source + event_result = {**result, "status": effective_status(result)} + write_keycheck_event(SERVICE, RESULTS_FILE, key, event_result, source, finding, detector) + + print_result(processed, source, key, result) + append_status_file(key, result) + append_checked_file(key, result) + record_validation_result(SERVICE, key, {**result, "status": effective_status(result)}, source, finding, detector) + + known_keys.add(key) + checked_statuses[key] = effective_status(result) + + # Small pause helps when many keys hit the same API/proxy. + time.sleep(0.1) + + print("\n--- Done ---") + print(f"Processed: {processed}") + print(f"Skipped: {skipped}") + print(f"Results: {RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/github/githubKeycheck.py b/app/keycheckers/github/githubKeycheck.py new file mode 100644 index 0000000..8f9754b --- /dev/null +++ b/app/keycheckers/github/githubKeycheck.py @@ -0,0 +1,212 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import json +import os +import re +import time + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + read_plain_keys, + recover_status_transaction, + request_error_message, + record_validation_result, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) + + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +PARENT_DIR = os.path.dirname(SCRIPT_DIR) +SERVICE = "github" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) + +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +PLAIN_FILE = os.path.join(OUTPUT_DIR, "github.txt") +CHECKED_FILE = os.path.join(OUTPUT_DIR, "githubChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "githubResults.jsonl") + +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "githubAlive.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "githubDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "githubRestricted.txt"), + "RATE_LIMITED": os.path.join(OUTPUT_DIR, "githubRateLimited.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "githubNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "githubUnknown.txt"), + "REFRESH_TOKEN": os.path.join(OUTPUT_DIR, "githubRefreshToken.txt"), +} + +GITHUB_TOKEN_RE = re.compile(r"\b(?:gh[pousr]_[A-Za-z0-9_]{20,}|github_pat_[A-Za-z0-9_]{20,})\b") + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), PLAIN_FILE]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def extract_candidates(input_file, plain_files): + seen_plain = set() + for item in iter_findings(input_file, ["Github", "GitHubOauth2"]): + text = "\n".join(str(value or "") for value in [item.get("raw"), item.get("raw_v2")]) + for match in GITHUB_TOKEN_RE.findall(text): + yield match, item["source"], item["finding"] + + for item in read_plain_keys(plain_files, GITHUB_TOKEN_RE): + key = item["key"] + if key not in seen_plain: + seen_plain.add(key) + yield key, item["source"], {} + + +def is_rate_limited(response): + remaining = response.headers.get("X-RateLimit-Remaining") + return response.status_code in (403, 429) and remaining == "0" + + +def check_token(token, proxy, timeout): + if token.startswith("ghr_"): + return { + "status": "REFRESH_TOKEN", + "message": "GitHub refresh tokens cannot be checked directly as bearer API tokens", + } + + if token.startswith("ghs_"): + url = "https://api.github.com/installation/repositories" + token_kind = "installation" + else: + url = "https://api.github.com/user" + token_kind = "user" + + headers = { + "Authorization": f"Bearer {token}", + "Accept": "application/vnd.github+json", + "X-GitHub-Api-Version": "2022-11-28", + "User-Agent": "local-keycheck-github", + } + try: + response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc), "token_kind": token_kind} + + message = request_error_message(response) + scopes = response.headers.get("X-OAuth-Scopes", "") + accepted_scopes = response.headers.get("X-Accepted-OAuth-Scopes", "") + rate_remaining = response.headers.get("X-RateLimit-Remaining", "") + rate_reset = response.headers.get("X-RateLimit-Reset", "") + + if response.status_code == 200: + payload = response.json() + extra = { + "token_kind": token_kind, + "scopes": scopes, + "accepted_scopes": accepted_scopes, + "rate_remaining": rate_remaining, + "rate_reset": rate_reset, + } + if token_kind == "installation": + extra["repo_count"] = payload.get("total_count") + return {"status": "VALID", "message": "installation token accepted", **extra} + return { + "status": "VALID", + "message": "user token accepted", + "login": payload.get("login"), + "account_type": payload.get("type"), + **extra, + } + + if response.status_code == 401: + return {"status": "DEAD", "http_status": 401, "message": message, "token_kind": token_kind} + if is_rate_limited(response): + return {"status": "RATE_LIMITED", "http_status": response.status_code, "message": message, "token_kind": token_kind, "rate_reset": rate_reset} + if response.status_code == 403: + return {"status": "RESTRICTED", "http_status": 403, "message": message, "token_kind": token_kind, "scopes": scopes} + if response.status_code in (404, 422): + return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "token_kind": token_kind} + if response.status_code >= 500: + return {"status": "NETWORK", "http_status": response.status_code, "message": message, "token_kind": token_kind} + return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "token_kind": token_kind} + + +def write_result(key, result, source, finding): + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding) + extra = result.get("login") or result.get("repo_count") or result.get("token_kind") or source + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra, + ) + record_validation_result(SERVICE, key, result, source, finding) + + +def parse_args(): + parser = argparse.ArgumentParser(description="GitHub token checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=20) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = set() + if args.retry_network: + retry_statuses.add("NETWORK") + if args.retry_limited: + retry_statuses.add("RATE_LIMITED") + if args.retry_unknown: + retry_statuses.add("UNKNOWN") + if args.retry_restricted: + retry_statuses.add("RESTRICTED") + + plain_files = args.plain or [PLAIN_FILE] + processed = 0 + skipped = 0 + for key, source, finding in extract_candidates(args.input, plain_files): + if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] GitHub candidate {mask_secret(key)} from {source}") + proxy = next(proxy_cycler) if proxy_cycler else None + result = check_token(key, proxy, args.timeout) + print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + time.sleep(0.1) + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/gitlab/gitlabKeycheck.py b/app/keycheckers/gitlab/gitlabKeycheck.py new file mode 100644 index 0000000..a0cf48c --- /dev/null +++ b/app/keycheckers/gitlab/gitlabKeycheck.py @@ -0,0 +1,178 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os +import re +import time + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + read_plain_keys, + recover_status_transaction, + request_error_message, + record_validation_result, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) + + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +PARENT_DIR = os.path.dirname(SCRIPT_DIR) +SERVICE = "gitlab" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) + +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +PLAIN_FILE = os.path.join(OUTPUT_DIR, "gitlab.txt") +CHECKED_FILE = os.path.join(OUTPUT_DIR, "gitlabChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "gitlabResults.jsonl") + +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "gitlabAlive.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "gitlabDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "gitlabRestricted.txt"), + "RATE_LIMITED": os.path.join(OUTPUT_DIR, "gitlabRateLimited.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "gitlabNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "gitlabUnknown.txt"), +} + +GITLAB_TOKEN_RE = re.compile(r"\b(?:glpat|gloas|glcbt|glimt|glrt|glft|glsoat)-[A-Za-z0-9_\-=]{20,}\b") + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), PLAIN_FILE]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def extract_candidates(input_file, plain_files): + seen_plain = set() + for item in iter_findings(input_file, ["Gitlab"]): + text = "\n".join(str(value or "") for value in [item.get("raw"), item.get("raw_v2")]) + for match in GITLAB_TOKEN_RE.findall(text): + yield match, item["source"], item["finding"] + + for item in read_plain_keys(plain_files, GITLAB_TOKEN_RE): + key = item["key"] + if key not in seen_plain: + seen_plain.add(key) + yield key, item["source"], {} + + +def check_token(token, base_url, proxy, timeout): + base_url = base_url.rstrip("/") + url = f"{base_url}/api/v4/user" + headers = {"Authorization": f"Bearer {token}", "User-Agent": "local-keycheck-gitlab"} + try: + response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc), "base_url": base_url} + + message = request_error_message(response) + retry_after = response.headers.get("Retry-After", "") + rate_remaining = response.headers.get("RateLimit-Remaining") or response.headers.get("X-RateLimit-Remaining") or "" + + if response.status_code == 200: + payload = response.json() + return { + "status": "VALID", + "message": "token accepted", + "username": payload.get("username"), + "name": payload.get("name"), + "user_id": payload.get("id"), + "base_url": base_url, + "rate_remaining": rate_remaining, + } + if response.status_code == 401: + return {"status": "DEAD", "http_status": 401, "message": message, "base_url": base_url} + if response.status_code == 403: + # TruffleHog treats 403 as a live token with insufficient scope or blocked account. + return {"status": "RESTRICTED", "http_status": 403, "message": message, "base_url": base_url} + if response.status_code == 429: + return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "base_url": base_url, "retry_after": retry_after} + if response.status_code >= 500: + return {"status": "NETWORK", "http_status": response.status_code, "message": message, "base_url": base_url} + return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "base_url": base_url} + + +def write_result(key, result, source, finding): + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding) + extra = result.get("username") or result.get("user_id") or source + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra, + ) + record_validation_result(SERVICE, key, result, source, finding) + + +def parse_args(): + parser = argparse.ArgumentParser(description="GitLab token checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--base-url", default="https://gitlab.com") + parser.add_argument("--timeout", type=int, default=20) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = set() + if args.retry_network: + retry_statuses.add("NETWORK") + if args.retry_limited: + retry_statuses.add("RATE_LIMITED") + if args.retry_unknown: + retry_statuses.add("UNKNOWN") + if args.retry_restricted: + retry_statuses.add("RESTRICTED") + + plain_files = args.plain or [PLAIN_FILE] + processed = 0 + skipped = 0 + for key, source, finding in extract_candidates(args.input, plain_files): + if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] GitLab candidate {mask_secret(key)} from {source}") + proxy = next(proxy_cycler) if proxy_cycler else None + result = check_token(key, args.base_url, proxy, args.timeout) + print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + time.sleep(0.1) + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/groq/groqKeycheck.py b/app/keycheckers/groq/groqKeycheck.py new file mode 100644 index 0000000..ab90d67 --- /dev/null +++ b/app/keycheckers/groq/groqKeycheck.py @@ -0,0 +1,275 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os +import re + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, + classify_common_http_status, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + read_plain_keys, + record_validation_result, + recover_status_transaction, + request_error_message, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) + + +SERVICE = "groq" +DETECTOR = "Groq" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() + +CHECKED_FILE = os.path.join(OUTPUT_DIR, "groqChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "groqResults.jsonl") +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "groqAlive.txt"), + "NO_BALANCE": os.path.join(OUTPUT_DIR, "groqNoBalance.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "groqDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "groqRestricted.txt"), + "LIMITED": os.path.join(OUTPUT_DIR, "groqLimited.txt"), + "NO_CONTEXT": os.path.join(OUTPUT_DIR, "groqNoContext.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "groqNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "groqUnknown.txt"), +} + +GROQ_KEY_REGEX = re.compile(r"\bgsk_[A-Za-z0-9_-]{20,}\b") +MODELS_URL = "https://api.groq.com/openai/v1/models" +CHAT_URL = "https://api.groq.com/openai/v1/chat/completions" +CHAT_MODEL_PRIORITY = ( + "llama-3.1-8b-instant", + "llama-3.3-70b-versatile", + "llama3-8b-8192", + "llama3-70b-8192", + "mixtral-8x7b-32768", + "gemma2-9b-it", +) +NO_BALANCE_MARKERS = ( + "quota", + "billing", + "balance", + "credit", + "payment", + "insufficient", + "depleted", +) + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def extract_candidates(input_file, plain_files): + seen_plain = set() + for item in iter_findings(input_file, [DETECTOR]): + key = item["raw"] + if key and GROQ_KEY_REGEX.fullmatch(key): + yield key, item["source"], item["finding"] + + for item in read_plain_keys(plain_files, GROQ_KEY_REGEX): + key = item["key"] + if key not in seen_plain: + seen_plain.add(key) + yield key, item["source"], {} + + +def classify_groq_response(response): + message = request_error_message(response).lower() + if response.status_code == 401: + return "DEAD" + if response.status_code == 403: + return "RESTRICTED" + if response.status_code == 429: + if any(marker in message for marker in NO_BALANCE_MARKERS): + return "NO_BALANCE" + return "LIMITED" + return classify_common_http_status(response.status_code) + + +def notable_models(payload): + models = payload.get("data", []) if isinstance(payload, dict) else [] + ids = [] + for item in models: + if isinstance(item, dict) and item.get("id"): + ids.append(str(item.get("id"))) + priority = [] + for marker in ("llama", "mixtral", "gemma", "whisper"): + for model in ids: + if marker in model.lower() and model not in priority: + priority.append(model) + return priority[:20], len(ids), ids + + +def choose_chat_model(model_ids): + model_ids = [str(model or "") for model in model_ids if model] + by_lower = {model.lower(): model for model in model_ids} + for model in CHAT_MODEL_PRIORITY: + if model.lower() in by_lower: + return by_lower[model.lower()] + for marker in ("llama", "mixtral", "gemma"): + for model in model_ids: + lowered = model.lower() + if marker in lowered and "whisper" not in lowered and "guard" not in lowered: + return model + return "" + + +def probe_chat_completion(key, model, proxy, timeout, debug=False): + if not model: + return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""} + headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"} + payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1} + try: + response = requests.post(CHAT_URL, headers=headers, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc), "model": model} + + if debug: + print(f" DEBUG chat ping {model}: HTTP {response.status_code}: {response.text[:500].replace(key, '***REDACTED***')}") + + if response.status_code == 200: + return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model} + return { + "status": classify_groq_response(response), + "http_status": response.status_code, + "message": request_error_message(response).replace(key, "***REDACTED***"), + "model": model, + } + + +def check_key(key, proxy, timeout, debug=False): + headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"} + try: + response = requests.get(MODELS_URL, headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc)} + + if debug: + print(f" DEBUG /models: HTTP {response.status_code}: {response.text[:500].replace(key, '***REDACTED***')}") + + if response.status_code == 200: + try: + payload = response.json() + except ValueError: + payload = {} + models, model_count, model_ids = notable_models(payload) + chat_model = choose_chat_model(model_ids) + probe = probe_chat_completion(key, chat_model, proxy, timeout, debug) + if probe.get("status") != "GENERATION_OK": + return { + "status": probe.get("status") or "UNKNOWN", + "message": probe.get("message", ""), + "model_count": model_count, + "models": models, + "llm_probe_status": probe.get("status"), + "llm_probe_model": probe.get("model", chat_model), + "llm_probe_http_status": probe.get("http_status"), + } + return { + "status": "VALID", + "message": f"chat ping ok; model={chat_model}; models={model_count}", + "model_count": model_count, + "models": models, + "llm_probe_status": probe.get("status"), + "llm_probe_model": chat_model, + } + + return { + "status": classify_groq_response(response), + "http_status": response.status_code, + "message": request_error_message(response).replace(key, "***REDACTED***"), + } + + +def write_result(key, result, source, finding): + status = result.get("status") or "UNKNOWN" + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR) + extra = ",".join(result.get("models") or [])[:500] if status == "VALID" else source + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra, + ) + record_validation_result(SERVICE, key, result, source, finding, DETECTOR) + + +def parse_args(): + parser = argparse.ArgumentParser(description="Groq key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=15) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--retry-no-balance", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + parser.add_argument("--debug", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = set() + if args.retry_network: + retry_statuses.add("NETWORK") + if args.retry_limited: + retry_statuses.add("LIMITED") + if args.retry_unknown: + retry_statuses.add("UNKNOWN") + if args.retry_restricted: + retry_statuses.add("RESTRICTED") + if args.retry_no_balance: + retry_statuses.add("NO_BALANCE") + if args.retry_valid: + retry_statuses.add("VALID") + + print("--- Groq key checker ---") + processed = 0 + skipped = 0 + for key, source, finding in extract_candidates(args.input, args.plain): + if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] Groq candidate {mask_secret(key)} from {source}") + proxy = next(proxy_cycler) if proxy_cycler else None + result = check_key(key, proxy, args.timeout, args.debug) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/huggingface/huggingfaceKeycheck.py b/app/keycheckers/huggingface/huggingfaceKeycheck.py new file mode 100644 index 0000000..cc115f9 --- /dev/null +++ b/app/keycheckers/huggingface/huggingfaceKeycheck.py @@ -0,0 +1,142 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os +import re + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, classify_common_http_status, commit_status_transaction, + default_input_file, default_proxy_file, ensure_output_files, iter_findings, + keycheck_input_mode, + load_checked_statuses, load_known_keys, load_proxies, mask_secret, + read_plain_keys, record_validation_result, recover_status_transaction, + request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event, +) + +SERVICE = "huggingface" +DETECTOR_NAMES = ["HuggingFace", "Huggingface"] +DETECTOR = "HuggingFace" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +CHECKED_FILE = os.path.join(OUTPUT_DIR, "huggingfaceChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "huggingfaceResults.jsonl") +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "huggingfaceAlive.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "huggingfaceDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "huggingfaceRestricted.txt"), + "LIMITED": os.path.join(OUTPUT_DIR, "huggingfaceLimited.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "huggingfaceNetwork.txt"), + "NO_CONTEXT": os.path.join(OUTPUT_DIR, "huggingfaceNoContext.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "huggingfaceUnknown.txt"), +} +KEY_REGEX = re.compile(r"\bhf_[A-Za-z0-9]{20,}\b") +WHOAMI_URL = "https://huggingface.co/api/whoami-v2" + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def iter_candidate_decisions(input_file, plain_files): + seen_plain = set() + for item in iter_findings(input_file, DETECTOR_NAMES): + key = item.get("credential_secret_text") or item["raw"] + if key: + yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key)) + for item in read_plain_keys(plain_files, KEY_REGEX): + key = item["key"] + if key not in seen_plain: + seen_plain.add(key) + yield key, item["source"], {}, True + + +def extract_candidates(input_file, plain_files): + for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files): + if valid_format: + yield key, source, finding + + +def check_key(key, proxy, timeout): + try: + response = requests.get(WHOAMI_URL, headers={"Authorization": f"Bearer {key}"}, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc)} + if response.status_code == 200: + data = response.json() if response.text else {} + return {"status": "VALID", "message": "whoami accepted", "username": data.get("name") or data.get("fullname") or ""} + status = "RESTRICTED" if response.status_code == 403 else classify_common_http_status(response.status_code) + return {"status": status, "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***")} + + +def write_result(key, result, source, finding): + status = result.get("status") or "UNKNOWN" + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR) + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), source, + ) + record_validation_result(SERVICE, key, result, source, finding, DETECTOR) + + +def parse_args(): + parser = argparse.ArgumentParser(description="HuggingFace key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=15) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = set() + if args.retry_network: retry_statuses.add("NETWORK") + if args.retry_limited: retry_statuses.add("LIMITED") + if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"}) + if args.retry_restricted: retry_statuses.add("RESTRICTED") + processed = skipped = 0 + print("--- HuggingFace key checker ---") + postgres_mode = keycheck_input_mode() == "postgres" + for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain): + if not valid_format and not postgres_mode: + skipped += 1 + continue + if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] HuggingFace candidate {mask_secret(key)} from {source}") + result = ( + check_key(key, next(proxy_cycler) if proxy_cycler else None, args.timeout) + if valid_format else + {"status": "NO_CONTEXT", "message": "candidate does not match canonical Hugging Face token format"} + ) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/keycheck_common.py b/app/keycheckers/keycheck_common.py new file mode 100644 index 0000000..763e130 --- /dev/null +++ b/app/keycheckers/keycheck_common.py @@ -0,0 +1,2465 @@ +import hashlib +import json +import os +import re +import shutil +import sqlite3 +import sys +import time +import uuid +from collections import OrderedDict +from contextlib import contextmanager +from datetime import datetime, timezone +from itertools import cycle + +import requests + +PROJECT_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +if PROJECT_DIR not in sys.path: + sys.path.append(PROJECT_DIR) + +from paths import apply_path_config, default_project_paths +from scanner_db import ScannerDB, extract_raw_secret, json_dumps, safe_json_loads, sha256_text +from lifecycle_authority import LifecycleAuthorityError, require_active_supervisor_child +from runtime_security import ( + canonical_path, + durable_replace, + durable_unlink, + PrivateFileLock, + harden_private_file, + read_private_json, + reject_reparse_components, + require_private_directory, + require_private_file, +) + + +STATUS_TRANSACTION_JOURNAL_FILENAME = ".status-transaction.pending.json" +STATUS_TRANSACTION_LOCK_FILENAME = ".status-transaction.lock" +STATUS_TRANSACTION_MAX_KEY_BYTES = 8192 +STATUS_TRANSACTION_MAX_LINE_BYTES = 8192 +STATUS_TRANSACTION_MAX_JOURNAL_BYTES = 64 * 1024 +GCP_STATUS_MAX_LINE_BYTES = 256 * 1024 +GCP_STATUS_MAX_JOURNAL_BYTES = 1024 * 1024 +GENERIC_SK_PROVIDERS = frozenset(("qwen", "deepseek", "kimi", "zai")) +QWEN_DEEPSEEK_PROVIDERS = frozenset(("qwen", "deepseek")) +AMBIGUOUS_QWEN_DEEPSEEK_HINT = "ambiguous_qwen_deepseek" +AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk" +EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE = "explicit_assignment" +QWEN_EXPLICIT_ROUTING_DETECTORS = frozenset(("qwendashscope", "qwen_dashscope")) +DEEPSEEK_EXPLICIT_ROUTING_DETECTORS = frozenset(("deepseekapikey", "deepseek_api_key")) +KIMI_EXPLICIT_ROUTING_DETECTORS = frozenset(("kimimoonshot", "moonshotai")) +ZAI_EXPLICIT_ROUTING_DETECTORS = frozenset(("zaiglm",)) +GENERIC_SK_NON_EXPLICIT_ROUTING_DETECTORS = frozenset(("qwen", "dashscope", "deepseek")) +PROVIDER_ROUTING_EVIDENCE_CACHE = OrderedDict() +PROVIDER_ROUTING_EVIDENCE_CACHE_AUTHORITY = None +PROVIDER_ROUTING_EVIDENCE_DB_FAILED = False +REVIEWED_CORRUPTION_MAX_ROWS = 100000 +KEYCHECK_CAPACITY_BLOCKED_EXIT = 76 +REVIEWED_CORRUPTION_LEDGER_MAX_BYTES = 512 * 1024 * 1024 +REVIEWED_CORRUPTION_CLASSIFICATIONS = frozenset(( + 'invalid_json', 'legacy_numeric_prefix_corrupt_json', 'invalid_utf8', + 'utf8_bom_prefix', 'unterminated_record', 'invalid_error_projection', + 'oversized_record', +)) +_POSTGRES_DB = None +_ACTIVE_DB_CANDIDATE = None + + +def keycheck_input_mode(): + default = 'postgres' if os.getenv('SCANNER_SUPERVISED') == '1' else 'jsonl' + return str(os.getenv('KEYCHECK_INPUT_MODE') or default).strip().lower() + + +def _postgres_candidate_db(): + global _POSTGRES_DB + if _POSTGRES_DB is not None and _POSTGRES_DB.enabled: + return _POSTGRES_DB + db_url = str(os.getenv('KEYCHECK_DB_URL') or os.getenv('SCANNER_DB_URL') or '').strip() + if not db_url: + raise RuntimeError('PostgreSQL keycheck input mode requires KEYCHECK_DB_URL') + _POSTGRES_DB = ScannerDB(db_url=db_url, initialize=False) + if not _POSTGRES_DB.enabled: + raise RuntimeError('PostgreSQL keycheck candidate database is unavailable') + _POSTGRES_DB.set_application_name(f'truf-keycheck-provider:{keycheck_service_name() or "unknown"}') + _POSTGRES_DB.require_runtime_safety_schema() + _POSTGRES_DB.require_final_cutover() + return _POSTGRES_DB + + +def now_iso(): + return datetime.now(timezone.utc).isoformat(timespec="seconds") + + +def env_int(name, default): + try: + return int(os.getenv(name, default)) + except (TypeError, ValueError): + return default + + +def require_provider_authority(service): + """Authenticate a leaf checker before it parses arguments or opens findings.""" + metadata = require_active_supervisor_child( + child_kind='keycheck-provider', + require_dsn=True, + handshake_timeout_retries=1, + ) + expected_service = str(service or '').strip().lower() + if not expected_service or str(os.getenv('KEYCHECK_SERVICE') or '').strip().lower() != expected_service: + raise LifecycleAuthorityError('keycheck provider capability does not match this service') + canonical_dsn = os.getenv('TRUF_MANAGED_POSTGRES_DSN') or '' + if os.getenv('KEYCHECK_DB_URL') != canonical_dsn: + raise LifecycleAuthorityError('keycheck provider database URL is not supervisor-managed') + if keycheck_input_mode() == 'postgres' and any( + str(value).lower() in ('--input', '--plain') + or str(value).lower().startswith(('--input=', '--plain=')) + for value in sys.argv[1:] + ): + raise LifecycleAuthorityError( + 'PostgreSQL keycheck providers forbid compatibility input/plain arguments' + ) + try: + import yaml + + with open(metadata['config_path'], 'r', encoding='utf-8') as handle: + config = apply_path_config(yaml.safe_load(handle) or {}, metadata['config_path']) + expected_output = os.path.join((config.get('global') or {})['keycheck_dir'], expected_service) + except Exception as exc: + raise LifecycleAuthorityError('keycheck provider could not validate its canonical output path') from exc + output_dir = os.getenv('KEYCHECK_OUTPUT_DIR') or '' + state_dir = os.getenv('KEYCHECK_STATE_DIR') or '' + if canonical_path(output_dir) != canonical_path(expected_output) or canonical_path(state_dir) != canonical_path(expected_output): + raise LifecycleAuthorityError('keycheck provider output capability is outside its canonical service directory') + return metadata + + +def load_keycheck_layout(config_path=None): + if config_path: + try: + import yaml + with open(config_path, "r", encoding="utf-8") as f: + config = apply_path_config(yaml.safe_load(f) or {}, config_path) + return config.get("global") or {} + except Exception: + pass + return default_project_paths() + + +def service_output_dir(service, config_path=None, output_dir=None): + if output_dir: + return output_dir + layout = load_keycheck_layout(config_path) + return os.path.join(layout["keycheck_dir"], service) + + +def default_input_file(config_path=None): + layout = load_keycheck_layout(config_path) + return os.path.join(layout["results_dir"], "found_secrets.jsonl") + + +def default_proxy_file(config_path=None): + layout = load_keycheck_layout(config_path) + return layout["proxy_file"] + + +def truthy_env(name, default=False): + value = os.getenv(name) + if value is None: + return default + return str(value).strip().lower() in ("1", "true", "yes", "on") + + +def keycheck_service_name(): + service = os.getenv("KEYCHECK_SERVICE") + if service: + return service.strip().lower() + output_dir = os.getenv("KEYCHECK_OUTPUT_DIR") + if output_dir: + return os.path.basename(os.path.normpath(output_dir)).lower() + return "" + + +def status_projection_line_max_bytes(): + if os.getenv("KEYCHECK_INPUT_LIST_MAX_LINE_BYTES") is not None: + return max(1, env_int("KEYCHECK_INPUT_LIST_MAX_LINE_BYTES", STATUS_TRANSACTION_MAX_LINE_BYTES)) + if keycheck_service_name() == "gcp": + return GCP_STATUS_MAX_LINE_BYTES + return STATUS_TRANSACTION_MAX_LINE_BYTES + + +def status_transaction_journal_max_bytes(): + if keycheck_service_name() == "gcp": + return GCP_STATUS_MAX_JOURNAL_BYTES + return STATUS_TRANSACTION_MAX_JOURNAL_BYTES + + +def keycheck_state_path(service=None): + service = service or keycheck_service_name() or "default" + state_dir = os.getenv("KEYCHECK_STATE_DIR") + if not state_dir: + output_dir = os.getenv("KEYCHECK_OUTPUT_DIR") + state_dir = output_dir or service_output_dir(service) + require_private_directory(state_dir, create=True) + return os.path.join(state_dir, "input_state.json") + + +def read_json_file(path, default=None): + try: + with open(path, "r", encoding="utf-8") as f: + return json.load(f) + except (OSError, ValueError): + return default + + +def write_json_file(path, data): + with private_atomic_writer(path) as f: + json.dump(data, f, ensure_ascii=False, indent=2, sort_keys=True) + + +def input_file_signature(path, handle=None): + details = os.fstat(handle.fileno()) if handle is not None else os.stat(path) + return { + "path": os.path.abspath(path), + "device": int(getattr(details, "st_dev", 0) or 0), + "size": int(details.st_size), + "mtime_ns": int(getattr(details, "st_mtime_ns", int(details.st_mtime * 1_000_000_000))), + "inode": int(getattr(details, "st_ino", 0) or 0), + } + + +def jsonl_manifest_path(path): + base, _ = os.path.splitext(path) + return f"{base}.manifest.json" + + +def load_jsonl_manifest(path): + try: + if os.path.getsize(jsonl_manifest_path(path)) > 1024 * 1024: + raise RuntimeError(f"keycheck JSONL manifest exceeds bounded size: {jsonl_manifest_path(path)}") + with open(jsonl_manifest_path(path), "r", encoding="utf-8") as f: + data = json.load(f) + return data if isinstance(data, dict) else {} + except FileNotFoundError: + return {} + + +def write_jsonl_manifest(path, manifest): + manifest_path = jsonl_manifest_path(path) + with private_atomic_writer(manifest_path) as f: + json.dump(manifest, f, ensure_ascii=False, indent=2, sort_keys=True) + + +@contextmanager +def private_atomic_writer(path, binary=False, suffix=".tmp"): + """Publish a same-directory file whose ACL is exact-private before payload writes.""" + path = os.path.abspath(path) + require_private_directory(os.path.dirname(path), create=True) + if os.path.lexists(path): + require_private_file(path) + temporary = f"{path}.{os.getpid()}.{uuid.uuid4().hex}{suffix}" + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL + if hasattr(os, "O_BINARY"): + flags |= os.O_BINARY + descriptor = os.open(temporary, flags, 0o600) + published = False + try: + harden_private_file(temporary) + require_private_file(temporary) + mode = "wb" if binary else "w" + options = {} if binary else {"encoding": "utf-8", "newline": ""} + with os.fdopen(descriptor, mode, **options) as handle: + descriptor = None + yield handle + handle.flush() + os.fsync(handle.fileno()) + require_private_file(temporary) + durable_replace(temporary, path) + published = True + require_private_file(path) + finally: + if descriptor is not None: + os.close(descriptor) + if not published and os.path.exists(temporary): + os.remove(temporary) + + +@contextmanager +def private_append_writer(path, binary=False): + """Open an exact-private output for append, rejecting unsafe existing files.""" + path = os.path.abspath(path) + ensure_output_files([path]) + require_private_file(path) + mode = "ab" if binary else "a" + options = {} if binary else {"encoding": "utf-8", "newline": ""} + with open(path, mode, **options) as handle: + yield handle + handle.flush() + os.fsync(handle.fileno()) + require_private_file(path) + + +def acquire_file_lock(lock_path, stale_sec=300, timeout_sec=30): + require_private_directory(os.path.dirname(os.path.abspath(lock_path)), create=True) + reject_reparse_components(os.path.dirname(os.path.abspath(lock_path))) + deadline = time.monotonic() + max(1, timeout_sec) + while True: + lock = PrivateFileLock(lock_path) + try: + return lock.acquire() + except BlockingIOError: + if time.monotonic() >= deadline: + raise TimeoutError(f"timed out acquiring lock {lock_path}") + time.sleep(0.05) + + +def release_file_lock(lock, lock_path): + try: + lock.release() + except (AttributeError, OSError): + return + + +def next_jsonl_segment_path(path, manifest): + base, ext = os.path.splitext(path) + seq = int(manifest.get("next_sequence") or 1) + while True: + segment = f"{base}.{seq:06d}{ext or '.jsonl'}" + if not os.path.exists(segment): + return segment, seq + seq += 1 + + +def physical_jsonl_segments(path): + root = os.path.dirname(os.path.abspath(path)) + base, extension = os.path.splitext(os.path.basename(path)) + pattern = re.compile(rf"^{re.escape(base)}\.(\d{{6}}){re.escape(extension)}$") + output = [] + inspect_limit = max(2, env_int("KEYCHECK_RESULTS_MAX_SEGMENTS", 16) + 1) + try: + with os.scandir(root) as entries: + for entry in entries: + match = pattern.fullmatch(entry.name) + if not match: + continue + if entry.is_symlink() or not entry.is_file(follow_symlinks=False): + raise RuntimeError(f"unsafe keycheck JSONL segment: {entry.path}") + output.append((int(match.group(1)), os.path.abspath(entry.path))) + if len(output) > inspect_limit: + raise RuntimeError(f"keycheck JSONL segment count exceeds its bound: {path}") + except FileNotFoundError: + return [] + return sorted(output) + + +def _keycheck_files_share_prefix(segment_path, current_path, size): + if size <= 0 or not os.path.isfile(current_path) or os.path.getsize(current_path) < size: + return False + remaining = size + with open(segment_path, "rb") as segment, open(current_path, "rb") as current: + while remaining: + amount = min(1024 * 1024, remaining) + left = segment.read(amount) + right = current.read(amount) + if not left or left != right: + return False + remaining -= len(left) + return remaining == 0 + + +def _publish_empty_keycheck_generation(path): + with private_atomic_writer(path, binary=True, suffix=".empty.tmp"): + pass + + +def _keycheck_remove_prefix(path, size): + current_size = os.path.getsize(path) + if current_size == size: + _publish_empty_keycheck_generation(path) + return + require_private_file(path) + with open(path, "rb") as source, private_atomic_writer(path, binary=True, suffix=".prefix.tmp") as destination: + source.seek(size) + shutil.copyfileobj(source, destination, 1024 * 1024) + + +def _keycheck_file_generation(path): + details = os.stat(path, follow_symlinks=False) + return { + "device": int(getattr(details, "st_dev", 0) or 0), + "size": int(details.st_size), + "mtime_ns": int(getattr(details, "st_mtime_ns", int(details.st_mtime * 1_000_000_000))), + "inode": int(getattr(details, "st_ino", 0) or 0), + } + + +def _keycheck_manifest_skip_valid(path, manifest, skip): + signature = manifest.get("current_skip_signature") if isinstance(manifest, dict) else None + if not isinstance(signature, dict) or not os.path.isfile(path): + return False + try: + current = _keycheck_file_generation(path) + return int(skip) <= current["size"] and all( + int(current[key]) == int(signature.get(key, -1)) + for key in ("device", "size", "mtime_ns", "inode") + ) + except (OSError, TypeError, ValueError): + return False + + +def repair_keycheck_jsonl_tail(path): + if not os.path.exists(path) or os.path.getsize(path) == 0: + return 0 + require_private_file(path) + scan_limit = max(1, env_int("KEYCHECK_JSONL_TAIL_SCAN_MAX_BYTES", 8 * 1024 * 1024)) + quarantine_limit = max(1, env_int("KEYCHECK_JSONL_TORN_QUARANTINE_MAX_BYTES", 64 * 1024)) + size = os.path.getsize(path) + with open(path, "r+b") as handle: + handle.seek(-1, os.SEEK_END) + if handle.read(1) == b"\n": + return 0 + start = max(0, size - scan_limit) + handle.seek(start) + tail = handle.read(size - start) + newline = tail.rfind(b"\n") + if newline < 0 and start: + raise RuntimeError(f"keycheck JSONL tail exceeds bounded repair window: {path}") + truncate_at = start + newline + 1 if newline >= 0 else 0 + torn = tail[newline + 1:] if newline >= 0 else tail + quarantine = f"{path}.torn-tail.bin" + with private_atomic_writer(quarantine, binary=True) as output: + output.write(torn[:quarantine_limit]) + handle.truncate(truncate_at) + handle.flush() + os.fsync(handle.fileno()) + return size - truncate_at + + +def reconcile_keycheck_jsonl_segments(path): + manifest = load_jsonl_manifest(path) + physical = physical_jsonl_segments(path) + listed = { + str(item.get("name") or os.path.basename(str(item.get("path") or ""))): item + for item in (manifest.get("segments") or []) if isinstance(item, dict) + } + segments = [] + missing = [] + for sequence, segment_path in physical: + item = listed.get(os.path.basename(segment_path)) + if item is None: + item = { + "name": os.path.basename(segment_path), "path": segment_path, + "bytes": os.path.getsize(segment_path), "closed_at": now_iso(), + "sequence": sequence, + } + missing.append(segment_path) + else: + item = dict(item, name=os.path.basename(segment_path), path=segment_path, sequence=sequence) + segments.append(item) + skip = max(0, int(manifest.get("current_skip_bytes") or 0)) + duplicate = None + for segment_path in list(reversed(missing)) + ([physical[-1][1]] if skip and physical else []): + size = os.path.getsize(segment_path) + if _keycheck_files_share_prefix(segment_path, path, size): + duplicate = (segment_path, size) + break + effective_skip = duplicate[1] if duplicate else 0 + if missing or segments != (manifest.get("segments") or []) or skip or manifest.get("current_skip_signature"): + manifest.update({ + "current": os.path.basename(path), "current_path": os.path.abspath(path), + "next_sequence": max([item[0] for item in physical] or [0]) + 1, + "segments": segments, + "current_skip_bytes": effective_skip, + "current_skip_signature": _keycheck_file_generation(path) if effective_skip and os.path.isfile(path) else None, + "updated_at": now_iso(), + }) + write_jsonl_manifest(path, manifest) + if duplicate: + _keycheck_remove_prefix(path, duplicate[1]) + manifest["current_skip_bytes"] = 0 + manifest["current_skip_signature"] = None + manifest["updated_at"] = now_iso() + write_jsonl_manifest(path, manifest) + return manifest + + +def rotate_jsonl_if_needed(path, max_bytes): + if not max_bytes or max_bytes <= 0 or not os.path.lexists(path): + return + reject_reparse_components(path) + repair_keycheck_jsonl_tail(path) + size = os.stat(path, follow_symlinks=False).st_size + if size < max_bytes: + return + manifest = reconcile_keycheck_jsonl_segments(path) + max_segments = max(1, env_int("KEYCHECK_RESULTS_MAX_SEGMENTS", 16)) + if len(physical_jsonl_segments(path)) >= max_segments: + raise RuntimeError( + f"keycheck JSONL segment bound reached for {path}; ingest and retire segments offline" + ) + segment_path, seq = next_jsonl_segment_path(path, manifest) + require_private_file(path) + with open(path, "rb") as source, private_atomic_writer(segment_path, binary=True) as destination: + shutil.copyfileobj(source, destination, 1024 * 1024) + segments = manifest.get("segments") if isinstance(manifest.get("segments"), list) else [] + segments.append({ + "name": os.path.basename(segment_path), + "path": segment_path, + "bytes": int(size), + "closed_at": now_iso(), + "sequence": seq, + }) + manifest.update({ + "current": os.path.basename(path), + "current_path": path, + "next_sequence": seq + 1, + "max_bytes": int(max_bytes), + "segments": segments, + "current_skip_bytes": int(size), + "current_skip_signature": _keycheck_file_generation(path), + "updated_at": now_iso(), + }) + write_jsonl_manifest(path, manifest) + _publish_empty_keycheck_generation(path) + manifest["current_skip_bytes"] = 0 + manifest["current_skip_signature"] = None + manifest["updated_at"] = now_iso() + write_jsonl_manifest(path, manifest) + + +def should_rotate_jsonl(path): + return os.path.basename(str(path or "")).lower().endswith("results.jsonl") + + +def tail_fallback_bytes(): + tail_mb = os.getenv("KEYCHECK_INPUT_TAIL_MB") or os.getenv("KEYCHECK_TAIL_MB") + try: + return int(float(tail_mb) * 1024 * 1024) if tail_mb else int(os.getenv("KEYCHECK_INPUT_TAIL_BYTES", "0") or 0) + except (TypeError, ValueError): + return 0 + + +def keycheck_input_max_line_bytes(): + return max(1024, env_int("KEYCHECK_INPUT_MAX_LINE_BYTES", 16 * 1024 * 1024)) + + +def keycheck_reconciliation_ledger_path(input_file): + base, _ = os.path.splitext(os.path.abspath(input_file)) + return base + '.publication-ledger.sqlite3' + + +def load_reviewed_keycheck_corruptions(input_file): + ledger_path = keycheck_reconciliation_ledger_path(input_file) + if not os.path.lexists(ledger_path): + return {} + require_private_file(ledger_path) + max_bytes = max( + 1024 * 1024, + env_int('KEYCHECK_RECONCILIATION_LEDGER_MAX_BYTES', REVIEWED_CORRUPTION_LEDGER_MAX_BYTES), + ) + if os.path.getsize(ledger_path) > max_bytes: + raise RuntimeError('keycheck reconciliation ledger exceeds its byte bound') + uri = 'file:' + ledger_path.replace('\\', '/') + '?mode=ro' + connection = sqlite3.connect(uri, uri=True, timeout=5) + connection.row_factory = sqlite3.Row + try: + connection.execute('PRAGMA query_only=ON') + connection.execute('PRAGMA busy_timeout=5000') + columns = { + row['name'] for row in connection.execute('PRAGMA table_info(reconciliation_issue)').fetchall() + } + required = { + 'identity_key', 'file_name', 'file_device', 'file_inode', 'file_size', + 'file_mtime_ns', 'byte_offset', 'byte_length', 'record_sha256', + 'classification', 'status', 'resolved_at', + } + if not required.issubset(columns): + raise RuntimeError('keycheck reconciliation ledger schema is incomplete') + marker = connection.execute( + 'SELECT value FROM publication_meta WHERE key = ?', ('bootstrapped:finding_uid',), + ).fetchone() + if not marker or str(marker['value'] or '') != '1': + raise RuntimeError('keycheck reconciliation ledger is not complete') + max_rows = min( + REVIEWED_CORRUPTION_MAX_ROWS, + max(1, env_int('KEYCHECK_REVIEWED_CORRUPTION_MAX_ROWS', REVIEWED_CORRUPTION_MAX_ROWS)), + ) + rows = connection.execute( + '''SELECT file_name, file_device, file_inode, file_size, file_mtime_ns, + byte_offset, byte_length, record_sha256, classification + FROM reconciliation_issue + WHERE identity_key = 'finding_uid' AND status = 'resolved' AND resolved_at IS NOT NULL + ORDER BY file_name, byte_offset LIMIT ?''', + (max_rows + 1,), + ).fetchall() + finally: + connection.close() + if len(rows) > max_rows: + raise RuntimeError('reviewed keycheck corruption count exceeds its bound') + reviewed = {} + for row in rows: + file_name = str(row['file_name'] or '') + offset = int(row['byte_offset']) + length = int(row['byte_length']) + digest = str(row['record_sha256'] or '').lower() + classification = str(row['classification'] or '') + if ( + not file_name or os.path.basename(file_name) != file_name + or offset < 0 or length <= 0 + or not re.fullmatch(r'[a-f0-9]{64}', digest) + or classification not in REVIEWED_CORRUPTION_CLASSIFICATIONS + ): + raise RuntimeError('reviewed keycheck corruption metadata is invalid') + file_rows = reviewed.setdefault(file_name, {}) + if offset in file_rows: + raise RuntimeError('reviewed keycheck corruption offsets are ambiguous') + file_rows[offset] = { + 'device': str(row['file_device']), + 'inode': str(row['file_inode']), + 'size': int(row['file_size']), + 'mtime_ns': str(row['file_mtime_ns']), + 'length': length, + 'sha256': digest, + 'classification': classification, + } + return reviewed + + +def reviewed_corruptions_for_file(reviewed, path, signature): + rows = (reviewed or {}).get(os.path.basename(path), {}) + if not rows: + return {} + actual = { + 'device': str(signature.get('device', 0)), + 'inode': str(signature.get('inode', 0)), + 'size': int(signature.get('size', -1)), + 'mtime_ns': str(signature.get('mtime_ns', -1)), + } + for issue in rows.values(): + expected = {key: issue[key] for key in ('device', 'inode', 'size', 'mtime_ns')} + if actual != expected: + raise RuntimeError(f'reviewed keycheck corruption file identity changed: {path}') + return rows + + +def skip_reviewed_corruption(path, offset, length, digest, reviewed): + issue = (reviewed or {}).get(int(offset)) + if issue is None: + return False + if int(issue['length']) != int(length) or issue['sha256'] != str(digest or '').lower(): + raise RuntimeError(f'reviewed keycheck corruption record changed: {path}:byte:{offset}') + print( + f"Warning: skipped exact reviewed keycheck corruption at {path}:byte:{offset}; " + f"classification={issue['classification']}", + flush=True, + ) + return True + + +def _drain_oversized_jsonl_identity(handle, first_chunk): + digest = hashlib.sha256(first_chunk) + length = len(first_chunk) + if first_chunk.endswith(b"\n"): + return True, length, digest.hexdigest() + while True: + chunk = handle.readline(64 * 1024) + if not chunk: + return False, length, digest.hexdigest() + digest.update(chunk) + length += len(chunk) + if chunk.endswith(b"\n"): + return True, length, digest.hexdigest() + + +def _drain_oversized_jsonl_record(handle, first_chunk): + return _drain_oversized_jsonl_identity(handle, first_chunk)[0] + + +def _warn_skipped_oversized(path, line_offset, max_line_bytes): + print( + f"Warning: skipped oversized committed keycheck input at {path}:byte:{line_offset}; " + f"line limit={max_line_bytes}", + flush=True, + ) + + +def checkpoint_matches_signature(path, state, signature): + if not isinstance(state, dict) or state.get("path") != os.path.abspath(path): + return False + state_inode = int(state.get("inode", 0) or 0) + signature_inode = int(signature.get("inode", 0) or 0) + if not state_inode or not signature_inode or state_inode != signature_inode: + return False + state_device = int(state.get("device", 0) or 0) + signature_device = int(signature.get("device", 0) or 0) + if (state_device or signature_device) and ( + not state_device or not signature_device or state_device != signature_device + ): + return False + state_size = int(state.get("size", -1)) + signature_size = int(signature.get("size", 0)) + if state_size < 0 or state_size > signature_size: + return False + if state_size == signature_size: + state_mtime = int(state.get("mtime_ns", -1)) + signature_mtime = int(signature.get("mtime_ns", -2)) + if state_mtime < 0 or state_mtime != signature_mtime: + return False + return True + + +def input_start_offset(input_file, state, signature): + if truthy_env("KEYCHECK_DISABLE_HIGH_WATERMARK"): + tail_bytes = tail_fallback_bytes() + return max(0, signature["size"] - tail_bytes) if tail_bytes > 0 else 0, "tail_disabled" if tail_bytes > 0 else "full_disabled" + + state = state or {} + if checkpoint_matches_signature(input_file, state, signature): + offset = int(state.get("offset", 0) or 0) + if 0 <= offset <= signature["size"]: + return offset, "high_watermark" + return 0, "full_replay" if state else "full_initial" + + +def iter_jsonl_input(input_file): + physical_segments = physical_jsonl_segments(input_file) + if not os.path.exists(input_file) and not physical_segments: + print(f"Warning: {input_file} not found. Nothing to read.") + return + manifest_path = jsonl_manifest_path(input_file) + if os.path.exists(manifest_path) or physical_segments: + yield from iter_segmented_jsonl_input(input_file, manifest_path) + return + service = keycheck_service_name() + state_file = keycheck_state_path(service) + state = read_json_file(state_file, {}) or {} + mode = "unopened" + final_offset = 0 + processed = 0 + skipped_oversized = 0 + previous_skipped_oversized = 0 + completed_read = False + checkpoint_signature = None + max_line_bytes = keycheck_input_max_line_bytes() + reviewed = load_reviewed_keycheck_corruptions(input_file) + skipped_reviewed = 0 + try: + with open(input_file, "rb") as f: + signature = input_file_signature(input_file, f) + file_reviews = reviewed_corruptions_for_file(reviewed, input_file, signature) + start_offset, mode = input_start_offset(input_file, state, signature) + previous_skipped_oversized = ( + int(state.get("skipped_oversized", 0) or 0) + if mode == "high_watermark" else 0 + ) + final_offset = start_offset + print( + f"Input reader: service={service or 'unknown'} mode={mode} " + f"offset={start_offset} size={signature['size']} state={state_file}", + flush=True, + ) + if start_offset > 0: + f.seek(start_offset) + if mode.startswith("tail"): + boundary_offset = f.tell() + skipped = f.readline(max_line_bytes + 1) + if len(skipped) > max_line_bytes: + if not _drain_oversized_jsonl_record(f, skipped): + raise RuntimeError(f"torn oversized keycheck tail boundary record in {input_file}") + _warn_skipped_oversized(input_file, boundary_offset, max_line_bytes) + skipped_oversized += 1 + elif skipped and not skipped.endswith(b"\n"): + raise RuntimeError(f"torn keycheck tail boundary record in {input_file}") + final_offset = f.tell() + while True: + line_offset = f.tell() + raw_line = f.readline(max_line_bytes + 1) + if not raw_line: + checkpoint_signature = input_file_signature(input_file, f) + completed_read = True + break + final_offset = f.tell() + if len(raw_line) > max_line_bytes: + complete, record_length, record_digest = _drain_oversized_jsonl_identity(f, raw_line) + if not complete: + final_offset = line_offset + completed_read = False + raise RuntimeError(f"torn oversized committed keycheck input at {input_file}:{line_offset}") + final_offset = f.tell() + if skip_reviewed_corruption( + input_file, line_offset, record_length, record_digest, file_reviews, + ): + skipped_oversized += 1 + skipped_reviewed += 1 + continue + skipped_oversized += 1 + _warn_skipped_oversized(input_file, line_offset, max_line_bytes) + continue + if not raw_line.endswith(b"\n"): + if skip_reviewed_corruption( + input_file, line_offset, len(raw_line), hashlib.sha256(raw_line).hexdigest(), file_reviews, + ): + skipped_reviewed += 1 + continue + final_offset = line_offset + completed_read = False + raise RuntimeError(f"torn committed keycheck input at {input_file}:{line_offset}") + try: + line = raw_line.decode("utf-8") + data = json.loads(line) + except (UnicodeDecodeError, ValueError) as exc: + if skip_reviewed_corruption( + input_file, line_offset, len(raw_line), hashlib.sha256(raw_line).hexdigest(), file_reviews, + ): + skipped_reviewed += 1 + continue + final_offset = line_offset + completed_read = False + kind = "torn" if not raw_line.endswith(b"\n") else "invalid" + raise RuntimeError(f"{kind} committed keycheck input at {input_file}:{line_offset}") from exc + processed += 1 + yield { + "source": f"{input_file}:byte:{line_offset}", + "line_offset": line_offset, + "data": data, + } + finally: + if completed_read and checkpoint_signature is not None: + new_state = { + **checkpoint_signature, + "offset": int(final_offset), + "processed": int(processed), + "skipped_oversized": previous_skipped_oversized + int(skipped_oversized), + "skipped_reviewed_corrupt": int(state.get('skipped_reviewed_corrupt', 0) or 0) + skipped_reviewed, + "mode": mode, + "updated_at": now_iso(), + } + write_json_file(state_file, new_state) + print( + f"Input reader done: service={service or 'unknown'} processed={processed} " + f"skipped_oversized={skipped_oversized} new_offset={final_offset}", + flush=True, + ) + else: + print( + f"Input reader stopped before EOF: service={service or 'unknown'} processed={processed}; " + f"state offset unchanged", + flush=True, + ) + + +def manifest_input_files(input_file, manifest): + root = os.path.dirname(os.path.abspath(input_file)) + physical = physical_jsonl_segments(input_file) + files = [path for _, path in physical] + listed_names = { + os.path.basename(str(segment.get("path") or segment.get("name") or "")) + for segment in (manifest.get("segments") or []) if isinstance(segment, dict) + } + missing = sorted(name for name in listed_names if name and name not in {os.path.basename(path) for path in files}) + if missing: + raise RuntimeError(f"manifest-listed keycheck segment is missing: {missing[0]}") + current = os.path.abspath(input_file) + if current and os.path.exists(current): + files.append(os.path.abspath(current)) + seen = set() + out = [] + for path in files: + if path not in seen: + seen.add(path) + out.append(path) + return out + + +def iter_segmented_jsonl_input(input_file, manifest_path): + service = keycheck_service_name() + state_file = keycheck_state_path(service) + state = read_json_file(state_file, {}) or {} + manifest = read_json_file(manifest_path, {}) or {} + files_state = state.get("files") if isinstance(state.get("files"), dict) else {} + files = manifest_input_files(input_file, manifest) + first_manifest_run = not bool(files_state) + force_replay = truthy_env("KEYCHECK_DISABLE_HIGH_WATERMARK") + current_path = os.path.abspath(manifest.get("current_path") or input_file) + if current_path != os.path.abspath(input_file): + raise RuntimeError("keycheck manifest current path does not match the configured input") + # Physical segments are always discoverable and downstream publication is + # idempotent. Never seek past current-file bytes based on mutable manifest + # state; a writer could truncate and append after any generation check. + skip_current = 0 + physical = physical_jsonl_segments(input_file) + total_processed = 0 + total_skipped_oversized = 0 + max_line_bytes = keycheck_input_max_line_bytes() + reviewed = load_reviewed_keycheck_corruptions(input_file) + total_skipped_reviewed = 0 + print( + f"Input reader: service={service or 'unknown'} mode=manifest files={len(files)} " + f"state={state_file}", + flush=True, + ) + try: + for path in files: + file_state = files_state.get(os.path.abspath(path)) or files_state.get(path) or {} + processed = 0 + skipped_oversized = 0 + skipped_reviewed = 0 + with open(path, "rb") as f: + signature = input_file_signature(path, f) + file_reviews = reviewed_corruptions_for_file(reviewed, path, signature) + if force_replay: + offset = 0 + previous_skipped_oversized = 0 + previous_skipped_reviewed = 0 + elif checkpoint_matches_signature(path, file_state, signature): + offset = int(file_state.get("offset", 0) or 0) + previous_skipped_oversized = int(file_state.get("skipped_oversized", 0) or 0) + previous_skipped_reviewed = int(file_state.get('skipped_reviewed_corrupt', 0) or 0) + else: + offset = 0 + previous_skipped_oversized = 0 + previous_skipped_reviewed = 0 + if offset < 0 or offset > signature["size"]: + offset = 0 + previous_skipped_oversized = 0 + previous_skipped_reviewed = 0 + if not force_replay and first_manifest_run and os.path.abspath(path) == current_path: + offset, _ = input_start_offset(path, {}, signature) + if os.path.abspath(path) == current_path: + offset = max(offset, min(skip_current, signature["size"])) + final_offset = offset + f.seek(offset) + while True: + line_offset = f.tell() + raw_line = f.readline(max_line_bytes + 1) + if not raw_line: + break + final_offset = f.tell() + if len(raw_line) > max_line_bytes: + complete, record_length, record_digest = _drain_oversized_jsonl_identity(f, raw_line) + if not complete: + final_offset = line_offset + raise RuntimeError(f"torn oversized committed keycheck input at {path}:{line_offset}") + final_offset = f.tell() + if skip_reviewed_corruption( + path, line_offset, record_length, record_digest, file_reviews, + ): + skipped_oversized += 1 + skipped_reviewed += 1 + total_skipped_reviewed += 1 + continue + skipped_oversized += 1 + _warn_skipped_oversized(path, line_offset, max_line_bytes) + continue + if not raw_line.endswith(b"\n"): + if skip_reviewed_corruption( + path, line_offset, len(raw_line), hashlib.sha256(raw_line).hexdigest(), file_reviews, + ): + skipped_reviewed += 1 + total_skipped_reviewed += 1 + continue + final_offset = line_offset + raise RuntimeError(f"torn committed keycheck input at {path}:{line_offset}") + try: + line = raw_line.decode("utf-8") + data = json.loads(line) + except (UnicodeDecodeError, ValueError) as exc: + if skip_reviewed_corruption( + path, line_offset, len(raw_line), hashlib.sha256(raw_line).hexdigest(), file_reviews, + ): + skipped_reviewed += 1 + total_skipped_reviewed += 1 + continue + final_offset = line_offset + kind = "torn" if not raw_line.endswith(b"\n") else "invalid" + raise RuntimeError(f"{kind} committed keycheck input at {path}:{line_offset}") from exc + processed += 1 + total_processed += 1 + yield { + "source": f"{path}:byte:{line_offset}", + "line_offset": line_offset, + "data": data, + } + checkpoint_signature = input_file_signature(path, f) + files_state[os.path.abspath(path)] = { + **checkpoint_signature, + "offset": int(final_offset), + "done": int(final_offset) >= int(checkpoint_signature["size"]), + "processed": int(processed), + "skipped_oversized": previous_skipped_oversized + int(skipped_oversized), + "skipped_reviewed_corrupt": previous_skipped_reviewed + skipped_reviewed, + "updated_at": now_iso(), + } + total_skipped_oversized += skipped_oversized + finally: + new_state = { + "manifest_path": os.path.abspath(manifest_path), + "current_file": os.path.abspath(input_file), + "files": files_state, + "processed": int(total_processed), + "skipped_oversized": sum( + int(item.get("skipped_oversized", 0) or 0) + for item in files_state.values() if isinstance(item, dict) + ), + "skipped_reviewed_corrupt": sum( + int(item.get('skipped_reviewed_corrupt', 0) or 0) + for item in files_state.values() if isinstance(item, dict) + ), + "mode": "manifest", + "updated_at": now_iso(), + } + write_json_file(state_file, new_state) + print( + f"Input reader done: service={service or 'unknown'} mode=manifest " + f"processed={total_processed} skipped_oversized={total_skipped_oversized}", + flush=True, + ) + + +def mask_secret(value): + if not value or len(value) < 12: + return value or "" + return f"{value[:8]}...{value[-4:]}" + + +def keycheck_status_group(status): + status = str(status or "UNKNOWN").strip().upper() + alive = {"ALIVE", "VALID", "VALID_2FA", "VALID_RATE_LIMITED", "BEDROCK", "ADMIN", "VERTEX", "CANARY", "FOUNDRY"} + dead = {"DEAD", "INVALID", "EXPIRED", "LEAKED_REVOKED", "INVALID_OR_REVOKED"} + restricted = {"RESTRICTED", "API_DISABLED", "ACCESS_DENIED", "QUARANTINED", "DISABLED"} + no_balance = {"NO_BALANCE", "NO_QUOTA", "LIMITED_OR_NO_BALANCE", "LIMITED_OR_QUOTA"} + no_context = {"NO_CONTEXT", "NO_USERNAME", "NO_TARGET", "NO_TARGET_MODELS", "NO_GENERATION_MODEL"} + limited = {"LIMITED", "RATE_LIMITED"} + network = {"NETWORK", "NETWORK_ERROR"} + if status in alive: + return "alive" + if status in dead: + return "dead" + if status in restricted: + return "restricted" + if status in no_balance: + return "no_balance" + if status in no_context: + return "no_context" + if status in limited: + return "limited" + if status in network: + return "network" + return "unknown" + + +def redact_value(value, raw_key): + if isinstance(value, dict): + return {k: redact_value(v, raw_key) for k, v in value.items() if k not in ("finding", "raw_secret", "raw", "raw_v2")} + if isinstance(value, list): + return [redact_value(item, raw_key) for item in value] + if isinstance(value, str): + text = value.replace(raw_key, "***REDACTED***") if raw_key else value + return text[:4000] + return value + + +def result_message(result): + if not isinstance(result, dict): + return "" + if result.get("message"): + return result.get("message") + error = result.get("error") + if isinstance(error, dict): + return error.get("message") or json.dumps(error, ensure_ascii=False)[:1000] + return error or "" + + +def finding_secret_hash(finding): + if not isinstance(finding, dict): + return "" + raw = extract_raw_secret(finding) + return sha256_text(raw) if raw else "" + + +def clear_provider_routing_evidence_cache(): + global PROVIDER_ROUTING_EVIDENCE_CACHE_AUTHORITY, PROVIDER_ROUTING_EVIDENCE_DB_FAILED + PROVIDER_ROUTING_EVIDENCE_CACHE.clear() + PROVIDER_ROUTING_EVIDENCE_CACHE_AUTHORITY = None + PROVIDER_ROUTING_EVIDENCE_DB_FAILED = False + + +def _provider_routing_explicit_evidence(finding): + if not isinstance(finding, dict): + return set() + detector = str(finding.get("DetectorName") or "").strip().lower() + detector_name = detector + if detector == "customregex": + extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {} + detector_name = str(extra.get("name") or "").strip().lower() + + evidence = set() + if detector_name in QWEN_EXPLICIT_ROUTING_DETECTORS: + evidence.add("qwen") + if detector_name in DEEPSEEK_EXPLICIT_ROUTING_DETECTORS: + evidence.add("deepseek") + if detector_name in KIMI_EXPLICIT_ROUTING_DETECTORS: + evidence.add("kimi") + if detector_name in ZAI_EXPLICIT_ROUTING_DETECTORS: + evidence.add("zai") + + context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {} + if context.get("provider_hint_source") != EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE: + return evidence + hint = context.get("provider_hint") + if hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT: + evidence.update(QWEN_DEEPSEEK_PROVIDERS) + return evidence + if hint == AMBIGUOUS_GENERIC_SK_HINT: + evidence.update(GENERIC_SK_PROVIDERS) + return evidence + if hint in GENERIC_SK_PROVIDERS: + evidence.add(hint) + return evidence + return set(GENERIC_SK_PROVIDERS) + + +def _provider_routing_cache_limits(): + return { + "items": min(100000, max(1, env_int("KEYCHECK_PROVIDER_EVIDENCE_CACHE_ITEMS", 4096))), + "rows": min(1024, max(1, env_int("KEYCHECK_PROVIDER_EVIDENCE_MAX_ROWS", 64))), + "json_chars": min( + 1024 * 1024, + max(1024, env_int("KEYCHECK_PROVIDER_EVIDENCE_MAX_JSON_CHARS", 256 * 1024)), + ), + } + + +def _provider_routing_database_authority(): + db_url = str(os.getenv("KEYCHECK_DB_URL") or "").strip() + db_path = "" if db_url else str(os.getenv("KEYCHECK_DB_PATH") or "").strip() + identity = db_url or (os.path.abspath(db_path) if db_path else "") + return db_url, db_path, sha256_text(identity) if identity else "" + + +def _load_provider_routing_evidence(secret_hash, limits): + db_url, db_path, _ = _provider_routing_database_authority() + if not db_url and not db_path: + raise RuntimeError("canonical keycheck database is unavailable") + db = ScannerDB(db_path=db_path or None, db_url=db_url or "", initialize=False) + try: + if not db.enabled: + raise RuntimeError("canonical keycheck database connection is unavailable") + db.require_runtime_safety_schema() + if getattr(db.conn, "is_postgres", False): + statement_timeout_ms = min( + 30000, + max(250, env_int("KEYCHECK_PROVIDER_EVIDENCE_STATEMENT_TIMEOUT_MS", 5000)), + ) + lock_timeout_ms = min( + statement_timeout_ms, + max(100, env_int("KEYCHECK_PROVIDER_EVIDENCE_LOCK_TIMEOUT_MS", 1000)), + ) + db.conn.execute("SELECT set_config('statement_timeout', ?, false)", (f"{statement_timeout_ms}ms",)) + db.conn.execute("SELECT set_config('lock_timeout', ?, false)", (f"{lock_timeout_ms}ms",)) + rows = db.provider_routing_finding_rows( + secret_hash, + limits["rows"] + 1, + limits["json_chars"], + ) + finally: + db.close() + + if len(rows) > limits["rows"]: + return frozenset(GENERIC_SK_PROVIDERS) + evidence = set() + for row in rows: + if bool(row["truncated"]): + return frozenset(GENERIC_SK_PROVIDERS) + raw_finding_json = row["raw_finding_json"] or "" + if not raw_finding_json: + detector_name = str(row["detector_name"] or "").strip().lower() + detector_evidence = _provider_routing_explicit_evidence({ + "DetectorName": detector_name, + }) + if detector_evidence: + evidence.update(detector_evidence) + elif detector_name not in GENERIC_SK_NON_EXPLICIT_ROUTING_DETECTORS: + return frozenset(GENERIC_SK_PROVIDERS) + if GENERIC_SK_PROVIDERS.issubset(evidence): + return frozenset(GENERIC_SK_PROVIDERS) + continue + try: + finding = json.loads(raw_finding_json) + except (TypeError, ValueError, json.JSONDecodeError): + return frozenset(GENERIC_SK_PROVIDERS) + if isinstance(finding, dict) and isinstance(finding.get("finding"), dict): + finding = finding["finding"] + if not isinstance(finding, dict): + return frozenset(GENERIC_SK_PROVIDERS) + raw = extract_raw_secret(finding) + if not raw or sha256_text(raw) != secret_hash: + return frozenset(GENERIC_SK_PROVIDERS) + evidence.update(_provider_routing_explicit_evidence(finding)) + if GENERIC_SK_PROVIDERS.issubset(evidence): + return frozenset(GENERIC_SK_PROVIDERS) + return frozenset(evidence) + + +def global_provider_routing_evidence(raw_key): + """Return persisted generic sk-key provider evidence without retaining the raw key.""" + global PROVIDER_ROUTING_EVIDENCE_CACHE_AUTHORITY, PROVIDER_ROUTING_EVIDENCE_DB_FAILED + PROVIDER_ROUTING_EVIDENCE_DB_FAILED = False + secret_hash = sha256_text(raw_key) + if not secret_hash: + return set(GENERIC_SK_PROVIDERS) + _, _, authority = _provider_routing_database_authority() + if authority != PROVIDER_ROUTING_EVIDENCE_CACHE_AUTHORITY: + PROVIDER_ROUTING_EVIDENCE_CACHE.clear() + PROVIDER_ROUTING_EVIDENCE_CACHE_AUTHORITY = authority + PROVIDER_ROUTING_EVIDENCE_DB_FAILED = False + cached = PROVIDER_ROUTING_EVIDENCE_CACHE.get(secret_hash) + if cached is not None: + evidence, retry_at = cached + if retry_at is None or retry_at > time.monotonic(): + PROVIDER_ROUTING_EVIDENCE_CACHE.move_to_end(secret_hash) + PROVIDER_ROUTING_EVIDENCE_DB_FAILED = retry_at is not None + return set(evidence) + del PROVIDER_ROUTING_EVIDENCE_CACHE[secret_hash] + + limits = _provider_routing_cache_limits() + try: + evidence = _load_provider_routing_evidence(secret_hash, limits) + except Exception: + PROVIDER_ROUTING_EVIDENCE_DB_FAILED = True + evidence = frozenset(GENERIC_SK_PROVIDERS) + retry_at = time.monotonic() + min( + 5, + max(1, env_int("KEYCHECK_PROVIDER_EVIDENCE_ERROR_CACHE_SEC", 1)), + ) + else: + retry_at = None + + if evidence == GENERIC_SK_PROVIDERS: + PROVIDER_ROUTING_EVIDENCE_CACHE[secret_hash] = (evidence, retry_at) + PROVIDER_ROUTING_EVIDENCE_CACHE.move_to_end(secret_hash) + while len(PROVIDER_ROUTING_EVIDENCE_CACHE) > limits["items"]: + PROVIDER_ROUTING_EVIDENCE_CACHE.popitem(last=False) + return set(evidence) + + +def provider_routing_database_failed(): + return bool(PROVIDER_ROUTING_EVIDENCE_DB_FAILED) + + +def combined_provider_routing_hint(raw_key, local_hint=""): + global PROVIDER_ROUTING_EVIDENCE_DB_FAILED + PROVIDER_ROUTING_EVIDENCE_DB_FAILED = False + evidence = set() + if local_hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT: + evidence.update(QWEN_DEEPSEEK_PROVIDERS) + elif local_hint == AMBIGUOUS_GENERIC_SK_HINT: + evidence.update(GENERIC_SK_PROVIDERS) + elif local_hint in GENERIC_SK_PROVIDERS: + evidence.add(local_hint) + elif keycheck_input_mode() == 'postgres': + service = keycheck_service_name() + if service in GENERIC_SK_PROVIDERS: + evidence.add(service) + if len(evidence) > 1: + return ( + AMBIGUOUS_QWEN_DEEPSEEK_HINT + if evidence == set(QWEN_DEEPSEEK_PROVIDERS) + else AMBIGUOUS_GENERIC_SK_HINT + ) + evidence.update(global_provider_routing_evidence(raw_key)) + if len(evidence) > 1: + return ( + AMBIGUOUS_QWEN_DEEPSEEK_HINT + if evidence == set(QWEN_DEEPSEEK_PROVIDERS) + else AMBIGUOUS_GENERIC_SK_HINT + ) + if not evidence: + return "" + provider = next(iter(evidence)) + if provider not in GENERIC_SK_PROVIDERS: + return AMBIGUOUS_GENERIC_SK_HINT + return provider + + +def strip_finding_nearby_context(finding): + if not isinstance(finding, dict): + return finding + output = dict(finding) + context = output.get("ScannerContext") + if isinstance(context, dict) and "nearby" in context: + output["ScannerContext"] = {key: value for key, value in context.items() if key != "nearby"} + return output + + +def finding_detector_secret_hash(finding): + if not isinstance(finding, dict): + return "" + raw = extract_raw_secret(finding) + secret_hash = sha256_text(raw) if raw else "" + fallback_hash = sha256_text(json_dumps(finding)) if not secret_hash else "" + detector = str(finding.get("DetectorName") or finding.get("DetectorType") or "") + return sha256_text("|".join([detector, secret_hash or fallback_hash])) if detector or secret_hash or fallback_hash else "" + + +def finding_uid(finding): + if not isinstance(finding, dict): + return "" + return str(finding.get("finding_uid") or finding.get("FindingUID") or "") + + +def keycheck_event_id(service, key_hash, status, source, finding_uid_value, checked_at, result_source, payload): + explicit = payload.get("event_id") if isinstance(payload, dict) else "" + if explicit: + return str(explicit) + cached_id = payload.get("cached_occurrence_id") if isinstance(payload, dict) else "" + if cached_id: + return sha256_text("|".join([str(service or ""), "cached", str(cached_id)])) + return sha256_text("|".join([ + str(service or ""), + str(key_hash or ""), + str(status or ""), + str(source or ""), + str(finding_uid_value or ""), + str(checked_at or ""), + str(result_source or "api_check"), + ])) + + +def write_keycheck_event(service, results_file, key, result, source="", finding=None, detector=None, result_source="api_check"): + result = result if isinstance(result, dict) else {"status": str(result or "UNKNOWN")} + finding = strip_finding_nearby_context(finding) + status = str(result.get("status") or "UNKNOWN").upper() + checked_at = result.get("checked_at") or now_iso() + detector_name = detector or get_detector_name(finding or {}) or result.get("detector") or "" + key_hash = result.get("key_hash") or sha256_text(key) + secret_hash = result.get("secret_hash") or finding_secret_hash(finding) or key_hash + detector_secret_hash = result.get("detector_secret_hash") or finding_detector_secret_hash(finding) + finding_uid_value = result.get("finding_uid") or finding_uid(finding) + payload = { + "key_masked": result.get("key_masked") or mask_secret(key), + "key_hash": key_hash, + "secret_hash": secret_hash, + "detector_secret_hash": detector_secret_hash, + "finding_uid": finding_uid_value, + "detector": detector_name, + "source": source, + "finding": finding or {}, + "checked_at": checked_at, + "result_source": result.get("result_source") or result_source or "api_check", + **result, + } + payload["status"] = status + payload["checked_at"] = checked_at + payload["result_source"] = result.get("result_source") or result_source or "api_check" + payload["event_id"] = keycheck_event_id( + service, + payload.get("key_hash"), + payload.get("status"), + source, + payload.get("finding_uid"), + payload.get("checked_at"), + payload.get("result_source"), + payload, + ) + if keycheck_input_mode() == 'postgres': + global _ACTIVE_DB_CANDIDATE + candidate = _ACTIVE_DB_CANDIDATE + if not candidate or str(candidate.get('service') or '') != str(service or ''): + raise RuntimeError('keycheck result has no exact active PostgreSQL candidate lease') + payload['event_id'] = sha256_text('|'.join(( + 'truf-keycheck-event-v1', str(candidate['id']), str(candidate['attempts']), + ))) + metadata = { + name: value for name, value in payload.items() + if name not in ('finding', 'key_masked', 'key_hash', 'secret_hash') + } + resolved_service = str(payload.get('resolved_provider') or '').strip().lower() + if resolved_service: + metadata['resolution_origin_service'] = str(service or '').strip().lower() + outcome = _postgres_candidate_db().complete_keycheck_candidate( + candidate['id'], candidate['lease_token'], payload['event_id'], + payload['status'], keycheck_status_group(payload['status']), + checked_at=payload['checked_at'], message=result_message(payload), + metadata=redact_value(metadata, key), result_source=payload['result_source'], + resolved_service=resolved_service, + ) + if not outcome or not outcome.get('completed'): + raise RuntimeError('PostgreSQL keycheck candidate completion lost its fence') + candidate['_completed'] = True + else: + append_jsonl(results_file, payload) + return payload + + +def record_validation_result(service, key, result, source="", finding=None, detector=None, db_path=None): + if keycheck_input_mode() == 'postgres': + candidate = _ACTIVE_DB_CANDIDATE + return bool(candidate and candidate.get('_completed')) + db_path = db_path or os.getenv("KEYCHECK_DB_PATH") or os.getenv("SCANNER_DB_PATH") or os.getenv("SCAN_DB_PATH") + db_url = os.getenv("KEYCHECK_DB_URL") or os.getenv("SCANNER_DB_URL") or os.getenv("DATABASE_URL") + if not db_path and not db_url: + return + if not truthy_env("KEYCHECK_DB_INLINE", False): + return False + result = result if isinstance(result, dict) else {"status": str(result or "UNKNOWN")} + finding = strip_finding_nearby_context(finding) + status = str(result.get("status") or "UNKNOWN").upper() + secret_hash = finding_secret_hash(finding) or sha256_text(key) + detector_secret_hash = finding_detector_secret_hash(finding) + metadata = redact_value({**result, "source_line": source, "detector_secret_hash": detector_secret_hash}, key) + result_source = metadata.get("result_source") or "api_check" + try: + db = ScannerDB(db_path=db_path, db_url=db_url, initialize=False) + try: + db.require_runtime_safety_schema() + if getattr(db.conn, "is_postgres", False): + statement_timeout_ms = max(1000, env_int("KEYCHECK_DB_STATEMENT_TIMEOUT_MS", 8000)) + lock_timeout_ms = max(250, env_int("KEYCHECK_DB_LOCK_TIMEOUT_MS", 2000)) + db.conn.execute("SELECT set_config('statement_timeout', ?, false)", (f"{statement_timeout_ms}ms",)) + db.conn.execute("SELECT set_config('lock_timeout', ?, false)", (f"{lock_timeout_ms}ms",)) + ok = db.record_keycheck_result( + service=service, + status=status, + status_group=keycheck_status_group(status), + checked_at=result.get("checked_at") or now_iso(), + key_hash=sha256_text(key), + secret_hash=secret_hash, + key_masked=result.get("key_masked") or mask_secret(key), + detector_name=detector or get_detector_name(finding or {}) or result.get("detector") or "", + message=result_message(result), + metadata=metadata, + link_findings=truthy_env("KEYCHECK_DB_LINK_FINDINGS", False), + source_line=source, + detector_secret_hash=detector_secret_hash, + ) + if not ok: + raise RuntimeError("keycheck DB write returned false") + finally: + db.close() + record_db_write_metric(service, True, result_source, status) + return True + except Exception as exc: + print(f"Warning: unable to record keycheck result in DB: {exc}") + record_db_write_metric(service, False, result_source, status, str(exc)) + return False + + +def record_db_write_metric(service, ok, result_source, status, error=""): + output_dir = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(service or keycheck_service_name() or "default") + require_private_directory(output_dir, create=True) + payload = { + "service": service or keycheck_service_name(), + "ok": bool(ok), + "result_source": result_source or "api_check", + "status": status, + "error": str(error or "")[:500], + "created_at": now_iso(), + } + try: + append_jsonl(os.path.join(output_dir, "db_write_metrics.jsonl"), payload) + if not ok: + with private_append_writer(os.path.join(output_dir, "db_write_failures.log")) as f: + f.write(f"{payload['created_at']}\t{payload['status']}\t{payload['result_source']}\t{payload['error']}\n") + except OSError: + pass + + +def normalize_status_key(line): + line = line.strip() + if not line: + return None + if "\t" in line: + return line.split("\t", 1)[0].strip() + return line.strip() + + +def iter_bounded_text_lines(path): + if keycheck_input_mode() == 'postgres': + return + max_bytes = max(1, env_int('KEYCHECK_INPUT_LIST_MAX_BYTES', 32 * 1024 * 1024)) + max_items = max(1, env_int('KEYCHECK_INPUT_LIST_MAX_ITEMS', 100000)) + max_line_bytes = status_projection_line_max_bytes() + if not os.path.exists(path): + return + if os.path.getsize(path) > max_bytes: + raise RuntimeError(f'keycheck input list exceeds its aggregate byte bound: {path}') + with open(path, 'rb') as handle: + for index, raw_line in enumerate(handle, 1): + if index > max_items: + raise RuntimeError(f'keycheck input list exceeds its item bound: {path}') + if len(raw_line) > max_line_bytes: + raise RuntimeError(f'keycheck input list line exceeds its byte bound: {path}:{index}') + yield raw_line.decode('utf-8', errors='replace') + + +def load_keys_from_file(path): + if keycheck_input_mode() == 'postgres': + return set() + if not os.path.exists(path): + return set() + return {key for key in (normalize_status_key(line) for line in iter_bounded_text_lines(path)) if key} + + +def load_checked_statuses(path): + statuses = {} + if keycheck_input_mode() == 'postgres': + return statuses + if not os.path.exists(path): + return statuses + for line in iter_bounded_text_lines(path): + parts = line.rstrip("\n").split("\t") + if not parts or not parts[0]: + continue + statuses[parts[0]] = parts[1] if len(parts) > 1 else "UNKNOWN" + return statuses + + +def ensure_output_files(paths): + if keycheck_input_mode() == 'postgres': + return + for path in paths: + parent = os.path.dirname(path) + if parent: + require_private_directory(parent, create=True) + if os.path.lexists(path): + require_private_file(path) + continue + with private_atomic_writer(path, binary=True, suffix=".empty.tmp"): + pass + + +def remove_key_from_files(key, paths): + if keycheck_input_mode() == 'postgres': + return + for path in paths: + if not os.path.exists(path): + continue + lock_path = f"{path}.lock" + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + require_private_file(path) + with open(path, "r", encoding="utf-8") as f: + lines = f.readlines() + with private_atomic_writer(path) as f: + for line in lines: + if normalize_status_key(line) != key: + f.write(line) + finally: + release_file_lock(lock, lock_path) + + +def append_status(path, key, status, message="", extra=""): + if keycheck_input_mode() == 'postgres': + return 'postgres-authoritative' + message = str(message or "").replace("\n", " ")[:1000] + extra = str(extra or "").replace("\n", " ")[:1000] + require_private_directory(os.path.dirname(os.path.abspath(path)), create=True) + if os.path.lexists(path): + reject_reparse_components(path) + lock_path = f"{path}.lock" + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + with private_append_writer(path) as f: + f.write(f"{key}\t{status}\t{message}\t{extra}\n") + finally: + release_file_lock(lock, lock_path) + + +def append_checked(path, key, status): + if keycheck_input_mode() == 'postgres': + return 'postgres-authoritative' + require_private_directory(os.path.dirname(os.path.abspath(path)), create=True) + if os.path.lexists(path): + reject_reparse_components(path) + lock_path = f"{path}.lock" + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + with private_append_writer(path) as f: + f.write(f"{key}\t{status}\t{now_iso()}\n") + finally: + release_file_lock(lock, lock_path) + + +def append_jsonl(path, payload): + if keycheck_input_mode() == 'postgres': + return 'postgres-authoritative' + parent = os.path.dirname(path) + if parent: + require_private_directory(parent, create=True) + if os.path.lexists(path): + reject_reparse_components(path) + fd = None + lock_path = f"{path}.lock" + try: + fd = acquire_file_lock(lock_path, env_int("KEYCHECK_JSONL_LOCK_STALE_SEC", 300)) + repair_keycheck_jsonl_tail(path) + reconcile_keycheck_jsonl_segments(path) + if should_rotate_jsonl(path): + max_mb = env_int("KEYCHECK_RESULTS_MAX_MB", 32) + rotate_jsonl_if_needed(path, max_mb * 1024 * 1024) + with private_append_writer(path) as f: + f.write(json.dumps(payload, ensure_ascii=False, default=str) + "\n") + finally: + if fd is not None: + release_file_lock(fd, lock_path) + + +def load_known_keys(checked_file, status_files): + return set(load_known_statuses(checked_file, status_files).keys()) + + +def normalize_status_files(status_files): + output = {} + if not status_files: + return output + if isinstance(status_files, dict): + for key, value in status_files.items(): + key_text = str(key) + value_text = str(value) + if key_text.lower().endswith(".txt") or os.sep in key_text or "/" in key_text: + output[value_text.upper()] = key_text + else: + output[key_text.upper()] = value_text + else: + for path in status_files: + output["UNKNOWN"] = str(path) + return output + + +def status_transaction_journal_path(checked_file): + return os.path.join( + os.path.dirname(os.path.abspath(checked_file)), + STATUS_TRANSACTION_JOURNAL_FILENAME, + ) + + +def status_transaction_lock_path(checked_file): + return os.path.join( + os.path.dirname(os.path.abspath(checked_file)), + STATUS_TRANSACTION_LOCK_FILENAME, + ) + + +def _status_transaction_layout(checked_file, status_files): + checked_file = os.path.abspath(checked_file) + output_dir = os.path.dirname(checked_file) + require_private_directory(output_dir, create=True) + normalized = { + str(status or "").strip().upper(): os.path.abspath(path) + for status, path in normalize_status_files(status_files).items() + } + if not normalized or any( + not re.fullmatch(r"[A-Z][A-Z0-9_]{0,127}", status) + for status in normalized + ): + raise ValueError("status transaction requires named status files") + status_paths = list(dict.fromkeys( + os.path.abspath(path) for path in normalized.values() + )) + paths = [checked_file, *status_paths] + if any(os.path.normcase(os.path.dirname(path)) != os.path.normcase(output_dir) for path in paths): + raise ValueError("status transaction files must share one service directory") + if os.path.normcase(checked_file) in {os.path.normcase(path) for path in status_paths}: + raise ValueError("status transaction checked and status paths must be distinct") + ensure_output_files(paths) + return checked_file, normalized + + +def _status_transaction_key(key): + key = str(key or "") + if not key or any(character in key for character in ("\x00", "\t", "\r", "\n")): + raise ValueError("status transaction key is empty or contains a control separator") + if len(key.encode("utf-8", errors="strict")) > status_projection_line_max_bytes(): + raise ValueError("status transaction key exceeds its byte bound") + return key + + +def _status_transaction_field(value): + return str(value or "").replace("\r", " ").replace("\n", " ")[:1000] + + +def _status_transaction_line(value, name): + if not isinstance(value, str): + raise ValueError(f"status transaction {name} must be text") + if not value.endswith("\n") or "\n" in value[:-1] or "\r" in value or "\x00" in value: + raise ValueError(f"status transaction {name} must be exactly one LF-terminated line") + if len(value.encode("utf-8", errors="strict")) > status_projection_line_max_bytes(): + raise ValueError(f"status transaction {name} exceeds its byte bound") + return value + + +def _status_projection_line_matches_key(line, key): + value = str(line or "").rstrip("\r\n") + if value == key: + return True + if not value.startswith(key): + return False + suffix = value[len(key):] + return suffix.startswith("\t") or suffix.startswith(":") + + +def _validate_status_projection_line(line, key): + line = _status_transaction_line(line, "status line") + if not _status_projection_line_matches_key(line, key): + raise ValueError("status transaction status line does not contain its exact key") + return line + + +def _validate_checked_projection_line(line, key, status): + line = _status_transaction_line(line, "checked line") + parts = line[:-1].split("\t") + if len(parts) < 2 or parts[0] != key or parts[1] != status: + raise ValueError("status transaction checked line does not match its key and status") + return line + + +def _new_status_transaction( + key, + status, + status_path, + message, + extra, + status_line=None, + checked_line=None, +): + key = _status_transaction_key(key) + status = str(status or "UNKNOWN").strip().upper() + message = _status_transaction_field(message) + extra = _status_transaction_field(extra) + checked_at = now_iso() + status_line = _validate_status_projection_line( + status_line if status_line is not None else f"{key}\t{status}\t{message}\t{extra}\n", + key, + ) + checked_line = _validate_checked_projection_line( + checked_line if checked_line is not None else f"{key}\t{status}\t{checked_at}\n", + key, + status, + ) + transaction = { + "version": 2, + "transaction_id": uuid.uuid4().hex, + "key": key, + "status": status, + "status_file": os.path.basename(status_path), + "message": message, + "extra": extra, + "checked_at": checked_at, + "status_line": status_line, + "checked_line": checked_line, + } + encoded = json.dumps(transaction, ensure_ascii=True, sort_keys=True, separators=(",", ":")).encode("utf-8") + if len(encoded) + 1 > status_transaction_journal_max_bytes(): + raise ValueError("status transaction journal exceeds its byte bound") + return transaction + + +def _validate_status_transaction(transaction, status_files): + if not isinstance(transaction, dict) or type(transaction.get("version")) is not int or transaction["version"] not in (1, 2): + raise RuntimeError("invalid status transaction journal version") + version = transaction["version"] + transaction_id = transaction.get("transaction_id") + if not isinstance(transaction_id, str) or not re.fullmatch(r"[a-f0-9]{32}", transaction_id): + raise RuntimeError("invalid status transaction identity") + raw_key = transaction.get("key") + if not isinstance(raw_key, str): + raise RuntimeError("invalid status transaction key") + try: + key = _status_transaction_key(raw_key) + except (UnicodeError, ValueError) as exc: + raise RuntimeError(f"invalid status transaction key: {exc}") from exc + raw_status = transaction.get("status") + if not isinstance(raw_status, str) or raw_status != raw_status.strip().upper(): + raise RuntimeError("status transaction status is not canonical") + status = raw_status + if status not in status_files: + raise RuntimeError("status transaction targets an unknown projection") + status_file = transaction.get("status_file") + if not isinstance(status_file, str) or os.path.normcase(status_file) != os.path.normcase( + os.path.basename(status_files[status]) + ): + raise RuntimeError("status transaction target does not match its projection") + message = _status_transaction_field(transaction.get("message")) + extra = _status_transaction_field(transaction.get("extra")) + if message != transaction.get("message") or extra != transaction.get("extra"): + raise RuntimeError("status transaction fields are not canonical") + checked_at = transaction.get("checked_at") + if not isinstance(checked_at, str): + raise RuntimeError("invalid status transaction timestamp") + if not checked_at or len(checked_at) > 64 or any(character in checked_at for character in ("\t", "\r", "\n")): + raise RuntimeError("invalid status transaction timestamp") + expected_status_line = f"{key}\t{status}\t{message}\t{extra}\n" + expected_checked_line = f"{key}\t{status}\t{checked_at}\n" + try: + if version == 1: + if transaction.get("status_line") != expected_status_line or transaction.get("checked_line") != expected_checked_line: + raise RuntimeError("status transaction projection data is inconsistent") + _validate_status_projection_line(transaction["status_line"], key) + _validate_checked_projection_line(transaction["checked_line"], key, status) + else: + _validate_status_projection_line(transaction.get("status_line"), key) + _validate_checked_projection_line(transaction.get("checked_line"), key, status) + except (TypeError, UnicodeError, ValueError) as exc: + raise RuntimeError(f"invalid status transaction projection data: {exc}") from exc + encoded = json.dumps(transaction, ensure_ascii=True, sort_keys=True, separators=(",", ":")).encode("utf-8") + if len(encoded) + 1 > status_transaction_journal_max_bytes(): + raise RuntimeError("status transaction journal exceeds its byte bound") + return transaction + + +def _rewrite_status_projection(path, key, replacement_line=None): + require_private_file(path) + with private_atomic_writer(path, suffix=".status.tmp") as output: + for line in iter_bounded_text_lines(path): + if _status_projection_line_matches_key(line, key): + continue + output.write(line if line.endswith("\n") else line + "\n") + if replacement_line is not None: + output.write(replacement_line) + + +def publish_status_transaction_target(transaction, status_files): + status_files = normalize_status_files(status_files) + _rewrite_status_projection( + status_files[transaction["status"]], + transaction["key"], + transaction["status_line"], + ) + + +def publish_status_transaction_checked(transaction, checked_file): + _rewrite_status_projection( + checked_file, + transaction["key"], + transaction["checked_line"], + ) + + +def remove_status_transaction_old_copies(transaction, status_files): + status_files = normalize_status_files(status_files) + target_path = os.path.normcase(os.path.abspath(status_files[transaction["status"]])) + visited = {target_path} + for _, path in sorted(status_files.items()): + normalized_path = os.path.normcase(os.path.abspath(path)) + if normalized_path in visited: + continue + visited.add(normalized_path) + _rewrite_status_projection(path, transaction["key"]) + + +def delete_status_transaction_journal(journal_path): + durable_unlink(journal_path) + + +def _apply_status_transaction(transaction, checked_file, status_files, journal_path): + transaction = _validate_status_transaction(transaction, status_files) + publish_status_transaction_target(transaction, status_files) + publish_status_transaction_checked(transaction, checked_file) + remove_status_transaction_old_copies(transaction, status_files) + delete_status_transaction_journal(journal_path) + + +def _recover_status_transaction_locked(checked_file, status_files, journal_path): + if not os.path.lexists(journal_path): + return False + transaction = read_private_json(journal_path, max_bytes=status_transaction_journal_max_bytes()) + _apply_status_transaction(transaction, checked_file, status_files, journal_path) + return True + + +def recover_status_transaction(checked_file, status_files): + if keycheck_input_mode() == 'postgres': + return False + journal_path = status_transaction_journal_path(checked_file) + if not os.path.lexists(journal_path): + return False + checked_file, status_files = _status_transaction_layout(checked_file, status_files) + journal_path = status_transaction_journal_path(checked_file) + lock_path = status_transaction_lock_path(checked_file) + timeout = min(120, max(1, env_int("KEYCHECK_STATUS_TRANSACTION_LOCK_TIMEOUT_SEC", 30))) + lock = acquire_file_lock(lock_path, timeout_sec=timeout) + try: + return _recover_status_transaction_locked(checked_file, status_files, journal_path) + finally: + release_file_lock(lock, lock_path) + + +def commit_status_transaction( + checked_file, + status_files, + key, + status, + message="", + extra="", + status_line=None, + checked_line=None, +): + if keycheck_input_mode() == 'postgres': + return 'postgres-authoritative' + checked_file, status_files = _status_transaction_layout(checked_file, status_files) + status = str(status or "UNKNOWN").strip().upper() + if status not in status_files: + status = "UNKNOWN" + if status not in status_files: + raise ValueError("status transaction has no UNKNOWN projection") + journal_path = status_transaction_journal_path(checked_file) + lock_path = status_transaction_lock_path(checked_file) + timeout = min(120, max(1, env_int("KEYCHECK_STATUS_TRANSACTION_LOCK_TIMEOUT_SEC", 30))) + lock = acquire_file_lock(lock_path, timeout_sec=timeout) + try: + _recover_status_transaction_locked(checked_file, status_files, journal_path) + transaction = _new_status_transaction( + key, + status, + status_files[status], + message, + extra, + status_line, + checked_line, + ) + encoded = json.dumps( + transaction, ensure_ascii=True, sort_keys=True, separators=(",", ":"), + ).encode("utf-8") + b"\n" + if len(encoded) > status_transaction_journal_max_bytes(): + raise ValueError("status transaction journal exceeds its byte bound") + with private_atomic_writer(journal_path, binary=True, suffix=".journal.tmp") as handle: + handle.write(encoded) + _apply_status_transaction(transaction, checked_file, status_files, journal_path) + return transaction["transaction_id"] + finally: + release_file_lock(lock, lock_path) + + +def load_known_statuses(checked_file, status_files): + if keycheck_input_mode() == 'postgres': + return {} + statuses = load_checked_statuses(checked_file) + for status, path in normalize_status_files(status_files).items(): + for key in load_keys_from_file(path): + statuses.setdefault(key, status) + return statuses + + +def cached_occurrence_identity(service, key, source, finding): + finding_hash = finding_secret_hash(finding) + if not finding_hash and isinstance(finding, dict): + finding_hash = sha256_text(json.dumps(finding, ensure_ascii=False, sort_keys=True, default=str)) + return sha256_text("|".join([str(service or ""), sha256_text(key), str(source or ""), finding_hash])) + + +def cached_occurrence_log_path(service=None): + service = service or keycheck_service_name() or "default" + output_dir = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(service) + require_private_directory(output_dir, create=True) + return os.path.join(output_dir, "cached_occurrences.tsv") + + +CACHED_OCCURRENCE_CACHE = {} +CACHED_OCCURRENCE_ROWS = {} +CACHED_OCCURRENCE_SIGNATURES = {} + + +def cached_occurrence_limits(): + return { + "items": max(1, env_int("KEYCHECK_CACHED_OCCURRENCE_MAX_ITEMS", 100000)), + "bytes": max(1, env_int("KEYCHECK_CACHED_OCCURRENCE_MAX_BYTES", 16 * 1024 * 1024)), + "ttl": max(60, env_int("KEYCHECK_CACHED_OCCURRENCE_TTL_SEC", 30 * 86400)), + } + + +def _cached_occurrence_timestamp(value): + try: + parsed = datetime.fromisoformat(str(value or "").replace("Z", "+00:00")) + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.timestamp() + except (TypeError, ValueError): + return 0.0 + + +def _cached_occurrence_signature(path): + try: + details = os.stat(path) + return (int(details.st_size), int(getattr(details, "st_mtime_ns", 0))) + except OSError: + return (0, 0) + + +def _load_bounded_cached_occurrences(path, limits): + if not os.path.exists(path): + return [], False + size = os.path.getsize(path) + start = max(0, size - limits["bytes"]) + rows = [] + truncated = start > 0 + cutoff = time.time() - limits["ttl"] + expired = False + with open(path, "rb") as handle: + handle.seek(start) + if start: + handle.readline(limits["bytes"] + 1) + payload = handle.read(limits["bytes"] + 1) + if len(payload) > limits["bytes"]: + payload = payload[-limits["bytes"]:] + first_newline = payload.find(b"\n") + payload = payload[first_newline + 1:] if first_newline >= 0 else b"" + truncated = True + for raw_line in payload.splitlines(): + try: + text = raw_line.decode("utf-8") + except UnicodeDecodeError: + truncated = True + continue + identity, separator, timestamp = text.partition("\t") + identity = identity.strip() + if not identity or not separator: + truncated = True + continue + epoch = _cached_occurrence_timestamp(timestamp.strip()) + if epoch < cutoff: + expired = True + continue + rows.append((identity, timestamp.strip(), epoch)) + deduped = {} + for identity, timestamp, epoch in rows: + deduped[identity] = (identity, timestamp, epoch) + retained = sorted(deduped.values(), key=lambda item: item[2])[-limits["items"]:] + return retained, truncated or expired or len(retained) != len(rows) + + +def _save_cached_occurrence_rows(path, rows, byte_limit): + encoded = [f"{identity}\t{timestamp}\n".encode("utf-8") for identity, timestamp, _ in rows] + total = sum(len(item) for item in encoded) + remove = 0 + while remove < len(encoded) and total > byte_limit: + total -= len(encoded[remove]) + remove += 1 + if remove: + encoded = encoded[remove:] + rows = rows[remove:] + with private_atomic_writer(path, binary=True) as handle: + for row in encoded: + handle.write(row) + return rows + + +def cached_occurrence_set(service=None): + service = service or keycheck_service_name() or "default" + path = cached_occurrence_log_path(service) + signature = _cached_occurrence_signature(path) + cached_rows = CACHED_OCCURRENCE_ROWS.get(service) or [] + cache_unexpired = not cached_rows or cached_rows[0][2] >= time.time() - cached_occurrence_limits()["ttl"] + if service in CACHED_OCCURRENCE_CACHE and CACHED_OCCURRENCE_SIGNATURES.get(service) == signature and cache_unexpired: + return CACHED_OCCURRENCE_CACHE[service] + limits = cached_occurrence_limits() + rows, needs_compaction = _load_bounded_cached_occurrences(path, limits) + if needs_compaction: + lock_path = f"{path}.lock" + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + rows, _ = _load_bounded_cached_occurrences(path, limits) + rows = _save_cached_occurrence_rows(path, rows, limits["bytes"]) + finally: + release_file_lock(lock, lock_path) + signature = _cached_occurrence_signature(path) + seen = {identity for identity, _, _ in rows} + CACHED_OCCURRENCE_CACHE[service] = seen + CACHED_OCCURRENCE_ROWS[service] = rows + CACHED_OCCURRENCE_SIGNATURES[service] = signature + return seen + + +def cached_occurrence_seen(identity, service=None): + return identity in cached_occurrence_set(service) + + +def mark_cached_occurrence_seen(identity, service=None): + service = service or keycheck_service_name() or "default" + path = cached_occurrence_log_path(service) + lock_path = f"{path}.lock" + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + limits = cached_occurrence_limits() + signature = _cached_occurrence_signature(path) + if CACHED_OCCURRENCE_SIGNATURES.get(service) == signature: + rows = list(CACHED_OCCURRENCE_ROWS.get(service) or []) + cutoff = time.time() - limits["ttl"] + retained = [row for row in rows if row[2] >= cutoff] + needs_compaction = len(retained) != len(rows) + rows = retained + else: + rows, needs_compaction = _load_bounded_cached_occurrences(path, limits) + seen = {value for value, _, _ in rows} + if identity in seen: + CACHED_OCCURRENCE_CACHE[service] = seen + CACHED_OCCURRENCE_ROWS[service] = rows + CACHED_OCCURRENCE_SIGNATURES[service] = signature + return False + timestamp = now_iso() + rows.append((str(identity), timestamp, time.time())) + line = f"{identity}\t{timestamp}\n".encode("utf-8") + current_size = signature[0] + if ( + needs_compaction + or len(rows) > limits["items"] + or current_size + len(line) > limits["bytes"] + ): + rows = rows[-limits["items"]:] + rows = _save_cached_occurrence_rows(path, rows, limits["bytes"]) + else: + with private_append_writer(path, binary=True) as handle: + handle.write(line) + seen = {value for value, _, _ in rows} + CACHED_OCCURRENCE_CACHE[service] = seen + CACHED_OCCURRENCE_ROWS[service] = rows + CACHED_OCCURRENCE_SIGNATURES[service] = _cached_occurrence_signature(path) + return True + finally: + release_file_lock(lock, lock_path) + + +def record_cached_keycheck_occurrence(service, key, status, source="", finding=None, detector=None): + if keycheck_input_mode() == 'postgres': + raise RuntimeError('PostgreSQL candidates cannot be completed from cached file state') + if not key or not service or not status: + return False + identity = cached_occurrence_identity(service, key, source, finding) + if cached_occurrence_seen(identity, service): + return False + finding = strip_finding_nearby_context(finding) + secret_hash = finding_secret_hash(finding) or sha256_text(key) + detector_secret_hash = finding_detector_secret_hash(finding) + result = { + "status": str(status).upper(), + "result_source": "cached_status", + "cached_status": True, + "cached_occurrence_id": identity, + "checked_at": now_iso(), + "message": "cached status occurrence; provider API not called", + "key_hash": sha256_text(key), + "secret_hash": secret_hash, + "detector_secret_hash": detector_secret_hash, + "finding_uid": finding_uid(finding), + } + output_dir = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(service) + require_private_directory(output_dir, create=True) + write_keycheck_event( + service, + os.path.join(output_dir, f"{service}Results.jsonl"), + key, + result, + source, + finding, + detector or get_detector_name(finding or {}) or "", + "cached_status", + ) + mark_cached_occurrence_seen(identity, service) + if not truthy_env("KEYCHECK_DB_INLINE", False): + return True + ok = record_validation_result(service, key, result, source, finding, detector) + return ok + + +def should_skip_key(key, checked_statuses, known_keys, args, retry_statuses=None, service=None, source="", finding=None, detector=None, known_statuses=None): + if keycheck_input_mode() == 'postgres': + if not _ACTIVE_DB_CANDIDATE: + raise RuntimeError('PostgreSQL provider probe has no active fenced candidate') + return False + attempted = getattr(args, "_keycheck_attempted_keys", None) + if not isinstance(attempted, set): + attempted = set() + setattr(args, "_keycheck_attempted_keys", attempted) + attempt_identity = (str(service or keycheck_service_name() or "default").lower(), key) + if attempt_identity in attempted: + if service and source: + cached_status = checked_statuses.get(key) or (known_statuses or {}).get(key) or "UNKNOWN" + record_cached_keycheck_occurrence(service, key, cached_status, source, finding, detector) + return True + + if getattr(args, "recheck_all", False): + attempted.add(attempt_identity) + return False + status = checked_statuses.get(key) or (known_statuses or {}).get(key) + retry_statuses = retry_statuses or set() + if status in retry_statuses: + attempted.add(attempt_identity) + return False + if status is None and retry_statuses and key in known_keys: + attempted.add(attempt_identity) + return False + skip = key in known_keys or key in checked_statuses + if skip and service and source: + cached_status = checked_statuses.get(key) or (known_statuses or {}).get(key) or "UNKNOWN" + record_cached_keycheck_occurrence(service, key, cached_status, source, finding, detector) + if not skip: + attempted.add(attempt_identity) + return skip + + +def load_proxies(proxy_file): + if not os.path.exists(proxy_file): + print(f"Info: {proxy_file} not found. Requests will go directly.") + return None + proxies = [] + with open(proxy_file, "r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + ip, port, login, password = line.split(":") + proxy_url = f"http://{login}:{password}@{ip}:{port}" + proxies.append({"http": proxy_url, "https": proxy_url}) + except ValueError: + print(f"Warning: bad proxy format: {line}. Skipping.") + if not proxies: + print(f"Warning: {proxy_file} is empty. Requests will go directly.") + return None + print(f"Loaded proxies: {len(proxies)}") + return cycle(proxies) + + +def get_detector_name(data): + if data.get("DetectorName"): + return data.get("DetectorName") + if data.get("detector"): + return data.get("detector") + finding = data.get("finding") + if isinstance(finding, dict): + return finding.get("DetectorName") + return None + + +def get_raw_values(data): + if data.get("DetectorName"): + return data.get("Raw"), data.get("RawV2"), data + if data.get("detector"): + finding = data.get("finding") if isinstance(data.get("finding"), dict) else data + return data.get("raw"), data.get("raw_v2"), finding + finding = data.get("finding") + if isinstance(finding, dict): + return finding.get("Raw"), finding.get("RawV2"), finding + return None, None, data + + +def iter_findings(input_file, detector_names): + if keycheck_input_mode() == 'postgres': + global _ACTIVE_DB_CANDIDATE + service = keycheck_service_name() + if not service: + raise RuntimeError('PostgreSQL keycheck input mode requires KEYCHECK_SERVICE') + owner = f'{service}:{os.getpid()}:{uuid.uuid4().hex}' + detector_names = set(detector_names) if detector_names is not None else None + slice_limit = max(0, env_int('KEYCHECK_PROVIDER_SLICE_KEYS', 0)) + yielded = 0 + try: + while True: + if slice_limit and yielded >= slice_limit: + return + candidate = _postgres_candidate_db().claim_keycheck_candidate( + service, owner, + lease_seconds=max(30, env_int('KEYCHECK_CANDIDATE_LEASE_SEC', 300)), + result_projection_reserve_bytes=max( + 3 * 1024 * 1024, + env_int('KEYCHECK_RESULT_PROJECTION_RESERVE_BYTES', 3 * 1024 * 1024), + ), + projection_max_items=max( + 1, env_int('KEYCHECK_PROJECTION_MAX_ITEMS', 10000), + ), + projection_max_bytes=max( + 3 * 1024 * 1024, + env_int('KEYCHECK_PROJECTION_MAX_BYTES', 2 * 1024 * 1024 * 1024), + ), + ) + if not candidate: + return + if candidate.get('_capacity_blocked'): + print( + 'PostgreSQL keycheck blocked: projection capacity is saturated', + flush=True, + ) + raise SystemExit(KEYCHECK_CAPACITY_BLOCKED_EXIT) + _ACTIVE_DB_CANDIDATE = candidate + yielded += 1 + finding = dict(candidate.get('finding') or {}) + candidate_metadata = safe_json_loads(candidate.get('metadata_json')) or {} + if not finding: + finding = { + 'DetectorName': candidate.get('detector_name') or candidate_metadata.get('detector_name'), + 'Raw': candidate.get('secret_text') or '', + 'RawV2': candidate_metadata.get('raw_v2') or candidate.get('secret_json') or '', + } + detector = finding.get('DetectorName') or candidate.get('detector_name') + candidate_kind = candidate.get('candidate_kind') or '' + structured_route = candidate_kind not in ('', 'provider_key') + provider_hint = str(candidate_metadata.get('provider_hint') or '').lower() + persisted_route = ( + provider_hint == service + or ( + service == 'provider_resolver' + and provider_hint in ( + AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT, + ) + ) + ) + if ( + detector_names is not None and detector not in detector_names + and not structured_route and not persisted_route + ): + _postgres_candidate_db().quarantine_keycheck_candidate( + candidate['id'], candidate['lease_token'], + 'candidate_provider_route_mismatch', + 'candidate detector does not match provider route', + ) + _ACTIVE_DB_CANDIDATE = None + continue + candidate_db = _postgres_candidate_db() + cached_completed = False + cached_state_retry_exhausted = False + for cached_attempt in range(2): + cached = candidate_db.keycheck_candidate_cached_status( + candidate['id'], candidate['lease_token'], + ) + if cached is None: + raise RuntimeError('PostgreSQL cached-status lookup lost its candidate fence') + if cached.get('probe_required'): + break + event_id = sha256_text('|'.join(( + 'truf-keycheck-cached-event-v1', str(candidate['id']), + ))) + cached_metadata = { + 'cached_status': True, + 'cached_state_version': cached['state_version'], + 'cached_result_id': cached['last_result_id'], + 'cached_checked_at': cached['checked_at'], + 'cached_result_source': cached['result_source'], + } + outcome = candidate_db.complete_keycheck_candidate( + candidate['id'], candidate['lease_token'], event_id, + cached['status'], cached['status_group'], checked_at=now_iso(), + message='cached current status occurrence; provider API not called', + metadata=cached_metadata, result_source='cached_status', + cached_state_version=cached['state_version'], + cached_last_result_id=cached['last_result_id'], + ) + if outcome and outcome.get('completed'): + candidate['_completed'] = True + _ACTIVE_DB_CANDIDATE = None + cached_completed = True + break + if outcome and outcome.get('cached_state_changed'): + cached_state_retry_exhausted = cached_attempt == 1 + continue + if outcome and outcome.get('probe_required'): + break + raise RuntimeError('PostgreSQL cached-status completion lost its candidate fence') + if cached_completed: + continue + if cached_state_retry_exhausted: + candidate_db.defer_keycheck_candidate( + candidate['id'], candidate['lease_token'], + 'current status changed during cached occurrence completion', 1, + ) + _ACTIVE_DB_CANDIDATE = None + continue + raw = finding.get('Raw') + raw_v2 = finding.get('RawV2') + if not raw and candidate.get('secret_text'): + raw = candidate['secret_text'] + if not raw_v2 and candidate.get('secret_json'): + raw_v2 = candidate['secret_json'] + yield { + 'line': candidate['id'], + 'detector': detector, + 'raw': raw or '', + 'raw_v2': raw_v2 or '', + 'finding': finding, + 'source': f'postgres:keycheck_candidates:{candidate["id"]}', + 'candidate_id': candidate['id'], + 'candidate_kind': candidate_kind, + 'credential_secret_text': candidate.get('secret_text') or '', + 'credential_secret_json': candidate.get('secret_json') or '', + 'credential_endpoint': candidate.get('endpoint') or '', + 'credential_principal': candidate.get('principal') or '', + 'candidate_metadata': candidate_metadata, + } + if not candidate.get('_completed'): + reason = 'provider iterator resumed without a committed result' + max_unconsumed = max( + 1, env_int('KEYCHECK_CANDIDATE_MAX_UNCONSUMED_ATTEMPTS', 3), + ) + if int(candidate.get('attempts') or 0) >= max_unconsumed: + _postgres_candidate_db().quarantine_keycheck_candidate( + candidate['id'], candidate['lease_token'], + 'provider_candidate_unconsumed', reason, + ) + else: + _postgres_candidate_db().defer_keycheck_candidate( + candidate['id'], candidate['lease_token'], reason, 60, + ) + _ACTIVE_DB_CANDIDATE = None + finally: + candidate = _ACTIVE_DB_CANDIDATE + if candidate and not candidate.get('_completed'): + try: + _postgres_candidate_db().defer_keycheck_candidate( + candidate['id'], candidate['lease_token'], + 'provider iterator closed without a committed result', 60, + ) + finally: + _ACTIVE_DB_CANDIDATE = None + return + detector_names = set(detector_names) if detector_names is not None else None + for item in iter_jsonl_input(input_file): + data = item.get("data") or {} + detector = get_detector_name(data) + if detector_names is not None and detector not in detector_names: + continue + raw, raw_v2, finding = get_raw_values(data) + yield { + "line": item.get("line_offset"), + "detector": detector, + "raw": raw or "", + "raw_v2": raw_v2 or "", + "finding": finding, + "source": item.get("source") or input_file, + } + + +def read_plain_keys(paths, regex): + if keycheck_input_mode() == 'postgres': + return + seen = set() + for path in paths or []: + if not os.path.exists(path): + continue + with open(path, "r", encoding="utf-8") as f: + content = f.read() + for match in regex.findall(content): + key = match if isinstance(match, str) else ":".join(match) + if key not in seen: + seen.add(key) + yield {"source": path, "key": key} + + +def request_error_message(response): + try: + payload = response.json() + except ValueError: + return response.text[:1000] + return json.dumps(payload, ensure_ascii=False)[:1000] + + +def classify_common_http_status(status_code): + if status_code in (401, 403): + return "DEAD" + if status_code == 429: + return "LIMITED" + if 500 <= status_code <= 599: + return "NETWORK" + return "UNKNOWN" + + +def requests_network_error(exc): + return isinstance(exc, requests.RequestException) diff --git a/app/keycheckers/kimi/kimiKeycheck.py b/app/keycheckers/kimi/kimiKeycheck.py new file mode 100644 index 0000000..93dcaaf --- /dev/null +++ b/app/keycheckers/kimi/kimiKeycheck.py @@ -0,0 +1,499 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import json +import os +import re +from urllib.parse import urlparse + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + combined_provider_routing_hint, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + keycheck_input_mode, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + provider_routing_database_failed, + read_plain_keys, + record_validation_result, + recover_status_transaction, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) +from keycheckers.provider_resolution import resolve_provider_key + + +SERVICE = "kimi" +DETECTOR = "KimiMoonshot" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() + +CHECKED_FILE = os.path.join(OUTPUT_DIR, "kimiChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "kimiResults.jsonl") +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "kimiAlive.txt"), + "NO_BALANCE": os.path.join(OUTPUT_DIR, "kimiNoBalance.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "kimiDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "kimiRestricted.txt"), + "LIMITED": os.path.join(OUTPUT_DIR, "kimiLimited.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "kimiNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "kimiUnknown.txt"), +} + +KIMI_DETECTOR_NAMES = {"kimimoonshot", "moonshotai", "moonshot", "kimi"} +KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"} +QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"} +DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"} +GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"} +AMBIGUOUS_QWEN_DEEPSEEK_HINT = "ambiguous_qwen_deepseek" +AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk" +EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment" +CANDIDATE_PROVIDER_ROUTE_FIELD = "_keycheck_candidate_provider_route" +KIMI_KEY_MAX_BYTES = 512 +KIMI_KEY_REGEX = re.compile( + r"(? KIMI_KEY_MAX_BYTES: + return f"candidate exceeds the {KIMI_KEY_MAX_BYTES}-byte key limit" + if value.count("sk-") != 1: + return "candidate contains multiple key prefixes" + if not KIMI_KEY_REGEX.fullmatch(value) or value.startswith(FOREIGN_KEY_PREFIXES): + return "candidate does not match the bounded Kimi/Moonshot key format" + return "" + + +def finding_provider_routing_hint(finding): + if not isinstance(finding, dict): + return "" + context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {} + persisted_hint = context.get("provider_hint") + if context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE and persisted_hint in ( + *GENERIC_SK_PROVIDERS, AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT, + ): + return persisted_hint + + parts = [str(context.get(key) or "") for key in ("nearby", "file")] + metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {} + data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {} + for details in data.values(): + if isinstance(details, dict): + parts.extend(str(details.get(key) or "") for key in ("file", "repository", "repo", "link", "image")) + + text = "\n".join(parts) + evidence = set() + if QWEN_CONTEXT_REGEX.search(text) or finding_has_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES): + evidence.add("qwen") + if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES): + evidence.add("deepseek") + if KIMI_CONTEXT_REGEX.search(text) or finding_has_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES): + evidence.add("kimi") + if persisted_hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT: + evidence.update(("qwen", "deepseek")) + elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT: + evidence.update(GENERIC_SK_PROVIDERS) + elif persisted_hint in GENERIC_SK_PROVIDERS: + evidence.add(persisted_hint) + if len(evidence) > 1: + return ( + AMBIGUOUS_QWEN_DEEPSEEK_HINT + if evidence == {"qwen", "deepseek"} + else AMBIGUOUS_GENERIC_SK_HINT + ) + return next(iter(evidence)) if evidence else "" + + +def iter_candidate_keys(input_file, plain_files, trusted_retry_files=None): + detector_names = [ + "KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi", + "kimimoonshot", "moonshotai", "moonshot", "kimi", "CustomRegex", + ] + routing_decisions = {} + seen = set() + for item in iter_findings(input_file, detector_names): + finding = dict(item.get("finding") or {}) + finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, None) + candidate_metadata = item.get("candidate_metadata") + persisted_route = "" + if keycheck_input_mode() == "postgres" and isinstance(candidate_metadata, dict): + persisted_route = str(candidate_metadata.get("provider_hint") or "").lower() + if persisted_route == SERVICE: + finding[CANDIDATE_PROVIDER_ROUTE_FIELD] = SERVICE + if persisted_route != SERVICE and not finding_has_detector(finding, KIMI_DETECTOR_NAMES): + continue + key = key_from_text(item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2")) + if not key or key_rejection_reason(key): + continue + if persisted_route == SERVICE: + hint = SERVICE + else: + local_hint = finding_provider_routing_hint(finding) + if key not in routing_decisions: + routing_decisions[key] = combined_provider_routing_hint(key, local_hint) + hint = routing_decisions[key] + if provider_routing_database_failed(): + raise RuntimeError("provider routing evidence lookup failed closed") + if hint == "kimi" or ( + keycheck_input_mode() == "postgres" + and hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT) + ): + seen.add(key) + yield key, item.get("source") or input_file, finding + + owned_retry_paths = { + os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values() + } + retry_files = [ + path for path in (trusted_retry_files or []) + if os.path.normcase(os.path.abspath(path)) in owned_retry_paths + ] + for item in read_plain_keys(retry_files, KIMI_KEY_REGEX): + key = item["key"] + if key_rejection_reason(key) or key in seen: + continue + hint = combined_provider_routing_hint(key, "kimi") + if provider_routing_database_failed(): + raise RuntimeError("provider routing evidence lookup failed closed") + if hint == "kimi": + seen.add(key) + yield key, item["source"], {} + + +def redact_text(value, key): + text = str(value or "")[:1000] + if key: + text = text.replace(key, "***REDACTED***") + return KIMI_KEY_REGEX.sub("***REDACTED***", text) + + +def parse_error(response, key): + try: + payload = response.json() + except ValueError: + payload = {} + error = payload.get("error") if isinstance(payload, dict) else {} + if not isinstance(error, dict): + error = {} + message = error.get("message") or response.text[:500] + return { + "http_status": response.status_code, + "code": error.get("code") or error.get("type") or "", + "message": redact_text(message, key), + } + + +def classify_error(error): + http_status = int(error.get("http_status") or 0) + code = str(error.get("code") or "").lower() + message = str(error.get("message") or "").lower() + if http_status == 401 or any(marker in code for marker in ("invalid_authentication", "invalid_api_key")): + return "DEAD" + if http_status == 403: + return "RESTRICTED" + if http_status == 429: + if any(marker in code + " " + message for marker in ("quota", "balance", "payment")): + return "NO_BALANCE" + return "LIMITED" + if 500 <= http_status <= 599: + return "NETWORK" + return "UNKNOWN" + + +def check_base_url(key, base_url, proxy, timeout, debug=False): + url = f"{normalize_base_url(base_url)}/users/me/balance" + headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"} + try: + response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return { + "status": "NETWORK", "base_url": base_url, + "region": endpoint_label(base_url), "message": redact_text(exc, key), + } + if debug: + print( + f" DEBUG {endpoint_label(base_url)} balance: HTTP {response.status_code}: " + f"{redact_text(response.text[:500], key)}" + ) + if response.status_code == 200: + try: + payload = response.json() + data = payload.get("data") if isinstance(payload, dict) else None + if not isinstance(data, dict) or "available_balance" not in data: + raise ValueError("missing data.available_balance") + available = float(data.get("available_balance")) + voucher = float(data.get("voucher_balance", 0) or 0) + cash = float(data.get("cash_balance", 0) or 0) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + return { + "status": "UNKNOWN", "base_url": base_url, + "region": endpoint_label(base_url), + "message": f"invalid balance response: {exc}", + } + status = "VALID" if available > 0 else "NO_BALANCE" + return { + "status": status, + "authenticated": True, + "base_url": base_url, + "region": endpoint_label(base_url), + "balance_usd": round(available, 6), + "voucher_balance_usd": round(voucher, 6), + "cash_balance_usd": round(cash, 6), + "message": f"available_balance=${available:.6f}", + } + error = parse_error(response, key) + return { + "status": classify_error(error), + "base_url": base_url, + "region": endpoint_label(base_url), + "http_status": response.status_code, + "error": error, + "message": error.get("message") or "", + } + + +def choose_final_status(attempts): + statuses = [attempt.get("status") for attempt in attempts] + for status in ("VALID", "NO_BALANCE", "LIMITED", "RESTRICTED", "UNKNOWN", "NETWORK"): + if status in statuses: + return status + return "DEAD" + + +def check_key(key, base_urls, proxy, timeout, debug=False): + rejection = key_rejection_reason(key) + if rejection: + return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True} + attempts = [] + for base_url in base_urls: + result = check_base_url(key, base_url, proxy, timeout, debug) + attempts.append(result) + if result.get("status") in ("VALID", "NO_BALANCE"): + return {**result, "attempts": attempts} + status = choose_final_status(attempts) + selected = next((attempt for attempt in attempts if attempt.get("status") == status), {}) + return {**selected, "status": status, "attempts": attempts} + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def write_result(key, result, source, finding): + status = result.get("status") or "UNKNOWN" + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR) + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, status, + result.get("message", ""), result.get("region") or source, + ) + record_validation_result(SERVICE, key, result, source, finding, DETECTOR) + + +def retry_statuses_from_args(args): + statuses = set() + if args.retry_network: + statuses.add("NETWORK") + if args.retry_limited: + statuses.add("LIMITED") + if args.retry_unknown: + statuses.add("UNKNOWN") + if args.retry_restricted: + statuses.add("RESTRICTED") + if args.retry_no_balance: + statuses.add("NO_BALANCE") + if args.retry_valid: + statuses.add("VALID") + return statuses + + +def retry_input_files_from_args(args): + if args.recheck_all: + statuses = list(STATUS_FILES) + else: + statuses = [ + status for flag, status in ( + (args.retry_network, "NETWORK"), + (args.retry_limited, "LIMITED"), + (args.retry_unknown, "UNKNOWN"), + (args.retry_restricted, "RESTRICTED"), + (args.retry_no_balance, "NO_BALANCE"), + (args.retry_valid, "VALID"), + ) if flag + ] + return [STATUS_FILES[status] for status in statuses] + + +def base_urls_from_args(args): + custom = [] + for value in args.base_url: + custom.extend(split_csv(value)) + custom.extend(split_csv(os.getenv("KIMI_BASE_URLS") or os.getenv("MOONSHOT_BASE_URLS"))) + defaults = [] if args.no_default_base_urls else DEFAULT_BASE_URLS + return unique_ordered([*custom, *defaults]) + + +def parse_args(): + parser = argparse.ArgumentParser(description="Kimi / Moonshot AI key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=15) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--base-url", action="append", default=[]) + parser.add_argument("--no-default-base-urls", action="store_true") + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--retry-no-balance", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + parser.add_argument("--debug", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = retry_statuses_from_args(args) + retry_files = retry_input_files_from_args(args) + base_urls = base_urls_from_args(args) + if not base_urls: + raise SystemExit("No Kimi/Moonshot base URLs configured") + + print("--- Kimi / Moonshot AI key checker ---") + print("base_urls: " + ", ".join(endpoint_label(url) for url in base_urls)) + processed = 0 + skipped = 0 + for key, source, finding in iter_candidate_keys(args.input, args.plain, retry_files): + finding = dict(finding or {}) + candidate_route = str(finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, "") or "").lower() + if should_skip_key( + key, checked, known, args, retry_statuses, service=SERVICE, + source=source, finding=finding, detector=DETECTOR, + ): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] Kimi/Moonshot candidate {mask_secret(key)} from {source}") + proxy = next(proxy_cycler) if proxy_cycler else None + routing_hint = "kimi" + if keycheck_input_mode() == "postgres": + if candidate_route == SERVICE: + routing_hint = SERVICE + else: + routing_hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding)) + if provider_routing_database_failed(): + raise RuntimeError("provider routing evidence lookup failed closed") + if routing_hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT): + result = resolve_provider_key( + key, finding, proxy, args.timeout, args.debug, + hint=routing_hint, origin_service=SERVICE, + ) + else: + result = check_key(key, base_urls, proxy, args.timeout, args.debug) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/openai/Keycheck.py b/app/keycheckers/openai/Keycheck.py new file mode 100644 index 0000000..4730c90 --- /dev/null +++ b/app/keycheckers/openai/Keycheck.py @@ -0,0 +1,572 @@ +import sys + +sys.dont_write_bytecode = True + +import requests +import json +import os +import argparse +import re +from itertools import cycle +import time + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files as ensure_private_output_files, + iter_findings, + iter_bounded_text_lines, + keycheck_input_mode, + load_known_statuses, + private_atomic_writer, + record_cached_keycheck_occurrence, + record_validation_result, + recover_status_transaction, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) + +try: + sys.stdout.reconfigure(encoding='utf-8', errors='replace') + sys.stderr.reconfigure(encoding='utf-8', errors='replace') +except Exception: + pass + +# --- Конфигурация --- +SERVICE = "openai" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +ALIVE_FILE = os.path.join(OUTPUT_DIR, "openaiAlive.txt") +DEAD_FILE = os.path.join(OUTPUT_DIR, "openaiDead.txt") +NETWORK_FILE = os.path.join(OUTPUT_DIR, "openaiNetwork.txt") +LIMITED_FILE = os.path.join(OUTPUT_DIR, "openaiLimited.txt") +RESTRICTED_FILE = os.path.join(OUTPUT_DIR, "openaiRestricted.txt") +UNKNOWN_FILE = os.path.join(OUTPUT_DIR, "openaiUnknown.txt") +NO_TARGET_FILE = os.path.join(OUTPUT_DIR, "openaiNoTarget.txt") +CHECKED_FILE = os.path.join(OUTPUT_DIR, "openaiChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "openaiResults.jsonl") +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() + +MODEL_PRIORITY_FOR_TEST = ( + 'gpt-5.6-sol', + 'gpt-5.6', + 'gpt-5.6-luna', + 'gpt-5.6-terra', + 'gpt-5', + 'o3-pro', + 'o3', +) +TARGET_MODELS = {'gpt-5', 'o3-pro', 'o3'} +NON_CHAT_MODEL_MARKERS = ( + 'embedding', 'image', 'audio', 'tts', 'transcribe', 'realtime', 'search', 'moderation', +) +PROBE_MAX_COMPLETION_TOKENS = 16 +STATUS_FILES = [ALIVE_FILE, DEAD_FILE, NETWORK_FILE, LIMITED_FILE, RESTRICTED_FILE, UNKNOWN_FILE, NO_TARGET_FILE] +STATUS_BY_FILE = { + ALIVE_FILE: 'ALIVE', + DEAD_FILE: 'DEAD', + NETWORK_FILE: 'NETWORK', + LIMITED_FILE: 'LIMITED', + RESTRICTED_FILE: 'RESTRICTED', + UNKNOWN_FILE: 'UNKNOWN', + NO_TARGET_FILE: 'NO_TARGET_MODELS', +} +OPENAI_KEY_REGEX = re.compile(r'sk-[A-Za-z0-9_-]{20,}') + +# --- Вспомогательные функции --- +def key_from_line(line): + line = line.strip() + if not line: + return None + if '\t' in line: + return line.split('\t', 1)[0].strip() + if ':[' in line: + return line.split(':[', 1)[0].strip() + match = OPENAI_KEY_REGEX.search(line) + if match: + return match.group(0) + return line.split()[0].strip() + +def load_set_from_file(filepath): + if keycheck_input_mode() == 'postgres': + return set() + if not os.path.exists(filepath): return set() + return {key for key in (key_from_line(line) for line in iter_bounded_text_lines(filepath)) if key} + + +def iter_plain_openai_keys(paths): + if keycheck_input_mode() == 'postgres': + return + seen = set() + items = [] + for path in paths or []: + if not path or not os.path.exists(path): + continue + for line in iter_bounded_text_lines(path): + key = key_from_line(line) + if not key or key in seen: + continue + seen.add(key) + items.append({'key': key, 'source': path, 'finding': {}}) + for item in items: + yield item + + +def retry_plain_files(args): + if keycheck_input_mode() == 'postgres': + return [] + files = list(args.plain or []) + if args.recheck_all: + files.extend(STATUS_FILES) + else: + if args.retry_limited: + files.append(LIMITED_FILE) + if args.retry_network: + files.append(NETWORK_FILE) + if args.retry_unknown: + files.append(UNKNOWN_FILE) + if args.retry_restricted: + files.append(RESTRICTED_FILE) + if args.retry_no_balance: + files.append(LIMITED_FILE) + out = [] + seen = set() + for path in files: + if path and path not in seen: + seen.add(path) + out.append(path) + return out + +def load_checked_statuses(): + if keycheck_input_mode() == 'postgres': + return {} + statuses = {} + if not os.path.exists(CHECKED_FILE): + return statuses + for line in iter_bounded_text_lines(CHECKED_FILE): + parts = line.rstrip('\n').split('\t') + if parts and parts[0]: + statuses[parts[0]] = parts[1] if len(parts) > 1 else 'UNKNOWN' + return statuses + +def load_known_keys(): + if keycheck_input_mode() == 'postgres': + return set() + known = set(load_checked_statuses().keys()) + for path in STATUS_FILES: + known.update(load_set_from_file(path)) + return known + +def ensure_output_files(): + ensure_private_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES]) + recover_status_transaction(CHECKED_FILE, STATUS_BY_FILE) + +def compact_status_file(path): + if keycheck_input_mode() == 'postgres': + return + if not os.path.exists(path): + return + last_by_key = {} + order = [] + for line in iter_bounded_text_lines(path): + key = key_from_line(line) + if not key: + continue + if key not in last_by_key: + order.append(key) + last_by_key[key] = line + for attempt in range(6): + try: + with private_atomic_writer(path) as f: + for key in order: + f.write(last_by_key[key]) + except PermissionError: + if attempt == 5: + print(f"Warning: unable to compact {path}; leaving existing file as-is") + return + time.sleep(0.1 * (attempt + 1)) + else: + return + +def compact_all_status_files(): + for path in [CHECKED_FILE, *STATUS_FILES]: + compact_status_file(path) + +def backfill_checked_file(): + if keycheck_input_mode() == 'postgres': + return + checked = load_checked_statuses() + changed = False + for path, status in STATUS_BY_FILE.items(): + for key in load_set_from_file(path): + if key not in checked: + checked[key] = status + changed = True + if not changed: + return + with private_atomic_writer(CHECKED_FILE) as f: + for key, status in sorted(checked.items()): + f.write(f"{key}\t{status}\tbackfilled\n") + +def load_proxies(proxy_file=None): + proxy_file = proxy_file or PROXY_FILE + if not os.path.exists(proxy_file): return None + proxies = [] + with open(proxy_file, 'r') as f: + for line in f: + line = line.strip() + if not line: continue + try: + ip, port, login, password = line.split(':') + proxy_url = f"http://{login}:{password}@{ip}:{port}" + proxies.append({"http": proxy_url, "https": proxy_url}) + except ValueError: + print(f"⚠️ Неверный формат прокси: '{line}'. Пропускаем.") + if not proxies: + print("⚠️ Файл proxy.txt пуст. Запросы будут идти напрямую.") + return None + print(f"✅ Загружено {len(proxies)} прокси.") + return cycle(proxies) + +def move_key_to_alive( + key, available_target_models, service_tier, source='', finding=None, + model_inventory=None, probe_model='', +): + """ + Перемещает ключ из DEAD_FILE в ALIVE_FILE, записывая модели и service_tier. + """ + models_str = ",".join(sorted(list(available_target_models))) + tier_str = str(service_tier or 'unknown').replace('\r', ' ').replace('\n', ' ')[:1000] + print(f" -> ✅ Ключ рабочий! Модели: {models_str}, Тир: {tier_str}. Перемещаем в {ALIVE_FILE}") + + model_inventory = sorted(set(model_inventory or available_target_models)) + result = { + 'status': 'ALIVE', + 'models': sorted(list(available_target_models)), + 'model_inventory': model_inventory, + 'model_count': len(model_inventory), + 'llm_probe_model': probe_model, + 'llm_probe_status': 'GENERATION_OK', + 'service_tier': service_tier, + 'message': f'chat ping ok; model={probe_model}; models={len(model_inventory)}', + } + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, 'OpenAI') + commit_status_transaction( + CHECKED_FILE, + STATUS_BY_FILE, + key, + 'ALIVE', + status_line=f"{key}:[{models_str}]:{tier_str}\n", + checked_line=f"{key}\tALIVE\t{time.strftime('%Y-%m-%dT%H:%M:%S')}\n", + ) + record_validation_result(SERVICE, key, result, source, finding, 'OpenAI') + +def redact_message(message, key): + message = str(message).replace('\r', ' ').replace('\n', ' ')[:1000] + if key: + message = message.replace(key, '***REDACTED***') + return OPENAI_KEY_REGEX.sub('***REDACTED***', message) + +def write_key_status(key, path, status, message='', source='', finding=None, metadata=None): + status_upper = status.upper() + projection_status = STATUS_BY_FILE.get(path) + if not projection_status: + raise ValueError(f'unknown OpenAI status projection: {path}') + message = redact_message(message, key) + result = {**(metadata or {}), 'status': status_upper, 'message': message} + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, 'OpenAI') + commit_status_transaction( + CHECKED_FILE, + STATUS_BY_FILE, + key, + projection_status, + message, + source, + status_line=f"{key}\t{status}\t{message}\n", + checked_line=f"{key}\t{projection_status}\t{time.strftime('%Y-%m-%dT%H:%M:%S')}\n", + ) + record_validation_result(SERVICE, key, result, source, finding, 'OpenAI') + +def extract_openai_key(data): + if data.get("DetectorName") == "OpenAI": + return data.get("Raw") or data.get("RawV2") + if data.get("detector") == "OpenAI": + return data.get("raw") or data.get("raw_v2") + finding = data.get("finding") + if isinstance(finding, dict) and finding.get("DetectorName") == "OpenAI": + return finding.get("Raw") or finding.get("RawV2") + return None + +# --- Функции проверки --- +def check_authentication(key, proxy): + print(f" [1/2] Проверка аутентификации...") + url = "https://api.openai.com/v1/models" + headers = {"Authorization": f"Bearer {key}"} + try: + response = requests.get(url, headers=headers, proxies=proxy, timeout=15) + if response.status_code == 200: + print(" -> Аутентификация пройдена.") + return 'valid', response.json().get('data', []) + elif response.status_code == 401: + print(" -> Ошибка 401: Ключ недействителен или отозван.") + return 'dead', response.text + elif response.status_code == 403: + print(f" -> Ошибка 403: ключ ограничен/заблокирован: {response.text[:300]}") + return 'restricted', response.text + elif response.status_code == 429: + print(f" -> Ошибка 429: rate limit / quota: {response.text[:300]}") + return 'limited', response.text + else: + print(f" -> Ошибка {response.status_code}: {response.text}") + return 'unknown', response.text + except requests.RequestException as e: + print(f" -> Ошибка сети: {e}") + return 'network', str(e) + +def reportable_target_models(model_ids): + return { + model for model in model_ids + if model in TARGET_MODELS or model.startswith('gpt-5.6-') or model == 'gpt-5.6' + } + + +def choose_probe_model(model_ids): + models = [str(model or '') for model in model_ids if model] + by_lower = {model.lower(): model for model in models} + for model in MODEL_PRIORITY_FOR_TEST: + if model.lower() in by_lower: + return by_lower[model.lower()] + for model in models: + lowered = model.lower() + if lowered.startswith(('gpt-', 'o')) and not any( + marker in lowered for marker in NON_CHAT_MODEL_MARKERS + ): + return model + return '' + + +def check_balance_and_tier(key, model_to_test, proxy): + """ + Проверяет баланс и возвращает service_tier в случае успеха. + """ + print(f" [2/2] Проверка баланса и тира...") + if not model_to_test: + print(" -> Не найдено подходящих моделей для теста баланса.") + return 'unknown', 'no chat-capable model from /models' + + print(f" -> Используем модель для теста: {model_to_test}") + url = "https://api.openai.com/v1/chat/completions" + headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"} + payload = { + "model": model_to_test, + "messages": [{"role": "user", "content": "Reply with one digit."}], + "max_completion_tokens": PROBE_MAX_COMPLETION_TOKENS, + } + + try: + response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=20) + if response.status_code == 200: + # Успех, извлекаем service_tier + response_data = response.json() + service_tier = response_data.get('service_tier') + return 'ok', service_tier + elif response.status_code == 429: + print(f" -> Ошибка 429: Нет баланса или превышен лимит.") + return 'limited', response.text + elif response.status_code == 401: + print(f" -> Ошибка 401: ключ недействителен или отозван.") + return 'dead', response.text + elif response.status_code == 403: + print(f" -> Ошибка 403: ключ ограничен/заблокирован: {response.text[:300]}") + return 'restricted', response.text + else: + print(f" -> Ошибка {response.status_code}: {response.text}") + return 'unknown', response.text + except requests.RequestException as e: + print(f" -> Ошибка сети: {e}") + return 'network', str(e) + +# --- Основной процесс --- +def parse_args(): + parser = argparse.ArgumentParser(description='OpenAI key checker') + parser.add_argument('--input', default=INPUT_FILE) + parser.add_argument('--plain', action='append', default=[]) + parser.add_argument('--proxy-file', default=PROXY_FILE) + parser.add_argument('--max-keys', type=int, default=0) + parser.add_argument('--retry-network', action='store_true') + parser.add_argument('--retry-limited', action='store_true') + parser.add_argument('--retry-unknown', action='store_true') + parser.add_argument('--retry-restricted', action='store_true') + parser.add_argument('--retry-no-balance', action='store_true') + parser.add_argument('--recheck-all', action='store_true') + return parser.parse_args() + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_output_files() + compact_all_status_files() + backfill_checked_file() + print("--- 🚀 Запуск чекера ключей OpenAI 🚀 ---") + proxy_cycler = load_proxies(args.proxy_file) + checked_statuses = load_checked_statuses() + known_statuses = load_known_statuses(CHECKED_FILE, STATUS_BY_FILE) + known_keys = set(known_statuses) + alive_keys = load_set_from_file(ALIVE_FILE) + retry_statuses = set() + if args.retry_network: + retry_statuses.add('NETWORK') + if args.retry_limited: + retry_statuses.update(('LIMITED', 'LIMITED_OR_NO_BALANCE', 'LIMITED_OR_QUOTA')) + if args.retry_unknown: + retry_statuses.add('UNKNOWN') + if args.retry_restricted: + retry_statuses.add('RESTRICTED') + if args.retry_no_balance: + retry_statuses.update(('NO_BALANCE', 'NO_QUOTA', 'LIMITED_OR_NO_BALANCE', 'LIMITED_OR_QUOTA')) + print(f"📖 Загружено: {len(alive_keys)} живых ключей, {len(known_keys)} уже классифицированных ключей, {len(checked_statuses)} checked.") + + if keycheck_input_mode() == 'jsonl' and not os.path.exists(args.input): + print(f"❌ Файл с секретами {args.input} не найден. Завершение.") + return + + processed = 0 + skipped = 0 + seen_this_run = set() + def candidates(): + for item in iter_findings(args.input, ['OpenAI']): + data = item.get('finding') or {} + key = item.get('raw') or extract_openai_key(data) + if key: + yield {'key': key, 'source': item.get('source') or args.input, 'finding': data} + seen_plain = set() + for item in iter_plain_openai_keys(retry_plain_files(args)): + key = item.get('key') + if key and key not in seen_plain: + seen_plain.add(key) + yield item + + for item in candidates(): + data = item.get('finding') or {} + source_line = item.get('source') or args.input + key = item.get('key') + if not key: + continue + if keycheck_input_mode() != 'postgres' and key in seen_this_run: + cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN' + record_cached_keycheck_occurrence(SERVICE, key, cached_status, source_line, data, 'OpenAI') + skipped += 1 + continue + seen_this_run.add(key) + if should_skip_key( + key, checked_statuses, known_keys, args, retry_statuses, + service=SERVICE, source=source_line, finding=data, detector='OpenAI', + known_statuses=known_statuses, + ): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + + print(f"\n[{processed}] 🎯 Новый кандидат: {key[:8]}...{key[-4:]} from {source_line}") + + current_proxy = next(proxy_cycler) if proxy_cycler else None + + auth_status, auth_data = check_authentication(key, current_proxy) + + if auth_status == 'valid' and auth_data: + all_available_models_data = auth_data + all_model_ids = sorted({ + str(model.get('id') or '') for model in all_available_models_data + if isinstance(model, dict) and model.get('id') + }) + found_target_models = reportable_target_models(all_model_ids) + model_to_test = choose_probe_model(all_model_ids) + + if not model_to_test: + print(" -> Ключ валиден, но не имеет подходящей chat-модели. Пропускаем.") + write_key_status( + key, NO_TARGET_FILE, 'no_target_models', ','.join(all_model_ids)[:500], + source_line, data, { + 'models': sorted(found_target_models), + 'model_inventory': all_model_ids, + 'model_count': len(all_model_ids), + 'llm_probe_status': 'NO_CONTEXT', + }, + ) + known_keys.add(key) + checked_statuses[key] = 'NO_TARGET_MODELS' + continue + + balance_status, balance_data = check_balance_and_tier(key, model_to_test, current_proxy) + probe_metadata = { + 'models': sorted(found_target_models or {model_to_test}), + 'model_inventory': all_model_ids, + 'model_count': len(all_model_ids), + 'llm_probe_model': model_to_test, + 'llm_probe_status': { + 'ok': 'GENERATION_OK', + 'limited': 'LIMITED', + 'network': 'NETWORK', + 'restricted': 'RESTRICTED', + 'dead': 'DEAD', + }.get(balance_status, 'UNKNOWN'), + } + + # Проверяем, что результат не None (успешная проверка баланса) + if balance_status == 'ok': + move_key_to_alive( + key, found_target_models or {model_to_test}, balance_data, source_line, data, + model_inventory=all_model_ids, probe_model=model_to_test, + ) + alive_keys.add(key) + final_status = 'ALIVE' + elif balance_status == 'limited': + write_key_status(key, LIMITED_FILE, 'limited_or_no_balance', balance_data, source_line, data, probe_metadata) + final_status = 'LIMITED_OR_NO_BALANCE' + elif balance_status == 'network': + write_key_status(key, NETWORK_FILE, 'network_error', balance_data, source_line, data, probe_metadata) + final_status = 'NETWORK' + elif balance_status == 'restricted': + write_key_status(key, RESTRICTED_FILE, 'restricted', balance_data, source_line, data, probe_metadata) + final_status = 'RESTRICTED' + elif balance_status == 'dead': + write_key_status(key, DEAD_FILE, 'invalid_or_revoked', balance_data, source_line, data, probe_metadata) + final_status = 'DEAD' + else: + write_key_status(key, UNKNOWN_FILE, 'unknown', balance_data, source_line, data, probe_metadata) + final_status = 'UNKNOWN' + known_keys.add(key) + checked_statuses[key] = final_status + elif auth_status == 'network': + write_key_status(key, NETWORK_FILE, 'network_error', auth_data, source_line, data) + known_keys.add(key) + checked_statuses[key] = 'NETWORK' + elif auth_status == 'limited': + write_key_status(key, LIMITED_FILE, 'limited_or_quota', auth_data, source_line, data) + known_keys.add(key) + checked_statuses[key] = 'LIMITED' + elif auth_status == 'restricted': + write_key_status(key, RESTRICTED_FILE, 'restricted', auth_data, source_line, data) + known_keys.add(key) + checked_statuses[key] = 'RESTRICTED' + elif auth_status == 'dead': + write_key_status(key, DEAD_FILE, 'invalid_or_revoked', auth_data, source_line, data) + known_keys.add(key) + checked_statuses[key] = 'DEAD' + else: + write_key_status(key, UNKNOWN_FILE, 'unknown', auth_data, source_line, data) + known_keys.add(key) + checked_statuses[key] = 'UNKNOWN' + + print("\n--- ✅ Проверка завершена. ---") + print(f"Processed={processed}, skipped={skipped}") + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/openrouter/OpenrouterKeycheck.py b/app/keycheckers/openrouter/OpenrouterKeycheck.py new file mode 100644 index 0000000..3b46e33 --- /dev/null +++ b/app/keycheckers/openrouter/OpenrouterKeycheck.py @@ -0,0 +1,498 @@ +import sys + +sys.dont_write_bytecode = True + +import requests +import json +import os +import argparse +from itertools import cycle +from datetime import datetime, timezone + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +try: + sys.stdout.reconfigure(encoding='utf-8', errors='replace') + sys.stderr.reconfigure(encoding='utf-8', errors='replace') +except Exception: + pass + +from keycheck_common import ( + acquire_file_lock, + append_checked, + append_jsonl, + commit_status_transaction, + default_input_file, + default_proxy_file, + env_int, + ensure_output_files as ensure_private_output_files, + finding_detector_secret_hash, + iter_findings, + iter_bounded_text_lines, + keycheck_input_mode, + load_known_statuses, + load_checked_statuses, + mask_secret, + now_iso, + physical_jsonl_segments, + private_append_writer, + private_atomic_writer, + reconcile_keycheck_jsonl_segments, + record_validation_result, + recover_status_transaction, + release_file_lock, + repair_keycheck_jsonl_tail, + require_provider_authority, + rotate_jsonl_if_needed, + service_output_dir, + should_skip_key, + sha256_text, + write_keycheck_event, +) +from runtime_security import reject_reparse_components, require_private_file + +# --- Конфигурация --- +SERVICE = "openrouter" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +ALIVE_FILE = os.path.join(OUTPUT_DIR, "openrouterAlive.txt") +DEAD_FILE = os.path.join(OUTPUT_DIR, "openrouterDead.txt") +LIMITED_FILE = os.path.join(OUTPUT_DIR, "openrouterLimited.txt") +NO_BALANCE_FILE = os.path.join(OUTPUT_DIR, "openrouterNoBalance.txt") +NETWORK_FILE = os.path.join(OUTPUT_DIR, "openrouterNetwork.txt") +UNKNOWN_FILE = os.path.join(OUTPUT_DIR, "openrouterUnknown.txt") +CHECKED_FILE = os.path.join(OUTPUT_DIR, "openrouterChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "openrouterResults.jsonl") +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +STATUS_FILES = { + "VALID": ALIVE_FILE, + "NO_BALANCE": NO_BALANCE_FILE, + "DEAD": DEAD_FILE, + "LIMITED": LIMITED_FILE, + "NETWORK": NETWORK_FILE, + "UNKNOWN": UNKNOWN_FILE, +} + +CREDITS_URL = "https://openrouter.ai/api/v1/credits" + +# --- Вспомогательные функции --- +def load_set_from_file(filepath): + """Загружает ключи из файла в set для быстрой проверки.""" + if not os.path.exists(filepath): + return set() + return {line.strip().split(':')[0] for line in iter_bounded_text_lines(filepath) if line.strip()} + +def ensure_output_files(): + ensure_private_output_files((CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values())) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def legacy_status_key(line): + value = line.strip() + if not value: + return None + if '\t' in value: + return value.split('\t', 1)[0].strip() + return value.split(':', 1)[0].strip() + + +def load_openrouter_keys(path): + if keycheck_input_mode() == 'postgres': + return set() + if not os.path.exists(path): + return set() + return {key for key in (legacy_status_key(line) for line in iter_bounded_text_lines(path)) if key} + + +def iter_plain_openrouter_keys(paths): + if keycheck_input_mode() == 'postgres': + return + seen = set() + items = [] + for path in paths or []: + if not path or not os.path.exists(path): + continue + for line in iter_bounded_text_lines(path): + key = legacy_status_key(line) + if not key or key in seen: + continue + seen.add(key) + items.append({'raw': key, 'source': path, 'finding': {}}) + for item in items: + yield item + + +def retry_plain_files(args): + if keycheck_input_mode() == 'postgres': + return [] + files = list(args.plain or []) + if args.recheck_all: + files.extend(STATUS_FILES.values()) + else: + if args.retry_valid: + files.append(ALIVE_FILE) + if args.retry_no_balance: + files.append(NO_BALANCE_FILE) + if args.retry_limited: + files.append(LIMITED_FILE) + if args.retry_network: + files.append(NETWORK_FILE) + if args.retry_unknown: + files.append(UNKNOWN_FILE) + out = [] + seen = set() + for path in files: + if path and path not in seen: + seen.add(path) + out.append(path) + return out + + +def migrate_legacy_checked(): + if keycheck_input_mode() == 'postgres': + return + checked = load_checked_statuses(CHECKED_FILE) + for key in sorted(load_openrouter_keys(ALIVE_FILE)): + if key not in checked: + write_keycheck_event(SERVICE, RESULTS_FILE, key, {'status': 'VALID', 'message': 'legacy alive status migration'}, 'legacy:openrouterAlive.txt', {}, 'OpenRouter', 'legacy_status') + append_checked(CHECKED_FILE, key, 'VALID') + checked[key] = 'VALID' + for key in sorted(load_openrouter_keys(DEAD_FILE)): + if key not in checked: + write_keycheck_event(SERVICE, RESULTS_FILE, key, {'status': 'DEAD', 'message': 'legacy dead status migration'}, 'legacy:openrouterDead.txt', {}, 'OpenRouter', 'legacy_status') + append_checked(CHECKED_FILE, key, 'DEAD') + checked[key] = 'DEAD' + + +def _legacy_alive_checked_at(path): + details = os.stat(path, follow_symlinks=False) + return datetime.fromtimestamp(details.st_mtime, timezone.utc).isoformat(timespec='seconds') + + +def _legacy_balance_event_payload(key, balance, checked_at): + balance_text = f'{balance:.6f}' + key_hash = sha256_text(key) + event_id = sha256_text('|'.join([ + SERVICE, + 'legacy_status', + 'openrouterAlive.txt', + key_hash, + 'NO_BALANCE', + balance_text, + ])) + return { + 'key_masked': mask_secret(key), + 'key_hash': key_hash, + 'secret_hash': key_hash, + 'detector_secret_hash': finding_detector_secret_hash({}), + 'finding_uid': '', + 'detector': 'OpenRouter', + 'source': 'legacy:openrouterAlive.txt', + 'finding': {}, + 'checked_at': checked_at, + 'result_source': 'legacy_status', + 'status': 'NO_BALANCE', + 'message': f'credits={balance_text}', + 'event_id': event_id, + } + + +def _publish_legacy_no_balance(moved): + lock_path = f'{NO_BALANCE_FILE}.lock' + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + require_private_file(NO_BALANCE_FILE) + existing = load_openrouter_keys(NO_BALANCE_FILE) + with private_append_writer(NO_BALANCE_FILE) as handle: + for key, balance in moved: + if key in existing: + continue + balance_text = f'{balance:.6f}' + handle.write(f'{key}\tNO_BALANCE\tcredits={balance_text}\tmigrated_from_alive\n') + existing.add(key) + finally: + release_file_lock(lock, lock_path) + + +def _existing_legacy_event_ids(expected): + found = set() + max_line_bytes = max(1024, env_int('KEYCHECK_INPUT_MAX_LINE_BYTES', 16 * 1024 * 1024)) + max_file_bytes = max( + max_line_bytes, + max(1, env_int('KEYCHECK_INPUT_LIST_MAX_BYTES', 32 * 1024 * 1024)), + max(0, env_int('KEYCHECK_RESULTS_MAX_MB', 32)) * 1024 * 1024 + max_line_bytes, + ) + paths = [path for _, path in physical_jsonl_segments(RESULTS_FILE)] + if os.path.isfile(RESULTS_FILE): + paths.append(os.path.abspath(RESULTS_FILE)) + for path in paths: + reject_reparse_components(path) + if os.path.getsize(path) > max_file_bytes: + raise RuntimeError(f'OpenRouter result file exceeds the bounded migration scan size: {path}') + with open(path, 'rb') as handle: + while True: + raw_line = handle.readline(max_line_bytes + 1) + if not raw_line: + break + if len(raw_line) > max_line_bytes: + raise RuntimeError(f'OpenRouter result line exceeds the bounded migration scan size: {path}') + if not raw_line.endswith(b'\n'): + raise RuntimeError(f'torn OpenRouter result line during legacy migration: {path}') + try: + payload = json.loads(raw_line.decode('utf-8', errors='replace')) + except (TypeError, ValueError): + continue + event_id = str(payload.get('event_id') or '') if isinstance(payload, dict) else '' + if event_id not in expected: + continue + wanted = expected[event_id] + for field in ('key_hash', 'status', 'source', 'result_source'): + if str(payload.get(field) or '') != str(wanted.get(field) or ''): + raise RuntimeError(f'conflicting OpenRouter legacy migration event: {event_id}') + found.add(event_id) + return found + + +def _publish_legacy_balance_events(moved, checked_at): + payloads = [_legacy_balance_event_payload(key, balance, checked_at) for key, balance in moved] + expected = {payload['event_id']: payload for payload in payloads} + lock_path = f'{RESULTS_FILE}.lock' + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + require_private_file(RESULTS_FILE) + repair_keycheck_jsonl_tail(RESULTS_FILE) + reconcile_keycheck_jsonl_segments(RESULTS_FILE) + existing = _existing_legacy_event_ids(expected) + max_bytes = max(0, env_int('KEYCHECK_RESULTS_MAX_MB', 32)) * 1024 * 1024 + for payload in payloads: + event_id = payload['event_id'] + if event_id in existing: + continue + rotate_jsonl_if_needed(RESULTS_FILE, max_bytes) + with private_append_writer(RESULTS_FILE) as handle: + handle.write(json.dumps(payload, ensure_ascii=False, default=str) + '\n') + existing.add(event_id) + finally: + release_file_lock(lock, lock_path) + + +def _publish_legacy_checked(moved, checked_at): + lock_path = f'{CHECKED_FILE}.lock' + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + require_private_file(CHECKED_FILE) + existing = set() + for line in iter_bounded_text_lines(CHECKED_FILE): + parts = line.rstrip('\r\n').split('\t') + if len(parts) >= 2: + existing.add((parts[0], parts[1])) + with private_append_writer(CHECKED_FILE) as handle: + for key, _ in moved: + identity = (key, 'NO_BALANCE') + if identity in existing: + continue + handle.write(f'{key}\tNO_BALANCE\t{checked_at}\n') + existing.add(identity) + finally: + release_file_lock(lock, lock_path) + + +def _rewrite_legacy_alive(keep): + with private_atomic_writer(ALIVE_FILE, binary=True, suffix='.legacy.tmp') as handle: + for line in keep: + handle.write(line.encode('utf-8')) + + +def migrate_legacy_alive_balances(): + if keycheck_input_mode() == 'postgres': + return + lock_path = f'{ALIVE_FILE}.lock' + lock = acquire_file_lock(lock_path, timeout_sec=30) + try: + if not os.path.exists(ALIVE_FILE): + return + require_private_file(ALIVE_FILE) + checked_at = _legacy_alive_checked_at(ALIVE_FILE) + keep = [] + moved = [] + seen = set() + for line in iter_bounded_text_lines(ALIVE_FILE): + text = line.strip() + if not text: + keep.append(line) + continue + key = legacy_status_key(text) + balance = None + if ':' in text and '\t' not in text: + try: + balance = float(text.rsplit(':', 1)[1]) + except ValueError: + balance = None + if key and balance is not None and balance <= 0: + if key not in seen: + moved.append((key, balance)) + seen.add(key) + else: + keep.append(line) + if not moved: + return + _publish_legacy_no_balance(moved) + _publish_legacy_balance_events(moved, checked_at) + _publish_legacy_checked(moved, checked_at) + _rewrite_legacy_alive(keep) + finally: + release_file_lock(lock, lock_path) + + +def load_proxies(proxy_file=None): + """Загружает и подготавливает прокси.""" + proxy_file = proxy_file or PROXY_FILE + if not os.path.exists(proxy_file): + print("ℹ️ Файл proxy.txt не найден, запросы будут идти напрямую.") + return None + proxies = [] + with open(proxy_file, 'r') as f: + for line in f: + line = line.strip() + if not line: continue + try: + ip, port, login, password = line.split(':') + proxy_url = f"http://{login}:{password}@{ip}:{port}" + proxies.append({"http": proxy_url, "https": proxy_url}) + except ValueError: + print(f"⚠️ Неверный формат прокси: '{line}'. Пропускаем.") + if not proxies: + print("⚠️ Файл proxy.txt пуст. Запросы будут идти напрямую.") + return None + print(f"✅ Загружено {len(proxies)} прокси.") + return cycle(proxies) + +def write_result(key, result, source_line, finding=None, previous_status=None): + status = result.get('status') or 'UNKNOWN' + credits = result.get('remaining_credits') + extra = f"credits={credits:.6f}" if isinstance(credits, (int, float)) else source_line + checked_at = now_iso() + result = {**result, 'checked_at': result.get('checked_at') or checked_at} + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source_line, finding, 'OpenRouter') + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, status, result.get('message', ''), extra, + ) + record_validation_result(SERVICE, key, result, source_line, finding, 'OpenRouter') + +# --- Функция проверки --- +def check_openrouter_key(key, proxy): + """Проверяет один ключ OpenRouter и возвращает normalized result.""" + headers = {"Authorization": f"Bearer {key}"} + try: + resp = requests.get(CREDITS_URL, headers=headers, proxies=proxy, timeout=15) + except requests.exceptions.RequestException as e: + return {'status': 'NETWORK', 'message': str(e)} + + if resp.status_code == 200: + try: + data = resp.json().get("data", {}) + total = float(data.get("total_credits", 0.0) or 0.0) + used = float(data.get("total_usage", 0.0) or 0.0) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + return {'status': 'UNKNOWN', 'http_status': resp.status_code, 'message': f'invalid credits response: {exc}'} + remaining = total - used + if remaining > 0: + return {'status': 'VALID', 'remaining_credits': remaining, 'message': f'credits={remaining:.6f}'} + return {'status': 'NO_BALANCE', 'remaining_credits': remaining, 'message': f'credits={remaining:.6f}'} + if resp.status_code in (401, 403): + return {'status': 'DEAD', 'http_status': resp.status_code, 'message': resp.text[:1000]} + if resp.status_code == 429: + return {'status': 'LIMITED', 'http_status': resp.status_code, 'message': resp.text[:1000]} + if 500 <= resp.status_code <= 599: + return {'status': 'NETWORK', 'http_status': resp.status_code, 'message': resp.text[:1000]} + return {'status': 'UNKNOWN', 'http_status': resp.status_code, 'message': resp.text[:1000]} + +def parse_args(): + parser = argparse.ArgumentParser(description='OpenRouter key checker') + parser.add_argument('--input', default=INPUT_FILE) + parser.add_argument('--plain', action='append', default=[]) + parser.add_argument('--proxy-file', default=PROXY_FILE) + parser.add_argument('--max-keys', type=int, default=0) + parser.add_argument('--retry-network', action='store_true') + parser.add_argument('--retry-limited', action='store_true') + parser.add_argument('--retry-unknown', action='store_true') + parser.add_argument('--retry-no-balance', action='store_true') + parser.add_argument('--retry-valid', action='store_true') + parser.add_argument('--recheck-all', action='store_true') + return parser.parse_args() + + +# --- Основной процесс --- +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_output_files() + migrate_legacy_alive_balances() + migrate_legacy_checked() + print("--- 🚀 Запуск чекера ключей OpenRouter 🚀 ---") + proxy_cycler = load_proxies(args.proxy_file) + + checked_statuses = load_checked_statuses(CHECKED_FILE) + known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES) + known_keys = set(known_statuses) + for path in STATUS_FILES.values(): + known_keys.update(load_openrouter_keys(path)) + retry_statuses = set() + if args.retry_network: + retry_statuses.add('NETWORK') + if args.retry_limited: + retry_statuses.add('LIMITED') + if args.retry_unknown: + retry_statuses.add('UNKNOWN') + if args.retry_no_balance: + retry_statuses.add('NO_BALANCE') + if args.retry_valid: + retry_statuses.add('VALID') + print(f"📖 Загружено: {len(load_openrouter_keys(ALIVE_FILE))} живых ключей, {len(known_keys)} классифицированных ключей.") + + if keycheck_input_mode() == 'jsonl' and not os.path.exists(args.input): + print(f"❌ Файл с секретами {args.input} не найден. Завершение.") + return + + processed = 0 + def candidates(): + for item in iter_findings(args.input, ["OpenRouter"]): + key = item.get("raw") or "" + if key: + yield item + seen = set() + for item in iter_plain_openrouter_keys(retry_plain_files(args)): + key = item.get("raw") or "" + if key and key not in seen: + seen.add(key) + yield item + + for item in candidates(): + key = item.get("raw") or "" + if not key: + continue + + source = item.get("source") or args.input + finding = item.get("finding") or {} + if should_skip_key( + key, checked_statuses, known_keys, args, retry_statuses, + service=SERVICE, source=source, finding=finding, detector='OpenRouter', known_statuses=known_statuses, + ): + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + + print(f"\n[{processed}] 🎯 Новый кандидат: {key[:8]}...{key[-4:]} from {source}") + current_proxy = next(proxy_cycler) if proxy_cycler else None + result = check_openrouter_key(key, current_proxy) + print(f" STATUS: {result.get('status')} | {str(result.get('message', ''))[:200]}") + previous_status = known_statuses.get(key) or checked_statuses.get(key) + write_result(key, result, source, finding, previous_status) + known_keys.add(key) + checked_statuses[key] = result.get('status') + + print("\n--- ✅ Проверка завершена. ---") + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/provider_resolution.py b/app/keycheckers/provider_resolution.py new file mode 100644 index 0000000..a2ca081 --- /dev/null +++ b/app/keycheckers/provider_resolution.py @@ -0,0 +1,189 @@ +import sys + +sys.dont_write_bytecode = True + +import os + + +SUPPORTED_PROVIDERS = ("deepseek", "zai", "qwen", "kimi") +DEFAULT_PROVIDER_ORDER = SUPPORTED_PROVIDERS +AMBIGUOUS_QWEN_DEEPSEEK_HINT = "ambiguous_qwen_deepseek" +AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk" + + +def split_csv(value): + if not value: + return [] + if isinstance(value, str): + values = value.split(",") + else: + values = value + return [str(item).strip().lower() for item in values if str(item).strip()] + + +def unique_supported(values): + output = [] + seen = set() + for value in values: + provider = str(value or "").strip().lower() + if provider in SUPPORTED_PROVIDERS and provider not in seen: + seen.add(provider) + output.append(provider) + return output + + +def providers_for_hint(hint): + hint = str(hint or "").strip().lower() + if hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT: + return ["qwen", "deepseek"] + if hint == AMBIGUOUS_GENERIC_SK_HINT: + return list(SUPPORTED_PROVIDERS) + return [hint] if hint in SUPPORTED_PROVIDERS else [] + + +def detector_provider(finding): + if not isinstance(finding, dict): + return "" + extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {} + names = { + str(finding.get("DetectorName") or finding.get("DetectorType") or "").strip().lower(), + str(extra.get("name") or "").strip().lower(), + } + mappings = ( + ("deepseek", {"deepseek", "deepseekapikey", "deepseek_api_key"}), + ("zai", {"zaiglm"}), + ("qwen", {"qwendashscope", "qwen_dashscope", "qwen", "dashscope"}), + ("kimi", {"kimimoonshot", "moonshotai", "moonshot", "kimi"}), + ) + for provider, detectors in mappings: + if names & detectors: + return provider + return "" + + +def ordered_providers(finding=None, hint="", origin_service="", configured_order=None): + context = finding.get("ScannerContext") if isinstance(finding, dict) and isinstance( + finding.get("ScannerContext"), dict + ) else {} + hint = str(hint or context.get("provider_hint") or "").strip().lower() + compatible = unique_supported(context.get("provider_candidates") or providers_for_hint(hint)) + if not compatible: + compatible = providers_for_hint(hint) + if not compatible: + compatible = list(SUPPORTED_PROVIDERS) + + configured = unique_supported( + configured_order + if configured_order is not None + else split_csv(os.getenv("KEYCHECK_PROVIDER_RESOLUTION_ORDER")) + ) + base_order = configured or list(DEFAULT_PROVIDER_ORDER) + origin = str(origin_service or detector_provider(finding)).strip().lower() + ordered = [] + if origin in compatible: + ordered.append(origin) + ordered.extend(provider for provider in base_order if provider in compatible) + ordered.extend(provider for provider in compatible if provider not in ordered) + return unique_supported(ordered) + + +def provider_result_outcome(result): + result = result if isinstance(result, dict) else {} + status = str(result.get("status") or "UNKNOWN").strip().upper() + if result.get("authenticated") is True or status in ("VALID", "ALIVE"): + return "match" + if result.get("candidate_rejected") or status in ( + "DEAD", "INVALID", "EXPIRED", "LEAKED_REVOKED", "INVALID_OR_REVOKED", + ): + return "no_match" + return "retry" + + +def bounded_attempt(provider, result, outcome): + result = result if isinstance(result, dict) else {} + error = result.get("error") if isinstance(result.get("error"), dict) else {} + return { + "provider": provider, + "outcome": outcome, + "status": str(result.get("status") or "UNKNOWN").upper(), + "authenticated": bool(result.get("authenticated")), + "http_status": int(result.get("http_status") or error.get("http_status") or 0), + "business_code": str(result.get("business_code") or error.get("code") or "")[:80], + "region": str(result.get("region") or "")[:160], + "message": str(result.get("message") or "").replace("\r", " ").replace("\n", " ")[:300], + } + + +def default_provider_probe(provider, key, proxy, timeout, debug=False): + if provider == "deepseek": + from keycheckers.deepseek import deepseekKeycheck + + return deepseekKeycheck.check_key(key, proxy, timeout) + if provider == "zai": + from keycheckers.zai import zaiKeycheck + + return zaiKeycheck.check_key( + key, zaiKeycheck.base_urls_from_environment(), proxy, timeout, debug, + ) + if provider == "qwen": + from keycheckers.qwen import qwenKeycheck + + custom = qwenKeycheck.split_csv( + os.getenv("QWEN_BASE_URLS") or os.getenv("DASHSCOPE_BASE_URLS") + ) + base_urls = qwenKeycheck.unique_ordered([*custom, *qwenKeycheck.DEFAULT_BASE_URLS]) + return qwenKeycheck.check_key(key, base_urls, bool(custom), proxy, timeout, debug) + if provider == "kimi": + from keycheckers.kimi import kimiKeycheck + + custom = kimiKeycheck.split_csv( + os.getenv("KIMI_BASE_URLS") or os.getenv("MOONSHOT_BASE_URLS") + ) + base_urls = kimiKeycheck.unique_ordered([*custom, *kimiKeycheck.DEFAULT_BASE_URLS]) + return kimiKeycheck.check_key(key, base_urls, proxy, timeout, debug) + raise ValueError(f"unsupported provider resolution adapter: {provider}") + + +def resolve_provider_key( + key, finding=None, proxy=None, timeout=15, debug=False, hint="", + origin_service="", configured_order=None, probe=None, +): + providers = ordered_providers(finding, hint, origin_service, configured_order) + probe = probe or default_provider_probe + attempts = [] + retry_results = [] + for provider in providers: + result = probe(provider, key, proxy, timeout, debug) + result = result if isinstance(result, dict) else {"status": "UNKNOWN"} + outcome = provider_result_outcome(result) + attempts.append(bounded_attempt(provider, result, outcome)) + if outcome == "match": + return { + **result, + "resolved_provider": provider, + "provider_resolution": "matched", + "provider_resolution_order": providers, + "provider_resolution_attempts": attempts, + "result_source": "provider_resolution", + } + if outcome == "retry": + retry_results.append(result) + + if retry_results: + selected = retry_results[0] + return { + **selected, + "provider_resolution": "retry", + "provider_resolution_order": providers, + "provider_resolution_attempts": attempts, + "result_source": "provider_resolution", + "message": str(selected.get("message") or "provider resolution remains inconclusive")[:1000], + } + return { + "status": "DEAD", + "provider_resolution": "exhausted", + "provider_resolution_order": providers, + "provider_resolution_attempts": attempts, + "result_source": "provider_resolution", + "message": "all compatible providers rejected the credential", + } diff --git a/app/keycheckers/provider_resolver/providerResolverKeycheck.py b/app/keycheckers/provider_resolver/providerResolverKeycheck.py new file mode 100644 index 0000000..e626953 --- /dev/null +++ b/app/keycheckers/provider_resolver/providerResolverKeycheck.py @@ -0,0 +1,203 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os +import re + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + record_validation_result, + recover_status_transaction, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) +from keycheckers.provider_resolution import ( + AMBIGUOUS_GENERIC_SK_HINT, + AMBIGUOUS_QWEN_DEEPSEEK_HINT, + resolve_provider_key, +) + + +SERVICE = "provider_resolver" +DETECTOR = "ProviderResolver" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() + +CHECKED_FILE = os.path.join(OUTPUT_DIR, "providerResolverChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "providerResolverResults.jsonl") +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "providerResolverAlive.txt"), + "NO_BALANCE": os.path.join(OUTPUT_DIR, "providerResolverNoBalance.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "providerResolverDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "providerResolverRestricted.txt"), + "LIMITED": os.path.join(OUTPUT_DIR, "providerResolverLimited.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "providerResolverNetwork.txt"), + "NO_CONTEXT": os.path.join(OUTPUT_DIR, "providerResolverNoContext.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "providerResolverUnknown.txt"), +} + +RESOLVABLE_KEY_REGEX = re.compile( + r"(? 512: + return "candidate exceeds the 512-byte key limit" + if value.startswith(FOREIGN_KEY_PREFIXES): + return "candidate has a foreign provider prefix" + if not RESOLVABLE_KEY_REGEX.fullmatch(value): + return "candidate does not match a bounded resolvable provider-key format" + return "" + + +def iter_candidate_keys(input_file): + detector_names = [ + "ProviderResolver", "CustomRegex", "QwenDashScope", "Qwen_DashScope", + "Qwen", "DashScope", "DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key", + "KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi", "ZaiGLM", + "qwendashscope", "qwen_dashscope", "qwen", "dashscope", "deepseek", + "deepseekapikey", "deepseek_api_key", "kimimoonshot", "moonshotai", + "moonshot", "kimi", "zaiglm", + ] + for item in iter_findings(input_file, detector_names): + finding = item.get("finding") or {} + key = item.get("credential_secret_text") or "" + if not key: + for value in (item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2")): + match = RESOLVABLE_KEY_REGEX.search(str(value or "")) + if match: + key = match.group(0) + break + if not key or key_rejection_reason(key): + continue + context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {} + metadata_hint = "" + active_metadata = item.get("candidate_metadata") + if isinstance(active_metadata, dict): + metadata_hint = str(active_metadata.get("provider_hint") or "") + metadata_candidates = active_metadata.get("provider_candidates") + if metadata_hint or isinstance(metadata_candidates, list): + context = dict(context) + if metadata_hint: + context.setdefault("provider_hint", metadata_hint) + if isinstance(metadata_candidates, list): + context.setdefault("provider_candidates", metadata_candidates) + finding = dict(finding) + finding["ScannerContext"] = context + hint = str(context.get("provider_hint") or metadata_hint or AMBIGUOUS_GENERIC_SK_HINT) + if hint not in AMBIGUOUS_HINTS: + hint = AMBIGUOUS_GENERIC_SK_HINT + yield key, item.get("source") or input_file, finding, hint + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def write_result(key, result, source, finding): + status = result.get("status") or "UNKNOWN" + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR) + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, status, + result.get("message", ""), result.get("resolved_provider") or source, + ) + record_validation_result(SERVICE, key, result, source, finding, DETECTOR) + + +def retry_statuses_from_args(args): + statuses = set() + for enabled, status in ( + (args.retry_network, "NETWORK"), + (args.retry_limited, "LIMITED"), + (args.retry_unknown, "UNKNOWN"), + (args.retry_restricted, "RESTRICTED"), + (args.retry_no_balance, "NO_BALANCE"), + (args.retry_valid, "VALID"), + ): + if enabled: + statuses.add(status) + return statuses + + +def parse_args(): + parser = argparse.ArgumentParser(description="Ambiguous generic provider key resolver") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=15) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--retry-no-balance", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + parser.add_argument("--debug", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = retry_statuses_from_args(args) + processed = 0 + skipped = 0 + for key, source, finding, hint in iter_candidate_keys(args.input): + if should_skip_key( + key, checked, known, args, retry_statuses, service=SERVICE, + source=source, finding=finding, detector=DETECTOR, + ): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] Ambiguous provider candidate {mask_secret(key)} from {source}") + proxy = next(proxy_cycler) if proxy_cycler else None + result = resolve_provider_key( + key, finding, proxy, args.timeout, args.debug, hint=hint, + ) + print( + f" STATUS: {result['status']} provider={result.get('resolved_provider', '')} " + f"| {result.get('message', '')[:200]}" + ) + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/qwen/qwenKeycheck.py b/app/keycheckers/qwen/qwenKeycheck.py new file mode 100644 index 0000000..0fc5487 --- /dev/null +++ b/app/keycheckers/qwen/qwenKeycheck.py @@ -0,0 +1,773 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import json +import os +import re +from urllib.parse import urlparse + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +try: + sys.stdout.reconfigure(encoding="utf-8", errors="replace") + sys.stderr.reconfigure(encoding="utf-8", errors="replace") +except (AttributeError, OSError, ValueError): + pass + +from keycheck_common import ( + combined_provider_routing_hint, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + keycheck_input_mode, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + provider_routing_database_failed, + read_plain_keys, + record_validation_result, + recover_status_transaction, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) +from keycheckers.provider_resolution import resolve_provider_key + + +SERVICE = "qwen" +DETECTOR = "QwenDashScope" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() + +CHECKED_FILE = os.path.join(OUTPUT_DIR, "qwenChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "qwenResults.jsonl") +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "qwenAlive.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "qwenDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "qwenRestricted.txt"), + "LIMITED": os.path.join(OUTPUT_DIR, "qwenLimited.txt"), + "NO_BALANCE": os.path.join(OUTPUT_DIR, "qwenNoBalance.txt"), + "NO_CONTEXT": os.path.join(OUTPUT_DIR, "qwenNoContext.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "qwenNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "qwenUnknown.txt"), +} + +QWEN_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope", "dashscope", "qwen"} +QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"} +DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"} +KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"} +QWEN_KEY_MAX_BYTES = 512 +QWEN_KEY_REGEX = re.compile( + r"(? QWEN_KEY_MAX_BYTES: + return f"candidate exceeds the {QWEN_KEY_MAX_BYTES}-byte key limit" + if OPENAI_LEGACY_KEY_MARKER in value: + return "candidate is a recognizable OpenAI legacy key" + if value.count("sk-") != 1: + return "candidate contains multiple concatenated key prefixes" + if not QWEN_KEY_REGEX.fullmatch(value) or value.startswith(FOREIGN_QWEN_KEY_PREFIXES): + return "candidate does not match the bounded Qwen key format" + return "" + + +def is_qwen_key(key): + return not qwen_key_rejection_reason(key) + + +def finding_has_oversized_qwen_key(finding, *raw_values): + values = list(raw_values) + if isinstance(finding, dict): + values.extend((finding.get("Raw"), finding.get("RawV2"), finding.get("raw"), finding.get("raw_v2"))) + nested = finding.get("finding") + if isinstance(nested, dict): + values.extend((nested.get("Raw"), nested.get("RawV2"), nested.get("raw"), nested.get("raw_v2"))) + return any( + QWEN_OVERSIZED_KEY_PREFIX_REGEX.search(str(value or "")) + for value in values + ) + + +def warn_rejected_candidate(reason, source, key=""): + reason = str(reason or "candidate rejected") + source = str(source or "unknown") + if key: + reason = reason.replace(key, "***REDACTED***") + source = source.replace(key, "***REDACTED***") + reason = reason.replace("\r", " ").replace("\n", " ")[:300] + source = source.replace("\r", " ").replace("\n", " ")[:300] + print(f"Warning: skipped Qwen candidate from {source}: {reason}", flush=True) + + +def warn_candidate_failure(reason, source, key=""): + reason = str(reason or "candidate failure") + source = str(source or "unknown") + if key: + reason = reason.replace(key, "***REDACTED***") + source = source.replace(key, "***REDACTED***") + reason = QWEN_KEY_REGEX.sub("***REDACTED***", reason).replace("\r", " ").replace("\n", " ")[:300] + source = QWEN_KEY_REGEX.sub("***REDACTED***", source).replace("\r", " ").replace("\n", " ")[:300] + print(f"Warning: Qwen candidate failure from {source}: {reason}", flush=True) + + +def extract_key_from_finding(data): + if not detector_name_from_finding(data): + return "" + if data.get("Raw") or data.get("RawV2"): + return key_from_text(data.get("Raw"), data.get("RawV2")) + if data.get("raw") or data.get("raw_v2"): + return key_from_text(data.get("raw"), data.get("raw_v2")) + finding = data.get("finding") + if isinstance(finding, dict): + return key_from_text(finding.get("Raw"), finding.get("RawV2")) + return "" + + +def finding_provider_routing_hint(finding): + if not isinstance(finding, dict): + return "" + context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {} + persisted_hint = context.get("provider_hint") + if ( + context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE + and persisted_hint in (*GENERIC_SK_PROVIDERS, AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT) + ): + return persisted_hint + parts = [str(context.get(key) or "") for key in ("nearby", "file")] + metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {} + data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {} + for details in data.values(): + if not isinstance(details, dict): + continue + parts.extend(str(details.get(key) or "") for key in ("file", "repository", "repo", "link", "image")) + + text = "\n".join(parts) + evidence = set() + if QWEN_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES): + evidence.add("qwen") + if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES): + evidence.add("deepseek") + if KIMI_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES): + evidence.add("kimi") + if persisted_hint == AMBIGUOUS_PROVIDER_HINT: + evidence.update(("qwen", "deepseek")) + elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT: + evidence.update(GENERIC_SK_PROVIDERS) + elif persisted_hint in GENERIC_SK_PROVIDERS: + evidence.add(persisted_hint) + if len(evidence) > 1: + return AMBIGUOUS_PROVIDER_HINT if evidence == {"qwen", "deepseek"} else AMBIGUOUS_GENERIC_SK_HINT + return next(iter(evidence)) if evidence else "" + + +def finding_has_ambiguous_provider_hint(finding): + return finding_provider_routing_hint(finding) in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT) + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def iter_candidate_keys(input_file, plain_files, trusted_retry_files=None): + seen_plain = set() + seen_candidates = set() + routing_decisions = {} + detector_names = [ + "QwenDashScope", "Qwen_DashScope", "qwendashscope", "qwen_dashscope", + "Qwen", "DashScope", "qwen", "dashscope", "CustomRegex", + ] + for item in iter_findings(input_file, detector_names): + data = dict(item.get("finding") or {}) + data.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, None) + candidate_metadata = item.get("candidate_metadata") + persisted_route = "" + if keycheck_input_mode() == "postgres" and isinstance(candidate_metadata, dict): + persisted_route = str(candidate_metadata.get("provider_hint") or "").lower() + if persisted_route == SERVICE: + data[CANDIDATE_PROVIDER_ROUTE_FIELD] = SERVICE + if finding_has_oversized_qwen_key(data, item.get("raw"), item.get("raw_v2")): + warn_rejected_candidate( + f"candidate exceeds the {QWEN_KEY_MAX_BYTES}-byte key limit", + item.get("source") or input_file, + ) + continue + if persisted_route == SERVICE: + key = key_from_text( + item.get("raw"), item.get("raw_v2"), + data.get("Raw"), data.get("RawV2"), + ) + else: + key = extract_key_from_finding(data) + if key and is_qwen_key(key): + if not key.startswith("sk-sp-"): + if persisted_route == SERVICE: + hint, lookup_failed = SERVICE, False + else: + local_hint = finding_provider_routing_hint(data) + if key in routing_decisions: + hint, lookup_failed = routing_decisions[key] + if local_hint == AMBIGUOUS_PROVIDER_HINT or ( + local_hint and hint and local_hint != hint + ): + hint = AMBIGUOUS_PROVIDER_HINT + elif not hint: + hint = local_hint + routing_decisions[key] = (hint, lookup_failed) + else: + hint = combined_provider_routing_hint(key, local_hint) + lookup_failed = provider_routing_database_failed() + routing_decisions[key] = (hint, lookup_failed) + if lookup_failed: + message = "provider routing evidence lookup failed closed" + warn_candidate_failure(message, item.get("source") or input_file, key) + raise RuntimeError(message) + if hint != "qwen" and not ( + keycheck_input_mode() == "postgres" + and hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT) + ): + continue + seen_candidates.add(key) + yield key, item.get("source") or input_file, data + + for item in read_plain_keys(plain_files, QWEN_KEY_REGEX): + key = item["key"] + if not is_qwen_key(key) or not key.startswith("sk-sp-"): + continue + if key not in seen_plain: + seen_plain.add(key) + seen_candidates.add(key) + yield key, item["source"], {} + + owned_retry_paths = { + os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values() + } + retry_files = [ + path for path in (trusted_retry_files or []) + if os.path.normcase(os.path.abspath(path)) in owned_retry_paths + ] + for item in read_plain_keys(retry_files, QWEN_KEY_REGEX): + key = item["key"] + if not is_qwen_key(key) or key in seen_candidates: + continue + if not key.startswith("sk-sp-"): + hint = combined_provider_routing_hint(key, "qwen") + if provider_routing_database_failed(): + message = "provider routing evidence lookup failed closed" + warn_candidate_failure(message, item["source"], key) + raise RuntimeError(message) + if hint != "qwen": + continue + seen_candidates.add(key) + yield key, item["source"], {} + + +def redact_text(text, key): + redacted = str(text or "")[:1000] + if key: + redacted = redacted.replace(key, "***REDACTED***") + return QWEN_KEY_REGEX.sub("***REDACTED***", redacted) + + +def parse_error_response(response, key): + try: + payload = response.json() + except ValueError: + payload = {} + error = payload.get("error") if isinstance(payload, dict) else {} + if not isinstance(error, dict): + error = {} + message = error.get("message") or response.text[:500] + return { + "http_status": response.status_code, + "code": error.get("code") or error.get("type") or "", + "type": error.get("type") or "", + "message": redact_text(message, key), + } + + +def classify_error(error): + http_status = int(error.get("http_status") or 0) + code = str(error.get("code") or "").lower() + message = str(error.get("message") or "").lower() + + if http_status == 401 or "invalid_api_key" in code or "incorrect api key" in message: + return "DEAD" + if http_status == 402 or "arrearage" in code or any(item in message for item in ( + "arrearage", "arrears", "billing", "balance", "overdue", "payment", + "insufficient credit", "credit balance", + )): + return "NO_BALANCE" + if http_status == 403: + return "RESTRICTED" + if http_status == 429: + return "LIMITED" + if 500 <= http_status <= 599: + return "NETWORK" + return "UNKNOWN" + + +def choose_chat_model(models): + models = [str(model or "").replace("models/", "") for model in models if model] + by_lower = {model.lower(): model for model in models} + for model in CHAT_MODEL_PRIORITY: + if model.lower() in by_lower: + return by_lower[model.lower()] + for model in models: + lowered = model.lower() + if any(marker in lowered for marker in NON_CHAT_MODEL_MARKERS): + continue + if any(marker in lowered for marker in ("qwen", "qwq", "qvq")): + return model + return "" + + +def probe_chat_completion(key, base_url, model, proxy, timeout, debug=False): + if not model: + return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""} + url = f"{normalize_base_url(base_url)}/chat/completions" + headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"} + payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1} + try: + response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc)[:1000], "model": model} + if debug: + print(f" DEBUG {endpoint_label(base_url)} chat ping {model}: HTTP {response.status_code}: {redact_text(response.text[:500], key)}") + if response.status_code == 200: + return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model} + error = parse_error_response(response, key) + return {"status": classify_error(error), "error": error, "message": error.get("message") or "", "model": model} + + +def parse_models(payload): + if not isinstance(payload, dict): + return [], [] + model_infos = payload.get("data") + if not isinstance(model_infos, list): + model_infos = payload.get("models") if isinstance(payload.get("models"), list) else [] + models = [] + for item in model_infos: + if not isinstance(item, dict): + continue + model_id = item.get("id") or item.get("model") or item.get("name") + if model_id: + models.append(str(model_id).replace("models/", "")) + return sorted(set(models)), model_infos + + +def notable_models(models): + notable = [] + for model in models: + lowered = model.lower() + if any(marker in lowered for marker in MODEL_MARKERS): + notable.append(model) + return notable[:30] + + +def check_base_url(key, base_url, proxy, timeout, debug=False): + url = f"{normalize_base_url(base_url)}/models" + headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"} + try: + response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return { + "base_url": base_url, + "region": endpoint_label(base_url), + "status": "NETWORK", + "message": str(exc)[:1000], + } + + if debug: + print(f" DEBUG {endpoint_label(base_url)} /models: HTTP {response.status_code}: {redact_text(response.text[:500], key)}") + + if response.status_code == 200: + try: + payload = response.json() + except ValueError: + payload = {} + models, model_infos = parse_models(payload) + chat_model = choose_chat_model(models) + probe = probe_chat_completion(key, base_url, chat_model, proxy, timeout, debug) + probe_status = probe.get("status") or "UNKNOWN" + status = "VALID" if probe_status in ("GENERATION_OK", "NO_CONTEXT") else probe_status + probe_message = probe.get("message") or json.dumps(probe.get("error") or {}, ensure_ascii=False)[:1000] + return { + "base_url": base_url, + "region": endpoint_label(base_url), + "status": status, + "authenticated": True, + "model_count": len(models), + "models": notable_models(models), + "all_model_count": len(models), + "model_infos_count": len(model_infos), + "llm_probe_status": probe_status, + "llm_probe_model": probe.get("model", chat_model), + "message": ( + f"chat ping ok; model={chat_model}; models={len(models)}" + if probe_status == "GENERATION_OK" + else f"models authenticated; generation_probe={probe_status}; models={len(models)}; {probe_message}" + )[:1000], + "error": probe.get("error") or {}, + } + + error = parse_error_response(response, key) + return { + "base_url": base_url, + "region": endpoint_label(base_url), + "status": classify_error(error), + "error": error, + "message": error.get("message") or "", + } + + +def choose_final_status(key, attempts, has_custom_base_urls): + statuses = [attempt.get("status") for attempt in attempts] + for status in ("VALID", "NO_BALANCE", "LIMITED", "RESTRICTED", "UNKNOWN", "NO_CONTEXT", "NETWORK"): + if status in statuses: + return status + return "DEAD" + + +def check_key(key, base_urls, has_custom_base_urls, proxy, timeout, debug=False): + rejection = qwen_key_rejection_reason(key) + if rejection: + return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True} + attempts = [] + for base_url in base_urls: + result = check_base_url(key, base_url, proxy, timeout, debug) + attempts.append(result) + if result.get("status") == "VALID": + return { + "status": "VALID", + "region": result.get("region"), + "base_url": result.get("base_url"), + "model_count": result.get("model_count", 0), + "models": result.get("models", []), + "llm_probe_status": result.get("llm_probe_status", ""), + "llm_probe_model": result.get("llm_probe_model", ""), + "authenticated": bool(result.get("authenticated")), + "attempts": attempts, + "message": f"models={result.get('model_count', 0)} region={result.get('region')}", + } + + status = choose_final_status(key, attempts, has_custom_base_urls) + message = "" + for attempt in attempts: + if attempt.get("status") == status: + message = attempt.get("message") or json.dumps(attempt.get("error") or {}, ensure_ascii=False)[:1000] + break + return {"status": status, "attempts": attempts, "message": message} + + +def write_result(key, result, source, finding): + rejection = qwen_key_rejection_reason(key) + if rejection: + raise ValueError(rejection) + status = result.get("status") or "UNKNOWN" + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR) + if status == "VALID": + extra = f"{result.get('region', '')};probe_model={result.get('llm_probe_model', '')};models={','.join(result.get('models', []))[:500]}" + else: + extra = source + commit_status_transaction( + CHECKED_FILE, + STATUS_FILES, + key, + status, + result.get("message", ""), + extra, + ) + record_validation_result(SERVICE, key, result, source, finding, DETECTOR) + + +def retry_statuses_from_args(args): + retry_statuses = set() + if args.retry_network: + retry_statuses.add("NETWORK") + if args.retry_limited: + retry_statuses.add("LIMITED") + if args.retry_unknown: + retry_statuses.update({"UNKNOWN", "NO_CONTEXT"}) + if args.retry_restricted: + retry_statuses.add("RESTRICTED") + if args.retry_no_balance: + retry_statuses.add("NO_BALANCE") + if args.retry_valid: + retry_statuses.add("VALID") + return retry_statuses + + +def retry_input_files_from_args(args): + if args.recheck_all: + statuses = list(STATUS_FILES) + else: + statuses = [] + if args.retry_network: + statuses.append("NETWORK") + if args.retry_limited: + statuses.append("LIMITED") + if args.retry_unknown: + statuses.extend(("UNKNOWN", "NO_CONTEXT")) + if args.retry_restricted: + statuses.append("RESTRICTED") + if args.retry_no_balance: + statuses.append("NO_BALANCE") + if args.retry_valid: + statuses.append("VALID") + return list(dict.fromkeys(STATUS_FILES[status] for status in statuses)) + + +def base_urls_from_args(args): + env_urls = split_csv(os.getenv("QWEN_BASE_URLS") or os.getenv("DASHSCOPE_BASE_URLS")) + custom_urls = [] + for value in args.base_url: + custom_urls.extend(split_csv(value)) + custom_urls.extend(env_urls) + default_urls = [] if args.no_default_base_urls else DEFAULT_BASE_URLS + return unique_ordered(custom_urls + default_urls), bool(custom_urls) + + +def parse_args(): + parser = argparse.ArgumentParser(description="Qwen/DashScope key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=15) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--base-url", action="append", default=[], help="Extra DashScope/OpenAI-compatible base URL; can be repeated") + parser.add_argument("--no-default-base-urls", action="store_true") + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--retry-no-balance", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + parser.add_argument("--debug", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = retry_statuses_from_args(args) + retry_input_files = retry_input_files_from_args(args) + base_urls, has_custom_base_urls = base_urls_from_args(args) + if not base_urls: + raise SystemExit("No Qwen/DashScope base URLs configured") + + print("--- Qwen/DashScope key checker ---") + print("base_urls: " + ", ".join(endpoint_label(url) for url in base_urls)) + processed = 0 + skipped = 0 + for key, source, finding in iter_candidate_keys(args.input, args.plain, retry_input_files): + finding = dict(finding or {}) + candidate_route = str(finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, "") or "").lower() + rejection = qwen_key_rejection_reason(key) + if rejection: + skipped += 1 + warn_rejected_candidate(rejection, source, key) + continue + try: + should_skip = should_skip_key( + key, checked, known, args, retry_statuses, service=SERVICE, + source=source, finding=finding, detector=DETECTOR, + ) + except Exception as exc: + warn_candidate_failure(f"candidate preparation failed: {exc}", source, key) + raise + if should_skip: + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + try: + print(f"\n[{processed}] Qwen/DashScope candidate {mask_secret(key)} from {source}") + proxy = next(proxy_cycler) if proxy_cycler else None + routing_hint = "qwen" + if keycheck_input_mode() == "postgres": + if candidate_route == SERVICE: + routing_hint = SERVICE + else: + routing_hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding)) + if provider_routing_database_failed(): + raise RuntimeError("provider routing evidence lookup failed closed") + if routing_hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT): + result = resolve_provider_key( + key, finding, proxy, args.timeout, args.debug, + hint=routing_hint, origin_service=SERVICE, + ) + else: + result = check_key(key, base_urls, has_custom_base_urls, proxy, args.timeout, args.debug) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + except Exception as exc: + warn_candidate_failure(f"candidate processing failed: {exc}", source, key) + raise + + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/replicate/replicateKeycheck.py b/app/keycheckers/replicate/replicateKeycheck.py new file mode 100644 index 0000000..6f1a4d5 --- /dev/null +++ b/app/keycheckers/replicate/replicateKeycheck.py @@ -0,0 +1,403 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os +import re +from collections import Counter + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, classify_common_http_status, commit_status_transaction, + default_input_file, default_proxy_file, ensure_output_files, iter_findings, + keycheck_input_mode, + load_checked_statuses, load_known_keys, load_proxies, mask_secret, + read_plain_keys, record_validation_result, recover_status_transaction, + request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event, +) + +SERVICE = "replicate" +DETECTOR = "Replicate" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +CHECKED_FILE = os.path.join(OUTPUT_DIR, "replicateChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "replicateResults.jsonl") +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "replicateAlive.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "replicateDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "replicateRestricted.txt"), + "LIMITED": os.path.join(OUTPUT_DIR, "replicateLimited.txt"), + "NO_BALANCE": os.path.join(OUTPUT_DIR, "replicateNoBalance.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "replicateNetwork.txt"), + "NO_CONTEXT": os.path.join(OUTPUT_DIR, "replicateNoContext.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "replicateUnknown.txt"), +} +KEY_REGEX = re.compile(r"\br8_[A-Za-z0-9]{30,}\b") +API_BASE = "https://api.replicate.com/v1" +ACCOUNT_URL = f"{API_BASE}/account" +RESOURCE_ENDPOINTS = { + "predictions": f"{API_BASE}/predictions", + "deployments": f"{API_BASE}/deployments", + "trainings": f"{API_BASE}/trainings", +} +NO_BALANCE_MARKERS = ( + "balance", + "billing", + "credit", + "credits", + "payment", + "insufficient", + "depleted", + "no credits", + "out of credit", + "run out of credit", +) + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def auth_headers(key): + return {"Authorization": f"Bearer {key}", "Accept": "application/json"} + + +def redacted_error_message(response, key): + return request_error_message(response).replace(key, "***REDACTED***") + + +def classify_replicate_response(response, key): + message = redacted_error_message(response, key).lower() + if response.status_code == 402 or any(marker in message for marker in NO_BALANCE_MARKERS): + return "NO_BALANCE" + if response.status_code == 403: + return "RESTRICTED" + return classify_common_http_status(response.status_code) + + +def api_get(key, url, proxy, timeout, debug=False): + try: + response = requests.get(url, headers=auth_headers(key), proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc)[:1000], "payload": None} + if debug: + detail = "ok" if response.status_code == 200 else redacted_error_message(response, key)[:500] + print(f" DEBUG GET {url}: HTTP {response.status_code}: {detail}") + if response.status_code != 200: + return { + "status": classify_replicate_response(response, key), + "http_status": response.status_code, + "message": redacted_error_message(response, key), + "payload": None, + } + try: + payload = response.json() if response.text else {} + except ValueError: + payload = {} + return {"status": "OK", "http_status": 200, "message": "ok", "payload": payload} + + +def paginated_items(payload): + if isinstance(payload, list): + return payload + if not isinstance(payload, dict): + return [] + for key in ("results", "data", "items"): + value = payload.get(key) + if isinstance(value, list): + return value + return [] + + +def text_value(value): + return str(value or "").strip() + + +def compact_model_ref(value): + if isinstance(value, str): + return value.strip() + if not isinstance(value, dict): + return "" + owner = text_value(value.get("owner") or value.get("model_owner")) + name = text_value(value.get("name") or value.get("model_name")) + if owner and name: + return f"{owner}/{name}" + for key in ("model", "id", "slug"): + item = text_value(value.get(key)) + if item: + return item + url = text_value(value.get("url") or value.get("web_url")) + if "replicate.com/" in url: + return url.rstrip("/").split("replicate.com/", 1)[-1] + return "" + + +def model_refs_from_item(item): + if not isinstance(item, dict): + return [] + refs = [] + for key in ("model", "destination", "source_model", "base_model"): + ref = compact_model_ref(item.get(key)) + if ref: + refs.append(ref) + for key in ("version", "latest_version", "current_release"): + value = item.get(key) + if isinstance(value, dict): + ref = compact_model_ref(value.get("model") or value.get("destination")) + if ref: + refs.append(ref) + return sorted(set(refs)) + + +def summarize_predictions(payload, limit=10): + items = paginated_items(payload) + models = sorted({text_value(item.get("model")) for item in items if isinstance(item, dict) and item.get("model")}) + statuses = Counter(text_value(item.get("status")) for item in items if isinstance(item, dict) and item.get("status")) + samples = [] + for item in items[:limit]: + if not isinstance(item, dict): + continue + samples.append({ + "id": text_value(item.get("id"))[:80], + "status": text_value(item.get("status")), + "model": text_value(item.get("model")), + "source": text_value(item.get("source")), + "data_removed": bool(item.get("data_removed")), + "created_at": text_value(item.get("created_at")), + "completed_at": text_value(item.get("completed_at")), + }) + return { + "prediction_count_sample": len(items), + "prediction_has_next_page": bool(isinstance(payload, dict) and payload.get("next")), + "prediction_status_counts": dict(statuses), + "prediction_models": models[:50], + "prediction_samples": samples, + } + + +def deployment_name(item): + owner = text_value(item.get("owner") or item.get("deployment_owner")) + name = text_value(item.get("name") or item.get("deployment_name")) + if owner and name and "/" not in name: + return f"{owner}/{name}" + return name or owner + + +def summarize_deployments(payload, limit=20): + items = paginated_items(payload) + models = sorted({ref for item in items for ref in model_refs_from_item(item)}) + deployments = [] + for item in items[:limit]: + if not isinstance(item, dict): + continue + current_release = item.get("current_release") if isinstance(item.get("current_release"), dict) else {} + deployments.append({ + "name": deployment_name(item), + "model": next(iter(model_refs_from_item(item)), ""), + "version": text_value(item.get("version") or current_release.get("version"))[:80], + "hardware": text_value(item.get("hardware") or current_release.get("hardware")), + "min_instances": item.get("min_instances"), + "max_instances": item.get("max_instances"), + }) + return { + "deployment_count": len(items), + "deployment_has_next_page": bool(isinstance(payload, dict) and payload.get("next")), + "deployment_models": models[:50], + "deployments": deployments, + } + + +def summarize_trainings(payload, limit=10): + items = paginated_items(payload) + models = sorted({ref for item in items for ref in model_refs_from_item(item)}) + statuses = Counter(text_value(item.get("status")) for item in items if isinstance(item, dict) and item.get("status")) + samples = [] + for item in items[:limit]: + if not isinstance(item, dict): + continue + samples.append({ + "id": text_value(item.get("id"))[:80], + "status": text_value(item.get("status")), + "model": next(iter(model_refs_from_item(item)), ""), + "created_at": text_value(item.get("created_at")), + "completed_at": text_value(item.get("completed_at")), + }) + return { + "training_count_sample": len(items), + "training_has_next_page": bool(isinstance(payload, dict) and payload.get("next")), + "training_status_counts": dict(statuses), + "training_models": models[:50], + "training_samples": samples, + } + + +def probe_account_resources(key, proxy, timeout, debug=False): + summaries = {} + endpoint_statuses = {} + model_refs = set() + ok_count = 0 + total_items = 0 + summarizers = { + "predictions": summarize_predictions, + "deployments": summarize_deployments, + "trainings": summarize_trainings, + } + for name, url in RESOURCE_ENDPOINTS.items(): + result = api_get(key, url, proxy, timeout, debug) + endpoint_statuses[name] = {k: v for k, v in result.items() if k in ("status", "http_status", "message")} + if result.get("status") != "OK": + continue + ok_count += 1 + summary = summarizers[name](result.get("payload")) + summaries.update(summary) + for key_name, value in summary.items(): + if key_name.endswith("_models") and isinstance(value, list): + model_refs.update(value) + total_items += sum( + int(summary.get(field, 0) or 0) + for field in ("prediction_count_sample", "deployment_count", "training_count_sample") + ) + if ok_count == len(RESOURCE_ENDPOINTS): + probe_status = "RESOURCE_OK" if total_items else "NO_RESOURCES" + elif ok_count: + probe_status = "PARTIAL" + else: + probe_status = next((item.get("status") for item in endpoint_statuses.values() if item.get("status")), "UNKNOWN") + return { + "probe": {"status": probe_status, "endpoints": endpoint_statuses}, + "models": sorted(model_refs)[:50], + "model_count": len(model_refs), + **summaries, + } + + +def iter_candidate_decisions(input_file, plain_files): + seen_plain = set() + for item in iter_findings(input_file, [DETECTOR]): + key = item.get("credential_secret_text") or item["raw"] + if key: + yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key)) + for item in read_plain_keys(plain_files, KEY_REGEX): + key = item["key"] + if key not in seen_plain: + seen_plain.add(key) + yield key, item["source"], {}, True + + +def extract_candidates(input_file, plain_files): + for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files): + if valid_format: + yield key, source, finding + + +def check_key(key, proxy, args): + account = api_get(key, ACCOUNT_URL, proxy, args.timeout, args.debug) + if account.get("status") != "OK": + return {k: v for k, v in account.items() if k != "payload"} + data = account.get("payload") if isinstance(account.get("payload"), dict) else {} + result = { + "status": "VALID", + "message": "account endpoint accepted", + "account": data.get("username") or data.get("name") or "", + "account_type": data.get("type") or "", + } + if not args.no_resource_probe: + result.update(probe_account_resources(key, proxy, args.timeout, args.debug)) + result["message"] = ( + f"account endpoint accepted; probe={result.get('probe', {}).get('status')}; " + f"models={result.get('model_count', 0)}; " + f"deployments={result.get('deployment_count', 0)}; " + f"predictions={result.get('prediction_count_sample', 0)}; " + f"trainings={result.get('training_count_sample', 0)}" + ) + else: + result.update({"probe": {"status": "not_probed"}, "models": [], "model_count": 0}) + return result + + +def write_result(key, result, source, finding): + status = result.get("status") or "UNKNOWN" + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR) + extra = ",".join(result.get("models") or [])[:1000] if status == "VALID" else source + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra or source, + ) + record_validation_result(SERVICE, key, result, source, finding, DETECTOR) + + +def parse_args(): + parser = argparse.ArgumentParser(description="Replicate key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=15) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--retry-no-balance", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + parser.add_argument("--no-resource-probe", action="store_true", help="Only call /account; skip read-only predictions/deployments/trainings probes.") + parser.add_argument("--debug", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = set() + if args.retry_network: retry_statuses.add("NETWORK") + if args.retry_limited: retry_statuses.add("LIMITED") + if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"}) + if args.retry_restricted: retry_statuses.add("RESTRICTED") + if args.retry_no_balance: retry_statuses.add("NO_BALANCE") + if args.retry_valid: retry_statuses.add("VALID") + processed = skipped = 0 + print("--- Replicate key checker ---") + print("Default mode: /account plus read-only /predictions, /deployments and /trainings probes. Use --no-resource-probe for /account only.") + print(f"proxy: {args.proxy_file}") + postgres_mode = keycheck_input_mode() == "postgres" + for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain): + if not valid_format and not postgres_mode: + skipped += 1 + continue + if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] Replicate candidate {mask_secret(key)} from {source}") + result = ( + check_key(key, next(proxy_cycler) if proxy_cycler else None, args) + if valid_format else + {"status": "NO_CONTEXT", "message": "candidate does not match canonical Replicate token format"} + ) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + if result.get("status") == "VALID": + print(f" ACCOUNT: {result.get('account') or 'unknown'}") + print(f" MODELS: {result.get('model_count', 0)} from account resources") + notable = result.get("models") or [] + if notable: + print(f" MODEL REFS: {', '.join(notable[:8])}") + print(f" PROBE: {(result.get('probe') or {}).get('status')}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/xai/xaiKeycheck.py b/app/keycheckers/xai/xaiKeycheck.py new file mode 100644 index 0000000..de695ac --- /dev/null +++ b/app/keycheckers/xai/xaiKeycheck.py @@ -0,0 +1,231 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os +import re + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + append_jsonl, classify_common_http_status, commit_status_transaction, + default_input_file, default_proxy_file, ensure_output_files, iter_findings, + keycheck_input_mode, + load_checked_statuses, load_known_keys, load_proxies, mask_secret, + read_plain_keys, record_validation_result, recover_status_transaction, + request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event, +) + +SERVICE = "xai" +DETECTOR_NAMES = ["XAI", "XAi", "Xai"] +DETECTOR = "XAI" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() +CHECKED_FILE = os.path.join(OUTPUT_DIR, "xaiChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "xaiResults.jsonl") +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "xaiAlive.txt"), + "NO_BALANCE": os.path.join(OUTPUT_DIR, "xaiNoBalance.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "xaiDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "xaiRestricted.txt"), + "LIMITED": os.path.join(OUTPUT_DIR, "xaiLimited.txt"), + "NO_CONTEXT": os.path.join(OUTPUT_DIR, "xaiNoContext.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "xaiNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "xaiUnknown.txt"), +} +KEY_REGEX = re.compile(r"\bxai-[A-Za-z0-9_-]{20,}\b") +MODELS_URL = "https://api.x.ai/v1/models" +CHAT_URL = "https://api.x.ai/v1/chat/completions" +CHAT_MODEL_PRIORITY = ( + "grok-4.6", + "grok-4.5", + "grok-4.3", + "grok-4.20-0309-reasoning", + "grok-4.20-0309-non-reasoning", +) +NO_BALANCE_MARKERS = ( + "quota", + "billing", + "balance", + "credit", + "credits", + "payment", + "insufficient", + "depleted", + "spending limit", + "no credits", + "used all available credits", + "doesn't have any credits", +) +DEAD_MARKERS = ("incorrect api key", "invalid api key", "api key provided", "invalid-argument") + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def iter_candidate_decisions(input_file, plain_files): + seen_plain = set() + for item in iter_findings(input_file, DETECTOR_NAMES): + key = item.get("credential_secret_text") or item["raw"] + if key: + yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key)) + for item in read_plain_keys(plain_files, KEY_REGEX): + key = item["key"] + if key not in seen_plain: + seen_plain.add(key) + yield key, item["source"], {}, True + + +def extract_candidates(input_file, plain_files): + for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files): + if valid_format: + yield key, source, finding + + +def check_key(key, proxy, timeout): + try: + response = requests.get(MODELS_URL, headers={"Authorization": f"Bearer {key}", "Accept": "application/json"}, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc)} + if response.status_code == 200: + data = response.json() if response.text else {} + models = [item.get("id") for item in data.get("data", []) if isinstance(item, dict) and item.get("id")] + model_inventory = sorted(set(models)) + model = choose_chat_model(models) + probe = probe_chat_completion(key, model, proxy, timeout) + if probe.get("status") != "GENERATION_OK": + return { + "status": probe.get("status") or "UNKNOWN", + "message": probe.get("message", ""), + "model_count": len(models), + "models": models[:20], + "model_inventory": model_inventory, + "llm_probe_status": probe.get("status"), + "llm_probe_model": probe.get("model", model), + "llm_probe_http_status": probe.get("http_status"), + } + return { + "status": "VALID", "message": f"chat ping ok; model={model}; models={len(models)}", + "model_count": len(models), "models": models[:20], "model_inventory": model_inventory, + "llm_probe_status": probe.get("status"), "llm_probe_model": model, + } + status = classify_xai_response(response) + return {"status": status, "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***")} + + +def classify_xai_response(response): + message = request_error_message(response).lower() + if any(marker in message for marker in DEAD_MARKERS): + return "DEAD" + if any(marker in message for marker in NO_BALANCE_MARKERS): + return "NO_BALANCE" + if response.status_code == 401: + return "DEAD" + if response.status_code == 403: + return "RESTRICTED" + if response.status_code == 429: + return "LIMITED" + return classify_common_http_status(response.status_code) + + +def choose_chat_model(models): + models = [str(model or "") for model in models if model] + by_lower = {model.lower(): model for model in models} + for model in CHAT_MODEL_PRIORITY: + if model.lower() in by_lower: + return by_lower[model.lower()] + for model in models: + if "grok" in model.lower(): + return model + return models[0] if models else "" + + +def probe_chat_completion(key, model, proxy, timeout): + if not model: + return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""} + headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"} + payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1} + try: + response = requests.post(CHAT_URL, headers=headers, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return {"status": "NETWORK", "message": str(exc), "model": model} + if response.status_code == 200: + return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model} + return {"status": classify_xai_response(response), "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***"), "model": model} + + +def write_result(key, result, source, finding): + status = result.get("status") or "UNKNOWN" + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR) + extra = ",".join(result.get("models") or [])[:500] if status == "VALID" else source + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra, + ) + record_validation_result(SERVICE, key, result, source, finding, DETECTOR) + + +def parse_args(): + parser = argparse.ArgumentParser(description="xAI key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=15) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--retry-no-balance", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = set() + if args.retry_network: retry_statuses.add("NETWORK") + if args.retry_limited: retry_statuses.add("LIMITED") + if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"}) + if args.retry_restricted: retry_statuses.add("RESTRICTED") + if args.retry_no_balance: retry_statuses.add("NO_BALANCE") + if args.retry_valid: retry_statuses.add("VALID") + processed = skipped = 0 + print("--- xAI key checker ---") + postgres_mode = keycheck_input_mode() == "postgres" + for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain): + if not valid_format and not postgres_mode: + skipped += 1 + continue + if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] xAI candidate {mask_secret(key)} from {source}") + result = ( + check_key(key, next(proxy_cycler) if proxy_cycler else None, args.timeout) + if valid_format else + {"status": "NO_CONTEXT", "message": "candidate does not match canonical xAI token format"} + ) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/keycheckers/zai/zaiKeycheck.py b/app/keycheckers/zai/zaiKeycheck.py new file mode 100644 index 0000000..b965da2 --- /dev/null +++ b/app/keycheckers/zai/zaiKeycheck.py @@ -0,0 +1,508 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import json +import os +import re +from urllib.parse import urlparse + +import requests + +sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from keycheck_common import ( + combined_provider_routing_hint, + commit_status_transaction, + default_input_file, + default_proxy_file, + ensure_output_files, + iter_findings, + keycheck_input_mode, + load_checked_statuses, + load_known_keys, + load_proxies, + mask_secret, + provider_routing_database_failed, + read_plain_keys, + record_validation_result, + recover_status_transaction, + require_provider_authority, + service_output_dir, + should_skip_key, + write_keycheck_event, +) +from keycheckers.provider_resolution import ( + AMBIGUOUS_GENERIC_SK_HINT, + AMBIGUOUS_QWEN_DEEPSEEK_HINT, + resolve_provider_key, +) + + +SERVICE = "zai" +DETECTOR = "ZaiGLM" +OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE) +INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file() +PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file() + +CHECKED_FILE = os.path.join(OUTPUT_DIR, "zaiChecked.txt") +RESULTS_FILE = os.path.join(OUTPUT_DIR, "zaiResults.jsonl") +STATUS_FILES = { + "VALID": os.path.join(OUTPUT_DIR, "zaiAlive.txt"), + "NO_BALANCE": os.path.join(OUTPUT_DIR, "zaiNoBalance.txt"), + "DEAD": os.path.join(OUTPUT_DIR, "zaiDead.txt"), + "RESTRICTED": os.path.join(OUTPUT_DIR, "zaiRestricted.txt"), + "LIMITED": os.path.join(OUTPUT_DIR, "zaiLimited.txt"), + "NETWORK": os.path.join(OUTPUT_DIR, "zaiNetwork.txt"), + "UNKNOWN": os.path.join(OUTPUT_DIR, "zaiUnknown.txt"), +} + +DEFAULT_BASE_URLS = ( + "https://api.z.ai/api/paas/v4", + "https://open.bigmodel.cn/api/paas/v4", +) +ZAI_KEY_REGEX = re.compile( + r"(? 512: + return "candidate exceeds the 512-byte key limit" + if value.startswith(FOREIGN_KEY_PREFIXES): + return "candidate has a foreign provider prefix" + if not ZAI_KEY_REGEX.fullmatch(value): + return "candidate does not match a bounded ZAI key format" + return "" + + +def finding_detector_names(finding): + if not isinstance(finding, dict): + return set() + extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {} + return { + name for name in ( + str(finding.get("DetectorName") or finding.get("DetectorType") or "").strip().lower(), + str(extra.get("name") or "").strip().lower(), + ) if name + } + + +def finding_provider_routing_hint(finding, key=""): + if not isinstance(finding, dict): + return "zai" if ZAI_DOTTED_KEY_REGEX.fullmatch(str(key or "")) else "" + context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {} + persisted = str(context.get("provider_hint") or "").strip().lower() + if persisted: + return persisted + if "zaiglm" in finding_detector_names(finding) or ZAI_DOTTED_KEY_REGEX.fullmatch(str(key or "")): + return "zai" + text = "\n".join(str(context.get(name) or "") for name in ("nearby", "file")) + return "zai" if ZAI_CONTEXT_REGEX.search(text) else "" + + +def iter_candidate_decisions(input_file, plain_files, trusted_retry_files=None): + detectors = [ + "ZaiGLM", "zaiglm", "CustomRegex", "QwenDashScope", "Qwen_DashScope", + "Qwen", "DashScope", "DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key", + "KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi", + ] + seen = set() + for item in iter_findings(input_file, detectors): + finding = item.get("finding") or {} + key = item.get("credential_secret_text") or key_from_text( + item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2"), + ) + if not key or key_rejection_reason(key): + continue + local_hint = finding_provider_routing_hint(finding, key) + hint = combined_provider_routing_hint(key, local_hint) + if provider_routing_database_failed(): + raise RuntimeError("provider routing evidence lookup failed closed") + seen.add(key) + yield key, item.get("source") or input_file, finding, hint + + owned_retry_paths = { + os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values() + } + retry_paths = [ + path for path in trusted_retry_files or () + if os.path.normcase(os.path.abspath(path)) in owned_retry_paths + ] + for item in read_plain_keys([*plain_files, *retry_paths], ZAI_KEY_REGEX): + key = item["key"] + if key in seen or key_rejection_reason(key): + continue + yield key, item["source"], {}, "zai" + + +def redact_text(value, key): + text = str(value or "")[:1000] + if key: + text = text.replace(key, "***REDACTED***") + return ZAI_KEY_REGEX.sub("***REDACTED***", text) + + +def parse_error(response, key): + try: + payload = response.json() + except ValueError: + payload = {} + error = payload.get("error") if isinstance(payload, dict) else {} + if not isinstance(error, dict): + error = {} + return { + "http_status": int(response.status_code), + "code": str(error.get("code") or (payload.get("code") if isinstance(payload, dict) else "") or ""), + "message": redact_text( + error.get("message") or error.get("msg") or ( + payload.get("message") or payload.get("msg") if isinstance(payload, dict) else "" + ) or response.text[:500], + key, + ), + } + + +def classify_error(error): + http_status = int(error.get("http_status") or 0) + code = str(error.get("code") or "") + message = str(error.get("message") or "").lower() + if http_status == 402 or any(marker in message for marker in ( + "insufficient balance", "balance is insufficient", "no balance", "account balance", + "recharge", "payment required", "billing arrears", "credit balance", + )): + return "NO_BALANCE", True + if code in AUTHENTICATED_NO_BALANCE_CODES: + return "NO_BALANCE", True + if any(marker in message for marker in ( + "quota", "rate limit", "rate-limit", "too many requests", "resource exhausted", + "concurrency limit", "usage limit", + )): + return "LIMITED", True + if code in AUTHENTICATED_LIMIT_CODES: + return "LIMITED", True + if any(marker in message for marker in ( + "permission denied", "access denied", "not authorized for", "model access", "forbidden", + )): + return "RESTRICTED", True + if code in AUTHENTICATED_RESTRICTED_CODES: + return "RESTRICTED", True + if http_status == 401 or code in AUTH_FAILURE_CODES: + return "DEAD", False + if 500 <= http_status <= 599 or code in {"1200", "1230", "1234", "1305"}: + return "NETWORK", False + if http_status == 429: + return "LIMITED", False + if http_status == 403: + return "RESTRICTED", False + return "UNKNOWN", False + + +def probe_chat_completion(key, base_url, model, proxy, timeout, debug=False): + if not model: + return { + "status": "UNKNOWN", "model": "", + "message": "no chat-capable model returned by /models", + } + url = f"{normalize_base_url(base_url)}/chat/completions" + headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"} + payload = { + "model": model, + "messages": [{"role": "user", "content": "ping"}], + "max_tokens": 1, + "stream": False, + } + try: + response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return { + "status": "NETWORK", "model": model, + "message": redact_text(exc, key), + } + if debug: + print( + f" DEBUG {endpoint_label(base_url)} chat probe {model}: HTTP {response.status_code}: " + f"{redact_text(response.text[:500], key)}" + ) + if response.status_code == 200: + try: + response_payload = response.json() + except ValueError as exc: + return { + "status": "UNKNOWN", "model": model, "http_status": response.status_code, + "message": f"invalid chat completion response: {exc}", + } + if isinstance(response_payload, dict) and response_payload.get("choices"): + return { + "status": "GENERATION_OK", "model": model, + "http_status": response.status_code, "message": "chat completion accepted", + } + error = parse_error(response, key) + status, authenticated = classify_error(error) + return { + "status": status, "authenticated": authenticated, "model": model, + "http_status": response.status_code, "business_code": error.get("code") or "", + "error": error, "message": error.get("message") or "", + } + + +def check_base_url(key, base_url, proxy, timeout, debug=False): + url = f"{normalize_base_url(base_url)}/models" + headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"} + try: + response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout) + except requests.RequestException as exc: + return { + "status": "NETWORK", "base_url": base_url, + "region": endpoint_label(base_url), "message": redact_text(exc, key), + } + if debug: + print( + f" DEBUG {endpoint_label(base_url)} /models: HTTP {response.status_code}: " + f"{redact_text(response.text[:500], key)}" + ) + if response.status_code == 200: + try: + payload = response.json() + data = payload.get("data") if isinstance(payload, dict) else None + if not isinstance(data, list): + raise ValueError("missing data model list") + models = sorted({ + str(item.get("id") or item.get("name") or "") + for item in data if isinstance(item, dict) and (item.get("id") or item.get("name")) + }) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + return { + "status": "UNKNOWN", "base_url": base_url, + "region": endpoint_label(base_url), "message": f"invalid models response: {exc}", + } + probe_model = PROBE_MODEL + probe = probe_chat_completion(key, base_url, probe_model, proxy, timeout, debug) + probe_status = probe.get("status") or "UNKNOWN" + if probe_status == "GENERATION_OK": + status = "VALID" + elif probe_status in {"NO_BALANCE", "LIMITED", "RESTRICTED", "NETWORK"}: + status = probe_status + elif probe_status == "DEAD": + status = "RESTRICTED" + else: + status = "UNKNOWN" + probe_message = probe.get("message") or json.dumps(probe.get("error") or {}, ensure_ascii=False) + message = ( + f"chat probe ok; model={probe_model}; models={len(data)}" + if status == "VALID" + else f"models authenticated; generation_probe={probe_status}; model={probe_model}; " + f"models={len(data)}; {probe_message}" + ) + return { + "status": status, "authenticated": True, "base_url": base_url, + "region": endpoint_label(base_url), "model_count": len(data), + "models": models[:30], "llm_probe_status": probe_status, + "llm_probe_model": probe.get("model") or probe_model, + "llm_probe_http_status": probe.get("http_status"), + "business_code": probe.get("business_code") or "", + "probe": probe, "error": probe.get("error") or {}, + "message": message[:1000], + } + error = parse_error(response, key) + status, authenticated = classify_error(error) + return { + "status": status, "authenticated": authenticated, "base_url": base_url, + "region": endpoint_label(base_url), "http_status": response.status_code, + "business_code": error.get("code") or "", "error": error, + "message": error.get("message") or "", + } + + +def check_key(key, base_urls, proxy, timeout, debug=False): + rejection = key_rejection_reason(key) + if rejection: + return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True} + attempts = [] + for base_url in base_urls: + result = check_base_url(key, base_url, proxy, timeout, debug) + attempts.append(result) + if result.get("authenticated"): + return {**result, "attempts": attempts} + statuses = [attempt.get("status") for attempt in attempts] + status = next( + (candidate for candidate in ("NETWORK", "LIMITED", "RESTRICTED", "UNKNOWN", "DEAD") if candidate in statuses), + "UNKNOWN", + ) + selected = next((attempt for attempt in attempts if attempt.get("status") == status), {}) + return {**selected, "status": status, "attempts": attempts} + + +def ensure_files(): + ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()]) + recover_status_transaction(CHECKED_FILE, STATUS_FILES) + + +def write_result(key, result, source, finding): + status = result.get("status") or "UNKNOWN" + write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR) + commit_status_transaction( + CHECKED_FILE, STATUS_FILES, key, status, + result.get("message", ""), result.get("region") or source, + ) + record_validation_result(SERVICE, key, result, source, finding, DETECTOR) + + +def retry_statuses_from_args(args): + statuses = set() + for enabled, status in ( + (args.retry_network, "NETWORK"), + (args.retry_limited, "LIMITED"), + (args.retry_unknown, "UNKNOWN"), + (args.retry_restricted, "RESTRICTED"), + (args.retry_no_balance, "NO_BALANCE"), + (args.retry_valid, "VALID"), + ): + if enabled: + statuses.add(status) + return statuses + + +def retry_input_files_from_args(args): + if args.recheck_all: + statuses = list(STATUS_FILES) + else: + statuses = list(retry_statuses_from_args(args)) + return [STATUS_FILES[status] for status in statuses] + + +def parse_args(): + parser = argparse.ArgumentParser(description="ZAI / Zhipu GLM key checker") + parser.add_argument("--input", default=INPUT_FILE) + parser.add_argument("--plain", action="append", default=[]) + parser.add_argument("--proxy-file", default=PROXY_FILE) + parser.add_argument("--timeout", type=int, default=15) + parser.add_argument("--max-keys", type=int, default=0) + parser.add_argument("--base-url", action="append", default=[]) + parser.add_argument("--no-default-base-urls", action="store_true") + parser.add_argument("--retry-network", action="store_true") + parser.add_argument("--retry-limited", action="store_true") + parser.add_argument("--retry-unknown", action="store_true") + parser.add_argument("--retry-restricted", action="store_true") + parser.add_argument("--retry-no-balance", action="store_true") + parser.add_argument("--retry-valid", action="store_true") + parser.add_argument("--recheck-all", action="store_true") + parser.add_argument("--debug", action="store_true") + return parser.parse_args() + + +def main(): + require_provider_authority(SERVICE) + args = parse_args() + ensure_files() + proxy_cycler = load_proxies(args.proxy_file) + checked = load_checked_statuses(CHECKED_FILE) + known = load_known_keys(CHECKED_FILE, STATUS_FILES) + retry_statuses = retry_statuses_from_args(args) + retry_files = retry_input_files_from_args(args) + base_urls = base_urls_from_environment(args.base_url, not args.no_default_base_urls) + if not base_urls: + raise SystemExit("No ZAI base URLs configured") + + processed = 0 + skipped = 0 + postgres_mode = keycheck_input_mode() == "postgres" + for key, source, finding, routing_hint in iter_candidate_decisions(args.input, args.plain, retry_files): + is_ambiguous = routing_hint in AMBIGUOUS_HINTS + if routing_hint != "zai" and not (postgres_mode and is_ambiguous): + skipped += 1 + continue + if should_skip_key( + key, checked, known, args, retry_statuses, service=SERVICE, + source=source, finding=finding, detector=DETECTOR, + ): + skipped += 1 + continue + if args.max_keys and processed >= args.max_keys: + break + processed += 1 + print(f"\n[{processed}] ZAI candidate {mask_secret(key)} from {source}") + proxy = next(proxy_cycler) if proxy_cycler else None + if is_ambiguous: + result = resolve_provider_key( + key, finding, proxy, args.timeout, args.debug, + hint=routing_hint, origin_service=SERVICE, + ) + else: + result = check_key(key, base_urls, proxy, args.timeout, args.debug) + print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}") + write_result(key, result, source, finding) + known.add(key) + checked[key] = result["status"] + + print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}") + + +if __name__ == "__main__": + main() diff --git a/app/lifecycle_authority.py b/app/lifecycle_authority.py new file mode 100644 index 0000000..af2ff15 --- /dev/null +++ b/app/lifecycle_authority.py @@ -0,0 +1,768 @@ +import hashlib +import hmac +import json +import os +import shutil +import socket +import stat + +from db_backend import parse_postgres_url +from process_identity import verify_retained_process +from runtime_security import ( + canonical_path, + private_file_ready, + read_private_json, + reject_reparse_components, + require_trusted_native_executable, + sha256_file, +) + + +CODE_MANIFEST_SCHEMA = 5 +APPLICATION_IMPORT_SUFFIXES = ('.py', '.pyw', '.pyc', '.pyd') +CONTROL_SCHEMA = 1 +DISCOVERY_PRODUCER_ROLE = 'discovery-producer' +DISCOVERY_PRODUCER_SOURCES = ('gitlab', 'dockerhub', 'huggingface') +PHASE_INACTIVE = 'INACTIVE' +PHASE_ACTIVATING = 'ACTIVATING' +PHASE_ACTIVE = 'ACTIVE' +PHASE_STOPPING = 'STOPPING' +PHASE_FAILED_HOLD = 'FAILED_HOLD' +LIFECYCLE_PHASES = { + PHASE_INACTIVE, + PHASE_ACTIVATING, + PHASE_ACTIVE, + PHASE_STOPPING, + PHASE_FAILED_HOLD, +} + +# These files collectively decide process ownership, database authority, and +# what data may be launched or persisted by the supervisor. +CODE_AUTHORITY_FILES = ( + 'owned_process.py', + 'supervisor.py', + 'supervisor_instance.py', + 'console_runner.py', + 'scanner.py', + 'docker_shadow.py', + 'keycheck_runner.py', + 'dashboard.py', + 'postgres_runtime.py', + 'process_identity.py', + 'runtime_security.py', + 'scanner_db.py', + 'db_backend.py', + 'result_spool.py', + 'janitor.py', + 'result_bundle.py', + 'result_ingester.py', + 'jsonl_projector.py', + 'admin_api.py', + 'worker_api.py', + 'worker_assignment.py', + 'worker_package.py', + 'scan_execution.py', + 'keycheck_candidates.py', + 'paths.py', + 'target_identity.py', + 'lifecycle_authority.py', + 'audit_github_tokens.py', + 'sync_alive_github_tokens.py', + 'child_bootstrap.py', + 'runtime_bootstrap.py', + 'runtime_document.py', + 'capacity_model.py', + 'runtime_document_io.py', + 'managed_files.py', + 'host_agent_client.py', + 'host_agent_protocol.py', + 'host_agent_reconcile.py', + 'host_agent_server.py', + 'host_agent_apply.py', + 'host_agent_lifecycle.py', + 'host_agent_runtime.py', + 'host_agent_state.py', + 'worker_contracts.py', + 'worker_assignment_runner.py', +) +REMOTE_WORKER_CODE_AUTHORITY_FILES = ( + 'db_backend.py', + 'docker_depth_experiment.py', + 'janitor.py', + 'keycheck_candidates.py', + 'lifecycle_authority.py', + 'owned_process.py', + 'paths.py', + 'process_identity.py', + 'query_policy.py', + 'remote_worker_bootstrap.py', + 'remote_worker_client.py', + 'result_bundle.py', + 'result_spool.py', + 'runtime_security.py', + 'scan_execution.py', + 'scanner.py', + 'scanner_db.py', + 'supervisor_instance.py', + 'target_identity.py', + 'worker_contracts.py', + 'worker_assignment_runner.py', + 'worker_cli.py', + 'worker_local_state.py', + 'worker_supervisor.py', + 'worker_package.py', +) +EXTERNAL_CODE_AUTHORITY_FILES = ( + '../runtime/check-openrouter-keys.ps1', + '../start_core_runtime.ps1', + '../start_runtime.ps1', + '../stop_runtime.ps1', +) if os.name == 'nt' else () +# The launchers execute before a manifest can be captured. First-launch trust +# therefore requires offline ACL hardening; manifests only detect later drift. +TRUFFLEHOG_MANIFEST_NAME = 'trufflehog' +GIT_MANIFEST_NAME = 'git' + +CHILD_INSTANCE_FILE_ENV = 'TRUF_SUPERVISOR_INSTANCE_FILE' +CHILD_INSTANCE_ID_ENV = 'TRUF_SUPERVISOR_INSTANCE_ID' +CHILD_TOKEN_ENV = 'TRUF_SUPERVISOR_TOKEN' +CHILD_CONFIG_HASH_ENV = 'TRUF_SUPERVISOR_CONFIG_SHA256' +CHILD_SCRIPT_HASH_ENV = 'TRUF_SUPERVISOR_SHA256' +CHILD_MANIFEST_HASH_ENV = 'TRUF_SUPERVISOR_CODE_MANIFEST_SHA256' +CHILD_DSN_HASH_ENV = 'TRUF_SUPERVISOR_DSN_SHA256' +CHILD_KIND_ENV = 'TRUF_SUPERVISOR_CHILD_KIND' +PRIVATE_CHILD_ENV_KEYS = ( + CHILD_INSTANCE_FILE_ENV, + CHILD_INSTANCE_ID_ENV, + CHILD_TOKEN_ENV, + CHILD_CONFIG_HASH_ENV, + CHILD_SCRIPT_HASH_ENV, + CHILD_MANIFEST_HASH_ENV, + CHILD_DSN_HASH_ENV, + CHILD_KIND_ENV, + 'SCANNER_SUPERVISED', + 'TRUF_MANAGED_POSTGRES_DSN', + 'SCANNER_DB_URL', + 'DATABASE_URL', + 'SCANNER_DASHBOARD_DB_URL', + 'KEYCHECK_DB_URL', + 'TRUF_DASHBOARD_CANONICAL_LAUNCH', + 'TRUF_DASHBOARD_HOST', +) +_LIBPQ_PRIVATE_ENV_KEYS = frozenset({ + 'PGPASSWORD', + 'PGUSER', + 'PGDATABASE', + 'PGHOST', + 'PGHOSTADDR', + 'PGPORT', + 'PGSERVICE', + 'PGSERVICEFILE', + 'PGPASSFILE', + 'PGOPTIONS', + 'PGSSLMODE', + 'PGSSLKEY', + 'PGSSLCERT', + 'PGSSLROOTCERT', +}) +_PRIVATE_EXTERNAL_ENV_KEYS = frozenset(PRIVATE_CHILD_ENV_KEYS) | _LIBPQ_PRIVATE_ENV_KEYS + + +class LifecycleAuthorityError(ValueError): + pass + + +def _manifest_payload(manifest): + return json.dumps( + manifest, + ensure_ascii=True, + sort_keys=True, + separators=(',', ':'), + ).encode('utf-8') + + +def code_manifest_sha256(manifest): + return hashlib.sha256(_manifest_payload(manifest)).hexdigest() + + +def _is_reparse_point(path): + details = os.lstat(path) + if stat.S_ISLNK(details.st_mode): + return True + attributes = getattr(details, 'st_file_attributes', 0) + reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0) + return bool(attributes & reparse_attribute) or getattr(os.path, 'isjunction', lambda _path: False)(path) + + +def _application_root(app_dir=None): + candidate = os.path.abspath(os.fspath(app_dir or os.path.dirname(os.path.abspath(__file__)))) + try: + details = os.lstat(candidate) + if _is_reparse_point(candidate): + raise LifecycleAuthorityError(f'application root reparse point is forbidden: {candidate}') + except OSError as exc: + raise LifecycleAuthorityError(f'application root is unavailable: {candidate}') from exc + if not stat.S_ISDIR(details.st_mode): + raise LifecycleAuthorityError(f'application root is not a directory: {candidate}') + return canonical_path(candidate) + + +def _external_authority_expected_path(root, name): + if name not in EXTERNAL_CODE_AUTHORITY_FILES: + raise LifecycleAuthorityError(f'code authority path is not an allowed external: {name}') + project_root = os.path.normcase(os.path.abspath(os.path.dirname(root))) + candidate = os.path.normcase(os.path.abspath(os.path.join(root, *name.split('/')))) + try: + contained = candidate != project_root and os.path.commonpath((project_root, candidate)) == project_root + except ValueError: + contained = False + if not contained: + raise LifecycleAuthorityError(f'external code authority path escapes the project root: {name}') + return candidate + + +def _require_external_authority_file(root, name): + path = _external_authority_expected_path(root, name) + try: + reject_reparse_components(path) + except OSError as exc: + raise LifecycleAuthorityError(f'external code authority reparse point is forbidden: {name}') from exc + try: + details = os.stat(path, follow_symlinks=False) + except OSError as exc: + raise LifecycleAuthorityError(f'required external code authority file is absent: {name}') from exc + if not stat.S_ISREG(details.st_mode): + raise LifecycleAuthorityError(f'external code authority file is not regular: {name}') + if canonical_path(path) != path: + raise LifecycleAuthorityError(f'external code authority path is not exact: {name}') + return path + + +def _application_code_files(root): + """Return the exact importable application code surface.""" + names = [] + + def raise_walk_error(exc): + raise LifecycleAuthorityError(f'unable to inspect the application root: {exc}') from exc + + for current, directories, files in os.walk(root, followlinks=False, onerror=raise_walk_error): + for name in directories: + candidate = os.path.join(current, name) + try: + linked = _is_reparse_point(candidate) + except OSError as exc: + raise LifecycleAuthorityError(f'unable to inspect application directory: {candidate}') from exc + if linked: + relative = os.path.relpath(candidate, root).replace(os.sep, '/') + raise LifecycleAuthorityError(f'application directory reparse point is forbidden: {relative}') + relative_current = os.path.relpath(current, root) + in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep)) + suffixes = ('.pyc',) if in_cache else APPLICATION_IMPORT_SUFFIXES + for name in files: + source_path = os.path.abspath(os.path.join(current, name)) + try: + linked = _is_reparse_point(source_path) + except OSError as exc: + raise LifecycleAuthorityError(f'unable to inspect application file: {source_path}') from exc + if linked: + relative = os.path.relpath(source_path, root).replace(os.sep, '/') + raise LifecycleAuthorityError(f'application file reparse point is forbidden: {relative}') + if not name.lower().endswith(suffixes): + continue + path = canonical_path(source_path) + try: + if os.path.commonpath((root, path)) != root: + raise LifecycleAuthorityError(f'code authority path escapes the application root: {path}') + except ValueError as exc: + raise LifecycleAuthorityError(f'code authority path escapes the application root: {path}') from exc + names.append(os.path.relpath(source_path, root).replace(os.sep, '/')) + return sorted(names) + + +def code_authority_file_names(app_dir=None, existing_only=False): + root = _application_root(app_dir) + names = list(_application_code_files(root)) + for name in (*CODE_AUTHORITY_FILES, *EXTERNAL_CODE_AUTHORITY_FILES): + path = canonical_path(os.path.join(root, name)) + if not existing_only or os.path.isfile(path): + names.append(name) + return tuple(dict.fromkeys(names)) + + +def resolve_manifest_executable(value, *, name=TRUFFLEHOG_MANIFEST_NAME, app_dir=None): + if name not in {TRUFFLEHOG_MANIFEST_NAME, GIT_MANIFEST_NAME}: + raise LifecycleAuthorityError('unsupported manifested executable') + label = 'TruffleHog' if name == TRUFFLEHOG_MANIFEST_NAME else 'Git' + text = str(value or '').strip() + if not text: + if name == GIT_MANIFEST_NAME: + text = 'git' + if os.name == 'nt': + private_git = os.path.join(os.path.dirname(_application_root(app_dir)), 'runtime', 'git', 'cmd', 'git.exe') + try: + reject_reparse_components(private_git) + except OSError as exc: + raise LifecycleAuthorityError('project-private Git path contains a reparse point') from exc + text = private_git if os.path.lexists(private_git) else text + else: + from paths import default_trufflehog_path + + text = default_trufflehog_path() + candidate = shutil.which(text) if not os.path.isabs(text) and not any(sep in text for sep in ('/', '\\')) else text + if not candidate: + raise LifecycleAuthorityError(f'configured {label} executable is unavailable: {text}') + if os.name != 'nt': + if not os.path.isabs(candidate) or candidate != os.path.normpath(candidate): + raise LifecycleAuthorityError(f'configured {label} executable path must be exact and absolute: {candidate}') + try: + reject_reparse_components(candidate) + except (OSError, ValueError) as exc: + raise LifecycleAuthorityError(f'configured {label} executable path is unsafe: {candidate}') from exc + path = canonical_path(candidate) + if not os.path.isfile(path): + raise LifecycleAuthorityError(f'configured {label} executable is not a regular file: {path}') + return path + + +def manifest_authority_paths( + app_dir=None, trufflehog_path=None, policy_paths=None, existing_only=False, *, + git_path=None, include_executables=True, +): + """List authority paths; exclude executables when applying private-file policy.""" + root = _application_root(app_dir) + paths = [] + names = list(code_authority_file_names(root, existing_only=existing_only)) + # Required externals may never disappear from read-only preflight or an + # offline hardening plan, even when optional paths use existing_only. + names.extend(name for name in EXTERNAL_CODE_AUTHORITY_FILES if name not in names) + for name in names: + if name in EXTERNAL_CODE_AUTHORITY_FILES: + paths.append(_require_external_authority_file(root, name)) + continue + path = canonical_path(os.path.join(root, *name.split('/'))) + if os.path.isfile(path): + paths.append(path) + elif not existing_only: + raise LifecycleAuthorityError(f'code authority file is absent or outside the application root: {name}') + if include_executables: + if trufflehog_path or not existing_only: + try: + paths.append(resolve_manifest_executable(trufflehog_path)) + except LifecycleAuthorityError: + if not existing_only or trufflehog_path: + raise + # Git is required even in preflight/offline hardening's existing-only mode. + paths.append(resolve_manifest_executable(git_path, name=GIT_MANIFEST_NAME, app_dir=root)) + for value in policy_paths or (): + if not value: + continue + path = canonical_path(value) + if os.path.isfile(path): + paths.append(path) + elif not existing_only: + raise LifecycleAuthorityError(f'configured policy authority file is unavailable: {path}') + return tuple(dict.fromkeys(paths)) + + +def build_code_manifest( + app_dir=None, trufflehog_path=None, policy_paths=None, *, git_path=None, + include_trufflehog=True, +): + root = _application_root(app_dir) + files = {} + for name in code_authority_file_names(root): + path = ( + _require_external_authority_file(root, name) + if name in EXTERNAL_CODE_AUTHORITY_FILES + else canonical_path(os.path.join(root, *name.split('/'))) + ) + try: + contained = os.path.commonpath((root, path)) == root + except ValueError: + contained = False + explicitly_external = ( + name in EXTERNAL_CODE_AUTHORITY_FILES + and path == _external_authority_expected_path(root, name) + ) + if (not contained and not explicitly_external) or not os.path.isfile(path): + raise LifecycleAuthorityError(f'code authority file is absent or outside the application root: {name}') + files[name] = {'path': path, 'sha256': sha256_file(path)} + git_executable = resolve_manifest_executable(git_path, name=GIT_MANIFEST_NAME, app_dir=root) + executables = { + GIT_MANIFEST_NAME: {'path': git_executable, 'sha256': sha256_file(git_executable)}, + } + if include_trufflehog: + executable = resolve_manifest_executable(trufflehog_path) + executables[TRUFFLEHOG_MANIFEST_NAME] = { + 'path': executable, + 'sha256': sha256_file(executable), + } + assets = {} + for value in policy_paths or (): + if not value: + continue + path = canonical_path(value) + if not os.path.isfile(path): + raise LifecycleAuthorityError(f'configured policy authority file is unavailable: {path}') + assets[path] = {'path': path, 'sha256': sha256_file(path)} + return { + 'schema': CODE_MANIFEST_SCHEMA, + 'root': root, + 'files': files, + 'executables': executables, + 'assets': assets, + } + + +def normalize_code_manifest(manifest, *, required_names=None, external_names=None): + if not isinstance(manifest, dict) or manifest.get('schema') != CODE_MANIFEST_SCHEMA: + raise LifecycleAuthorityError('unsupported code authority manifest schema') + root_value = manifest.get('root') or '' + root = _application_root(root_value) if root_value else '' + values = manifest.get('files') + expected_names = set(values) if isinstance(values, dict) else set() + external_names = set( + EXTERNAL_CODE_AUTHORITY_FILES if external_names is None else external_names + ) + if not external_names <= set(EXTERNAL_CODE_AUTHORITY_FILES): + raise LifecycleAuthorityError('code authority manifest has unsupported external files') + required_names = set(CODE_AUTHORITY_FILES if required_names is None else required_names) + required_names.update(external_names) + if not root or not isinstance(values, dict) or not required_names.issubset(expected_names): + raise LifecycleAuthorityError('code authority manifest has an incomplete file set') + files = {} + for name in sorted(expected_names): + value = values.get(name) + if not isinstance(value, dict): + raise LifecycleAuthorityError(f'invalid code authority entry: {name}') + if name in external_names: + path = os.path.normcase(os.path.abspath(os.fspath(value.get('path') or ''))) + expected_path = _external_authority_expected_path(root, name) + else: + path = canonical_path(value.get('path') or '') + expected_path = canonical_path(os.path.join(root, *name.split('/'))) + try: + contained = os.path.commonpath((root, expected_path)) == root + except ValueError: + contained = False + if not contained: + raise LifecycleAuthorityError(f'code authority path is not an allowed external: {name}') + digest = str(value.get('sha256') or '') + if path != expected_path: + raise LifecycleAuthorityError(f'code authority path mismatch: {name}') + if len(digest) != 64 or any(ch not in '0123456789abcdef' for ch in digest): + raise LifecycleAuthorityError(f'invalid code authority digest: {name}') + files[name] = {'path': path, 'sha256': digest} + executables = manifest.get('executables') + executable_names = set(executables) if isinstance(executables, dict) else set() + if executable_names not in ( + {GIT_MANIFEST_NAME}, + {TRUFFLEHOG_MANIFEST_NAME, GIT_MANIFEST_NAME}, + ): + raise LifecycleAuthorityError('code authority manifest has an incomplete executable set') + normalized_executables = {} + for name, label in ((TRUFFLEHOG_MANIFEST_NAME, 'TruffleHog'), (GIT_MANIFEST_NAME, 'Git')): + if name not in executables: + continue + executable = executables[name] + if not isinstance(executable, dict): + raise LifecycleAuthorityError(f'invalid {label} authority entry') + raw_path = executable.get('path') + executable_digest = str(executable.get('sha256') or '') + if not isinstance(raw_path, str) or not os.path.isabs(raw_path) or '\x00' in raw_path or len(executable_digest) != 64 or any(ch not in '0123456789abcdef' for ch in executable_digest): + raise LifecycleAuthorityError(f'invalid {label} authority identity') + executable_path = canonical_path(raw_path) + if os.name != 'nt' and executable_path != raw_path: + raise LifecycleAuthorityError(f'{label} authority path is not exact') + normalized_executables[name] = {'path': executable_path, 'sha256': executable_digest} + assets_value = manifest.get('assets') + if not isinstance(assets_value, dict): + raise LifecycleAuthorityError('code authority manifest has an invalid asset set') + assets = {} + for name in sorted(assets_value): + value = assets_value[name] + if not isinstance(value, dict): + raise LifecycleAuthorityError(f'invalid policy authority entry: {name}') + path = canonical_path(value.get('path') or '') + digest = str(value.get('sha256') or '') + if name != path or not os.path.isabs(path) or len(digest) != 64 or any(ch not in '0123456789abcdef' for ch in digest): + raise LifecycleAuthorityError(f'invalid policy authority identity: {name}') + assets[name] = {'path': path, 'sha256': digest} + return { + 'schema': CODE_MANIFEST_SCHEMA, + 'root': root, + 'files': files, + 'executables': normalized_executables, + 'assets': assets, + } + + +def verify_code_manifest( + manifest, expected_sha256=None, require_private_acl=False, *, + required_names=None, external_names=None, +): + normalized = normalize_code_manifest( + manifest, required_names=required_names, external_names=external_names, + ) + digest = code_manifest_sha256(normalized) + if expected_sha256 and not hmac.compare_digest(digest, str(expected_sha256)): + raise LifecycleAuthorityError('code authority manifest digest mismatch') + for name in (EXTERNAL_CODE_AUTHORITY_FILES if external_names is None else external_names): + if normalized['files'][name]['path'] != _require_external_authority_file(normalized['root'], name): + raise LifecycleAuthorityError(f'code authority path mismatch: {name}') + current_code = set(_application_code_files(normalized['root'])) + manifested_code = { + name for name, value in normalized['files'].items() + if name.lower().endswith(APPLICATION_IMPORT_SUFFIXES) + and os.path.commonpath((normalized['root'], value['path'])) == normalized['root'] + } + if current_code != manifested_code: + added = sorted(current_code - manifested_code) + removed = sorted(manifested_code - current_code) + detail = added[0] if added else removed[0] if removed else 'unknown' + raise LifecycleAuthorityError(f'application code authority file set drifted: {detail}') + entries = [(name, value, False) for name, value in normalized['files'].items()] + [ + (f'executable:{name}', value, True) for name, value in normalized['executables'].items() + ] + [(f'asset:{name}', value, False) for name, value in normalized['assets'].items()] + for name, value, native in entries: + if require_private_acl: + if native and os.name != 'nt': + try: + require_trusted_native_executable(value['path']) + except (OSError, ValueError) as exc: + raise LifecycleAuthorityError(f'code authority executable is not trusted: {name}: {exc}') from exc + elif not private_file_ready(value['path']): + raise LifecycleAuthorityError(f'code authority ACL is not exact-private: {name}') + try: + current = sha256_file(value['path']) + except OSError as exc: + raise LifecycleAuthorityError(f'unable to verify code authority file: {name}') from exc + if not hmac.compare_digest(current, value['sha256']): + raise LifecycleAuthorityError(f'code authority drifted: {name}') + return normalized + + +def dsn_sha256(dsn): + value = str(dsn or '') + return hashlib.sha256(value.encode('utf-8')).hexdigest() if value else '' + + +def supervised_child_environment(metadata, canonical_dsn, child_kind): + return { + 'SCANNER_SUPERVISED': '1', + CHILD_INSTANCE_FILE_ENV: str(metadata['instance_file']), + CHILD_INSTANCE_ID_ENV: str(metadata['instance_id']), + CHILD_TOKEN_ENV: str(metadata['token']), + CHILD_CONFIG_HASH_ENV: str(metadata['config_sha256']), + CHILD_SCRIPT_HASH_ENV: str(metadata['supervisor_sha256']), + CHILD_MANIFEST_HASH_ENV: str(metadata['code_manifest_sha256']), + CHILD_DSN_HASH_ENV: str(metadata.get('canonical_dsn_sha256') or ''), + CHILD_KIND_ENV: str(child_kind), + 'TRUF_MANAGED_POSTGRES_DSN': str(canonical_dsn or ''), + } + + +def strip_supervisor_credentials(env): + for key in list(env): + normalized = str(key).upper() + if normalized.startswith('TRUF_POSTGRES_') or normalized in _PRIVATE_EXTERNAL_ENV_KEYS: + env.pop(key, None) + return env + + +def _send_handshake(metadata, timeout=3): + control = metadata.get('control') or {} + request = { + 'schema': CONTROL_SCHEMA, + 'instance_id': metadata['instance_id'], + 'token': metadata['token'], + 'action': 'handshake', + } + payload = json.dumps(request, ensure_ascii=True, separators=(',', ':')).encode('utf-8') + b'\n' + with socket.create_connection((control.get('host'), int(control.get('port') or 0)), timeout=timeout) as sock: + sock.settimeout(timeout) + sock.sendall(payload) + sock.shutdown(socket.SHUT_WR) + chunks = [] + total = 0 + while True: + chunk = sock.recv(65536) + if not chunk: + break + total += len(chunk) + if total > 1024 * 1024: + raise LifecycleAuthorityError('supervisor handshake response is too large') + chunks.append(chunk) + try: + response = json.loads(b''.join(chunks).decode('utf-8')) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise LifecycleAuthorityError('invalid supervisor handshake response') from exc + if ( + not isinstance(response, dict) + or response.get('schema') != CONTROL_SCHEMA + or response.get('instance_id') != metadata['instance_id'] + or response.get('ok') is not True + or not isinstance(response.get('result'), dict) + ): + raise LifecycleAuthorityError('authenticated supervisor handshake failed') + return response['result'] + + +def _send_handshake_with_timeout_retry(metadata, timeout_retries=0): + retries = min(1, max(0, int(timeout_retries or 0))) + for attempt in range(retries + 1): + try: + return _send_handshake(metadata) + except TimeoutError: + if attempt >= retries: + raise + + +def verify_supervisor_command_line(arguments, supervisor_path, config_path): + """Verify the unique script binding and config value retained by the OS.""" + arguments = [str(argument) for argument in arguments] + options = { + '--runtime-bootstrap-entrypoint': [], + '--config': [], + } + option_value_indices = set() + for index, argument in enumerate(arguments): + for option in options: + if argument == option: + value = arguments[index + 1] if index + 1 < len(arguments) else '' + options[option].append(value) + if index + 1 < len(arguments): + option_value_indices.add(index + 1) + elif argument.startswith(option + '='): + options[option].append(argument.split('=', 1)[1]) + + expected_supervisor = canonical_path(supervisor_path) + direct_bindings = [] + for index, argument in enumerate(arguments): + if index in option_value_indices or not argument or argument.startswith('-'): + continue + try: + if canonical_path(argument) == expected_supervisor: + direct_bindings.append(argument) + except (OSError, TypeError, ValueError): + continue + + bindings = options['--runtime-bootstrap-entrypoint'] + direct_bindings + if len(bindings) != 1: + raise LifecycleAuthorityError('supervisor command line must contain exactly one bound supervisor script') + try: + binding_matches = canonical_path(bindings[0]) == expected_supervisor + except (OSError, TypeError, ValueError): + binding_matches = False + if not binding_matches: + raise LifecycleAuthorityError('supervisor command line bound supervisor script mismatch') + + configs = options['--config'] + if len(configs) != 1: + raise LifecycleAuthorityError('supervisor command line must contain exactly one bound config argument') + try: + config_matches = canonical_path(configs[0]) == canonical_path(config_path) + except (OSError, TypeError, ValueError): + config_matches = False + if not config_matches: + raise LifecycleAuthorityError('supervisor command line bound config argument mismatch') + + +def _verify_supervisor_process(metadata): + process = verify_retained_process( + metadata['pid'], + metadata['process_creation_time'], + metadata['executable'], + ) + try: + arguments = process.command_line() + verify_supervisor_command_line( + arguments, + metadata['supervisor_path'], + metadata['config_path'], + ) + finally: + process.close() + + +def require_active_supervisor_child( + config_path=None, child_kind=None, require_dsn=True, handshake_timeout_retries=0, +): + """Authenticate a mutating child before it reads application inputs.""" + instance_file = os.getenv(CHILD_INSTANCE_FILE_ENV) or '' + inherited_id = os.getenv(CHILD_INSTANCE_ID_ENV) or '' + inherited_token = os.getenv(CHILD_TOKEN_ENV) or '' + inherited_config_hash = os.getenv(CHILD_CONFIG_HASH_ENV) or '' + inherited_script_hash = os.getenv(CHILD_SCRIPT_HASH_ENV) or '' + inherited_manifest_hash = os.getenv(CHILD_MANIFEST_HASH_ENV) or '' + inherited_dsn_hash = os.getenv(CHILD_DSN_HASH_ENV) or '' + inherited_kind = os.getenv(CHILD_KIND_ENV) or '' + if not all((instance_file, inherited_id, inherited_token, inherited_config_hash, inherited_script_hash, inherited_manifest_hash)): + raise LifecycleAuthorityError('direct mutation is retired; use an authenticated active supervisor command') + if child_kind and inherited_kind != str(child_kind): + raise LifecycleAuthorityError('supervised child kind does not match the requested mutation entrypoint') + + # Imported lazily to avoid a module cycle while supervisor metadata support + # itself imports the manifest helpers above. + from supervisor_instance import load_instance_metadata + + try: + metadata = load_instance_metadata(instance_file) + except (OSError, ValueError) as exc: + raise LifecycleAuthorityError('private supervisor instance metadata is unavailable') from exc + if not hmac.compare_digest(metadata['instance_id'], inherited_id): + raise LifecycleAuthorityError('supervisor child instance identity mismatch') + if not hmac.compare_digest(metadata['token'], inherited_token): + raise LifecycleAuthorityError('supervisor child credential mismatch') + if metadata.get('activation_state') != PHASE_ACTIVE: + raise LifecycleAuthorityError('supervisor is not ACTIVE; mutation is refused') + if canonical_path(instance_file) != canonical_path(metadata.get('instance_file') or instance_file): + raise LifecycleAuthorityError('supervisor child instance path mismatch') + if config_path and canonical_path(config_path) != metadata['config_path']: + raise LifecycleAuthorityError('supervisor child config path mismatch') + + expected_pairs = ( + ('config_sha256', inherited_config_hash), + ('supervisor_sha256', inherited_script_hash), + ('code_manifest_sha256', inherited_manifest_hash), + ) + for key, inherited in expected_pairs: + if not hmac.compare_digest(str(metadata.get(key) or ''), inherited): + raise LifecycleAuthorityError(f'supervisor child {key} authority mismatch') + if not hmac.compare_digest(sha256_file(metadata['config_path']), metadata['config_sha256']): + raise LifecycleAuthorityError('supervisor config authority drifted') + verify_code_manifest( + metadata['code_manifest'], + metadata['code_manifest_sha256'], + require_private_acl=True, + ) + _verify_supervisor_process(metadata) + + dsn = os.getenv('TRUF_MANAGED_POSTGRES_DSN') or '' + if require_dsn: + try: + parsed = parse_postgres_url(dsn) + except ValueError as exc: + raise LifecycleAuthorityError('a canonical managed PostgreSQL DSN is required') from exc + if parsed['host'] != '127.0.0.1': + raise LifecycleAuthorityError('managed PostgreSQL DSN is not loopback-bound') + actual_dsn_hash = dsn_sha256(dsn) + if not inherited_dsn_hash or not hmac.compare_digest(actual_dsn_hash, inherited_dsn_hash): + raise LifecycleAuthorityError('managed PostgreSQL DSN authority mismatch') + if not hmac.compare_digest(str(metadata.get('canonical_dsn_sha256') or ''), inherited_dsn_hash): + raise LifecycleAuthorityError('private metadata PostgreSQL DSN authority mismatch') + for key in ('SCANNER_DB_URL', 'DATABASE_URL'): + if not hmac.compare_digest(str(os.getenv(key) or ''), dsn): + raise LifecycleAuthorityError(f'{key} does not match the managed PostgreSQL DSN') + if any(key.upper().startswith('PG') for key in os.environ): + raise LifecycleAuthorityError('libpq PG* environment overrides are forbidden for managed children') + + handshake = _send_handshake_with_timeout_retry(metadata, handshake_timeout_retries) + if handshake.get('instance_id') != metadata['instance_id'] or handshake.get('activation_state') != PHASE_ACTIVE: + raise LifecycleAuthorityError('supervisor handshake did not confirm ACTIVE authority') + for key, value in expected_pairs: + if not hmac.compare_digest(str(handshake.get(key) or ''), value): + raise LifecycleAuthorityError(f'supervisor handshake {key} mismatch') + if require_dsn and not hmac.compare_digest(str(handshake.get('canonical_dsn_sha256') or ''), inherited_dsn_hash): + raise LifecycleAuthorityError('supervisor handshake PostgreSQL DSN authority mismatch') + return metadata diff --git a/app/managed_files.py b/app/managed_files.py new file mode 100644 index 0000000..759583b --- /dev/null +++ b/app/managed_files.py @@ -0,0 +1,1440 @@ +from contextlib import contextmanager +import ctypes +from dataclasses import dataclass, field +from enum import Enum +import errno +import hashlib +import hmac +import os +import posixpath +import re +import secrets +import stat +import sys +import tempfile +import threading + +try: + import fcntl +except ImportError: # pragma: no cover - non-Linux import portability + fcntl = None + + +MAX_MANAGED_ROOTS = 8 +MAX_MANAGED_ROOT_ID_BYTES = 64 +MAX_MANAGED_ROOT_PATH_BYTES = 4096 +MAX_RELATIVE_PATH_BYTES = 4096 +MAX_COMPONENT_BYTES = 255 +MAX_PATH_DEPTH = 32 +MAX_LISTING_ENTRIES = 1000 +MAX_LISTING_BYTES = 1024 * 1024 +MAX_FILE_BYTES = 64 * 1024 * 1024 +MAX_RESULT_FILE_BYTES = 256 * 1024 * 1024 + +RUNTIME_LOG_ROOT_ID = 'runtime-logs' +RUNTIME_LOG_ROOT_PATH = '/data/runtime-linux/logs' +RUNTIME_KEYCHECK_ROOT_ID = 'runtime-keychecks' +RUNTIME_KEYCHECK_ROOT_PATH = '/data/runtime-linux/keychecks' +RUNTIME_RESULT_ROOT_ID = 'runtime-results' +RUNTIME_RESULT_ROOT_PATH = '/data/runtime-linux/results' +MANAGED_DATA_ROOT = '/data/managed-files' + +_PREDEFINED_READ_ONLY_ROOTS = { + RUNTIME_KEYCHECK_ROOT_ID: RUNTIME_KEYCHECK_ROOT_PATH, + RUNTIME_LOG_ROOT_ID: RUNTIME_LOG_ROOT_PATH, + RUNTIME_RESULT_ROOT_ID: RUNTIME_RESULT_ROOT_PATH, +} + +_ROOT_ID = re.compile(r'^[a-z][a-z0-9-]{0,63}$') +_DRIVE_PATH = re.compile(r'^/?[A-Za-z]:') +_DRIVE_COMPONENT = re.compile(r'^[A-Za-z]:') +_HASH = re.compile(r'^[0-9a-f]{64}$') +_RESULT_PROJECTION_FILE = re.compile( + r'^(?:found_secrets|scan_results)(?:\.g[0-9]{6})?\.jsonl$', +) +_TEMPORARY_PREFIX = '.truf-managed-file-' +_TEMPORARY_ATTEMPTS = 16 +_READ_CHUNK_BYTES = 64 * 1024 +_RENAME_NOREPLACE = 1 +_PERMISSION_KEYS = frozenset({'list', 'read', 'create_replace', 'delete'}) +_LIMIT_BOUNDS = { + 'max_relative_path_bytes': MAX_RELATIVE_PATH_BYTES, + 'max_component_bytes': MAX_COMPONENT_BYTES, + 'max_path_depth': MAX_PATH_DEPTH, + 'max_listing_entries': MAX_LISTING_ENTRIES, + 'max_listing_bytes': MAX_LISTING_BYTES, + 'max_file_bytes': MAX_FILE_BYTES, +} + + +class ManagedFileConfigurationError(ValueError): + def __init__(self, category, field): + self.category = category + self.field = tuple(field) + super().__init__('managed file root configuration is invalid') + + +class ManagedFileAccessError(RuntimeError): + def __init__(self, category): + self.category = category + super().__init__('managed file access failed') + + +class ManagedFileOperation(str, Enum): + LIST = 'list' + READ = 'read' + CREATE_REPLACE = 'create-replace' + DELETE = 'delete' + + +@dataclass(frozen=True) +class ManagedFilePermissions: + allow_list: bool + allow_read: bool + allow_create_replace: bool + allow_delete: bool + + def allows(self, operation): + if operation is ManagedFileOperation.LIST: + return self.allow_list + if operation is ManagedFileOperation.READ: + return self.allow_read + if operation is ManagedFileOperation.CREATE_REPLACE: + return self.allow_create_replace + if operation is ManagedFileOperation.DELETE: + return self.allow_delete + return False + + +@dataclass(frozen=True) +class ManagedFileLimits: + max_relative_path_bytes: int + max_component_bytes: int + max_path_depth: int + max_listing_entries: int + max_listing_bytes: int + max_file_bytes: int + + +@dataclass(frozen=True) +class ManagedFileRoot: + root_id: str + absolute_path: str = field(repr=False) + permissions: ManagedFilePermissions + limits: ManagedFileLimits + + +@dataclass(frozen=True) +class ManagedFileRootRegistry: + roots: tuple[ManagedFileRoot, ...] = () + + def get(self, root_id): + if type(root_id) is not str: + return None + return next((root for root in self.roots if root.root_id == root_id), None) + + def root_ids(self): + return tuple(root.root_id for root in self.roots) + + +@dataclass(frozen=True) +class ManagedFileIdentity: + sha256: str + byte_count: int + + +@dataclass(frozen=True) +class ManagedFileDirectoryEntry: + name: str + kind: str + byte_count: int | None + + +@dataclass(frozen=True) +class ManagedFileListing: + entries: tuple[ManagedFileDirectoryEntry, ...] + name_bytes: int + + +class ManagedFileSnapshot: + __slots__ = ('_handle', '_release', '_lock') + + def __init__(self, handle, release): + self._handle = handle + self._release = release + self._lock = threading.Lock() + + def chunks(self): + try: + while True: + with self._lock: + handle = self._handle + chunk = None if handle is None else handle.read(_READ_CHUNK_BYTES) + if not chunk: + break + yield chunk + finally: + self.close() + + def close(self): + with self._lock: + handle = self._handle + release = self._release + self._handle = None + self._release = None + if handle is not None: + try: + handle.close() + except OSError: + pass + if release is not None: + release(self) + + def __repr__(self): + return '' + + +@dataclass(frozen=True) +class ManagedFileDownload: + identity: ManagedFileIdentity + content: bytes | None = field(repr=False, default=None) + snapshot: ManagedFileSnapshot | None = field(repr=False, default=None) + + def __post_init__(self): + if (self.content is None) == (self.snapshot is None): + raise ValueError('managed file download requires one body source') + + +@dataclass(frozen=True) +class ManagedFileMutation: + before: ManagedFileIdentity | None + after: ManagedFileIdentity | None + written: bool + + +class ManagedFileOpenedTarget: + __slots__ = ('_descriptor', '_details') + + def __init__(self, descriptor, details): + self._descriptor = descriptor + self._details = details + + def fileno(self): + if self._descriptor is None: + raise ManagedFileAccessError('closed') + return self._descriptor + + @property + def details(self): + return self._details + + def _close(self): + descriptor = self._descriptor + self._descriptor = None + if descriptor is not None: + try: + os.close(descriptor) + except OSError: + pass + + def __repr__(self): + return '' + + +def _error(category, *field): + raise ManagedFileConfigurationError(category, ('root',) + field) + + +def _absolute_root_path(value): + if type(value) is not str: + _error('type', 'path') + try: + encoded = value.encode('utf-8') + except UnicodeEncodeError: + _error('bounds', 'path') + if ( + not value or len(encoded) > MAX_MANAGED_ROOT_PATH_BYTES + or '\\' in value or '\x00' in value or _DRIVE_PATH.match(value) + or not value.startswith('/') or value == '/' or value.endswith('/') + or '//' in value + ): + _error('deployment_path', 'path') + components = value.split('/')[1:] + if any( + component in ('', '.', '..') + or len(component.encode('utf-8')) > MAX_COMPONENT_BYTES + for component in components + ) or posixpath.normpath(value) != value: + _error('deployment_path', 'path') + return value + + +def _permissions(value): + if type(value) is not dict: + _error('type', 'permissions') + if set(value) != _PERMISSION_KEYS: + category = 'unknown_key' if set(value) - _PERMISSION_KEYS else 'schema' + _error(category, 'permissions') + if any(type(value[name]) is not bool for name in _PERMISSION_KEYS): + _error('type', 'permissions') + return ManagedFilePermissions( + allow_list=value['list'], + allow_read=value['read'], + allow_create_replace=value['create_replace'], + allow_delete=value['delete'], + ) + + +def _limits(value, root_id): + if type(value) is not dict: + _error('type', 'limits') + expected = set(_LIMIT_BOUNDS) + if set(value) != expected: + category = 'unknown_key' if set(value) - expected else 'schema' + _error(category, 'limits') + for name, maximum in _LIMIT_BOUNDS.items(): + if name == 'max_file_bytes' and root_id == RUNTIME_RESULT_ROOT_ID: + maximum = MAX_RESULT_FILE_BYTES + if type(value[name]) is not int or not 1 <= value[name] <= maximum: + _error('bounds', 'limits', name) + if ( + value['max_component_bytes'] > value['max_relative_path_bytes'] + or value['max_listing_bytes'] < value['max_component_bytes'] + ): + _error('bounds', 'limits') + return ManagedFileLimits(**{name: value[name] for name in _LIMIT_BOUNDS}) + + +def _allowed_root(root_id, path, permissions): + predefined_path = _PREDEFINED_READ_ONLY_ROOTS.get(root_id) + predefined_id = next(( + candidate_id for candidate_id, candidate_path + in _PREDEFINED_READ_ONLY_ROOTS.items() + if candidate_path == path + ), None) + if predefined_path is not None or predefined_id is not None: + if predefined_path != path or permissions != ManagedFilePermissions( + allow_list=True, + allow_read=True, + allow_create_replace=False, + allow_delete=False, + ): + _error('deployment_path', 'path') + return + parent, name = posixpath.split(path) + if parent != MANAGED_DATA_ROOT or not name: + _error('deployment_path', 'path') + + +def normalize_managed_file_roots(value): + if type(value) is not dict: + raise ManagedFileConfigurationError('type', ('root',)) + if len(value) > MAX_MANAGED_ROOTS: + raise ManagedFileConfigurationError('bounds', ('root',)) + + if any(type(root_id) is not str for root_id in value): + _error('bounds', 'id') + normalized = {} + roots = [] + paths = set() + for root_id in sorted(value): + if ( + type(root_id) is not str or not _ROOT_ID.fullmatch(root_id) + or len(root_id.encode('utf-8')) > MAX_MANAGED_ROOT_ID_BYTES + ): + _error('bounds', 'id') + raw = value[root_id] + if type(raw) is not dict: + _error('type') + expected = {'path', 'permissions', 'limits'} + if set(raw) != expected: + category = 'unknown_key' if set(raw) - expected else 'schema' + _error(category) + path = _absolute_root_path(raw['path']) + permissions = _permissions(raw['permissions']) + limits = _limits(raw['limits'], root_id) + _allowed_root(root_id, path, permissions) + if path in paths: + _error('deployment_path', 'path') + paths.add(path) + root = ManagedFileRoot(root_id, path, permissions, limits) + roots.append(root) + normalized[root_id] = { + 'path': path, + 'permissions': { + 'list': permissions.allow_list, + 'read': permissions.allow_read, + 'create_replace': permissions.allow_create_replace, + 'delete': permissions.allow_delete, + }, + 'limits': {name: getattr(limits, name) for name in _LIMIT_BOUNDS}, + } + return normalized, ManagedFileRootRegistry(tuple(roots)) + + +def managed_file_root_registry_from_config(config): + if type(config) is not dict: + raise ManagedFileConfigurationError('type', ('root',)) + supervisor = config.get('supervisor', {}) + if type(supervisor) is not dict: + raise ManagedFileConfigurationError('type', ('root',)) + worker_api = supervisor.get('worker_api', {}) + if type(worker_api) is not dict: + raise ManagedFileConfigurationError('type', ('root',)) + admin = worker_api.get('admin', {}) + if admin is None: + admin = {} + if type(admin) is not dict: + raise ManagedFileConfigurationError('type', ('root',)) + return normalize_managed_file_roots(admin.get('managed_file_roots', {}))[1] + + +def parse_managed_relative_path(value, limits): + if type(value) is not str or not isinstance(limits, ManagedFileLimits): + raise ManagedFileAccessError('invalid_path') + try: + encoded = value.encode('utf-8') + except UnicodeEncodeError: + raise ManagedFileAccessError('invalid_path') from None + if ( + not value or len(encoded) > limits.max_relative_path_bytes + or value.startswith('/') or '\\' in value or '\x00' in value + or _DRIVE_PATH.match(value) + ): + raise ManagedFileAccessError('invalid_path') + components = value.split('/') + if ( + len(components) > limits.max_path_depth + or any( + component in ('', '.', '..') + or _DRIVE_COMPONENT.match(component) + or component.startswith(_TEMPORARY_PREFIX) + or len(component.encode('utf-8')) > limits.max_component_bytes + for component in components + ) + ): + raise ManagedFileAccessError('invalid_path') + return tuple(components) + + +def _close_descriptor(descriptor): + if descriptor is None: + return + try: + os.close(descriptor) + except OSError: + pass + + +def _descriptor_flags(): + required = ( + 'O_PATH', 'O_DIRECTORY', 'O_NOFOLLOW', 'O_CLOEXEC', + 'O_NONBLOCK', 'O_NOCTTY', 'O_CREAT', 'O_EXCL', + ) + if ( + not sys.platform.startswith('linux') + or os.open not in getattr(os, 'supports_dir_fd', ()) + or any(type(getattr(os, name, None)) is not int for name in required) + ): + raise ManagedFileAccessError('filesystem_unavailable') + return { + 'directory': ( + os.O_PATH | os.O_DIRECTORY | os.O_NOFOLLOW | os.O_CLOEXEC + ), + 'list': os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW | os.O_CLOEXEC, + 'inspect': os.O_PATH | os.O_NOFOLLOW | os.O_CLOEXEC, + 'read': os.O_RDONLY | os.O_NONBLOCK | os.O_NOCTTY | os.O_CLOEXEC, + 'temporary': ( + os.O_RDWR | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW | os.O_CLOEXEC + ), + } + + +def _filesystem_error_category(exc): + if exc.errno == errno.ENOENT: + return 'not_found' + if exc.errno in { + errno.ELOOP, errno.ENOTDIR, errno.EISDIR, errno.ENXIO, + errno.ENODEV, errno.EMLINK, + getattr(errno, 'EOPNOTSUPP', -1), + getattr(errno, 'ENOTSUP', -1), + }: + return 'unsafe_target' + return 'filesystem_unavailable' + + +def _file_revision(details): + return ( + details.st_dev, + details.st_ino, + stat.S_IFMT(details.st_mode), + details.st_nlink, + details.st_size, + getattr(details, 'st_mtime_ns', None), + getattr(details, 'st_ctime_ns', None), + ) + + +def _regular_single_link(details): + return stat.S_ISREG(details.st_mode) and details.st_nlink == 1 + + +def _valid_hash(value): + if type(value) is not str or not _HASH.fullmatch(value): + raise ManagedFileAccessError('invalid_hash') + return value + + +def _rename_noreplace(source, target, directory_descriptor): + try: + function = ctypes.CDLL(None, use_errno=True).renameat2 + function.argtypes = ( + ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, + ctypes.c_uint, + ) + function.restype = ctypes.c_int + ctypes.set_errno(0) + result = function( + directory_descriptor, source.encode('utf-8'), + directory_descriptor, target.encode('utf-8'), + _RENAME_NOREPLACE, + ) + except (AttributeError, OSError, TypeError, ValueError): + raise ManagedFileAccessError('filesystem_unavailable') from None + if result == 0: + return + error_number = ctypes.get_errno() + if error_number == errno.EEXIST: + raise ManagedFileAccessError('hash_conflict') + raise ManagedFileAccessError('filesystem_unavailable') + + +class ManagedFileTraversal: + def __init__(self, registry): + if not isinstance(registry, ManagedFileRootRegistry): + raise ValueError('managed file root registry is invalid') + if any(not isinstance(root, ManagedFileRoot) for root in registry.roots): + raise ValueError('managed file root registry is invalid') + root_ids = tuple(root.root_id for root in registry.roots) + try: + roots_are_valid = all( + _ROOT_ID.fullmatch(root.root_id) + and _absolute_root_path(root.absolute_path) == root.absolute_path + and isinstance(root.permissions, ManagedFilePermissions) + and isinstance(root.limits, ManagedFileLimits) + for root in registry.roots + ) + except (AttributeError, ManagedFileConfigurationError, TypeError): + roots_are_valid = False + if not roots_are_valid or len(root_ids) != len(set(root_ids)): + raise ValueError('managed file root registry is invalid') + + self._registry = registry + self._lock = threading.Lock() + self._mutation_lock = threading.Lock() + self._snapshot_gate = threading.BoundedSemaphore(1) + self._snapshots = set() + self._closed = False + self._roots = {} + self._flags = None + if not registry.roots: + return + + self._flags = _descriptor_flags() + try: + for root in registry.roots: + self._roots[root.root_id] = self._open_root(root.absolute_path) + except BaseException: + descriptors = tuple(self._roots.values()) + self._roots.clear() + self._closed = True + for descriptor in descriptors: + _close_descriptor(descriptor) + raise + + def _open_root(self, absolute_path): + current = None + try: + current = os.open('/', self._flags['directory']) + if not stat.S_ISDIR(os.fstat(current).st_mode): + raise ManagedFileAccessError('root_unavailable') + for component in absolute_path.split('/')[1:]: + child = None + try: + child = os.open( + component, self._flags['directory'], dir_fd=current, + ) + if not stat.S_ISDIR(os.fstat(child).st_mode): + raise ManagedFileAccessError('root_unavailable') + except BaseException: + _close_descriptor(child) + raise + previous = current + current = child + child = None + _close_descriptor(previous) + descriptor = current + current = None + return descriptor + except OSError: + raise ManagedFileAccessError('root_unavailable') from None + finally: + _close_descriptor(current) + + def close(self): + with self._lock: + if self._closed: + return + self._closed = True + descriptors = tuple(self._roots.values()) + snapshots = tuple(self._snapshots) + self._roots.clear() + self._snapshots.clear() + for descriptor in descriptors: + _close_descriptor(descriptor) + for snapshot in snapshots: + snapshot.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc_value, traceback): + self.close() + + def _root_for_operation(self, root_id, operation): + root = self._registry.get(root_id) + if root is None: + raise ManagedFileAccessError('unknown_root') + if not root.permissions.allows(operation): + raise ManagedFileAccessError('operation_not_allowed') + return root + + def _duplicate_root(self, root_id): + with self._lock: + if self._closed: + raise ManagedFileAccessError('closed') + descriptor = self._roots.get(root_id) + if descriptor is None: + raise ManagedFileAccessError('root_unavailable') + try: + return os.dup(descriptor) + except OSError: + raise ManagedFileAccessError('root_unavailable') from None + + @staticmethod + def _relative_target_allowed(root_id, components, target_kind): + if root_id != RUNTIME_RESULT_ROOT_ID: + return True + if target_kind == 'directory': + return not components + return ( + len(components) == 1 + and _RESULT_PROJECTION_FILE.fullmatch(components[0]) is not None + ) + + def _require_relative_target(self, root_id, components, target_kind): + if not self._relative_target_allowed(root_id, components, target_kind): + raise ManagedFileAccessError('not_found') + + def _snapshot_closed(self, snapshot): + with self._lock: + self._snapshots.discard(snapshot) + self._snapshot_gate.release() + + def _new_snapshot_file(self, root_id): + root_descriptor = self._duplicate_root(root_id) + try: + return tempfile.TemporaryFile( + mode='w+b', dir=f'/proc/self/fd/{root_descriptor}', + ) + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + finally: + _close_descriptor(root_descriptor) + + @staticmethod + def _write_snapshot(handle, payload): + view = memoryview(payload) + written = 0 + while written < len(view): + try: + count = handle.write(view[written:]) + except InterruptedError: + continue + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + if not count: + raise ManagedFileAccessError('filesystem_unavailable') + written += count + + def _snapshot_descriptor(self, root, descriptor): + handle = None + try: + before = os.fstat(descriptor) + if not _regular_single_link(before): + raise ManagedFileAccessError('unsafe_target') + if before.st_size > root.limits.max_file_bytes: + raise ManagedFileAccessError('limit_exceeded') + handle = self._new_snapshot_file(root.root_id) + os.lseek(descriptor, 0, os.SEEK_SET) + digest = hashlib.sha256() + byte_count = 0 + while True: + remaining = root.limits.max_file_bytes - byte_count + if remaining < 0: + raise ManagedFileAccessError('limit_exceeded') + try: + chunk = os.read( + descriptor, min(_READ_CHUNK_BYTES, remaining + 1), + ) + except InterruptedError: + continue + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + if not chunk: + break + byte_count += len(chunk) + if byte_count > root.limits.max_file_bytes: + raise ManagedFileAccessError('limit_exceeded') + digest.update(chunk) + self._write_snapshot(handle, chunk) + after = os.fstat(descriptor) + if ( + not _regular_single_link(after) + or _file_revision(before) != _file_revision(after) + or byte_count != before.st_size + ): + raise ManagedFileAccessError('concurrent_change') + handle.flush() + handle.seek(0) + with self._lock: + if self._closed: + raise ManagedFileAccessError('closed') + snapshot = ManagedFileSnapshot(handle, self._snapshot_closed) + self._snapshots.add(snapshot) + handle = None + return ManagedFileIdentity(digest.hexdigest(), byte_count), snapshot + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + finally: + if handle is not None: + try: + handle.close() + except OSError: + pass + + def _open_relative(self, root, components, target_kind): + current = self._duplicate_root(root.root_id) + leaf = None + try: + for component in components[:-1]: + child = None + try: + child = os.open( + component, self._flags['directory'], dir_fd=current, + ) + if not stat.S_ISDIR(os.fstat(child).st_mode): + raise ManagedFileAccessError('unsafe_target') + except BaseException: + _close_descriptor(child) + raise + previous = current + current = child + child = None + _close_descriptor(previous) + + leaf_name = components[-1] if components else '.' + if target_kind == 'directory': + leaf = os.open(leaf_name, self._flags['list'], dir_fd=current) + details = os.fstat(leaf) + safe = stat.S_ISDIR(details.st_mode) + else: + leaf, details = self._open_file_at(current, leaf_name) + safe = True + if not safe: + raise ManagedFileAccessError('unsafe_target') + target = ManagedFileOpenedTarget(leaf, details) + leaf = None + return target + except OSError as exc: + raise ManagedFileAccessError(_filesystem_error_category(exc)) from None + finally: + _close_descriptor(leaf) + _close_descriptor(current) + + def _open_file_at(self, directory_descriptor, leaf_name): + inspected = readable = None + try: + inspected = os.open( + leaf_name, self._flags['inspect'], dir_fd=directory_descriptor, + ) + inspected_details = os.fstat(inspected) + if not _regular_single_link(inspected_details): + raise ManagedFileAccessError('unsafe_target') + try: + readable = os.open( + f'/proc/self/fd/{inspected}', self._flags['read'], + ) + readable_details = os.fstat(readable) + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + if ( + not _regular_single_link(readable_details) + or readable_details.st_dev != inspected_details.st_dev + or readable_details.st_ino != inspected_details.st_ino + ): + raise ManagedFileAccessError('unsafe_target') + result = readable + readable = None + return result, readable_details + finally: + _close_descriptor(readable) + _close_descriptor(inspected) + + @contextmanager + def _opened_parent(self, root, components): + current = operational = None + try: + current = self._duplicate_root(root.root_id) + for component in components[:-1]: + child = None + try: + child = os.open( + component, self._flags['directory'], dir_fd=current, + ) + if not stat.S_ISDIR(os.fstat(child).st_mode): + raise ManagedFileAccessError('unsafe_target') + except BaseException: + _close_descriptor(child) + raise + previous = current + current = child + child = None + _close_descriptor(previous) + anchored = os.fstat(current) + operational = os.open('.', self._flags['list'], dir_fd=current) + opened = os.fstat(operational) + if ( + not stat.S_ISDIR(opened.st_mode) + or opened.st_dev != anchored.st_dev + or opened.st_ino != anchored.st_ino + ): + raise ManagedFileAccessError('unsafe_target') + descriptor = operational + operational = None + try: + yield descriptor, components[-1] + finally: + _close_descriptor(descriptor) + except OSError as exc: + raise ManagedFileAccessError(_filesystem_error_category(exc)) from None + finally: + _close_descriptor(operational) + _close_descriptor(current) + + def _read_descriptor(self, descriptor, max_bytes, *, include_content): + payload = bytearray() if include_content else None + try: + before = os.fstat(descriptor) + if not _regular_single_link(before): + raise ManagedFileAccessError('unsafe_target') + if before.st_size > max_bytes: + raise ManagedFileAccessError('limit_exceeded') + try: + os.lseek(descriptor, 0, os.SEEK_SET) + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + digest = hashlib.sha256() + byte_count = 0 + while True: + remaining = max_bytes - byte_count + if remaining < 0: + raise ManagedFileAccessError('limit_exceeded') + try: + chunk = os.read( + descriptor, min(_READ_CHUNK_BYTES, remaining + 1), + ) + except InterruptedError: + continue + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + if not chunk: + break + byte_count += len(chunk) + if byte_count > max_bytes: + raise ManagedFileAccessError('limit_exceeded') + digest.update(chunk) + if payload is not None: + payload.extend(chunk) + after = os.fstat(descriptor) + if ( + not _regular_single_link(after) + or _file_revision(before) != _file_revision(after) + or byte_count != before.st_size + ): + raise ManagedFileAccessError('concurrent_change') + identity = ManagedFileIdentity(digest.hexdigest(), byte_count) + content = bytes(payload) if payload is not None else None + if payload is not None: + payload.clear() + return identity, before, content + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + except BaseException: + if payload is not None: + payload.clear() + raise + + def _read_file_at(self, directory_descriptor, leaf_name, max_bytes, *, content): + descriptor = None + try: + descriptor, details = self._open_file_at( + directory_descriptor, leaf_name, + ) + identity, stable, payload = self._read_descriptor( + descriptor, max_bytes, include_content=content, + ) + if _file_revision(details) != _file_revision(stable): + raise ManagedFileAccessError('concurrent_change') + return identity, stable, payload + except OSError as exc: + raise ManagedFileAccessError(_filesystem_error_category(exc)) from None + finally: + _close_descriptor(descriptor) + + def _revalidate_named_file(self, directory_descriptor, leaf_name, expected): + descriptor = None + try: + descriptor = os.open( + leaf_name, self._flags['inspect'], dir_fd=directory_descriptor, + ) + current = os.fstat(descriptor) + if ( + not _regular_single_link(current) + or _file_revision(current) != _file_revision(expected) + ): + raise ManagedFileAccessError('concurrent_change') + return current + except ManagedFileAccessError: + raise + except OSError: + raise ManagedFileAccessError('concurrent_change') from None + finally: + _close_descriptor(descriptor) + + def _new_temporary_file(self, directory_descriptor): + for _ in range(_TEMPORARY_ATTEMPTS): + name = f'{_TEMPORARY_PREFIX}{secrets.token_hex(12)}.tmp' + try: + descriptor = os.open( + name, self._flags['temporary'], 0o600, + dir_fd=directory_descriptor, + ) + return name, descriptor + except FileExistsError: + continue + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + raise ManagedFileAccessError('filesystem_unavailable') + + def _stage_temporary_file(self, directory_descriptor, payload, max_bytes): + name = descriptor = view = None + try: + name, descriptor = self._new_temporary_file(directory_descriptor) + os.fchmod(descriptor, 0o600) + details = os.fstat(descriptor) + if ( + not _regular_single_link(details) + or details.st_uid != os.geteuid() + or stat.S_IMODE(details.st_mode) != 0o600 + ): + raise ManagedFileAccessError('unsafe_target') + view = memoryview(payload) + written = 0 + while written < len(view): + try: + count = os.write(descriptor, view[written:]) + except InterruptedError: + continue + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + if count <= 0: + raise ManagedFileAccessError('filesystem_unavailable') + written += count + try: + os.fsync(descriptor) + except OSError: + raise ManagedFileAccessError('filesystem_unavailable') from None + identity, stable, _ = self._read_descriptor( + descriptor, max_bytes, include_content=False, + ) + if ( + identity.byte_count != len(payload) + or stable.st_uid != os.geteuid() + or stat.S_IMODE(stable.st_mode) != 0o600 + ): + raise ManagedFileAccessError('unsafe_target') + return name, descriptor, identity, stable + except OSError: + failure = ManagedFileAccessError('filesystem_unavailable') + self._release_unpublished_temporary( + directory_descriptor, name, descriptor, failure, + ) + descriptor = None + raise failure from None + except BaseException as failure: + self._release_unpublished_temporary( + directory_descriptor, name, descriptor, failure, + ) + descriptor = None + raise + finally: + if view is not None: + view.release() + payload = None + + def _release_unpublished_temporary( + self, directory_descriptor, name, descriptor, failure=None): + cleanup_failure = close_failure = None + if name is not None: + try: + self._cleanup_temporary_file(directory_descriptor, name) + except BaseException as exc: + cleanup_failure = exc + try: + _close_descriptor(descriptor) + except BaseException as exc: + close_failure = exc + if failure is not None: + if isinstance(failure, Exception): + if cleanup_failure is not None: + raise cleanup_failure + if close_failure is not None: + raise close_failure + return + if cleanup_failure is not None: + raise cleanup_failure + if close_failure is not None: + raise close_failure + + @staticmethod + def _cleanup_temporary_file(directory_descriptor, name): + try: + os.unlink(name, dir_fd=directory_descriptor) + except FileNotFoundError: + pass + except OSError: + raise ManagedFileAccessError('durability_uncertain') from None + try: + os.fsync(directory_descriptor) + except OSError: + raise ManagedFileAccessError('durability_uncertain') from None + + @staticmethod + def _fsync_directory(directory_descriptor): + try: + os.fsync(directory_descriptor) + except OSError: + raise ManagedFileAccessError('durability_uncertain') from None + + @staticmethod + def _lock_mutation_directory(directory_descriptor): + if fcntl is None: + raise ManagedFileAccessError('filesystem_unavailable') + while True: + try: + fcntl.flock(directory_descriptor, fcntl.LOCK_EX) + return + except InterruptedError: + continue + except OSError: + raise ManagedFileAccessError( + 'filesystem_unavailable', + ) from None + + @staticmethod + def _unlock_mutation_directory(directory_descriptor): + if fcntl is None: + return + try: + fcntl.flock(directory_descriptor, fcntl.LOCK_UN) + except OSError: + pass + + @contextmanager + def _locked_parent(self, root, components): + with self._opened_parent(root, components) as parent: + directory_descriptor, _ = parent + self._lock_mutation_directory(directory_descriptor) + try: + yield parent + finally: + self._unlock_mutation_directory(directory_descriptor) + + def _publish_temporary( + self, directory_descriptor, temporary_name, leaf_name, *, create): + failure = None + try: + if create: + _rename_noreplace( + temporary_name, leaf_name, directory_descriptor, + ) + else: + try: + os.replace( + temporary_name, leaf_name, + src_dir_fd=directory_descriptor, + dst_dir_fd=directory_descriptor, + ) + except OSError: + raise ManagedFileAccessError( + 'concurrent_change', + ) from None + except BaseException as exc: + failure = exc + raise + finally: + try: + self._fsync_directory(directory_descriptor) + except BaseException as sync_failure: + if ( + failure is None + or isinstance(failure, Exception) + and not isinstance(sync_failure, Exception) + ): + raise + + def _verify_private_file( + self, directory_descriptor, leaf_name, max_bytes, + expected_identity, expected_inode): + identity, details, _ = self._read_file_at( + directory_descriptor, leaf_name, max_bytes, content=False, + ) + if ( + identity != expected_identity + or details.st_dev != expected_inode.st_dev + or details.st_ino != expected_inode.st_ino + or details.st_uid != os.geteuid() + or stat.S_IMODE(details.st_mode) != 0o600 + ): + raise ManagedFileAccessError('concurrent_change') + return details + + @contextmanager + def _opened_target(self, root, components, target_kind): + target = self._open_relative(root, components, target_kind) + try: + yield target + finally: + target._close() + + def open_list_directory(self, root_id, relative_path=None): + root = self._root_for_operation(root_id, ManagedFileOperation.LIST) + components = ( + () if relative_path is None + else parse_managed_relative_path(relative_path, root.limits) + ) + self._require_relative_target(root.root_id, components, 'directory') + return self._opened_target(root, components, 'directory') + + def open_read_file(self, root_id, relative_path): + root = self._root_for_operation(root_id, ManagedFileOperation.READ) + components = parse_managed_relative_path(relative_path, root.limits) + self._require_relative_target(root.root_id, components, 'file') + return self._opened_target(root, components, 'file') + + def list_directory(self, root_id, relative_path=None): + root = self._root_for_operation(root_id, ManagedFileOperation.LIST) + components = ( + () if relative_path is None + else parse_managed_relative_path(relative_path, root.limits) + ) + self._require_relative_target(root.root_id, components, 'directory') + entries = [] + name_bytes = scanned = 0 + try: + with self._opened_target(root, components, 'directory') as opened: + try: + iterator = os.scandir(opened.fileno()) + with iterator: + for item in iterator: + scanned += 1 + if scanned > root.limits.max_listing_entries: + raise ManagedFileAccessError('limit_exceeded') + name = item.name + if type(name) is not str or name.startswith( + _TEMPORARY_PREFIX): + continue + if not self._relative_target_allowed( + root.root_id, (*components, name), 'file'): + continue + try: + parse_managed_relative_path( + '/'.join((*components, name)), root.limits, + ) + encoded = name.encode('utf-8') + details = os.stat( + name, dir_fd=opened.fileno(), + follow_symlinks=False, + ) + except ManagedFileAccessError: + continue + except FileNotFoundError: + continue + except OSError: + raise ManagedFileAccessError( + 'filesystem_unavailable', + ) from None + if stat.S_ISDIR(details.st_mode): + kind = 'directory' + byte_count = None + elif ( + _regular_single_link(details) + and details.st_size <= root.limits.max_file_bytes + ): + kind = 'file' + byte_count = details.st_size + else: + continue + if not self._relative_target_allowed( + root.root_id, (*components, name), kind): + continue + name_bytes += len(encoded) + if name_bytes > root.limits.max_listing_bytes: + raise ManagedFileAccessError('limit_exceeded') + entries.append(ManagedFileDirectoryEntry( + name, kind, byte_count, + )) + except OSError: + raise ManagedFileAccessError( + 'filesystem_unavailable', + ) from None + entries.sort(key=lambda entry: entry.name.encode('utf-8')) + return ManagedFileListing(tuple(entries), name_bytes) + except BaseException: + entries.clear() + raise + + def download_file(self, root_id, relative_path): + root = self._root_for_operation(root_id, ManagedFileOperation.READ) + components = parse_managed_relative_path(relative_path, root.limits) + self._require_relative_target(root.root_id, components, 'file') + if root.root_id == RUNTIME_RESULT_ROOT_ID: + if not self._snapshot_gate.acquire(blocking=False): + raise ManagedFileAccessError('download_busy') + try: + for attempt in range(2): + try: + with self._opened_target(root, components, 'file') as opened: + identity, snapshot = self._snapshot_descriptor( + root, opened.fileno(), + ) + return ManagedFileDownload( + identity, snapshot=snapshot, + ) + except ManagedFileAccessError as exc: + if attempt or exc.category not in ( + 'concurrent_change', 'unsafe_target'): + raise + except BaseException: + self._snapshot_gate.release() + raise + raise ManagedFileAccessError('concurrent_change') + for attempt in range(2): + try: + with self._opened_target(root, components, 'file') as opened: + identity, _, content = self._read_descriptor( + opened.fileno(), root.limits.max_file_bytes, + include_content=True, + ) + return ManagedFileDownload(identity, content) + except ManagedFileAccessError as exc: + if attempt or exc.category not in ('concurrent_change', 'unsafe_target'): + raise + raise ManagedFileAccessError('concurrent_change') + + def mutation_file_identity( + self, root_id, relative_path, operation, *, require_private_sha256=None): + if operation not in ( + ManagedFileOperation.CREATE_REPLACE, ManagedFileOperation.DELETE, + ): + raise ManagedFileAccessError('operation_not_allowed') + root = self._root_for_operation(root_id, operation) + components = parse_managed_relative_path(relative_path, root.limits) + with self._opened_parent(root, components) as parent: + directory_descriptor, leaf_name = parent + try: + try: + identity, details, _ = self._read_file_at( + directory_descriptor, leaf_name, + root.limits.max_file_bytes, content=False, + ) + except ManagedFileAccessError as exc: + if exc.category != 'not_found': + raise + self._fsync_directory(directory_descriptor) + try: + self._read_file_at( + directory_descriptor, leaf_name, + root.limits.max_file_bytes, content=False, + ) + except ManagedFileAccessError as repeated: + if repeated.category == 'not_found': + raise exc + raise + raise ManagedFileAccessError('concurrent_change') + # Replay may follow a crash after namespace publication. A + # successful directory fsync turns the observed state into + # durable evidence before terminal operation reconciliation. + self._fsync_directory(directory_descriptor) + self._revalidate_named_file( + directory_descriptor, leaf_name, details, + ) + if ( + require_private_sha256 is not None + and hmac.compare_digest(identity.sha256, require_private_sha256) + and ( + details.st_uid != os.geteuid() + or stat.S_IMODE(details.st_mode) != 0o600 + ) + ): + raise ManagedFileAccessError('unsafe_target') + except ManagedFileAccessError: + raise + return identity + + def _create_replace_file_payload( + self, root_id, relative_path, payload_box, *, expected_sha256): + root = self._root_for_operation( + root_id, ManagedFileOperation.CREATE_REPLACE, + ) + components = parse_managed_relative_path(relative_path, root.limits) + if len(payload_box) != 1 or type(payload_box[0]) is not bytes: + raise ManagedFileAccessError('invalid_content') + if len(payload_box[0]) > root.limits.max_file_bytes: + raise ManagedFileAccessError('limit_exceeded') + if expected_sha256 is not None: + expected_sha256 = _valid_hash(expected_sha256) + proposed = ManagedFileIdentity( + hashlib.sha256(payload_box[0]).hexdigest(), len(payload_box[0]), + ) + + with self._mutation_lock, self._locked_parent(root, components) as parent: + directory_descriptor, leaf_name = parent + before = before_details = None + if expected_sha256 is not None: + before, before_details, _ = self._read_file_at( + directory_descriptor, leaf_name, + root.limits.max_file_bytes, content=False, + ) + if not hmac.compare_digest(before.sha256, expected_sha256): + raise ManagedFileAccessError('hash_conflict') + if before == proposed: + self._revalidate_named_file( + directory_descriptor, leaf_name, before_details, + ) + return ManagedFileMutation(before, before, False) + + temporary_name = temporary_descriptor = None + published = False + failure = None + try: + ( + temporary_name, temporary_descriptor, + staged_identity, staged_details, + ) = self._stage_temporary_file( + directory_descriptor, payload_box[0], + root.limits.max_file_bytes, + ) + if staged_identity != proposed: + raise ManagedFileAccessError('concurrent_change') + if expected_sha256 is not None: + self._revalidate_named_file( + directory_descriptor, leaf_name, before_details, + ) + self._publish_temporary( + directory_descriptor, temporary_name, leaf_name, + create=expected_sha256 is None, + ) + published = True + try: + verified = self._verify_private_file( + directory_descriptor, leaf_name, + root.limits.max_file_bytes, proposed, staged_details, + ) + self._revalidate_named_file( + directory_descriptor, leaf_name, verified, + ) + except ManagedFileAccessError as exc: + if exc.category == 'concurrent_change': + raise + raise ManagedFileAccessError( + 'durability_uncertain', + ) from None + return ManagedFileMutation(before, proposed, True) + except BaseException as exc: + failure = exc + raise + finally: + if temporary_name is not None and not published: + self._release_unpublished_temporary( + directory_descriptor, temporary_name, + temporary_descriptor, failure, + ) + temporary_descriptor = None + try: + _close_descriptor(temporary_descriptor) + except BaseException: + if failure is None: + raise + + def create_replace_file( + self, root_id, relative_path, payload, *, expected_sha256): + payload_box = [payload] + payload = None + try: + return self._create_replace_file_payload( + root_id, relative_path, payload_box, + expected_sha256=expected_sha256, + ) + finally: + payload_box.clear() + payload = payload_box = None + + def delete_file(self, root_id, relative_path, *, expected_sha256): + root = self._root_for_operation(root_id, ManagedFileOperation.DELETE) + components = parse_managed_relative_path(relative_path, root.limits) + expected_sha256 = _valid_hash(expected_sha256) + with self._mutation_lock, self._locked_parent(root, components) as parent: + directory_descriptor, leaf_name = parent + before, before_details, _ = self._read_file_at( + directory_descriptor, leaf_name, + root.limits.max_file_bytes, content=False, + ) + if not hmac.compare_digest(before.sha256, expected_sha256): + raise ManagedFileAccessError('hash_conflict') + self._revalidate_named_file( + directory_descriptor, leaf_name, before_details, + ) + failure = None + try: + try: + os.unlink(leaf_name, dir_fd=directory_descriptor) + except OSError: + raise ManagedFileAccessError( + 'concurrent_change', + ) from None + except BaseException as exc: + failure = exc + raise + finally: + try: + self._fsync_directory(directory_descriptor) + except BaseException as sync_failure: + if ( + failure is None + or isinstance(failure, Exception) + and not isinstance(sync_failure, Exception) + ): + raise + return ManagedFileMutation(before, None, True) diff --git a/app/migrate_layout.py b/app/migrate_layout.py new file mode 100644 index 0000000..67c366f --- /dev/null +++ b/app/migrate_layout.py @@ -0,0 +1,324 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import fnmatch +import hashlib +import os +import shutil + +from db_backend import database_url_from_env, is_postgres_url +from paths import apply_path_config, default_project_paths +from migrate_runtime_safety import require_runtime_hardening_stopped +from postgres_runtime import load_postgres_environment +from runtime_security import ClusterAuthorityLock, reject_reparse_components + + +DESKTOP_HF = r"C:\Users\pro100noob\Desktop\HugginFace" +EXCLUDED_DIRS = {"__pycache__", ".git", ".opencode", "node_modules", "tmp", "runtime"} +APP_FILES = [ + "app.py", + "console_runner.py", + "dashboard.py", + "keycheck_runner.py", + "migrate_layout.py", + "paths.py", + "scan_manager.py", + "scanner.py", + "scanner_db.py", + "supervisor.py", + "ui_components.py", + "config.yaml", + "secrets.yaml", + "requirements.txt", + "CHEATSHEET.md", + "DETECTOR_NOTES.md", +] +APP_DIRS = [".streamlit", "keycheckers"] +RESULT_FILES = ["found_secrets.jsonl", "scan_results.jsonl", "scan_errors.log", "scanner.db", "scanner.db-wal", "scanner.db-shm"] +KEYCHECK_SERVICE_SCRIPTS = { + "anthropic": [os.path.join("anthropic", "anthropicKeycheck.py")], + "aws": [os.path.join("aws", "awsKeycheck.py")], + "azure": [os.path.join("azure", "azureKeycheck.py")], + "deepseek": [os.path.join("deepseek", "deepseekKeycheck.py")], + "dockerhub": [os.path.join("dockerhub", "dockerhubKeycheck.py"), os.path.join("dockerhub", "dockerhub.txt")], + "gcp": [os.path.join("gcp", "gcpKeycheck.py"), os.path.join("gcp", "gcp.txt")], + "gemini": [os.path.join("gemini", "geminiKeycheck.py"), os.path.join("gemini", "gem.txt")], + "groq": [os.path.join("groq", "groqKeycheck.py")], + "github": [os.path.join("github", "githubKeycheck.py"), os.path.join("github", "github.txt")], + "gitlab": [os.path.join("gitlab", "gitlabKeycheck.py"), os.path.join("gitlab", "gitlab.txt")], + "kimi": [os.path.join("kimi", "kimiKeycheck.py")], + "openai": ["Keycheck.py"], + "openrouter": ["OpenrouterKeycheck.py"], + "provider_resolver": [os.path.join("provider_resolver", "providerResolverKeycheck.py")], + "qwen": [os.path.join("qwen", "qwenKeycheck.py")], + "replicate": [os.path.join("replicate", "replicateKeycheck.py")], + "xai": [os.path.join("xai", "xaiKeycheck.py")], + "huggingface": [os.path.join("huggingface", "huggingfaceKeycheck.py")], + "zai": [os.path.join("zai", "zaiKeycheck.py")], +} +KEYCHECK_TOP_LEVEL_PREFIXES = { + "openai": "openai", + "openrouter": "openrouter", +} + + +def load_config(config_path): + if not config_path: + return {'global': default_project_paths()} + try: + import yaml + except ImportError as e: + raise SystemExit("PyYAML is required for --config") from e + with open(config_path, "r", encoding="utf-8") as f: + config = apply_path_config(yaml.safe_load(f) or {}, config_path) + return config + + +def load_layout(config_path): + return load_config(config_path).get('global') or {} + + +def mkdir(path, dry_run=False): + if dry_run: + print(f"mkdir {path}") + return + os.makedirs(path, exist_ok=True) + + +def copy_file(src, dst, overwrite=False, dry_run=False): + if not os.path.exists(src): + return False + if os.path.lexists(dst) and not overwrite: + print(f"skip existing {dst}") + return False + parent = os.path.dirname(dst) + if parent: + mkdir(parent, dry_run) + if dry_run: + print(f"copy {src} -> {dst}") + return True + try: + reject_reparse_components(parent or os.path.dirname(os.path.abspath(dst))) + if os.path.lexists(dst): + reject_reparse_components(dst) + shutil.copy2(src, dst) + except OSError as e: + print(f"skip locked/unavailable {src}: {e}") + return False + print(f"copied {src} -> {dst}") + return True + + +def ignore_app_dir(_dir, names): + ignored = set() + for name in names: + if name in EXCLUDED_DIRS: + ignored.add(name) + if fnmatch.fnmatch(name, "*.pyc"): + ignored.add(name) + return ignored + + +def copy_dir(src, dst, overwrite=False, dry_run=False, verified_apply=False): + if not os.path.isdir(src): + return False + if os.path.lexists(dst) and not overwrite: + print(f"skip existing {dst}") + return False + if dry_run: + print(f"copytree {src} -> {dst}") + return True + if os.path.lexists(dst) and overwrite: + raise RuntimeError('legacy directory replacement is retired; existing directories are never replaced') + shutil.copytree(src, dst, ignore=ignore_app_dir, dirs_exist_ok=False) + print(f"copied {src} -> {dst}") + return True + + +def create_layout(layout, dry_run=False): + for key in ("project_dir", "runtime_dir", "results_dir", "queue_dir", "log_dir", "state_dir", "keycheck_dir", "work_dir"): + mkdir(layout[key], dry_run) + mkdir(os.path.join(layout["runtime_dir"], "imports"), dry_run) + + +def copy_app_files(source_dir, layout, overwrite=False, dry_run=False, verified_apply=False): + project_dir = layout["project_dir"] + if os.path.abspath(source_dir) == os.path.abspath(project_dir): + print("app source is already project_dir; app copy skipped") + return + for name in APP_FILES: + copy_file(os.path.join(source_dir, name), os.path.join(project_dir, name), overwrite, dry_run) + for name in APP_DIRS: + copy_dir(os.path.join(source_dir, name), os.path.join(project_dir, name), overwrite, dry_run, verified_apply) + + +def copy_scanner_runtime(old_root, layout, overwrite=False, dry_run=False, verified_apply=False): + for name in RESULT_FILES: + copy_file(os.path.join(old_root, name), os.path.join(layout["results_dir"], name), overwrite, dry_run) + for pattern in ("todo_*.txt", "checked_*.txt"): + if not os.path.isdir(old_root): + continue + for name in os.listdir(old_root): + if fnmatch.fnmatch(name, pattern): + copy_file(os.path.join(old_root, name), os.path.join(layout["queue_dir"], name), overwrite, dry_run) + copy_dir(os.path.join(old_root, "logs"), layout["log_dir"], overwrite, dry_run, verified_apply) + copy_dir(os.path.join(old_root, "state"), layout["state_dir"], overwrite, dry_run, verified_apply) + copy_file(os.path.join(old_root, "runner_state.json"), os.path.join(layout["state_dir"], "runner_state.json"), overwrite, dry_run) + + +def line_hash(line): + return hashlib.sha256(line.strip().encode("utf-8", errors="replace")).hexdigest() + + +def existing_line_hashes(path): + hashes = set() + if not os.path.exists(path): + return hashes + with open(path, "r", encoding="utf-8", errors="replace") as f: + for line in f: + if line.strip(): + hashes.add(line_hash(line)) + return hashes + + +def import_jsonl_dedupe(inputs, output, dry_run=False): + hashes = existing_line_hashes(output) + added = 0 + if dry_run: + print(f"dedupe import {len(inputs)} file(s) -> {output}") + return 0 + mkdir(os.path.dirname(output), dry_run=False) + with open(output, "a", encoding="utf-8") as dst: + for path in inputs: + if not os.path.exists(path): + continue + with open(path, "r", encoding="utf-8", errors="replace") as src: + for line in src: + if not line.strip(): + continue + digest = line_hash(line) + if digest in hashes: + continue + dst.write(line if line.endswith("\n") else line + "\n") + hashes.add(digest) + added += 1 + print(f"imported {added} unique finding line(s) into {output}") + return added + + +def copy_legacy_keychecker_outputs(layout, desktop_dir=DESKTOP_HF, overwrite=False, dry_run=False): + if not os.path.isdir(desktop_dir): + return + for service in KEYCHECK_SERVICE_SCRIPTS: + source_dir = os.path.join(desktop_dir, service) + target_dir = os.path.join(layout["keycheck_dir"], service) + if os.path.isdir(source_dir): + for name in os.listdir(source_dir): + if name.lower().endswith((".txt", ".jsonl")): + copy_file(os.path.join(source_dir, name), os.path.join(target_dir, name), overwrite, dry_run) + for service, prefix in KEYCHECK_TOP_LEVEL_PREFIXES.items(): + target_dir = os.path.join(layout["keycheck_dir"], service) + for name in os.listdir(desktop_dir): + lower = name.lower() + if lower.startswith(prefix) and lower.endswith((".txt", ".jsonl")): + copy_file(os.path.join(desktop_dir, name), os.path.join(target_dir, name), overwrite, dry_run) + + +def merge_unique_lines(inputs, output, dry_run=False): + values = [] + seen = existing_values = set() + if os.path.exists(output): + with open(output, "r", encoding="utf-8", errors="replace") as f: + existing_values = {line.strip().lstrip("\ufeff") for line in f if line.strip()} + seen = set(existing_values) + for path in inputs: + if not os.path.exists(path): + continue + with open(path, "r", encoding="utf-8", errors="replace") as f: + for line in f: + value = line.strip().lstrip("\ufeff") + if value and value not in seen: + values.append(value) + seen.add(value) + if dry_run: + print(f"merge {len(values)} unique line(s) -> {output}") + return + mkdir(os.path.dirname(output), dry_run=False) + with open(output, "a", encoding="utf-8") as f: + for value in values: + f.write(value + "\n") + print(f"appended {len(values)} unique line(s) -> {output}") + + +def import_huggingface_desktop(layout, desktop_dir=DESKTOP_HF, overwrite=False, dry_run=False): + if not os.path.isdir(desktop_dir): + print(f"Desktop HugginFace directory not found: {desktop_dir}") + return + jsonl_inputs = [os.path.join(desktop_dir, name) for name in os.listdir(desktop_dir) if fnmatch.fnmatch(name, "found_secrets*.jsonl")] + import_jsonl_dedupe(jsonl_inputs, os.path.join(layout["results_dir"], "found_secrets.jsonl"), dry_run) + merge_unique_lines([os.path.join(desktop_dir, "checked.txt")], os.path.join(layout["queue_dir"], "checked_huggingface.txt"), dry_run) + merge_unique_lines([os.path.join(desktop_dir, "todo.txt")], os.path.join(layout["queue_dir"], "todo_huggingface.txt"), dry_run) + copy_file(os.path.join(desktop_dir, "proxy.txt"), layout["proxy_file"], overwrite=False, dry_run=dry_run) + copy_file(os.path.join(desktop_dir, "requirements-keycheckers.txt"), os.path.join(layout["project_dir"], "requirements-keycheckers.txt"), overwrite, dry_run) + copy_file(os.path.join(desktop_dir, "KEYCHECKERS.md"), os.path.join(layout["project_dir"], "KEYCHECKERS.md"), overwrite, dry_run) + for service, rel_paths in KEYCHECK_SERVICE_SCRIPTS.items(): + for rel_path in rel_paths: + src = os.path.join(desktop_dir, rel_path) + dst = os.path.join(layout["project_dir"], "keycheckers", service, os.path.basename(rel_path)) + copy_file(src, dst, overwrite, dry_run) + copy_legacy_keychecker_outputs(layout, desktop_dir, overwrite, dry_run) + + +def parse_args(): + parser = argparse.ArgumentParser(description="Copy/import legacy scanner files into the unified D:\\truf layout.") + parser.add_argument("--config", default="config.yaml") + parser.add_argument("--source-app", default=os.path.dirname(os.path.abspath(__file__))) + parser.add_argument("--target-app", help="Destination app directory. Defaults to \\app.") + parser.add_argument("--in-place", action="store_true", help="Use project_dir from config instead of copying to \\app.") + parser.add_argument("--old-root", default=r"D:\truf") + parser.add_argument("--desktop-hf", default=DESKTOP_HF) + parser.add_argument("--overwrite", action="store_true") + parser.add_argument("--dry-run", action="store_true") + parser.add_argument("--apply", action="store_true", help="Retired; production layout mutation is disabled") + parser.add_argument("--no-app-copy", action="store_true") + parser.add_argument("--no-desktop-import", action="store_true") + return parser.parse_args() + + +def main(): + args = parse_args() + if args.apply and args.dry_run: + raise SystemExit('--apply and --dry-run are mutually exclusive') + if args.apply: + raise SystemExit( + 'migrate_layout --apply is retired because the production layout is already migrated. ' + 'Use reviewed offline backup/restore tooling for any future relocation.' + ) + config = load_config(args.config) + layout = config.get('global') or {} + if not args.in_place: + layout["project_dir"] = args.target_app or os.path.join(layout["root_dir"], "app") + dry_run = not args.apply + load_postgres_environment(os.path.abspath(args.config), config) + endpoint_dsn = database_url_from_env() or layout.get('database_url') + if not is_postgres_url(endpoint_dsn): + raise SystemExit('A caller-selected canonical PostgreSQL DSN is required for maintenance authority') + with ClusterAuthorityLock(config, endpoint_dsn=endpoint_dsn): + require_runtime_hardening_stopped(config) + create_layout(layout, dry_run) + if not args.no_app_copy: + copy_app_files(args.source_app, layout, args.overwrite, dry_run, verified_apply=args.apply) + copy_scanner_runtime(args.old_root, layout, args.overwrite, dry_run, verified_apply=args.apply) + if not args.no_desktop_import: + import_huggingface_desktop(layout, args.desktop_hf, args.overwrite, dry_run) + if dry_run: + print("Dry-run complete. --apply is retired; use reviewed offline backup/restore tooling for relocation.") + return 0 + print("Migration copy/import finished. Originals were left in place.") + return 0 + + +if __name__ == "__main__": + main() diff --git a/app/migrate_observability_db.py b/app/migrate_observability_db.py new file mode 100644 index 0000000..7b3debb --- /dev/null +++ b/app/migrate_observability_db.py @@ -0,0 +1,137 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os +import sqlite3 + +from scanner_db import ScannerDB + + +TABLES = [ + 'runs', + 'source_cycles', + 'target_scans', + 'target_queue', + 'scan_publication_outbox', + 'findings', + 'errors', + 'queue_snapshots', + 'config_snapshots', + 'package_repo_candidates', + 'keycheck_results', + 'keycheck_event_map', + 'finding_uid_map', +] + + +def sqlite_connect_ro(path): + uri = 'file:' + os.path.abspath(path).replace('\\', '/') + '?mode=ro' + conn = sqlite3.connect(uri, uri=True) + conn.row_factory = sqlite3.Row + return conn + + +def sqlite_columns(conn, table): + return [row['name'] for row in conn.execute(f'PRAGMA table_info({table})').fetchall()] + + +def sqlite_count(conn, table): + return int(conn.execute(f'SELECT COUNT(*) AS count FROM {table}').fetchone()['count']) + + +def sqlite_row_estimate(conn, table, exact=False): + if exact: + return sqlite_count(conn, table) + try: + row = conn.execute('SELECT seq FROM sqlite_sequence WHERE name = ?', (table,)).fetchone() + if row and row['seq'] is not None: + return int(row['seq']) + except sqlite3.Error: + pass + return None + + +def pg_reset_identity(db, table): + db.conn.execute(f''' + SELECT setval( + pg_get_serial_sequence('{table}', 'id'), + COALESCE((SELECT MAX(id) FROM {table}), 1), + (SELECT MAX(id) IS NOT NULL FROM {table}) + ) + ''') + + +def postgres_safe_value(value): + if isinstance(value, str) and '\x00' in value: + return value.replace('\x00', '\\u0000') + return value + + +def copy_table(source, target, table, batch_size, dry_run=False, exact_counts=False): + columns = sqlite_columns(source, table) + if not columns: + print(f'{table}: missing or empty schema in SQLite, skipped', flush=True) + return 0 + total = sqlite_row_estimate(source, table, exact=exact_counts) + total_label = total if total is not None else 'unknown' + print(f'{table}: source_rows={total_label}' + ('' if exact_counts else ' estimated'), flush=True) + if dry_run or total == 0: + return int(total or 0) + + column_sql = ', '.join(columns) + placeholders = ', '.join('?' for _ in columns) + conflict_sql = ' ON CONFLICT (id) DO NOTHING' if 'id' in columns else '' + insert_sql = f'INSERT INTO {table} ({column_sql}) VALUES ({placeholders}){conflict_sql}' + + copied = 0 + cursor = source.execute(f'SELECT {column_sql} FROM {table} ORDER BY id' if 'id' in columns else f'SELECT {column_sql} FROM {table}') + while True: + rows = cursor.fetchmany(batch_size) + if not rows: + break + values = [[postgres_safe_value(row[column]) for column in columns] for row in rows] + if target.conn.is_postgres: + with target.conn._conn.cursor() as pg_cursor: + with pg_cursor.copy(f'COPY {table} ({column_sql}) FROM STDIN') as copy: + for value_row in values: + copy.write_row(value_row) + else: + for value_row in values: + target.conn.execute(insert_sql, value_row) + copied += len(rows) + target.conn.commit() + print(f'{table}: copied={copied}/{total_label}', flush=True) + if 'id' in columns: + pg_reset_identity(target, table) + target.conn.commit() + return copied + + +def truncate_target(db): + table_sql = ', '.join(TABLES) + db.conn.execute(f'TRUNCATE TABLE {table_sql} RESTART IDENTITY CASCADE') + db.conn.commit() + + +def parse_args(): + parser = argparse.ArgumentParser(description='Offline migrate scanner observability SQLite DB to PostgreSQL.') + parser.add_argument('--sqlite', required=True, help='Path to scanner_active.db or archived SQLite DB') + parser.add_argument('--db-url', default=os.getenv('SCANNER_DB_URL') or os.getenv('DATABASE_URL'), help='PostgreSQL DSN') + parser.add_argument('--batch-size', type=int, default=1000) + parser.add_argument('--exact-counts', action='store_true', help='Use exact COUNT(*) per table; slow on multi-GB SQLite files') + parser.add_argument('--truncate', action='store_true', help='Delete existing Postgres observability rows before import') + parser.add_argument('--dry-run', action='store_true') + return parser.parse_args() + + +def main(): + raise SystemExit( + 'migrate_observability_db.py is retired for PostgreSQL. Use migrate_runtime_safety.py ' + '--apply --sources-stopped with the bound cluster; perform legacy data import only with a reviewed, cluster-bound tool.' + ) + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/app/migrate_runtime_safety.py b/app/migrate_runtime_safety.py new file mode 100644 index 0000000..099ac7c --- /dev/null +++ b/app/migrate_runtime_safety.py @@ -0,0 +1,3504 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import contextlib +import hashlib +import json +import os +import re +import sqlite3 +import stat +import time +from datetime import datetime, timezone + +from paths import apply_path_config +from query_policy import validate_rejected_query_policy +from db_backend import ( + DatabaseUrlError, + POSTGRES_APPLICATION_SCHEMA, + canonical_postgres_url, + database_url_from_env, + is_postgres_url, +) +from postgres_runtime import ( + _listener_present, + configured_cluster_values, + load_postgres_environment, + postgres_runtime_paths, + verify_cluster_identity, +) +from scanner_db import ( + PIPELINE_MIGRATION_VERSIONS, + ScanEventConflictError, + ScannerDB, + json_dumps, + migrate_runtime_safety_schema, + normalize_target, + utc_now_iso, +) +from process_identity import current_process_identity +from result_ingester import ResultIngester +from result_spool import ResultSpool +from result_bundle import ResultBundleReader +from runtime_security import ( + atomic_write_private_json, + ClusterAuthorityLock, + PrivatePathState, + PrivateFileError, + canonical_cluster_data_directory, + canonical_path, + harden_private_directory, + harden_private_file, + harden_private_tree, + durable_publish, + durable_unlink, + ensure_private_directory, + fsync_directory, + is_reparse_point, + private_directory_ready, + private_file_ready, + inspect_private_relative_path, + read_private_json, + require_private_directory, + require_private_file, + require_trusted_native_executable, + reject_reparse_components, + sha256_file, + preflight_lifecycle_paths, +) + + +RECONCILIATION_ABSOLUTE_ROW_BYTES = 16 * 1024 * 1024 +RECONCILIATION_MUTATION_RETRIES = 3 +TARGET_QUEUE_POLICY_MANIFEST_MAX_BYTES = 16 * 1024 * 1024 +REVIEWED_LEGACY_RESULT_SPOOL = os.path.normcase( + os.path.abspath(r'D:\truf\runtime\result_spool') +) + + +class ReconciliationFileChanged(RuntimeError): + pass + + +def reviewed_legacy_result_spool(config): + global_config = (config or {}).get('global') or {} + value = global_config.get('legacy_result_spool_dir') or global_config.get('result_spool_dir') + if not value or '{' in str(value) or '}' in str(value): + raise RuntimeError('legacy result spool path is unresolved') + if os.name != 'nt': + runtime = global_config.get('runtime_dir') + for label, path in (('runtime', runtime), ('legacy result spool', value)): + if ( + not isinstance(path, str) or not os.path.isabs(path) + or path != os.path.normpath(path) + or any(char in path for char in ('{', '}', '\\', '\x00')) + ): + raise RuntimeError(f'{label} path must be resolved, exact and absolute') + reject_reparse_components(path) + expected = os.path.join(runtime, 'result_spool') + if value != expected: + raise RuntimeError('legacy result spool must be exactly runtime_dir/result_spool') + for path in (runtime, value): + require_private_directory(path, create=False) + resolved = canonical_path(value) + if resolved != os.path.join(canonical_path(runtime), 'result_spool'): + raise RuntimeError('legacy result spool resolved outside its runtime authority') + for name in ('reservations', 'quarantine'): + path = os.path.join(value, name) + if os.path.lexists(path): + require_private_directory(path, create=False) + return resolved + resolved = os.path.normcase(os.path.abspath(value)) + if resolved != REVIEWED_LEGACY_RESULT_SPOOL: + raise RuntimeError( + 'legacy result spool path does not match reviewed D:\\truf\\runtime\\result_spool authority' + ) + return os.path.abspath(value) + + +def _mtime_ns(stat_result): + return int(getattr(stat_result, 'st_mtime_ns', int(stat_result.st_mtime * 1_000_000_000))) + + +def _stat_identity(stat_result): + inode = int(getattr(stat_result, 'st_ino', 0) or 0) + device = int(getattr(stat_result, 'st_dev', 0) or 0) + return device, inode, int(stat_result.st_size), _mtime_ns(stat_result) + + +def todo_file_identity(path, stat_result=None): + stat_result = stat_result or os.stat(path, follow_symlinks=False) + device, inode, size, mtime_ns = _stat_identity(stat_result) + return f'{device}:{inode}:{size}:{mtime_ns}:{os.path.realpath(os.path.abspath(path))}' + + +def _safe_target_preview(value): + if isinstance(value, bytes): + payload = value + else: + payload = str(value or '').encode('utf-8', errors='replace') + return f'' + + +def _same_file_snapshot(path, handle_stat): + try: + current_handle = os.fstat(handle_stat[0]) if isinstance(handle_stat, tuple) else handle_stat + current_path = os.stat(path, follow_symlinks=False) + except OSError: + return False + expected = handle_stat[1] if isinstance(handle_stat, tuple) else _stat_identity(handle_stat) + return _stat_identity(current_handle) == expected and _stat_identity(current_path) == expected + + +def docker_target_is_bare(target): + text = str(target or '').strip() + if not text or '@' in text: + return False + return ':' not in text.rsplit('/', 1)[-1] + + +def _contains_nul(value): + if isinstance(value, str): + return '\x00' in value + if isinstance(value, dict): + return any(_contains_nul(key) or _contains_nul(item) for key, item in value.items()) + if isinstance(value, (list, tuple)): + return any(_contains_nul(item) for item in value) + return False + + +def reconciliation_target_error(target, platform): + if not target: + return 'empty row' + if '\x00' in target: + return 'NUL byte in row' + if platform == 'docker' and docker_target_is_bare(target): + return 'unresolved bare Docker repository' + if platform in ('npm', 'pypi', 'package_git', 'postman', 'github_actions', 'gitlab_ci'): + try: + value = json.loads(target) + except (TypeError, ValueError, json.JSONDecodeError): + return f'malformed {platform} JSON target' + if not isinstance(value, dict): + return f'malformed {platform} JSON target' + if _contains_nul(value): + return 'NUL byte in decoded JSON row' + if not normalize_target(target, platform): + return 'target normalizes to an empty value' + return '' + + +def require_legacy_cutover_clear(db, config, max_entries=10000): + """Refuse v2 activation while any legacy publication state is unresolved.""" + conn = getattr(db, 'conn', None) + if not conn: + raise RuntimeError('database connection is unavailable') + outbox_rows = 0 + if conn.table_exists('scan_publication_outbox'): + row = conn.execute( + 'SELECT COUNT(*) AS count FROM scan_publication_outbox' + ).fetchone() + outbox_rows = int(row['count'] or 0) + if outbox_rows: + raise RuntimeError( + f'final cutover refused: scan_publication_outbox has {outbox_rows} unresolved row(s)' + ) + legacy_raw_rows = 0 + if conn.table_exists('target_scans'): + row = conn.execute( + '''SELECT COUNT(*) AS count FROM target_scans + WHERE raw_result_storage != 'normalized_v2' AND raw_result_json IS NOT NULL''' + ).fetchone() + legacy_raw_rows = int(row['count'] or 0) + if legacy_raw_rows: + raise RuntimeError( + f'final cutover refused: target_scans has {legacy_raw_rows} legacy raw result row(s); ' + 'run --backfill-normalized-results in bounded offline batches' + ) + global_config = (config or {}).get('global') or {} + spool_dir = ( + reviewed_legacy_result_spool(config) + if getattr(conn, 'is_postgres', False) + else ( + global_config.get('legacy_result_spool_dir') + or global_config.get('result_spool_dir') + or os.path.join(global_config.get('runtime_dir') or '', 'result_spool') + ) + ) + inspected = 0 + unresolved = [] + spool_present = False + if spool_dir: + try: + spool_details = os.stat(spool_dir, follow_symlinks=False) + except FileNotFoundError: + spool_details = None + except OSError as exc: + raise RuntimeError('final cutover refused: legacy spool state is unknown') from exc + if spool_details is not None and not stat.S_ISDIR(spool_details.st_mode): + raise RuntimeError('final cutover refused: legacy spool path is not a directory') + spool_present = spool_details is not None + if spool_present: + for relative in ('', 'reservations', 'quarantine'): + parent = os.path.join(spool_dir, relative) if relative else spool_dir + try: + parent_details = os.stat(parent, follow_symlinks=False) + except FileNotFoundError: + continue + except OSError as exc: + raise RuntimeError('final cutover refused: legacy spool shard state is unknown') from exc + if not stat.S_ISDIR(parent_details.st_mode): + raise RuntimeError('final cutover refused: legacy spool shard is not a directory') + if os.name != 'nt': + require_private_directory(parent, create=False) + with os.scandir(parent) as entries: + for entry in entries: + inspected += 1 + if inspected > max(1, int(max_entries)): + raise RuntimeError('final cutover refused: legacy spool inspection exceeded its entry bound') + if os.name != 'nt' and (entry.is_symlink() or is_reparse_point(entry.path)): + raise RuntimeError('final cutover refused: legacy spool contains a link') + if entry.is_file(follow_symlinks=False) and entry.name.lower().endswith('.json'): + unresolved.append(os.path.join(relative, entry.name)) + if len(unresolved) >= 10: + break + if len(unresolved) >= 10: + break + if unresolved: + raise RuntimeError( + 'final cutover refused: legacy result spool has unresolved objects: ' + + ', '.join(unresolved) + ) + results_dir = global_config.get('results_dir') + ledgers_checked = 0 + if results_dir: + for name in ('scan_results', 'found_secrets', 'scan_errors'): + ledger_path = os.path.join(results_dir, f'{name}.publication-ledger.sqlite3') + try: + ledger_details = os.stat(ledger_path, follow_symlinks=False) + except FileNotFoundError: + continue + except OSError as exc: + raise RuntimeError(f'final cutover refused: {name} ledger state is unknown') from exc + if not stat.S_ISREG(ledger_details.st_mode): + raise RuntimeError(f'final cutover refused: {name} ledger is not a regular file') + ledgers_checked += 1 + uri = 'file:' + os.path.abspath(ledger_path).replace('\\', '/') + '?mode=ro' + ledger = sqlite3.connect(uri, uri=True, timeout=5) + try: + table = ledger.execute( + "SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = 'publication_identity'" + ).fetchone() + if table: + pending = ledger.execute( + "SELECT 1 FROM publication_identity WHERE state = 'prepared' LIMIT 1" + ).fetchone() + if pending: + raise RuntimeError( + f'final cutover refused: {name} publication ledger has a prepared append' + ) + finally: + ledger.close() + if conn.is_postgres: + conn.commit() + return { + 'legacy_outbox_rows': outbox_rows, + 'legacy_raw_result_rows': legacy_raw_rows, + 'legacy_spool_present': spool_present, + 'legacy_spool_canonical': os.path.normcase(os.path.abspath(spool_dir)), + 'legacy_spool_entries_inspected': inspected, + 'legacy_projection_ledgers_checked': ledgers_checked, + 'legacy_prepared_appends': 0, + } + + +def drain_legacy_scan_outbox(db, config, max_rows=1000): + max_rows = min(10000, max(1, int(max_rows))) + from console_runner import apply_global_config, drain_scan_publication_outbox + from scanner import initialize_scanner_runtime + + global_config = (config or {}).get('global') or {} + apply_global_config(global_config) + initialize_scanner_runtime(preflight_complete=True, register_cleanup=False) + drained = drain_scan_publication_outbox( + db, max_rows, require_v2_schema=False, + ) + remaining = db.conn.execute( + 'SELECT COUNT(*) AS count FROM scan_publication_outbox' + ).fetchone() + db.conn.commit() + return {'drained': int(drained), 'remaining': int(remaining['count'] or 0)} + + +def import_legacy_outbox_to_projection(db, max_rows=1000): + conn = getattr(db, 'conn', None) + if not conn or not conn.is_postgres: + raise RuntimeError('legacy outbox projection transfer requires PostgreSQL') + max_rows = min(10000, max(1, int(max_rows))) + now = utc_now_iso() + transferred = 0 + try: + rows = conn.execute( + '''SELECT o.id, o.target_scan_id, ts.scan_event_id, ts.scan_event_hash, + COALESCE(ts.findings_count, 0) AS findings_count, + COALESCE(ts.error_count, 0) AS error_count, + CASE WHEN ts.raw_result_json IS NULL THEN 0 + ELSE octet_length(ts.raw_result_json) END AS raw_result_bytes + FROM scan_publication_outbox o + JOIN target_scans ts ON ts.id = o.target_scan_id + WHERE NOT EXISTS ( + SELECT 1 FROM pipeline_quarantine q + WHERE q.subsystem = 'migration' AND q.object_type = 'legacy_outbox' + AND q.object_id = o.id AND q.review_status = 'pending' + ) + ORDER BY o.id LIMIT ? FOR UPDATE OF o, ts''', + (max_rows,), + ).fetchall() + for row in rows: + raw_bytes = int(row['raw_result_bytes'] or 0) + event_id = str(row['scan_event_id'] or f'legacy-target-scan-{row["target_scan_id"]}') + event_hash = str(row['scan_event_hash'] or '') + if not event_hash: + if raw_bytes <= 0 or raw_bytes > 192 * 1024 * 1024: + conn.execute( + '''INSERT INTO pipeline_quarantine( + subsystem, object_type, object_id, reason_code, reason_detail, + byte_count, capacity_credit_applied, review_status, detected_at + ) VALUES ('migration','legacy_outbox',?, + 'legacy_payload_oversized',?,?,0,'pending',?) + ON CONFLICT DO NOTHING''', + ( + row['id'], + f'legacy outbox payload is {raw_bytes} bytes and has no event hash', + max(0, raw_bytes), utc_now_iso(), + ), + ) + continue + payload = conn.execute( + '''SELECT raw_result_json FROM target_scans + WHERE id = ? AND raw_result_json IS NOT NULL + AND octet_length(raw_result_json) = ?''', + (row['target_scan_id'], raw_bytes), + ).fetchone() + if not payload: + raise RuntimeError('legacy outbox payload changed during bounded identity recovery') + event_hash = hashlib.sha256( + str(payload['raw_result_json']).encode('utf-8') + ).hexdigest() + stream_mask = ( + 1 | (2 if int(row['findings_count']) else 0) + | (4 if int(row['error_count']) else 0) + ) + job = conn.execute( + '''SELECT id, event_hash FROM projection_jobs + WHERE job_kind = 'scan_event' AND event_id = ? FOR UPDATE''', + (event_id,), + ).fetchone() + if job and str(job['event_hash']) != event_hash: + raise RuntimeError(f'legacy outbox event hash conflicts with projection job: {event_id}') + if not job: + capacity_bytes = max(1, raw_bytes) * 2 + conn.execute('SELECT id FROM pipeline_capacity WHERE id = 1 FOR UPDATE') + conn.insert_returning_id( + '''INSERT INTO projection_jobs( + job_kind, event_id, event_hash, target_scan_id, status, + required_stream_mask, capacity_items, capacity_bytes, + created_at, updated_at + ) VALUES ('scan_event', ?, ?, ?, 'pending', ?, 1, ?, ?, ?)''', + ( + event_id, event_hash, row['target_scan_id'], stream_mask, + capacity_bytes, now, now, + ), + ) + conn.execute( + '''UPDATE pipeline_capacity SET projection_items = projection_items + 1, + projection_bytes = projection_bytes + ?, updated_at = ? WHERE id = 1''', + (capacity_bytes, now), + ) + deleted = conn.execute( + 'DELETE FROM scan_publication_outbox WHERE id = ? AND target_scan_id = ?', + (row['id'], row['target_scan_id']), + ) + if int(deleted.rowcount or 0) != 1: + raise RuntimeError(f'legacy outbox row changed during projection transfer: {row["id"]}') + transferred += 1 + conn.commit() + remaining = conn.execute( + 'SELECT COUNT(*) AS count FROM scan_publication_outbox' + ).fetchone() + conn.commit() + return {'transferred': transferred, 'remaining': int(remaining['count'] or 0)} + except Exception: + conn.rollback() + raise + + +def backfill_normalized_raw_results( + db, max_rows=1000, max_bytes=192 * 1024 * 1024, max_seconds=30, + max_object_bytes=192 * 1024 * 1024, +): + conn = getattr(db, 'conn', None) + if not conn or not conn.is_postgres: + raise RuntimeError('normalized raw-result backfill requires PostgreSQL') + max_rows = min(10000, max(1, int(max_rows))) + max_bytes = max(1, int(max_bytes)) + deadline = time.monotonic() + max(0.1, float(max_seconds)) + processed = 0 + consumed = 0 + blocked_id = None + reviewed = 0 + while processed < max_rows and time.monotonic() < deadline: + try: + candidate = conn.execute( + '''SELECT s.id, octet_length(s.raw_result_json) AS payload_bytes + FROM target_scans s + WHERE s.raw_result_storage != 'normalized_v2' + AND s.raw_result_json IS NOT NULL + AND NOT EXISTS ( + SELECT 1 FROM pipeline_quarantine q + WHERE q.subsystem = 'migration' + AND q.object_type = 'legacy_target_scan' + AND q.object_id = s.id AND q.review_status = 'pending' + ) + ORDER BY s.id LIMIT 1''' + ).fetchone() + if not candidate: + conn.commit() + break + candidate_id = int(candidate['id']) + payload_size = int(candidate['payload_bytes'] or 0) + if payload_size <= 0 or payload_size > max(1, int(max_object_bytes)): + conn.execute( + '''INSERT INTO pipeline_quarantine( + subsystem, object_type, object_id, reason_code, reason_detail, + byte_count, capacity_credit_applied, review_status, detected_at + ) VALUES ('migration','legacy_target_scan',?, + 'legacy_payload_oversized',?,?,0,'pending',?) + ON CONFLICT DO NOTHING''', + ( + candidate_id, + f'legacy raw result is {payload_size} bytes; reviewed bounded import required', + max(0, payload_size), utc_now_iso(), + ), + ) + conn.commit() + reviewed += 1 + continue + if consumed + payload_size > max_bytes: + blocked_id = candidate_id + conn.commit() + break + row = conn.execute( + '''SELECT id, raw_result_json FROM target_scans + WHERE id = ? AND raw_result_storage != 'normalized_v2' + AND raw_result_json IS NOT NULL + AND octet_length(raw_result_json) = ? FOR UPDATE''', + (candidate_id, payload_size), + ).fetchone() + if not row: + conn.rollback() + continue + prepared = [] + payload = str(row['raw_result_json'] or '') + if len(payload.encode('utf-8')) != payload_size: + raise RuntimeError(f'legacy target scan {candidate_id} changed during bounded fetch') + try: + result = json.loads(payload) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise RuntimeError(f'legacy target scan {candidate_id} contains invalid JSON') from exc + if not isinstance(result, dict): + raise RuntimeError(f'legacy target scan {candidate_id} is not a JSON object') + findings = result.get('findings') or [] + errors = result.get('errors') or [] + if not isinstance(findings, list) or not isinstance(errors, list): + raise RuntimeError(f'legacy target scan {candidate_id} has invalid finding/error arrays') + metadata = dict(result) + metadata.pop('findings', None) + metadata.pop('errors', None) + encoded_metadata = json.dumps( + metadata, ensure_ascii=True, sort_keys=True, + separators=(',', ':'), default=str, + ).encode('utf-8') + if len(encoded_metadata) > 16 * 1024 * 1024: + raise RuntimeError( + f'legacy target scan {candidate_id} metadata exceeds its normalized bound' + ) + prepared.append({ + 'id': candidate_id, 'payload': payload, + 'payload_bytes': payload_size, 'findings': findings, 'errors': errors, + 'metadata': encoded_metadata, + 'metadata_sha256': hashlib.sha256(encoded_metadata).hexdigest(), + }) + ids = [item['id'] for item in prepared] + placeholders = ','.join('?' for _ in ids) + finding_rows = conn.execute( + f'''SELECT id, target_scan_id, finding_uid FROM findings + WHERE target_scan_id IN ({placeholders}) ORDER BY target_scan_id, id''', + ids, + ).fetchall() + findings_by_scan = {item['id']: [] for item in prepared} + for finding_row in finding_rows: + findings_by_scan[int(finding_row['target_scan_id'])].append(finding_row) + error_rows = conn.execute( + f'''SELECT target_scan_id, COUNT(*) AS count FROM errors + WHERE target_scan_id IN ({placeholders}) GROUP BY target_scan_id''', + ids, + ).fetchall() + errors_by_scan = {int(row['target_scan_id']): int(row['count']) for row in error_rows} + existing_metadata_rows = conn.execute( + f'''SELECT target_scan_id, metadata_sha256, metadata_bytes FROM scan_result_compat + WHERE target_scan_id IN ({placeholders})''', + ids, + ).fetchall() + existing_metadata = {int(row['target_scan_id']): row for row in existing_metadata_rows} + now = utc_now_iso() + metadata_inserts = [] + finding_inserts = [] + finding_updates = [] + for item in prepared: + finding_group = findings_by_scan[item['id']] + if ( + len(finding_group) != len(item['findings']) + or errors_by_scan.get(item['id'], 0) != len(item['errors']) + ): + raise RuntimeError( + f'legacy target scan {item["id"]} normalized child counts do not match raw history' + ) + existing = existing_metadata.get(item['id']) + if existing and ( + existing['metadata_sha256'] != item['metadata_sha256'] + or int(existing['metadata_bytes']) != len(item['metadata']) + ): + raise RuntimeError( + f'legacy target scan {item["id"]} has conflicting compatibility metadata' + ) + if not existing: + metadata_inserts.append(( + item['id'], item['metadata'].decode('utf-8'), + item['metadata_sha256'], len(item['metadata']), now, + )) + for finding_row, finding in zip(finding_group, item['findings']): + if not isinstance(finding, dict): + raise RuntimeError( + f'legacy target scan {item["id"]} contains a non-object finding' + ) + finding_uid = str(finding.get('finding_uid') or '') + if finding_uid and finding_uid != str(finding_row['finding_uid'] or ''): + raise RuntimeError( + f'legacy target scan {item["id"]} finding order/identity changed' + ) + compat = db._compat_payload(finding) + finding_inserts.append(( + finding_row['id'], compat['raw_value'], compat['raw_v2_value'], + compat['structured_data_json'], compat['extra_data_json'], + compat['analysis_info_json'], compat['extension_json'], + compat['payload_sha256'], compat['payload_bytes'], + compat['payload_omitted'], now, + )) + finding_updates.append((finding_row['id'], compat)) + + if metadata_inserts: + values = ','.join('(?, 2, ?, ?, ?, \'bounded\', ?)' for _ in metadata_inserts) + conn.execute( + '''INSERT INTO scan_result_compat( + target_scan_id, schema_version, metadata_json, metadata_sha256, + metadata_bytes, reconstruction_status, created_at + ) VALUES ''' + values, + tuple(value for row in metadata_inserts for value in row), + ) + if finding_inserts: + finding_ids = [int(row[0]) for row in finding_inserts] + existing_payloads = {} + for start in range(0, len(finding_ids), 10000): + id_batch = finding_ids[start:start + 10000] + finding_placeholders = ','.join('?' for _ in id_batch) + existing_payload_rows = conn.execute( + f'''SELECT finding_id, payload_sha256, payload_bytes, payload_omitted + FROM finding_compat_payloads + WHERE finding_id IN ({finding_placeholders})''', + id_batch, + ).fetchall() + existing_payloads.update({ + int(row['finding_id']): row for row in existing_payload_rows + }) + missing_payloads = [] + for values_row in finding_inserts: + existing = existing_payloads.get(int(values_row[0])) + if existing and ( + existing['payload_sha256'] != values_row[7] + or int(existing['payload_bytes']) != int(values_row[8]) + or bool(existing['payload_omitted']) != bool(values_row[9]) + ): + raise RuntimeError('legacy finding has conflicting compatibility data') + if not existing: + missing_payloads.append(values_row) + if missing_payloads: + for start in range(0, len(missing_payloads), 5000): + payload_batch = missing_payloads[start:start + 5000] + values = ','.join( + '(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)' for _ in payload_batch + ) + conn.execute( + '''INSERT INTO finding_compat_payloads( + finding_id, raw_value, raw_v2_value, structured_data_json, + extra_data_json, analysis_info_json, extension_json, + payload_sha256, payload_bytes, payload_omitted, created_at + ) VALUES ''' + values, + tuple(value for row in payload_batch for value in row), + ) + for start in range(0, len(finding_updates), 10000): + update_batch = finding_updates[start:start + 10000] + values = ','.join( + '(?::bigint, ?::text, ?::bigint, ?::integer)' for _ in update_batch + ) + conn.execute( + '''UPDATE findings AS f SET raw_finding_json = NULL, + raw_payload_sha256 = v.payload_sha256, + raw_payload_bytes = v.payload_bytes, + raw_payload_omitted = v.payload_omitted + FROM (VALUES ''' + values + ''') AS v( + id, payload_sha256, payload_bytes, payload_omitted + ) WHERE f.id = v.id''', + tuple( + value + for finding_id, compat in update_batch + for value in ( + finding_id, compat['payload_sha256'], compat['payload_bytes'], + compat['payload_omitted'], + ) + ), + ) + updated = conn.execute( + '''UPDATE target_scans SET raw_result_json = NULL, compat_schema_version = 2, + raw_result_storage = 'normalized_v2' + WHERE id IN (''' + placeholders + ''') + AND raw_result_storage != 'normalized_v2' AND raw_result_json IS NOT NULL''', + ids, + ) + if int(updated.rowcount or 0) != len(prepared): + raise RuntimeError('legacy target scan normalization lost its row fence') + conn.commit() + except Exception: + conn.rollback() + raise + processed += len(prepared) + consumed += sum(item['payload_bytes'] for item in prepared) + if blocked_id is not None: + break + remaining = conn.execute( + '''SELECT COUNT(*) AS count FROM target_scans + WHERE raw_result_storage != 'normalized_v2' AND raw_result_json IS NOT NULL''' + ).fetchone() + conn.commit() + return { + 'processed': processed, + 'bytes': consumed, + 'remaining': int(remaining['count'] or 0), + 'blocked_target_scan_id': blocked_id, + 'reviewed_oversized': reviewed, + } + + +def import_legacy_result_spool(db, config, max_rows=1000): + conn = getattr(db, 'conn', None) + if not conn or not conn.is_postgres: + raise RuntimeError('legacy result-spool import requires PostgreSQL') + global_config = (config or {}).get('global') or {} + spool_dir = reviewed_legacy_result_spool(config) + spool = ResultSpool( + spool_dir, + max_event_bytes=int(global_config.get( + 'legacy_result_spool_max_event_bytes', global_config.get('result_spool_max_event_bytes', 192 * 1024 * 1024), + )), + max_events=int(global_config.get( + 'legacy_result_spool_max_events', global_config.get('result_spool_max_events', 10000), + )), + max_total_bytes=int(global_config.get( + 'legacy_result_spool_max_total_bytes', global_config.get('result_spool_max_total_bytes', 3 * 1024 * 1024 * 1024), + )), + min_free_bytes=0, + ) + max_rows = min(10000, max(1, int(max_rows))) + if not db.try_acquire_result_spool_publisher(): + raise RuntimeError('legacy result-spool publisher advisory lock is unavailable') + imported = 0 + try: + while imported < max_rows: + record = spool.next_pending_event() + if record is None: + break + try: + outcome = db.ingest_scan_event(record.envelope) + except ScanEventConflictError as exc: + spool.quarantine_event( + record.event_id, f'database scan-event hash conflict during offline import: {exc}', + ) + raise RuntimeError( + f'legacy scan event {record.event_id} was quarantined after a hash conflict' + ) from exc + if ( + not outcome or not outcome.get('ingested') + or str(outcome.get('scan_event_id') or '') != record.event_id + or str(outcome.get('scan_event_hash') or '') != record.event_hash + ): + raise RuntimeError(f'legacy scan event {record.event_id} was not confirmed by PostgreSQL') + if spool.acknowledge(record.event_id, record.event_hash) is not True: + raise RuntimeError(f'legacy scan event {record.event_id} acknowledgement was not confirmed') + imported += 1 + remaining = spool.next_pending_event() is not None + return {'imported': imported, 'remaining': int(remaining)} + finally: + if db.release_result_spool_publisher() is not True: + raise RuntimeError('legacy result-spool publisher advisory lock release was not confirmed') + + +def review_pipeline_quarantine_manifest(db, manifest_path, max_rows=1000, config=None): + require_private_file(manifest_path) + manifest = read_private_json(manifest_path, max_bytes=1024 * 1024) + if not isinstance(manifest, dict) or manifest.get('type') != 'truf-pipeline-quarantine-review-v1': + raise RuntimeError('pipeline quarantine review manifest type is invalid') + entries = manifest.get('entries') + if not isinstance(entries, list) or len(entries) > min(10000, max(1, int(max_rows))): + raise RuntimeError('pipeline quarantine review manifest exceeds its row bound') + manifest_bytes = json.dumps( + manifest, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8') + manifest_hash = hashlib.sha256(manifest_bytes).hexdigest() + reviewed = 0 + duplicates = 0 + for entry in entries: + if not isinstance(entry, dict): + raise RuntimeError('pipeline quarantine review entry is not an object') + required = ('id', 'reason_code', 'payload_sha256', 'action') + if any(name not in entry for name in required): + raise RuntimeError('pipeline quarantine review entry is incomplete') + canonical = json.dumps( + entry, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8') + audit_hash = hashlib.sha256( + b'truf-pipeline-quarantine-review-v1|' + manifest_hash.encode('ascii') + b'|' + canonical + ).hexdigest() + row = db.pipeline_quarantine_for_review(entry['id']) + if not row: + raise RuntimeError('pipeline quarantine review row is absent') + global_config = (config or {}).get('global') or {} + if row['object_type'] == 'result_bundle': + bundle_root = global_config.get('result_bundle_dir') + if not bundle_root: + raise RuntimeError('bundle quarantine review root is unavailable') + reservation = db.conn.execute( + '''SELECT id, state, ready_relative_path, docker_layer_plan_json + FROM result_reservations + WHERE id = ?''', + (row['reservation_id'],), + ).fetchone() + db.conn.commit() + if not reservation: + raise RuntimeError('bundle quarantine reservation is absent') + if reservation['state'] != 'quarantined': + ready = inspect_private_relative_path( + bundle_root, reservation['ready_relative_path'], + ) + quarantined = inspect_private_relative_path( + bundle_root, row['source_relative_path'], + ) + if ( + ready.state == PrivatePathState.UNKNOWN + or quarantined.state == PrivatePathState.UNKNOWN + ): + raise RuntimeError('prepared bundle quarantine physical state is unknown') + if ( + ready.state == PrivatePathState.PRESENT + and quarantined.state == PrivatePathState.ABSENT + ): + durable_publish(ready.path, quarantined.path) + quarantined = inspect_private_relative_path( + bundle_root, row['source_relative_path'], + ) + ready = inspect_private_relative_path( + bundle_root, reservation['ready_relative_path'], + ) + if not ( + ready.state == PrivatePathState.ABSENT + and quarantined.state == PrivatePathState.PRESENT + ): + raise RuntimeError( + 'prepared bundle quarantine cannot be finalized from conflicting physical state' + ) + db.quarantine_result_bundle( + reservation['id'], row['reason_code'], row['reason_detail'] or '', + row['source_relative_path'], payload_sha256=row['payload_sha256'] or '', + byte_count=row['byte_count'], physical_confirmed=True, + ) + row = db.pipeline_quarantine_for_review(entry['id']) + if str(entry['action']).lower() == 'rescan' and ( + row['object_type'] != 'result_bundle' + or reservation['docker_layer_plan_json'] is None + ): + raise RuntimeError('quarantine rescan requires a Docker layer result bundle') + if str(entry['action']).lower() == 'retry' and row['object_type'] == 'result_bundle': + global_config = (config or {}).get('global') or {} + bundle_root = global_config.get('result_bundle_dir') + if not bundle_root: + raise RuntimeError('bundle quarantine retry root is unavailable') + reservation = db.conn.execute( + '''SELECT id, bundle_id, scan_event_id, ready_relative_path, + docker_layer_plan_json + FROM result_reservations WHERE id = ?''', + (row['reservation_id'],), + ).fetchone() + db.conn.commit() + if not reservation: + raise RuntimeError('bundle quarantine retry reservation is absent') + if reservation['docker_layer_plan_json'] is not None: + raise RuntimeError( + 'Docker layer bundle quarantine requires a fresh parent claim' + ) + quarantined = inspect_private_relative_path(bundle_root, row['source_relative_path']) + ready = inspect_private_relative_path(bundle_root, reservation['ready_relative_path']) + if quarantined.state == PrivatePathState.UNKNOWN or ready.state == PrivatePathState.UNKNOWN: + raise RuntimeError('bundle quarantine retry physical state is unknown') + if quarantined.state == PrivatePathState.PRESENT and ready.state == PrivatePathState.ABSENT: + candidate_path = quarantined.path + elif quarantined.state == PrivatePathState.ABSENT and ready.state == PrivatePathState.PRESENT: + candidate_path = ready.path + else: + raise RuntimeError('bundle quarantine retry requires exactly one physical artifact') + if not private_file_ready(candidate_path): + raise RuntimeError('bundle quarantine retry artifact is not an exact private file') + if os.path.getsize(candidate_path) != int(row['byte_count']): + raise RuntimeError('bundle quarantine retry byte count conflicts with reviewed evidence') + if sha256_file(candidate_path) != str(row['payload_sha256'] or ''): + raise RuntimeError('bundle quarantine retry payload hash conflicts with reviewed evidence') + validated = ResultBundleReader( + candidate_path, + max_event_bytes=int(global_config.get('result_bundle_max_event_bytes', 192 * 1024 * 1024)), + ).validate() + if ( + int(validated.reservation_id) != int(reservation['id']) + or str(validated.bundle_id) != str(reservation['bundle_id']) + or str(validated.scan_event_id) != str(reservation['scan_event_id']) + ): + raise RuntimeError('bundle quarantine retry identity conflicts with its reservation') + if quarantined.state == PrivatePathState.PRESENT: + ensure_private_directory(os.path.dirname(ready.path), reject_reparse=True) + durable_publish(quarantined.path, ready.path) + quarantined = inspect_private_relative_path(bundle_root, row['source_relative_path']) + ready = inspect_private_relative_path(bundle_root, reservation['ready_relative_path']) + if not ( + quarantined.state == PrivatePathState.ABSENT + and ready.state == PrivatePathState.PRESENT + and private_file_ready(ready.path) + ): + raise RuntimeError('bundle quarantine retry publication was not confirmed') + if str(entry['action']).lower() in ('discard', 'rescan') and row.get('source_relative_path'): + root = ( + global_config.get('result_bundle_dir') + if row['subsystem'] == 'result_ingester' + else global_config.get('results_dir') + ) + if not root: + raise RuntimeError('physical quarantine review root is unavailable') + inspection = inspect_private_relative_path(root, row['source_relative_path']) + if inspection.state == PrivatePathState.UNKNOWN: + raise RuntimeError('physical quarantine artifact state is unknown') + if inspection.state == PrivatePathState.PRESENT: + durable_unlink(inspection.path) + inspection = inspect_private_relative_path(root, row['source_relative_path']) + if inspection.state != PrivatePathState.ABSENT: + raise RuntimeError('physical quarantine artifact unlink was not confirmed') + if row['object_type'] == 'projection_tail': + db.mark_pipeline_artifact_deleted(row['object_id']) + elif row['object_type'] == 'result_bundle': + artifact = db.conn.execute( + '''SELECT id FROM pipeline_artifacts + WHERE subsystem = 'result_bundle' + AND artifact_kind = 'bundle_quarantine' AND owner_id = ?''', + (row['reservation_id'],), + ).fetchone() + db.conn.commit() + if not artifact: + raise RuntimeError('bundle quarantine artifact index is absent') + db.mark_pipeline_artifact_deleted(artifact['id']) + outcome = db.review_pipeline_quarantine( + entry['id'], entry['reason_code'], entry['payload_sha256'], + entry['action'], audit_hash, + ) + reviewed += int(bool(outcome and outcome.get('reviewed'))) + duplicates += int(bool(outcome and outcome.get('duplicate'))) + return { + 'manifest_sha256': manifest_hash, + 'reviewed': reviewed, + 'duplicates': duplicates, + } + + +def rebuild_jsonl_projections( + db, output_root, max_rows=1000, max_bytes=192 * 1024 * 1024, + max_seconds=30, +): + from jsonl_projector import JsonlProjector + + if not db.conn or not db.conn.is_postgres: + raise RuntimeError('full JSONL rebuild requires PostgreSQL') + db.require_runtime_safety_schema() + db.require_final_cutover() + output_root = os.path.abspath(output_root) + existed = os.path.lexists(output_root) + output_root = ensure_private_directory(output_root, reject_reparse=True) + state_path = os.path.join(output_root, '.rebuild-state.json') + identity = db.conn.execute( + '''SELECT current_database() AS database_name, current_user AS database_user, + inet_server_port() AS database_port''' + ).fetchone() + db.conn.commit() + cutover = db.final_cutover_status() + authority = hashlib.sha256(json.dumps({ + 'database': dict(identity), + 'cutover_evidence_sha256': cutover['evidence_sha256'], + }, ensure_ascii=True, sort_keys=True, separators=(',', ':')).encode('utf-8')).hexdigest() + if not os.path.lexists(state_path): + if existed: + with os.scandir(output_root) as entries: + if next(entries, None) is not None: + raise RuntimeError('new JSONL rebuild output directory must be empty') + state = { + 'schema': 1, + 'authority_sha256': authority, + 'phase': 'scans', + 'last_scan_id': 0, + 'last_keycheck_result_id': 0, + 'offsets': {}, + 'prepared': None, + 'completed': False, + } + atomic_write_private_json(state_path, state) + state = read_private_json(state_path, max_bytes=1024 * 1024) + if ( + not isinstance(state, dict) or state.get('schema') != 1 + or state.get('authority_sha256') != authority + ): + raise RuntimeError('JSONL rebuild state is invalid') + + results_dir = ensure_private_directory( + os.path.join(output_root, 'results'), reject_reparse=True, + ) + keycheck_dir = ensure_private_directory( + os.path.join(output_root, 'keychecks'), reject_reparse=True, + ) + projector = JsonlProjector( + db, results_dir, 'offline-rebuild', keycheck_dir=keycheck_dir, + artifact_tracking=False, + ) + + def output_path(relative): + path = os.path.abspath(os.path.join(output_root, str(relative).replace('/', os.sep))) + if path == output_root or os.path.commonpath((output_root, path)) != output_root: + raise RuntimeError('JSONL rebuild path escapes its output root') + ensure_private_directory(os.path.dirname(path), reject_reparse=True) + if os.path.lexists(path): + require_private_file(path) + else: + descriptor = os.open( + path, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), + 0o600, + ) + os.close(descriptor) + harden_private_file(path) + fsync_directory(os.path.dirname(path)) + return path + + prepared = state.get('prepared') + if prepared: + if not isinstance(prepared, dict) or not isinstance(prepared.get('offsets'), dict): + raise RuntimeError('JSONL rebuild prepared state is invalid') + for relative, offset in prepared['offsets'].items(): + path = output_path(relative) + size = os.path.getsize(path) + offset = int(offset) + if size < offset: + raise RuntimeError('JSONL rebuild output is shorter than its committed offset') + if size != offset: + with open(path, 'r+b', buffering=0) as handle: + handle.truncate(offset) + handle.flush() + os.fsync(handle.fileno()) + state['prepared'] = None + atomic_write_private_json(state_path, state) + + max_rows = min(100000, max(1, int(max_rows))) + max_bytes = max(1, int(max_bytes)) + deadline = time.monotonic() + max(0.1, float(max_seconds)) + rows_written = 0 + bytes_written = 0 + blocked = None + + def stream_relative(stream_name, service=''): + if stream_name == 'scan_results': + return 'results/scan_results.jsonl' + if stream_name == 'found_secrets': + return 'results/found_secrets.jsonl' + if stream_name == 'scan_errors': + return 'results/scan_errors.log' + if stream_name.endswith(':results'): + return f'keychecks/{service}/{service}Results.jsonl' + return f'keychecks/{service}/{service}Checked.txt' + + def append_serialized(job, streams, service=''): + nonlocal bytes_written, blocked + total = sum(int(item.byte_length) for item in streams) + if bytes_written + total > max_bytes: + blocked = {'kind': job['job_kind'], 'id': int(job['id']), 'bytes': total} + return False + offsets = {} + destinations = [] + for item in streams: + relative = stream_relative(item.stream_name, service) + path = output_path(relative) + expected = int(state['offsets'].get(relative, 0)) + if os.path.getsize(path) != expected: + raise RuntimeError('JSONL rebuild output/cursor mismatch') + offsets[relative] = expected + destinations.append((item, relative, path)) + state['prepared'] = { + 'kind': job['job_kind'], 'id': int(job['id']), 'offsets': offsets, + } + atomic_write_private_json(state_path, state) + for item, relative, path in destinations: + with open(path, 'ab', buffering=0) as target, open(item.path, 'rb', buffering=0) as source: + while True: + block = source.read(1024 * 1024) + if not block: + break + target.write(block) + target.flush() + os.fsync(target.fileno()) + state['offsets'][relative] = offsets[relative] + int(item.byte_length) + bytes_written += total + state['prepared'] = None + return True + + try: + while rows_written < max_rows and time.monotonic() < deadline and not state['completed']: + streams = [] + if state['phase'] == 'scans': + row = db.conn.execute( + '''SELECT id, scan_event_id, scan_event_hash FROM target_scans + WHERE id > ? ORDER BY id LIMIT 1''', + (int(state['last_scan_id']),), + ).fetchone() + db.conn.commit() + if not row: + state['phase'] = 'keychecks' + atomic_write_private_json(state_path, state) + continue + event_id = str(row['scan_event_id'] or f'legacy-target-scan-{row["id"]}') + event_hash = str(row['scan_event_hash'] or hashlib.sha256( + f'legacy-target-scan|{row["id"]}'.encode('utf-8') + ).hexdigest()) + job = { + 'id': int(row['id']), 'job_kind': 'scan_event', + 'target_scan_id': int(row['id']), 'event_id': event_id, + 'event_hash': event_hash, 'required_stream_mask': 7, + 'capacity_bytes': 192 * 1024 * 1024, + } + streams = projector._serialize(job) + if not append_serialized(job, streams): + break + state['last_scan_id'] = int(row['id']) + else: + row = db.conn.execute( + '''SELECT id, event_id, service FROM keycheck_results + WHERE id > ? ORDER BY id LIMIT 1''', + (int(state['last_keycheck_result_id']),), + ).fetchone() + db.conn.commit() + if not row: + state['completed'] = True + atomic_write_private_json(state_path, state) + break + event_id = str(row['event_id'] or f'legacy-keycheck-result-{row["id"]}') + event_hash = hashlib.sha256( + f'keycheck-rebuild|{row["id"]}|{event_id}'.encode('utf-8') + ).hexdigest() + job = { + 'id': int(row['id']), 'job_kind': 'keycheck_event', + 'keycheck_result_id': int(row['id']), 'event_id': event_id, + 'event_hash': event_hash, 'required_stream_mask': 8, + 'capacity_bytes': 192 * 1024 * 1024, + } + streams = projector._serialize(job) + if not append_serialized(job, streams, str(row['service'])): + break + state['last_keycheck_result_id'] = int(row['id']) + rows_written += 1 + atomic_write_private_json(state_path, state) + for item in streams: + try: + os.remove(item.path) + except OSError: + pass + finally: + for item in locals().get('streams', []): + try: + os.remove(item.path) + except OSError: + pass + + return { + 'rows_written': rows_written, + 'bytes_written': bytes_written, + 'phase': state['phase'], + 'completed': bool(state['completed']), + 'blocked': blocked, + 'output_root': output_root, + } + + +def initialize_projection_cursors_from_existing_files(db, config): + conn = getattr(db, 'conn', None) + if not conn: + raise RuntimeError('projection cursor initialization requires a database') + global_config = ((config or {}).get('global') or {}) + results_value = global_config.get('results_dir') + if not results_value and global_config.get('runtime_dir'): + results_value = os.path.join(global_config['runtime_dir'], 'results') + if not results_value: + raise RuntimeError('projection cursor initialization requires results_dir') + results_dir = os.path.abspath(results_value) + initialized = {} + try: + for stream_name in ('scan_results', 'found_secrets', 'scan_errors'): + suffix = ' FOR UPDATE' if conn.is_postgres else '' + row = conn.execute( + f'''SELECT s.base_relative_path, c.committed_offset, c.last_append_id, + c.last_job_id + FROM projection_streams s + JOIN projection_cursors c ON c.stream_name = s.stream_name + WHERE s.stream_name = ?{suffix}''', + (stream_name,), + ).fetchone() + if not row: + raise RuntimeError(f'projection cursor is absent: {stream_name}') + path = os.path.abspath(os.path.join(results_dir, str(row['base_relative_path']))) + if os.path.commonpath((results_dir, path)) != results_dir or path == results_dir: + raise RuntimeError(f'projection stream path escapes results_dir: {stream_name}') + size = 0 + try: + details = os.stat(path, follow_symlinks=False) + except FileNotFoundError: + details = None + except OSError as exc: + raise RuntimeError( + f'projection cursor file state is unknown: {stream_name}' + ) from exc + if details is not None: + if not stat.S_ISREG(details.st_mode): + raise RuntimeError(f'projection stream is not a regular file: {stream_name}') + require_private_file(path) + size = int(details.st_size) + offset = int(row['committed_offset'] or 0) + if offset == size: + initialized[stream_name] = size + continue + append_count = conn.execute( + 'SELECT COUNT(*) AS count FROM projection_appends WHERE stream_name = ?', + (stream_name,), + ).fetchone() + if ( + offset != 0 or row['last_append_id'] is not None + or row['last_job_id'] is not None or int(append_count['count'] or 0) != 0 + ): + raise RuntimeError( + f'projection cursor/file mismatch requires reviewed recovery: {stream_name}' + ) + conn.execute( + '''UPDATE projection_cursors SET committed_offset = ?, updated_at = ? + WHERE stream_name = ? AND committed_offset = 0 + AND last_append_id IS NULL AND last_job_id IS NULL''', + (size, utc_now_iso(), stream_name), + ) + initialized[stream_name] = size + conn.commit() + return initialized + except Exception: + conn.rollback() + raise + + +def reconcile_todo_file( + db, + todo_path, + source, + platform, + query='reconciled', + max_rows=1000, + max_bytes=4 * 1024 * 1024, + max_seconds=5.0, +): + if not db or not getattr(db, 'conn', None): + raise RuntimeError('database connection is unavailable') + db.require_runtime_safety_schema() + todo_path = os.path.normcase(os.path.abspath(todo_path)) + reject_reparse_components(todo_path) + if 'checked' in os.path.basename(todo_path).lower(): + raise ValueError('checked files are not accepted as reconciliation input') + if is_reparse_point(todo_path) or not os.path.isfile(todo_path): + raise FileNotFoundError(todo_path) + max_rows = max(1, int(max_rows)) + max_bytes = max(1, int(max_bytes)) + max_seconds = max(0.001, float(max_seconds)) + conn = db.conn + last_change = None + for mutation_attempt in range(RECONCILIATION_MUTATION_RETRIES): + started = time.monotonic() + descriptor = None + try: + flags = os.O_RDONLY | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0) + descriptor = os.open(todo_path, flags) + initial_stat = os.fstat(descriptor) + if not stat.S_ISREG(initial_stat.st_mode): + raise ValueError('reconciliation input must be a regular file') + snapshot = _stat_identity(initial_stat) + if not _same_file_snapshot(todo_path, (descriptor, snapshot)): + raise ReconciliationFileChanged('reconciliation input changed while opening') + identity = todo_file_identity(todo_path, initial_stat) + file_size = snapshot[2] + file_mtime_ns = snapshot[3] + + if conn.is_sqlite: + conn.execute('BEGIN IMMEDIATE') + cursor = conn.execute( + '''SELECT file_identity, source, platform, byte_offset, line_number, + discarding_oversized, oversized_line_start, + cumulative_rows, cumulative_bytes, cumulative_inserted, cumulative_rejected, + completed_at + FROM target_queue_reconciliation_cursors WHERE source_file = ?''', + (todo_path,), + ).fetchone() + if cursor and (str(cursor['source']) != str(source) or str(cursor['platform']) != str(platform)): + raise ValueError( + f'reconciliation cursor is already bound to {cursor["source"]}/{cursor["platform"]}; ' + f'it cannot be reused for {source}/{platform}' + ) + same_identity = bool(cursor and cursor['file_identity'] == identity) + offset = int(cursor['byte_offset'] or 0) if same_identity else 0 + line_number = int(cursor['line_number'] or 0) if same_identity else 0 + discarding = bool(cursor['discarding_oversized']) if same_identity else False + oversized_line_start = int(cursor['oversized_line_start'] or 0) if same_identity and cursor['oversized_line_start'] is not None else None + cumulative_rows = int(cursor['cumulative_rows'] or 0) if same_identity else 0 + cumulative_bytes = int(cursor['cumulative_bytes'] or 0) if same_identity else 0 + cumulative_inserted = int(cursor['cumulative_inserted'] or 0) if same_identity else 0 + cumulative_rejected = int(cursor['cumulative_rejected'] or 0) if same_identity else 0 + previous_completed_at = cursor['completed_at'] if same_identity else None + if offset < 0 or offset > file_size: + offset = line_number = 0 + discarding = False + oversized_line_start = None + cumulative_rows = cumulative_bytes = cumulative_inserted = cumulative_rejected = 0 + + rows_read = 0 + bytes_read = 0 + rejected = 0 + eligible = [] + resolver_rows = [] + seen = set() + malformed = [] + unresolved_docker = [] + bounded_row = None + issues = [] + next_offset = offset + next_line = line_number + + with os.fdopen(descriptor, 'rb', closefd=False) as handle: + handle.seek(offset) + while time.monotonic() - started < max_seconds: + if not discarding and rows_read >= max_rows: + break + if bytes_read >= max_bytes and (rows_read or discarding): + break + row_start = handle.tell() + if discarding: + read_limit = min( + RECONCILIATION_ABSOLUTE_ROW_BYTES + 1, + max(1, max_bytes - bytes_read), + ) + else: + read_limit = RECONCILIATION_ABSOLUTE_ROW_BYTES + 1 + raw = handle.readline(read_limit) + if not raw: + break + candidate_offset = handle.tell() + reached_eof = candidate_offset >= file_size + + if discarding: + bytes_read += len(raw) + next_offset = candidate_offset + if raw.endswith(b'\n') or reached_eof: + discarding = False + oversized_line_start = None + next_line += 1 + continue + + oversized = len(raw) > RECONCILIATION_ABSOLUTE_ROW_BYTES and not raw.endswith(b'\n') + if oversized: + if rows_read and bytes_read + len(raw) > max_bytes: + handle.seek(row_start) + break + bytes_read += len(raw) + next_offset = candidate_offset + issue_line = next_line + 1 + rows_read += 1 + rejected += 1 + oversized_line_start = row_start + discarding = not reached_eof + if not discarding: + next_line += 1 + oversized_line_start = None + reason = f'row exceeds absolute {RECONCILIATION_ABSOLUTE_ROW_BYTES}-byte reconciliation row limit' + bounded_row = {'line': issue_line, 'offset': row_start, 'reason': reason} + issue = { + 'line': issue_line, 'offset': row_start, 'reason': reason, + 'target': _safe_target_preview(raw), + } + malformed.append(issue) + issues.append(issue) + continue + + if rows_read and bytes_read + len(raw) > max_bytes: + handle.seek(row_start) + break + + decode_error = None + target = '' + try: + target = raw.decode('utf-8').strip().lstrip('\ufeff') + except UnicodeDecodeError as exc: + decode_error = exc + target_error = reconciliation_target_error(target, platform) if target and not decode_error else '' + + bytes_read += len(raw) + next_offset = candidate_offset + rows_read += 1 + next_line += 1 + if decode_error is not None: + rejected += 1 + issue = { + 'line': next_line, 'offset': row_start, + 'reason': f'invalid UTF-8 at byte {decode_error.start}', + 'target': _safe_target_preview(raw), + } + malformed.append(issue) + issues.append(issue) + continue + if not target: + continue + error = target_error + if error: + rejected += 1 + item = { + 'line': next_line, 'offset': row_start, + 'target': _safe_target_preview(target), 'reason': error, + } + issues.append(item) + if error == 'unresolved bare Docker repository': + unresolved_docker.append(item) + normalized = normalize_target(target, platform) + if normalized and normalized not in seen: + seen.add(normalized) + resolver_rows.append((target, normalized)) + else: + malformed.append(item) + continue + normalized = normalize_target(target, platform) + if normalized in seen: + continue + seen.add(normalized) + eligible.append((target, normalized)) + + now = utc_now_iso() + for issue in issues: + conn.execute( + '''INSERT INTO target_queue_reconciliation_issues ( + source_file, file_identity, source, platform, line_number, byte_offset, + reason, target_preview, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(source_file, file_identity, line_number, byte_offset, reason) DO NOTHING''', + ( + todo_path, identity, source, platform, issue['line'], issue['offset'], + issue['reason'], issue.get('target'), now, + ), + ) + + inserted = 0 + for target, normalized in eligible: + cur = conn.execute( + '''INSERT INTO target_queue ( + source, platform, query, target, normalized_target, status, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'pending', ?, ?) + ON CONFLICT(source, normalized_target) DO NOTHING''', + (source, platform, query, target, normalized, now, now), + ) + inserted += max(0, int(getattr(cur, 'rowcount', 0) or 0)) + + resolver_due = datetime.fromtimestamp(time.time() + 3600, timezone.utc).isoformat(timespec='seconds') + resolver_inserted = 0 + for target, normalized in resolver_rows: + cur = conn.execute( + '''INSERT INTO target_queue ( + source, platform, query, target, normalized_target, status, available_after, + resolver_state, resolver_due_at, last_error, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'deferred', ?, 'pending', ?, + 'Docker tag resolution unresolved', ?, ?) + ON CONFLICT(source, normalized_target) DO NOTHING''', + (source, platform, query, target, normalized, resolver_due, resolver_due, now, now), + ) + resolver_inserted += max(0, int(getattr(cur, 'rowcount', 0) or 0)) + inserted += resolver_inserted + + if not _same_file_snapshot(todo_path, (descriptor, snapshot)): + raise ReconciliationFileChanged('reconciliation input changed before cursor update') + at_eof = next_offset >= file_size and not discarding + completed_at = (previous_completed_at or now) if at_eof else None + cumulative_rows += rows_read + cumulative_bytes += bytes_read + cumulative_inserted += inserted + cumulative_rejected += rejected + unresolved_row = conn.execute( + '''SELECT COUNT(*) AS count FROM target_queue_reconciliation_issues + WHERE source_file = ? AND source = ? AND platform = ? AND resolved_at IS NULL''', + (todo_path, source, platform), + ).fetchone() + unresolved_count = int(unresolved_row['count'] or 0) + report = { + 'source_file': todo_path, + 'file_identity': identity, + 'file_size': file_size, + 'file_mtime_ns': file_mtime_ns, + 'source': source, + 'platform': platform, + 'start_offset': offset, + 'byte_offset': next_offset, + 'line_number': next_line, + 'rows_read': rows_read, + 'bytes_read': bytes_read, + 'eligible_rows': len(eligible), + 'inserted_rows': inserted, + 'resolver_rows_inserted': resolver_inserted, + 'rejected_rows': rejected, + 'malformed_rows': malformed, + 'unresolved_docker_rows': unresolved_docker, + 'unresolved_issues': unresolved_count, + 'bounded_row': bounded_row, + 'discarding_oversized': discarding, + 'at_eof': at_eof, + 'completed_at': completed_at, + 'cumulative_rows': cumulative_rows, + 'cumulative_bytes': cumulative_bytes, + 'cumulative_inserted': cumulative_inserted, + 'cumulative_rejected': cumulative_rejected, + 'elapsed_sec': max(0.0, time.monotonic() - started), + } + conn.execute( + '''INSERT INTO target_queue_reconciliation_cursors ( + source_file, file_identity, file_size, file_mtime_ns, source, platform, + byte_offset, line_number, discarding_oversized, oversized_line_start, + cumulative_rows, cumulative_bytes, cumulative_inserted, cumulative_rejected, + completed_at, last_report_json, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(source_file) DO UPDATE SET + file_identity = excluded.file_identity, + file_size = excluded.file_size, + file_mtime_ns = excluded.file_mtime_ns, + byte_offset = excluded.byte_offset, + line_number = excluded.line_number, + discarding_oversized = excluded.discarding_oversized, + oversized_line_start = excluded.oversized_line_start, + cumulative_rows = excluded.cumulative_rows, + cumulative_bytes = excluded.cumulative_bytes, + cumulative_inserted = excluded.cumulative_inserted, + cumulative_rejected = excluded.cumulative_rejected, + completed_at = excluded.completed_at, + last_report_json = excluded.last_report_json, + updated_at = excluded.updated_at''', + ( + todo_path, identity, file_size, file_mtime_ns, source, platform, + next_offset, next_line, 1 if discarding else 0, oversized_line_start, + cumulative_rows, cumulative_bytes, cumulative_inserted, cumulative_rejected, + completed_at, json_dumps(report), now, + ), + ) + if not _same_file_snapshot(todo_path, (descriptor, snapshot)): + raise ReconciliationFileChanged('reconciliation input changed before commit') + conn.commit() + return report + except ReconciliationFileChanged as exc: + last_change = exc + try: + conn.rollback() + except Exception: + pass + if mutation_attempt + 1 >= RECONCILIATION_MUTATION_RETRIES: + raise + time.sleep(0.01) + except Exception: + try: + conn.rollback() + except Exception: + pass + raise + finally: + if descriptor is not None: + try: + os.close(descriptor) + except OSError: + pass + raise last_change or ReconciliationFileChanged('reconciliation input remained unstable') + + +def resolve_reconciliation_issues(db, issue_ids): + requested = sorted({int(value) for value in issue_ids or []}) + if not requested: + return 0 + conn = getattr(db, 'conn', None) + if not conn: + raise RuntimeError('database connection is unavailable') + placeholders = ','.join('?' for _ in requested) + try: + if conn.is_sqlite: + conn.execute('BEGIN IMMEDIATE') + lock_suffix = ' FOR UPDATE' if conn.is_postgres else '' + rows = conn.execute( + f'''SELECT id, resolved_at FROM target_queue_reconciliation_issues + WHERE id IN ({placeholders}){lock_suffix}''', + requested, + ).fetchall() + found = {int(row['id']): row['resolved_at'] for row in rows} + if set(found) != set(requested) or any(found[issue_id] is not None for issue_id in requested): + raise RuntimeError('one or more requested reconciliation issues were absent or already resolved') + cursor = conn.execute( + f'''UPDATE target_queue_reconciliation_issues SET resolved_at = ? + WHERE id IN ({placeholders}) AND resolved_at IS NULL''', + (utc_now_iso(), *requested), + ) + if int(getattr(cursor, 'rowcount', 0) or 0) != len(requested): + raise RuntimeError('reconciliation issue set changed before resolution commit') + conn.commit() + return len(requested) + except Exception: + try: + conn.rollback() + except Exception: + pass + raise + + +def _canonical_json_sha256(value): + encoded = json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8') + return hashlib.sha256(encoded).hexdigest() + + +def _target_queue_platform_for_source(source): + source = str(source or '').strip() + return 'docker' if source in ('docker', 'dockerhub') else source + + +def configured_target_queue_query_policy(config): + policy = {} + document = [] + rejected_entries = validate_rejected_query_policy(config) + rejected = {} + sources = (config or {}).get('sources') or {} + if not isinstance(sources, dict): + raise RuntimeError('configured source policy is not a mapping') + for source in sorted(sources): + source_config = sources[source] + if not isinstance(source_config, dict) or 'queries' not in source_config: + continue + raw_queries = source_config.get('queries') + if isinstance(raw_queries, str): + raw_queries = raw_queries.split(',') + if not isinstance(raw_queries, (list, tuple)): + raise RuntimeError(f'configured query policy is invalid for source {source}') + queries = [] + for raw_query in raw_queries: + query = str(raw_query or '').strip() + if not query: + raise RuntimeError(f'configured query policy contains an empty query for source {source}') + if query in queries: + raise RuntimeError(f'configured query policy contains a duplicate query for source {source}') + queries.append(query) + platform = _target_queue_platform_for_source(source) + key = (str(source), platform) + policy[key] = tuple(queries) + document.append({ + 'source': str(source), 'platform': platform, 'queries': sorted(queries), + }) + for entry in rejected_entries: + source = entry['source'] + platform = _target_queue_platform_for_source(source) + rejected.setdefault((source, platform), []).append(entry['query']) + rejected = { + key: tuple(sorted(queries)) for key, queries in sorted(rejected.items()) + } + registry_present = (config or {}).get('query_policy') is not None + policy_document = document + if registry_present: + policy_document = { + 'active': document, + 'rejected': [ + {**entry, 'platform': _target_queue_platform_for_source(entry['source'])} + for entry in rejected_entries + ], + } + return { + 'queries': policy, + 'rejected': rejected, + 'rejected_registry_present': registry_present, + 'document': policy_document, + 'policy_sha256': _canonical_json_sha256(policy_document), + } + + +def _stale_target_queue_rows(db, policy, source, platform, max_rows): + source = str(source or '').strip() + platform = str(platform or '').strip() + queries = policy['queries'].get((source, platform)) + if queries is None or not queries: + raise RuntimeError('target queue policy scope has no configured query allowlist') + maximum = min(100000, max(1, int(max_rows))) + placeholders = ','.join('?' for _ in queries) + rejected = policy.get('rejected', {}).get((source, platform), ()) + registry_present = bool(policy.get('rejected_registry_present')) + if registry_present and not rejected: + return [], [] + if registry_present: + rejected_placeholders = ','.join('?' for _ in rejected) + query_filter = ( + f'AND queue.query IN ({rejected_placeholders}) ' + f'AND queue.query NOT IN ({placeholders})' + ) + query_params = (*rejected, *queries) + else: + query_filter = f'AND queue.query NOT IN ({placeholders})' + query_params = tuple(queries) + rows = db.conn.execute( + f'''SELECT queue.*, + EXISTS ( + SELECT 1 FROM result_reservations reservation + WHERE reservation.queue_id = queue.id + AND reservation.state IN ('scanning','ready','ingesting','db_committed') + ) AS active_reservation, + EXISTS ( + SELECT 1 + FROM docker_image_blob_coverage coverage + JOIN docker_content_blobs blob + ON blob.digest = coverage.blob_digest + AND blob.coverage_policy_sha256 = coverage.coverage_policy_sha256 + WHERE coverage.queue_id = queue.id + AND blob.state IN ('leased','submitted') + ) AS active_docker_blob + FROM target_queue queue + WHERE queue.source = ? AND queue.platform = ? + AND queue.status IN ('pending','deferred','in_progress') + AND queue.query IS NOT NULL AND BTRIM(queue.query) <> '' + {query_filter} + ORDER BY queue.id + LIMIT ?''', + (source, platform, *query_params, maximum + 1), + ).fetchall() + db.conn.commit() + if len(rows) > maximum: + raise RuntimeError(f'stale target queue selection exceeds its {maximum} row bound') + eligible = [] + blockers = [] + fence_fields = ( + 'lease_owner', 'lease_token', 'claim_batch', 'leased_at', 'lease_expires_at', + 'current_result_reservation_id', 'claim_event_id', 'resolver_token', + ) + for row in rows: + blocked = ( + row['status'] not in ('pending', 'deferred') + or any(row[field] is not None for field in fence_fields) + or str(row['resolver_state'] or '') == 'resolving' + or bool(row['active_reservation']) + or bool(row['active_docker_blob']) + ) + if blocked: + blockers.append(int(row['id'])) + continue + eligible.append({ + 'queue_id': int(row['id']), + 'source': str(row['source']), + 'platform': str(row['platform']), + 'query': str(row['query']), + 'prior_status': str(row['status']), + 'prior_updated_at': str(row['updated_at']), + }) + return eligible, blockers + + +def plan_stale_target_queue_cold( + db, config, *, source, platform, config_sha256, max_rows=10000, +): + policy = configured_target_queue_query_policy(config) + entries, blockers = _stale_target_queue_rows( + db, policy, source, platform, max_rows, + ) + if blockers: + raise RuntimeError( + f'stale target queue policy selection has {len(blockers)} fenced row(s)' + ) + manifest = { + 'schema': 1, + 'type': 'truf-target-queue-cold-review-v1', + 'config_sha256': str(config_sha256), + 'policy_sha256': policy['policy_sha256'], + 'source': str(source), + 'platform': str(platform), + 'reason_code': ( + 'rejected_zero_alive' + if policy.get('rejected_registry_present') + else 'query_not_in_canonical_policy' + ), + 'selection_sha256': _canonical_json_sha256(entries), + 'generated_at': utc_now_iso(), + 'entries': entries, + } + return manifest + + +def _cold_reactivation_rows(db, source, platform, query, max_rows): + maximum = min(100000, max(1, int(max_rows))) + rows = db.conn.execute( + '''SELECT queue.*, cold_event.id AS cold_event_id, + cold_event.prior_status AS restore_status, + reverse_event.id AS reverse_event_id, + EXISTS ( + SELECT 1 FROM result_reservations reservation + WHERE reservation.queue_id = queue.id + AND reservation.state IN ('scanning','ready','ingesting','db_committed') + ) AS active_reservation, + EXISTS ( + SELECT 1 + FROM docker_image_blob_coverage coverage + JOIN docker_content_blobs blob + ON blob.digest = coverage.blob_digest + AND blob.coverage_policy_sha256 = coverage.coverage_policy_sha256 + WHERE coverage.queue_id = queue.id + AND blob.state IN ('leased','submitted') + ) AS active_docker_blob + FROM target_queue queue + JOIN target_queue_policy_events cold_event + ON cold_event.queue_id = queue.id AND cold_event.action = 'cold' + AND cold_event.experiment_id IS NULL + LEFT JOIN target_queue_policy_events reverse_event + ON reverse_event.reverses_event_id = cold_event.id + WHERE queue.source = ? AND queue.platform = ? AND queue.query = ? + AND queue.status = 'cold' AND reverse_event.id IS NULL + ORDER BY queue.id + LIMIT ?''', + (str(source), str(platform), str(query), maximum + 1), + ).fetchall() + db.conn.commit() + if len(rows) > maximum: + raise RuntimeError(f'cold target queue selection exceeds its {maximum} row bound') + entries = [] + fence_fields = ( + 'lease_owner', 'lease_token', 'claim_batch', 'leased_at', 'lease_expires_at', + 'current_result_reservation_id', 'claim_event_id', 'resolver_token', + ) + for row in rows: + if ( + any(row[field] is not None for field in fence_fields) + or str(row['resolver_state'] or '') == 'resolving' + or bool(row['active_reservation']) + or bool(row['active_docker_blob']) + ): + raise RuntimeError('cold target queue reactivation has a fenced row') + entries.append({ + 'queue_id': int(row['id']), + 'source': str(row['source']), + 'platform': str(row['platform']), + 'query': str(row['query']), + 'cold_event_id': int(row['cold_event_id']), + 'restore_status': str(row['restore_status']), + 'prior_updated_at': str(row['updated_at']), + }) + return entries + + +def plan_cold_target_queue_reactivation( + db, config, *, source, platform, query, config_sha256, max_rows=10000, +): + policy = configured_target_queue_query_policy(config) + entries = _cold_reactivation_rows(db, source, platform, query, max_rows) + return { + 'schema': 1, + 'type': 'truf-target-queue-reactivation-review-v1', + 'config_sha256': str(config_sha256), + 'policy_sha256': policy['policy_sha256'], + 'source': str(source), + 'platform': str(platform), + 'query': str(query), + 'reason_code': 'reviewed_policy_reactivation', + 'selection_sha256': _canonical_json_sha256(entries), + 'generated_at': utc_now_iso(), + 'entries': entries, + } + + +def load_target_queue_policy_manifest(path, expected_type, max_rows=10000): + require_private_file(path) + manifest = read_private_json(path, max_bytes=TARGET_QUEUE_POLICY_MANIFEST_MAX_BYTES) + if not isinstance(manifest, dict) or manifest.get('type') != expected_type: + raise RuntimeError('target queue policy manifest type is invalid') + expected_keys = { + 'schema', 'type', 'config_sha256', 'policy_sha256', 'source', 'platform', + 'reason_code', 'selection_sha256', 'generated_at', 'entries', + } + if expected_type == 'truf-target-queue-reactivation-review-v1': + expected_keys.add('query') + if set(manifest) != expected_keys or manifest.get('schema') != 1: + raise RuntimeError('target queue policy manifest shape is invalid') + entries = manifest.get('entries') + maximum = min(100000, max(1, int(max_rows))) + if not isinstance(entries, list) or not entries or len(entries) > maximum: + raise RuntimeError('target queue policy manifest entry count is outside its bound') + action = 'cold' if expected_type == 'truf-target-queue-cold-review-v1' else 'reactivate' + entries = ScannerDB._normalize_target_queue_policy_entries(entries, action, maximum) + if entries != manifest['entries']: + raise RuntimeError('target queue policy manifest entries are not canonical') + for name in ('config_sha256', 'policy_sha256', 'selection_sha256'): + if not re.fullmatch(r'[a-f0-9]{64}', str(manifest.get(name) or '')): + raise RuntimeError(f'target queue policy manifest {name} is invalid') + if manifest['selection_sha256'] != _canonical_json_sha256(entries): + raise RuntimeError('target queue policy selection hash is invalid') + return manifest, _canonical_json_sha256(manifest) + + +def _manifest_already_applied(db, manifest, manifest_sha256, action): + rows = db.conn.execute( + '''SELECT id, queue_id, action FROM target_queue_policy_events + WHERE manifest_sha256 = ? ORDER BY queue_id''', + (manifest_sha256,), + ).fetchall() + db.conn.commit() + if not rows: + return False + expected_ids = [int(entry['queue_id']) for entry in manifest['entries']] + actual_ids = [int(row['queue_id']) for row in rows] + if actual_ids != expected_ids or any(row['action'] != action for row in rows): + raise RuntimeError('target queue policy manifest was only partially or differently applied') + return True + + +def apply_stale_target_queue_cold( + db, config, manifest, manifest_sha256, *, config_sha256, max_rows=10000, +): + policy = configured_target_queue_query_policy(config) + if manifest['config_sha256'] != config_sha256: + raise RuntimeError('target queue policy manifest config identity drifted') + if manifest['policy_sha256'] != policy['policy_sha256']: + raise RuntimeError('target queue policy manifest query policy drifted') + if not _manifest_already_applied(db, manifest, manifest_sha256, 'cold'): + current = plan_stale_target_queue_cold( + db, config, source=manifest['source'], platform=manifest['platform'], + config_sha256=config_sha256, max_rows=max_rows, + ) + if ( + current['selection_sha256'] != manifest['selection_sha256'] + or current['entries'] != manifest['entries'] + ): + raise RuntimeError('target queue policy selection drifted after review') + return db.cold_target_queue_rows( + manifest['entries'], reason_code=manifest['reason_code'], + config_sha256=config_sha256, policy_sha256=policy['policy_sha256'], + manifest_sha256=manifest_sha256, max_rows=max_rows, + ) + + +def apply_cold_target_queue_reactivation( + db, config, manifest, manifest_sha256, *, config_sha256, max_rows=10000, +): + policy = configured_target_queue_query_policy(config) + if manifest['config_sha256'] != config_sha256: + raise RuntimeError('target queue reactivation manifest config identity drifted') + if manifest['policy_sha256'] != policy['policy_sha256']: + raise RuntimeError('target queue reactivation manifest query policy drifted') + if not _manifest_already_applied(db, manifest, manifest_sha256, 'reactivate'): + current = plan_cold_target_queue_reactivation( + db, config, source=manifest['source'], platform=manifest['platform'], + query=manifest['query'], config_sha256=config_sha256, max_rows=max_rows, + ) + if ( + current['selection_sha256'] != manifest['selection_sha256'] + or current['entries'] != manifest['entries'] + ): + raise RuntimeError('target queue reactivation selection drifted after review') + return db.reactivate_cold_target_queue_rows( + manifest['entries'], reason_code=manifest['reason_code'], + config_sha256=config_sha256, policy_sha256=policy['policy_sha256'], + manifest_sha256=manifest_sha256, max_rows=max_rows, + ) + + +def load_config(path): + try: + import yaml + except ImportError as exc: + raise SystemExit('PyYAML is required for runtime safety migration') from exc + with open(path, 'r', encoding='utf-8') as handle: + return apply_path_config(yaml.safe_load(handle) or {}, path) + + +def _supervisor_metadata_paths(config): + global_config = config.get('global') or {} + supervisor = config.get('supervisor') or {} + log_dir = supervisor.get('log_dir') or global_config.get('log_dir') or '' + control_dir = supervisor.get('control_dir') or global_config.get('control_dir') or '' + paths = [ + supervisor.get('instance_file') or os.path.join(control_dir, 'supervisor.instance.json'), + os.path.join(log_dir, 'supervisor.instance.json'), + os.path.join(log_dir, 'supervisor.pid'), + ] + return list(dict.fromkeys(os.path.abspath(path) for path in paths if path)) + + +def require_local_sources_stopped(config, inspect_scan_slots=True): + for path in _supervisor_metadata_paths(config): + if os.path.lexists(path): + try: + metadata = read_private_json(path) + from process_identity import exact_process_identity_state + + state = exact_process_identity_state( + metadata.get('pid'), metadata.get('process_creation_time'), + metadata.get('executable'), + ) + except (OSError, ValueError): + state = 'unknown' + if state not in ('dead', 'reused'): + raise RuntimeError( + f'refusing migration while supervisor metadata identity is {state}: {path}' + ) + global_config = config.get('global') or {} + limiter_path = global_config.get('scan_limiter_db') or os.path.join(global_config.get('state_dir') or '', 'scan_limiter.db') + if inspect_scan_slots and limiter_path and os.path.isfile(limiter_path): + try: + uri = 'file:' + os.path.abspath(limiter_path).replace('\\', '/') + '?mode=ro' + with sqlite3.connect(uri, uri=True, timeout=1) as local_db: + row = local_db.execute("SELECT COUNT(*) FROM scan_slots").fetchone() + if row and int(row[0] or 0) > 0: + raise RuntimeError(f'refusing migration while {row[0]} local scan slot(s) are active') + except sqlite3.OperationalError as exc: + if 'no such table' not in str(exc).lower(): + raise RuntimeError(f'unable to prove local scan slots are stopped: {exc}') from exc + + +def recover_dead_scan_slots(config, identity_live=None, now=None, max_rows=10000): + from scanner import exact_process_identity_live + + identity_live = identity_live or exact_process_identity_live + now = time.time() if now is None else float(now) + max_rows = min(10000, max(1, int(max_rows))) + global_config = config.get('global') or {} + path = global_config.get('scan_limiter_db') or os.path.join( + global_config.get('state_dir') or '', 'scan_limiter.db', + ) + path = os.path.abspath(path) + if not os.path.isfile(path): + return {'path': path, 'scanned': 0, 'deleted': 0, 'remaining': 0, 'rows': []} + require_private_file(path) + connection = sqlite3.connect(path, timeout=30) + connection.row_factory = sqlite3.Row + try: + connection.execute('PRAGMA busy_timeout=30000') + connection.execute('BEGIN IMMEDIATE') + columns = {row[1] for row in connection.execute('PRAGMA table_info(scan_slots)').fetchall()} + required = { + 'slot_id', 'owner_pid', 'owner_thread', 'owner_source', + 'owner_creation_time', 'owner_executable', 'child_pid', + 'child_creation_time', 'child_executable', 'acquired_at', 'updated_at', + } + if not required.issubset(columns): + raise RuntimeError('scan-slot table lacks exact process identity columns') + rows = connection.execute( + '''SELECT slot_id, owner_pid, owner_thread, owner_source, + owner_creation_time, owner_executable, child_pid, + child_creation_time, child_executable, acquired_at, updated_at + FROM scan_slots ORDER BY acquired_at LIMIT ?''', + (max_rows + 1,), + ).fetchall() + if len(rows) > max_rows: + raise RuntimeError(f'scan-slot recovery exceeds its {max_rows} row bound') + report_rows = [] + deleted = 0 + for row in rows: + checks = [] + for prefix in ('owner', 'child'): + try: + live = identity_live( + row[f'{prefix}_pid'], + row[f'{prefix}_creation_time'], + row[f'{prefix}_executable'], + ) + except Exception: + live = None + checks.append(live if live is None else bool(live)) + owner_live, child_live = checks + deleted_row = False + if owner_live is False and child_live is False: + cursor = connection.execute( + '''DELETE FROM scan_slots + WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ? + AND owner_source IS ? AND owner_creation_time IS ? + AND owner_executable IS ? AND child_pid IS ? + AND child_creation_time IS ? AND child_executable IS ? + AND acquired_at = ? AND updated_at = ?''', + ( + row['slot_id'], row['owner_pid'], row['owner_thread'], + row['owner_source'], row['owner_creation_time'], row['owner_executable'], + row['child_pid'], row['child_creation_time'], row['child_executable'], + row['acquired_at'], row['updated_at'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError(f'scan-slot identity changed during recovery: {row["slot_id"]}') + deleted += 1 + deleted_row = True + report_rows.append({ + 'slot_id': row['slot_id'], + 'owner_pid': row['owner_pid'], + 'owner_thread': row['owner_thread'], + 'owner_source': row['owner_source'], + 'owner_creation_time': row['owner_creation_time'], + 'owner_executable': row['owner_executable'], + 'owner_live': owner_live, + 'child_pid': row['child_pid'], + 'child_creation_time': row['child_creation_time'], + 'child_executable': row['child_executable'], + 'child_live': child_live, + 'acquired_at': row['acquired_at'], + 'updated_at': row['updated_at'], + 'age_seconds': max(0.0, now - float(row['updated_at'] or row['acquired_at'])), + 'deleted': deleted_row, + }) + remaining = int(connection.execute('SELECT COUNT(*) FROM scan_slots').fetchone()[0]) + connection.commit() + return { + 'path': path, + 'scanned': len(rows), + 'deleted': deleted, + 'remaining': remaining, + 'rows': report_rows, + } + except BaseException: + connection.rollback() + raise + finally: + connection.close() + harden_private_file(path) + + +def _offline_result_pipeline_state(db): + conn = getattr(db, 'conn', None) + if not conn or not conn.is_postgres: + raise RuntimeError('offline result-pipeline recovery requires PostgreSQL') + return { + 'worker_leases': int(conn.execute( + "SELECT COUNT(*) AS count FROM pipeline_leases WHERE state NOT IN ('released','failed')" + ).fetchone()['count'] or 0), + 'result_reservations': int(conn.execute( + """SELECT COUNT(*) AS count FROM result_reservations + WHERE state IN ('scanning','ready','ingesting','db_committed')""" + ).fetchone()['count'] or 0), + 'queue_leases': int(conn.execute( + """SELECT COUNT(*) AS count FROM target_queue q + LEFT JOIN result_reservations r ON r.id = q.current_result_reservation_id + WHERE q.status = 'in_progress' + OR r.state IN ('scanning','ready','ingesting','db_committed')""" + ).fetchone()['count'] or 0), + 'blob_leases': int(conn.execute( + """SELECT COUNT(*) AS count FROM docker_content_blobs + WHERE state IN ('leased','submitted') OR lease_reservation_id IS NOT NULL""" + ).fetchone()['count'] or 0), + } + + +def _require_offline_result_pipeline_schema(db): + conn = getattr(db, 'conn', None) + if not conn or not conn.is_postgres: + raise RuntimeError('offline result-pipeline recovery requires PostgreSQL') + marker = '20260901_13_docker_layer_content_scanning' + if not conn.table_exists('runtime_schema_migrations') or not conn.execute( + 'SELECT 1 AS present FROM runtime_schema_migrations WHERE version = ?', (marker,), + ).fetchone(): + raise RuntimeError('offline result-pipeline recovery requires the marker-13 schema') + required = { + 'pipeline_leases': {'worker_name', 'generation', 'lease_token', 'state'}, + 'result_reservations': { + 'id', 'state', 'queue_id', 'claim_lease_token', 'scan_event_id', + 'producer_pid', 'producer_creation_time', 'producer_executable', + 'ready_relative_path', 'bundle_id', 'reservation_token', + }, + 'result_bundles': {'reservation_id', 'state', 'relative_path', 'scan_event_hash'}, + 'target_queue': {'id', 'status', 'current_result_reservation_id', 'claim_event_id'}, + 'docker_content_blobs': {'state', 'lease_reservation_id'}, + 'keycheck_candidates': {'candidate_uid', 'credential_id', 'service', 'secret_hash'}, + } + for table, columns in required.items(): + if not conn.table_exists(table) or not columns.issubset(conn.table_columns(table)): + raise RuntimeError(f'offline result-pipeline recovery schema is incomplete: {table}') + conn.commit() + + +def recover_stale_result_pipeline(db, config, max_rows=1000, max_seconds=300.0): + _require_offline_result_pipeline_schema(db) + max_rows = min(10000, max(1, int(max_rows))) + max_seconds = min(3600.0, max(1.0, float(max_seconds))) + initial = _offline_result_pipeline_state(db) + if initial['result_reservations'] > max_rows: + raise RuntimeError('offline result-pipeline recovery exceeds its reservation bound') + + global_config = config.get('global') or {} + worker_config = ((config.get('supervisor') or {}).get('result_ingester') or {}) + lease_seconds = min(3600, max(60, int(worker_config.get('lease_seconds', 300)))) + instance_id = f'offline-result-pipeline-{os.getpid()}' + identity = current_process_identity() + projector_lease = None + worker = None + processed = 0 + recovery_passes = 0 + deadline = time.monotonic() + max_seconds + try: + projector_lease = db.acquire_pipeline_lease( + 'jsonl_projector', instance_id, identity, + lease_seconds=lease_seconds, initial_state='recovering', + ) + if not projector_lease: + raise RuntimeError('jsonl projector advisory lock is held') + worker = ResultIngester( + db, global_config['result_bundle_dir'], instance_id, + lease_seconds=lease_seconds, + quarantine_max_items=int(global_config.get('pipeline_quarantine_max_items', 10000)), + quarantine_max_bytes=int( + global_config.get('pipeline_quarantine_max_bytes', 1024 * 1024 * 1024) + ), + recover_expired_ready=True, + ) + worker.lease = db.acquire_pipeline_lease( + 'result_ingester', instance_id, identity, + lease_seconds=lease_seconds, initial_state='recovering', + ) + if not worker.lease: + raise RuntimeError('result ingester advisory lock is held') + + previous_active = None + while time.monotonic() < deadline: + current = _offline_result_pipeline_state(db)['result_reservations'] + if current == 0: + break + if previous_active is not None and current >= previous_active: + break + previous_active = current + worker.recovery_after_id = 0 + worker.recover( + page_size=min(100, max_rows), + max_pages=max(1, (max_rows + 99) // 100), + ) + recovery_passes += 1 + if not worker.heartbeat('recovering'): + raise RuntimeError('offline result ingester heartbeat fence was lost') + while processed < max_rows and time.monotonic() < deadline: + if not worker.process_one(): + break + processed += 1 + if not worker.heartbeat('recovering'): + raise RuntimeError('offline result ingester heartbeat fence was lost') + + if worker and not worker.stop(): + raise RuntimeError('offline result ingester lease release was not fenced') + worker = None + if not db.release_pipeline_lease( + 'jsonl_projector', projector_lease['generation'], projector_lease['lease_token'], + ): + raise RuntimeError('offline JSONL projector lease release was not fenced') + projector_lease = None + final = _offline_result_pipeline_state(db) + return { + 'initial': initial, + 'processed_bundles': processed, + 'recovery_passes': recovery_passes, + 'final': final, + 'quiescent': not any(final.values()), + } + except BaseException: + if worker is not None: + try: + worker.stop('offline_result_pipeline_recovery_failed') + except Exception: + pass + if projector_lease is not None: + try: + db.release_pipeline_lease( + 'jsonl_projector', projector_lease['generation'], + projector_lease['lease_token'], state='failed', + error='offline_result_pipeline_recovery_failed', + ) + except Exception: + pass + raise + + +def _pid_running(pid): + try: + pid = int(pid) + if pid <= 0 or pid == os.getpid(): + return False + if os.name == 'nt': + from process_identity import open_process + process = open_process(pid) + try: + return process.is_running() + finally: + process.close() + os.kill(pid, 0) + return True + except (OSError, TypeError, ValueError): + return False + + +def known_application_process_markers(config): + global_config = config.get('global') or {} + roots = [global_config.get('work_dir')] + active = [] + for root in roots: + if not root or not os.path.isdir(root) or is_reparse_point(root): + continue + with os.scandir(root) as entries: + for index, entry in enumerate(entries): + if index >= 256: + raise RuntimeError(f'unable to bound known application process markers under {root}') + if entry.is_symlink() or is_reparse_point(entry.path) or not entry.is_dir(follow_symlinks=False): + continue + marker = os.path.join(entry.path, '.scanner-owner.json') + if not os.path.isfile(marker) or not private_file_ready(marker): + continue + try: + with open(marker, 'r', encoding='utf-8') as handle: + value = json.load(handle) + prefix = 'owner' if value.get('owner_pid') else 'parent' + pid = int(value.get(f'{prefix}_pid') or 0) + creation_time = value.get(f'{prefix}_creation_time') + executable = value.get(f'{prefix}_executable') + except (OSError, TypeError, ValueError, json.JSONDecodeError): + continue + from process_identity import exact_process_identity_state + state = exact_process_identity_state(pid, creation_time, executable) + if state == 'alive' or (state == 'unknown' and _pid_running(pid)): + active.append((pid, marker)) + return active + + +def require_runtime_hardening_stopped(config): + require_local_sources_stopped(config, inspect_scan_slots=False) + paths = postgres_runtime_paths(config) + postmaster_pid = os.path.join(paths['data_dir'], 'postmaster.pid') + if os.path.lexists(postmaster_pid): + raise RuntimeError(f'refusing hardening while postmaster.pid exists: {postmaster_pid}') + port = configured_cluster_values()['port'] + config_path = str((config.get('global') or {}).get('project_dir') or '') + env_path = find_postgres_environment_path(os.path.join(config_path, 'config.yaml'), config) if config_path else None + if env_path: + try: + with open(env_path, 'r', encoding='utf-8') as handle: + for line in handle: + key, separator, value = line.strip().partition('=') + if separator and key.strip() == 'TRUF_POSTGRES_PORT': + port = int(value.strip().strip('"').strip("'")) + if not 0 < port <= 65535: + raise ValueError('port outside valid range') + break + except (OSError, ValueError) as exc: + raise RuntimeError(f'unable to determine the offline PostgreSQL listener port: {exc}') from exc + if _listener_present(port): + raise RuntimeError(f'refusing hardening while a listener is present on 127.0.0.1:{port}') + active = known_application_process_markers(config) + if active: + pid, marker = active[0] + raise RuntimeError(f'refusing hardening while known application process {pid} is active: {marker}') + + +def find_postgres_environment_path(config_path, config): + global_config = (config or {}).get('global') or {} + candidates = [] + if global_config.get('root_dir'): + candidates.append(os.path.join(global_config['root_dir'], '.env.postgres')) + candidates.extend(( + os.path.join(os.path.dirname(config_path), '..', '.env.postgres'), + os.path.join(os.path.dirname(config_path), '.env.postgres'), + )) + return next((os.path.abspath(path) for path in candidates if os.path.isfile(path)), None) + + +def _online_postgres_identity(db, dsn, config, identity=None): + identity = identity or verify_cluster_identity(config) + canonical = canonical_postgres_url(dsn, identity['database'], identity['user'], identity['port']) + if canonical != dsn: + db.url = canonical + row = db.conn.execute( + '''SELECT pg_catalog.current_database() AS database, CURRENT_USER AS user_name, + pg_catalog.current_setting('data_directory') AS data_directory, + pg_catalog.current_setting('port')::integer AS port, + (SELECT system_identifier::text FROM pg_catalog.pg_control_system()) AS system_identifier''' + ).fetchone() + checks = { + 'database': (str(row['database']), identity['database']), + 'user': (str(row['user_name']), identity['user']), + 'data_directory': ( + os.path.normcase(os.path.realpath(os.path.abspath(row['data_directory']))), + os.path.normcase(os.path.realpath(os.path.abspath(identity['data_directory']))), + ), + 'port': (int(row['port']), int(identity['port'])), + 'system_identifier': (str(row['system_identifier']), str(identity['system_identifier'])), + } + for name, (actual, expected) in checks.items(): + if actual != expected: + raise RuntimeError(f'online migration target {name} does not match cluster_identity.json') + db.conn.commit() + return identity, canonical + + +def _require_no_postgres_application_sessions(db): + rows = db.conn.execute( + '''SELECT pid, application_name, state + FROM pg_catalog.pg_stat_activity + WHERE datname = pg_catalog.current_database() AND pid <> pg_catalog.pg_backend_pid() + AND backend_type = 'client backend' LIMIT 20''' + ).fetchall() + db.conn.commit() + if rows: + names = ', '.join(str(row['application_name'] or '')[:80] for row in rows[:5]) + raise RuntimeError(f'refusing migration while {len(rows)} other database application session(s) are active: {names}') + + +@contextlib.contextmanager +def postgres_migration_guard(db): + db.conn.execute(f'SET search_path = {POSTGRES_APPLICATION_SCHEMA}') + schema_row = db.conn.execute( + """SELECT pg_catalog.current_schema() AS schema_name, + pg_catalog.current_setting('search_path') AS search_path, + EXISTS ( + SELECT 1 + FROM pg_catalog.pg_namespace n + CROSS JOIN LATERAL pg_catalog.aclexplode( + COALESCE(n.nspacl, pg_catalog.acldefault('n', n.nspowner)) + ) acl + WHERE n.nspname = 'public' + AND acl.grantee = 0 + AND acl.privilege_type = 'CREATE' + ) AS public_create""" + ).fetchone() + normalized_path = ''.join(str(schema_row['search_path'] or '').lower().replace('"', '').split()) if schema_row else '' + if ( + not schema_row + or schema_row['schema_name'] != POSTGRES_APPLICATION_SCHEMA + or normalized_path != 'public' + or bool(schema_row.get('public_create', False)) + ): + raise RuntimeError('migration PostgreSQL schema/search_path validation failed') + db.conn.execute("SET application_name = 'truf-offline-migration'") + db.conn.execute("SET statement_timeout = 0") + db.conn.execute("SET lock_timeout = '10s'") + db.conn.execute("SET idle_in_transaction_session_timeout = 0") + db.conn.commit() + _require_no_postgres_application_sessions(db) + row = db.conn.execute('SELECT pg_catalog.pg_try_advisory_lock(781273968142991337) AS locked').fetchone() + db.conn.commit() + if not row or not row['locked']: + raise RuntimeError('another PostgreSQL runtime-safety migration holds the advisory lock') + try: + _require_no_postgres_application_sessions(db) + yield + finally: + try: + db.conn.execute('SELECT pg_catalog.pg_advisory_unlock(781273968142991337)') + db.conn.commit() + except Exception: + try: + db.conn.rollback() + except Exception: + pass + + +def harden_runtime_paths(config, env_path=None, config_path=None): + global_config = config.get('global') or {} + supervisor_config = config.get('supervisor') or {} + bundle_dir = global_config.get('result_bundle_dir') + if bundle_dir and os.name == 'nt' and os.path.splitdrive(os.path.abspath(bundle_dir))[0].upper() != 'S:': + raise RuntimeError('production result_bundle_dir must be on S:') + directories = [] + for key in ( + 'root_dir', 'project_dir', 'runtime_dir', 'results_dir', 'result_spool_dir', + 'result_bundle_dir', + 'control_dir', 'queue_dir', 'state_dir', 'keycheck_dir', 'postman_cache_dir', + 'gharchive_cache_dir', 'log_dir', 'work_dir', + ): + if global_config.get(key): + directories.append(global_config[key]) + for key in ('control_dir', 'log_dir', 'state_dir'): + if supervisor_config.get(key): + directories.append(supervisor_config[key]) + if bundle_dir: + directories.extend( + os.path.join(bundle_dir, name) for name in ('tmp', 'ready', 'quarantine') + ) + postgres_paths = postgres_runtime_paths(config) + postgres_paths['data_dir'] = canonical_cluster_data_directory(config) + directories.append(postgres_paths['postgres_dir']) + try: + data_inside_postgres_dir = ( + os.path.commonpath((postgres_paths['postgres_dir'], postgres_paths['data_dir'])) + == postgres_paths['postgres_dir'] + ) + except ValueError: + data_inside_postgres_dir = False + if not data_inside_postgres_dir: + directories.append(postgres_paths['data_dir']) + + normalized_directories = sorted( + {os.path.normcase(os.path.abspath(path)) for path in directories if path}, + key=lambda path: (path.count(os.sep), path), + ) + files = [ + config_path, + env_path, + global_config.get('secrets_file'), + global_config.get('proxy_file'), + global_config.get('api_proxy_file'), + global_config.get('download_proxy_file'), + global_config.get('trufflehog_config'), + ] + from lifecycle_authority import ( + GIT_MANIFEST_NAME, + TRUFFLEHOG_MANIFEST_NAME, + manifest_authority_paths, + resolve_manifest_executable, + ) + + authority_files = manifest_authority_paths( + global_config.get('project_dir'), + global_config.get('trufflehog_path'), + policy_paths=[global_config.get('trufflehog_config')], + existing_only=True, + include_executables=os.name == 'nt', + ) + native_files = [] + if os.name != 'nt': + for name, value in ( + (TRUFFLEHOG_MANIFEST_NAME, global_config.get('trufflehog_path')), + (GIT_MANIFEST_NAME, None), + ): + native_files.append(require_trusted_native_executable(resolve_manifest_executable( + value, name=name, app_dir=global_config.get('project_dir'), + ))) + for key in ('postgres', 'pg_ctl', 'pg_isready', 'pg_controldata', 'initdb', 'psql'): + native_files.append(require_trusted_native_executable(postgres_paths[key])) + files.extend(authority_files) + required_authority_files = { + os.path.normcase(os.path.abspath(path)) for path in authority_files + } + config_dir = os.path.dirname(os.path.abspath(config_path)) if config_path else None + for parent in ( + global_config.get('root_dir'), + global_config.get('project_dir'), + config_dir, + os.path.dirname(config_dir) if config_dir else None, + ): + if parent: + candidate = os.path.join(parent, '.env.postgres') + if candidate not in files: + files.append(candidate) + # Private trees may not encompass immutable native code, including its + # parent. Reject conflicting configuration rather than chmod system paths. + private_parents = normalized_directories + [ + os.path.dirname(os.path.abspath(path)) for path in files if path + ] + for native in native_files: + parent = os.path.dirname(native) + for directory in private_parents: + if os.path.commonpath((directory, parent)) in (directory, parent): + raise PrivateFileError(f'private runtime authority overlaps a native executable directory: {parent}') + hardened = 0 + for directory in normalized_directories: + reject_reparse_components(os.path.dirname(directory)) + os.makedirs(directory, mode=0o700, exist_ok=True) + reject_reparse_components(directory) + harden_private_directory(directory) + hardened += 1 + for path in files: + if not path: + continue + absolute = os.path.abspath(path) + if not os.path.lexists(absolute): + if os.path.normcase(absolute) in required_authority_files: + raise PrivateFileError(f'required code authority file is absent: {absolute}') + if config_path and os.path.normcase(absolute) == os.path.normcase(os.path.abspath(config_path)): + raise PrivateFileError(f'required config file is absent: {absolute}') + continue + reject_reparse_components(absolute) + if not os.path.isfile(absolute): + raise PrivateFileError(f'sensitive configured file is not regular: {absolute}') + harden_private_directory(os.path.dirname(absolute)) + harden_private_file(absolute) + hardened += 1 + + # Preserve recursive hardening of runtime-owned content after every + # configured parent has been created and hardened in a deterministic order. + tree_roots = [] + for key in ('runtime_dir', 'results_dir', 'result_spool_dir', 'result_bundle_dir', 'queue_dir', 'state_dir', 'keycheck_dir', 'postman_cache_dir', 'log_dir', 'work_dir'): + if global_config.get(key): + tree_roots.append(global_config[key]) + tree_roots.append(postgres_paths['postgres_dir']) + if not data_inside_postgres_dir: + tree_roots.append(postgres_paths['data_dir']) + seen_trees = set() + for path in tree_roots: + absolute = os.path.normcase(os.path.abspath(path)) + if absolute in seen_trees: + continue + seen_trees.add(absolute) + hardened += harden_private_tree(absolute) + + for directory in normalized_directories: + if not private_directory_ready(directory): + raise PrivateFileError(f'private directory verification failed after hardening: {directory}') + for path in files: + if path and os.path.isfile(path) and not private_file_ready(path): + raise PrivateFileError(f'private file verification failed after hardening: {path}') + for path in authority_files: + if not private_file_ready(path): + raise PrivateFileError(f'required code authority verification failed after hardening: {path}') + for path in native_files: + require_trusted_native_executable(path) + return hardened + + +def load_jsonl_issue_review_manifest(path): + path = os.path.abspath(path) + require_private_file(path) + if os.path.getsize(path) > 16 * 1024 * 1024: + raise RuntimeError('JSONL issue review manifest exceeds its byte bound') + with open(path, 'r', encoding='utf-8') as handle: + value = json.load(handle) + if not isinstance(value, dict) or set(value) != {'schema', 'type', 'audit_sha256', 'issues'}: + raise RuntimeError('JSONL issue review manifest root is invalid') + if value.get('schema') != 1 or value.get('type') != 'truf-jsonl-projection-review': + raise RuntimeError('JSONL issue review manifest schema is unsupported') + if not re.fullmatch(r'[a-f0-9]{64}', str(value.get('audit_sha256') or '')): + raise RuntimeError('JSONL issue review manifest audit identity is invalid') + issues = value.get('issues') + if not isinstance(issues, list) or not issues or len(issues) > 100000: + raise RuntimeError('JSONL issue review manifest issue count is outside its bound') + allowed_classifications = { + 'invalid_json', 'legacy_numeric_prefix_corrupt_json', 'invalid_utf8', + 'utf8_bom_prefix', 'unterminated_record', 'invalid_error_projection', + 'oversized_record', + } + normalized = [] + seen = set() + expected_keys = { + 'ledger_kind', 'file', 'offset', 'length', 'sha256', + 'classification', 'safe_identity', + } + for issue in issues: + if not isinstance(issue, dict) or set(issue) != expected_keys: + raise RuntimeError('JSONL issue review manifest entry is invalid') + ledger_kind = str(issue.get('ledger_kind') or '') + file_name = str(issue.get('file') or '') + offset = int(issue.get('offset', -1)) + length = int(issue.get('length', 0)) + digest = str(issue.get('sha256') or '').lower() + classification = str(issue.get('classification') or '') + if ( + ledger_kind not in ('scan_results', 'found_secrets', 'scan_errors') + or not file_name or os.path.basename(file_name) != file_name + or offset < 0 or length <= 0 + or not re.fullmatch(r'[a-f0-9]{64}', digest) + or classification not in allowed_classifications + or issue.get('safe_identity') is not False + ): + raise RuntimeError('JSONL issue review manifest metadata is invalid') + identity = (ledger_kind, file_name, offset, digest) + if identity in seen: + raise RuntimeError('JSONL issue review manifest contains duplicate metadata') + seen.add(identity) + normalized.append({ + 'ledger_kind': ledger_kind, + 'file': file_name, + 'offset': offset, + 'length': length, + 'sha256': digest, + 'classification': classification, + }) + return normalized + + +def load_jsonl_conflict_review_manifest(path): + path = os.path.abspath(path) + require_private_file(path) + if os.path.getsize(path) > 16 * 1024 * 1024: + raise RuntimeError('JSONL conflict review manifest exceeds its byte bound') + with open(path, 'r', encoding='utf-8') as handle: + value = json.load(handle) + if not isinstance(value, dict) or set(value) != {'schema', 'type', 'audit_sha256', 'variants'}: + raise RuntimeError('JSONL conflict review manifest root is invalid') + if value.get('schema') != 1 or value.get('type') != 'truf-jsonl-projection-conflict-review': + raise RuntimeError('JSONL conflict review manifest schema is unsupported') + if not re.fullmatch(r'[a-f0-9]{64}', str(value.get('audit_sha256') or '')): + raise RuntimeError('JSONL conflict review manifest audit identity is invalid') + variants = value.get('variants') + if not isinstance(variants, list) or not variants or len(variants) > 100000: + raise RuntimeError('JSONL conflict review manifest variant count is outside its bound') + expected_keys = { + 'ledger_kind', 'file', 'offset', 'length', 'identity_sha256', + 'payload_sha256', 'field_name_set_sha256', 'classification', 'occurrences', + } + normalized = [] + seen = set() + for variant in variants: + if not isinstance(variant, dict) or set(variant) != expected_keys: + raise RuntimeError('JSONL conflict review manifest entry is invalid') + ledger_kind = str(variant.get('ledger_kind') or '') + file_name = str(variant.get('file') or '') + offset = int(variant.get('offset', -1)) + length = int(variant.get('length', 0)) + identity_sha256 = str(variant.get('identity_sha256') or '').lower() + payload_sha256 = str(variant.get('payload_sha256') or '').lower() + field_name_set_sha256 = str(variant.get('field_name_set_sha256') or '').lower() + occurrences = int(variant.get('occurrences', 0)) + if ( + ledger_kind not in ('scan_results', 'found_secrets', 'scan_errors') + or not file_name or os.path.basename(file_name) != file_name + or offset < 0 or length <= 0 or occurrences <= 0 + or not re.fullmatch(r'[a-f0-9]{64}', identity_sha256) + or not re.fullmatch(r'[a-f0-9]{64}', payload_sha256) + or not re.fullmatch(r'[a-f0-9]{64}', field_name_set_sha256) + or variant.get('classification') != 'historical_payload_variant' + ): + raise RuntimeError('JSONL conflict review manifest metadata is invalid') + identity = (ledger_kind, identity_sha256, payload_sha256) + if identity in seen: + raise RuntimeError('JSONL conflict review manifest contains duplicate variants') + seen.add(identity) + normalized.append({ + 'ledger_kind': ledger_kind, + 'file': file_name, + 'offset': offset, + 'length': length, + 'identity_sha256': identity_sha256, + 'payload_sha256': payload_sha256, + 'field_name_set_sha256': field_name_set_sha256, + 'classification': 'historical_payload_variant', + 'occurrences': occurrences, + }) + return normalized + + +def reconcile_jsonl_projection_ledgers(config, args): + from scanner import ( + approve_projection_conflict_variants, + approve_projection_reconciliation_issues, + reconcile_projection_ledger_batch, + ) + + global_config = config.get('global') or {} + results_dir = global_config.get('results_dir') + if not results_dir: + raise RuntimeError('configured results_dir is required for JSONL projection reconciliation') + specs = ( + ('scan_results.jsonl', 'scan_event_id'), + ('found_secrets.jsonl', 'finding_uid'), + ('scan_errors.log', 'error_row_id'), + ) + reviewed = {name: [] for name, _ in specs} + for value in args.resolve_jsonl_issue: + parts = str(value or '').rsplit(':', 2) + if len(parts) != 3: + raise SystemExit('--resolve-jsonl-issue requires FILE:OFFSET:SHA256') + physical_file, raw_offset, digest = parts + selected = None + for name, identity_key in specs: + stem, extension = os.path.splitext(name) + if physical_file == name or ( + physical_file.startswith(stem + '.') and physical_file.endswith(extension) + ): + selected = (name, identity_key) + break + if selected is None: + raise SystemExit(f'unsupported JSONL projection issue file: {physical_file}') + name, identity_key = selected + reviewed[name].append({ + 'file': physical_file, + 'offset': int(raw_offset), + 'sha256': digest, + }) + if args.resolve_jsonl_issue_manifest: + for issue in load_jsonl_issue_review_manifest(args.resolve_jsonl_issue_manifest): + reviewed[issue['ledger_kind'] + ('.log' if issue['ledger_kind'] == 'scan_errors' else '.jsonl')].append(issue) + for name, identity_key in specs: + if not reviewed[name]: + continue + report = approve_projection_reconciliation_issues( + os.path.join(results_dir, name), + identity_key, + reviewed[name], + max_record_bytes=int(global_config.get('result_spool_max_event_bytes', 192 * 1024 * 1024) or 192 * 1024 * 1024), + ) + print(json.dumps({ + 'resolved_issue_review': {'ledger': name, **report}, + }, ensure_ascii=True, sort_keys=True), flush=True) + if args.resolve_jsonl_conflict_manifest: + conflict_reviews = {name: [] for name, _ in specs} + for variant in load_jsonl_conflict_review_manifest(args.resolve_jsonl_conflict_manifest): + logical_name = variant['ledger_kind'] + ( + '.log' if variant['ledger_kind'] == 'scan_errors' else '.jsonl' + ) + conflict_reviews[logical_name].append(variant) + for name, identity_key in specs: + if not conflict_reviews[name]: + continue + report = approve_projection_conflict_variants( + os.path.join(results_dir, name), + identity_key, + conflict_reviews[name], + max_record_bytes=int(global_config.get('result_spool_max_event_bytes', 192 * 1024 * 1024) or 192 * 1024 * 1024), + progress_callback=lambda value, ledger_name=name: print(json.dumps({ + 'conflict_review_progress': {'ledger': ledger_name, **value}, + }, ensure_ascii=True, sort_keys=True), flush=True), + ) + print(json.dumps({ + 'resolved_conflict_review': {'ledger': name, **report}, + }, ensure_ascii=True, sort_keys=True), flush=True) + all_complete = True + for name, identity_key in specs: + path = os.path.join(results_dir, name) + report = None + batch_limit = max(1, int(args.max_batches)) if args.until_complete else 1 + for _ in range(batch_limit): + report = reconcile_projection_ledger_batch( + path, + identity_key, + max_rows=args.max_rows, + max_bytes=args.max_bytes, + max_seconds=args.max_seconds, + row_limit=int(global_config.get('jsonl_ledger_max_rows', 1000000) or 1000000), + ledger_byte_limit=int(global_config.get('jsonl_ledger_max_bytes', 512 * 1024 * 1024) or 512 * 1024 * 1024), + max_record_bytes=int(global_config.get('result_spool_max_event_bytes', 192 * 1024 * 1024) or 192 * 1024 * 1024), + ) + print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True) + if report.get('complete'): + break + if not report or not report.get('complete'): + all_complete = False + return 0 if all_complete else 2 + + +def _metadata_is_docker_command_timeout(metadata_json): + try: + metadata = json.loads(metadata_json or '{}') + except (TypeError, ValueError): + return False + if not isinstance(metadata, dict): + return False + scan_meta = metadata.get('scan_meta') + return bool( + isinstance(scan_meta, dict) and scan_meta.get('command_timed_out') + or metadata.get('error_class') == 'timeout' + ) + + +def repair_docker_timeout_attempts(db, max_attempts=3, max_rows=1000): + max_attempts = max(1, int(max_attempts or 3)) + max_rows = max(1, int(max_rows or 1000)) + now = utc_now_iso() + report = {'examined': 0, 'repaired': 0, 'terminal': 0, 'unchanged': 0} + candidates = db.conn.execute( + '''SELECT id, attempts + FROM target_queue + WHERE source = 'dockerhub' AND platform = 'docker' + AND status = 'deferred' + AND normalized_target LIKE '%@sha256:%' + AND lease_owner IS NULL AND lease_token IS NULL + AND current_result_reservation_id IS NULL AND claim_event_id IS NULL + AND resolver_token IS NULL + ORDER BY id + LIMIT ?''', + (max_rows,), + ).fetchall() + try: + for candidate in candidates: + report['examined'] += 1 + rows = db.conn.execute( + '''SELECT s.id, c.metadata_json + FROM target_scans s + LEFT JOIN scan_result_compat c ON c.target_scan_id = s.id + WHERE s.queue_id = ? + ORDER BY s.id''', + (candidate['id'],), + ).fetchall() + if not rows or not _metadata_is_docker_command_timeout(rows[-1]['metadata_json']): + report['unchanged'] += 1 + continue + timeout_attempts = sum( + 1 for row in rows if _metadata_is_docker_command_timeout(row['metadata_json']) + ) + repaired_attempts = min( + max_attempts, + max(int(candidate['attempts'] or 0), timeout_attempts), + ) + if repaired_attempts == int(candidate['attempts'] or 0): + report['unchanged'] += 1 + continue + terminal = repaired_attempts >= max_attempts + cursor = db.conn.execute( + '''UPDATE target_queue SET + attempts = ?, + status = CASE WHEN ? != 0 THEN 'failed' ELSE status END, + available_after = CASE WHEN ? != 0 THEN NULL ELSE available_after END, + completed_at = CASE WHEN ? != 0 THEN COALESCE(completed_at, ?) ELSE completed_at END, + last_error = CASE WHEN ? != 0 + THEN 'Docker timeout attempts exhausted by guarded repair' + ELSE last_error END, + updated_at = ? + WHERE id = ? AND source = 'dockerhub' AND platform = 'docker' + AND status = 'deferred' + AND lease_owner IS NULL AND lease_token IS NULL + AND current_result_reservation_id IS NULL AND claim_event_id IS NULL + AND resolver_token IS NULL''', + ( + repaired_attempts, + 1 if terminal else 0, + 1 if terminal else 0, + 1 if terminal else 0, + now, + 1 if terminal else 0, + now, + candidate['id'], + ), + ) + if cursor.rowcount != 1: + raise RuntimeError('Docker timeout repair lost its queue fence') + report['repaired'] += 1 + report['terminal'] += 1 if terminal else 0 + db.conn.commit() + return report + except Exception: + db.conn.rollback() + raise + + +def run_target_queue_policy_action(config_path, config, args): + cold_action = bool(args.cold_stale_backlog) + reactivate_action = bool(args.reactivate_cold_backlog) + if cold_action == reactivate_action: + raise SystemExit('Select exactly one target queue policy action') + if bool(args.dry_run) == bool(args.apply): + raise SystemExit('Target queue policy action requires exactly one of --dry-run or --apply') + if not args.sources_stopped: + raise SystemExit('Target queue policy action requires --sources-stopped') + if not args.policy_manifest: + raise SystemExit('Target queue policy action requires --policy-manifest') + policy_source = str(args.policy_source or '').strip() + policy_platform = str(args.policy_platform or '').strip() + policy_query = str(args.policy_query or '') + if not policy_source or not policy_platform: + raise SystemExit('Target queue policy action requires --policy-source and --policy-platform') + if cold_action and policy_query: + raise SystemExit('--policy-query is only valid for cold-backlog reactivation') + if reactivate_action and not policy_query: + raise SystemExit('Cold-backlog reactivation requires --policy-query') + if any(( + args.sqlite, args.initialize_base, args.todo, args.source, args.platform, + args.resolve_issue, args.harden_runtime, args.reconcile_jsonl_projections, + args.resolve_jsonl_issue, args.resolve_jsonl_issue_manifest, + args.resolve_jsonl_conflict_manifest, args.recover_dead_scan_slots, + args.recover_stale_result_pipeline, args.repair_docker_timeout_attempts, + args.drain_legacy_outbox, args.import_legacy_outbox_to_projection, + args.backfill_normalized_results, args.import_legacy_spool, + args.review_pipeline_quarantine, args.rebuild_jsonl_output, + args.until_complete, + )): + raise SystemExit('Target queue policy action cannot be combined with another migration action') + + global_config = config.get('global') or {} + load_postgres_environment(config_path, config) + requested_db_url = ( + args.database_url or database_url_from_env() or global_config.get('database_url') + ) + if not is_postgres_url(requested_db_url): + raise SystemExit('A canonical PostgreSQL DSN is required for target queue policy review') + preflight_lifecycle_paths(config_path, config, authority_profile='server') + manifest_path = os.path.abspath(args.policy_manifest) + with ClusterAuthorityLock(config, endpoint_dsn=requested_db_url): + require_local_sources_stopped(config) + identity = verify_cluster_identity(config) + db_url = canonical_postgres_url( + requested_db_url, identity['database'], identity['user'], identity['port'], + ) + db = ScannerDB(db_url=db_url, initialize=False) + try: + if not db.enabled: + raise RuntimeError('target queue policy PostgreSQL connection is unavailable') + _online_postgres_identity(db, db_url, config, identity=identity) + db.set_application_name('truf-offline-target-queue-policy') + config_sha256 = sha256_file(config_path) + with postgres_migration_guard(db): + db.require_runtime_safety_schema() + if args.dry_run: + if cold_action: + manifest = plan_stale_target_queue_cold( + db, config, source=policy_source, platform=policy_platform, + config_sha256=config_sha256, max_rows=args.max_rows, + ) + action = 'cold' + else: + manifest = plan_cold_target_queue_reactivation( + db, config, source=policy_source, platform=policy_platform, + query=policy_query, config_sha256=config_sha256, + max_rows=args.max_rows, + ) + action = 'reactivate' + atomic_write_private_json( + manifest_path, manifest, + max_bytes=TARGET_QUEUE_POLICY_MANIFEST_MAX_BYTES, + ) + require_private_file(manifest_path) + report = { + 'action': action, + 'planned': len(manifest['entries']), + 'config_sha256': manifest['config_sha256'], + 'policy_sha256': manifest['policy_sha256'], + 'selection_sha256': manifest['selection_sha256'], + 'manifest_sha256': _canonical_json_sha256(manifest), + } + else: + expected_type = ( + 'truf-target-queue-cold-review-v1' + if cold_action + else 'truf-target-queue-reactivation-review-v1' + ) + manifest, manifest_sha256 = load_target_queue_policy_manifest( + manifest_path, expected_type, max_rows=args.max_rows, + ) + if ( + manifest['source'] != policy_source + or manifest['platform'] != policy_platform + or ( + reactivate_action + and manifest['query'] != policy_query + ) + ): + raise RuntimeError('target queue policy manifest scope does not match CLI scope') + if cold_action: + report = apply_stale_target_queue_cold( + db, config, manifest, manifest_sha256, + config_sha256=config_sha256, max_rows=args.max_rows, + ) + else: + report = apply_cold_target_queue_reactivation( + db, config, manifest, manifest_sha256, + config_sha256=config_sha256, max_rows=args.max_rows, + ) + report = { + 'action': 'cold' if cold_action else 'reactivate', + **report, + } + print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True) + return 0 + finally: + db.close() + + +def parse_args(): + parser = argparse.ArgumentParser(description='Offline/idempotent runtime safety schema migration and bounded todo reconciliation.') + parser.add_argument('--config', default=os.path.join(os.path.dirname(__file__), 'config.yaml')) + parser.add_argument('--database-url') + parser.add_argument('--sqlite') + parser.add_argument('--initialize-base', action='store_true', help='Create the fresh-install base schema before applying additive migration') + parser.add_argument('--todo', help='Optional todo file to reconcile; checked files are intentionally unsupported') + parser.add_argument('--source') + parser.add_argument('--platform') + parser.add_argument('--query', default='offline-reconciliation') + parser.add_argument('--max-rows', type=int, default=1000) + parser.add_argument('--max-bytes', type=int, default=4 * 1024 * 1024) + parser.add_argument('--max-seconds', type=float, default=5.0) + parser.add_argument('--until-complete', action='store_true', help='Run bounded batches until EOF or --max-batches') + parser.add_argument('--max-batches', type=int, default=100) + parser.add_argument('--resolve-issue', action='append', type=int, default=[], help='Mark a reviewed reconciliation issue resolved') + parser.add_argument('--harden-runtime', action='store_true', help='Offline recursive hardening/verification of sensitive runtime trees') + parser.add_argument('--reconcile-jsonl-projections', action='store_true', help='Offline bounded/resumable publication-ledger initialization') + parser.add_argument('--resolve-jsonl-issue', action='append', default=[], help='Approve exact reviewed FILE:OFFSET:SHA256 corrupt record metadata') + parser.add_argument('--resolve-jsonl-issue-manifest', help='Approve a private exact reviewed issue manifest') + parser.add_argument('--resolve-jsonl-conflict-manifest', help='Approve a private exact historical payload-variant manifest') + parser.add_argument('--recover-dead-scan-slots', action='store_true', help='Explicitly remove only exact-identity dead local scan slots') + parser.add_argument('--recover-stale-result-pipeline', action='store_true', help='Explicitly reconcile fenced stale result reservations and singleton worker leases') + parser.add_argument('--repair-docker-timeout-attempts', action='store_true', help='Repair only unfenced Docker targets whose latest durable result is a command timeout') + parser.add_argument('--drain-legacy-outbox', action='store_true', help='Explicit bounded offline projection of legacy scan_publication_outbox rows') + parser.add_argument('--import-legacy-outbox-to-projection', action='store_true', help='Transfer bounded legacy outbox references into PostgreSQL projection jobs') + parser.add_argument('--backfill-normalized-results', action='store_true', help='Boundedly convert legacy raw PostgreSQL result blobs to normalized-v2 compatibility rows') + parser.add_argument('--import-legacy-spool', action='store_true', help='Import a bounded batch from the retired durable result spool into PostgreSQL') + parser.add_argument('--review-pipeline-quarantine', help='Apply an exact private audited quarantine review manifest') + parser.add_argument('--rebuild-jsonl-output', help='Build a full resumable PostgreSQL-derived compatibility projection in this dedicated directory') + parser.add_argument('--cold-stale-backlog', action='store_true', help='Plan or apply exact policy-stale target queue cold transitions') + parser.add_argument('--reactivate-cold-backlog', action='store_true', help='Plan or apply exact reviewed cold target queue reactivation') + parser.add_argument('--policy-manifest', help='Private exact target queue policy review manifest path') + parser.add_argument('--policy-source', help='Exact source scope for target queue policy review') + parser.add_argument('--policy-platform', help='Exact platform scope for target queue policy review') + parser.add_argument('--policy-query', help='Exact query scope for cold target queue reactivation') + parser.add_argument('--dry-run', action='store_true') + parser.add_argument('--apply', action='store_true') + parser.add_argument('--sources-stopped', action='store_true') + return parser.parse_args() + + +def main(): + args = parse_args() + for name in ( + 'drain_legacy_outbox', 'import_legacy_outbox_to_projection', + 'backfill_normalized_results', 'import_legacy_spool', + 'review_pipeline_quarantine', + 'rebuild_jsonl_output', + 'recover_stale_result_pipeline', + ): + if not hasattr(args, name): + setattr(args, name, False) + config_path = os.path.abspath(args.config) + config = load_config(config_path) + if args.cold_stale_backlog or args.reactivate_cold_backlog: + return run_target_queue_policy_action(config_path, config, args) + if args.dry_run: + raise SystemExit('--dry-run is only supported for a target queue policy action') + if not args.apply or not args.sources_stopped: + raise SystemExit('Refusing migration without both --apply and --sources-stopped') + global_config = config.get('global', {}) + if args.recover_stale_result_pipeline: + if any(( + args.sqlite, args.initialize_base, args.todo, args.resolve_issue, + args.harden_runtime, args.reconcile_jsonl_projections, + args.resolve_jsonl_issue, args.resolve_jsonl_issue_manifest, + args.resolve_jsonl_conflict_manifest, args.recover_dead_scan_slots, + args.repair_docker_timeout_attempts, args.drain_legacy_outbox, + args.import_legacy_outbox_to_projection, args.backfill_normalized_results, + args.import_legacy_spool, args.review_pipeline_quarantine, + args.rebuild_jsonl_output, args.until_complete, + )): + raise SystemExit('--recover-stale-result-pipeline is a separate PostgreSQL recovery action') + load_postgres_environment(config_path, config) + requested_db_url = args.database_url or database_url_from_env() or global_config.get('database_url') + if not is_postgres_url(requested_db_url): + raise SystemExit('A canonical PostgreSQL DSN is required for result-pipeline recovery') + preflight_lifecycle_paths(config_path, config, authority_profile='server') + with ClusterAuthorityLock(config, endpoint_dsn=requested_db_url): + require_local_sources_stopped(config) + identity = verify_cluster_identity(config) + db_url = canonical_postgres_url( + requested_db_url, identity['database'], identity['user'], identity['port'], + ) + db = ScannerDB(db_url=db_url, initialize=False) + try: + if not db.enabled: + raise RuntimeError('result-pipeline recovery PostgreSQL connection is unavailable') + _online_postgres_identity(db, db_url, config, identity=identity) + db.set_application_name('truf-offline-result-recovery') + with postgres_migration_guard(db): + report = recover_stale_result_pipeline( + db, config, max_rows=args.max_rows, max_seconds=args.max_seconds, + ) + print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True) + return 0 if report['quiescent'] else 2 + except BaseException as exc: + raise SystemExit( + f'offline result-pipeline recovery failed: {type(exc).__name__}' + ) from None + finally: + db.close() + if args.rebuild_jsonl_output: + if any(( + args.sqlite, args.initialize_base, args.todo, args.resolve_issue, + args.harden_runtime, args.reconcile_jsonl_projections, + args.resolve_jsonl_issue, args.resolve_jsonl_issue_manifest, + args.resolve_jsonl_conflict_manifest, args.recover_dead_scan_slots, + args.drain_legacy_outbox, args.import_legacy_outbox_to_projection, + args.backfill_normalized_results, args.import_legacy_spool, + args.review_pipeline_quarantine, + )): + raise SystemExit('--rebuild-jsonl-output is a separate bounded compatibility action') + load_postgres_environment(config_path, config) + requested_db_url = args.database_url or database_url_from_env() or global_config.get('database_url') + if not is_postgres_url(requested_db_url): + raise SystemExit('A canonical PostgreSQL DSN is required for full JSONL rebuild') + preflight_lifecycle_paths(config_path, config, authority_profile='server') + with ClusterAuthorityLock(config, endpoint_dsn=requested_db_url): + require_local_sources_stopped(config) + identity = verify_cluster_identity(config) + db_url = canonical_postgres_url( + requested_db_url, identity['database'], identity['user'], identity['port'], + ) + db = ScannerDB(db_url=db_url, initialize=False) + try: + if not db.enabled: + raise RuntimeError('full JSONL rebuild PostgreSQL connection is unavailable') + _online_postgres_identity(db, db_url, config, identity=identity) + with postgres_migration_guard(db): + report = rebuild_jsonl_projections( + db, args.rebuild_jsonl_output, + args.max_rows, args.max_bytes, args.max_seconds, + ) + print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True) + return 0 if report['completed'] else 2 + finally: + db.close() + if args.review_pipeline_quarantine: + if any(( + args.sqlite, args.initialize_base, args.todo, args.resolve_issue, + args.harden_runtime, args.reconcile_jsonl_projections, + args.resolve_jsonl_issue, args.resolve_jsonl_issue_manifest, + args.resolve_jsonl_conflict_manifest, args.recover_dead_scan_slots, + args.drain_legacy_outbox, args.import_legacy_outbox_to_projection, + args.backfill_normalized_results, args.import_legacy_spool, + )): + raise SystemExit('--review-pipeline-quarantine is a separate audited action') + load_postgres_environment(config_path, config) + requested_db_url = args.database_url or database_url_from_env() or global_config.get('database_url') + if not is_postgres_url(requested_db_url): + raise SystemExit('A canonical PostgreSQL DSN is required for quarantine review') + preflight_lifecycle_paths(config_path, config, authority_profile='server') + with ClusterAuthorityLock(config, endpoint_dsn=requested_db_url): + require_local_sources_stopped(config) + identity = verify_cluster_identity(config) + db_url = canonical_postgres_url( + requested_db_url, identity['database'], identity['user'], identity['port'], + ) + db = ScannerDB(db_url=db_url, initialize=False) + try: + if not db.enabled: + raise RuntimeError('quarantine review PostgreSQL connection is unavailable') + _online_postgres_identity(db, db_url, config, identity=identity) + with postgres_migration_guard(db): + db.require_runtime_safety_schema() + report = review_pipeline_quarantine_manifest( + db, os.path.abspath(args.review_pipeline_quarantine), args.max_rows, + config, + ) + print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True) + return 0 + finally: + db.close() + if args.import_legacy_spool: + if any(( + args.sqlite, args.initialize_base, args.todo, args.resolve_issue, + args.harden_runtime, args.reconcile_jsonl_projections, + args.resolve_jsonl_issue, args.resolve_jsonl_issue_manifest, + args.resolve_jsonl_conflict_manifest, args.recover_dead_scan_slots, + args.drain_legacy_outbox, args.import_legacy_outbox_to_projection, + args.backfill_normalized_results, + )): + raise SystemExit('--import-legacy-spool is a separate bounded compatibility action') + load_postgres_environment(config_path, config) + requested_db_url = args.database_url or database_url_from_env() or global_config.get('database_url') + if not is_postgres_url(requested_db_url): + raise SystemExit('A canonical PostgreSQL DSN is required for legacy spool import') + preflight_lifecycle_paths(config_path, config, authority_profile='server') + with ClusterAuthorityLock(config, endpoint_dsn=requested_db_url): + require_local_sources_stopped(config) + identity = verify_cluster_identity(config) + db_url = canonical_postgres_url( + requested_db_url, identity['database'], identity['user'], identity['port'], + ) + db = ScannerDB(db_url=db_url, initialize=False) + try: + if not db.enabled: + raise RuntimeError('legacy spool import PostgreSQL connection is unavailable') + _online_postgres_identity(db, db_url, config, identity=identity) + with postgres_migration_guard(db): + db.require_runtime_safety_schema() + db.revoke_final_cutover() + report = import_legacy_result_spool(db, config, args.max_rows) + print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True) + return 0 if report['remaining'] == 0 else 2 + finally: + db.close() + if args.backfill_normalized_results: + if any(( + args.sqlite, args.initialize_base, args.todo, args.resolve_issue, + args.harden_runtime, args.reconcile_jsonl_projections, + args.resolve_jsonl_issue, args.resolve_jsonl_issue_manifest, + args.resolve_jsonl_conflict_manifest, args.recover_dead_scan_slots, + args.drain_legacy_outbox, args.import_legacy_outbox_to_projection, + )): + raise SystemExit('--backfill-normalized-results is a separate bounded compatibility action') + load_postgres_environment(config_path, config) + requested_db_url = args.database_url or database_url_from_env() or global_config.get('database_url') + if not is_postgres_url(requested_db_url): + raise SystemExit('A canonical PostgreSQL DSN is required for normalized result backfill') + preflight_lifecycle_paths(config_path, config, authority_profile='server') + with ClusterAuthorityLock(config, endpoint_dsn=requested_db_url): + require_local_sources_stopped(config) + identity = verify_cluster_identity(config) + db_url = canonical_postgres_url( + requested_db_url, identity['database'], identity['user'], identity['port'], + ) + db = ScannerDB(db_url=db_url, initialize=False) + try: + if not db.enabled: + raise RuntimeError('normalized result backfill PostgreSQL connection is unavailable') + _online_postgres_identity(db, db_url, config, identity=identity) + with postgres_migration_guard(db): + db.require_runtime_safety_schema() + db.revoke_final_cutover() + batch_limit = max(1, int(args.max_batches)) if args.until_complete else 1 + report = None + for batch_number in range(1, batch_limit + 1): + report = backfill_normalized_raw_results( + db, args.max_rows, args.max_bytes, args.max_seconds, + ) + print(json.dumps( + {'batch': batch_number, **report}, + ensure_ascii=True, sort_keys=True, + ), flush=True) + if report['remaining'] == 0 or report['processed'] == 0: + break + return 0 if report['remaining'] == 0 else 2 + finally: + db.close() + if args.import_legacy_outbox_to_projection: + if any(( + args.sqlite, args.initialize_base, args.todo, args.resolve_issue, + args.harden_runtime, args.reconcile_jsonl_projections, + args.resolve_jsonl_issue, args.resolve_jsonl_issue_manifest, + args.resolve_jsonl_conflict_manifest, args.recover_dead_scan_slots, + args.drain_legacy_outbox, + )): + raise SystemExit('--import-legacy-outbox-to-projection is a separate bounded compatibility action') + load_postgres_environment(config_path, config) + requested_db_url = args.database_url or database_url_from_env() or global_config.get('database_url') + if not is_postgres_url(requested_db_url): + raise SystemExit('A canonical PostgreSQL DSN is required for legacy outbox transfer') + preflight_lifecycle_paths(config_path, config, authority_profile='server') + with ClusterAuthorityLock(config, endpoint_dsn=requested_db_url): + require_local_sources_stopped(config) + identity = verify_cluster_identity(config) + db_url = canonical_postgres_url( + requested_db_url, identity['database'], identity['user'], identity['port'], + ) + db = ScannerDB(db_url=db_url, initialize=False) + try: + if not db.enabled: + raise RuntimeError('legacy outbox transfer PostgreSQL connection is unavailable') + _online_postgres_identity(db, db_url, config, identity=identity) + with postgres_migration_guard(db): + db.require_runtime_safety_schema() + db.revoke_final_cutover() + report = import_legacy_outbox_to_projection(db, args.max_rows) + print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True) + return 0 if report['remaining'] == 0 else 2 + finally: + db.close() + if args.drain_legacy_outbox: + if any(( + args.sqlite, args.initialize_base, args.todo, args.resolve_issue, + args.harden_runtime, args.reconcile_jsonl_projections, + args.resolve_jsonl_issue, args.resolve_jsonl_issue_manifest, + args.resolve_jsonl_conflict_manifest, args.recover_dead_scan_slots, + )): + raise SystemExit('--drain-legacy-outbox is a separate bounded compatibility action') + load_postgres_environment(config_path, config) + requested_db_url = args.database_url or database_url_from_env() or global_config.get('database_url') + if not is_postgres_url(requested_db_url): + raise SystemExit('A canonical PostgreSQL DSN is required for legacy outbox drain') + preflight_lifecycle_paths(config_path, config, authority_profile='server') + with ClusterAuthorityLock(config, endpoint_dsn=requested_db_url): + require_local_sources_stopped(config) + identity = verify_cluster_identity(config) + db_url = canonical_postgres_url( + requested_db_url, identity['database'], identity['user'], identity['port'], + ) + db = ScannerDB(db_url=db_url, initialize=False) + try: + if not db.enabled: + raise RuntimeError('legacy outbox PostgreSQL connection is unavailable') + _online_postgres_identity(db, db_url, config, identity=identity) + with postgres_migration_guard(db): + db.require_runtime_safety_schema() + db.revoke_final_cutover() + report = drain_legacy_scan_outbox(db, config, args.max_rows) + print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True) + return 0 if report['remaining'] == 0 else 2 + finally: + db.close() + if args.recover_dead_scan_slots: + if any(( + args.database_url, args.sqlite, args.initialize_base, args.todo, + args.resolve_issue, args.harden_runtime, args.reconcile_jsonl_projections, + args.resolve_jsonl_issue, args.resolve_jsonl_issue_manifest, + args.resolve_jsonl_conflict_manifest, args.until_complete, + )): + raise SystemExit('--recover-dead-scan-slots is a separate local recovery action') + load_postgres_environment(config_path, config) + requested_db_url = global_config.get('database_url') or database_url_from_env() + if not is_postgres_url(requested_db_url): + raise SystemExit('A caller-selected canonical PostgreSQL DSN is required for maintenance authority') + preflight_lifecycle_paths(config_path, config, authority_profile='server') + with ClusterAuthorityLock(config, create_parent=True, endpoint_dsn=requested_db_url): + require_local_sources_stopped(config, inspect_scan_slots=False) + report = recover_dead_scan_slots(config) + print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True) + return 0 if report['remaining'] == 0 else 2 + if ( + args.resolve_jsonl_issue + or args.resolve_jsonl_issue_manifest + or args.resolve_jsonl_conflict_manifest + ) and not args.reconcile_jsonl_projections: + raise SystemExit('JSONL review requires --reconcile-jsonl-projections') + if args.harden_runtime and args.reconcile_jsonl_projections: + raise SystemExit('--harden-runtime and --reconcile-jsonl-projections are separate actions') + if args.harden_runtime: + if any((args.database_url, args.sqlite, args.initialize_base, args.todo, args.resolve_issue, args.until_complete)): + raise SystemExit('--harden-runtime is a separate filesystem-only action and cannot be combined with migration/reconciliation options') + load_postgres_environment(config_path, config) + requested_db_url = global_config.get('database_url') or database_url_from_env() + if not is_postgres_url(requested_db_url): + raise SystemExit('A caller-selected canonical PostgreSQL DSN is required for maintenance authority') + with ClusterAuthorityLock(config, create_parent=True, endpoint_dsn=requested_db_url): + require_runtime_hardening_stopped(config) + hardened = harden_runtime_paths( + config, + find_postgres_environment_path(config_path, config), + config_path=config_path, + ) + print(f'Runtime hardening verification: OK ({hardened} entries)') + return 0 + + if args.reconcile_jsonl_projections: + if any((args.database_url, args.sqlite, args.initialize_base, args.todo, args.resolve_issue)): + raise SystemExit('--reconcile-jsonl-projections cannot be combined with database migration/reconciliation options') + load_postgres_environment(config_path, config) + requested_db_url = global_config.get('database_url') or database_url_from_env() + if not is_postgres_url(requested_db_url): + raise SystemExit('A caller-selected canonical PostgreSQL DSN is required for maintenance authority') + with ClusterAuthorityLock(config, create_parent=True, endpoint_dsn=requested_db_url): + require_runtime_hardening_stopped(config) + return reconcile_jsonl_projection_ledgers(config, args) + + load_postgres_environment(config_path, config) + requested_db_url = ( + args.database_url + or database_url_from_env() + or global_config.get('database_url') + ) + if not args.sqlite and not is_postgres_url(requested_db_url): + raise SystemExit('A PostgreSQL DSN is required unless --sqlite is explicit; refusing scanner_active.db fallback') + if args.sqlite and not requested_db_url: + raise SystemExit('A caller-selected canonical PostgreSQL DSN is required for maintenance authority') + preflight_lifecycle_paths(config_path, config, authority_profile='server') + with ClusterAuthorityLock( + config, + endpoint_dsn=requested_db_url, + ): + db_url = requested_db_url or database_url_from_env() + if not args.sqlite and not is_postgres_url(db_url): + raise SystemExit('A PostgreSQL DSN is required unless --sqlite is explicit; refusing scanner_active.db fallback') + if args.sqlite and args.database_url: + raise SystemExit('--sqlite and --database-url are mutually exclusive') + if args.todo and (not args.source or not args.platform): + raise SystemExit('--todo requires --source and --platform') + if args.until_complete and not args.todo: + raise SystemExit('--until-complete requires --todo') + + require_local_sources_stopped(config) + db = None + exit_code = 0 + try: + identity = None + if not args.sqlite: + try: + identity = verify_cluster_identity(config) + db_url = canonical_postgres_url( + db_url, identity['database'], identity['user'], identity['port'], + ) + except (DatabaseUrlError, OSError, ValueError) as exc: + raise RuntimeError(f'PostgreSQL migration authority validation failed: {exc}') from exc + db = ScannerDB(db_path=args.sqlite, db_url='' if args.sqlite else db_url, initialize=False) + if not db.enabled: + raise RuntimeError(f'Unable to open migration database: {db.db_display}') + guard = contextlib.nullcontext() + if db.conn.is_postgres: + _online_postgres_identity(db, db_url, config, identity=identity) + guard = postgres_migration_guard(db) + with guard: + require_local_sources_stopped(config) + if db.conn.is_postgres and db.conn.table_exists('runtime_final_cutover'): + db.revoke_final_cutover() + migrate_runtime_safety_schema(db, initialize_base=args.initialize_base) + cursor_report = initialize_projection_cursors_from_existing_files(db, config) + legacy_evidence = require_legacy_cutover_clear(db, config) + db.require_runtime_safety_schema() + if db.conn.is_postgres: + migrations = db.conn.execute( + 'SELECT version, code_sha256 FROM runtime_schema_migrations ORDER BY version' + ).fetchall() + db.conn.commit() + marker = db.record_final_cutover({ + 'legacy': legacy_evidence, + 'projection_cursors': cursor_report, + 'schema_migrations': [dict(row) for row in migrations], + }) + db.require_final_cutover() + print( + f'Final PostgreSQL cutover marker: {marker["evidence_sha256"]}', + flush=True, + ) + print('Runtime safety schema migration: OK') + + if args.repair_docker_timeout_attempts: + report = repair_docker_timeout_attempts( + db, + max_attempts=int(global_config.get('target_retry_max_attempts', 3) or 3), + max_rows=args.max_rows, + ) + print('Docker timeout attempt repair: ' + json.dumps(report, sort_keys=True)) + + if args.resolve_issue: + resolve_reconciliation_issues(db, args.resolve_issue) + + if args.todo: + batch_limit = max(1, int(args.max_batches)) if args.until_complete else 1 + report = None + for _ in range(batch_limit): + previous_offset = report.get('byte_offset') if report else None + report = reconcile_todo_file( + db, args.todo, args.source, args.platform, args.query, + args.max_rows, args.max_bytes, args.max_seconds, + ) + print(json.dumps(report, indent=2, sort_keys=True)) + if report['at_eof']: + break + if previous_offset is not None and report['byte_offset'] <= previous_offset: + raise RuntimeError('bounded reconciliation made no forward progress') + if not report or not report['at_eof'] or int(report.get('unresolved_issues') or 0) != 0: + exit_code = 2 + finally: + if db is not None: + db.close() + return exit_code + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/app/optimize_dashboard_db.py b/app/optimize_dashboard_db.py new file mode 100644 index 0000000..b2d6d9a --- /dev/null +++ b/app/optimize_dashboard_db.py @@ -0,0 +1,21 @@ +"""Retired direct database mutation entrypoint. + +Dashboard indexes are part of the authority-locked offline runtime-safety +migration. Keeping a second live DDL path would bypass cluster identity and +stopped-source checks. +""" + +import sys + +sys.dont_write_bytecode = True + + +def main(): + raise SystemExit( + 'optimize_dashboard_db is retired; run migrate_runtime_safety.py ' + 'offline under cluster authority' + ) + + +if __name__ == '__main__': + main() diff --git a/app/owned_process.py b/app/owned_process.py new file mode 100644 index 0000000..ca2d734 --- /dev/null +++ b/app/owned_process.py @@ -0,0 +1,1346 @@ +import sys + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError("owned process host could not disable bytecode writes") + +import ctypes +import errno +import json +import math +import os +import queue +import signal +import struct +import subprocess +import threading +import time + + +# Keep this module standard-library-only: it is re-executed as the isolated +# containment host before any application code is allowed to run. +_HOST_FLAG = "--owned-process-host" +_MAX_PACKET_SIZE = 16 * 1024 * 1024 +_CREATE_BREAKAWAY_FROM_JOB = 0x01000000 +_TEST_FAIL_JOB_ASSIGNMENT = "OWNED_PROCESS_TEST_FAIL_JOB_ASSIGNMENT" +_WINDOWS_INHERIT_LOCK = threading.Lock() +_JOB_OBJECT_LIMIT_JOB_MEMORY = 0x00000200 +_JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE = 0x00002000 +_JOB_OBJECT_CPU_RATE_CONTROL_INFORMATION = 15 +_JOB_OBJECT_CPU_RATE_CONTROL_ENABLE = 0x1 +_JOB_OBJECT_CPU_RATE_CONTROL_WEIGHT_BASED = 0x2 +_PROCESS_MEMORY_PRIORITY = 0 +_MAX_JOB_PROCESS_IDS = 4096 +_MAX_REAP_BATCH = 64 +_HOST_PRIVATE_ENV_KEYS = frozenset({ + "SCANNER_SUPERVISED", + "TRUF_MANAGED_POSTGRES_DSN", + "SCANNER_DB_URL", + "DATABASE_URL", + "SCANNER_DASHBOARD_DB_URL", + "KEYCHECK_DB_URL", + "PGPASSWORD", + "PGUSER", + "PGDATABASE", + "PGHOST", + "PGHOSTADDR", + "PGPORT", + "PGSERVICE", + "PGSERVICEFILE", + "PGPASSFILE", + "PGOPTIONS", + "PGSSLMODE", + "PGSSLKEY", + "PGSSLCERT", + "PGSSLROOTCERT", +}) + + +if os.name == "nt": + from ctypes import wintypes + + class _IO_COUNTERS(ctypes.Structure): + _fields_ = [ + ("ReadOperationCount", ctypes.c_ulonglong), + ("WriteOperationCount", ctypes.c_ulonglong), + ("OtherOperationCount", ctypes.c_ulonglong), + ("ReadTransferCount", ctypes.c_ulonglong), + ("WriteTransferCount", ctypes.c_ulonglong), + ("OtherTransferCount", ctypes.c_ulonglong), + ] + + class _JOBOBJECT_BASIC_LIMIT_INFORMATION(ctypes.Structure): + _fields_ = [ + ("PerProcessUserTimeLimit", ctypes.c_longlong), + ("PerJobUserTimeLimit", ctypes.c_longlong), + ("LimitFlags", wintypes.DWORD), + ("MinimumWorkingSetSize", ctypes.c_size_t), + ("MaximumWorkingSetSize", ctypes.c_size_t), + ("ActiveProcessLimit", wintypes.DWORD), + ("Affinity", ctypes.c_size_t), + ("PriorityClass", wintypes.DWORD), + ("SchedulingClass", wintypes.DWORD), + ] + + class _JOBOBJECT_EXTENDED_LIMIT_INFORMATION(ctypes.Structure): + _fields_ = [ + ("BasicLimitInformation", _JOBOBJECT_BASIC_LIMIT_INFORMATION), + ("IoInfo", _IO_COUNTERS), + ("ProcessMemoryLimit", ctypes.c_size_t), + ("JobMemoryLimit", ctypes.c_size_t), + ("PeakProcessMemoryUsed", ctypes.c_size_t), + ("PeakJobMemoryUsed", ctypes.c_size_t), + ] + + class _JOBOBJECT_BASIC_ACCOUNTING_INFORMATION(ctypes.Structure): + _fields_ = [ + ("TotalUserTime", ctypes.c_longlong), + ("TotalKernelTime", ctypes.c_longlong), + ("ThisPeriodTotalUserTime", ctypes.c_longlong), + ("ThisPeriodTotalKernelTime", ctypes.c_longlong), + ("TotalPageFaultCount", wintypes.DWORD), + ("TotalProcesses", wintypes.DWORD), + ("ActiveProcesses", wintypes.DWORD), + ("TotalTerminatedProcesses", wintypes.DWORD), + ] + + class _JOBOBJECT_BASIC_PROCESS_ID_LIST(ctypes.Structure): + _fields_ = [ + ("NumberOfAssignedProcesses", wintypes.DWORD), + ("NumberOfProcessIdsInList", wintypes.DWORD), + ("ProcessIdList", ctypes.c_size_t * _MAX_JOB_PROCESS_IDS), + ] + + class _JOBOBJECT_CPU_RATE_CONTROL_INFORMATION(ctypes.Structure): + _fields_ = [ + ("ControlFlags", wintypes.DWORD), + ("Weight", wintypes.DWORD), + ] + + class _MEMORY_PRIORITY_INFORMATION(ctypes.Structure): + _fields_ = [("MemoryPriority", wintypes.ULONG)] + + class _FILETIME(ctypes.Structure): + _fields_ = [ + ("dwLowDateTime", wintypes.DWORD), + ("dwHighDateTime", wintypes.DWORD), + ] + + _P_BOOL = ctypes.POINTER(wintypes.BOOL) + _P_DWORD = ctypes.POINTER(wintypes.DWORD) + _P_FILETIME = ctypes.POINTER(_FILETIME) + + _KERNEL32 = ctypes.WinDLL("kernel32", use_last_error=True) + _CREATE_JOB_OBJECT = _KERNEL32.CreateJobObjectW + _CREATE_JOB_OBJECT.argtypes = [ctypes.c_void_p, wintypes.LPCWSTR] + _CREATE_JOB_OBJECT.restype = wintypes.HANDLE + _SET_INFORMATION_JOB_OBJECT = _KERNEL32.SetInformationJobObject + _SET_INFORMATION_JOB_OBJECT.argtypes = [ + wintypes.HANDLE, ctypes.c_int, ctypes.c_void_p, wintypes.DWORD, + ] + _SET_INFORMATION_JOB_OBJECT.restype = wintypes.BOOL + _QUERY_INFORMATION_JOB_OBJECT = _KERNEL32.QueryInformationJobObject + _QUERY_INFORMATION_JOB_OBJECT.argtypes = [ + wintypes.HANDLE, ctypes.c_int, ctypes.c_void_p, wintypes.DWORD, _P_DWORD, + ] + _QUERY_INFORMATION_JOB_OBJECT.restype = wintypes.BOOL + _ASSIGN_PROCESS_TO_JOB_OBJECT = _KERNEL32.AssignProcessToJobObject + _ASSIGN_PROCESS_TO_JOB_OBJECT.argtypes = [wintypes.HANDLE, wintypes.HANDLE] + _ASSIGN_PROCESS_TO_JOB_OBJECT.restype = wintypes.BOOL + _IS_PROCESS_IN_JOB = _KERNEL32.IsProcessInJob + _IS_PROCESS_IN_JOB.argtypes = [wintypes.HANDLE, wintypes.HANDLE, _P_BOOL] + _IS_PROCESS_IN_JOB.restype = wintypes.BOOL + _GET_CURRENT_PROCESS = _KERNEL32.GetCurrentProcess + _GET_CURRENT_PROCESS.argtypes = [] + _GET_CURRENT_PROCESS.restype = wintypes.HANDLE + _SET_HANDLE_INFORMATION = _KERNEL32.SetHandleInformation + _SET_HANDLE_INFORMATION.argtypes = [wintypes.HANDLE, wintypes.DWORD, wintypes.DWORD] + _SET_HANDLE_INFORMATION.restype = wintypes.BOOL + _GET_HANDLE_INFORMATION = _KERNEL32.GetHandleInformation + _GET_HANDLE_INFORMATION.argtypes = [wintypes.HANDLE, _P_DWORD] + _GET_HANDLE_INFORMATION.restype = wintypes.BOOL + _CLOSE_HANDLE = _KERNEL32.CloseHandle + _CLOSE_HANDLE.argtypes = [wintypes.HANDLE] + _CLOSE_HANDLE.restype = wintypes.BOOL + _TERMINATE_JOB_OBJECT = _KERNEL32.TerminateJobObject + _TERMINATE_JOB_OBJECT.argtypes = [wintypes.HANDLE, wintypes.UINT] + _TERMINATE_JOB_OBJECT.restype = wintypes.BOOL + _EXIT_PROCESS = _KERNEL32.ExitProcess + _EXIT_PROCESS.argtypes = [wintypes.UINT] + _EXIT_PROCESS.restype = None + _GET_PROCESS_TIMES = _KERNEL32.GetProcessTimes + _GET_PROCESS_TIMES.argtypes = [ + wintypes.HANDLE, _P_FILETIME, _P_FILETIME, _P_FILETIME, _P_FILETIME, + ] + _GET_PROCESS_TIMES.restype = wintypes.BOOL + _QUERY_FULL_PROCESS_IMAGE_NAME = _KERNEL32.QueryFullProcessImageNameW + _QUERY_FULL_PROCESS_IMAGE_NAME.argtypes = [ + wintypes.HANDLE, wintypes.DWORD, wintypes.LPWSTR, _P_DWORD, + ] + _QUERY_FULL_PROCESS_IMAGE_NAME.restype = wintypes.BOOL + _SET_PROCESS_INFORMATION = _KERNEL32.SetProcessInformation + _SET_PROCESS_INFORMATION.argtypes = [ + wintypes.HANDLE, ctypes.c_int, ctypes.c_void_p, wintypes.DWORD, + ] + _SET_PROCESS_INFORMATION.restype = wintypes.BOOL + _GET_PROCESS_INFORMATION = _KERNEL32.GetProcessInformation + _GET_PROCESS_INFORMATION.argtypes = [ + wintypes.HANDLE, ctypes.c_int, ctypes.c_void_p, wintypes.DWORD, + ] + _GET_PROCESS_INFORMATION.restype = wintypes.BOOL +else: + _IO_COUNTERS = _JOBOBJECT_BASIC_LIMIT_INFORMATION = None + _JOBOBJECT_EXTENDED_LIMIT_INFORMATION = None + _JOBOBJECT_BASIC_ACCOUNTING_INFORMATION = None + _JOBOBJECT_BASIC_PROCESS_ID_LIST = _FILETIME = None + _JOBOBJECT_CPU_RATE_CONTROL_INFORMATION = _MEMORY_PRIORITY_INFORMATION = None + _KERNEL32 = None + + +class OwnedProcessError(OSError): + def __init__(self, stage, message, error_number=None, windows_error=None): + super().__init__(error_number or 0, f"owned process {stage} failed: {message}") + self.stage = stage + self.winerror = windows_error + + +def _write_all(fd, data): + view = memoryview(data) + while view: + written = os.write(fd, view) + if written <= 0: + raise BrokenPipeError("control pipe closed") + view = view[written:] + + +def _read_exact(fd, size): + chunks = [] + remaining = size + while remaining: + chunk = os.read(fd, remaining) + if not chunk: + raise EOFError("pipe closed") + chunks.append(chunk) + remaining -= len(chunk) + return b"".join(chunks) + + +def _write_packet(fd, value): + data = json.dumps(value, ensure_ascii=True, separators=(",", ":")).encode("utf-8") + if len(data) > _MAX_PACKET_SIZE: + raise ValueError("owned process control packet is too large") + _write_all(fd, struct.pack("!I", len(data)) + data) + + +def _read_packet(fd): + size = struct.unpack("!I", _read_exact(fd, 4))[0] + if size > _MAX_PACKET_SIZE: + raise ValueError("owned process control packet is too large") + return json.loads(_read_exact(fd, size).decode("utf-8")) + + +def _serializable_command(args): + if isinstance(args, (str, os.PathLike)): + return os.fspath(args) + if isinstance(args, bytes): + raise TypeError("OwnedProcess does not accept bytes command arguments") + values = [] + for value in args: + if isinstance(value, bytes): + raise TypeError("OwnedProcess does not accept bytes command arguments") + values.append(os.fspath(value) if isinstance(value, os.PathLike) else str(value)) + return values + + +def _serializable_environment(env): + source = os.environ if env is None else env + result = {} + for key, value in source.items(): + if isinstance(key, bytes) or isinstance(value, bytes): + raise TypeError("OwnedProcess does not accept bytes environment entries") + result[str(key)] = str(value) + result.pop(_TEST_FAIL_JOB_ASSIGNMENT, None) + return result + + +def _validate_job_memory_limit_bytes(value): + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError("job_memory_limit_bytes must be a non-negative integer") + if value < 0: + raise ValueError("job_memory_limit_bytes must be a non-negative integer") + if value > (1 << (ctypes.sizeof(ctypes.c_size_t) * 8)) - 1: + raise ValueError("job_memory_limit_bytes exceeds the host address size") + return value + + +def _validate_job_cpu_weight(value): + if isinstance(value, bool) or not isinstance(value, int) or value < 0 or value > 9: + raise ValueError("job_cpu_weight must be 0 or an integer from 1 through 9") + return value + + +def _validate_process_memory_priority(value): + if isinstance(value, bool) or not isinstance(value, int) or value < 0 or value > 5: + raise ValueError("process_memory_priority must be 0 or an integer from 1 through 5") + return value + + +def _host_environment(): + """The containment host never needs application authority credentials.""" + env = os.environ.copy() + for key in list(env): + normalized = str(key).upper() + if ( + normalized.startswith("TRUF_SUPERVISOR_") + or normalized.startswith("TRUF_POSTGRES_") + or normalized in _HOST_PRIVATE_ENV_KEYS + ): + env.pop(key, None) + return env + + +def _require_isolated_host(): + if not ( + sys.flags.isolated + and sys.flags.no_site + and sys.flags.dont_write_bytecode + ): + raise RuntimeError("owned process host requires -I -S -B") + + +def _status_reader(fd_owner, result_queue): + try: + fd = fd_owner.get_nowait() + except queue.Empty: + return + try: + result_queue.put((True, _read_packet(fd))) + except BaseException as exc: + result_queue.put((False, exc)) + finally: + try: + os.close(fd) + except OSError: + pass + + +def _validated_identity(value, pid): + prefix = "windows-filetime:" if os.name == "nt" else "proc-start-ticks:" + if not isinstance(value, dict) or type(pid) is not int or pid <= 0: + raise OwnedProcessError("identity", "missing process identity") + creation = value.get("creation_time") + timestamp = value.get("creation_time_unix") + executable = value.get("executable") + try: + valid_timestamp = type(timestamp) in (int, float) and math.isfinite(timestamp) and timestamp > 0 + except OverflowError: + valid_timestamp = False + if ( + type(value.get("pid")) is not int or value["pid"] != pid + or not isinstance(creation, str) or not creation.startswith(prefix) + or not creation[len(prefix):].isascii() or not creation[len(prefix):].isdigit() + or not valid_timestamp + or not isinstance(executable, str) or not os.path.isabs(executable) or "\0" in executable + ): + raise OwnedProcessError("identity", "incomplete or mismatched process identity") + return dict(value) + + +class OwnedProcess: + """Popen-like process whose payload tree is owned by a host containment boundary.""" + + def __init__( + self, + args, + bufsize=-1, + executable=None, + stdin=None, + stdout=None, + stderr=None, + preexec_fn=None, + close_fds=True, + shell=False, + cwd=None, + env=None, + universal_newlines=None, + startupinfo=None, + creationflags=0, + restore_signals=True, + start_new_session=False, + pass_fds=(), + *, + user=None, + group=None, + extra_groups=None, + encoding=None, + errors=None, + text=None, + umask=-1, + pipesize=-1, + process_group=None, + job_memory_limit_bytes=0, + job_cpu_weight=0, + process_memory_priority=0, + _startup_timeout=15, + ): + if preexec_fn is not None: + raise ValueError("OwnedProcess does not support preexec_fn") + if not close_fds: + raise ValueError("OwnedProcess requires close_fds=True") + if startupinfo is not None: + raise ValueError("OwnedProcess does not support caller startupinfo") + if start_new_session: + raise ValueError("OwnedProcess owns the payload session") + if pass_fds: + raise ValueError("OwnedProcess does not pass arbitrary file descriptors") + if user is not None or group is not None or extra_groups is not None or umask != -1: + raise ValueError("OwnedProcess does not support credential or umask changes") + if process_group not in (None, -1): + raise ValueError("OwnedProcess owns the payload process group") + if int(creationflags or 0) & _CREATE_BREAKAWAY_FROM_JOB: + raise ValueError("CREATE_BREAKAWAY_FROM_JOB is forbidden for owned processes") + job_memory_limit_bytes = _validate_job_memory_limit_bytes(job_memory_limit_bytes) + job_cpu_weight = _validate_job_cpu_weight(job_cpu_weight) + process_memory_priority = _validate_process_memory_priority(process_memory_priority) + + self.args = args + self._owner_pid = os.getpid() + self._host_process = None + self._control_fd = None + self._control_lock = threading.Lock() + self._payload_pid = None + self._payload_identity = None + self._host_identity = None + self._job_membership_verified = False + self._job_accounting = {} + self._job_members = [] + self._status_fd_owner = queue.SimpleQueue() + self._configuration_sent = False + + config = { + "args": _serializable_command(args), + "bufsize": int(bufsize), + "executable": os.fspath(executable) if executable is not None else None, + "shell": bool(shell), + "cwd": os.fspath(cwd) if cwd is not None else None, + "env": _serializable_environment(env), + "creationflags": int(creationflags or 0), + "restore_signals": bool(restore_signals), + "job_memory_limit_bytes": job_memory_limit_bytes, + "job_cpu_weight": job_cpu_weight, + "process_memory_priority": process_memory_priority, + } + + # Windows venv executables are redirecting launchers. Retain the actual + # stdlib-only observer, not a launcher with a different process identity. + host_python = sys._base_executable if os.name == "nt" else sys.executable + control_read, control_write = os.pipe() + status_read, status_write = os.pipe() + self._status_fd_owner.put(status_read) + self._control_fd = control_write + host_command = [ + host_python, + "-I", + "-S", + "-B", + os.path.abspath(__file__), + _HOST_FLAG, + ] + popen_options = { + "stdin": stdin, + "stdout": stdout, + "stderr": stderr, + "bufsize": bufsize, + "universal_newlines": universal_newlines, + "encoding": encoding, + "errors": errors, + "text": text, + "close_fds": True, + "pipesize": pipesize, + "env": _host_environment(), + } + + try: + if os.name == "nt": + import msvcrt + + control_handle = msvcrt.get_osfhandle(control_read) + status_handle = msvcrt.get_osfhandle(status_write) + host_command.extend([str(control_handle), str(status_handle)]) + startup = subprocess.STARTUPINFO() + startup.lpAttributeList = {"handle_list": [control_handle, status_handle]} + popen_options["startupinfo"] = startup + popen_options["creationflags"] = self._host_creationflags(config["creationflags"]) + with _WINDOWS_INHERIT_LOCK: + os.set_inheritable(control_read, True) + os.set_inheritable(status_write, True) + try: + self._host_process = subprocess.Popen(host_command, **popen_options) + finally: + os.set_inheritable(control_read, False) + os.set_inheritable(status_write, False) + else: + host_command.extend([str(control_read), str(status_write)]) + popen_options["pass_fds"] = (control_read, status_write) + # An inner observer must survive the outer payload-group stop. + popen_options["start_new_session"] = True + self._host_process = subprocess.Popen(host_command, **popen_options) + except BaseException: + self._close_fd(control_read) + self._close_fd(control_write) + self._cancel_status_reader() + self._close_fd(status_write) + self._control_fd = None + raise + + self._close_fd(control_read) + self._close_fd(status_write) + try: + _write_packet(self._control_fd, config) + self._configuration_sent = True + status = self._wait_for_startup(float(_startup_timeout)) + if not isinstance(status, dict) or status.get("ok") is not True: + detail = status if isinstance(status, dict) else {} + raise OwnedProcessError( + str(detail.get("stage") or "startup"), + str(detail.get("message") or "host exited without a valid handshake"), + detail.get("errno"), + detail.get("winerror"), + ) + if status.get("job_membership_verified") is not True: + raise OwnedProcessError("job_membership", "host did not prove payload containment") + self._payload_pid = status.get("pid") + if self._payload_pid == self._host_process.pid: + raise OwnedProcessError("identity", "payload cannot be its own observer") + self._payload_identity = _validated_identity(status.get("payload_identity"), self._payload_pid) + self._host_identity = _validated_identity(status.get("host_identity"), self._host_process.pid) + members = status.get("job_members") + if not isinstance(members, list) or len(members) > _MAX_JOB_PROCESS_IDS: + raise OwnedProcessError("identity", "invalid containment member list") + self._job_members = [ + _validated_identity(item, item.get("pid") if isinstance(item, dict) else None) + for item in members + ] + if self._payload_identity not in self._job_members or self._host_identity not in self._job_members: + raise OwnedProcessError("identity", "retained identities missing from containment members") + accounting = status.get("job_accounting") + if not isinstance(accounting, dict) or ( + os.name != "nt" and accounting.get("child_subreaper_verified") is not True + ): + raise OwnedProcessError("job_membership", "missing containment accounting or subreaper proof") + self._job_accounting = dict(accounting) + self._job_membership_verified = True + except BaseException: + # An unclaimed read end can keep the host blocked on a full pipe. + self._cancel_status_reader() + self._request_stop() + self._reap_failed_start() + raise + finally: + self._cancel_status_reader() + + @staticmethod + def _close_fd(fd): + if fd is None: + return + try: + os.close(fd) + except OSError: + pass + + @staticmethod + def _host_creationflags(payload_flags): + # Keep host visibility and scheduling aligned with the payload without + # copying payload-only process-group flags. + allowed = ( + getattr(subprocess, "CREATE_NO_WINDOW", 0x08000000) + | getattr(subprocess, "IDLE_PRIORITY_CLASS", 0x00000040) + | getattr(subprocess, "BELOW_NORMAL_PRIORITY_CLASS", 0x00004000) + | getattr(subprocess, "NORMAL_PRIORITY_CLASS", 0x00000020) + | getattr(subprocess, "ABOVE_NORMAL_PRIORITY_CLASS", 0x00008000) + | getattr(subprocess, "HIGH_PRIORITY_CLASS", 0x00000080) + | getattr(subprocess, "REALTIME_PRIORITY_CLASS", 0x00000100) + ) + return int(payload_flags) & allowed + + def _cancel_status_reader(self): + try: + fd = self._status_fd_owner.get_nowait() + except queue.Empty: + return + self._close_fd(fd) + + def _wait_for_startup(self, timeout): + result_queue = queue.Queue(maxsize=1) + # Either the reader claims the FD or failed construction cancels it. + # Thread.start() interruption cannot make both sides own the same FD. + thread = threading.Thread(target=_status_reader, args=(self._status_fd_owner, result_queue), daemon=True) + thread.start() + try: + ok, result = result_queue.get(timeout=max(0.1, timeout)) + except queue.Empty as exc: + raise subprocess.TimeoutExpired(self.args, timeout) from exc + if not ok: + if self._host_process.poll() is not None: + return { + "ok": False, + "stage": "startup", + "message": f"host exited with code {self._host_process.returncode}", + } + raise result + return result + + def _reap_failed_start(self): + if self._host_process is None: + return + if os.name != "nt": + # Keep the observer retained even if startup cleanup is interrupted. + # Its exit, not a timeout or repeated Ctrl+C, acknowledges teardown. + while True: + try: + self._host_process.wait() + return + except BaseException: + try: + time.sleep(0.05) + except BaseException: + pass + try: + self._host_process.wait(timeout=2) + return + except subprocess.TimeoutExpired: + pass + self._host_process.terminate() + try: + self._host_process.wait(timeout=2) + except subprocess.TimeoutExpired: + self._host_process.kill() + self._host_process.wait() + + def _request_stop(self): + if os.getpid() != self._owner_pid: + # A forked proxy owns only its local FD, not the parent's job or lock. + fd, self._control_fd = self._control_fd, None + self._close_fd(fd) + return + with self._control_lock: + fd = self._control_fd + self._control_fd = None + if fd is not None: + try: + if self._configuration_sent: + # EOF alone is insufficient when a fork inherited a writer. + if os.name != "nt": + os.set_blocking(fd, False) + os.write(fd, b"\0") + except OSError: + pass + finally: + self._close_fd(fd) + + @property + def pid(self): + return self._payload_pid + + @property + def host_pid(self): + return self._host_process.pid + + @property + def payload_identity(self): + return dict(self._payload_identity or {}) + + @property + def host_identity(self): + return dict(self._host_identity or {}) + + @property + def job_membership_verified(self): + return bool(self._job_membership_verified) + + def ownership_snapshot(self): + return { + "host_identity": self.host_identity, + "payload_identity": self.payload_identity, + "job_membership_verified": self.job_membership_verified, + "job_accounting": dict(self._job_accounting), + "active_members": [dict(item) for item in self._job_members], + } + + @property + def stdin(self): + return self._host_process.stdin + + @property + def stdout(self): + return self._host_process.stdout + + @property + def stderr(self): + return self._host_process.stderr + + @property + def returncode(self): + return self._host_process.returncode + + def poll(self): + code = self._host_process.poll() + if code is not None: + self._request_stop() + return code + + def wait(self, timeout=None): + try: + try: + return self._host_process.wait(timeout=timeout) + except subprocess.TimeoutExpired as exc: + raise subprocess.TimeoutExpired(self.args, timeout, output=exc.output, stderr=exc.stderr) from None + finally: + if self._host_process.returncode is not None: + self._request_stop() + + def communicate(self, input=None, timeout=None): + try: + try: + return self._host_process.communicate(input=input, timeout=timeout) + except subprocess.TimeoutExpired as exc: + raise subprocess.TimeoutExpired(self.args, timeout, output=exc.output, stderr=exc.stderr) from None + finally: + if self._host_process.returncode is not None: + self._request_stop() + + def terminate(self): + self._request_stop() + + def kill(self): + self._request_stop() + if os.name == "nt" and self._host_process.poll() is None: + # Terminating only the retained host closes its sole Job handle. + self._host_process.kill() + + def send_signal(self, sig): + terminating = {signal.SIGTERM} + if hasattr(signal, "SIGKILL"): + terminating.add(signal.SIGKILL) + if hasattr(signal, "CTRL_BREAK_EVENT"): + terminating.add(signal.CTRL_BREAK_EVENT) + if hasattr(signal, "CTRL_C_EVENT"): + terminating.add(signal.CTRL_C_EVENT) + if sig not in terminating: + raise ValueError("OwnedProcess only supports containment-scoped stop signals") + if hasattr(signal, "SIGKILL") and sig == signal.SIGKILL: + self.kill() + else: + self.terminate() + + def __enter__(self): + return self + + def __exit__(self, exc_type, value, traceback): + if self.stdout: + self.stdout.close() + if self.stderr: + self.stderr.close() + try: + if self.stdin: + self.stdin.close() + finally: + self.wait() + + def __del__(self): + try: + self._request_stop() + except BaseException: + pass + + +def run_owned(*popenargs, input=None, capture_output=False, timeout=None, check=False, **kwargs): + if input is not None: + if kwargs.get("stdin") is not None: + raise ValueError("stdin and input arguments may not both be used") + kwargs["stdin"] = subprocess.PIPE + if capture_output: + if kwargs.get("stdout") is not None or kwargs.get("stderr") is not None: + raise ValueError("stdout/stderr and capture_output may not both be used") + kwargs["stdout"] = subprocess.PIPE + kwargs["stderr"] = subprocess.PIPE + + with OwnedProcess(*popenargs, **kwargs) as process: + try: + stdout, stderr = process.communicate(input, timeout=timeout) + except subprocess.TimeoutExpired as exc: + process.kill() + exc.stdout, exc.stderr = process.communicate() + raise + except BaseException: + process.kill() + process.wait() + raise + code = process.poll() + if check and code: + raise subprocess.CalledProcessError(code, popenargs[0], output=stdout, stderr=stderr) + return subprocess.CompletedProcess(popenargs[0], code, stdout, stderr) + + +def _open_inherited_fd(value, write=False): + if os.name != "nt": + fd = int(value) + os.set_inheritable(fd, False) + return fd + + import msvcrt + + flags = os.O_WRONLY if write else os.O_RDONLY + flags |= getattr(os, "O_BINARY", 0) + fd = msvcrt.open_osfhandle(int(value), flags) + os.set_inheritable(fd, False) + return fd + + +def _windows_job_limit_information(job_memory_limit_bytes=0): + job_memory_limit_bytes = _validate_job_memory_limit_bytes(job_memory_limit_bytes) + limits = _JOBOBJECT_EXTENDED_LIMIT_INFORMATION() + limits.BasicLimitInformation.LimitFlags = _JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE + if job_memory_limit_bytes: + limits.BasicLimitInformation.LimitFlags |= _JOB_OBJECT_LIMIT_JOB_MEMORY + limits.JobMemoryLimit = job_memory_limit_bytes + return limits + + +def _windows_job(job_memory_limit_bytes=0, job_cpu_weight=0): + job = _CREATE_JOB_OBJECT(None, None) + if not job: + raise ctypes.WinError(ctypes.get_last_error()) + try: + handle_flag_inherit = 0x00000001 + if not _SET_HANDLE_INFORMATION(job, handle_flag_inherit, 0): + raise ctypes.WinError(ctypes.get_last_error()) + handle_flags = wintypes.DWORD() + if not _GET_HANDLE_INFORMATION(job, ctypes.byref(handle_flags)): + raise ctypes.WinError(ctypes.get_last_error()) + if handle_flags.value & handle_flag_inherit: + raise OSError(errno.EPERM, "Job handle remained inheritable") + + limits = _windows_job_limit_information(job_memory_limit_bytes) + if not _SET_INFORMATION_JOB_OBJECT(job, 9, ctypes.byref(limits), ctypes.sizeof(limits)): + raise ctypes.WinError(ctypes.get_last_error()) + if job_cpu_weight: + cpu_policy = _JOBOBJECT_CPU_RATE_CONTROL_INFORMATION() + cpu_policy.ControlFlags = ( + _JOB_OBJECT_CPU_RATE_CONTROL_ENABLE + | _JOB_OBJECT_CPU_RATE_CONTROL_WEIGHT_BASED + ) + cpu_policy.Weight = int(job_cpu_weight) + if not _SET_INFORMATION_JOB_OBJECT( + job, _JOB_OBJECT_CPU_RATE_CONTROL_INFORMATION, + ctypes.byref(cpu_policy), ctypes.sizeof(cpu_policy), + ): + raise ctypes.WinError(ctypes.get_last_error()) + if os.getenv(_TEST_FAIL_JOB_ASSIGNMENT) == "1": + raise OSError(errno.EPERM, "deliberate Job assignment failure") + current_process = _GET_CURRENT_PROCESS() + if not _ASSIGN_PROCESS_TO_JOB_OBJECT(job, current_process): + raise ctypes.WinError(ctypes.get_last_error()) + in_job = wintypes.BOOL() + if not _IS_PROCESS_IN_JOB(current_process, job, ctypes.byref(in_job)): + raise ctypes.WinError(ctypes.get_last_error()) + if not in_job.value: + raise OSError(errno.EPERM, "host Job membership verification failed") + return _KERNEL32, job + except BaseException: + _CLOSE_HANDLE(job) + raise + + +def _windows_process_identity(handle, pid): + creation = _FILETIME() + ignored_exit = _FILETIME() + ignored_kernel = _FILETIME() + ignored_user = _FILETIME() + if not _GET_PROCESS_TIMES( + handle, ctypes.byref(creation), ctypes.byref(ignored_exit), + ctypes.byref(ignored_kernel), ctypes.byref(ignored_user), + ): + raise ctypes.WinError(ctypes.get_last_error()) + filetime = (int(creation.dwHighDateTime) << 32) | int(creation.dwLowDateTime) + path = ctypes.create_unicode_buffer(32768) + length = wintypes.DWORD(len(path)) + if not _QUERY_FULL_PROCESS_IMAGE_NAME(handle, 0, path, ctypes.byref(length)): + raise ctypes.WinError(ctypes.get_last_error()) + return { + "pid": int(pid), + "creation_time": f"windows-filetime:{filetime}", + "creation_time_unix": (filetime - 116444736000000000) / 10000000.0, + "executable": os.path.normcase(os.path.realpath(os.path.abspath(path.value))), + } + + +def _windows_process_memory_priority(handle): + policy = _MEMORY_PRIORITY_INFORMATION() + if not _GET_PROCESS_INFORMATION( + handle, _PROCESS_MEMORY_PRIORITY, ctypes.byref(policy), ctypes.sizeof(policy), + ): + raise ctypes.WinError(ctypes.get_last_error()) + return int(policy.MemoryPriority) + + +def _windows_set_process_memory_priority(handle, priority): + policy = _MEMORY_PRIORITY_INFORMATION() + policy.MemoryPriority = int(priority) + if not _SET_PROCESS_INFORMATION( + handle, _PROCESS_MEMORY_PRIORITY, ctypes.byref(policy), ctypes.sizeof(policy), + ): + raise ctypes.WinError(ctypes.get_last_error()) + if _windows_process_memory_priority(handle) != int(priority): + raise OSError(errno.EPERM, "process memory-priority verification failed") + + +def _windows_verify_payload_job( + job, payload, job_memory_limit_bytes=0, job_cpu_weight=0, + process_memory_priority=0, +): + payload_handle = wintypes.HANDLE(int(payload._handle)) + in_job = wintypes.BOOL() + if not _IS_PROCESS_IN_JOB(payload_handle, job, ctypes.byref(in_job)): + raise ctypes.WinError(ctypes.get_last_error()) + if not in_job.value: + raise OSError(errno.EPERM, "payload is not in the exact retained Job") + + process_ids = _JOBOBJECT_BASIC_PROCESS_ID_LIST() + returned = wintypes.DWORD() + if not _QUERY_INFORMATION_JOB_OBJECT( + job, 3, ctypes.byref(process_ids), ctypes.sizeof(process_ids), ctypes.byref(returned), + ): + raise ctypes.WinError(ctypes.get_last_error()) + count = int(process_ids.NumberOfProcessIdsInList) + assigned = int(process_ids.NumberOfAssignedProcesses) + if count > _MAX_JOB_PROCESS_IDS or assigned > _MAX_JOB_PROCESS_IDS: + raise OSError(errno.EOVERFLOW, "Job member list exceeds its configured bound") + members = [int(process_ids.ProcessIdList[index]) for index in range(count)] + if int(payload.pid) not in members: + raise OSError(errno.EPERM, "payload PID is absent from the exact Job member list") + + accounting = _JOBOBJECT_BASIC_ACCOUNTING_INFORMATION() + if not _QUERY_INFORMATION_JOB_OBJECT( + job, 1, ctypes.byref(accounting), ctypes.sizeof(accounting), ctypes.byref(returned), + ): + raise ctypes.WinError(ctypes.get_last_error()) + limits = _JOBOBJECT_EXTENDED_LIMIT_INFORMATION() + if not _QUERY_INFORMATION_JOB_OBJECT( + job, 9, ctypes.byref(limits), ctypes.sizeof(limits), ctypes.byref(returned), + ): + raise ctypes.WinError(ctypes.get_last_error()) + limit_flags = int(limits.BasicLimitInformation.LimitFlags) + if not limit_flags & _JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE: + raise OSError(errno.EPERM, "Job kill-on-close limit is missing") + if job_memory_limit_bytes: + if not limit_flags & _JOB_OBJECT_LIMIT_JOB_MEMORY: + raise OSError(errno.EPERM, "Job memory limit is missing") + if int(limits.JobMemoryLimit) != int(job_memory_limit_bytes): + raise OSError(errno.EPERM, "Job memory limit verification failed") + + if job_cpu_weight: + cpu_policy = _JOBOBJECT_CPU_RATE_CONTROL_INFORMATION() + if not _QUERY_INFORMATION_JOB_OBJECT( + job, _JOB_OBJECT_CPU_RATE_CONTROL_INFORMATION, + ctypes.byref(cpu_policy), ctypes.sizeof(cpu_policy), ctypes.byref(returned), + ): + raise ctypes.WinError(ctypes.get_last_error()) + expected_flags = ( + _JOB_OBJECT_CPU_RATE_CONTROL_ENABLE + | _JOB_OBJECT_CPU_RATE_CONTROL_WEIGHT_BASED + ) + if int(cpu_policy.ControlFlags) != expected_flags or int(cpu_policy.Weight) != int(job_cpu_weight): + raise OSError(errno.EPERM, "Job CPU-weight verification failed") + + payload_memory_priority = _windows_process_memory_priority(payload_handle) + if process_memory_priority and payload_memory_priority != int(process_memory_priority): + raise OSError(errno.EPERM, "payload memory-priority verification failed") + payload_identity = _windows_process_identity(payload_handle, payload.pid) + host_identity = _windows_process_identity(_GET_CURRENT_PROCESS(), os.getpid()) + member_identities = [host_identity] + if payload_identity["pid"] != host_identity["pid"]: + member_identities.append(payload_identity) + return { + "payload_identity": payload_identity, + "host_identity": host_identity, + "job_members": member_identities, + "job_accounting": { + "assigned_processes": assigned, + "listed_processes": count, + "active_processes": int(accounting.ActiveProcesses), + "total_processes": int(accounting.TotalProcesses), + "terminated_processes": int(accounting.TotalTerminatedProcesses), + "peak_job_memory_bytes": int(limits.PeakJobMemoryUsed), + "job_memory_limit_bytes": int(limits.JobMemoryLimit), + "job_limit_flags": limit_flags, + "job_cpu_weight": int(job_cpu_weight), + "process_memory_priority": payload_memory_priority, + }, + } + + +def _host_error(stage, exc): + return { + "ok": False, + "stage": stage, + "message": str(exc) or type(exc).__name__, + "errno": getattr(exc, "errno", None), + "winerror": getattr(exc, "winerror", None), + } + + +def _host_stdio(fd): + try: + os.fstat(fd) + return fd + except OSError: + return None + + +def _terminate_posix_group(process_group): + try: + os.killpg(process_group, signal.SIGKILL) + except ProcessLookupError: + pass + + +def _linux_process_identity(pid): + try: + with open(f"/proc/{pid}/stat", "rb") as handle: + head, separator, tail = handle.read().rpartition(b") ") + fields = tail.split() + if not separator or int(head.partition(b" (")[0]) != pid: + raise ValueError("invalid proc process ID") + start_ticks = int(fields[19]) + if start_ticks < 0: + raise ValueError("invalid proc start time") + executable = os.readlink(f"/proc/{pid}/exe") + if not os.path.isabs(executable): + raise ValueError("nonabsolute proc executable") + executable = os.path.normcase(os.path.realpath(os.path.abspath(executable))) + clock_ticks = int(os.sysconf("SC_CLK_TCK")) + boot_time = None + with open("/proc/stat", "rb") as handle: + for line in handle: + if line.startswith(b"btime "): + boot_time = int(line.split()[1]) + break + if not boot_time or boot_time < 0 or clock_ticks <= 0: + raise ValueError("proc boot time or clock rate unavailable") + identity = { + "pid": pid, + "creation_time": f"proc-start-ticks:{start_ticks}", + "creation_time_unix": boot_time + start_ticks / clock_ticks, + "executable": executable, + } + return _validated_identity(identity, pid) + except (OSError, ValueError, IndexError, OverflowError) as exc: + raise OwnedProcessError("identity", f"unable to inspect retained Linux process {pid}") from exc + + +def _linux_enable_subreaper(): + if sys.platform != "linux" or not all( + hasattr(os, name) for name in ( + "waitid", "WNOWAIT", "WEXITED", "WNOHANG", "P_ALL", "CLD_EXITED", "CLD_KILLED", "CLD_DUMPED", + ) + ): + raise OSError(errno.ENOSYS, "owned POSIX processes require Linux subreapers and waitid") + prctl = ctypes.CDLL(None, use_errno=True).prctl + prctl.argtypes = [ctypes.c_int] + [ctypes.c_ulong] * 4 + prctl.restype = ctypes.c_int + enabled = ctypes.c_int() + if prctl(36, 1, 0, 0, 0) != 0 or prctl(37, ctypes.addressof(enabled), 0, 0, 0) != 0: + raise OSError(ctypes.get_errno(), "unable to establish child subreaper") + if enabled.value != 1: + raise OSError(errno.EPERM, "child subreaper verification failed") + # An inherited SIGCHLD ignore disposition would auto-reap the group leader. + signal.signal(signal.SIGCHLD, signal.SIG_DFL) + if os.getpgid(0) != os.getpid() or os.getsid(0) != os.getpid(): + raise OSError(errno.EPERM, "Linux observer is not in its own session") + + +def _linux_payload_returncode(payload): + # Drain adoptees while the root is alive, but bound each pass so sustained + # child exits cannot starve control requests. Never reap the root here. + for _ in range(_MAX_REAP_BATCH): + status = os.waitid(os.P_ALL, 0, os.WEXITED | os.WNOHANG | os.WNOWAIT) + if status is None: + return None + if status.si_code == os.CLD_EXITED: + returncode = status.si_status + elif status.si_code in (os.CLD_KILLED, os.CLD_DUMPED): + returncode = -status.si_status + else: + raise OSError(errno.ECHILD, "unexpected retained child wait status") + if status.si_pid == payload.pid: + return returncode + if status.si_pid <= 0: + raise OSError(errno.ECHILD, "invalid retained child wait process") + os.waitpid(status.si_pid, 0) + return None + + +def _linux_direct_children(): + try: + with open( + f"/proc/{os.getpid()}/task/{os.getpid()}/children", + "r", encoding="ascii", + ) as handle: + values = handle.read(1024 * 1024).split() + except OSError as exc: + raise OwnedProcessError("descendant_enumeration", "unable to enumerate owned Linux descendants") from exc + if len(values) > _MAX_JOB_PROCESS_IDS: + raise OwnedProcessError("descendant_enumeration", "owned Linux descendant count exceeds its bound") + children = [] + for value in values: + try: + pid = int(value) + except ValueError as exc: + raise OwnedProcessError("descendant_enumeration", "invalid owned Linux descendant identity") from exc + if pid <= 0 or pid == os.getpid(): + raise OwnedProcessError("descendant_enumeration", "invalid owned Linux descendant process ID") + children.append(pid) + return children + + +def _linux_kill_exact_descendant(pid): + try: + identity = _linux_process_identity(pid) + except OwnedProcessError: + return + pidfd = None + try: + if hasattr(os, "pidfd_open"): + try: + pidfd = os.pidfd_open(pid, 0) + except OSError: + pidfd = None + sender = getattr(signal, "pidfd_send_signal", None) + if pidfd is not None and sender is not None: + sender(pidfd, signal.SIGKILL, None, 0) + else: + current = _linux_process_identity(pid) + if ( + current["creation_time"] != identity["creation_time"] + or current["executable"] != identity["executable"] + ): + return + os.kill(pid, signal.SIGKILL) + try: + process_group = os.getpgid(pid) + except ProcessLookupError: + return + if process_group > 0 and process_group != os.getpgrp(): + _terminate_posix_group(process_group) + except ProcessLookupError: + pass + finally: + if pidfd is not None: + os.close(pidfd) + + +def _linux_reap_adopted_batch(): + reaped = 0 + for _ in range(_MAX_REAP_BATCH): + try: + pid, _status = os.waitpid(-1, getattr(os, "WNOHANG", 1)) + except InterruptedError: + continue + except ChildProcessError: + return reaped + if pid == 0: + return reaped + reaped += 1 + return reaped + + +def _linux_stop_and_reap(payload): + if payload.returncode is None: + # Keep the leader unreaped until after killpg, pinning its numeric PGID. + _terminate_posix_group(payload.pid) + payload.wait() + while True: + _linux_reap_adopted_batch() + children = _linux_direct_children() + if not children: + # Recheck after a scheduling point so a just-killed intermediate + # cannot publish an adopted child after teardown was acknowledged. + time.sleep(0.01) + _linux_reap_adopted_batch() + if not _linux_direct_children(): + try: + os.waitpid(-1, getattr(os, "WNOHANG", 1)) + except ChildProcessError: + return + continue + for pid in children: + _linux_kill_exact_descendant(pid) + # Never exit while an escaped-session descendant remains. Each pass is + # bounded; an unkillable process keeps this observer alive and prevents + # the owner from mistaking host exit for successful containment teardown. + time.sleep(0.01) + + +def _exit_like_payload(returncode): + if os.name == "nt": + # Process teardown closes the host's sole Job handle after preserving + # the payload's full DWORD exit code, killing any remaining members. + _EXIT_PROCESS(int(returncode) & 0xFFFFFFFF) + raise AssertionError("ExitProcess returned") + if returncode < 0: + signum = -int(returncode) + if signum != signal.SIGKILL: + signal.signal(signum, signal.SIG_DFL) + signal.pthread_sigmask(signal.SIG_UNBLOCK, {signum}) + os.kill(os.getpid(), signum) + os._exit(int(returncode) & 0xFF) + + +def _host_main(control_value, status_value): + control_fd = _open_inherited_fd(control_value) + status_fd = _open_inherited_fd(status_value, write=True) + status_sent = False + stage = "configuration" + payload = None + job_api = None + job = None + stop_event = threading.Event() + signal_stop_requested = False + + def request_stop(signum=None, frame=None): + nonlocal signal_stop_requested + signal_stop_requested = True + + if os.name != "nt": + signal.signal(signal.SIGTERM, request_stop) + signal.signal(signal.SIGINT, request_stop) + + try: + _require_isolated_host() + config = _read_packet(control_fd) + if int(config.get("creationflags") or 0) & _CREATE_BREAKAWAY_FROM_JOB: + raise ValueError("CREATE_BREAKAWAY_FROM_JOB is forbidden for owned processes") + job_memory_limit_bytes = _validate_job_memory_limit_bytes(config.get("job_memory_limit_bytes", 0)) + job_cpu_weight = _validate_job_cpu_weight(config.get("job_cpu_weight", 0)) + process_memory_priority = _validate_process_memory_priority( + config.get("process_memory_priority", 0) + ) + + if os.name == "nt": + stage = "job_assignment" + if process_memory_priority: + _windows_set_process_memory_priority( + _GET_CURRENT_PROCESS(), process_memory_priority, + ) + job_api, job = _windows_job(job_memory_limit_bytes, job_cpu_weight) + else: + stage = "containment" + _linux_enable_subreaper() + + if signal_stop_requested or stop_event.is_set(): + raise InterruptedError("parent stopped before payload launch") + + stage = "payload_launch" + payload_options = { + "stdin": _host_stdio(0), + "stdout": _host_stdio(1), + "stderr": _host_stdio(2), + "bufsize": int(config.get("bufsize", -1)), + "executable": config.get("executable"), + "shell": bool(config.get("shell")), + "cwd": config.get("cwd"), + "env": config.get("env"), + "restore_signals": bool(config.get("restore_signals", True)), + "close_fds": True, + } + if os.name == "nt": + payload_options["creationflags"] = int(config.get("creationflags") or 0) + else: + payload_options["start_new_session"] = True + payload = subprocess.Popen(config["args"], **payload_options) + if os.name == "nt": + stage = "job_membership" + membership = _windows_verify_payload_job( + job, payload, job_memory_limit_bytes, job_cpu_weight, + process_memory_priority, + ) + status = { + "ok": True, + "pid": payload.pid, + "job_membership_verified": True, + **membership, + } + else: + stage = "identity" + payload_identity = _linux_process_identity(payload.pid) + host_identity = _linux_process_identity(os.getpid()) + if os.getpgid(payload.pid) != payload.pid or os.getsid(payload.pid) != payload.pid: + raise OSError(errno.EPERM, "payload is not in its owned Linux session") + status = { + "ok": True, + "pid": payload.pid, + "job_membership_verified": True, + "payload_identity": payload_identity, + "host_identity": host_identity, + "job_members": [host_identity, payload_identity], + "job_accounting": {"child_subreaper_verified": True, "listed_processes": 2}, + } + _write_packet(status_fd, status) + status_sent = True + os.close(status_fd) + status_fd = None + + def watch_parent(): + try: + os.read(control_fd, 1) + except OSError: + pass + stop_event.set() + + threading.Thread(target=watch_parent, daemon=True).start() + while True: + returncode = payload.poll() if os.name == "nt" else _linux_payload_returncode(payload) + if returncode is not None: + if os.name != "nt": + _linux_stop_and_reap(payload) + return _exit_like_payload(returncode) + if signal_stop_requested or stop_event.wait(0.05): + if os.name == "nt": + job_api.TerminateJobObject(job, 1) + job_api.ExitProcess(1) + else: + _linux_stop_and_reap(payload) + os._exit(1) + except BaseException as exc: + if not status_sent and status_fd is not None: + try: + _write_packet(status_fd, _host_error(stage, exc)) + except BaseException: + pass + if payload is not None and os.name != "nt": + while True: + try: + _linux_stop_and_reap(payload) + break + except OSError: + # Cleanup failure must leave the observer alive, not let a + # caller mistake its exit for confirmed containment teardown. + time.sleep(0.05) + if os.name == "nt" and job is not None: + job_api.ExitProcess(127) + return 127 + finally: + if status_fd is not None: + try: + os.close(status_fd) + except OSError: + pass + try: + os.close(control_fd) + except OSError: + pass + + +if __name__ == "__main__" and len(sys.argv) == 4 and sys.argv[1] == _HOST_FLAG: + sys.exit(_host_main(sys.argv[2], sys.argv[3])) diff --git a/app/paths.py b/app/paths.py new file mode 100644 index 0000000..b0b51e7 --- /dev/null +++ b/app/paths.py @@ -0,0 +1,237 @@ +import ntpath +import os +import re + +from query_policy import validate_rejected_query_policy + + +APP_DIR = os.path.dirname(os.path.abspath(__file__)) +CANONICAL_ROOT = os.path.dirname(APP_DIR) +DEFAULT_TRUFFLEHOG = r"C:\Tools\trufflehog.exe" + +PLACEHOLDER_RE = re.compile(r"\{([A-Za-z_][A-Za-z0-9_]*)\}") + + +class PathResolutionError(ValueError): + pass + + +def _norm(path): + return os.path.normpath(str(path)) + + +def _config_dir(config_path=None): + if config_path: + return os.path.dirname(os.path.abspath(config_path)) + return APP_DIR + + +def _expand(value, context): + text = str(value) + missing = sorted({name for name in PLACEHOLDER_RE.findall(text) if name not in context}) + if missing: + raise PathResolutionError(f"Unknown path placeholder(s): {', '.join(missing)} in {text!r}") + for name in PLACEHOLDER_RE.findall(text): + text = text.replace("{" + name + "}", str(context[name])) + return os.path.expandvars(os.path.expanduser(text)) + + +def is_command_name(value): + text = str(value or "") + return bool(text) and not os.path.isabs(text) and "\\" not in text and "/" not in text + + +def is_database_url(value): + return str(value or "").strip().lower().startswith(("postgresql://", "postgres://")) + + +def resolve_path(value, context=None, base_dir=None, allow_command=False, required=False): + if value is None or str(value).strip() == "": + if required: + raise PathResolutionError("Required path is empty") + return value + + context = context or {} + text = _expand(value, context) + if os.name != 'nt' and (ntpath.splitdrive(text)[0] or '\\' in text): + raise PathResolutionError(f"Windows path is not supported on this platform: {text!r}") + if allow_command and is_command_name(text): + return text + if os.path.isabs(text): + return _norm(text) + base = base_dir or context.get("project_dir") or context.get("config_dir") or os.getcwd() + if os.name != 'nt' and (ntpath.splitdrive(str(base))[0] or '\\' in str(base)): + raise PathResolutionError(f"Windows base path is not supported on this platform: {base!r}") + return _norm(os.path.join(base, text)) + + +def default_trufflehog_path(): + return DEFAULT_TRUFFLEHOG if os.name == 'nt' and os.path.exists(DEFAULT_TRUFFLEHOG) else "trufflehog" + + +def resolve_project_paths(global_config=None, config_path=None): + global_config = global_config or {} + context = {"config_dir": _config_dir(config_path)} + + root_raw = ( + global_config.get("root_dir") + or os.getenv("SCANNER_ROOT_DIR") + or os.getenv("SCANNER_PROJECT_ROOT") + or CANONICAL_ROOT + ) + context["root_dir"] = resolve_path(root_raw, context, base_dir=context["config_dir"], required=True) + + project_raw = global_config.get("project_dir") or os.getenv("SCANNER_PROJECT_DIR") or context["config_dir"] + context["project_dir"] = resolve_path(project_raw, context, base_dir=context["config_dir"], required=True) + + ordered_defaults = [ + ("runtime_dir", os.getenv("SCANNER_RUNTIME_DIR") or os.path.join(context["root_dir"], "runtime")), + ("result_bundle_dir", os.getenv("SCANNER_RESULT_BUNDLE_DIR") or "{runtime_dir}/result_bundles"), + ("result_spool_dir", "{runtime_dir}/result_spool"), + ("results_dir", os.getenv("SCAN_RESULTS_DIR") or "{runtime_dir}/results"), + ("queue_dir", "{runtime_dir}/queues"), + ("state_dir", "{runtime_dir}/state"), + ("log_dir", "{runtime_dir}/logs"), + ("control_dir", "{runtime_dir}/control"), + ("keycheck_dir", "{runtime_dir}/keychecks"), + ("postman_cache_dir", "{runtime_dir}/postman_cache"), + ("gharchive_cache_dir", "{state_dir}/gharchive_cache"), + ("work_dir", os.getenv("TRUFFLEHOG_WORK_DIR") or os.path.join(context["root_dir"], "tmp")), + ("proxy_file", "{runtime_dir}/proxy.txt"), + ("database_path", os.getenv("SCANNER_DB_PATH") or os.getenv("SCAN_DB_PATH") or "{results_dir}/scanner.db"), + ("state_file", "{state_dir}/runner_state.json"), + ("secrets_file", "{project_dir}/secrets.yaml"), + ] + + for key, default in ordered_defaults: + raw = global_config.get(key) or default + context[key] = resolve_path(raw, context, base_dir=context["project_dir"], required=True) + + managed_database_url = os.getenv("TRUF_MANAGED_POSTGRES_DSN") or "" + database_url = managed_database_url or global_config.get("database_url") or os.getenv("SCANNER_DB_URL") or os.getenv("DATABASE_URL") or "" + context["database_url"] = _expand(database_url, context) if database_url else "" + dashboard_db_url = managed_database_url or global_config.get("dashboard_db_url") or os.getenv("SCANNER_DASHBOARD_DB_URL") or context["database_url"] + context["dashboard_db_url"] = _expand(dashboard_db_url, context) if dashboard_db_url else "" + + trufflehog_raw = global_config.get("trufflehog_path") or os.getenv("TRUFFLEHOG_PATH") or default_trufflehog_path() + context["trufflehog_path"] = resolve_path( + trufflehog_raw, + context, + base_dir=context["project_dir"], + allow_command=True, + required=True, + ) + return context + + +def default_project_paths(): + return resolve_project_paths({}, None) + + +def resolve_optional_path(value, path_context, base_dir=None, allow_command=False): + if not value: + return value + return resolve_path(value, path_context, base_dir=base_dir or path_context.get("project_dir"), allow_command=allow_command) + + +def resolve_postgres_data_dir(global_config=None, runtime_dir=None, base_dir=None): + global_config = global_config or {} + runtime_dir = runtime_dir or global_config.get('runtime_dir') + if not runtime_dir: + root_dir = global_config.get('root_dir') or CANONICAL_ROOT + runtime_dir = os.path.join(root_dir, 'runtime') + context = dict(global_config) + context['runtime_dir'] = runtime_dir + raw = global_config.get('postgres_data_dir') or os.path.join(runtime_dir, 'postgres', 'data') + return resolve_path( + raw, + context, + base_dir=base_dir or global_config.get('project_dir') or global_config.get('root_dir'), + required=True, + ) + + +def resolve_postgres_bin_dir(global_config=None, runtime_dir=None, base_dir=None): + global_config = global_config or {} + runtime_dir = runtime_dir or global_config.get('runtime_dir') + if not runtime_dir: + runtime_dir = os.path.join(global_config.get('root_dir') or CANONICAL_ROOT, 'runtime') + context = dict(global_config, runtime_dir=runtime_dir) + return resolve_path( + global_config.get('postgres_bin_dir') or os.path.join(runtime_dir, 'postgres', 'pgsql', 'bin'), + context, + base_dir=base_dir or global_config.get('project_dir') or global_config.get('root_dir'), + required=True, + ) + + +def apply_path_config(config, config_path=None): + config = config or {} + validate_rejected_query_policy(config) + global_config = config.setdefault("global", {}) + path_context = resolve_project_paths(global_config, config_path) + + for key, value in path_context.items(): + global_config[key] = value + + if global_config.get('legacy_result_spool_dir'): + global_config['legacy_result_spool_dir'] = resolve_path( + global_config['legacy_result_spool_dir'], + path_context, + base_dir=path_context['project_dir'], + required=True, + ) + + if global_config.get('postgres_data_dir'): + global_config['postgres_data_dir'] = resolve_postgres_data_dir( + global_config, + path_context['runtime_dir'], + base_dir=path_context['project_dir'], + ) + + if global_config.get('postgres_bin_dir'): + global_config['postgres_bin_dir'] = resolve_postgres_bin_dir( + global_config, + path_context['runtime_dir'], + base_dir=path_context['project_dir'], + ) + + for key in ( + 'api_proxy_file', 'download_proxy_file', 'trufflehog_config', + 'dashboard_db_path', 'scan_limiter_db', 'dockerhub_tag_cache_path', + ): + if global_config.get(key): + global_config[key] = resolve_path( + global_config[key], + path_context, + base_dir=path_context['project_dir'], + required=True, + ) + + supervisor = config.setdefault("supervisor", {}) + supervisor_defaults = { + "log_dir": "{log_dir}", + "control_dir": "{control_dir}", + "instance_file": "{control_dir}/supervisor.instance.json", + "lock_file": "{control_dir}/supervisor.lock", + "supervisor_log": "{log_dir}/supervisor.log", + "status_file": "{log_dir}/supervisor.status.txt", + "dashboard_log": "{log_dir}/dashboard.log", + "state_dir": "{state_dir}", + } + for key, default in supervisor_defaults.items(): + supervisor[key] = resolve_path( + supervisor.get(key) or default, + path_context, + base_dir=path_context["project_dir"], + required=True, + ) + + return config + + +def ensure_directories(paths, keys): + for key in keys: + path = paths.get(key) + if path: + os.makedirs(path, exist_ok=True) diff --git a/app/postgres_runtime.py b/app/postgres_runtime.py new file mode 100644 index 0000000..7644392 --- /dev/null +++ b/app/postgres_runtime.py @@ -0,0 +1,1955 @@ +import sys +import os + +if __name__ == '__main__': + if sys.platform != 'linux' or not os.path.isfile('/.dockerenv') or os.path.abspath(__file__) != '/opt/truf/app/postgres_runtime.py': + raise SystemExit('Docker development copy: runtime control is disabled outside the prepared container. See DOCKER_MIGRATION.md.') + import runpy + runpy.run_path('/opt/truf/app/container_runtime.py')['require_container']() + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('PostgreSQL runtime could not disable bytecode writes') + +import argparse +import contextlib +import json +import re +import shlex +import signal +import socket +import stat +import subprocess +import tempfile +import time +from concurrent.futures import ThreadPoolExecutor +from dataclasses import dataclass +from datetime import datetime, timezone +from enum import Enum +from urllib.parse import quote + +from db_backend import DatabaseUrlError, canonical_postgres_url, connect_postgres, database_url_from_env, parse_postgres_url +from paths import apply_path_config, resolve_postgres_bin_dir, resolve_postgres_data_dir +from process_identity import ( + ProcessExitedError, + ProcessIdentityError, + current_process_identity, + open_process, +) +from runtime_security import ( + ClusterAuthorityLock, + canonical_cluster_data_directory, + canonical_path, + harden_private_file, + is_reparse_point, + private_file_ready, + read_private_json, + preflight_lifecycle_paths, + reject_reparse_components, + require_private_directory, + require_trusted_native_executable, + sha256_file, + write_private_json_exclusive, +) + + +CREATE_NO_WINDOW = 0x08000000 +IS_WINDOWS = os.name == 'nt' +CLUSTER_IDENTITY_SCHEMA = 1 +DEFAULT_CONNECT_TIMEOUT_SEC = 5 +DEFAULT_QUERY_TIMEOUT_MS = 5000 +HOST_AGENT_POSTGRES_SOCKET_DIRECTORY = '/run/truf-postgres' +_HOST_AGENT_AUTH_BEGIN = '# BEGIN TRUF HOST AGENT AUTHORITY\n' +_HOST_AGENT_AUTH_END = '# END TRUF HOST AGENT AUTHORITY\n' + + +class PostgresState(str, Enum): + DISABLED = 'DISABLED' + VERIFYING = 'VERIFYING' + STARTING = 'STARTING' + RECOVERING = 'RECOVERING' + STABILIZING = 'STABILIZING' + READY = 'READY' + BACKOFF = 'BACKOFF' + FOREIGN_OR_CONFIG_ERROR = 'FOREIGN_OR_CONFIG_ERROR' + STOPPING = 'STOPPING' + STOPPED = 'STOPPED' + STOP_FAILED = 'STOP_FAILED' + + +def _write_managed_postgres_auth(path, lines): + path = os.path.abspath(os.fspath(path)) + parent = os.path.dirname(path) + reject_reparse_components(parent) + details = os.stat(path, follow_symlinks=False) + if ( + not stat.S_ISREG(details.st_mode) + or details.st_nlink != 1 + or (not IS_WINDOWS and details.st_uid != os.geteuid()) + or stat.S_IMODE(details.st_mode) & 0o077 + ): + raise ClusterIdentityError('PostgreSQL authentication file is unsafe') + descriptor = None + try: + descriptor = os.open( + path, + os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) + | getattr(os, 'O_NOFOLLOW', 0), + ) + before = os.fstat(descriptor) + with os.fdopen(descriptor, 'rb') as handle: + descriptor = None + payload = handle.read(1024 * 1024 + 1) + after = os.fstat(handle.fileno()) + current = os.stat(path, follow_symlinks=False) + identity = lambda item: ( + item.st_dev, item.st_ino, item.st_size, + getattr(item, 'st_mtime_ns', None), + getattr(item, 'st_ctime_ns', None), + ) + if identity(before) != identity(after) or identity(after) != identity(current): + raise OSError('changed') + except Exception: + raise ClusterIdentityError('PostgreSQL authentication file is unsafe') from None + finally: + if descriptor is not None: + os.close(descriptor) + if len(payload) > 1024 * 1024 or b'\x00' in payload: + raise ClusterIdentityError('PostgreSQL authentication file is invalid') + begin = _HOST_AGENT_AUTH_BEGIN.encode('ascii') + end = _HOST_AGENT_AUTH_END.encode('ascii') + block = begin + ''.join(line + '\n' for line in lines).encode('ascii') + end + if begin in payload or end in payload: + if payload.count(begin) != 1 or payload.count(end) != 1 or not payload.startswith(block): + raise ClusterIdentityError('PostgreSQL host-agent authority conflicts with managed policy') + return + stage = path + '.truf-host-agent-stage' + descriptor = None + try: + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_NOFOLLOW', 0) + descriptor = os.open(stage, flags, 0o600) + with os.fdopen(descriptor, 'wb') as handle: + descriptor = None + handle.write(block) + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + os.replace(stage, path) + directory = os.open(parent, os.O_RDONLY | getattr(os, 'O_DIRECTORY', 0)) + try: + os.fsync(directory) + finally: + os.close(directory) + except Exception: + try: + os.unlink(stage) + except FileNotFoundError: + pass + raise ClusterIdentityError('PostgreSQL host-agent authority could not be installed') from None + finally: + if descriptor is not None: + os.close(descriptor) + + +def _configure_host_agent_peer_authority(paths, values): + if IS_WINDOWS or values.get('user') != 'truf' or values.get('database') != 'truf': + return + data_dir = paths['data_dir'] + _write_managed_postgres_auth( + os.path.join(data_dir, 'pg_hba.conf'), + ('local truf truf peer map=truf_host_agent',), + ) + _write_managed_postgres_auth( + os.path.join(data_dir, 'pg_ident.conf'), + ('truf_host_agent root truf', 'truf_host_agent truf truf'), + ) + + +class ProbeKind(str, Enum): + READY = 'READY' + RECOVERING = 'RECOVERING' + STOPPED = 'STOPPED' + OWNED_START_UNCERTAIN = 'OWNED_START_UNCERTAIN' + FOREIGN_OR_CONFIG_ERROR = 'FOREIGN_OR_CONFIG_ERROR' + + +@dataclass(frozen=True) +class ProbeResult: + kind: ProbeKind + detail: str = '' + postmaster_epoch: str = '' + + +@dataclass(frozen=True) +class StartResult: + accepted: bool + detail: str = '' + foreign_or_config_error: bool = False + uncertain: bool = False + + +@dataclass(frozen=True) +class StopResult: + completed: bool + stopped: bool + detail: str = '' + + +class ClusterIdentityError(ValueError): + pass + + +class PostmasterProcessAbsent(ClusterIdentityError): + pass + + +class OnlineUnavailable(OSError): + pass + + +def utc_now_iso(): + return datetime.now(timezone.utc).isoformat(timespec='seconds') + + +def postgres_runtime_paths(config): + global_config = (config or {}).get('global') or {} + runtime_dir = global_config.get('runtime_dir') + if not runtime_dir: + root_dir = global_config.get('root_dir') or os.path.dirname(os.path.dirname(os.path.abspath(__file__))) + runtime_dir = os.path.join(root_dir, 'runtime') + postgres_dir = os.path.join(runtime_dir, 'postgres') + data_dir = resolve_postgres_data_dir(global_config, runtime_dir) + suffix = '.exe' if os.name == 'nt' else '' + bin_dir = resolve_postgres_bin_dir(global_config, runtime_dir) + if os.name != 'nt': + for path in (postgres_dir, data_dir, bin_dir): + reject_reparse_components(path) + paths = { + 'postgres_dir': canonical_path(postgres_dir), + 'bin_dir': canonical_path(bin_dir), + 'data_dir': canonical_path(data_dir), + 'log_dir': canonical_path(os.path.join(postgres_dir, 'logs')), + 'log_path': os.path.join(postgres_dir, 'postgres.log'), + 'identity_path': os.path.join(postgres_dir, 'cluster_identity.json'), + } + for name in ('postgres', 'pg_ctl', 'pg_isready', 'pg_controldata', 'initdb', 'psql'): + path = os.path.join(bin_dir, name + suffix) + if os.name != 'nt': + reject_reparse_components(path) + paths[name] = canonical_path(path) + return paths + + +def _numbered_log_path(path): + base, extension = os.path.splitext(path) + sequence = 1 + while True: + candidate = f'{base}.{sequence:06d}{extension or ".log"}' + if not os.path.exists(candidate): + return candidate + sequence += 1 + + +def rotate_bounded_postgres_startup_log(path, max_bytes, keep): + absolute = os.path.abspath(os.fspath(path)) + parent = canonical_path(require_private_directory(os.path.dirname(absolute), create=False)) + if not os.path.lexists(absolute): + return + if is_reparse_point(absolute): + raise ClusterIdentityError(f'unsafe PostgreSQL startup log: {absolute}') + reject_reparse_components(absolute) + resolved = canonical_path(absolute) + try: + contained = os.path.commonpath((parent, resolved)) == parent + except ValueError: + contained = False + if not contained or os.path.dirname(resolved) != parent: + raise ClusterIdentityError(f'escaping PostgreSQL startup log: {absolute}') + details = os.stat(absolute, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or details.st_nlink != 1: + raise ClusterIdentityError(f'unsafe PostgreSQL startup log: {absolute}') + harden_private_file(absolute) + verified = os.stat(absolute, follow_symlinks=False) + if ( + not os.path.samestat(details, verified) + or not stat.S_ISREG(verified.st_mode) + or verified.st_nlink != 1 + or is_reparse_point(absolute) + or not private_file_ready(absolute) + ): + raise ClusterIdentityError(f'PostgreSQL startup log hardening verification failed: {absolute}') + if os.path.getsize(absolute) >= max(1, int(max_bytes)): + if os.path.getsize(absolute) > max(1, int(max_bytes)): + with open(absolute, 'r+b') as handle: + handle.truncate(max(1, int(max_bytes))) + handle.flush() + os.fsync(handle.fileno()) + destination = _numbered_log_path(absolute) + os.replace(absolute, destination) + harden_private_file(destination) + base, extension = os.path.splitext(os.path.basename(absolute)) + pattern = re.compile(rf'^{re.escape(base)}\.\d{{6}}{re.escape(extension or ".log")}$') + rotated = [] + with os.scandir(parent) as entries: + for entry in entries: + if ( + pattern.fullmatch(entry.name) + and entry.is_file(follow_symlinks=False) + and not entry.is_symlink() + and not is_reparse_point(entry.path) + ): + rotated.append(entry.path) + rotated.sort(key=lambda value: os.path.getmtime(value), reverse=True) + for old in rotated[max(0, int(keep or 0)):]: + reject_reparse_components(old) + os.remove(old) + + +def prune_postgres_collector_logs(log_dir, keep, max_bytes=0): + if not os.path.lexists(log_dir): + return 0 + verified_log_dir = require_private_directory(log_dir, create=False) + verified_root = canonical_path(verified_log_dir) + pattern = re.compile(r'postgresql-[0-9]{8}-[0-9]{6}\.log') + logs = [] + with os.scandir(verified_log_dir) as entries: + for entry in entries: + if not pattern.fullmatch(entry.name): + raise ClusterIdentityError(f'unknown PostgreSQL collector log entry: {entry.path}') + if entry.is_symlink() or is_reparse_point(entry.path): + raise ClusterIdentityError(f'unsafe PostgreSQL collector log entry: {entry.path}') + details = os.stat(entry.path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or details.st_nlink != 1: + raise ClusterIdentityError(f'unsafe PostgreSQL collector log entry: {entry.path}') + if os.path.dirname(canonical_path(entry.path)) != verified_root: + raise ClusterIdentityError(f'escaping PostgreSQL collector log entry: {entry.path}') + harden_private_file(entry.path) + if not private_file_ready(entry.path): + raise ClusterIdentityError(f'PostgreSQL collector log is not private: {entry.path}') + logs.append(entry.path) + logs.sort(key=lambda value: os.path.getmtime(value), reverse=True) + removed = 0 + keep = max(1, int(keep or 1)) + max_bytes = max(0, int(max_bytes or 0)) + for index, old in enumerate(logs): + size = os.path.getsize(old) + if index == 0 and max_bytes and size > max_bytes * 2: + raise ClusterIdentityError('active PostgreSQL collector log exceeds the verified rotation bound') + if index >= keep or (index > 0 and max_bytes and size > max_bytes): + reject_reparse_components(old) + os.remove(old) + removed += 1 + return removed + + +def configured_cluster_values(): + return { + 'database': str(os.getenv('TRUF_POSTGRES_DB') or 'truf'), + 'user': str(os.getenv('TRUF_POSTGRES_USER') or 'truf'), + 'port': int(os.getenv('TRUF_POSTGRES_PORT') or 5432), + } + + +def canonical_database_url(): + values = configured_cluster_values() + url = database_url_from_env() + if not url and os.getenv('TRUF_POSTGRES_PASSWORD') is not None: + password = os.getenv('TRUF_POSTGRES_PASSWORD') or '' + url = ( + f'postgresql://{quote(values["user"], safe="")}:{quote(password, safe="")}' + f'@127.0.0.1:{values["port"]}/{quote(values["database"], safe="")}' + ) + if not url: + return '' + return canonical_postgres_url(url, values['database'], values['user'], values['port']) + + +def _postgres_subprocess_environment(overrides=None): + environment = os.environ.copy() + if os.name != 'nt': + environment = {key: value for key, value in environment.items() if not key.upper().startswith('PG')} + environment['LC_ALL'] = 'C' + environment['LANG'] = 'C' + environment.update(overrides or {}) + return environment + + +def _run_bounded(command, timeout, capture_output=True, environment_overrides=None): + if os.name != 'nt': + command = [require_trusted_native_executable(command[0]), *command[1:]] + return subprocess.run( + command, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE if capture_output else subprocess.DEVNULL, + stderr=subprocess.STDOUT if capture_output else subprocess.DEVNULL, + text=True, + errors='replace', + timeout=max(1, float(timeout)), + env=_postgres_subprocess_environment(environment_overrides), + creationflags=CREATE_NO_WINDOW if os.name == 'nt' else 0, + close_fds=True, + ) + + +def _system_identifier(paths): + result = _run_bounded([paths['pg_controldata'], paths['data_dir']], timeout=15) + if result.returncode != 0: + raise ClusterIdentityError((result.stdout or '').strip() or 'pg_controldata failed') + match = re.search(r'^Database system identifier\s*:\s*(\d+)\s*$', result.stdout or '', re.MULTILINE | re.IGNORECASE) + if not match: + raise ClusterIdentityError('pg_controldata did not report a system identifier') + return match.group(1) + + +def _data_directory_major(paths): + version_path = os.path.join(paths['data_dir'], 'PG_VERSION') + try: + with open(version_path, 'r', encoding='ascii') as handle: + cluster_major = handle.read().strip().split('.')[0] + except OSError as exc: + raise ClusterIdentityError(f'unable to read {version_path}') from exc + if not cluster_major.isdigit(): + raise ClusterIdentityError(f'invalid PostgreSQL cluster version: {cluster_major!r}') + return int(cluster_major) + + +def _bundled_postgres_major(paths): + result = _run_bounded([paths['postgres'], '--version'], timeout=10) + match = re.search(r'(\d+)(?:\.\d+)?', result.stdout or '') + if result.returncode != 0 or not match: + raise ClusterIdentityError('unable to determine bundled postgres executable major version') + return int(match.group(1)) + + +def _bootstrap_postgres_major(paths): + cluster_major = _data_directory_major(paths) + if _bundled_postgres_major(paths) != cluster_major: + raise ClusterIdentityError('bundled postgres executable does not match the data directory major version') + return cluster_major + + +def _listener_present(port, timeout=0.25): + try: + with socket.create_connection(('127.0.0.1', int(port)), timeout=max(0.05, float(timeout))): + return True + except OSError: + return False + + +def _require_bootstrap_offline(config, paths, values): + supervisor = (config or {}).get('supervisor') or {} + global_config = (config or {}).get('global') or {} + log_dir = supervisor.get('log_dir') or global_config.get('log_dir') or os.path.join(global_config.get('runtime_dir') or '', 'logs') + control_dir = supervisor.get('control_dir') or global_config.get('control_dir') or os.path.join(global_config.get('runtime_dir') or '', 'control') + instance_path = supervisor.get('instance_file') or os.path.join(control_dir, 'supervisor.instance.json') + if os.path.lexists(instance_path): + raise ClusterIdentityError(f'refusing bootstrap while supervisor metadata exists: {instance_path}') + legacy_instance_path = os.path.join(log_dir, 'supervisor.instance.json') + if os.path.abspath(legacy_instance_path) != os.path.abspath(instance_path) and os.path.lexists(legacy_instance_path): + raise ClusterIdentityError(f'refusing bootstrap while legacy supervisor metadata exists: {legacy_instance_path}') + legacy_pid_path = os.path.join(log_dir, 'supervisor.pid') + if os.path.lexists(legacy_pid_path): + raise ClusterIdentityError(f'refusing bootstrap while legacy supervisor status metadata exists: {legacy_pid_path}') + postmaster_pid = os.path.join(paths['data_dir'], 'postmaster.pid') + if os.path.lexists(postmaster_pid): + raise ClusterIdentityError(f'refusing bootstrap while postmaster.pid exists: {postmaster_pid}') + if _listener_present(values['port']): + raise ClusterIdentityError(f'refusing bootstrap while a listener is present on 127.0.0.1:{values["port"]}') + + +def bootstrap_cluster_identity(config): + paths = postgres_runtime_paths(config) + values = configured_cluster_values() + if os.path.lexists(paths['identity_path']): + raise ClusterIdentityError('refusing to replace an existing cluster identity') + for key in ('data_dir', 'postgres', 'pg_ctl', 'pg_isready', 'pg_controldata'): + if key == 'data_dir' and not os.path.isdir(paths[key]): + raise ClusterIdentityError(f'bundled PostgreSQL data directory not found: {paths[key]}') + if key != 'data_dir' and not os.path.isfile(paths[key]): + raise ClusterIdentityError(f'bundled PostgreSQL executable not found: {paths[key]}') + if key != 'data_dir' and os.name != 'nt': + require_trusted_native_executable(paths[key]) + _require_bootstrap_offline(config, paths, values) + major = _bootstrap_postgres_major(paths) + executable_keys = ('postgres', 'pg_ctl', 'pg_isready', 'pg_controldata') + identity = { + 'schema': CLUSTER_IDENTITY_SCHEMA, + 'private_file_ready': True, + 'data_directory': paths['data_dir'], + 'pg_major': major, + 'executables': {key: paths[key] for key in executable_keys}, + 'executable_sha256': {key: sha256_file(paths[key]) for key in executable_keys}, + 'database': values['database'], + 'user': values['user'], + 'port': values['port'], + 'system_identifier': _system_identifier(paths), + 'created_at': utc_now_iso(), + } + write_private_json_exclusive(paths['identity_path'], identity) + return verify_cluster_identity(config) + + +def _wait_initialization_child(process, *, stop=False): + # A direct, unreaped child is the owner, never a PID guessed from PGDATA. + # Do not unwind the authority lock or remove its private socket on uncertainty. + while True: + try: + if stop and process.poll() is None: + process.send_signal(signal.SIGINT) + return process.wait(timeout=60) + except (OSError, subprocess.SubprocessError, KeyboardInterrupt, SystemExit): + try: + print('PostgreSQL initialize-empty FAILED_HOLD: retaining authority until the owned child exits.', flush=True) + time.sleep(1) + except (OSError, KeyboardInterrupt, SystemExit): + pass + + +def initialize_empty(config): + """Initialize a prepared, empty, independent POSIX PGDATA and bind it once. + + The configured user is initdb's bootstrap superuser. No application schema or + cutover evidence is installed here; those belong to the ordinary migration. + """ + if os.name == 'nt' or not hasattr(os, 'geteuid') or os.geteuid() == 0: + raise ClusterIdentityError('initialize-empty requires a non-root POSIX runtime user') + global_config = (config or {}).get('global') or {} + for key in ('runtime_dir', 'postgres_data_dir'): + value = global_config.get(key) + if not value or not os.path.isabs(value) or any(char in str(value) for char in ('{', '}', '\x00', '\r', '\n')): + raise ClusterIdentityError(f'initialize-empty requires an explicit resolved absolute {key}') + reject_reparse_components(value) + paths = postgres_runtime_paths(config) + if paths['data_dir'] != canonical_cluster_data_directory(config): + raise ClusterIdentityError('initialize-empty data directory authority mismatch') + for key in ('runtime_dir', 'root_dir', 'project_dir'): + if global_config.get(key): + other = canonical_path(reject_reparse_components(global_config[key])) + if os.path.commonpath((other, paths['data_dir'])) in (other, paths['data_dir']): + raise ClusterIdentityError('initialize-empty PGDATA must be independent of application and runtime trees') + for path in (global_config['runtime_dir'], paths['postgres_dir'], paths['data_dir']): + require_private_directory(path, create=False) + endpoint_dsn = canonical_database_url() + if not endpoint_dsn: + raise ClusterIdentityError('initialize-empty requires a canonical managed PostgreSQL DSN') + credentials = parse_postgres_url(endpoint_dsn) + password = credentials['password'] + if not password or any(char in password for char in ('\x00', '\r', '\n')): + raise ClusterIdentityError('initialize-empty requires a nonempty single-line PostgreSQL password') + values = configured_cluster_values() + for key in ('user', 'database'): + value = values[key] + if not value or len(value.encode('utf-8')) > 63 or any(char in value for char in ('\x00', '\r', '\n')): + raise ClusterIdentityError(f'initialize-empty PostgreSQL {key} is invalid') + if values['database'] in ('template0', 'template1') or values['user'].startswith('pg_'): + raise ClusterIdentityError('initialize-empty cannot use a reserved database or role name') + + with ClusterAuthorityLock(config, endpoint_dsn=endpoint_dsn): + if os.path.lexists(paths['identity_path']): + raise ClusterIdentityError('initialize-empty refuses an existing cluster identity') + _require_bootstrap_offline(config, paths, values) + with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as port_probe: + try: + port_probe.bind(('127.0.0.1', values['port'])) + except OSError as exc: + raise ClusterIdentityError('initialize-empty PostgreSQL TCP port is occupied or unavailable') from exc + with os.scandir(paths['data_dir']) as entries: + if next(entries, None) is not None: + raise ClusterIdentityError('initialize-empty refuses nonempty PGDATA; no adoption or repair is performed') + for key in ('postgres', 'pg_ctl', 'pg_isready', 'pg_controldata', 'initdb', 'psql'): + require_trusted_native_executable(paths[key]) + if _bundled_postgres_major(paths) != 16: + raise ClusterIdentityError('initialize-empty requires PostgreSQL 16') + + backend = PostgresBackend(config) + backend._prepare_logging() + with tempfile.TemporaryDirectory(prefix='init-', dir=paths['postgres_dir']) as private_dir: + require_private_directory(private_dir, create=False) + password_path = os.path.join(private_dir, 'password') + passfile_path = os.path.join(private_dir, 'pgpass') + escaped_user = values['user'].replace('\\', '\\\\').replace(':', '\\:') + escaped_password = password.replace('\\', '\\\\').replace(':', '\\:') + for path, payload in ( + (password_path, password + '\n'), + (passfile_path, f'*:{values["port"]}:*:{escaped_user}:{escaped_password}\n'), + ): + descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_NOFOLLOW', 0), 0o600) + with os.fdopen(descriptor, 'w', encoding='utf-8', newline='\n') as handle: + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + if not private_file_ready(path): + raise ClusterIdentityError('initialize-empty password file is not private') + + with open(paths['log_path'], 'ab', buffering=0) as startup_log: + initdb = subprocess.Popen( + [require_trusted_native_executable(paths['initdb']), '-D', paths['data_dir'], + '--encoding=UTF8', '--locale=C', '--auth-local=scram-sha-256', + '--auth-host=scram-sha-256', '--username=' + values['user'], '--pwfile=' + password_path], + stdin=subprocess.DEVNULL, stdout=startup_log, stderr=subprocess.STDOUT, + close_fds=True, start_new_session=True, env=_postgres_subprocess_environment(), + ) + try: + initdb_result = initdb.wait(timeout=120) + finally: + _wait_initialization_child(initdb) + if initdb_result != 0: + raise ClusterIdentityError('initdb failed; PGDATA is retained for review, see the private PostgreSQL log') + _configure_host_agent_peer_authority(paths, values) + system_identifier = _system_identifier(paths) + socket_setting = 'unix_socket_directories="' + private_dir.replace('"', '""') + '"' + backend._start_requested_wall_time = time.time() + process = subprocess.Popen( + [require_trusted_native_executable(paths['postgres']), '-D', paths['data_dir'], + '-c', 'data_directory=' + paths['data_dir'], '-c', 'listen_addresses=', + '-c', socket_setting, '-c', 'unix_socket_permissions=0700', + '-c', f'port={values["port"]}', '-c', 'logging_collector=off'], + stdin=subprocess.DEVNULL, stdout=startup_log, stderr=subprocess.STDOUT, + close_fds=True, start_new_session=True, env=_postgres_subprocess_environment(), + ) + try: + deadline = time.monotonic() + 30 + while True: + if process.poll() is not None: + raise ClusterIdentityError('temporary PostgreSQL exited before initialization') + ready = _run_bounded( + [paths['pg_isready'], '-h', private_dir, '-p', str(values['port']), + '-U', values['user'], '-d', 'postgres', '-t', '1'], timeout=5, + ) + if ready.returncode == 0: + break + if time.monotonic() >= deadline: + raise ClusterIdentityError('temporary PostgreSQL socket readiness timed out') + time.sleep(0.1) + retained = backend._open_postmaster({ + 'data_directory': paths['data_dir'], 'executables': {'postgres': paths['postgres']}, + }, allow_new_after_start=True) + try: + if retained.pid != process.pid: + raise ClusterIdentityError('temporary postmaster is not the directly owned child') + finally: + retained.close() + + def execute(database, statement): + result = _run_bounded( + [paths['psql'], '-X', '-w', '-A', '-t', '--set=ON_ERROR_STOP=1', + '-h', private_dir, '-p', str(values['port']), '-U', values['user'], + '-d', database, '-c', statement], timeout=30, + environment_overrides={'PGPASSFILE': passfile_path, 'PGCONNECT_TIMEOUT': '5'}, + ) + if result.returncode != 0: + raise ClusterIdentityError('temporary PostgreSQL SQL setup failed: ' + (result.stdout or '').strip()) + return result.stdout or '' + + online = json.loads(execute('postgres', """ + SELECT pg_catalog.json_build_object( + 'data_directory', pg_catalog.current_setting('data_directory'), + 'system_identifier', (SELECT system_identifier::text FROM pg_catalog.pg_control_system()), + 'user', CURRENT_USER, 'port', pg_catalog.current_setting('port')::integer, + 'listen_addresses', pg_catalog.current_setting('listen_addresses'), + 'unix_socket', pg_catalog.inet_server_addr() IS NULL) + """)) + if online != { + 'data_directory': paths['data_dir'], 'system_identifier': system_identifier, + 'user': values['user'], 'port': values['port'], 'listen_addresses': '', 'unix_socket': True, + }: + raise ClusterIdentityError('temporary PostgreSQL online identity mismatch') + role = '"' + values['user'].replace('"', '""') + '"' + database = '"' + values['database'].replace('"', '""') + '"' + if values['database'] != 'postgres': + execute('postgres', f"CREATE DATABASE {database} OWNER {role} ENCODING 'UTF8' TEMPLATE template0") + execute(values['database'], f'REVOKE CREATE ON SCHEMA public FROM PUBLIC; ALTER ROLE {role} SET search_path TO public') + finally: + stop_result = _wait_initialization_child(process, stop=True) + if stop_result != 0: + raise ClusterIdentityError('temporary PostgreSQL did not exit cleanly; no cluster identity was bound') + + _require_bootstrap_offline(config, paths, values) + control = _run_bounded([paths['pg_controldata'], paths['data_dir']], timeout=15) + if control.returncode != 0 or not re.search(r'^Database cluster state\s*:\s*shut down\s*$', control.stdout or '', re.MULTILINE): + raise ClusterIdentityError('initialize-empty could not confirm a cleanly stopped cluster') + return bootstrap_cluster_identity(config) + + +def _validated_identity(value): + if not isinstance(value, dict) or value.get('schema') != CLUSTER_IDENTITY_SCHEMA: + raise ClusterIdentityError('unsupported cluster identity schema') + if value.get('private_file_ready') is not True: + raise ClusterIdentityError('cluster identity is not private-file-ready') + required_strings = ('data_directory', 'database', 'user', 'system_identifier', 'created_at') + for key in required_strings: + if not isinstance(value.get(key), str) or not value[key]: + raise ClusterIdentityError(f'invalid cluster identity field: {key}') + if not value['system_identifier'].isdigit(): + raise ClusterIdentityError('invalid cluster system identifier') + try: + major = int(value.get('pg_major')) + port = int(value.get('port')) + except (TypeError, ValueError) as exc: + raise ClusterIdentityError('invalid cluster version or port') from exc + executables = value.get('executables') + hashes = value.get('executable_sha256') + keys = ('postgres', 'pg_ctl', 'pg_isready', 'pg_controldata') + if not isinstance(executables, dict) or not isinstance(hashes, dict): + raise ClusterIdentityError('invalid cluster executable identity') + normalized = dict(value) + normalized['pg_major'] = major + normalized['port'] = port + normalized['data_directory'] = canonical_path(value['data_directory']) + normalized['executables'] = {key: canonical_path(executables.get(key, '')) for key in keys} + normalized['executable_sha256'] = {key: str(hashes.get(key) or '') for key in keys} + return normalized + + +def verify_cluster_identity(config, *, verify_offline_system_identifier=True): + paths = postgres_runtime_paths(config) + try: + identity = _validated_identity(read_private_json(paths['identity_path'])) + except OSError as exc: + raise ClusterIdentityError(str(exc)) from exc + values = configured_cluster_values() + expected = { + 'data_directory': paths['data_dir'], + 'database': values['database'], + 'user': values['user'], + 'port': values['port'], + } + for key, expected_value in expected.items(): + if identity[key] != expected_value: + raise ClusterIdentityError(f'cluster identity {key} mismatch') + for key in ('postgres', 'pg_ctl', 'pg_isready', 'pg_controldata'): + if identity['executables'][key] != paths[key] or not os.path.isfile(paths[key]): + raise ClusterIdentityError(f'cluster executable path mismatch: {key}') + if os.name != 'nt': + require_trusted_native_executable(paths[key]) + if identity['executable_sha256'][key] != sha256_file(paths[key]): + raise ClusterIdentityError(f'cluster executable hash mismatch: {key}') + if identity['pg_major'] != _data_directory_major(paths): + raise ClusterIdentityError('cluster PostgreSQL major version changed') + if verify_offline_system_identifier and identity['system_identifier'] != _system_identifier(paths): + raise ClusterIdentityError('offline cluster system identifier changed') + return identity + + +def _database_url_matches_identity(url, identity): + try: + canonical_postgres_url(url, identity['database'], identity['user'], identity['port']) + except (TypeError, ValueError): + return False + return True + + +@dataclass(frozen=True) +class PostmasterPidRecord: + pid: int + data_directory: str + start_time: float + + +def _parse_postmaster_pid_record(data_dir): + path = os.path.join(data_dir, 'postmaster.pid') + try: + with open(path, 'r', encoding='ascii') as handle: + lines = [handle.readline().strip() for _ in range(3)] + pid = int(lines[0]) + start_time = float(lines[2]) + if pid <= 0 or not lines[1]: + return None + return PostmasterPidRecord(pid, canonical_path(lines[1]), start_time) + except (OSError, ValueError, IndexError): + return None + + +def _parse_postmaster_pid(data_dir): + record = _parse_postmaster_pid_record(data_dir) + return record.pid if record else None + + +def _timestamp(value): + if hasattr(value, 'timestamp'): + return float(value.timestamp()) + return datetime.fromisoformat(str(value).replace('Z', '+00:00')).timestamp() + + +class PostgresBackend: + """Bounded ordinary-subprocess operations used only by one controller worker.""" + + def __init__( + self, + config, + connect_timeout_sec=DEFAULT_CONNECT_TIMEOUT_SEC, + query_timeout_ms=DEFAULT_QUERY_TIMEOUT_MS, + stop_timeout_sec=60, + start_settle_timeout_sec=30, + log_max_mb=64, + log_keep=24, + ): + self.config = config + self.paths = postgres_runtime_paths(config) + self.connect_timeout_sec = max(1, int(connect_timeout_sec)) + self.query_timeout_ms = max(1000, int(query_timeout_ms)) + self.stop_timeout_sec = max(5, int(stop_timeout_sec)) + self.start_settle_timeout_sec = min(120.0, max(0.1, float(start_settle_timeout_sec))) + self.log_max_mb = max(1, int(log_max_mb)) + self.log_keep = max(1, int(log_keep)) + self._expected_process = None + self._start_requested_wall_time = None + self._accepted_start_at_monotonic = None + self._started_postmaster_observed = False + self._verified_identity = None + self._authenticated_postmaster = None + + @property + def owns_start(self): + return ( + getattr(self, '_start_requested_wall_time', None) is not None + or getattr(self, '_accepted_start_at_monotonic', None) is not None + or getattr(self, '_started_postmaster_observed', False) + ) + + def _prepare_logging(self): + log_dir = self.paths.get('log_dir') + if not log_dir: + return + require_private_directory(os.path.dirname(os.path.abspath(self.paths['log_path'])), create=False) + require_private_directory(log_dir, create=True) + rotate_bounded_postgres_startup_log( + self.paths['log_path'], + max(1, int(getattr(self, 'log_max_mb', 64))) * 1024 * 1024, + max(1, int(getattr(self, 'log_keep', 24))), + ) + if not os.path.lexists(self.paths['log_path']): + with open(self.paths['log_path'], 'xb'): + pass + harden_private_file(self.paths['log_path']) + prune_postgres_collector_logs( + log_dir, + max(1, int(getattr(self, 'log_keep', 24))), + max(1, int(getattr(self, 'log_max_mb', 64))) * 1024 * 1024, + ) + + def _prune_logging(self): + log_dir = self.paths.get('log_dir') + if log_dir and os.path.lexists(log_dir): + prune_postgres_collector_logs( + log_dir, + max(1, int(getattr(self, 'log_keep', 24))), + max(1, int(getattr(self, 'log_max_mb', 64))) * 1024 * 1024, + ) + + def _root_job_error(self): + if os.name != 'nt': + return None + try: + identity = current_process_identity() + except (OSError, ValueError) as exc: + return f'unable to verify root supervisor Job membership: {exc}' + if identity.in_job is not False: + return 'root supervisor is inside an application or unknown Windows Job' + return None + + def _remember_process(self, process): + current = self._expected_process + if (current and current.is_running() and current.pid == process.pid + and current.identity.creation_time == process.identity.creation_time): + process.close() + return current + if current: + current.close() + self._expected_process = process + return process + + def _forget_dead_process(self): + if self._expected_process and not self._expected_process.is_running(): + self._expected_process.close() + self._expected_process = None + self._authenticated_postmaster = None + + def _expected_is_live(self, identity): + self._forget_dead_process() + process = self._expected_process + if not process or not process.is_running(): + return False + pid = _parse_postmaster_pid(identity['data_directory']) + return pid == process.pid + + def _open_postmaster(self, identity, allow_new_after_start=False): + record = _parse_postmaster_pid_record(identity['data_directory']) + if not record: + raise ClusterIdentityError('postmaster.pid is absent or invalid') + if record.data_directory != identity['data_directory']: + raise ClusterIdentityError('postmaster.pid data directory does not match bound cluster') + try: + process = open_process(record.pid) + except ProcessExitedError as exc: + raise PostmasterProcessAbsent( + f'postmaster.pid references exited process {record.pid}' + ) from exc + except ProcessIdentityError as exc: + cause = exc.__cause__ + winerror = getattr(cause, 'winerror', None) or getattr(exc, 'winerror', None) + errno_value = getattr(cause, 'errno', None) or getattr(exc, 'errno', None) + if (IS_WINDOWS and winerror in (87, 1168)) or ( + not IS_WINDOWS and errno_value in (2, 3) + ): + raise PostmasterProcessAbsent( + f'postmaster.pid references absent process {record.pid}' + ) from exc + raise ClusterIdentityError(str(exc)) from exc + if process.identity.executable != identity['executables']['postgres']: + process.close() + raise ClusterIdentityError('postmaster executable path mismatch') + if os.name == 'nt' and process.identity.in_job is not False: + process.close() + raise ClusterIdentityError('verified postmaster is inside an application or unknown Windows Job') + if IS_WINDOWS and abs(process.identity.creation_time_unix - record.start_time) > 10.0: + process.close() + raise ClusterIdentityError('postmaster.pid start time does not match the live process') + if IS_WINDOWS and allow_new_after_start: + threshold = float(self._start_requested_wall_time or 0) - 5.0 + if not threshold or process.identity.creation_time_unix < threshold: + process.close() + raise ClusterIdentityError('postmaster creation time does not match this start request') + return process + + def _online_query(self, identity): + url = database_url_from_env() + try: + url = canonical_postgres_url(url, identity['database'], identity['user'], identity['port']) + except DatabaseUrlError as exc: + raise ClusterIdentityError(str(exc)) from exc + connection = None + try: + connection = connect_postgres( + url, + connect_timeout_sec=self.connect_timeout_sec, + statement_timeout_ms=self.query_timeout_ms, + lock_timeout_ms=min(2000, self.query_timeout_ms), + idle_in_transaction_timeout_ms=self.query_timeout_ms, + tcp_user_timeout_ms=self.query_timeout_ms, + ) + row = connection.execute( + """SELECT pg_catalog.current_database() AS database, + CURRENT_USER AS user_name, + pg_catalog.current_setting('data_directory') AS data_directory, + pg_catalog.current_setting('port')::integer AS port, + pg_catalog.pg_is_in_recovery() AS in_recovery, + pg_catalog.pg_postmaster_start_time() AS postmaster_start_time""" + ).fetchone() + system_identifier = None + control_system_available = False + if identity['pg_major'] >= 10: + availability = connection.execute( + """SELECT pg_catalog.to_regprocedure('pg_catalog.pg_control_system()') IS NOT NULL AS present, + CASE + WHEN pg_catalog.to_regprocedure('pg_catalog.pg_control_system()') IS NULL THEN false + ELSE pg_catalog.has_function_privilege( + CURRENT_USER, + pg_catalog.to_regprocedure('pg_catalog.pg_control_system()'), + 'EXECUTE' + ) + END AS permitted""" + ).fetchone() + control_system_available = bool(availability and availability['present'] and availability['permitted']) + if control_system_available: + system_row = connection.execute('SELECT system_identifier::text AS system_identifier FROM pg_catalog.pg_control_system()').fetchone() + system_identifier = system_row['system_identifier'] if system_row else None + except ClusterIdentityError: + raise + except Exception as exc: + raise OnlineUnavailable(str(exc)) from exc + finally: + if connection is not None: + try: + connection.close() + except Exception: + pass + if not row: + raise ClusterIdentityError('authenticated readiness query returned no row') + checks = { + 'database': (str(row['database']), identity['database']), + 'user': (str(row['user_name']), identity['user']), + 'data_directory': (canonical_path(row['data_directory']), identity['data_directory']), + 'port': (int(row['port']), identity['port']), + } + for name, (actual, expected) in checks.items(): + if actual != expected: + raise ClusterIdentityError(f'online cluster {name} mismatch') + if control_system_available and str(system_identifier or '') != identity['system_identifier']: + raise ClusterIdentityError('online cluster system identifier mismatch') + process = self._open_postmaster(identity) + try: + if IS_WINDOWS and abs(process.identity.creation_time_unix - _timestamp(row['postmaster_start_time'])) > 10.0: + raise ClusterIdentityError('postmaster process creation time mismatch') + self._remember_process(process) + # A CLI handoff may lose SQL readiness without losing this exact process. + if control_system_available: + self._authenticated_postmaster = (dict(identity), self._expected_process) + except BaseException: + if process is not self._expected_process: + process.close() + raise + return bool(row['in_recovery']), self._expected_process.identity.creation_time + + def _hint_running(self, identity): + try: + result = _run_bounded([identity['executables']['pg_ctl'], '-D', identity['data_directory'], 'status'], timeout=10) + return result.returncode == 0 + except (OSError, subprocess.SubprocessError): + return False + + def probe(self): + root_error = self._root_job_error() + if root_error: + return ProbeResult(ProbeKind.FOREIGN_OR_CONFIG_ERROR, root_error) + try: + identity = verify_cluster_identity(self.config, verify_offline_system_identifier=False) + except (OSError, ValueError, subprocess.SubprocessError) as exc: + self._verified_identity = None + return ProbeResult(ProbeKind.FOREIGN_OR_CONFIG_ERROR, str(exc)) + self._verified_identity = identity + try: + self._prune_logging() + except (OSError, ValueError) as exc: + return ProbeResult(ProbeKind.FOREIGN_OR_CONFIG_ERROR, f'PostgreSQL log retention validation failed: {exc}') + try: + in_recovery, postmaster_epoch = self._online_query(identity) + if getattr(self, '_accepted_start_at_monotonic', None) is not None: + self._accepted_start_at_monotonic = None + self._started_postmaster_observed = True + if in_recovery: + return ProbeResult( + ProbeKind.RECOVERING, + 'authenticated expected PostgreSQL reports recovery mode', + postmaster_epoch, + ) + return ProbeResult(ProbeKind.READY, 'authenticated cluster identity verified', postmaster_epoch) + except PostmasterProcessAbsent as exc: + unavailable_detail = str(exc) + except ClusterIdentityError as exc: + return ProbeResult(ProbeKind.FOREIGN_OR_CONFIG_ERROR, str(exc)) + except OnlineUnavailable as exc: + unavailable_detail = str(exc) + + if self._expected_is_live(identity): + if getattr(self, '_accepted_start_at_monotonic', None) is not None: + self._accepted_start_at_monotonic = None + self._started_postmaster_observed = True + return ProbeResult( + ProbeKind.RECOVERING, + f'expected postmaster is live but unavailable: {unavailable_detail}', + self._expected_process.identity.creation_time, + ) + pid = _parse_postmaster_pid(identity['data_directory']) + if pid and self._start_requested_wall_time: + try: + self._remember_process(self._open_postmaster(identity, allow_new_after_start=True)) + self._accepted_start_at_monotonic = None + self._started_postmaster_observed = True + return ProbeResult(ProbeKind.RECOVERING, f'new expected postmaster is still starting: {unavailable_detail}') + except PostmasterProcessAbsent: + pid = None + except ClusterIdentityError as exc: + return ProbeResult(ProbeKind.FOREIGN_OR_CONFIG_ERROR, str(exc)) + if pid: + try: + self._remember_process(self._open_postmaster(identity)) + return ProbeResult( + ProbeKind.RECOVERING, + f'bound bundled postmaster is live and awaiting authenticated identity: {unavailable_detail}', + self._expected_process.identity.creation_time, + ) + except PostmasterProcessAbsent: + pid = None + except ClusterIdentityError as exc: + return ProbeResult(ProbeKind.FOREIGN_OR_CONFIG_ERROR, str(exc)) + if _listener_present(identity['port']): + return ProbeResult(ProbeKind.FOREIGN_OR_CONFIG_ERROR, f'unauthenticated or foreign listener on 127.0.0.1:{identity["port"]}') + if self._hint_running(identity): + return ProbeResult(ProbeKind.FOREIGN_OR_CONFIG_ERROR, 'pg_ctl reports an unverified running postmaster') + if getattr(self, '_accepted_start_at_monotonic', None) is not None: + elapsed = time.monotonic() - self._accepted_start_at_monotonic + if elapsed >= float(getattr(self, 'start_settle_timeout_sec', 30.0)): + return ProbeResult( + ProbeKind.OWNED_START_UNCERTAIN, + 'controller-owned PostgreSQL start did not materialize within the bounded settle interval', + ) + return ProbeResult( + ProbeKind.RECOVERING, + 'accepted PostgreSQL start remains unobservable; start ownership is retained', + ) + try: + verify_cluster_identity(self.config) + except (OSError, ValueError, subprocess.SubprocessError) as exc: + self._verified_identity = None + return ProbeResult(ProbeKind.FOREIGN_OR_CONFIG_ERROR, str(exc)) + return ProbeResult(ProbeKind.STOPPED, 'bound cluster is offline') + + def start(self): + probe = self.probe() + if probe.kind == ProbeKind.FOREIGN_OR_CONFIG_ERROR: + return StartResult(False, probe.detail, foreign_or_config_error=True) + if probe.kind != ProbeKind.STOPPED: + return StartResult(False, f'start refused while cluster state is {probe.kind.value}') + identity = self._verified_identity + if not identity: + return StartResult(False, 'verified cluster identity was not retained', foreign_or_config_error=True) + self._start_requested_wall_time = time.time() + self._accepted_start_at_monotonic = None + self._started_postmaster_observed = False + try: + self._prepare_logging() + except (OSError, ValueError) as exc: + self._start_requested_wall_time = None + return StartResult(False, f'PostgreSQL logging setup failed: {exc}', foreign_or_config_error=True) + port = str(identity['port']) + host_agent_authority = ( + not IS_WINDOWS + and identity.get('user') == 'truf' + and identity.get('database') == 'truf' + ) + if host_agent_authority: + require_private_directory( + HOST_AGENT_POSTGRES_SOCKET_DIRECTORY, create=False, + ) + _configure_host_agent_peer_authority(self.paths, identity) + options = [ + '-c', f'data_directory={identity["data_directory"]}', + '-c', 'listen_addresses=127.0.0.1', + '-c', f'port={port}', + ] + if not IS_WINDOWS: + options.extend([ + '-c', 'unix_socket_directories=' + ( + HOST_AGENT_POSTGRES_SOCKET_DIRECTORY + if host_agent_authority else '' + ), + '-c', 'unix_socket_permissions=0700', + ]) + if self.paths.get('log_dir'): + options.extend([ + '-c', 'logging_collector=on', + '-c', 'log_destination=stderr', + '-c', f'log_directory={self.paths["log_dir"]}', + '-c', 'log_filename=postgresql-%Y%m%d-%H%M%S.log', + '-c', 'log_rotation_age=60', + '-c', f'log_rotation_size={max(1, int(getattr(self, "log_max_mb", 64)))}MB', + '-c', 'log_truncate_on_rotation=on', + '-c', 'log_file_mode=0600', + ]) + # Retain launch ownership even when an interrupt prevents a launcher + # result from reaching us. Only stop() may settle an uncertain launch. + self._accepted_start_at_monotonic = time.monotonic() + if IS_WINDOWS: + try: + with open(self.paths['log_path'], 'ab', buffering=0) as startup_log: + process = subprocess.Popen( + [identity['executables']['postgres'], '-D', identity['data_directory'], *options], + stdin=subprocess.DEVNULL, + stdout=startup_log, + stderr=subprocess.STDOUT, + close_fds=True, + creationflags=CREATE_NO_WINDOW, + ) + time.sleep(0.1) + exit_code = process.poll() + except (OSError, subprocess.SubprocessError) as exc: + return StartResult(True, f'detached postgres launch outcome is uncertain: {exc}', uncertain=True) + if exit_code is not None: + self._start_requested_wall_time = None + self._accepted_start_at_monotonic = None + return StartResult(False, f'detached postgres exited during launch with code {exit_code}') + self._accepted_start_at_monotonic = time.monotonic() + return StartResult(True, 'detached postgres start request accepted') + + option_text = shlex.join(options) + command = [ + identity['executables']['pg_ctl'], '-D', identity['data_directory'], + '-l', self.paths['log_path'], '-o', option_text, + 'start', '-W', + ] + try: + result = _run_bounded(command, timeout=30, capture_output=False) + except subprocess.TimeoutExpired as exc: + self._accepted_start_at_monotonic = time.monotonic() + return StartResult( + True, + f'pg_ctl start timed out after {exc.timeout}s; start outcome is uncertain', + uncertain=True, + ) + except (OSError, subprocess.SubprocessError) as exc: + return StartResult(True, f'pg_ctl launch outcome is uncertain: {exc}', uncertain=True) + if result.returncode != 0: + return StartResult(True, (result.stdout or '').strip() or 'pg_ctl start outcome is uncertain', uncertain=True) + self._accepted_start_at_monotonic = time.monotonic() + return StartResult(True, 'pg_ctl accepted the start request') + + def _stop_retained_postmaster(self, identity): + process = self._expected_process + pid = _parse_postmaster_pid(identity['data_directory']) + if not process or not process.is_running() or process.pid != pid: + return StopResult(False, False, 'retained postmaster identity no longer matches postmaster.pid') + command = [ + identity['executables']['pg_ctl'], '-D', identity['data_directory'], + 'stop', '-m', 'fast', '-w', '-t', str(self.stop_timeout_sec), + ] + try: + result = _run_bounded(command, timeout=self.stop_timeout_sec + 10) + except (OSError, subprocess.SubprocessError) as exc: + return StopResult(False, False, f'bounded PostgreSQL stop failed: {exc}') + if result.returncode != 0: + return StopResult(False, False, (result.stdout or '').strip() or 'pg_ctl stop failed') + if process.is_running(): + return StopResult(False, False, 'pg_ctl returned success but the retained postmaster is still running') + process.close() + self._expected_process = None + self._authenticated_postmaster = None + self._accepted_start_at_monotonic = None + self._started_postmaster_observed = False + self._start_requested_wall_time = None + return StopResult(True, True, 'identity-verified PostgreSQL stop completed') + + def stop(self): + probe = self.probe() + accepted_at = getattr(self, '_accepted_start_at_monotonic', None) + if accepted_at is not None and not self._expected_process: + deadline = time.monotonic() + float(getattr(self, 'start_settle_timeout_sec', 30.0)) + while not self._expected_process: + if probe.kind == ProbeKind.FOREIGN_OR_CONFIG_ERROR: + return StopResult(False, False, f'uncertain PostgreSQL start could not be identity-stopped: {probe.detail}') + remaining = deadline - time.monotonic() + if remaining <= 0: + break + time.sleep(min(0.1, remaining)) + probe = self.probe() + if not self._expected_process: + identity = self._verified_identity + if ( + identity + and probe.kind in (ProbeKind.STOPPED, ProbeKind.OWNED_START_UNCERTAIN) + and not _listener_present(identity['port']) + and not self._hint_running(identity) + ): + self._accepted_start_at_monotonic = None + self._started_postmaster_observed = False + self._start_requested_wall_time = None + return StopResult( + True, + True, + 'accepted PostgreSQL start did not materialize during bounded compensation probes', + ) + return StopResult( + False, + False, + 'accepted PostgreSQL start remained unobservable after bounded shutdown probes; stopped state is uncertain', + ) + if probe.kind == ProbeKind.FOREIGN_OR_CONFIG_ERROR: + return StopResult(False, False, f'identity-verified stop refused: {probe.detail}') + if accepted_at is not None and self._expected_process: + identity = self._verified_identity + if not identity: + return StopResult(False, False, 'uncertain PostgreSQL start retained a process without verified cluster identity') + return self._stop_retained_postmaster(identity) + if probe.kind == ProbeKind.STOPPED: + return StopResult(True, True, 'cluster already stopped') + if probe.kind == ProbeKind.RECOVERING: + identity = self._verified_identity + if ( + (accepted_at is not None or getattr(self, '_started_postmaster_observed', False) + or getattr(self, '_authenticated_postmaster', None) == (identity, self._expected_process)) + and identity + and self._expected_process + ): + return self._stop_retained_postmaster(identity) + return StopResult(False, False, 'live expected recovery was left running because online identity is unavailable') + if probe.kind != ProbeKind.READY: + return StopResult(False, False, f'identity-verified stop refused: {probe.detail}') + identity = self._verified_identity + if not identity: + return StopResult(False, False, 'verified cluster identity was not retained') + return self._stop_retained_postmaster(identity) + + def close(self): + if self._expected_process: + self._expected_process.close() + self._expected_process = None + self._authenticated_postmaster = None + + +class PostgresController: + """Nonblocking PostgreSQL state machine. tick() only polls/submits one worker operation.""" + + def __init__( + self, + backend, + enabled=True, + health_interval_sec=5, + stable_ready_interval_sec=60, + ready_loss_grace_sec=0, + backoff_base_sec=30, + backoff_max_sec=600, + executor=None, + authority_check=None, + shutdown_timeout_sec=120, + ): + self.backend = backend + self.state = PostgresState.VERIFYING if enabled else PostgresState.DISABLED + self.health_interval_sec = max(0.1, float(health_interval_sec)) + self.stable_ready_interval_sec = max(0.0, float(stable_ready_interval_sec)) + self.ready_loss_grace_sec = max(0.0, float(ready_loss_grace_sec)) + self.backoff_base_sec = max(30.0, float(backoff_base_sec)) + self.backoff_max_sec = min(600.0, max(self.backoff_base_sec, float(backoff_max_sec))) + self.failures = 0 + self.detail = '' + self.next_action_at = 0.0 + self.stable_since = None + self._stable_epoch = '' + self._ready_epoch = '' + self._ready_loss_since = None + self._executor = executor or ThreadPoolExecutor(max_workers=1, thread_name_prefix='postgres-runtime') + self._owns_executor = executor is None + self._future = None + self._operation = None + self._operation_generation = None + self._generation = 0 + self._shutdown_requested = False + self._stop_submitted = False + self._lifecycle_inert = False + self._automatic_inhibited = False + self._owned_start = False + self._compensating_start_failure = False + self._authority_release_safe = True + self.shutdown_timeout_sec = max(0.1, float(shutdown_timeout_sec)) + self.authority_check = authority_check + + @property + def ready(self): + return self.state == PostgresState.READY + + @property + def terminal(self): + return self.state in (PostgresState.DISABLED, PostgresState.STOPPED, PostgresState.STOP_FAILED) + + @property + def stop_succeeded(self): + return self.state in (PostgresState.DISABLED, PostgresState.STOPPED) + + @property + def authority_release_safe(self): + return self._authority_release_safe and self._future is None and self.stop_succeeded + + @property + def has_inflight_start(self): + return self._operation == 'start' and self._future is not None + + @property + def lifecycle_action_required(self): + """Whether shutdown must preserve or complete controller-owned lifecycle work.""" + if self.state in (PostgresState.DISABLED, PostgresState.STOPPED): + return False + return ( + self._owned_start + or self.has_inflight_start + or self._compensating_start_failure + or self._operation == 'stop' + or self._shutdown_requested + ) + + def _submit(self, operation): + if self._future is not None: + return False + if operation == 'start' and self.authority_check is not None and not self.authority_check(): + self._automatic_inhibited = True + self._lifecycle_inert = True + self.state = PostgresState.FOREIGN_OR_CONFIG_ERROR + self.detail = 'runtime authority drifted before PostgreSQL start' + return False + self._generation += 1 + generation = self._generation + function = getattr(self.backend, operation) + self._future = self._executor.submit(function) + self._operation = operation + self._operation_generation = generation + if operation == 'stop': + self._stop_submitted = True + return True + + def _enter_backoff(self, now, detail): + self.failures += 1 + delay = min(self.backoff_base_sec * (2 ** min(self.failures - 1, 20)), self.backoff_max_sec) + self.state = PostgresState.BACKOFF + self.detail = detail + self.next_action_at = now + delay + self.stable_since = None + self._stable_epoch = '' + self._ready_epoch = '' + self._owned_start = False + + def _handle_probe(self, result, now): + if not isinstance(result, ProbeResult): + self.state = PostgresState.FOREIGN_OR_CONFIG_ERROR + self.detail = 'invalid PostgreSQL probe result' + self._lifecycle_inert = True + return + if result.kind == ProbeKind.FOREIGN_OR_CONFIG_ERROR: + self.state = PostgresState.FOREIGN_OR_CONFIG_ERROR + self.detail = result.detail + self._lifecycle_inert = True + self.stable_since = None + self._stable_epoch = '' + self._ready_epoch = '' + self._ready_loss_since = None + self.next_action_at = now + self.health_interval_sec + return + if result.kind == ProbeKind.OWNED_START_UNCERTAIN: + if not self._owned_start: + self.state = PostgresState.FOREIGN_OR_CONFIG_ERROR + self.detail = 'unowned PostgreSQL start uncertainty was reported' + self._lifecycle_inert = True + return + self._compensating_start_failure = True + self._authority_release_safe = False + self.state = PostgresState.STOPPING + self.detail = result.detail or 'controller-owned start requires bounded compensation' + self._stop_submitted = False + if not self._submit('stop'): + self.state = PostgresState.STOP_FAILED + self.detail = 'unable to submit identity-safe compensation for an uncertain owned start' + return + if result.kind == ProbeKind.RECOVERING: + if ( + self.state == PostgresState.READY + and self.ready_loss_grace_sec > 0 + and result.postmaster_epoch + and result.postmaster_epoch == self._ready_epoch + ): + if self._ready_loss_since is None: + self._ready_loss_since = now + if now - self._ready_loss_since < self.ready_loss_grace_sec: + self.detail = f'transient readiness loss: {result.detail}' + self.next_action_at = now + self.health_interval_sec + return + self._ready_loss_since = None + self.state = PostgresState.RECOVERING + self.detail = result.detail + # A controller-owned start may pass through recovery without becoming + # a foreign adoption. A process found before our own start remains + # observational and lifecycle-inert. + self._lifecycle_inert = not self._owned_start + self.stable_since = None + self._stable_epoch = '' + self._ready_epoch = '' + self.next_action_at = now + self.health_interval_sec + return + if result.kind == ProbeKind.STOPPED: + self._ready_loss_since = None + if self._lifecycle_inert and not self._owned_start: + self.state = PostgresState.FOREIGN_OR_CONFIG_ERROR + self.detail = (result.detail or 'cluster is offline') + '; automatic lifecycle action remains inhibited' + self.next_action_at = now + self.health_interval_sec + return + if self.state in (PostgresState.STARTING, PostgresState.STABILIZING, PostgresState.READY, PostgresState.RECOVERING): + self._enter_backoff(now, result.detail or 'PostgreSQL died before stable readiness') + else: + self.state = PostgresState.VERIFYING + self.detail = result.detail + self._submit('start') + return + # Readiness verifies an observed cluster but does not adopt its lifecycle. + self._ready_loss_since = None + self._lifecycle_inert = not self._owned_start + if self.state == PostgresState.READY and ( + not result.postmaster_epoch or result.postmaster_epoch == self._ready_epoch + ): + self.detail = result.detail + self.next_action_at = now + self.health_interval_sec + return + if ( + self.state != PostgresState.STABILIZING + or self.stable_since is None + or (result.postmaster_epoch and result.postmaster_epoch != self._stable_epoch) + ): + self.state = PostgresState.STABILIZING + self.stable_since = now + self._stable_epoch = result.postmaster_epoch + self.detail = result.detail + if self.stable_ready_interval_sec == 0: + self.state = PostgresState.READY + self.failures = 0 + self._ready_epoch = result.postmaster_epoch + elif now - self.stable_since >= self.stable_ready_interval_sec: + self.state = PostgresState.READY + self.failures = 0 + self._ready_epoch = result.postmaster_epoch + self.detail = result.detail + self.next_action_at = now + self.health_interval_sec + + def _handle_start(self, result, now): + if not isinstance(result, StartResult): + self._enter_backoff(now, 'invalid PostgreSQL start result') + return + if result.foreign_or_config_error: + self.state = PostgresState.FOREIGN_OR_CONFIG_ERROR + self.detail = result.detail + self._lifecycle_inert = True + self.stable_since = None + self._stable_epoch = '' + self._ready_epoch = '' + self.next_action_at = now + self.health_interval_sec + return + if not result.accepted: + self._enter_backoff(now, result.detail or 'PostgreSQL start request failed') + return + self.state = PostgresState.STARTING + self._owned_start = True + self._lifecycle_inert = False + self.detail = result.detail + self.next_action_at = now + + def _consume_completion(self, now): + if self._future is None or not self._future.done(): + return False + future = self._future + operation = self._operation + generation = self._operation_generation + self._future = None + self._operation = None + self._operation_generation = None + try: + result = future.result() + except BaseException as exc: + if generation != self._generation: + if self._shutdown_requested: + self.state = PostgresState.STOPPING + return True + if self._shutdown_requested or operation == 'stop': + self.state = PostgresState.STOP_FAILED + self.detail = f'bounded PostgreSQL stop operation failed: {exc}' + elif operation == 'probe': + self.state = PostgresState.FOREIGN_OR_CONFIG_ERROR + self.detail = f'PostgreSQL verification failed: {exc}' + self._lifecycle_inert = True + else: + self._enter_backoff(now, f'PostgreSQL start operation failed: {exc}') + return True + if generation != self._generation: + if operation == 'start' and isinstance(result, StartResult): + if result.accepted: + self._shutdown_requested = True + self._automatic_inhibited = True + self.state = PostgresState.STOPPING + self._owned_start = True + self.detail = 'late accepted PostgreSQL start requires identity-safe compensating stop' + if not self._submit('stop'): + self.state = PostgresState.STOP_FAILED + self.detail = 'unable to schedule identity-safe stop for late accepted PostgreSQL start' + elif self._shutdown_requested: + self._automatic_inhibited = True + self._lifecycle_inert = True + self._owned_start = False + self._authority_release_safe = True + self.state = PostgresState.STOPPED + self.detail = ( + 'PostgreSQL observer closed after the in-flight start was rejected; ' + 'no controller-owned side effect was created or stopped' + ) + if result.detail: + self.detail += f': {result.detail}' + return True + if operation == 'probe': + self._handle_probe(result, now) + elif operation == 'start': + self._handle_start(result, now) + elif operation == 'stop': + if isinstance(result, StopResult) and result.completed and result.stopped: + self._owned_start = False + self._authority_release_safe = True + if self._compensating_start_failure and not self._shutdown_requested: + self._compensating_start_failure = False + self._stop_submitted = False + self._enter_backoff( + now, + result.detail or 'uncertain owned PostgreSQL start was safely compensated', + ) + else: + self.state = PostgresState.STOPPED + self.detail = result.detail or 'identity-verified PostgreSQL stop completed' + else: + self._compensating_start_failure = False + self.state = PostgresState.STOP_FAILED + self._authority_release_safe = False + self.detail = result.detail if isinstance(result, StopResult) else 'invalid PostgreSQL stop result' + return True + + def tick(self, now=None): + now = time.monotonic() if now is None else float(now) + self._consume_completion(now) + if self.state in (PostgresState.DISABLED, PostgresState.STOPPED, PostgresState.STOP_FAILED): + return self.state + if self._shutdown_requested: + self.state = PostgresState.STOPPING + if self._future is None and not self._stop_submitted: + self._submit('stop') + return self.state + if self._automatic_inhibited: + return self.state + if self._future is not None or now < self.next_action_at: + return self.state + if self.state == PostgresState.BACKOFF: + self.state = PostgresState.VERIFYING + if self.state in ( + PostgresState.VERIFYING, + PostgresState.STARTING, + PostgresState.RECOVERING, + PostgresState.STABILIZING, + PostgresState.READY, + PostgresState.FOREIGN_OR_CONFIG_ERROR, + ): + self._submit('probe') + return self.state + + def request_stop(self): + if self.state == PostgresState.DISABLED: + return + if self.state == PostgresState.STOPPED: + return + if self.state == PostgresState.STOP_FAILED: + return + if self._shutdown_requested: + self.state = PostgresState.STOPPING + return + if self._operation == 'stop': + self._shutdown_requested = True + self._authority_release_safe = False + self.state = PostgresState.STOPPING + self.detail = 'coordinated shutdown is waiting for in-flight PostgreSQL compensation' + return + self._shutdown_requested = True + self._authority_release_safe = False + self._generation += 1 + self.state = PostgresState.STOPPING + self.detail = 'coordinated shutdown requested' + + def inhibit_lifecycle(self, detail): + """Stop all automatic lifecycle submissions without taking process action.""" + if self._shutdown_requested: + return + self._automatic_inhibited = True + self._lifecycle_inert = True + if self._operation != 'stop': + self._generation += 1 + self.state = PostgresState.FOREIGN_OR_CONFIG_ERROR + self.detail = str(detail or 'automatic PostgreSQL lifecycle is inhibited') + self.stable_since = None + self._stable_epoch = '' + self._ready_epoch = '' + + def retry_failed_stop(self): + """Re-arm an identity-safe stop while endpoint authority is still held.""" + if self.state != PostgresState.STOP_FAILED or self._future is not None: + return False + self._generation += 1 + self._shutdown_requested = True + self._stop_submitted = False + self._authority_release_safe = False + self.state = PostgresState.STOPPING + self.detail = 'retrying identity-safe PostgreSQL compensation under retained authority' + return True + + def snapshot(self): + return { + 'state': self.state.value, + 'ready': self.ready, + 'failures': self.failures, + 'detail': self.detail, + 'operation': self._operation, + 'generation': self._generation, + 'stop_succeeded': self.stop_succeeded, + 'lifecycle_inert': self._lifecycle_inert, + 'automatic_inhibited': self._automatic_inhibited, + 'authority_release_safe': self.authority_release_safe, + 'inflight_start': self.has_inflight_start, + 'lifecycle_action_required': self.lifecycle_action_required, + } + + def close(self, wait=False, timeout_sec=None): + """Drain lifecycle work and return whether endpoint authority may be released.""" + timeout = self.shutdown_timeout_sec if timeout_sec is None else max(0.0, float(timeout_sec)) + deadline = time.monotonic() + timeout + observer_only = ( + self.state not in (PostgresState.DISABLED, PostgresState.STOPPED) + and not self.lifecycle_action_required + ) + if observer_only: + self._automatic_inhibited = True + self._lifecycle_inert = True + self._generation += 1 + if self._future is not None: + self._future.cancel() + while self._future is not None and time.monotonic() <= deadline: + self._consume_completion(time.monotonic()) + if self._future is not None: + time.sleep(0.01) + if self._future is None: + self.state = PostgresState.STOPPED + self._authority_release_safe = True + self.detail = ( + 'observer-only PostgreSQL controller closed without stopping a cluster ' + 'this controller did not start' + ) + else: + if self.state == PostgresState.DISABLED: + self._authority_release_safe = True + elif self.state == PostgresState.STOP_FAILED: + self.retry_failed_stop() + elif self.state != PostgresState.STOPPED: + self.request_stop() + + while not self.authority_release_safe and time.monotonic() <= deadline: + self.tick() + if self.state == PostgresState.STOP_FAILED and self._future is None: + break + time.sleep(0.01) + + safe = self.authority_release_safe + if not safe: + operation = self._operation or 'PostgreSQL compensation' + self.state = PostgresState.STOP_FAILED + self._authority_release_safe = False + self.detail = ( + f'bounded close left {operation} unresolved; endpoint authority release is unsafe' + if self._future is not None + else self.detail or 'PostgreSQL compensation did not confirm stopped state' + ) + return False + + if self._owns_executor: + self._executor.shutdown(wait=True, cancel_futures=False) + try: + self.backend.close() + except Exception: + pass + return True + + +def controller_from_config(config, supervisor_config): + return PostgresController( + PostgresBackend( + config, + connect_timeout_sec=int(supervisor_config.get('postgres_connect_timeout_sec', DEFAULT_CONNECT_TIMEOUT_SEC) or DEFAULT_CONNECT_TIMEOUT_SEC), + query_timeout_ms=int(supervisor_config.get('postgres_query_timeout_ms', DEFAULT_QUERY_TIMEOUT_MS) or DEFAULT_QUERY_TIMEOUT_MS), + stop_timeout_sec=int(supervisor_config.get('postgres_stop_timeout_sec', 60) or 60), + start_settle_timeout_sec=float(supervisor_config.get('postgres_start_settle_timeout_sec', 30) or 30), + log_max_mb=int(supervisor_config.get('postgres_log_max_mb', 64) or 64), + log_keep=int(supervisor_config.get('postgres_log_keep', 24) or 24), + ), + enabled=True, + health_interval_sec=float(supervisor_config.get('postgres_health_interval_sec', 15) or 15), + stable_ready_interval_sec=float(supervisor_config.get('postgres_stable_ready_sec', 60) or 60), + ready_loss_grace_sec=float(supervisor_config.get('postgres_ready_loss_grace_sec', 45) or 45), + backoff_base_sec=30, + backoff_max_sec=600, + shutdown_timeout_sec=float(supervisor_config.get('postgres_shutdown_timeout_sec', 120) or 120), + ) + + +def _load_config(path): + try: + import yaml + except ImportError as exc: + raise SystemExit('PyYAML is required') from exc + with open(path, 'r', encoding='utf-8') as handle: + return apply_path_config(yaml.safe_load(handle) or {}, path) + + +def load_postgres_environment(config_path, config): + global_config = (config or {}).get('global') or {} + candidates = [] + if global_config.get('root_dir'): + candidates.append(os.path.join(global_config['root_dir'], '.env.postgres')) + candidates.extend(( + os.path.join(os.path.dirname(config_path), '..', '.env.postgres'), + os.path.join(os.path.dirname(config_path), '.env.postgres'), + )) + loaded = None + for candidate in candidates: + candidate = os.path.abspath(candidate) + if not os.path.isfile(candidate): + continue + with open(candidate, 'r', encoding='utf-8') as handle: + for line in handle: + text = line.strip() + if not text or text.startswith('#') or '=' not in text: + continue + key, value = text.split('=', 1) + key = key.strip() + value = value.strip().strip('"').strip("'") + if key and value and not os.getenv(key): + os.environ[key] = value + loaded = candidate + break + try: + url = canonical_database_url() + except DatabaseUrlError as exc: + raise ClusterIdentityError(str(exc)) from exc + if url: + for key in list(os.environ): + if key.upper().startswith('PG'): + os.environ.pop(key, None) + os.environ['SCANNER_DB_URL'] = url + os.environ['DATABASE_URL'] = url + return loaded + + +_load_postgres_environment = load_postgres_environment + + +@contextlib.contextmanager +def _maintenance_shutdown_requests(): + requested = False + + def request_shutdown(_signum, _frame): + nonlocal requested + requested = True + + previous = {} + try: + for signum in (signal.SIGINT, signal.SIGTERM): + previous[signum] = signal.signal(signum, request_shutdown) + yield lambda: requested + finally: + for signum, handler in previous.items(): + signal.signal(signum, handler) + + +def _owned_maintenance(config, *, start, shutdown_requested): + """Caller holds cluster authority throughout this backend's ownership.""" + backend = PostgresBackend(config) + failure = None + stopped = False + result = None + try: + if start: + result = maintenance_start(config, backend, shutdown_requested=shutdown_requested) + else: + result = maintenance_stop(config, backend) + stopped = True + if shutdown_requested(): + raise ClusterIdentityError('maintenance PostgreSQL shutdown requested') + label = 'authenticated-ready' if start else 'stopped' + print(f'Maintenance PostgreSQL {label}: {result.detail}', flush=True) + except BaseException as exc: + failure = exc + + # Once compensation is required, mutable backend flags cannot replace a + # completed StopResult (an interrupt may have lost that result). + requires_stop = not start or result is not None or getattr(backend, 'owns_start', True) + while True: + try: + if failure is None and shutdown_requested(): + failure = ClusterIdentityError('maintenance PostgreSQL shutdown requested') + if failure is not None and not stopped: + # A rejected observational start owns no lifecycle. A stop, + # or any possibly launched start, needs positive stop proof. + if requires_stop: + try: + print('Maintenance PostgreSQL FAILED_HOLD: retaining backend and cluster authority until identity-verified stop.', flush=True) + except BaseException: + pass + maintenance_stop(config, backend) + stopped = True + # READY is an intentional successful handoff to the next CLI. + # Every failure path instead gets here only after safe compensation. + backend.close() + break + except BaseException as exc: + if failure is None: + failure = exc + try: + time.sleep(1) + except BaseException: + pass + + if failure is not None: + action = 'start' if start else 'stop' + raise ClusterIdentityError( + f'maintenance PostgreSQL {action} failed: {type(failure).__name__}: {failure}' + ) from failure + return result + + +def maintenance_start(config, backend=None, *, shutdown_requested=None): + """A supplied backend stays caller-owned, including failed/uncertain starts.""" + if backend is None: + with _maintenance_shutdown_requests() as requested: + return _owned_maintenance(config, start=True, shutdown_requested=requested) + if shutdown_requested is not None and shutdown_requested(): + raise ClusterIdentityError('maintenance PostgreSQL shutdown requested') + print('Maintenance PostgreSQL start request: validating offline cluster', flush=True) + result = backend.start() + if not result.accepted: + raise ClusterIdentityError(result.detail or 'maintenance PostgreSQL start was refused') + deadline = time.monotonic() + max(30.0, float(getattr(backend, 'start_settle_timeout_sec', 30.0)) + 10.0) + while time.monotonic() < deadline: + if shutdown_requested is not None and shutdown_requested(): + raise ClusterIdentityError('maintenance PostgreSQL shutdown requested') + probe = backend.probe() + if probe.kind == ProbeKind.READY: + return probe + if probe.kind in (ProbeKind.STOPPED, ProbeKind.FOREIGN_OR_CONFIG_ERROR, ProbeKind.OWNED_START_UNCERTAIN): + raise ClusterIdentityError(probe.detail or f'maintenance PostgreSQL entered {probe.kind.value}') + time.sleep(0.25) + raise ClusterIdentityError('maintenance PostgreSQL did not become authenticated-ready before timeout') + + +def maintenance_stop(config, backend=None): + """A supplied backend gets one bounded stop and is never implicitly closed.""" + if backend is None: + with _maintenance_shutdown_requests() as requested: + return _owned_maintenance(config, start=False, shutdown_requested=requested) + result = backend.stop() + if not isinstance(result, StopResult) or result.completed is not True or result.stopped is not True: + detail = result.detail if isinstance(result, StopResult) else '' + raise ClusterIdentityError(detail or 'maintenance PostgreSQL stop did not complete') + return result + + +def main(): + parser = argparse.ArgumentParser(description='Bootstrap or verify bundled PostgreSQL cluster authority while runtime sources are stopped.') + parser.add_argument('action', choices=('initialize-empty', 'bootstrap', 'verify', 'maintenance-start', 'maintenance-stop')) + parser.add_argument('--config', default=os.path.join(os.path.dirname(__file__), 'config.yaml')) + args = parser.parse_args() + config_path = os.path.abspath(args.config) + config = _load_config(config_path) + try: + preflight_lifecycle_paths(config_path, config, authority_profile='server') + load_postgres_environment(config_path, config) + endpoint_dsn = canonical_database_url() + if not endpoint_dsn: + raise ClusterIdentityError('canonical managed PostgreSQL DSN is required') + if args.action.startswith('maintenance-'): + print('Maintenance PostgreSQL preflight: OK', flush=True) + with _maintenance_shutdown_requests() as requested: + with ClusterAuthorityLock(config, endpoint_dsn=endpoint_dsn): + _owned_maintenance( + config, start=args.action == 'maintenance-start', shutdown_requested=requested, + ) + return + if args.action == 'initialize-empty': + identity = initialize_empty(config) + print(f'Initialized and bound stopped PostgreSQL cluster: {postgres_runtime_paths(config)["identity_path"]}') + else: + with ClusterAuthorityLock(config, endpoint_dsn=endpoint_dsn): + if args.action == 'bootstrap': + identity = bootstrap_cluster_identity(config) + print(f'Wrote verified private cluster identity: {postgres_runtime_paths(config)["identity_path"]}') + elif args.action == 'verify': + identity = verify_cluster_identity(config) + print(f'Verified private cluster identity: {postgres_runtime_paths(config)["identity_path"]}') + except (OSError, ValueError, subprocess.SubprocessError) as exc: + raise SystemExit(f'PostgreSQL cluster identity {args.action} failed closed: {exc}') from exc + print(f'PostgreSQL {identity["pg_major"]} system_identifier={identity["system_identifier"]}') + + +if __name__ == '__main__': + main() diff --git a/app/process_identity.py b/app/process_identity.py new file mode 100644 index 0000000..8be4e8b --- /dev/null +++ b/app/process_identity.py @@ -0,0 +1,449 @@ +import ctypes +import os +import select +import signal +import time +from dataclasses import dataclass + +from runtime_security import canonical_path + + +if os.name == 'nt': + from ctypes import wintypes + + class _FILETIME(ctypes.Structure): + _fields_ = [('dwLowDateTime', wintypes.DWORD), ('dwHighDateTime', wintypes.DWORD)] + + class _UNICODE_STRING(ctypes.Structure): + _fields_ = [ + ('Length', wintypes.USHORT), + ('MaximumLength', wintypes.USHORT), + ('Buffer', ctypes.c_void_p), + ] + + _P_DWORD = ctypes.POINTER(wintypes.DWORD) + _P_ULONG = ctypes.POINTER(wintypes.ULONG) + _P_BOOL = ctypes.POINTER(wintypes.BOOL) + _P_FILETIME = ctypes.POINTER(_FILETIME) + _P_UNICODE_STRING = ctypes.POINTER(_UNICODE_STRING) + _P_INT = ctypes.POINTER(ctypes.c_int) + _P_LPWSTR = ctypes.POINTER(wintypes.LPWSTR) + + _KERNEL32 = ctypes.WinDLL('kernel32', use_last_error=True) + _NTDLL = ctypes.WinDLL('ntdll', use_last_error=True) + _SHELL32 = ctypes.WinDLL('shell32', use_last_error=True) + + _GET_EXIT_CODE_PROCESS = _KERNEL32.GetExitCodeProcess + _GET_EXIT_CODE_PROCESS.argtypes = [wintypes.HANDLE, _P_DWORD] + _GET_EXIT_CODE_PROCESS.restype = wintypes.BOOL + _WAIT_FOR_SINGLE_OBJECT = _KERNEL32.WaitForSingleObject + _WAIT_FOR_SINGLE_OBJECT.argtypes = [wintypes.HANDLE, wintypes.DWORD] + _WAIT_FOR_SINGLE_OBJECT.restype = wintypes.DWORD + _CLOSE_HANDLE = _KERNEL32.CloseHandle + _CLOSE_HANDLE.argtypes = [wintypes.HANDLE] + _CLOSE_HANDLE.restype = wintypes.BOOL + _GET_PROCESS_TIMES = _KERNEL32.GetProcessTimes + _GET_PROCESS_TIMES.argtypes = [ + wintypes.HANDLE, _P_FILETIME, _P_FILETIME, _P_FILETIME, _P_FILETIME, + ] + _GET_PROCESS_TIMES.restype = wintypes.BOOL + _QUERY_FULL_PROCESS_IMAGE_NAME = _KERNEL32.QueryFullProcessImageNameW + _QUERY_FULL_PROCESS_IMAGE_NAME.argtypes = [ + wintypes.HANDLE, wintypes.DWORD, wintypes.LPWSTR, _P_DWORD, + ] + _QUERY_FULL_PROCESS_IMAGE_NAME.restype = wintypes.BOOL + _IS_PROCESS_IN_JOB = _KERNEL32.IsProcessInJob + _IS_PROCESS_IN_JOB.argtypes = [wintypes.HANDLE, wintypes.HANDLE, _P_BOOL] + _IS_PROCESS_IN_JOB.restype = wintypes.BOOL + _OPEN_PROCESS = _KERNEL32.OpenProcess + _OPEN_PROCESS.argtypes = [wintypes.DWORD, wintypes.BOOL, wintypes.DWORD] + _OPEN_PROCESS.restype = wintypes.HANDLE + _GET_CURRENT_PROCESS = _KERNEL32.GetCurrentProcess + _GET_CURRENT_PROCESS.argtypes = [] + _GET_CURRENT_PROCESS.restype = wintypes.HANDLE + _TERMINATE_PROCESS = _KERNEL32.TerminateProcess + _TERMINATE_PROCESS.argtypes = [wintypes.HANDLE, wintypes.UINT] + _TERMINATE_PROCESS.restype = wintypes.BOOL + _LOCAL_FREE = _KERNEL32.LocalFree + _LOCAL_FREE.argtypes = [wintypes.HLOCAL] + _LOCAL_FREE.restype = wintypes.HLOCAL + _NT_QUERY_INFORMATION_PROCESS = _NTDLL.NtQueryInformationProcess + _NT_QUERY_INFORMATION_PROCESS.argtypes = [ + wintypes.HANDLE, wintypes.ULONG, ctypes.c_void_p, wintypes.ULONG, _P_ULONG, + ] + _NT_QUERY_INFORMATION_PROCESS.restype = ctypes.c_long + _COMMAND_LINE_TO_ARGV = _SHELL32.CommandLineToArgvW + _COMMAND_LINE_TO_ARGV.argtypes = [wintypes.LPCWSTR, _P_INT] + _COMMAND_LINE_TO_ARGV.restype = _P_LPWSTR +else: + _FILETIME = _UNICODE_STRING = None + _KERNEL32 = _NTDLL = _SHELL32 = None + + +class ProcessIdentityError(OSError): + pass + + +class ProcessExitedError(ProcessIdentityError): + pass + + +@dataclass(frozen=True) +class ProcessIdentity: + pid: int + creation_time: str + creation_time_unix: float + executable: str + in_job: object + + def as_dict(self): + return { + 'pid': int(self.pid), + 'creation_time': str(self.creation_time), + 'creation_time_unix': float(self.creation_time_unix), + 'executable': str(self.executable), + 'in_job': self.in_job, + } + + +class RetainedProcess: + def __init__(self, identity, handle=None, pidfd=None): + self.identity = identity + self._handle = handle + self._pidfd = pidfd + self._closed = False + + @property + def pid(self): + return self.identity.pid + + def is_running(self): + if self._closed: + return False + if os.name == 'nt': + result = _WAIT_FOR_SINGLE_OBJECT(self._handle, 0) + if result == 258: + return True + if result == 0: + return False + raise ctypes.WinError(ctypes.get_last_error()) + try: + current = _posix_identity(self.pid) + return current.creation_time == self.identity.creation_time + except ProcessIdentityError: + return False + + def wait(self, timeout): + timeout = max(0.0, float(timeout)) + if os.name == 'nt': + milliseconds = min(int(timeout * 1000), 0xFFFFFFFE) + result = _WAIT_FOR_SINGLE_OBJECT(self._handle, milliseconds) + if result == 0: + return True + if result == 258: + return False + raise ctypes.WinError(ctypes.get_last_error()) + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if not self.is_running(): + return True + time.sleep(min(0.05, max(0.0, deadline - time.monotonic()))) + return not self.is_running() + + def exit_code(self): + if self._closed: + raise ProcessIdentityError('retained process handle is closed') + if os.name != 'nt': + return None + code = wintypes.DWORD() + if not _GET_EXIT_CODE_PROCESS(self._handle, ctypes.byref(code)): + raise ProcessIdentityError(f'unable to read process exit status: {ctypes.WinError(ctypes.get_last_error())}') + if code.value == 259: + return None + return int(code.value) + + def terminate(self): + if self._closed: + raise ProcessIdentityError('retained process handle is closed') + if os.name == 'nt': + if not _TERMINATE_PROCESS(self._handle, 1): + raise ctypes.WinError(ctypes.get_last_error()) + return + sender = getattr(signal, 'pidfd_send_signal', None) + if self._pidfd is not None and sender is not None: + sender(self._pidfd, signal.SIGTERM, None, 0) + return + current = _posix_identity(self.pid) + if ( + current.creation_time != self.identity.creation_time + or current.executable != self.identity.executable + ): + raise ProcessIdentityError(f'process identity changed before signaling PID {self.pid}') + os.kill(self.pid, signal.SIGTERM) + + def command_line(self): + if self._closed: + raise ProcessIdentityError('retained process handle is closed') + if os.name != 'nt': + try: + with open(f'/proc/{self.pid}/cmdline', 'rb') as handle: + return [item.decode(errors='surrogateescape') for item in handle.read().split(b'\0') if item] + except OSError as exc: + raise ProcessIdentityError(f'unable to read process {self.pid} command line') from exc + + needed = wintypes.ULONG() + _NT_QUERY_INFORMATION_PROCESS(self._handle, 60, None, 0, ctypes.byref(needed)) + if not needed.value: + raise ProcessIdentityError(f'unable to size process {self.pid} command line') + buffer = ctypes.create_string_buffer(needed.value) + status = _NT_QUERY_INFORMATION_PROCESS( + self._handle, 60, buffer, needed.value, ctypes.byref(needed), + ) + if status < 0: + raise ProcessIdentityError(f'unable to read process {self.pid} command line (NTSTATUS 0x{status & 0xFFFFFFFF:08X})') + value = ctypes.cast(buffer, _P_UNICODE_STRING).contents + command = ctypes.wstring_at(value.Buffer, value.Length // ctypes.sizeof(ctypes.c_wchar)) + argc = ctypes.c_int() + argv = _COMMAND_LINE_TO_ARGV(command, ctypes.byref(argc)) + if not argv: + raise ProcessIdentityError(f'unable to parse process {self.pid} command line') + try: + return [argv[index] for index in range(argc.value)] + finally: + _LOCAL_FREE(argv) + + def close(self): + if self._closed: + return + self._closed = True + if os.name == 'nt' and self._handle: + _CLOSE_HANDLE(self._handle) + elif self._pidfd is not None: + try: + os.close(self._pidfd) + except OSError: + pass + + def __enter__(self): + return self + + def __exit__(self, exc_type, value, traceback): + self.close() + + def __del__(self): + try: + self.close() + except BaseException: + pass + + +def _windows_identity(handle, pid): + creation = _FILETIME() + ignored_exit = _FILETIME() + ignored_kernel = _FILETIME() + ignored_user = _FILETIME() + if not _GET_PROCESS_TIMES( + handle, ctypes.byref(creation), ctypes.byref(ignored_exit), + ctypes.byref(ignored_kernel), ctypes.byref(ignored_user), + ): + raise ctypes.WinError(ctypes.get_last_error()) + filetime = (int(creation.dwHighDateTime) << 32) | int(creation.dwLowDateTime) + path_buffer = ctypes.create_unicode_buffer(32768) + path_size = wintypes.DWORD(len(path_buffer)) + if not _QUERY_FULL_PROCESS_IMAGE_NAME(handle, 0, path_buffer, ctypes.byref(path_size)): + raise ctypes.WinError(ctypes.get_last_error()) + in_job = wintypes.BOOL() + if not _IS_PROCESS_IN_JOB(handle, None, ctypes.byref(in_job)): + raise ctypes.WinError(ctypes.get_last_error()) + unix_time = (filetime - 116444736000000000) / 10000000.0 + return ProcessIdentity( + pid=int(pid), + creation_time=f'windows-filetime:{filetime}', + creation_time_unix=unix_time, + executable=canonical_path(path_buffer.value), + in_job=bool(in_job.value), + ) + + +def _posix_identity(pid): + stat_path = f'/proc/{int(pid)}/stat' + try: + with open(stat_path, 'r', encoding='ascii') as handle: + value = handle.read() + close_paren = value.rfind(')') + fields = value[close_paren + 2:].split() + start_ticks = int(fields[19]) + executable = canonical_path(os.readlink(f'/proc/{int(pid)}/exe')) + clock_ticks = int(os.sysconf('SC_CLK_TCK')) + boot_time = None + with open('/proc/stat', 'r', encoding='ascii') as handle: + for line in handle: + if line.startswith('btime '): + boot_time = float(line.split()[1]) + break + if boot_time is None: + raise ValueError('boot time unavailable') + except (OSError, ValueError, IndexError) as exc: + raise ProcessIdentityError(f'unable to inspect process {pid}') from exc + return ProcessIdentity( + pid=int(pid), + creation_time=f'proc-start-ticks:{start_ticks}', + creation_time_unix=boot_time + (start_ticks / float(clock_ticks)), + executable=executable, + in_job=False, + ) + + +def _pidfd_live(pidfd): + poller = select.poll() + poller.register(pidfd, select.POLLIN) + return not bool(poller.poll(0)) + + +def open_process(pid, *, terminate=False): + pid = int(pid) + if pid <= 0: + raise ProcessIdentityError(f'invalid process ID: {pid}') + if os.name == 'nt': + rights = 0x00100000 | 0x00001000 + if terminate: + rights |= 0x00000001 + handle = _OPEN_PROCESS(rights, False, pid) + if not handle: + native_error = ctypes.WinError(ctypes.get_last_error()) + raise ProcessIdentityError(f'unable to open process {pid}: {native_error}') from native_error + try: + wait_result = _WAIT_FOR_SINGLE_OBJECT(handle, 0) + if wait_result == 0: + exit_code = wintypes.DWORD() + code = int(exit_code.value) if _GET_EXIT_CODE_PROCESS( + handle, ctypes.byref(exit_code), + ) else -1 + raise ProcessExitedError( + f'process {pid} has already exited with code {code}' + ) + if wait_result != 258: + raise ProcessIdentityError( + f'unable to wait on process {pid}: {ctypes.WinError(ctypes.get_last_error())}' + ) + exit_code = wintypes.DWORD() + if not _GET_EXIT_CODE_PROCESS(handle, ctypes.byref(exit_code)): + native_error = ctypes.WinError(ctypes.get_last_error()) + raise ProcessIdentityError( + f'unable to read process {pid} exit status: {native_error}' + ) from native_error + try: + identity = _windows_identity(handle, pid) + except OSError as exc: + retry_exit_code = wintypes.DWORD() + if ( + _GET_EXIT_CODE_PROCESS(handle, ctypes.byref(retry_exit_code)) + and retry_exit_code.value != 259 + ): + raise ProcessExitedError( + f'process {pid} exited during identity inspection ' + f'with code {int(retry_exit_code.value)}' + ) from exc + raise ProcessIdentityError(f'unable to inspect process {pid}') from exc + final_wait = _WAIT_FOR_SINGLE_OBJECT(handle, 0) + if final_wait == 0: + raise ProcessExitedError( + f'process {pid} exited during identity inspection' + ) + if final_wait != 258: + raise ProcessIdentityError( + f'unable to confirm process {pid} liveness: ' + f'{ctypes.WinError(ctypes.get_last_error())}' + ) + return RetainedProcess(identity, handle=handle) + except BaseException: + _CLOSE_HANDLE(handle) + raise + pidfd = None + if hasattr(os, 'pidfd_open'): + try: + pidfd = os.pidfd_open(pid, 0) + except ProcessLookupError as exc: + raise ProcessExitedError(f'process {pid} has already exited') from exc + except OSError as exc: + raise ProcessIdentityError( + f'unable to pin process {pid} with pidfd', + ) from exc + try: + if pidfd is not None and not _pidfd_live(pidfd): + raise ProcessExitedError(f'process {pid} exited before identity binding') + identity = _posix_identity(pid) + if pidfd is not None and not _pidfd_live(pidfd): + raise ProcessExitedError(f'process {pid} exited during identity binding') + verified = _posix_identity(pid) + if ( + verified.creation_time != identity.creation_time + or verified.executable != identity.executable + ): + raise ProcessIdentityError( + f'process {pid} identity changed during pidfd binding', + ) + if pidfd is not None and not _pidfd_live(pidfd): + raise ProcessExitedError(f'process {pid} exited after identity binding') + return RetainedProcess(identity, pidfd=pidfd) + except BaseException: + if pidfd is not None: + os.close(pidfd) + raise + + +def current_process_identity(): + if os.name == 'nt': + return _windows_identity(_GET_CURRENT_PROCESS(), os.getpid()) + return _posix_identity(os.getpid()) + + +def verify_retained_process(pid, creation_time, executable, *, terminate=False): + process = open_process(pid, terminate=True) if terminate else open_process(pid) + expected_executable = canonical_path(executable) + if process.identity.creation_time != str(creation_time) or process.identity.executable != expected_executable: + process.close() + raise ProcessIdentityError(f'process identity mismatch for PID {pid}') + return process + + +def serialize_process_identity(identity): + if isinstance(identity, ProcessIdentity): + return identity.as_dict() + raise TypeError('expected ProcessIdentity') + + +def exact_process_identity_state(pid, creation_time, executable): + """Return alive, dead, reused, or unknown without PID-only inference.""" + try: + pid = int(pid) + except (TypeError, ValueError): + return 'unknown' + if pid <= 0 or not creation_time or not executable: + return 'unknown' + try: + process = open_process(pid) + except ProcessExitedError: + return 'dead' + except ProcessIdentityError as exc: + cause = exc.__cause__ + winerror = getattr(cause, 'winerror', None) or getattr(exc, 'winerror', None) + errno_value = getattr(cause, 'errno', None) or getattr(exc, 'errno', None) + if os.name == 'nt' and winerror in (87, 1168): + return 'dead' + if os.name != 'nt' and errno_value in (2, 3): + return 'dead' + return 'unknown' + try: + if not process.is_running(): + return 'dead' + if ( + str(process.identity.creation_time) != str(creation_time) + or canonical_path(process.identity.executable) != canonical_path(executable) + ): + return 'reused' + return 'alive' + except (OSError, ValueError): + return 'unknown' + finally: + process.close() diff --git a/app/query_policy.py b/app/query_policy.py new file mode 100644 index 0000000..b6547c5 --- /dev/null +++ b/app/query_policy.py @@ -0,0 +1,126 @@ +from datetime import datetime +import re + + +REJECTED_QUERY_STATUS = 'rejected_zero_alive' +REJECTED_QUERY_KEYS = { + 'source', 'query', 'status', 'evidence_cutoff', 'successful_scans', + 'findings', 'unique_credentials', 'pending_candidates', + 'ever_alive_credentials', 'reviewed_queue_rows', +} +REJECTED_QUERY_COUNT_KEYS = { + 'successful_scans', 'findings', 'unique_credentials', + 'pending_candidates', 'ever_alive_credentials', 'reviewed_queue_rows', +} +OPERATIONAL_QUERY_SENTINELS = { + ('github_archive', 'gharchive'), + ('github_archive_files', 'gharchive-files'), + ('github_gists', 'gists'), + ('github_actions', 'logs'), + ('gitlab_ci', 'logs'), + ('huggingface', 'spaces'), +} +SOURCE_RE = re.compile(r'[a-z][a-z0-9_]{0,63}') +MAX_QUERY_LENGTH = 512 +MAX_EVIDENCE_COUNT = (1 << 63) - 1 + + +class QueryPolicyError(ValueError): + pass + + +def _active_queries(source, source_config): + raw_queries = source_config.get('queries', []) + if isinstance(raw_queries, str): + raw_queries = raw_queries.split(',') + if not isinstance(raw_queries, (list, tuple)): + raise QueryPolicyError(f'active query policy is invalid for source {source}') + queries = set() + for raw_query in raw_queries: + query = str(raw_query or '').strip() + if not query: + raise QueryPolicyError(f'active query policy contains an empty query for source {source}') + queries.add(query) + return queries + + +def validate_rejected_query_policy(config): + config = config or {} + raw_policy = config.get('query_policy') + if raw_policy is None: + return () + if not isinstance(raw_policy, dict) or set(raw_policy) != {'rejected'}: + raise QueryPolicyError('query_policy must contain only the rejected registry') + raw_entries = raw_policy.get('rejected') + if not isinstance(raw_entries, list): + raise QueryPolicyError('query_policy.rejected must be a list') + + sources = config.get('sources') or {} + if not isinstance(sources, dict): + raise QueryPolicyError('configured sources must be a mapping') + + normalized = [] + seen = set() + for raw_entry in raw_entries: + if not isinstance(raw_entry, dict) or set(raw_entry) != REJECTED_QUERY_KEYS: + raise QueryPolicyError('rejected query evidence shape is invalid') + source = raw_entry.get('source') + query = raw_entry.get('query') + if not isinstance(source, str) or not SOURCE_RE.fullmatch(source): + raise QueryPolicyError('rejected query source is invalid') + if source not in sources or not isinstance(sources[source], dict): + raise QueryPolicyError(f'rejected query source is not configured: {source}') + if ( + not isinstance(query, str) + or query != query.strip() + or not query + or len(query) > MAX_QUERY_LENGTH + ): + raise QueryPolicyError(f'rejected query text is invalid for source {source}') + pair = (source, query) + if pair in seen: + raise QueryPolicyError('rejected query registry contains a duplicate pair') + if pair in OPERATIONAL_QUERY_SENTINELS: + raise QueryPolicyError('operational query sentinel cannot be rejected') + if query in _active_queries(source, sources[source]): + raise QueryPolicyError('active and rejected query policy overlap') + if raw_entry.get('status') != REJECTED_QUERY_STATUS: + raise QueryPolicyError('rejected query status is invalid') + + cutoff = raw_entry.get('evidence_cutoff') + if not isinstance(cutoff, str) or not cutoff or cutoff != cutoff.strip(): + raise QueryPolicyError('rejected query evidence cutoff is invalid') + try: + parsed_cutoff = datetime.fromisoformat(cutoff.replace('Z', '+00:00')) + except ValueError as exc: + raise QueryPolicyError('rejected query evidence cutoff is invalid') from exc + if parsed_cutoff.tzinfo is None or parsed_cutoff.utcoffset() is None: + raise QueryPolicyError('rejected query evidence cutoff must include a timezone') + + counts = {} + for name in REJECTED_QUERY_COUNT_KEYS: + value = raw_entry.get(name) + if ( + isinstance(value, bool) + or not isinstance(value, int) + or value < 0 + or value > MAX_EVIDENCE_COUNT + ): + raise QueryPolicyError(f'rejected query {name} is invalid') + counts[name] = value + if counts['successful_scans'] < 1000: + raise QueryPolicyError('rejected query has fewer than 1000 successful scans') + if counts['pending_candidates'] != 0: + raise QueryPolicyError('rejected query still has pending candidates') + if counts['ever_alive_credentials'] != 0: + raise QueryPolicyError('rejected query has historical alive credentials') + + seen.add(pair) + normalized.append({ + 'source': source, + 'query': query, + 'status': REJECTED_QUERY_STATUS, + 'evidence_cutoff': cutoff, + **{name: counts[name] for name in sorted(REJECTED_QUERY_COUNT_KEYS)}, + }) + return tuple(sorted(normalized, key=lambda entry: (entry['source'], entry['query']))) diff --git a/app/remote_worker_bootstrap.py b/app/remote_worker_bootstrap.py new file mode 100644 index 0000000..e5315fe --- /dev/null +++ b/app/remote_worker_bootstrap.py @@ -0,0 +1,162 @@ +"""Stdlib-only integrity boundary for packaged remote worker clients.""" + +import hashlib +import json +import os +import runpy +import stat +import sys + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('worker bootstrap could not disable bytecode writes') + + +MAX_MANIFEST_BYTES = 1024 * 1024 +MAX_MANIFEST_FILES = 512 +WORKER_PACKAGE_SCHEMA = 3 +WORKER_PROTOCOL_VERSION = 2 + + +def _canonical(path): + return os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(path)))) + + +def _is_reparse_point(path): + details = os.lstat(path) + if stat.S_ISLNK(details.st_mode): + return True + attributes = getattr(details, 'st_file_attributes', 0) + reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0) + return bool(attributes & reparse_attribute) or getattr( + os.path, 'isjunction', lambda _path: False, + )(path) + + +def _relative(value, label): + value = str(value or '') + if ( + not value or len(value) > 512 or '\\' in value or '\x00' in value + or value.startswith('/') or value.endswith('/') + ): + raise RuntimeError(f'invalid worker package {label} path') + if any(part in ('', '.', '..') for part in value.split('/')): + raise RuntimeError(f'invalid worker package {label} path') + return value + + +def _sha256(path): + digest = hashlib.sha256() + with open(path, 'rb', buffering=0) as handle: + for block in iter(lambda: handle.read(1024 * 1024), b''): + digest.update(block) + return digest.hexdigest() + + +def _load_manifest(path): + details = os.stat(path, follow_symlinks=False) + if _is_reparse_point(path) or not stat.S_ISREG(details.st_mode): + raise RuntimeError('worker package manifest is not a regular file') + with open(path, 'rb') as handle: + payload = handle.read(MAX_MANIFEST_BYTES + 1) + if len(payload) > MAX_MANIFEST_BYTES: + raise RuntimeError('worker package manifest exceeds its byte bound') + value = json.loads(payload.decode('utf-8', errors='strict')) + if ( + not isinstance(value, dict) + or type(value.get('schema')) is not int + or value['schema'] != WORKER_PACKAGE_SCHEMA + or type(value.get('protocol_version')) is not int + or value['protocol_version'] != WORKER_PROTOCOL_VERSION + ): + raise RuntimeError('worker package manifest is invalid') + return value + + +def _application_files(app_dir): + files = set() + + def raise_walk_error(exc): + raise RuntimeError(f'unable to inspect worker application root: {exc}') from exc + + for current, directories, names in os.walk( + app_dir, followlinks=False, onerror=raise_walk_error, + ): + for name in directories: + candidate = os.path.join(current, name) + if _is_reparse_point(candidate) or name.lower() == '__pycache__': + raise RuntimeError('worker application directory is unsupported') + for name in names: + candidate = os.path.join(current, name) + details = os.stat(candidate, follow_symlinks=False) + if _is_reparse_point(candidate) or not stat.S_ISREG(details.st_mode): + raise RuntimeError('worker application file is not regular') + files.add(os.path.relpath(candidate, app_dir).replace(os.sep, '/')) + return files + + +def _verify_application(package_root, manifest): + app_root = _relative(manifest.get('app_root'), 'application root') + app_dir = _canonical(os.path.join(package_root, *app_root.split('/'))) + if app_dir != _canonical(os.path.dirname(__file__)) or _is_reparse_point(app_dir): + raise RuntimeError('worker package application root is not canonical') + values = manifest.get('files') + if not isinstance(values, dict) or not 1 <= len(values) <= MAX_MANIFEST_FILES: + raise RuntimeError('worker package file set is invalid') + expected = set() + for name, entry in values.items(): + name = _relative(name, 'file name') + if not isinstance(entry, dict) or set(entry) != {'path', 'sha256'}: + raise RuntimeError('worker package file entry is invalid') + relative = _relative(entry.get('path'), f'file {name}') + if relative != f'{app_root}/{name}': + raise RuntimeError('worker package file path is not canonical') + digest = str(entry.get('sha256') or '') + if len(digest) != 64 or any(ch not in '0123456789abcdef' for ch in digest): + raise RuntimeError('worker package file digest is invalid') + path = _canonical(os.path.join(package_root, *relative.split('/'))) + try: + contained = os.path.commonpath((app_dir, path)) == app_dir + except ValueError: + contained = False + if not contained or _is_reparse_point(path) or _sha256(path) != digest: + raise RuntimeError('worker package application integrity check failed') + expected.add(name) + if _application_files(app_dir) != expected: + raise RuntimeError('worker package application file set drifted') + return app_dir + + +def main(): + if not ( + sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode + ): + raise RuntimeError( + 'remote worker bootstrap requires isolated no-site bytecode-free startup (-I -S -B)' + ) + sys.dont_write_bytecode = True + if len(sys.argv) < 2 or sys.argv[1] != '--': + raise RuntimeError('usage: remote_worker_bootstrap.py -- ') + + package_root = _canonical(os.path.dirname(os.path.dirname(__file__))) + manifest = _load_manifest(os.path.join(package_root, 'worker-package.json')) + app_dir = _verify_application(package_root, manifest) + dependency_dir = os.path.join(app_dir, 'dependencies') + if ( + not os.path.isdir(dependency_dir) or _is_reparse_point(dependency_dir) + or _canonical(dependency_dir) == app_dir + ): + raise RuntimeError('package-local worker dependencies are unavailable') + + entrypoint = os.path.join(app_dir, 'worker_cli.py') + sys.path.insert(0, dependency_dir) + sys.path.insert(0, app_dir) + sys.argv = [entrypoint, *sys.argv[2:]] + runpy.run_path(entrypoint, run_name='__main__') + + +if __name__ == '__main__': + try: + main() + except Exception as exc: + raise SystemExit('remote worker bootstrap rejected launch') from exc diff --git a/app/remote_worker_client.py b/app/remote_worker_client.py new file mode 100644 index 0000000..941504d --- /dev/null +++ b/app/remote_worker_client.py @@ -0,0 +1,2625 @@ +import argparse +from contextlib import nullcontext +import hashlib +import http.client +import json +import ntpath +import os +import posixpath +import re +import secrets +import ssl +import subprocess +import sys +import threading +import time +import traceback +from datetime import datetime, timedelta, timezone +from urllib.parse import urlsplit + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('remote worker client could not disable bytecode writes') + +from owned_process import OwnedProcess +from process_identity import exact_process_identity_state, verify_retained_process +from result_bundle import BundleReservation, ResultBundleReader, bundle_ready_path +from runtime_security import ( + MAX_EXTENDED_PRIVATE_JSON_BYTES, + PrivateFileLock, + atomic_write_private_json, + ensure_private_directory, + read_private_json, + reject_reparse_components, +) +import scanner +from scan_execution import ( + ScanExecutionError, + WorkerBuildCompatibility, + validate_protocol2_remote_assignment, +) +from worker_package import verify_worker_package +from worker_contracts import ( + AssignmentOutcome, + DiagnosticCategory, + DiagnosticExceptionContext, + DiagnosticHTTPContext, + DiagnosticKind, + DiagnosticProcessContext, + MAX_DIAGNOSTIC_BODY_BYTES, + MAX_DIAGNOSTIC_LOG_BYTES, + ScanOutcome, + PROGRESS_OUTBOX_SCHEMA, + WorkerContractError, + WorkerPhase, + build_diagnostic_envelope, + decode_diagnostic_envelope, + encode_diagnostic_envelope, + make_diagnostic_material, + make_log_material, +) +from worker_assignment_runner import ( + RunnerProtocolError, + adopt_runner_bundle, + bind_runner_owner, + bind_transferred_runner_owner, + build_runner_input, + cleanup_runner_root, + create_runner_root, + load_generation_terminal, + load_runner_input, + publish_generation_terminal, + publish_start_gate, + read_runner_events, + runner_paths, + runner_root_name, + transfer_runner_to_janitor, + validate_terminal_against_journal, +) +from worker_local_state import WorkerLocalStateError, prepare_progress_outbox_cursor + + +MAX_API_RESPONSE_BYTES = 64 * 1024 * 1024 +MAX_PENDING_BYTES = MAX_EXTENDED_PRIVATE_JSON_BYTES +DIGEST_RE = re.compile(r'^[a-f0-9]{64}$') +CLIENT_STORAGE_FAILURE_DETAIL = 'local worker storage operation failed' +CLIENT_PROCESS_FAILURE_DETAIL = 'local worker execution failed' +SLOT_STATE_SCHEMA = 2 +RUNNER_STOP_TIMEOUT_SECONDS = 10.0 +TIMEOUT_BUNDLE_PUBLICATION_SECONDS = 30.0 +PROGRESS_RETRY_MAX_SECONDS = 30.0 +PROGRESS_FINAL_DRAIN_SECONDS = 5.0 +PROGRESS_REQUEST_TIMEOUT_SECONDS = 2.0 +NO_WORK_REASONS = frozenset(( + 'empty_queue', 'assignment_cap', 'dispatch_paused', 'capacity', + 'compatibility', +)) +TERMINAL_REPORT_MAX_BYTES = 16 * 1024 + + +class WorkerClientError(RuntimeError): + pass + + +class WorkerAssignmentCompatibilityError(WorkerClientError): + pass + + +class RunnerContainmentPending(WorkerClientError): + pass + + +class RunnerStageTimeout(RunnerProtocolError): + def __init__(self, phase, scan_started_at, scan_deadline_at, message): + self.phase = WorkerPhase(phase).value + self.scan_started_at = str(scan_started_at) + self.scan_deadline_at = str(scan_deadline_at) + super().__init__(str(message)) + + +class WorkerHTTPError(WorkerClientError): + def __init__(self, status_code, code, message, body=None): + self.status_code = int(status_code) + self.code = str(code or 'request_rejected') + self.body = body + super().__init__( + f'worker API request failed ({self.status_code}, {self.code}): {message}' + ) + + +class WorkerNetworkError(OSError): + pass + + +def safe_worker_error_summary(error): + if isinstance(error, WorkerHTTPError): + return f'worker API request failed (HTTP {error.status_code})' + if isinstance(error, WorkerNetworkError): + return 'worker network operation failed' + if isinstance(error, WorkerClientError): + return 'worker protocol or state validation failed' + if isinstance(error, OSError): + return 'local I/O operation failed' + return 'worker operation failed' + + +class WorkerHTTPClient: + def __init__(self, server_url, token, timeout_seconds=120): + parsed = urlsplit(str(server_url or '').rstrip('/')) + if parsed.scheme != 'https' or not parsed.hostname or parsed.username or parsed.password: + raise ValueError('worker server URL must be an HTTPS origin without credentials') + if parsed.query or parsed.fragment or parsed.path not in ('', '/'): + raise ValueError('worker server URL must not contain a path, query, or fragment') + self.host = parsed.hostname + self.port = parsed.port or 443 + self.token = str(token or '') + if not 16 <= len(self.token) <= 512: + raise ValueError('worker token length is invalid') + self.timeout_seconds = max(1, min(3600, int(timeout_seconds))) + self.ssl_context = ssl.create_default_context() + self._progress_connections = set() + self._progress_connections_lock = threading.Lock() + + def _connection(self, timeout_seconds=None): + timeout = self.timeout_seconds + if timeout_seconds is not None: + timeout = min(timeout, max(0.05, float(timeout_seconds))) + return http.client.HTTPSConnection( + self.host, self.port, timeout=timeout, + context=self.ssl_context, + ) + + @staticmethod + def _network_call(operation, *args, **kwargs): + try: + return operation(*args, **kwargs) + except (OSError, http.client.HTTPException) as exc: + raise WorkerNetworkError() from exc + + @staticmethod + def _decode_response(response): + payload = WorkerHTTPClient._network_call( + response.read, MAX_API_RESPONSE_BYTES + 1, + ) + if len(payload) > MAX_API_RESPONSE_BYTES: + raise WorkerClientError('worker API response exceeded its byte bound') + if not payload: + return None + try: + value = json.loads(payload.decode('utf-8', errors='strict')) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise WorkerClientError('worker API returned invalid JSON') from exc + if not isinstance(value, dict): + raise WorkerClientError('worker API returned a non-object response') + return value + + @staticmethod + def _rejection(response, value): + error = (value or {}).get('error') or {} + body = ( + json.dumps(value, ensure_ascii=True, sort_keys=True, separators=(',', ':')).encode('utf-8') + if value is not None else None + ) + return WorkerHTTPError( + response.status, error.get('code'), + error.get('message') or 'request rejected', + body=body, + ) + + def _json_request( + self, method, path, payload, expected, response_wait_callback=None, + include_no_work_reason=False, timeout_seconds=None, + cancelable_progress=False, + ): + body = json.dumps( + payload, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('utf-8') + connection = self._connection( + timeout_seconds + ) if timeout_seconds is not None else self._connection() + if cancelable_progress: + with self._progress_connections_lock: + self._progress_connections.add(connection) + deadline_timer = None + if timeout_seconds is not None: + deadline_timer = threading.Timer( + max(0.01, float(timeout_seconds)), connection.close, + ) + deadline_timer.daemon = True + deadline_timer.start() + try: + self._network_call( + connection.request, + method, path, body=body, + headers={ + 'Authorization': f'Bearer {self.token}', + 'Content-Type': 'application/json', + 'Content-Length': str(len(body)), + }, + ) + if response_wait_callback is not None: + response_wait_callback() + response = self._network_call(connection.getresponse) + value = self._decode_response(response) + if response.status not in expected: + raise self._rejection(response, value) + result = (response.status, value, response.getheader('Retry-After')) + if include_no_work_reason: + return (*result, response.getheader('X-Truf-No-Work-Reason')) + return result + finally: + if deadline_timer is not None: + deadline_timer.cancel() + if cancelable_progress: + with self._progress_connections_lock: + self._progress_connections.discard(connection) + self._network_call(connection.close) + + def cancel_progress_requests(self): + with self._progress_connections_lock: + connections = tuple(self._progress_connections) + for connection in connections: + try: + connection.close() + except OSError: + pass + + def claim(self, request_id, compatibility): + status, value, retry_after, no_work_reason = self._json_request( + 'POST', '/api/v1/worker/claim', + {'request_id': request_id, 'build': compatibility}, + {200, 201, 204}, + include_no_work_reason=True, + ) + if status == 204: + try: + retry_after = int(retry_after) + except (TypeError, ValueError, OverflowError) as exc: + raise WorkerClientError('worker API claim retry delay is invalid') from exc + if not 1 <= retry_after <= 300: + raise WorkerClientError('worker API claim retry delay is invalid') + if no_work_reason is not None and no_work_reason not in NO_WORK_REASONS: + raise WorkerClientError('worker API no-work reason is invalid') + return { + 'retry_after_seconds': retry_after, + 'reason': no_work_reason, + } + if status == 200: + resolution = dict((value or {}).get('resolution') or {}) + if ( + int(resolution.get('reservation_id') or 0) <= 0 + or resolution.get('resolution') not in { + 'bundle_accepted', 'prebundle_report', 'expired', + } + or not DIGEST_RE.fullmatch(str(resolution.get('receipt_id') or '')) + or not re.fullmatch(r'[a-f0-9]{32,64}', str(resolution.get('bundle_id') or '')) + or not re.fullmatch(r'[a-f0-9]{32,64}', str(resolution.get('scan_event_id') or '')) + ): + raise WorkerClientError('worker API claim resolution is invalid') + return {'claim_resolution': resolution} + assignment = (value or {}).get('assignment') + if not isinstance(assignment, dict): + raise WorkerClientError('worker API claim response has no assignment') + return assignment + + def status(self, reservation_id): + connection = self._connection() + try: + self._network_call( + connection.request, + 'GET', f'/api/v1/worker/assignments/{int(reservation_id)}', + headers={'Authorization': f'Bearer {self.token}'}, + ) + response = self._network_call(connection.getresponse) + value = self._decode_response(response) + if response.status != 200: + raise self._rejection(response, value) + return value + finally: + self._network_call(connection.close) + + def progress(self, reservation_id, event, *, timeout_seconds=None): + _, value, _ = self._json_request( + 'POST', f'/api/v1/worker/assignments/{int(reservation_id)}/progress', + event, {200}, + timeout_seconds=( + PROGRESS_REQUEST_TIMEOUT_SECONDS + if timeout_seconds is None else min( + PROGRESS_REQUEST_TIMEOUT_SECONDS, float(timeout_seconds), + ) + ), + cancelable_progress=True, + ) + if ( + not isinstance(value, dict) + or value.get('accepted') is not True + or int(value.get('reservation_id') or 0) != int(reservation_id) + or int(value.get('sequence') or 0) != int(event.get('sequence') or 0) + or not isinstance(value.get('received_at'), str) + or type(value.get('replayed')) is not bool + ): + raise WorkerClientError('worker API progress acceptance is invalid') + return value + + def terminal( + self, reservation_id, report, + response_wait_callback=None, + ): + report = dict(report or {}) + encoded = json.dumps( + report, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + if len(encoded) > TERMINAL_REPORT_MAX_BYTES: + raise WorkerClientError('pending terminal report exceeds its byte bound') + _, value, _ = self._json_request( + 'POST', f'/api/v1/worker/assignments/{int(reservation_id)}/terminal', + report, {200}, + response_wait_callback=response_wait_callback, + ) + return value + + def upload(self, reservation_id, path, response_wait_callback=None): + reject_reparse_components(path) + if not os.path.isfile(path) or os.path.islink(path): + raise WorkerClientError('pending result bundle is not a regular file') + byte_count = os.path.getsize(path) + digest = hashlib.sha256() + with open(path, 'rb', buffering=0) as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b''): + digest.update(chunk) + payload_sha256 = digest.hexdigest() + connection = self._connection() + try: + self._network_call( + connection.putrequest, + 'PUT', f'/api/v1/worker/assignments/{int(reservation_id)}/bundle', + ) + self._network_call( + connection.putheader, 'Authorization', f'Bearer {self.token}', + ) + self._network_call( + connection.putheader, 'Content-Type', 'application/octet-stream', + ) + self._network_call( + connection.putheader, 'Content-Length', str(byte_count), + ) + self._network_call( + connection.putheader, 'X-Truf-Payload-SHA256', payload_sha256, + ) + self._network_call(connection.endheaders) + with open(path, 'rb', buffering=0) as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b''): + self._network_call(connection.send, chunk) + if response_wait_callback is not None: + response_wait_callback() + response = self._network_call(connection.getresponse) + value = self._decode_response(response) + if response.status not in (200, 201): + raise self._rejection(response, value) + if str((value or {}).get('payload_sha256') or '') != payload_sha256: + raise WorkerClientError('worker API receipt does not match the uploaded bundle') + return value + finally: + self._network_call(connection.close) + + +class ProgressOutbox: + def __init__(self, api, state_dir, event_reader): + if not callable(event_reader): + raise TypeError('progress outbox event reader must be callable') + self.api = api + self.event_reader = event_reader + try: + self.path, self.sequence = prepare_progress_outbox_cursor( + state_dir, create=True, + ) + except (OSError, ValueError, WorkerLocalStateError) as exc: + raise WorkerClientError( + 'persisted progress outbox cursor is invalid or conflicting' + ) from exc + + def _checkpoint(self, sequence): + sequence = int(sequence) + if sequence <= self.sequence: + raise WorkerClientError('progress outbox cursor did not advance') + atomic_write_private_json(self.path, { + 'schema': PROGRESS_OUTBOX_SCHEMA, + 'sequence': sequence, + }, max_bytes=64 * 1024) + self.sequence = sequence + + def publish_once(self, *, deadline=None): + events = self.event_reader(self.sequence, 128) + if not events: + return False + for event in events: + sequence = int(event.get('sequence') or 0) + if sequence <= self.sequence: + raise WorkerClientError('progress outbox event order is invalid') + reservation_id = event.get('reservation_id') + if reservation_id is not None: + timeout_seconds = PROGRESS_REQUEST_TIMEOUT_SECONDS + if deadline is not None: + timeout_seconds = deadline - time.monotonic() + if timeout_seconds <= 0: + raise TimeoutError('progress publication deadline elapsed') + timeout_seconds = min( + PROGRESS_REQUEST_TIMEOUT_SECONDS, timeout_seconds, + ) + try: + self.api.progress( + int(reservation_id), event, + timeout_seconds=timeout_seconds, + ) + except WorkerHTTPError as exc: + if not ( + exc.status_code == 410 and exc.code == 'progress_stale' + ): + raise + self._checkpoint(sequence) + return True + + def run(self, stopping): + retry = 0.5 + while not stopping.is_set(): + try: + worked = self.publish_once( + deadline=time.monotonic() + PROGRESS_REQUEST_TIMEOUT_SECONDS, + ) + except Exception: + stopping.wait(retry) + retry = min(PROGRESS_RETRY_MAX_SECONDS, retry * 2) + continue + retry = 0.5 + if not worked: + stopping.wait(0.5) + + def drain(self, timeout_seconds=PROGRESS_FINAL_DRAIN_SECONDS): + deadline = time.monotonic() + max(0.0, float(timeout_seconds)) + retry = 0.05 + while time.monotonic() < deadline: + try: + worked = self.publish_once(deadline=deadline) + except Exception: + remaining = deadline - time.monotonic() + if remaining <= 0: + break + time.sleep(min(retry, remaining)) + retry = min(PROGRESS_RETRY_MAX_SECONDS, retry * 2) + continue + retry = 0.05 + if not worked: + return True + return False + + +class WorkerSlot: + def __init__( + self, slot_id, api, compatibility, state_dir, bundle_root, *, + work_root=None, package_runtime=None, claim_enabled=True, event_callback=None, + terminal_callback=None, diagnostic_callback=None, + process_factory=OwnedProcess, monotonic=time.monotonic, + ): + self.slot_id = int(slot_id) + self.api = api + self.compatibility = WorkerBuildCompatibility.from_mapping(compatibility) + self.state_dir = os.path.abspath(state_dir) + self.state_path = os.path.join(state_dir, f'slot-{self.slot_id}.json') + self.stale_path = os.path.join(state_dir, f'slot-{self.slot_id}-stale.json') + self.bundle_root = bundle_root + self.work_root = ensure_private_directory( + os.path.abspath(work_root or os.path.join(state_dir, 'work')), + reject_reparse=True, + ) + self.package_runtime = dict(package_runtime or {}) + self.claim_enabled = bool(claim_enabled) + self.retry_after_seconds = None + self.event_callback = event_callback + self.terminal_callback = terminal_callback + self.diagnostic_callback = diagnostic_callback + self._diagnostics = [] + self._transport_diagnostics = [] + self._event_phase = None + self._process_factory = process_factory + self._monotonic = monotonic + + @staticmethod + def _event_timestamp(value): + if not value: + return None + try: + parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) + except ValueError as exc: + raise WorkerClientError('worker assignment deadline is invalid') from exc + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.astimezone(timezone.utc).isoformat().replace('+00:00', 'Z') + + @staticmethod + def _assignment_details(state): + assignment = dict((state or {}).get('assignment') or {}) + reservation = dict(assignment.get('reservation') or {}) + deadlines = dict(assignment.get('deadlines') or {}) + source = ( + reservation.get('source') or assignment.get('source') + or reservation.get('platform') or assignment.get('platform') + ) + return { + 'reservation_id': int(reservation.get('reservation_id') or 0) or None, + 'source': str(source) if source else None, + 'scan_deadline_at': WorkerSlot._event_timestamp( + ((state or {}).get('runner') or {}).get('scan_deadline_at') + or assignment.get('scan_deadline_at') + or deadlines.get('scan_deadline_at') + or reservation.get('scan_deadline_at') + ), + 'assignment_deadline_at': WorkerSlot._event_timestamp( + assignment.get('assignment_deadline_at') + or deadlines.get('assignment_deadline_at') + or reservation.get('remote_expires_at') + ), + 'attempt': max(1, int(reservation.get('attempts') or 1)), + } + + @staticmethod + def _runner_identity(value): + if value is None: + return None + if not isinstance(value, dict) or set(value) != { + 'host', 'payload', 'job_membership_verified', + } or type(value.get('job_membership_verified')) is not bool: + raise WorkerClientError('persisted runner identity shape is invalid') + normalized = {'job_membership_verified': value['job_membership_verified']} + for name in ('host', 'payload'): + identity = value.get(name) + if not isinstance(identity, dict) or set(identity) != { + 'pid', 'creation_time', 'executable', + } or type(identity.get('pid')) is not int or identity['pid'] <= 0 or not all( + isinstance(identity.get(field), str) and identity[field] + for field in ('creation_time', 'executable') + ): + raise WorkerClientError('persisted runner process identity is invalid') + normalized[name] = dict(identity) + return normalized + + @staticmethod + def _runner_record(value): + if value is None: + return None + fields = { + 'generation', 'input_sha256', 'root_name', 'input_ref', 'events_ref', + 'start_ref', 'terminal_ref', 'bundle_ref', 'scan_started_at', + 'scan_deadline_at', 'watchdog_deadline_at', + 'operation', 'timeout_phase', 'attempt', 'status', 'identity', 'last_event', + 'terminal_sha256', + } + if not isinstance(value, dict) or set(value) != fields: + raise WorkerClientError('persisted runner record shape is invalid') + generation = str(value.get('generation') or '') + input_sha256 = str(value.get('input_sha256') or '') + if not re.fullmatch(r'[a-f0-9]{32}', generation) or not DIGEST_RE.fullmatch(input_sha256): + raise WorkerClientError('persisted runner generation identity is invalid') + root_name = str(value.get('root_name') or '') + root_match = re.fullmatch( + r'worker-assignment-(0|[1-9][0-9]*)-([1-9][0-9]*)-([a-f0-9]{32})', + root_name, + ) + if root_match is None or root_match.group(3) != generation: + raise WorkerClientError('persisted runner root reference is invalid') + expected = { + 'input_ref': f'{root_name}/input.json', + 'events_ref': f'{root_name}/events.jsonl', + 'start_ref': f'{root_name}/start.json', + 'terminal_ref': f'{root_name}/terminal.json', + } + if any(value.get(name) != reference for name, reference in expected.items()): + raise WorkerClientError('persisted runner protocol reference is invalid') + bundle_ref = value.get('bundle_ref') + if bundle_ref is not None and ( + not isinstance(bundle_ref, str) + or not bundle_ref.startswith(f'{root_name}/bundle/ready/') + or '\\' in bundle_ref or '..' in bundle_ref.split('/') + ): + raise WorkerClientError('persisted runner bundle reference is invalid') + started = str(value.get('scan_started_at') or '') + deadline = str(value.get('scan_deadline_at') or '') + normalized_started = WorkerSlot._event_timestamp(started) + normalized_deadline = WorkerSlot._event_timestamp(deadline) + watchdog_deadline = str(value.get('watchdog_deadline_at') or '') + WorkerSlot._event_timestamp(watchdog_deadline) + if datetime.fromisoformat(normalized_deadline.replace('Z', '+00:00')) <= datetime.fromisoformat(normalized_started.replace('Z', '+00:00')): + raise WorkerClientError('persisted runner deadline ordering is invalid') + if value.get('status') not in { + 'created', 'running', 'stopping', 'exited', 'timed_out', 'fenced', + }: + raise WorkerClientError('persisted runner status is invalid') + operation = value.get('operation') + timeout_phase = value.get('timeout_phase') + if operation not in {'execute', 'timeout_bundle'}: + raise WorkerClientError('persisted runner operation is invalid') + if operation == 'execute' and timeout_phase is not None: + raise WorkerClientError('persisted scan runner timeout phase is invalid') + if operation == 'timeout_bundle': + try: + timeout_phase = WorkerPhase(timeout_phase).value + except ValueError as exc: + raise WorkerClientError('persisted timeout runner phase is invalid') from exc + if timeout_phase in { + WorkerPhase.IDLE.value, WorkerPhase.CLAIMING.value, + WorkerPhase.UPLOADING.value, WorkerPhase.AWAITING_RECEIPT.value, + WorkerPhase.BACKOFF.value, WorkerPhase.DRAINING.value, + WorkerPhase.STOPPED.value, + }: + raise WorkerClientError('persisted timeout runner phase is outside the scan stage') + if type(value.get('attempt')) is not int or not 1 <= value['attempt'] <= 3: + raise WorkerClientError('persisted runner attempt is invalid') + identity = WorkerSlot._runner_identity(value.get('identity')) + last_event = value.get('last_event') + if last_event is not None: + from worker_assignment_runner import validate_runner_event + try: + last_event = validate_runner_event( + last_event, generation=generation, input_sha256=input_sha256, + ) + except RunnerProtocolError as exc: + raise WorkerClientError('persisted runner event is invalid') from exc + terminal_sha256 = value.get('terminal_sha256') + if terminal_sha256 is not None and not DIGEST_RE.fullmatch(str(terminal_sha256)): + raise WorkerClientError('persisted runner terminal reference is invalid') + return { + **value, + 'scan_started_at': started, + 'scan_deadline_at': deadline, + 'watchdog_deadline_at': watchdog_deadline, + 'identity': identity, + 'last_event': last_event, + } + + @staticmethod + def _slot_state(value, *, slot_id=None): + if not isinstance(value, dict): + raise WorkerClientError('persisted worker slot state is invalid') + state = dict(value) + if 'schema' not in state: + phase = state.get('phase') + legacy = { + 'claiming': ({'phase', 'request_id'}, None), + 'assigned': ({'phase', 'assignment'}, 'runner'), + 'bundle_ready': ({'phase', 'assignment'}, 'runner'), + 'terminal_pending': ({'phase', 'assignment', 'terminal'}, 'runner'), + 'awaiting_resolution': ({'phase', 'assignment', 'stale'}, 'runner'), + } + expected, runner_field = legacy.get(phase, (None, None)) + if expected is None or not expected <= set(state) or set(state) - expected - {'bundle'}: + raise WorkerClientError('legacy worker slot state shape is invalid') + state['schema'] = SLOT_STATE_SCHEMA + if runner_field: + state[runner_field] = None + state['retained_work'] = [] + if phase == 'bundle_ready' and 'bundle' not in state: + state['bundle'] = {} + if state.get('schema') != SLOT_STATE_SCHEMA: + raise WorkerClientError('persisted worker slot schema is invalid') + phase = state.get('phase') + required = { + 'claiming': {'schema', 'phase', 'request_id'}, + 'assigned': {'schema', 'phase', 'assignment', 'runner', 'retained_work'}, + 'bundle_ready': {'schema', 'phase', 'assignment', 'runner', 'retained_work', 'bundle'}, + 'terminal_pending': {'schema', 'phase', 'assignment', 'runner', 'retained_work', 'terminal'}, + 'awaiting_resolution': {'schema', 'phase', 'assignment', 'runner', 'retained_work', 'stale'}, + }.get(phase) + optional = ( + {'bundle', 'terminal', 'transport_conflict'} + if phase == 'awaiting_resolution' + else {'bundle', 'transport_conflict'} + if phase == 'terminal_pending' + else {'transport_conflict'} if phase == 'bundle_ready' + else set() + ) + if required is None or not required <= set(state) or set(state) - required - optional: + raise WorkerClientError('persisted worker slot state shape is invalid') + if phase == 'claiming': + if not re.fullmatch(r'[a-f0-9]{32}', str(state.get('request_id') or '')): + raise WorkerClientError('persisted claim request identity is invalid') + return state + if not isinstance(state.get('assignment'), dict): + raise WorkerClientError('persisted worker assignment is invalid') + retained_work = state.get('retained_work') + if ( + not isinstance(retained_work, list) or len(retained_work) > 16 + or any( + not isinstance(item, str) + or re.fullmatch( + r'abandoned/worker-assignment-[0-9]+-[1-9][0-9]*-[a-f0-9]{32}', + item, + ) is None + for item in retained_work + ) + or len(set(retained_work)) != len(retained_work) + ): + raise WorkerClientError('persisted retained runner work is invalid') + state['runner'] = WorkerSlot._runner_record(state.get('runner')) + if state['runner'] is not None: + root_match = re.fullmatch( + r'worker-assignment-(0|[1-9][0-9]*)-([1-9][0-9]*)-([a-f0-9]{32})', + state['runner']['root_name'], + ) + reservation_id = int( + (state['assignment'].get('reservation') or {}).get('reservation_id') or 0 + ) + if ( + int(root_match.group(2)) != reservation_id + or (slot_id is not None and int(root_match.group(1)) != int(slot_id)) + ): + raise WorkerClientError('persisted runner root conflicts with slot authority') + if 'bundle' in state and not isinstance(state.get('bundle'), dict): + raise WorkerClientError('persisted worker bundle state is invalid') + if 'terminal' in state and ( + not isinstance(state.get('terminal'), dict) + or set(state['terminal']) not in ( + {'failure_code', 'detail'}, + {'failure_code', 'detail', 'diagnostics'}, + ) + or any(type(state['terminal'].get(name)) is not str for name in ('failure_code', 'detail')) + ): + raise WorkerClientError('persisted terminal report is invalid') + if 'terminal' in state and 'diagnostics' in state['terminal']: + diagnostics = state['terminal']['diagnostics'] + if not isinstance(diagnostics, list): + raise WorkerClientError('persisted terminal diagnostics are invalid') + try: + normalized = [ + json.loads(encode_diagnostic_envelope( + decode_diagnostic_envelope(json.dumps( + diagnostic, ensure_ascii=True, sort_keys=True, + separators=(',', ':'), + ).encode('ascii')) + ).decode('ascii')) + for diagnostic in diagnostics + ] + except (TypeError, ValueError, UnicodeError) as exc: + raise WorkerClientError('persisted terminal diagnostics are invalid') from exc + if normalized != diagnostics: + raise WorkerClientError('persisted terminal diagnostics are not canonical') + if 'stale' in state and ( + not isinstance(state.get('stale'), dict) + or set(state['stale']) != {'status_code', 'code'} + ): + raise WorkerClientError('persisted stale reconciliation is invalid') + if 'transport_conflict' in state: + conflict = state['transport_conflict'] + if ( + not isinstance(conflict, dict) + or set(conflict) != {'status_code', 'code', 'attempts'} + or conflict.get('status_code') != 409 + or not re.fullmatch(r'[a-z0-9_]{1,64}', str(conflict.get('code') or '')) + or type(conflict.get('attempts')) is not int + or not 1 <= conflict['attempts'] <= 2 + ): + raise WorkerClientError('persisted transport conflict is invalid') + return state + + def _emit( + self, phase, state=None, progress=None, *, timestamp=None, + phase_started_at=None, + ): + phase = WorkerPhase(phase) + details = self._assignment_details(state) + measured = {'attempt': details.pop('attempt')} + measured.update(dict(progress or {})) + if state is not None and state.get('retained_work'): + measured['retained_work'] = list(state['retained_work']) + if self.event_callback is not None: + self.event_callback({ + 'slot_id': self.slot_id, + 'phase': phase.value, + **details, + 'progress': measured, + 'timestamp': timestamp, + 'phase_started_at': phase_started_at, + }) + self._event_phase = phase + + def _resume_events(self, state): + if self._event_phase is not None: + return + phase = str((state or {}).get('phase') or '') + if not phase: + self._emit(WorkerPhase.IDLE, progress={'reason': 'startup'}) + return + if phase == 'claiming': + self._emit(WorkerPhase.CLAIMING, state, {'recovered': True}) + return + runner = dict((state or {}).get('runner') or {}) + if phase == 'assigned' and runner: + # Recovery decides whether this generation completed or must be + # fenced before publishing the first event for the new instance. + return + if phase == 'assigned': + self._emit(WorkerPhase.ASSIGNED, state, {'recovered': True}) + return + if phase == 'bundle_ready': + self._emit(WorkerPhase.BACKOFF, state, { + 'recovered': True, 'reason': 'bundle_upload_recovery', + }) + return + self._emit(WorkerPhase.BACKOFF, state, { + 'recovered': True, 'reason': 'terminal_reconciliation_recovery', + }) + + def _finish_idle(self, state, reason): + self._emit(WorkerPhase.IDLE, progress={'reason': reason}) + + def retire(self, reason): + if self._event_phase != WorkerPhase.DRAINING: + self._emit(WorkerPhase.DRAINING, progress={'reason': str(reason)}) + self._emit(WorkerPhase.STOPPED, progress={'reason': str(reason)}) + + def _notify_terminal(self, state, outcome, receipt): + if self.terminal_callback is None: + return + details = self._assignment_details(state) + reservation = dict((state.get('assignment') or {}).get('reservation') or {}) + reservation_id = int(details['reservation_id'] or 0) + receipt_id = str((receipt or {}).get('receipt_id') or '') + history_id = receipt_id or hashlib.sha256( + f'{reservation_id}:{outcome}:{(receipt or {}).get("code", "")}'.encode('utf-8') + ).hexdigest() + completed_at = datetime.now(timezone.utc) + started_at = self._event_timestamp( + reservation.get('remote_issued_at') or reservation.get('issued_at') + ) + duration_seconds = None + if started_at: + try: + started = datetime.fromisoformat(str(started_at).replace('Z', '+00:00')) + if started.tzinfo is None: + started = started.replace(tzinfo=timezone.utc) + duration_seconds = max(0.0, (completed_at - started.astimezone(timezone.utc)).total_seconds()) + except ValueError: + duration_seconds = None + self.terminal_callback({ + 'history_id': history_id, + 'slot_id': self.slot_id, + 'reservation_id': reservation_id, + 'source': details['source'], + 'outcome': str(outcome), + 'receipt': dict(receipt or {}), + 'started_at': started_at, + 'completed_at': completed_at.isoformat().replace('+00:00', 'Z'), + 'duration_seconds': duration_seconds, + 'first_sequence': None, + 'diagnostics': list(self._diagnostics), + }) + + def _clear_assignment_diagnostics(self): + self._diagnostics.clear() + self._transport_diagnostics.clear() + + @staticmethod + def _captured_bytes(error, *names): + for name in names: + value = getattr(error, name, None) + if value is None: + continue + if isinstance(value, bytes): + return value + if isinstance(value, str): + return value.encode('utf-8') + return None + + @staticmethod + def _utf8_prefix(value, maximum): + payload = str(value or '').encode('utf-8') + if len(payload) <= maximum: + return payload.decode('utf-8') + return payload[:maximum].decode('utf-8', errors='ignore') + + @staticmethod + def _allocate_bytes(desired, budget): + desired = [max(0, int(value)) for value in desired] + budget = min(sum(desired), max(0, int(budget))) + if not desired or not budget or not sum(desired): + return [0 for _value in desired] + total = sum(desired) + values = [min(value, (budget * value) // total) for value in desired] + remaining = budget - sum(values) + order = sorted( + range(len(desired)), + key=lambda index: ( + -((budget * desired[index]) % total), index, + ), + ) + for index in order: + if remaining <= 0: + break + if values[index] < desired[index]: + values[index] += 1 + remaining -= 1 + return values + + @classmethod + def _diagnostic_material_limits(cls, spec, budget): + body = spec.get('body') + stdout = spec.get('stdout') + stderr = spec.get('stderr') + log_desired = cls._allocate_bytes([ + min(len(stdout), MAX_DIAGNOSTIC_LOG_BYTES) if stdout is not None else 0, + min(len(stderr), MAX_DIAGNOSTIC_LOG_BYTES) if stderr is not None else 0, + ], MAX_DIAGNOSTIC_LOG_BYTES) + desired = [ + min(len(body), MAX_DIAGNOSTIC_BODY_BYTES) if body is not None else 0, + *log_desired, + ] + return cls._allocate_bytes(desired, budget) + + @classmethod + def _build_diagnostic_from_spec(cls, spec, material_budget): + body_limit, stdout_limit, stderr_limit = cls._diagnostic_material_limits( + spec, material_budget, + ) + process = None + if ( + spec.get('stdout') is not None + or spec.get('stderr') is not None + or spec.get('return_code') is not None + ): + process = DiagnosticProcessContext( + name=spec['process_name'], + exit_code=spec['return_code'], + signal=None, + timed_out=spec['timed_out'], + stdout=( + make_log_material(spec['stdout'], maximum=stdout_limit) + if spec.get('stdout') is not None else None + ), + stderr=( + make_log_material(spec['stderr'], maximum=stderr_limit) + if spec.get('stderr') is not None else None + ), + ) + http = None + if spec.get('status_code') is not None: + http = DiagnosticHTTPContext( + operation='worker-api', + status_code=spec['status_code'], + content_type=spec.get('content_type'), + request_id=spec.get('request_id'), + body=( + make_diagnostic_material(spec['body'], maximum=body_limit) + if spec.get('body') is not None else None + ), + ) + return build_diagnostic_envelope( + occurrence_id=spec['occurrence_id'], + reservation_id=spec['reservation_id'], + scan_event_id=spec['scan_event_id'], + slot_id=spec['slot_id'], + source=spec['source'], + phase=spec['phase'], + kind=spec['kind'], + category=spec['category'], + code=spec['code'], + summary=spec['summary'], + retryable=False, + attempt=spec['attempt'], + assignment_outcome=AssignmentOutcome.PREBUNDLE_FAILED, + scan_outcome=ScanOutcome.UNAVAILABLE, + occurred_at=spec['timestamp'], + captured_at=spec['timestamp'], + http=http, + process=process, + exception=DiagnosticExceptionContext( + type=spec['exception_type'], + message=spec['exception_message'], + fingerprint=spec['fingerprint'], + ), + ) + + def _fitted_terminal_report(self, failure_code, detail): + base = { + 'failure_code': str(failure_code), + 'detail': str(detail or '')[:1000], + } + if not self._transport_diagnostics: + return base + desired = [ + sum(self._diagnostic_material_limits( + spec, MAX_DIAGNOSTIC_BODY_BYTES + MAX_DIAGNOSTIC_LOG_BYTES, + )) + for spec in self._transport_diagnostics + ] + + def candidate(total_budget): + budgets = self._allocate_bytes(desired, total_budget) + value = dict(base) + try: + value['diagnostics'] = [ + json.loads(encode_diagnostic_envelope( + self._build_diagnostic_from_spec(spec, budget) + ).decode('ascii')) + for spec, budget in zip(self._transport_diagnostics, budgets) + ] + except ValueError: + return value, TERMINAL_REPORT_MAX_BYTES + 1 + payload = json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + return value, len(payload) + + low, high = 0, sum(desired) + fitted = None + while low <= high: + middle = (low + high) // 2 + value, size = candidate(middle) + if size <= TERMINAL_REPORT_MAX_BYTES: + fitted = value + low = middle + 1 + else: + high = middle - 1 + if fitted is None: + raise WorkerClientError('canonical terminal report cannot fit its byte bound') + return fitted + + def _archive_fitted_terminal_diagnostics(self, report): + if self.diagnostic_callback is None: + return + values = report.get('diagnostics') or [] + if len(values) != len(self._transport_diagnostics): + raise WorkerClientError( + 'fitted terminal diagnostics lost their local evidence identity' + ) + for spec, value in zip(self._transport_diagnostics, values): + envelope = decode_diagnostic_envelope(json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')) + try: + reference = self.diagnostic_callback(envelope, { + 'body': spec.get('body'), + 'stdout': spec.get('stdout'), + 'stderr': spec.get('stderr'), + }) + except Exception: + continue + if reference is not None: + self._diagnostics.append(reference) + + def _capture_failure_diagnostic( + self, state, error, code, category, *, operation='assignment_execute', + ): + details = self._assignment_details(state) + assignment = dict(state.get('assignment') or {}) + reservation = dict(assignment.get('reservation') or {}) + body = self._captured_bytes(error, 'body', 'response_body') + stdout = self._captured_bytes(error, 'stdout', 'output') + stderr = self._captured_bytes(error, 'stderr') + return_code = getattr(error, 'returncode', None) + status_code = getattr(error, 'status_code', None) + now = datetime.now(timezone.utc).isoformat().replace('+00:00', 'Z') + if isinstance(error, WorkerContractError): + field = str(error.category) + code = f'worker_contract_invalid.{field}' + category = DiagnosticCategory.PROTOCOL + full_message = f'worker contract field is invalid: {field}' + else: + full_message = str(error) + message = self._utf8_prefix(full_message, 1000) + runner = dict(state.get('runner') or {}) + exception_detail = { + 'message': full_message, + 'operation': str(operation), + 'phase': (self._event_phase or WorkerPhase.ASSIGNED).value, + 'runner_generation': runner.get('generation'), + 'errno': getattr(error, 'errno', None), + 'winerror': getattr(error, 'winerror', None), + 'filename': getattr(error, 'filename', None), + 'filename2': getattr(error, 'filename2', None), + 'traceback': ''.join(traceback.format_exception( + type(error), error, error.__traceback__, + )), + } + exception_message = self._utf8_prefix(json.dumps( + exception_detail, ensure_ascii=False, sort_keys=True, + separators=(',', ':'), + ), 4000) + fingerprint = hashlib.sha256( + f'{type(error).__module__}.{type(error).__qualname__}:{full_message}'.encode('utf-8') + ).hexdigest() + spec = { + 'occurrence_id': secrets.token_hex(16), + 'reservation_id': int(details['reservation_id'] or 0), + 'scan_event_id': str(reservation.get('scan_event_id') or '') or None, + 'slot_id': self.slot_id, + 'source': str(details['source'] or reservation.get('platform') or 'unknown'), + 'phase': self._event_phase or WorkerPhase.ASSIGNED, + 'kind': ( + DiagnosticKind.SCANNER_PROCESS + if stdout is not None or stderr is not None or type(return_code) is int + else DiagnosticKind.PROVIDER_HTTP + if type(status_code) is int else DiagnosticKind.EXCEPTION + ), + 'category': category, + 'code': str(code), + 'summary': message or type(error).__name__, + 'attempt': details['attempt'], + 'timestamp': now, + 'exception_type': f'{type(error).__module__}.{type(error).__qualname__}', + 'exception_message': exception_message, + 'fingerprint': fingerprint, + 'process_name': str(getattr(error, 'process_name', None) or 'worker-operation'), + 'return_code': int(return_code) if type(return_code) is int else None, + 'timed_out': bool(getattr(error, 'timed_out', False)), + 'status_code': int(status_code) if type(status_code) is int else None, + 'content_type': getattr(error, 'content_type', None), + 'request_id': getattr(error, 'request_id', None), + 'body': body, + 'stdout': stdout, + 'stderr': stderr, + } + self._transport_diagnostics.append(spec) + return None + + def _save(self, value): + normalized = self._slot_state(value, slot_id=self.slot_id) + atomic_write_private_json( + self.state_path, + normalized, + max_bytes=MAX_PENDING_BYTES, + ) + return normalized + + def _load(self): + if not os.path.exists(self.state_path): + return None + loaded = read_private_json(self.state_path, max_bytes=MAX_PENDING_BYTES) + normalized = self._slot_state(loaded, slot_id=self.slot_id) + if normalized != loaded: + atomic_write_private_json( + self.state_path, normalized, max_bytes=MAX_PENDING_BYTES, + ) + return normalized + + def _remove_state(self): + if os.path.lexists(self.state_path): + reject_reparse_components(self.state_path) + os.remove(self.state_path) + + def _resolved(self, state, status): + if not isinstance(status, dict) or not status.get('resolution'): + return False + assignment = state.get('assignment') or {} + reservation = assignment.get('reservation') or {} + reservation_id = int(reservation.get('reservation_id') or 0) + bundle_id = str(reservation.get('bundle_id') or '') + scan_event_id = str(reservation.get('scan_event_id') or '') + resolution = str(status.get('resolution') or '') + if ( + int(status.get('reservation_id') or 0) != reservation_id + or str(status.get('bundle_id') or '') != bundle_id + or str(status.get('scan_event_id') or '') != scan_event_id + or resolution not in { + 'bundle_accepted', 'prebundle_report', 'expired', + } + or not DIGEST_RE.fullmatch(str(status.get('receipt_id') or '')) + ): + raise WorkerClientError('worker API receipt identity is invalid') + phase = str(state.get('phase') or '') + if resolution == 'bundle_accepted': + payload_sha256 = str(status.get('payload_sha256') or '') + if phase not in ('assigned', 'bundle_ready', 'awaiting_resolution') or not DIGEST_RE.fullmatch( + payload_sha256 + ): + raise WorkerClientError('worker bundle receipt does not match pending state') + path = bundle_ready_path(self.bundle_root, bundle_id) + if os.path.lexists(path): + reject_reparse_components(path) + if not os.path.isfile(path) or os.path.islink(path): + raise WorkerClientError('resolved bundle path is not a regular file') + metadata = ResultBundleReader( + path, max_event_bytes=int(reservation.get('declared_bundle_bytes') or 0), + ).validate() + if metadata.header != BundleReservation.from_mapping(reservation).header(): + raise WorkerClientError('resolved bundle identity conflicts with pending state') + digest = hashlib.sha256() + with open(path, 'rb', buffering=0) as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b''): + digest.update(chunk) + if digest.hexdigest() != payload_sha256: + raise WorkerClientError('worker bundle receipt payload digest conflicts') + elif resolution == 'prebundle_report': + terminal = dict(state.get('terminal') or {}) + if phase != 'terminal_pending' or set(terminal) not in ( + {'failure_code', 'detail'}, + {'failure_code', 'detail', 'diagnostics'}, + ): + raise WorkerClientError('worker terminal receipt does not match pending state') + encoded = json.dumps( + terminal, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('utf-8') + if ( + str(status.get('failure_code') or '') != terminal['failure_code'] + or str(status.get('payload_sha256') or '') != hashlib.sha256(encoded).hexdigest() + ): + raise WorkerClientError('worker terminal receipt payload conflicts') + if bundle_id: + path = bundle_ready_path(self.bundle_root, bundle_id) + if os.path.lexists(path): + reject_reparse_components(path) + if not os.path.isfile(path) or os.path.islink(path): + raise WorkerClientError('resolved bundle path is not a regular file') + os.remove(path) + self._notify_terminal(state, resolution, status) + self._clear_assignment_diagnostics() + self._remove_state() + self._finish_idle(state, 'terminal_reconciled') + return True + + def _cleanup_recorded_stale(self, state): + if not os.path.exists(self.stale_path): + return False + record = read_private_json(self.stale_path, max_bytes=MAX_PENDING_BYTES) + assignment = dict(state.get('assignment') or {}) + reservation = dict(assignment.get('reservation') or {}) + if not isinstance(record, dict) or set(record) != { + 'schema', 'outcome', 'reservation_id', 'bundle_id', 'scan_event_id', + 'status_code', 'code', 'recorded_at', + } or ( + record.get('schema') != 1 + or record.get('outcome') != 'discarded_stale' + or int(record.get('reservation_id') or 0) != int(reservation.get('reservation_id') or 0) + or str(record.get('bundle_id') or '') != str(reservation.get('bundle_id') or '') + or str(record.get('scan_event_id') or '') != str(reservation.get('scan_event_id') or '') + ): + return False + bundle_id = str(record['bundle_id']) + if bundle_id: + path = bundle_ready_path(self.bundle_root, bundle_id) + if os.path.lexists(path): + reject_reparse_components(path) + if not os.path.isfile(path) or os.path.islink(path): + raise WorkerClientError('stale bundle path is not a regular file') + os.remove(path) + self._notify_terminal(state, 'discarded_stale', record) + self._clear_assignment_diagnostics() + self._remove_state() + self._finish_idle(state, 'stale_reconciled') + return True + + def _mark_stale(self, state, error): + assignment = dict(state.get('assignment') or {}) + reservation = dict(assignment.get('reservation') or {}) + code = str(error.code or '') + if not re.fullmatch(r'[a-z0-9_]{1,64}', code): + code = 'request_rejected' + record = { + 'schema': 1, + 'outcome': 'discarded_stale', + 'reservation_id': int(reservation.get('reservation_id') or 0), + 'bundle_id': str(reservation.get('bundle_id') or ''), + 'scan_event_id': str(reservation.get('scan_event_id') or ''), + 'status_code': int(error.status_code), + 'code': code, + 'recorded_at': datetime.now(timezone.utc).isoformat(), + } + if record['reservation_id'] <= 0 or not record['bundle_id'] or not record['scan_event_id']: + raise WorkerClientError('stale assignment identity is invalid') + atomic_write_private_json( + self.stale_path, record, max_bytes=MAX_PENDING_BYTES, + ) + if not self._cleanup_recorded_stale(state): + raise WorkerClientError('stale outcome could not be reconciled') + return True + + def _reconcile_http_failure(self, state, error): + if error.status_code not in (404, 409, 410): + raise error + try: + status = self.api.status( + int((state.get('assignment') or {}).get('reservation', {}).get('reservation_id') or 0) + ) + except WorkerHTTPError as status_error: + if status_error.status_code in (404, 409, 410): + return self._mark_stale(state, status_error) + raise + if self._resolved(state, status): + return True + if error.status_code == 409 and str(status.get('state') or '') == 'scanning': + previous = dict(state.get('transport_conflict') or {}) + attempts = ( + int(previous.get('attempts') or 0) + 1 + if previous.get('code') == str(error.code or '') else 1 + ) + if attempts == 1: + state['transport_conflict'] = { + 'status_code': 409, + 'code': str(error.code or 'request_rejected'), + 'attempts': 1, + } + self._save(state) + return False + state.pop('transport_conflict', None) + if state.get('phase') == 'bundle_ready': + self._capture_failure_diagnostic( + state, error, 'client_result_conflict', + DiagnosticCategory.PROTOCOL, + operation='bundle_upload_reconciliation', + ) + reservation_id = int( + (state.get('assignment') or {}).get( + 'reservation', {}, + ).get('reservation_id') or 0 + ) + return self._terminal( + state, reservation_id, 'client_process_failed', + CLIENT_PROCESS_FAILURE_DETAIL, + ) + state['phase'] = 'awaiting_resolution' + state['stale'] = { + 'status_code': error.status_code, + 'code': str(error.code or 'request_rejected'), + } + self._save(state) + return False + if error.status_code == 410: + state['phase'] = 'awaiting_resolution' + state['stale'] = {'status_code': error.status_code, 'code': error.code} + self._save(state) + return False + return self._mark_stale(state, error) + + def _adopt_ready_bundle(self, state, *, persist=True): + assignment = dict(state['assignment']) + reservation = BundleReservation.from_mapping(assignment['reservation']) + path = bundle_ready_path(self.bundle_root, reservation.bundle_id) + if not os.path.lexists(path): + return None + reject_reparse_components(path) + if not os.path.isfile(path) or os.path.islink(path): + raise WorkerClientError('pending result bundle is not a regular file') + metadata = ResultBundleReader( + path, max_event_bytes=reservation.declared_bytes, + ).validate() + if metadata.header != reservation.header(): + raise WorkerClientError('pending result bundle identity conflicts with its assignment') + runner = state.get('runner') + if runner is not None: + terminal = self._load_runner_terminal(state) + if ( + terminal is None or terminal['decision'] != 'completed' + or terminal['outcome']['status'] != 'succeeded' + ): + closed = self._fence_recovered_runner( + state, 'ready_bundle_without_completion', + ) + if closed == 'unknown': + raise RunnerContainmentPending( + 'ready bundle runner containment remains live', + ) + raise RunnerProtocolError( + 'canonical ready bundle lacks a valid completed generation', + ) + digest = hashlib.sha256() + with open(path, 'rb', buffering=0) as handle: + for block in iter(lambda: handle.read(1024 * 1024), b''): + digest.update(block) + if digest.hexdigest() != terminal['outcome']['bundle']['payload_sha256']: + raise RunnerProtocolError( + 'canonical ready bundle conflicts with runner terminal payload', + ) + if self._stop_persisted_runner(runner) != 'dead': + raise RunnerContainmentPending( + 'completed ready-bundle runner containment remains live', + ) + self._retain_runner_work(state, runner) + state['runner'] = None + state['phase'] = 'bundle_ready' + state['bundle'] = metadata.as_dict() + runner = state.get('runner') + if runner is not None: + reference = f"abandoned/{runner['root_name']}" + destination = os.path.join(self.work_root, *reference.split('/')) + if os.path.isdir(destination) and not os.path.exists( + self._runner_paths(runner)['root'] + ): + if reference not in state.setdefault('retained_work', []): + state['retained_work'].append(reference) + state['runner'] = None + if persist: + self._save(state) + return state + + @staticmethod + def _deadline_active(value): + text = str(value or '').strip() + if not text: + return False + try: + parsed = datetime.fromisoformat(text.replace('Z', '+00:00')) + except ValueError as exc: + raise WorkerClientError('worker assignment deadline is invalid') from exc + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.astimezone(timezone.utc) > datetime.now(timezone.utc) + + @staticmethod + def _identity_value(value): + return { + name: value[name] + for name in ('pid', 'creation_time', 'executable') + } + + def _new_runner( + self, state, *, scan_started_at=None, scan_deadline_at=None, + watchdog_deadline_at=None, operation='execute', timeout_phase=None, + attempt=1, + ): + assignment = dict(state['assignment']) + if set(self.package_runtime) != { + 'code_manifest', 'code_manifest_sha256', 'trufflehog_path', + 'git_path', 'detector_policy_path', 'capabilities', 'bootstrap_path', + }: + raise WorkerClientError('verified worker package runtime is unavailable') + packaged_capabilities = self.package_runtime['capabilities'] + if not isinstance(packaged_capabilities, tuple) or not packaged_capabilities: + raise WorkerClientError('verified worker package capabilities are unavailable') + try: + validate_protocol2_remote_assignment( + assignment, self.compatibility, packaged_capabilities, + ) + except (ScanExecutionError, TypeError, ValueError) as exc: + raise WorkerAssignmentCompatibilityError( + 'worker assignment is incompatible with this package' + ) from exc + reservation = dict(assignment['reservation']) + deadlines = dict(assignment['deadlines']) + if scan_started_at is None or scan_deadline_at is None: + started = datetime.now(timezone.utc) + assignment_deadline = self._event_timestamp( + deadlines.get('assignment_deadline_at') or reservation.get('remote_expires_at') + ) + assignment_deadline_value = datetime.fromisoformat( + assignment_deadline.replace('Z', '+00:00'), + ) + scan_deadline_value = min( + started + timedelta(seconds=int(deadlines['target_scan_timeout_seconds'])), + assignment_deadline_value, + ) + if scan_deadline_value <= started: + raise WorkerClientError('worker assignment deadline has passed') + scan_started_at = started.isoformat(timespec='milliseconds').replace('+00:00', 'Z') + scan_deadline_at = scan_deadline_value.isoformat(timespec='milliseconds').replace('+00:00', 'Z') + else: + self._event_timestamp(scan_started_at) + self._event_timestamp(scan_deadline_at) + scan_started_at = str(scan_started_at) + scan_deadline_at = str(scan_deadline_at) + watchdog_deadline_at = str(watchdog_deadline_at or scan_deadline_at) + self._event_timestamp(watchdog_deadline_at) + generation = secrets.token_hex(16) + root_name = runner_root_name( + self.slot_id, reservation['reservation_id'], generation, + ) + runner_input = build_runner_input( + assignment, + generation=generation, + slot_id=self.slot_id, + scan_started_at=scan_started_at, + scan_deadline_at=scan_deadline_at, + watchdog_deadline_at=watchdog_deadline_at, + operation=operation, + timeout_phase=timeout_phase, + ) + created = threading.Event() + cancelled = threading.Event() + creation = {} + + def create_protocol_root(): + try: + creation['value'] = create_runner_root( + self.work_root, root_name, runner_input, + ) + if cancelled.is_set(): + cleanup_runner_root(self.work_root, root_name) + except BaseException as exc: + creation['error'] = exc + finally: + created.set() + + threading.Thread( + target=create_protocol_root, + name=f'worker-runner-protocol-{self.slot_id}', daemon=True, + ).start() + watchdog_deadline = datetime.fromisoformat( + watchdog_deadline_at.replace('Z', '+00:00'), + ) + remaining = max( + 0.0, (watchdog_deadline - datetime.now(timezone.utc)).total_seconds(), + ) + if not created.wait(remaining): + cancelled.set() + raise RunnerStageTimeout( + WorkerPhase.PREPARING, scan_started_at, scan_deadline_at, + 'runner protocol initialization exceeded its absolute deadline', + ) + if 'error' in creation: + raise creation['error'] + _, input_sha256 = creation['value'] + state['runner'] = { + 'generation': generation, + 'input_sha256': input_sha256, + 'root_name': root_name, + 'input_ref': f'{root_name}/input.json', + 'events_ref': f'{root_name}/events.jsonl', + 'start_ref': f'{root_name}/start.json', + 'terminal_ref': f'{root_name}/terminal.json', + 'bundle_ref': None, + 'scan_started_at': scan_started_at, + 'scan_deadline_at': scan_deadline_at, + 'watchdog_deadline_at': watchdog_deadline_at, + 'operation': operation, + 'timeout_phase': timeout_phase, + 'attempt': int(attempt), + 'status': 'created', + 'identity': None, + 'last_event': None, + 'terminal_sha256': None, + } + self._save(state) + return state['runner'] + + def _runner_paths(self, runner): + return runner_paths(self.work_root, runner['root_name']) + + def _drain_runner_events(self, state, *, emit=True): + runner = state['runner'] + paths = self._runner_paths(runner) + after = int((runner.get('last_event') or {}).get('sequence') or 0) + events = read_runner_events( + paths['events'], generation=runner['generation'], + input_sha256=runner['input_sha256'], operation=runner['operation'], + after_sequence=after, + ) + for event in events: + if emit: + progress = dict(event['progress']) + progress.update({ + 'runner_generation': runner['generation'], + 'runner_sequence': event['sequence'], + }) + self._emit( + event['phase'], state, progress, + timestamp=event['timestamp'], + phase_started_at=event['phase_started_at'], + ) + runner['last_event'] = event + self._save(state) + return events + + @staticmethod + def _stop_exact_identity(identity, timeout): + state = exact_process_identity_state( + identity['pid'], identity['creation_time'], identity['executable'], + ) + if state in {'dead', 'reused'}: + return True + if state != 'alive': + return False + try: + with verify_retained_process( + identity['pid'], identity['creation_time'], identity['executable'], + terminate=True, + ) as process: + process.terminate() + return process.wait(timeout) + except OSError: + return exact_process_identity_state( + identity['pid'], identity['creation_time'], identity['executable'], + ) in {'dead', 'reused'} + + def _stop_persisted_runner(self, runner): + identity = runner.get('identity') + if identity is None: + return 'dead' + host_dead = self._stop_exact_identity( + identity['host'], RUNNER_STOP_TIMEOUT_SECONDS / 2, + ) + payload_dead = self._stop_exact_identity( + identity['payload'], RUNNER_STOP_TIMEOUT_SECONDS / 2, + ) + return 'dead' if host_dead and payload_dead else 'unknown' + + def _fence_recovered_runner(self, state, reason): + runner = state['runner'] + paths = self._runner_paths(runner) + try: + terminal, _won = publish_generation_terminal( + paths['terminal'], generation=runner['generation'], + input_sha256=runner['input_sha256'], decision='fenced', + reason=str(reason), + ) + if terminal['decision'] == 'completed': + return 'completed' + except RunnerProtocolError: + # A malformed complete terminal record is itself a closed but + # unusable generation. Exact containment still has to be stopped. + pass + runner['status'] = 'fenced' + self._save(state) + return self._stop_persisted_runner(runner) + + def _runner_command(self, root): + bootstrap = os.path.abspath(self.package_runtime['bootstrap_path']) + if not os.path.isfile(bootstrap) or os.path.islink(bootstrap): + raise WorkerClientError('verified worker bootstrap is unavailable') + return [ + sys._base_executable or sys.executable, + '-I', '-S', '-B', bootstrap, '--', + '_assignment_runner', '--root', root, + ] + + def _load_runner_terminal(self, state): + runner = state['runner'] + paths = self._runner_paths(runner) + if not os.path.exists(paths['terminal']): + return None + runner_input, input_digest = load_runner_input(paths['input']) + if input_digest != runner['input_sha256']: + raise RunnerProtocolError('runner input hash conflicts with slot authority') + terminal, digest = load_generation_terminal( + paths['terminal'], generation=runner['generation'], + input_sha256=runner['input_sha256'], + ) + events = read_runner_events( + paths['events'], generation=runner['generation'], + input_sha256=runner['input_sha256'], operation=runner['operation'], + ) + terminal = validate_terminal_against_journal( + terminal, events, runner_input, + ) + if terminal['decision'] != 'completed': + runner['terminal_sha256'] = digest + runner['status'] = ( + 'timed_out' if terminal['decision'] == 'timed_out' else 'fenced' + ) + self._save(state) + return terminal + if runner['status'] in {'stopping', 'timed_out', 'fenced'}: + raise RunnerProtocolError('completed output belongs to a closed runner generation') + deadline_name = ( + 'scan_deadline_at' if runner['operation'] == 'execute' + else 'watchdog_deadline_at' + ) + decided = datetime.fromisoformat( + terminal['decided_at'].replace('Z', '+00:00'), + ) + deadline = datetime.fromisoformat( + runner[deadline_name].replace('Z', '+00:00'), + ) + if decided > deadline: + raise RunnerProtocolError('runner completed after its absolute deadline') + outcome = terminal['outcome'] + assignment = state['assignment'] + reservation = assignment['reservation'] + identity = outcome['identity'] + if identity != { + 'slot_id': self.slot_id, + 'reservation_id': int(reservation['reservation_id']), + 'bundle_id': str(reservation['bundle_id']), + 'scan_event_id': str(reservation['scan_event_id']), + 'execution_snapshot_sha256': str(assignment['execution_snapshot_sha256']), + }: + raise RunnerProtocolError('runner outcome identity conflicts with slot authority') + self._drain_runner_events(state) + last_sequence = int((runner.get('last_event') or {}).get('sequence') or 0) + if outcome['last_event_sequence'] != last_sequence: + raise RunnerProtocolError('runner outcome event tail is incomplete or inconsistent') + runner['terminal_sha256'] = digest + runner['status'] = 'exited' + if outcome['bundle'] is not None: + runner['bundle_ref'] = ( + f"{runner['root_name']}/bundle/" + f"{outcome['bundle']['ready_relative_path']}" + ) + self._save(state) + return terminal + + def _adopt_runner_outcome(self, state, outcome): + if outcome['status'] != 'succeeded': + self._retain_runner_work(state, state['runner']) + state['runner'] = None + self._save(state) + error = outcome['error'] + raise RunnerProtocolError( + f"assignment runner failed in {outcome.get('final_phase') or 'startup'}: " + f"{error['code']}" + ) + metadata = adopt_runner_bundle( + self.work_root, state['runner']['root_name'], outcome, + state['assignment'], self.bundle_root, + ) + state['phase'] = 'bundle_ready' + state['bundle'] = { + **outcome['bundle']['commit'], + 'canonical_scan_event_hash': metadata.scan_event_hash, + } + self._retain_runner_work(state, state['runner']) + state['runner'] = None + self._save(state) + return state + + def _retain_runner_work(self, state, runner): + try: + reference = transfer_runner_to_janitor( + self.work_root, runner['root_name'], runner['generation'], + ) + except (OSError, RunnerProtocolError) as exc: + raise RunnerContainmentPending( + 'runner work has not reached durable janitor ownership', + ) from exc + retained = state.setdefault('retained_work', []) + if reference not in retained: + retained.append(reference) + return reference + + def _retry_closed_runner(self, state, reason): + runner = state['runner'] + if runner['operation'] == 'timeout_bundle': + if runner['attempt'] >= 3: + self._retain_runner_work(state, runner) + state['runner'] = None + self._save(state) + raise RunnerProtocolError( + 'timeout bundle runner exhausted its bounded generation retries', + ) + return self._timeout_bundle(state) + remaining = ( + datetime.fromisoformat( + runner['scan_deadline_at'].replace('Z', '+00:00'), + ) - datetime.now(timezone.utc) + ).total_seconds() + if remaining < 1: + runner['status'] = 'timed_out' + self._save(state) + return self._timeout_bundle(state) + self._retain_runner_work(state, runner) + if runner['attempt'] >= 3: + state['runner'] = None + self._save(state) + raise RunnerProtocolError('assignment runner exhausted its bounded generation retries') + scan_started_at = runner['scan_started_at'] + scan_deadline_at = runner['scan_deadline_at'] + attempt = runner['attempt'] + 1 + if self._event_phase is not None: + self._emit(self._event_phase, state, { + 'runner_retry_reason': str(reason), + 'next_runner_attempt': attempt, + }) + state['runner'] = None + self._save(state) + try: + self._new_runner( + state, + scan_started_at=scan_started_at, + scan_deadline_at=scan_deadline_at, + watchdog_deadline_at=scan_deadline_at, + operation='execute', attempt=attempt, + ) + except RunnerStageTimeout as exc: + return self._start_timeout_bundle( + state, + scan_started_at=exc.scan_started_at, + scan_deadline_at=exc.scan_deadline_at, + final_phase=exc.phase, + ) + return self._launch_runner(state) + + def _timeout_bundle(self, state): + runner = state['runner'] + final_phase = str( + runner.get('timeout_phase') + or (runner.get('last_event') or {}).get('phase') + or 'preparing' + ) + scan_started_at = runner['scan_started_at'] + scan_deadline_at = runner['scan_deadline_at'] + timeout_attempt = ( + runner['attempt'] + 1 + if runner['operation'] == 'timeout_bundle' else 1 + ) + self._retain_runner_work(state, runner) + state['runner'] = None + self._save(state) + return self._start_timeout_bundle( + state, + scan_started_at=scan_started_at, + scan_deadline_at=scan_deadline_at, + final_phase=final_phase, + attempt=timeout_attempt, + ) + + def _start_timeout_bundle( + self, state, *, scan_started_at, scan_deadline_at, + final_phase, attempt=1, + ): + assignment = state['assignment'] + assignment_deadline = datetime.fromisoformat( + self._event_timestamp( + assignment['deadlines']['assignment_deadline_at'], + ).replace('Z', '+00:00'), + ) + remaining = (assignment_deadline - datetime.now(timezone.utc)).total_seconds() + if remaining <= 0: + raise RunnerProtocolError('assignment authority expired before timeout publication') + watchdog_deadline_at = ( + datetime.now(timezone.utc) + timedelta(seconds=min( + TIMEOUT_BUNDLE_PUBLICATION_SECONDS, remaining, + )) + ).isoformat(timespec='milliseconds').replace('+00:00', 'Z') + try: + self._new_runner( + state, + scan_started_at=scan_started_at, + scan_deadline_at=scan_deadline_at, + watchdog_deadline_at=watchdog_deadline_at, + operation='timeout_bundle', + timeout_phase=final_phase, + attempt=attempt, + ) + except RunnerStageTimeout as exc: + raise RunnerContainmentPending( + 'timeout result authority persistence remains pending', + ) from exc + return self._launch_runner(state, allow_timeout_fallback=False) + + def _launch_process_bounded(self, runner, paths, remaining): + completed = threading.Event() + cancelled = threading.Event() + holder = {} + + def launch(): + try: + process = self._process_factory( + self._runner_command(paths['root']), + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + close_fds=True, + creationflags=( + subprocess.CREATE_NO_WINDOW if os.name == 'nt' else 0 + ), + _startup_timeout=max(0.01, min(15.0, remaining)), + ) + if cancelled.is_set(): + process.kill() + try: + process.wait(timeout=RUNNER_STOP_TIMEOUT_SECONDS) + except subprocess.TimeoutExpired: + pass + else: + try: + transferred = bind_transferred_runner_owner( + self.work_root, runner['root_name'], + self._identity_value(process.payload_identity), + ) + if not transferred: + bind_runner_owner( + self.work_root, runner['root_name'], + self._identity_value(process.payload_identity), + ) + except (OSError, RunnerProtocolError): + pass + else: + holder['process'] = process + except BaseException as exc: + holder['error'] = exc + finally: + completed.set() + + self._runner_launch_cleanup = completed + threading.Thread( + target=launch, name=f"worker-runner-launch-{self.slot_id}", daemon=True, + ).start() + if not completed.wait(max(0.0, remaining)): + cancelled.set() + publish_generation_terminal( + paths['terminal'], generation=runner['generation'], + input_sha256=runner['input_sha256'], decision='timed_out', + reason='startup_deadline', + ) + return None + if 'error' in holder: + raise holder['error'] + return holder.get('process') + + def _launch_runner(self, state, *, allow_timeout_fallback=True): + runner = state['runner'] + paths = self._runner_paths(runner) + watchdog_deadline = datetime.fromisoformat( + runner['watchdog_deadline_at'].replace('Z', '+00:00'), + ) + remaining = (watchdog_deadline - datetime.now(timezone.utc)).total_seconds() + if remaining <= 0: + publish_generation_terminal( + paths['terminal'], generation=runner['generation'], + input_sha256=runner['input_sha256'], decision='timed_out', + reason='startup_deadline', + ) + runner['status'] = 'timed_out' + self._save(state) + if allow_timeout_fallback and runner['operation'] == 'execute': + return self._timeout_bundle(state) + raise RunnerProtocolError('timeout result bundle publication exceeded its watchdog') + try: + process = self._launch_process_bounded(runner, paths, remaining) + except BaseException: + runner['status'] = 'fenced' + self._save(state) + try: + cleanup_runner_root(self.work_root, runner['root_name']) + except OSError: + pass + raise + if process is None: + runner['status'] = 'timed_out' + self._save(state) + if allow_timeout_fallback and runner['operation'] == 'execute': + return self._timeout_bundle(state) + raise RunnerProtocolError('timeout result bundle publication exceeded its watchdog') + watchdog_stop = threading.Event() + watchdog_expired = threading.Event() + + def watchdog(): + wait = max( + 0.0, + (watchdog_deadline - datetime.now(timezone.utc)).total_seconds(), + ) + if watchdog_stop.wait(wait): + return + watchdog_expired.set() + process.kill() + try: + publish_generation_terminal( + paths['terminal'], generation=runner['generation'], + input_sha256=runner['input_sha256'], decision='timed_out', + reason=( + 'scan_stage_deadline' if runner['operation'] == 'execute' + else 'timeout_bundle_deadline' + ), + ) + except (OSError, RunnerProtocolError): + pass + + watchdog_thread = threading.Thread( + target=watchdog, name=f"worker-runner-watchdog-{self.slot_id}", daemon=True, + ) + watchdog_thread.start() + try: + runner['identity'] = { + 'host': self._identity_value(process.host_identity), + 'payload': self._identity_value(process.payload_identity), + 'job_membership_verified': bool(process.job_membership_verified), + } + if not runner['identity']['job_membership_verified']: + raise RunnerProtocolError('assignment runner containment is unverified') + runner['status'] = 'running' + self._save(state) + if watchdog_expired.is_set(): + raise RunnerContainmentPending('runner deadline crossed during identity persistence') + bind_runner_owner( + self.work_root, runner['root_name'], runner['identity']['payload'], + ) + if watchdog_expired.is_set(): + raise RunnerContainmentPending('runner deadline crossed during owner persistence') + publish_start_gate( + paths['start'], generation=runner['generation'], + input_sha256=runner['input_sha256'], + host=runner['identity']['host'], payload=runner['identity']['payload'], + ) + while process.poll() is None: + if watchdog_expired.is_set(): + break + self._drain_runner_events(state) + time.sleep(0.1) + if watchdog_expired.is_set() and process.poll() is None: + try: + process.wait(timeout=RUNNER_STOP_TIMEOUT_SECONDS) + except subprocess.TimeoutExpired as exc: + runner['status'] = 'stopping' + self._save(state) + raise RunnerContainmentPending( + 'runner containment teardown remains pending', + ) from exc + if not watchdog_expired.is_set(): + watchdog_stop.set() + terminal = self._load_runner_terminal(state) + if terminal is None: + decision, _won = publish_generation_terminal( + paths['terminal'], generation=runner['generation'], + input_sha256=runner['input_sha256'], decision='fenced', + reason='runner_crash', + ) + if decision['decision'] == 'completed': + terminal = self._load_runner_terminal(state) + else: + runner['status'] = 'fenced' + self._save(state) + self._drain_runner_events(state) + return self._retry_closed_runner(state, 'runner_crash') + if terminal['decision'] == 'completed': + runner['status'] = 'exited' + self._save(state) + return self._adopt_runner_outcome(state, terminal['outcome']) + runner['status'] = ( + 'timed_out' if terminal['decision'] == 'timed_out' else 'fenced' + ) + self._save(state) + if ( + terminal['decision'] == 'timed_out' + and allow_timeout_fallback + and runner['operation'] == 'execute' + ): + return self._timeout_bundle(state) + raise RunnerProtocolError('runner generation closed without an adoptable outcome') + finally: + watchdog_stop.set() + if process.poll() is None: + process.kill() + try: + process.wait(timeout=RUNNER_STOP_TIMEOUT_SECONDS) + except subprocess.TimeoutExpired: + pass + + def _execute(self, state): + assignment = dict(state['assignment']) + packaged_capabilities = self.package_runtime.get('capabilities') + try: + validate_protocol2_remote_assignment( + assignment, self.compatibility, packaged_capabilities, + ) + except (ScanExecutionError, TypeError, ValueError) as exc: + raise WorkerAssignmentCompatibilityError( + 'worker assignment is incompatible with this package' + ) from exc + adopted = self._adopt_ready_bundle(state) + if adopted is not None: + return adopted + runner = state.get('runner') + if runner is not None: + source = self._runner_paths(runner)['root'] + abandoned = os.path.join( + self.work_root, 'abandoned', runner['root_name'], + ) + if not os.path.exists(source) and os.path.isdir(abandoned): + return self._retry_closed_runner( + state, 'janitor_transfer_recovery', + ) + try: + terminal = self._load_runner_terminal(state) + except RunnerProtocolError: + terminal = None + closed = self._fence_recovered_runner(state, 'malformed_output') + if closed == 'completed': + runner['status'] = 'fenced' + self._save(state) + closed = self._stop_persisted_runner(runner) + if closed == 'unknown': + raise RunnerContainmentPending( + 'malformed runner generation containment remains live', + ) + return self._retry_closed_runner(state, 'malformed_output') + if terminal is not None and terminal['decision'] == 'completed': + if self._stop_persisted_runner(runner) != 'dead': + raise RunnerContainmentPending( + 'completed runner containment remains live during recovery', + ) + return self._adopt_runner_outcome(state, terminal['outcome']) + closed = self._fence_recovered_runner(state, 'controller_recovery') + if closed == 'completed': + terminal = self._load_runner_terminal(state) + return self._adopt_runner_outcome(state, terminal['outcome']) + if closed == 'unknown': + raise RunnerContainmentPending( + 'recovered assignment runner could not be proven dead', + ) + try: + self._drain_runner_events(state) + except RunnerProtocolError: + return self._retry_closed_runner(state, 'malformed_event_tail') + if terminal is not None and terminal['decision'] == 'timed_out': + return self._timeout_bundle(state) + return self._retry_closed_runner(state, 'controller_recovery') + try: + self._new_runner(state) + except RunnerStageTimeout as exc: + if self._event_phase != WorkerPhase.PREPARING: + self._emit(WorkerPhase.PREPARING, state, { + 'reason': 'prelaunch_stage_timeout', + }) + return self._start_timeout_bundle( + state, + scan_started_at=exc.scan_started_at, + scan_deadline_at=exc.scan_deadline_at, + final_phase=exc.phase, + ) + return self._launch_runner(state) + + def _terminal(self, state, reservation_id, failure_code=None, detail=None): + if failure_code is not None: + state['phase'] = 'terminal_pending' + report = self._fitted_terminal_report(failure_code, detail) + self._archive_fitted_terminal_diagnostics(report) + encoded = json.dumps( + report, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + if len(encoded) > TERMINAL_REPORT_MAX_BYTES: + raise WorkerClientError('pending terminal report exceeds its byte bound') + state['terminal'] = report + self._save(state) + terminal = dict(state.get('terminal') or {}) + if set(terminal) not in ( + {'failure_code', 'detail'}, + {'failure_code', 'detail', 'diagnostics'}, + ): + raise WorkerClientError('pending terminal report is invalid') + if self._event_phase not in { + WorkerPhase.UPLOADING, WorkerPhase.AWAITING_RECEIPT, + }: + self._emit(WorkerPhase.UPLOADING, state, {'reason': 'terminal_report'}) + try: + awaiting_emitted = False + + def awaiting_receipt(): + nonlocal awaiting_emitted + if not awaiting_emitted: + self._emit( + WorkerPhase.AWAITING_RECEIPT, state, + {'reason': 'terminal_report'}, + ) + awaiting_emitted = True + + if isinstance(self.api, WorkerHTTPClient): + receipt = self.api.terminal( + reservation_id, terminal, + response_wait_callback=awaiting_receipt, + ) + else: + receipt = self.api.terminal(reservation_id, terminal) + awaiting_receipt() + except WorkerHTTPError as exc: + if self._event_phase in { + WorkerPhase.UPLOADING, WorkerPhase.AWAITING_RECEIPT, + }: + self._emit(WorkerPhase.BACKOFF, state, {'reason': 'terminal_retry'}) + return self._reconcile_http_failure(state, exc) + except Exception: + if self._event_phase in { + WorkerPhase.UPLOADING, WorkerPhase.AWAITING_RECEIPT, + }: + self._emit(WorkerPhase.BACKOFF, state, {'reason': 'terminal_retry'}) + raise + if not self._resolved(state, receipt): + raise WorkerClientError('worker API terminal receipt is not terminal') + return True + + def step(self): + state = self._load() + self._resume_events(state) + if state is None: + if not self.claim_enabled: + return False + self._emit(WorkerPhase.CLAIMING) + state = {'phase': 'claiming', 'request_id': secrets.token_hex(16)} + state = self._save(state) + if state.get('assignment') and self._cleanup_recorded_stale(state): + return True + if state.get('phase') == 'claiming': + claim_result = self.api.claim( + state['request_id'], self.compatibility.as_dict(), + ) + if claim_result is None: + self._remove_state() + self._emit(WorkerPhase.IDLE) + return False + if set(claim_result) == {'retry_after_seconds', 'reason'}: + retry_after = int(claim_result['retry_after_seconds']) + if not 1 <= retry_after <= 300: + raise WorkerClientError('worker API claim retry delay is invalid') + self.retry_after_seconds = retry_after + self._remove_state() + progress = { + 'retry_after_seconds': retry_after, + 'next_claim_at': datetime.fromtimestamp( + time.time() + retry_after, timezone.utc, + ).isoformat().replace('+00:00', 'Z'), + } + if claim_result['reason'] is not None: + progress['reason'] = claim_result['reason'] + self._emit(WorkerPhase.BACKOFF, progress=progress) + return False + if set(claim_result) == {'claim_resolution'}: + self._remove_state() + self._emit(WorkerPhase.IDLE, progress={'reason': 'claim_reconciled'}) + return True + assignment = claim_result + self._clear_assignment_diagnostics() + state = {'phase': 'assigned', 'assignment': assignment} + state = self._save(state) + self._emit(WorkerPhase.ASSIGNED, state) + + assignment = dict(state.get('assignment') or {}) + reservation = dict(assignment.get('reservation') or {}) + reservation_id = int(reservation.get('reservation_id') or 0) + if reservation_id <= 0: + raise WorkerClientError('pending assignment has an invalid reservation identity') + try: + status = self.api.status(reservation_id) + except WorkerHTTPError as exc: + if exc.status_code in (404, 409, 410): + return self._mark_stale(state, exc) + raise + if self._resolved(state, status): + return True + if state.get('phase') == 'awaiting_resolution': + return False + if str(status.get('state') or '') != 'scanning' and state.get('phase') != 'bundle_ready': + raise WorkerClientError('pending assignment is no longer executable') + + if state.get('phase') == 'terminal_pending': + return self._terminal(state, reservation_id) + + if state.get('phase') == 'assigned': + deadline = status.get('expires_at') or reservation.get('remote_expires_at') + if not self._deadline_active(deadline): + raise WorkerClientError('worker assignment deadline has passed') + try: + state = self._execute(state) + except WorkerAssignmentCompatibilityError: + raise + except RunnerContainmentPending: + raise + except RunnerProtocolError as exc: + self._capture_failure_diagnostic( + state, exc, 'runner_protocol_failed', DiagnosticCategory.PROTOCOL, + ) + return self._terminal( + state, reservation_id, 'client_process_failed', + CLIENT_PROCESS_FAILURE_DETAIL, + ) + except OSError as exc: + self._capture_failure_diagnostic( + state, exc, 'client_storage_failed', DiagnosticCategory.STORAGE, + operation='assignment_execute', + ) + return self._terminal( + state, reservation_id, 'client_storage_failed', + CLIENT_STORAGE_FAILURE_DETAIL, + ) + except Exception as exc: + self._capture_failure_diagnostic( + state, exc, 'client_process_failed', DiagnosticCategory.INTERNAL, + ) + return self._terminal( + state, reservation_id, 'client_process_failed', + CLIENT_PROCESS_FAILURE_DETAIL, + ) + + if state.get('phase') == 'bundle_ready': + path = bundle_ready_path(self.bundle_root, reservation['bundle_id']) + if not os.path.isfile(path) or os.path.islink(path): + self._capture_failure_diagnostic( + state, WorkerClientError('pending result bundle is missing'), + 'client_storage_failed', DiagnosticCategory.STORAGE, + ) + return self._terminal( + state, reservation_id, 'client_storage_failed', + 'pending result bundle is missing', + ) + try: + if self._event_phase in { + WorkerPhase.ASSIGNED, WorkerPhase.BUNDLING, WorkerPhase.BACKOFF, + }: + self._emit(WorkerPhase.UPLOADING, state) + awaiting_emitted = False + + def awaiting_receipt(): + nonlocal awaiting_emitted + if not awaiting_emitted: + self._emit(WorkerPhase.AWAITING_RECEIPT, state) + awaiting_emitted = True + + if isinstance(self.api, WorkerHTTPClient): + receipt = self.api.upload( + reservation_id, path, + response_wait_callback=awaiting_receipt, + ) + else: + receipt = self.api.upload(reservation_id, path) + awaiting_receipt() + except WorkerHTTPError as exc: + if self._event_phase in {WorkerPhase.UPLOADING, WorkerPhase.AWAITING_RECEIPT}: + self._emit(WorkerPhase.BACKOFF, state, {'reason': 'upload_retry'}) + return self._reconcile_http_failure(state, exc) + except Exception: + if self._event_phase in {WorkerPhase.UPLOADING, WorkerPhase.AWAITING_RECEIPT}: + self._emit(WorkerPhase.BACKOFF, state, {'reason': 'upload_retry'}) + raise + if not self._resolved(state, receipt): + raise WorkerClientError('worker API upload receipt is not terminal') + return True + raise WorkerClientError('pending slot phase is invalid') + + +def persisted_slot_ids(state_dir): + slot_ids = set() + entries = 0 + with os.scandir(state_dir) as iterator: + for entry in iterator: + entries += 1 + if entries > 4096: + raise WorkerClientError('worker state directory exceeds its entry bound') + match = re.fullmatch(r'slot-(0|[1-9][0-9]*)\.json', entry.name) + if not match: + continue + if entry.is_symlink() or not entry.is_file(follow_symlinks=False): + raise WorkerClientError('worker slot state is not a regular file') + slot_id = int(match.group(1)) + if slot_id > 100000: + raise WorkerClientError('worker slot identity exceeds its bound') + slot_ids.add(slot_id) + return slot_ids + + +def persisted_runner_root_names(state_dir): + roots = set() + for slot_id in persisted_slot_ids(state_dir): + path = os.path.join(state_dir, f'slot-{slot_id}.json') + try: + state = WorkerSlot._slot_state( + read_private_json(path, max_bytes=MAX_PENDING_BYTES), + slot_id=slot_id, + ) + except FileNotFoundError: + continue + runner = state.get('runner') + if runner is not None: + roots.add(runner['root_name']) + roots.add(f"abandoned/{runner['root_name']}") + return roots + + +def default_worker_paths(module_path=None, *, platform_name=None, environ=None, home=None): + platform_name = str(platform_name or os.name) + path_module = ntpath if platform_name == 'nt' else posixpath + environment = os.environ if environ is None else environ + package_root = path_module.dirname(path_module.dirname(path_module.abspath( + module_path or __file__, + ))) + home = path_module.abspath(home or os.path.expanduser('~')) + + if platform_name == 'nt': + state_base = str(environment.get('LOCALAPPDATA') or '') + if state_base and not path_module.isabs(state_base): + raise ValueError('LOCALAPPDATA must be absolute') + if not state_base: + state_base = path_module.join(home, 'AppData', 'Local') + state_dir = path_module.join(state_base, 'TRUF', 'RemoteWorker') + bundle_dir = path_module.join(state_dir, 'bundles') + work_dir = path_module.join(state_dir, 'work') + else: + state_base = str(environment.get('XDG_STATE_HOME') or '') + data_base = str(environment.get('XDG_DATA_HOME') or '') + if state_base and not path_module.isabs(state_base): + raise ValueError('XDG_STATE_HOME must be absolute') + if data_base and not path_module.isabs(data_base): + raise ValueError('XDG_DATA_HOME must be absolute') + state_base = state_base or path_module.join(home, '.local', 'state') + data_base = data_base or path_module.join(home, '.local', 'share') + state_dir = path_module.join(state_base, 'truf', 'remote-worker') + bundle_dir = path_module.join(data_base, 'truf', 'remote-worker', 'bundles') + work_dir = path_module.join(data_base, 'truf', 'remote-worker', 'work') + return { + 'package_manifest': path_module.join(package_root, 'worker-package.json'), + 'state_dir': state_dir, + 'bundle_dir': bundle_dir, + 'work_dir': work_dir, + } + + +def run_client( + args, *, drain_event=None, stop_event=None, event_callback=None, + terminal_callback=None, log_callback=None, acquire_lock=True, + started_callback=None, diagnostic_callback=None, + progress_event_reader=None, +): + package = verify_worker_package(args.package_manifest) + manifest = package.pop('manifest') + compatibility = WorkerBuildCompatibility.from_mapping( + package.pop('build_compatibility'), + ) + package.pop('runtime_trees') + package['capabilities'] = tuple( + ( + capability['source'], capability['platform'], + capability['planning_kind'], + ) + for capability in manifest['capabilities'] + ) + package['bootstrap_path'] = os.path.join( + os.path.dirname(os.path.abspath(args.package_manifest)), + 'app', 'remote_worker_bootstrap.py', + ) + state_dir = ensure_private_directory(os.path.abspath(args.state_dir), reject_reparse=True) + bundle_root = ensure_private_directory(os.path.abspath(args.bundle_dir), reject_reparse=True) + work_root = ensure_private_directory(os.path.abspath(args.work_dir), reject_reparse=True) + for name in ('tmp', 'ready', 'quarantine'): + ensure_private_directory(os.path.join(bundle_root, name), reject_reparse=True) + scanner.scan_config.work_dir = work_root + api = WorkerHTTPClient(args.server, args.token, args.http_timeout) + stopping = stop_event or threading.Event() + draining = drain_event or threading.Event() + progress_outbox = ( + ProgressOutbox(api, state_dir, progress_event_reader) + if progress_event_reader is not None else None + ) + progress_stopping = threading.Event() + progress_thread = ( + threading.Thread( + target=progress_outbox.run, + args=(progress_stopping,), + name='worker-progress-outbox', daemon=True, + ) + if progress_outbox is not None else None + ) + configured_slot_ids = set(range(args.parallelism)) + slot_ids = sorted(configured_slot_ids | persisted_slot_ids(state_dir)) + start_gate = threading.Event() + ready_events = {slot_id: threading.Event() for slot_id in slot_ids} + startup_errors = {} + startup_lock = threading.Lock() + + def loop(slot_id): + try: + slot = WorkerSlot( + slot_id, api, compatibility.as_dict(), state_dir, bundle_root, + work_root=work_root, + package_runtime=package, + claim_enabled=slot_id in configured_slot_ids, + event_callback=event_callback, + terminal_callback=terminal_callback, + diagnostic_callback=diagnostic_callback, + ) + except BaseException as exc: + with startup_lock: + startup_errors[slot_id] = exc + ready_events[slot_id].set() + return + ready_events[slot_id].set() + start_gate.wait() + while not stopping.is_set(): + if draining.is_set(): + slot.claim_enabled = False + if not slot.claim_enabled and not os.path.exists(slot.state_path): + if hasattr(slot, 'retire'): + slot.retire( + 'graceful_drain' if draining.is_set() + else 'lowered_parallelism_recovery_complete' + ) + return + try: + worked = slot.step() + delay = 0 if worked else ( + slot.retry_after_seconds + if slot.retry_after_seconds is not None else args.poll_seconds + ) + slot.retry_after_seconds = None + except RunnerContainmentPending: + message = f'worker slot {slot_id}: containment authority remains pending' + print(message, flush=True) + if log_callback is not None: + log_callback(message) + return + except Exception as exc: + print( + f'worker slot {slot_id}: {safe_worker_error_summary(exc)}', + flush=True, + ) + if log_callback is not None: + log_callback( + f'worker slot {slot_id}: {safe_worker_error_summary(exc)}' + ) + delay = args.error_delay_seconds + stopping.wait(max(0.1, float(delay))) + + threads = [ + threading.Thread(target=loop, args=(slot_id,), name=f'worker-slot-{slot_id}', daemon=True) + for slot_id in slot_ids + ] + active_threads = threads[:args.parallelism] + lock = ( + PrivateFileLock(os.path.join(state_dir, 'remote-worker.lock')) + if acquire_lock else nullcontext() + ) + with lock: + forced_stop = False + unexpected_exit = False + try: + if progress_thread is not None: + progress_thread.start() + for thread in threads: + thread.start() + startup_deadline = time.monotonic() + 5.0 + for slot_id in slot_ids: + remaining = startup_deadline - time.monotonic() + if remaining <= 0 or not ready_events[slot_id].wait(remaining): + raise WorkerClientError('worker slot startup timed out') + if startup_errors or not all(thread.is_alive() for thread in threads): + raise WorkerClientError('worker slot startup failed') + if started_callback is not None: + started_callback() + start_gate.set() + while True: + if stopping.is_set(): + forced_stop = True + break + monitored = threads if (draining.is_set() or stopping.is_set()) else active_threads + if not monitored or not all(thread.is_alive() for thread in monitored): + if not (draining.is_set() or stopping.is_set()) or not any( + thread.is_alive() for thread in threads + ): + unexpected_exit = not (draining.is_set() or stopping.is_set()) + break + time.sleep(0.5) + except KeyboardInterrupt: + forced_stop = True + stopping.set() + finally: + stopping.set() + start_gate.set() + if not forced_stop: + for thread in threads: + thread.join(timeout=5) + if progress_thread is not None: + progress_shutdown_started = time.monotonic() + progress_stopping.set() + api.cancel_progress_requests() + progress_thread.join(timeout=PROGRESS_REQUEST_TIMEOUT_SECONDS + 0.5) + if progress_thread.is_alive(): + raise WorkerClientError( + 'progress publisher exceeded its absolute shutdown bound' + ) + if not forced_stop: + progress_outbox.drain(max( + 0.0, + PROGRESS_FINAL_DRAIN_SECONDS + - (time.monotonic() - progress_shutdown_started), + )) + return 2 if forced_stop else (1 if unexpected_exit else 0) + + +def parse_args(argv=None): + defaults = default_worker_paths() + parser = argparse.ArgumentParser(description='Trusted TRUF remote scan worker') + parser.add_argument('--server', required=True) + parser.add_argument('--token', required=True) + parser.add_argument('--parallelism', type=int, default=1) + parser.set_defaults( + package_manifest=defaults['package_manifest'], + state_dir=defaults['state_dir'], + bundle_dir=defaults['bundle_dir'], + work_dir=defaults['work_dir'], + poll_seconds=5.0, + error_delay_seconds=15.0, + http_timeout=120, + ) + args = parser.parse_args(argv) + if not 1 <= args.parallelism <= 128: + parser.error('--parallelism must be between 1 and 128') + return args + + +def main(argv=None): + run_client(parse_args(argv)) + + +if __name__ == '__main__': + main() diff --git a/app/requirements-keycheckers.txt b/app/requirements-keycheckers.txt new file mode 100644 index 0000000..dd2a66f --- /dev/null +++ b/app/requirements-keycheckers.txt @@ -0,0 +1,4 @@ +requests +boto3 +botocore +psycopg[binary]>=3.2 diff --git a/app/requirements.txt b/app/requirements.txt new file mode 100644 index 0000000..e448d65 --- /dev/null +++ b/app/requirements.txt @@ -0,0 +1,10 @@ +streamlit +starlette>=0.47.3,<1 +uvicorn>=0.53,<1 +python-multipart>=0.0.10 +pandas +requests +plotly +PyYAML +psycopg[binary]>=3.2 +zstandard==0.23.0 diff --git a/app/result_bundle.py b/app/result_bundle.py new file mode 100644 index 0000000..4904ab4 --- /dev/null +++ b/app/result_bundle.py @@ -0,0 +1,757 @@ +import hashlib +import json +import os +import re +import struct +from dataclasses import dataclass + +from runtime_security import ( + PrivatePathState, + durable_publish, + ensure_private_directory, + harden_private_file, + inspect_private_relative_path, + private_file_ready, + reject_reparse_components, + require_private_directory, +) +from worker_contracts import ( + AssignmentOutcome, + MAX_DIAGNOSTIC_AGGREGATE_BYTES, + MAX_DIAGNOSTICS_PER_ASSIGNMENT, + decode_diagnostic_envelope, + build_legacy_error_frame_diagnostics, + encode_diagnostic_envelope, +) + + +MAGIC = b'TRUF-RB2\n' +FORMAT_VERSION = 2 +FRAME_HEADER = struct.Struct('!cI') +FRAME_TYPES = frozenset((b'H', b'F', b'E', b'D', b'K', b'M', b'C')) +FRAME_ORDER = {name: index for index, name in enumerate((b'H', b'F', b'E', b'D', b'K', b'M', b'C'))} +ID_RE = re.compile(r'^[a-f0-9]{32,64}$') +DEFAULT_MAX_EVENT_BYTES = 64 * 1024 * 1024 +DEFAULT_MAX_FRAME_BYTES = 16 * 1024 * 1024 +FRAME_BOUNDS = { + b'H': 1024 * 1024, + b'F': 16 * 1024 * 1024, + b'E': 1024 * 1024, + b'D': 64 * 1024, + b'K': 2 * 1024 * 1024, + b'M': 16 * 1024 * 1024, + b'C': 1024 * 1024, +} +MAX_FINDING_FRAMES = 20000 +MAX_ERROR_FRAMES = 2000 +MAX_CANDIDATE_FRAMES = 2000 +MAX_TOTAL_FRAMES = 24003 + MAX_DIAGNOSTICS_PER_ASSIGNMENT + + +class ResultBundleError(ValueError): + pass + + +class ResultBundleConflictError(ResultBundleError): + pass + + +class ResultBundleUnavailableError(OSError): + pass + + +def canonical_json_bytes(value): + try: + return json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('utf-8') + except (TypeError, ValueError) as exc: + raise ResultBundleError('bundle frame is not canonical JSON data') from exc + + +def _validated_id(value, name): + text = str(value or '').lower() + if not ID_RE.fullmatch(text): + raise ResultBundleError(f'invalid {name}') + return text + + +def _relative_ready_path(bundle_id): + bundle_id = _validated_id(bundle_id, 'bundle_id') + return os.path.join('ready', bundle_id[:2], f'{bundle_id}.trb') + + +def bundle_ready_path(root, bundle_id): + return os.path.join(os.path.abspath(root), _relative_ready_path(bundle_id)) + + +def bundle_partial_relative_path(bundle_id, reservation_token): + bundle_id = _validated_id(bundle_id, 'bundle_id') + producer_token = hashlib.sha256(str(reservation_token).encode('utf-8')).hexdigest()[:24] + return os.path.join('tmp', bundle_id[:2], f'{bundle_id}.{producer_token}.partial') + + +def bundle_partial_path(root, bundle_id, reservation_token): + return os.path.join( + os.path.abspath(root), bundle_partial_relative_path(bundle_id, reservation_token), + ) + + +def ensure_bundle_reservation_paths(root, reservation): + root = require_private_directory(os.path.abspath(root), create=False) + reservation = ( + reservation if isinstance(reservation, BundleReservation) + else BundleReservation.from_mapping(reservation) + ) + for name in ('tmp', 'ready', 'quarantine'): + ensure_private_directory( + os.path.join(root, name, reservation.bundle_id[:2]), + reject_reparse=True, + ) + return reservation + + +@dataclass(frozen=True) +class BundleReservation: + reservation_id: int + reservation_token: str + bundle_id: str + scan_event_id: str + queue_id: int + claim_lease_token: str + declared_bytes: int + ready_path: str + source: str = '' + platform: str = '' + query: str = '' + target: str = '' + normalized_target: str = '' + run_id: int | None = None + cycle_id: int | None = None + producer_instance_id: str = '' + producer_pid: int = 0 + producer_creation_time: str = '' + producer_executable: str = '' + + @classmethod + def from_mapping(cls, value): + data = dict(value or {}) + ready_path = data.get('ready_path') or data.get('ready_relative_path') or '' + return cls( + reservation_id=int(data['reservation_id'] if 'reservation_id' in data else data['id']), + reservation_token=str(data['reservation_token']), + bundle_id=_validated_id(data['bundle_id'], 'bundle_id'), + scan_event_id=_validated_id(data['scan_event_id'], 'scan_event_id'), + queue_id=int(data['queue_id']), + claim_lease_token=str(data.get('claim_lease_token') or data.get('claim_lease_token_value') or ''), + declared_bytes=int(data.get('declared_bytes') or data.get('declared_bundle_bytes') or 0), + ready_path=str(ready_path), + source=str(data.get('source') or ''), + platform=str(data.get('platform') or ''), + query=str(data.get('query') or ''), + target=str(data.get('target') or ''), + normalized_target=str(data.get('normalized_target') or ''), + run_id=data.get('run_id'), + cycle_id=data.get('cycle_id'), + producer_instance_id=str(data.get('producer_instance_id') or ''), + producer_pid=int(data.get('producer_pid') or 0), + producer_creation_time=str(data.get('producer_creation_time') or ''), + producer_executable=str(data.get('producer_executable') or ''), + ) + + def header(self): + return { + 'format_version': FORMAT_VERSION, + 'reservation_id': self.reservation_id, + 'reservation_token': self.reservation_token, + 'bundle_id': self.bundle_id, + 'scan_event_id': self.scan_event_id, + 'queue_id': self.queue_id, + 'claim_lease_token': self.claim_lease_token, + 'declared_bytes': self.declared_bytes, + 'ready_relative_path': self.ready_path.replace('\\', '/'), + 'source': self.source, + 'platform': self.platform, + 'query': self.query, + 'target': self.target, + 'normalized_target': self.normalized_target, + 'run_id': self.run_id, + 'cycle_id': self.cycle_id, + 'producer_instance_id': self.producer_instance_id, + 'producer_pid': self.producer_pid, + 'producer_creation_time': self.producer_creation_time, + 'producer_executable': self.producer_executable, + } + + +@dataclass(frozen=True) +class BundleCommit: + reservation_id: int + bundle_id: str + scan_event_id: str + scan_event_hash: str + relative_path: str + actual_bytes: int + frame_count: int + finding_count: int + error_count: int + candidate_count: int + + def as_dict(self): + return dict(self.__dict__) + + +@dataclass(frozen=True) +class BundleMetadata(BundleCommit): + header: dict + result_metadata: dict + diagnostic_count: int + diagnostic_bytes: int + + +class ResultBundleWriter: + def __init__(self, root, reservation, handle, partial_path, ready_path, fault=None): + self.root = root + self.reservation = reservation + self.handle = handle + self.partial_path = partial_path + self.ready_path = ready_path + self.fault = fault + self.digest = hashlib.sha256() + self.bytes_written = 0 + self.frame_count = 0 + self.finding_count = 0 + self.error_count = 0 + self.candidate_count = 0 + self.diagnostic_count = 0 + self.diagnostic_bytes = 0 + self.diagnostic_aggregate_bytes = 0 + self._diagnostic_uids = set() + self._last_frame_order = FRAME_ORDER[b'H'] + self.finished = False + self.ready_published = False + self._write_bytes(MAGIC) + self._write_frame(b'H', reservation.header()) + + @classmethod + def open(cls, root, reservation, fault=None, require_s_drive=False): + root = require_private_directory(os.path.abspath(root), create=False) + drive = os.path.splitdrive(root)[0].upper() + if require_s_drive and drive != 'S:': + raise ResultBundleError('production result bundle root must be on S:') + reservation = ensure_bundle_reservation_paths(root, reservation) + if reservation.declared_bytes <= len(MAGIC) or reservation.declared_bytes > DEFAULT_MAX_EVENT_BYTES: + raise ResultBundleError('declared bundle byte bound is invalid') + expected_relative = _relative_ready_path(reservation.bundle_id) + supplied_relative = str(reservation.ready_path or expected_relative).replace('/', os.sep) + if os.path.normcase(os.path.normpath(supplied_relative)) != os.path.normcase(os.path.normpath(expected_relative)): + raise ResultBundleError('reservation ready path is not deterministic for its bundle ID') + tmp_dir = os.path.join(root, 'tmp', reservation.bundle_id[:2]) + ready_dir = os.path.join(root, 'ready', reservation.bundle_id[:2]) + partial_path = bundle_partial_path( + root, reservation.bundle_id, reservation.reservation_token, + ) + ready_path = os.path.join(ready_dir, f'{reservation.bundle_id}.trb') + if os.path.lexists(ready_path): + raise ResultBundleConflictError('deterministic ready bundle path already exists') + descriptor = os.open( + partial_path, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), + 0o600, + ) + try: + os.close(descriptor) + descriptor = None + harden_private_file(partial_path) + handle = open(partial_path, 'w+b', buffering=0) + return cls(root, reservation, handle, partial_path, ready_path, fault=fault) + except BaseException: + if descriptor is not None: + os.close(descriptor) + try: + os.remove(partial_path) + except OSError: + pass + raise + + def _inject(self, stage): + if self.fault is not None: + self.fault(stage, self) + + def _write_bytes(self, payload): + if self.bytes_written + len(payload) > self.reservation.declared_bytes: + raise ResultBundleError('bundle exceeded its pre-reserved byte bound') + self.handle.write(payload) + self.digest.update(payload) + self.bytes_written += len(payload) + + def _write_frame(self, frame_type, value): + if self.finished or frame_type not in FRAME_TYPES or frame_type == b'C': + raise ResultBundleError('invalid bundle frame write') + if self.frame_count >= MAX_TOTAL_FRAMES - 1: + raise ResultBundleError('bundle frame count exceeds its bound') + if FRAME_ORDER[frame_type] < self._last_frame_order: + raise ResultBundleError('bundle frame order is not deterministic') + payload = canonical_json_bytes(value) + bound = min(FRAME_BOUNDS[frame_type], self.reservation.declared_bytes) + if len(payload) > bound: + raise ResultBundleError(f'{frame_type.decode()} frame exceeds its byte bound') + framed = FRAME_HEADER.pack(frame_type, len(payload)) + payload + self._inject(f'before_frame_{frame_type.decode()}') + self._write_bytes(framed) + self.frame_count += 1 + self._last_frame_order = FRAME_ORDER[frame_type] + self._inject(f'after_frame_{frame_type.decode()}') + return len(payload) + + def write_finding(self, finding): + if self.finding_count >= MAX_FINDING_FRAMES: + raise ResultBundleError('bundle finding count exceeds its bound') + self._write_frame(b'F', finding) + self.finding_count += 1 + + def write_error(self, error): + if self.error_count >= MAX_ERROR_FRAMES: + raise ResultBundleError('bundle error count exceeds its bound') + self._write_frame(b'E', {'error': str(error)}) + self.error_count += 1 + + def write_diagnostic(self, diagnostic): + if self.diagnostic_count >= MAX_DIAGNOSTICS_PER_ASSIGNMENT: + raise ResultBundleError('bundle diagnostic count exceeds its bound') + try: + if isinstance(diagnostic, dict): + envelope = decode_diagnostic_envelope(canonical_json_bytes(diagnostic)) + else: + envelope = decode_diagnostic_envelope( + encode_diagnostic_envelope(diagnostic) + ) + payload = encode_diagnostic_envelope(envelope) + value = json.loads(payload.decode('ascii')) + except (TypeError, ValueError, UnicodeError) as exc: + raise ResultBundleError('bundle diagnostic frame is invalid') from exc + if envelope.diagnostic_uid in self._diagnostic_uids: + raise ResultBundleError('bundle diagnostic identity is duplicated') + if ( + envelope.reservation_id != self.reservation.reservation_id + or envelope.scan_event_id != self.reservation.scan_event_id + or envelope.source != self.reservation.source + ): + raise ResultBundleError('bundle diagnostic identity conflicts with its reservation') + if ( + self.diagnostic_aggregate_bytes + len(payload) + 1 + > MAX_DIAGNOSTIC_AGGREGATE_BYTES + ): + raise ResultBundleError('bundle diagnostic aggregate exceeds its byte bound') + self._write_frame(b'D', value) + self._diagnostic_uids.add(envelope.diagnostic_uid) + self.diagnostic_count += 1 + self.diagnostic_bytes += FRAME_HEADER.size + len(payload) + self.diagnostic_aggregate_bytes += len(payload) + 1 + + def write_candidate(self, candidate): + if self.candidate_count >= MAX_CANDIDATE_FRAMES: + raise ResultBundleError('bundle candidate count exceeds its bound') + self._write_frame(b'K', candidate) + self.candidate_count += 1 + + def finish(self, metadata): + if self.finished: + raise ResultBundleError('bundle writer is already finished') + self._write_frame(b'M', metadata) + content_hash = self.digest.hexdigest() + footer = { + 'format_version': FORMAT_VERSION, + 'content_sha256': content_hash, + 'content_bytes': self.bytes_written, + 'byte_count': 0, + 'frame_count': self.frame_count + 1, + 'finding_count': self.finding_count, + 'error_count': self.error_count, + 'candidate_count': self.candidate_count, + } + if self.diagnostic_count: + footer.update({ + 'diagnostic_count': self.diagnostic_count, + 'diagnostic_bytes': self.diagnostic_bytes, + }) + while True: + payload = canonical_json_bytes(footer) + framed = FRAME_HEADER.pack(b'C', len(payload)) + payload + total = self.bytes_written + len(framed) + if footer['byte_count'] == total: + break + footer['byte_count'] = total + if total > self.reservation.declared_bytes: + raise ResultBundleError('bundle footer exceeds its pre-reserved byte bound') + self._inject('before_footer') + self.handle.write(framed) + self.bytes_written = total + self.frame_count += 1 + self._inject('after_footer') + self._inject('before_fsync') + self.handle.flush() + os.fsync(self.handle.fileno()) + self._inject('after_fsync') + self.handle.close() + self.handle = None + if not private_file_ready(self.partial_path): + raise ResultBundleError('private bundle ACL verification failed before publication') + if os.path.getsize(self.partial_path) != self.bytes_written: + raise ResultBundleError('bundle size changed before publication') + self._inject('before_rename') + durable_publish(self.partial_path, self.ready_path) + self.ready_published = True + self._inject('after_rename') + if not private_file_ready(self.ready_path): + inspection = inspect_private_relative_path( + self.root, os.path.relpath(self.ready_path, self.root), + ) + if inspection.state == PrivatePathState.UNKNOWN: + raise ResultBundleUnavailableError( + 'ready bundle state is unavailable after publication' + ) + if inspection.state == PrivatePathState.PRESENT: + raise ResultBundleError('ready bundle ACL verification failed') + self.finished = True + relative = os.path.relpath(self.ready_path, self.root) + return BundleCommit( + reservation_id=self.reservation.reservation_id, + bundle_id=self.reservation.bundle_id, + scan_event_id=self.reservation.scan_event_id, + scan_event_hash=content_hash, + relative_path=relative.replace(os.sep, '/'), + actual_bytes=self.bytes_written, + frame_count=self.frame_count, + finding_count=self.finding_count, + error_count=self.error_count, + candidate_count=self.candidate_count, + ) + + def abort(self): + if self.handle is not None: + self.handle.close() + self.handle = None + if not self.ready_published: + try: + os.remove(self.partial_path) + except FileNotFoundError: + pass + + def __enter__(self): + return self + + def __exit__(self, exc_type, value, traceback): + if not self.finished: + self.abort() + + +class ResultBundleReader: + def __init__(self, path, max_event_bytes=DEFAULT_MAX_EVENT_BYTES): + self.path = os.path.abspath(path) + self.max_event_bytes = max(1, int(max_event_bytes)) + self._validated = None + self._validated_fingerprint = None + + @classmethod + def from_reservation(cls, root, reservation, max_event_bytes=DEFAULT_MAX_EVENT_BYTES): + reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation) + return cls(bundle_ready_path(root, reservation.bundle_id), max_event_bytes=max_event_bytes) + + @staticmethod + def _stat_fingerprint(value): + return ( + int(value.st_dev), int(value.st_ino), int(value.st_mode), + int(value.st_size), int(value.st_mtime_ns), + ) + + def _file_fingerprint(self): + reject_reparse_components(self.path) + if not private_file_ready(self.path): + raise ResultBundleUnavailableError('bundle path is not currently available as an exact private regular file') + return self._stat_fingerprint(os.stat(self.path, follow_symlinks=False)) + + def _iter_frames(self, expected_fingerprint=None): + fingerprint = self._file_fingerprint() + if expected_fingerprint is not None and fingerprint != expected_fingerprint: + raise ResultBundleError('bundle changed after validation') + size = fingerprint[3] + if size <= len(MAGIC) or size > self.max_event_bytes: + raise ResultBundleError('bundle aggregate byte bound is invalid') + with open(self.path, 'rb', buffering=0) as handle: + opened_fingerprint = self._stat_fingerprint(os.fstat(handle.fileno())) + if opened_fingerprint != fingerprint: + raise ResultBundleUnavailableError('bundle identity changed while it was opened') + magic = handle.read(len(MAGIC)) + if magic != MAGIC: + raise ResultBundleError('bundle magic/version mismatch') + offset = len(MAGIC) + while offset < size: + header = handle.read(FRAME_HEADER.size) + if len(header) != FRAME_HEADER.size: + raise ResultBundleError('truncated bundle frame header') + frame_type, payload_length = FRAME_HEADER.unpack(header) + if frame_type not in FRAME_TYPES: + raise ResultBundleError('unknown bundle frame type') + bound = min(FRAME_BOUNDS[frame_type], self.max_event_bytes) + if payload_length > bound or offset + FRAME_HEADER.size + payload_length > size: + raise ResultBundleError('bundle frame length exceeds its bound') + payload = handle.read(payload_length) + if len(payload) != payload_length: + raise ResultBundleError('truncated bundle frame payload') + try: + value = json.loads(payload.decode('utf-8', errors='strict')) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ResultBundleError('bundle frame contains invalid UTF-8 JSON') from exc + if canonical_json_bytes(value) != payload: + raise ResultBundleError('bundle frame JSON is not canonical') + offset += FRAME_HEADER.size + payload_length + yield frame_type, value, header + payload, offset + if offset != size: + raise ResultBundleError('bundle byte count is inconsistent') + if self._stat_fingerprint(os.fstat(handle.fileno())) != opened_fingerprint: + raise ResultBundleError('bundle changed while it was read') + if self._file_fingerprint() != fingerprint: + raise ResultBundleError('bundle path changed while it was read') + + def validate(self): + if self._validated is not None: + return self._validated + digest = hashlib.sha256(MAGIC) + header_value = None + metadata_value = None + footer = None + counts = {b'F': 0, b'E': 0, b'D': 0, b'K': 0} + diagnostic_bytes = 0 + diagnostic_aggregate_bytes = 0 + diagnostic_uids = set() + diagnostic_scan_outcomes = [] + frame_count = 0 + last_frame_order = -1 + final_offset = len(MAGIC) + content_bytes = None + fingerprint = self._file_fingerprint() + for frame_type, value, framed, offset in self._iter_frames(fingerprint): + frame_count += 1 + if frame_count > MAX_TOTAL_FRAMES: + raise ResultBundleError('bundle frame count exceeds its bound') + if FRAME_ORDER[frame_type] < last_frame_order: + raise ResultBundleError('bundle frame order is not deterministic') + last_frame_order = FRAME_ORDER[frame_type] + final_offset = offset + if footer is not None: + raise ResultBundleError('commit footer is not the final frame') + if metadata_value is not None and frame_type != b'C': + raise ResultBundleError('result metadata is not immediately before the commit footer') + if frame_count == 1 and frame_type != b'H': + raise ResultBundleError('bundle header is not the first frame') + if frame_type in FRAME_TYPES and not isinstance(value, dict): + raise ResultBundleError('bundle typed frame must contain a JSON object') + if frame_type == b'E' and not isinstance(value.get('error'), str): + raise ResultBundleError('bundle error frame is invalid') + if frame_type == b'H': + if header_value is not None: + raise ResultBundleError('bundle contains duplicate headers') + header_value = value + elif frame_type == b'M': + if metadata_value is not None: + raise ResultBundleError('bundle contains duplicate metadata') + metadata_value = value + elif frame_type == b'C': + footer = value + content_bytes = offset - len(framed) + continue + elif frame_type == b'D': + try: + envelope = decode_diagnostic_envelope(canonical_json_bytes(value)) + except (TypeError, ValueError, UnicodeError) as exc: + raise ResultBundleError('bundle diagnostic frame is invalid') from exc + if envelope.diagnostic_uid in diagnostic_uids: + raise ResultBundleError('bundle diagnostic identity is duplicated') + if header_value is None or ( + envelope.reservation_id != int(header_value.get('reservation_id') or 0) + or envelope.scan_event_id != str(header_value.get('scan_event_id') or '') + or envelope.source != str(header_value.get('source') or '') + ): + raise ResultBundleError('bundle diagnostic identity conflicts with its header') + diagnostic_uids.add(envelope.diagnostic_uid) + if envelope.assignment_outcome is not AssignmentOutcome.ACCEPTED: + raise ResultBundleError( + 'bundle diagnostic assignment outcome is invalid' + ) + diagnostic_scan_outcomes.append(envelope.scan_outcome.value) + counts[b'D'] += 1 + if counts[b'D'] > MAX_DIAGNOSTICS_PER_ASSIGNMENT: + raise ResultBundleError('bundle typed frame count exceeds its bound') + envelope_bytes = len(encode_diagnostic_envelope(envelope)) + diagnostic_bytes += len(framed) + diagnostic_aggregate_bytes += envelope_bytes + 1 + if ( + diagnostic_aggregate_bytes > MAX_DIAGNOSTIC_AGGREGATE_BYTES + ): + raise ResultBundleError('bundle diagnostic aggregate exceeds its byte bound') + elif frame_type in counts: + counts[frame_type] += 1 + limit = { + b'F': MAX_FINDING_FRAMES, + b'E': MAX_ERROR_FRAMES, + b'D': MAX_DIAGNOSTICS_PER_ASSIGNMENT, + b'K': MAX_CANDIDATE_FRAMES, + }[frame_type] + if counts[frame_type] > limit: + raise ResultBundleError('bundle typed frame count exceeds its bound') + digest.update(framed) + if not isinstance(header_value, dict) or not isinstance(metadata_value, dict) or not isinstance(footer, dict): + raise ResultBundleError('bundle is missing required header, metadata, or footer') + if frame_count < 3 or footer.get('format_version') != FORMAT_VERSION: + raise ResultBundleError('bundle footer version is invalid') + expected_scan_outcome = { + 'clean': 'clean', + 'found': 'found', + 'degraded': 'degraded', + 'error': 'error', + 'skipped': 'skipped', + }.get(str(metadata_value.get('status') or 'clean'), 'error') + if any( + outcome != expected_scan_outcome + for outcome in diagnostic_scan_outcomes + ): + raise ResultBundleError( + 'bundle diagnostic scan outcome conflicts with result metadata' + ) + base_footer_fields = { + 'format_version', 'content_sha256', 'content_bytes', 'byte_count', + 'frame_count', 'finding_count', 'error_count', 'candidate_count', + } + expected_footer_fields = ( + base_footer_fields | {'diagnostic_count', 'diagnostic_bytes'} + if counts[b'D'] else base_footer_fields + ) + if set(footer) != expected_footer_fields: + raise ResultBundleError('bundle footer shape is invalid') + expected = { + 'content_sha256': digest.hexdigest(), + 'content_bytes': content_bytes, + 'byte_count': final_offset, + 'frame_count': frame_count, + 'finding_count': counts[b'F'], + 'error_count': counts[b'E'], + 'candidate_count': counts[b'K'], + } + if counts[b'D']: + expected.update({ + 'diagnostic_count': counts[b'D'], + 'diagnostic_bytes': diagnostic_bytes, + }) + if footer.get('content_sha256') != expected['content_sha256']: + raise ResultBundleError('bundle content hash mismatch') + for key in ( + 'content_bytes', 'byte_count', 'frame_count', 'finding_count', + 'error_count', 'candidate_count', + ): + value = footer.get(key) + if isinstance(value, bool) or not isinstance(value, int) or value != expected[key]: + raise ResultBundleError(f'bundle footer {key} mismatch') + diagnostic_footer_fields = {'diagnostic_count', 'diagnostic_bytes'} & set(footer) + if counts[b'D']: + if diagnostic_footer_fields != {'diagnostic_count', 'diagnostic_bytes'}: + raise ResultBundleError('bundle footer diagnostic accounting is missing') + for key in ('diagnostic_count', 'diagnostic_bytes'): + value = footer.get(key) + if isinstance(value, bool) or not isinstance(value, int) or value != expected[key]: + raise ResultBundleError(f'bundle footer {key} mismatch') + elif diagnostic_footer_fields: + raise ResultBundleError('D-less bundle has unexpected diagnostic accounting') + bundle_id = _validated_id(header_value.get('bundle_id'), 'bundle_id') + event_id = _validated_id(header_value.get('scan_event_id'), 'scan_event_id') + reservation_id = header_value.get('reservation_id') + if isinstance(reservation_id, bool) or not isinstance(reservation_id, int) or reservation_id <= 0: + raise ResultBundleError('invalid reservation_id') + self._validated = BundleMetadata( + reservation_id=reservation_id, + bundle_id=bundle_id, + scan_event_id=event_id, + scan_event_hash=expected['content_sha256'], + relative_path='', + actual_bytes=final_offset, + frame_count=frame_count, + finding_count=counts[b'F'], + error_count=counts[b'E'], + candidate_count=counts[b'K'], + header=header_value, + result_metadata=metadata_value, + diagnostic_count=counts[b'D'], + diagnostic_bytes=diagnostic_bytes, + ) + self._validated_fingerprint = fingerprint + return self._validated + + def _values(self, wanted): + validated = self.validate() + digest = hashlib.sha256(MAGIC) + footer = None + for frame_type, value, framed, _ in self._iter_frames( + self._validated_fingerprint + ): + if frame_type == b'C': + footer = value + else: + digest.update(framed) + if frame_type == wanted: + yield value + if ( + digest.hexdigest() != validated.scan_event_hash + or not isinstance(footer, dict) + or footer.get('content_sha256') != validated.scan_event_hash + ): + raise ResultBundleError('bundle content changed after validation') + + def iter_findings(self): + return self._values(b'F') + + def iter_errors(self): + for value in self._values(b'E'): + yield value.get('error') if isinstance(value, dict) else value + + def iter_candidates(self): + return self._values(b'K') + + def iter_diagnostics(self): + for value in self._values(b'D'): + envelope = decode_diagnostic_envelope(canonical_json_bytes(value)) + yield json.loads(encode_diagnostic_envelope(envelope).decode('ascii')) + + def effective_diagnostics(self): + validated = self.validate() + if validated.diagnostic_count: + return tuple(self.iter_diagnostics()) + metadata = self.metadata() + header = self.header() + envelopes = build_legacy_error_frame_diagnostics( + reservation_id=validated.reservation_id, + scan_event_id=validated.scan_event_id, + slot_id=0, + source=str(header.get('source') or ''), + timestamp=( + metadata.get('timestamp') or metadata.get('scan_started_at') + ), + errors=tuple(self.iter_errors()), + retryable=bool(metadata.get('retryable', False)), + attempt=max(1, int(metadata.get('attempt') or 1)), + ) + return tuple( + json.loads(encode_diagnostic_envelope(envelope).decode('ascii')) + for envelope in envelopes + ) + + def metadata(self): + self.validate() + if self._file_fingerprint() != self._validated_fingerprint: + raise ResultBundleError('bundle changed after validation') + return dict(self.validate().result_metadata) + + def header(self): + self.validate() + if self._file_fingerprint() != self._validated_fingerprint: + raise ResultBundleError('bundle changed after validation') + return dict(self.validate().header) diff --git a/app/result_ingester.py b/app/result_ingester.py new file mode 100644 index 0000000..21560d5 --- /dev/null +++ b/app/result_ingester.py @@ -0,0 +1,605 @@ +import sys + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('result ingester could not disable bytecode writes') + +import argparse +import json +import logging +import os +import time + +from lifecycle_authority import require_active_supervisor_child +from paths import apply_path_config +from process_identity import current_process_identity, exact_process_identity_state +from result_bundle import ( + ResultBundleError, + ResultBundleReader, + bundle_partial_relative_path, +) +from runtime_security import ( + PrivatePathState, + durable_publish, + durable_unlink, + ensure_private_directory, + private_file_ready, + inspect_private_relative_path, + require_private_directory, + sha256_file, +) +from scanner_db import ( + DockerCoverageDispositionConflictError, + DockerFindingAttributionLimitError, + ScanEventConflictError, + ScannerDB, +) + + +logger = logging.getLogger(__name__) + + +class ResultIngester: + def __init__( + self, db, bundle_root, supervisor_instance_id, lease_seconds=300, fault=None, + quarantine_max_items=10000, quarantine_max_bytes=1024 * 1024 * 1024, + metadata_retention_days=30, metadata_retirement_batch=100, + recover_expired_ready=False, + ): + self.db = db + self.bundle_root = require_private_directory(bundle_root, create=False) + self.supervisor_instance_id = str(supervisor_instance_id) + self.lease_seconds = max(30, int(lease_seconds)) + self.fault = fault + self.lease = None + self.recovery_after_id = 0 + self.quarantine_max_items = max(0, int(quarantine_max_items)) + self.quarantine_max_bytes = max(0, int(quarantine_max_bytes)) + self.metadata_retention_seconds = max(1, int(metadata_retention_days)) * 86400 + self.metadata_retirement_batch = min(500, max(1, int(metadata_retirement_batch))) + self.next_metadata_retirement = 0.0 + self.recover_expired_ready = recover_expired_ready is True + + def _inject(self, stage, value=None): + if self.fault is not None: + self.fault(stage, value) + + def start(self): + self.db.require_runtime_safety_schema() + self.db.require_final_cutover() + identity = current_process_identity() + self.lease = self.db.acquire_pipeline_lease( + 'result_ingester', self.supervisor_instance_id, identity, + lease_seconds=self.lease_seconds, initial_state='recovering', + ) + if not self.lease: + raise RuntimeError('another result ingester owns the singleton advisory lock') + self.reconcile_terminal_artifacts() + self.retire_terminal_metadata() + self.recover() + if not self.heartbeat('ready'): + raise RuntimeError('result ingester ready lease publication failed') + return self + + def heartbeat(self, state='ready', error=''): + return self.db.heartbeat_pipeline_lease( + 'result_ingester', self.lease['generation'], self.lease['lease_token'], + lease_seconds=self.lease_seconds, state=state, error=error, + ) + + def stop(self, error=''): + if self.lease: + released = self.db.release_pipeline_lease( + 'result_ingester', self.lease['generation'], self.lease['lease_token'], + state='failed' if error else 'released', error=error, + ) + self.lease = None + return released + return True + + def _path(self, relative): + normalized = str(relative or '').replace('/', os.sep) + path = os.path.abspath(os.path.join(self.bundle_root, normalized)) + if os.path.commonpath((self.bundle_root, path)) != self.bundle_root or path == self.bundle_root: + raise ResultBundleError('bundle database path escapes its configured root') + return path + + def _quarantine_path(self, reservation): + bundle_id = str(reservation['bundle_id']) + return os.path.join( + self.bundle_root, 'quarantine', bundle_id[:2], f'{bundle_id}.trb', + ) + + def _quarantine_relative_path(self, reservation): + bundle_id = str(reservation['bundle_id']) + return f'quarantine/{bundle_id[:2]}/{bundle_id}.trb' + + def _ensure_quarantine_shard(self, reservation): + bundle_id = str(reservation['bundle_id']) + return ensure_private_directory( + os.path.join(self.bundle_root, 'quarantine', bundle_id[:2]), + reject_reparse=True, + ) + + def _inspect(self, relative_path): + return inspect_private_relative_path(self.bundle_root, relative_path) + + def _defer_reservation_cleanup(self, reservation, error): + try: + self.db.defer_result_reservation_cleanup(reservation['id'], str(error)) + except Exception: + logger.warning( + 'reservation cleanup backoff could not be recorded: %s', reservation['id'] + ) + + def reconcile_terminal_artifacts(self, max_pages=100): + if not hasattr(self.db, 'bundle_terminal_temp_artifacts'): + return + for _ in range(max(1, int(max_pages))): + rows = self.db.bundle_terminal_temp_artifacts(100) + if not rows: + return + progressed = False + for row in rows: + try: + inspection = self._inspect(row['relative_path']) + if inspection.state == PrivatePathState.UNKNOWN: + self.db.defer_pipeline_artifact_cleanup( + row['id'], inspection.detail or 'artifact storage state is unknown', + ) + continue + if inspection.state == PrivatePathState.PRESENT: + if not private_file_ready(inspection.path): + self.db.defer_pipeline_artifact_cleanup( + row['id'], 'artifact is not an exact private file', + ) + continue + durable_unlink(inspection.path) + inspection = self._inspect(row['relative_path']) + if inspection.state == PrivatePathState.ABSENT: + self.db.mark_pipeline_artifact_deleted(row['id']) + progressed = True + else: + self.db.defer_pipeline_artifact_cleanup( + row['id'], 'artifact unlink was not confirmed', + ) + except OSError as exc: + try: + self.db.defer_pipeline_artifact_cleanup(row['id'], str(exc)) + except Exception: + logger.warning( + 'artifact cleanup backoff could not be recorded: %s', row['id'] + ) + continue + if not progressed: + return + if len(rows) < 100: + return + return + + def retire_terminal_metadata(self): + if time.monotonic() < self.next_metadata_retirement: + return + self.next_metadata_retirement = time.monotonic() + 60 + try: + self.db.retire_admission_intents( + self.metadata_retention_seconds, self.metadata_retirement_batch, + ) + self.db.retire_deleted_pipeline_artifacts( + self.metadata_retention_seconds, self.metadata_retirement_batch, + ) + except Exception as exc: + logger.warning('bounded terminal metadata retirement deferred: %s', type(exc).__name__) + + def quarantine(self, reservation, ready_path, reason_code, detail): + ready_relative = str(reservation['ready_relative_path']).replace('\\', '/') + quarantine_relative = self._quarantine_relative_path(reservation) + self._ensure_quarantine_shard(reservation) + ready = self._inspect(ready_relative) + quarantine = self._inspect(quarantine_relative) + if ready.state == PrivatePathState.UNKNOWN or quarantine.state == PrivatePathState.UNKNOWN: + raise ResultBundleError('bundle quarantine path state is unknown') + if ready.state == PrivatePathState.PRESENT and quarantine.state == PrivatePathState.PRESENT: + raise ResultBundleError('both ready and quarantine paths exist for one reservation') + if ready.state == PrivatePathState.ABSENT and quarantine.state == PrivatePathState.ABSENT: + raise ResultBundleError('bundle disappeared before quarantine') + quarantine_path = quarantine.path + ensure_private_directory(os.path.dirname(quarantine_path), reject_reparse=True) + source_path = ready.path if ready.state == PrivatePathState.PRESENT else quarantine.path + if not private_file_ready(source_path): + raise ResultBundleError('bundle quarantine source is not an exact private file') + byte_count = os.path.getsize(source_path) + payload_hash = sha256_file(source_path) if byte_count else '' + relative = quarantine_relative + quarantine_id = self.db.quarantine_result_bundle( + reservation['id'], reason_code, detail, relative, + payload_sha256=payload_hash, byte_count=byte_count, + quarantine_max_items=self.quarantine_max_items, + quarantine_max_bytes=self.quarantine_max_bytes, + physical_confirmed=ready.state != PrivatePathState.PRESENT, + ) + if ready.state == PrivatePathState.PRESENT: + durable_publish(ready.path, quarantine_path) + confirmed = self._inspect(quarantine_relative) + if confirmed.state != PrivatePathState.PRESENT: + raise ResultBundleError('bundle quarantine publication was not confirmed') + quarantine_id = self.db.quarantine_result_bundle( + reservation['id'], reason_code, detail, relative, + payload_sha256=payload_hash, byte_count=byte_count, + quarantine_max_items=self.quarantine_max_items, + quarantine_max_bytes=self.quarantine_max_bytes, + physical_confirmed=True, + ) + return quarantine_id + + def recover(self, page_size=100, max_pages=100): + pages = 0 + while pages < max(1, int(max_pages)): + rows = self.db.active_result_reservations(self.recovery_after_id, page_size) + if not rows: + self.recovery_after_id = 0 + return pages + pages += 1 + for reservation in rows: + self.recovery_after_id = int(reservation['id']) + if ( + reservation['state'] == 'scanning' + and str(reservation.get('assignment_kind') or 'local') == 'remote' + ): + continue + ready_relative = str(reservation['ready_relative_path']).replace('\\', '/') + ready = self._inspect(ready_relative) + identity_state = None + if reservation['state'] == 'scanning': + identity_state = exact_process_identity_state( + reservation['producer_pid'], reservation['producer_creation_time'], + reservation['producer_executable'], + ) + if ( + ready.state == PrivatePathState.UNKNOWN + and identity_state in ('dead', 'reused') + ): + try: + ensure_private_directory( + os.path.join( + self.bundle_root, 'ready', + str(reservation['bundle_id'])[:2], + ), + reject_reparse=True, + ) + ensure_private_directory( + os.path.join( + self.bundle_root, 'tmp', + str(reservation['bundle_id'])[:2], + ), + reject_reparse=True, + ) + ready = self._inspect(ready_relative) + except OSError as exc: + self._defer_reservation_cleanup(reservation, exc) + continue + quarantine_relative = self._quarantine_relative_path(reservation) + try: + self._ensure_quarantine_shard(reservation) + except OSError as exc: + self._defer_reservation_cleanup(reservation, exc) + continue + quarantine = self._inspect(quarantine_relative) + if quarantine.state == PrivatePathState.PRESENT: + try: + self.db.quarantine_result_bundle( + reservation['id'], + reservation.get('last_error_code') or 'recovered_quarantine', + reservation.get('last_error_detail') or 'recovered deterministic quarantine file', + quarantine_relative, + payload_sha256=sha256_file(quarantine.path), + byte_count=os.path.getsize(quarantine.path), + quarantine_max_items=self.quarantine_max_items, + quarantine_max_bytes=self.quarantine_max_bytes, + physical_confirmed=True, + ) + except OSError as exc: + self._defer_reservation_cleanup(reservation, exc) + continue + if quarantine.state == PrivatePathState.UNKNOWN: + self._defer_reservation_cleanup( + reservation, quarantine.detail or 'quarantine state unknown', + ) + continue + if ready.state == PrivatePathState.UNKNOWN: + self._defer_reservation_cleanup( + reservation, ready.detail or 'ready state unknown', + ) + continue + prepared_quarantine = self.db.pending_result_bundle_quarantine( + reservation['id'] + ) + if prepared_quarantine and ready.state == PrivatePathState.PRESENT: + try: + self.quarantine( + reservation, ready.path, + prepared_quarantine['reason_code'], + prepared_quarantine.get('reason_detail') or 'recovered prepared quarantine', + ) + except OSError as exc: + self._defer_reservation_cleanup(reservation, exc) + continue + if reservation['state'] == 'scanning': + if ready.state == PrivatePathState.PRESENT: + try: + metadata = ResultBundleReader(ready.path).validate() + recovered = metadata.as_dict() + recovered['relative_path'] = str( + reservation['ready_relative_path'] + ).replace('\\', '/') + marked_ready = self.db.mark_result_bundle_ready( + reservation['id'], recovered, + ) + if ( + not marked_ready + and self.recover_expired_ready + and identity_state in ('dead', 'reused') + ): + self.db.recover_expired_result_bundle_ready( + reservation['id'], recovered, + ) + except OSError as exc: + self._defer_reservation_cleanup(reservation, exc) + except (ValueError, ResultBundleError) as exc: + try: + self.quarantine( + reservation, ready.path, 'bundle_validation_failed', str(exc), + ) + except OSError as cleanup_exc: + self._defer_reservation_cleanup(reservation, cleanup_exc) + continue + if identity_state in ('dead', 'reused'): + try: + ensure_private_directory( + os.path.join(self.bundle_root, 'ready', str(reservation['bundle_id'])[:2]), + reject_reparse=True, + ) + ensure_private_directory( + os.path.join(self.bundle_root, 'tmp', str(reservation['bundle_id'])[:2]), + reject_reparse=True, + ) + except OSError as exc: + self._defer_reservation_cleanup(reservation, exc) + continue + ready = self._inspect(ready_relative) + if ready.state != PrivatePathState.ABSENT: + continue + if identity_state in ('dead', 'reused'): + partial_relative = bundle_partial_relative_path( + reservation['bundle_id'], reservation['reservation_token'], + ).replace(os.sep, '/') + partial = self._inspect(partial_relative) + if partial.state == PrivatePathState.UNKNOWN: + self._defer_reservation_cleanup( + reservation, partial.detail or 'partial state unknown', + ) + continue + if partial.state == PrivatePathState.PRESENT: + if not private_file_ready(partial.path): + continue + try: + durable_unlink(partial.path) + except OSError as exc: + self._defer_reservation_cleanup(reservation, exc) + continue + partial = self._inspect(partial_relative) + if partial.state != PrivatePathState.ABSENT: + self._defer_reservation_cleanup( + reservation, 'partial unlink was not confirmed', + ) + continue + self.db.refund_uncommitted_reservation( + reservation['id'], + { + 'pid': reservation['producer_pid'], + 'creation_time': reservation['producer_creation_time'], + 'executable': reservation['producer_executable'], + }, + f'producer identity is {identity_state} and exact ready path is absent', + partial_absence_confirmed=True, + ) + elif reservation['state'] in ('ready', 'ingesting'): + if ready.state == PrivatePathState.ABSENT: + self.db.quarantine_result_bundle( + reservation['id'], 'ready_bundle_missing', + 'database ready row has a definitively absent exact ready path', + '', byte_count=0, + quarantine_max_items=self.quarantine_max_items, + quarantine_max_bytes=self.quarantine_max_bytes, + physical_confirmed=True, + ) + elif reservation['state'] == 'db_committed': + bundle = self.db.result_bundle_for_reservation(reservation['id']) + if not bundle: + continue + if ready.state == PrivatePathState.PRESENT: + if not private_file_ready(ready.path): + continue + try: + durable_unlink(ready.path) + except OSError as exc: + self._defer_reservation_cleanup(reservation, exc) + continue + ready = self._inspect(ready_relative) + if ready.state == PrivatePathState.ABSENT: + event = self.db.confirm_scan_event( + reservation['scan_event_id'], bundle['scan_event_hash'], + ) + if event: + self.db.acknowledge_removed_bundle( + reservation['id'], reservation['scan_event_id'], + bundle['scan_event_hash'], + ) + return pages + + def process_one(self): + claimed = self.db.claim_ready_result_bundle( + self.lease['generation'], self.lease['lease_token'], self.lease_seconds, + ) + if not claimed: + return False + reservation = claimed['reservation'] + bundle = claimed['bundle'] + ready_relative = str(bundle['relative_path']).replace('\\', '/') + ready = self._inspect(ready_relative) + try: + if bundle['state'] == 'db_committed': + event = self.db.confirm_scan_event(bundle['scan_event_id'], bundle['scan_event_hash']) + if not event: + raise ResultBundleError('db_committed bundle has no exact authoritative event') + else: + if ready.state == PrivatePathState.UNKNOWN: + return False + if ready.state == PrivatePathState.ABSENT: + self.db.quarantine_result_bundle( + reservation['id'], 'ready_bundle_missing', + 'claimed ready bundle is definitively absent', '', + quarantine_max_items=self.quarantine_max_items, + quarantine_max_bytes=self.quarantine_max_bytes, + physical_confirmed=True, + ) + return True + self._inject('before_validation', reservation) + reader = ResultBundleReader(ready.path) + validated = reader.validate() + self._inject('after_validation', validated) + if ( + validated.bundle_id != str(bundle['bundle_id']) + or validated.scan_event_id != str(bundle['scan_event_id']) + or validated.scan_event_hash != str(bundle['scan_event_hash']) + or validated.actual_bytes != int(bundle['actual_bytes']) + ): + raise ResultBundleError('validated bundle totals conflict with its claimed database row') + self._inject('before_db_commit', reservation) + self.db.ingest_result_bundle(reader, reservation, bundle) + self._inject('after_db_commit', reservation) + event = self.db.confirm_scan_event(bundle['scan_event_id'], bundle['scan_event_hash']) + self._inject('after_confirmation', event) + if not event: + raise ResultBundleError('database commit was not confirmed by exact event ID and hash') + ready = self._inspect(ready_relative) + if ready.state == PrivatePathState.UNKNOWN: + return False + if ready.state == PrivatePathState.PRESENT: + self._inject('before_unlink', reservation) + durable_unlink(ready.path) + self._inject('after_unlink', reservation) + ready = self._inspect(ready_relative) + if ready.state != PrivatePathState.ABSENT: + return False + self._inject('before_capacity_release', reservation) + if not self.db.acknowledge_removed_bundle( + reservation['id'], bundle['scan_event_id'], bundle['scan_event_hash'], + ): + raise RuntimeError('bundle capacity acknowledgement was not fenced') + self._inject('after_capacity_release', reservation) + return True + except DockerCoverageDispositionConflictError as exc: + ready = self._inspect(ready_relative) + if ready.state == PrivatePathState.PRESENT: + try: + self.quarantine( + reservation, ready.path, + 'docker_coverage_disposition_conflict', str(exc), + ) + except OSError as cleanup_exc: + self._defer_reservation_cleanup(reservation, cleanup_exc) + return False + return True + except DockerFindingAttributionLimitError as exc: + ready = self._inspect(ready_relative) + if ready.state == PrivatePathState.PRESENT: + try: + self.quarantine( + reservation, ready.path, + 'docker_attribution_limit_exceeded', str(exc), + ) + except OSError as cleanup_exc: + self._defer_reservation_cleanup(reservation, cleanup_exc) + return False + return True + except ScanEventConflictError as exc: + ready = self._inspect(ready_relative) + if ready.state == PrivatePathState.PRESENT: + try: + self.quarantine(reservation, ready.path, 'scan_event_hash_conflict', str(exc)) + except OSError as cleanup_exc: + self._defer_reservation_cleanup(reservation, cleanup_exc) + return False + return True + except (ResultBundleError, ValueError) as exc: + ready = self._inspect(ready_relative) + if ready.state == PrivatePathState.PRESENT: + try: + self.quarantine(reservation, ready.path, 'bundle_validation_failed', str(exc)) + except OSError as cleanup_exc: + self._defer_reservation_cleanup(reservation, cleanup_exc) + return False + return True + raise + except OSError as exc: + self._defer_reservation_cleanup(reservation, exc) + return False + + +def parse_args(): + parser = argparse.ArgumentParser(description='Singleton durable result bundle ingester') + parser.add_argument('--config', required=True) + return parser.parse_args() + + +def main(): + metadata = require_active_supervisor_child(child_kind='result-ingester', require_dsn=True) + args = parse_args() + import yaml + + with open(args.config, 'r', encoding='utf-8') as handle: + config = apply_path_config(yaml.safe_load(handle) or {}, args.config) + global_config = config.get('global') or {} + db = ScannerDB(db_url=global_config['database_url'], initialize=False) + if not db.enabled: + raise SystemExit('result ingester PostgreSQL connection is unavailable') + db.set_application_name('truf-result-ingester') + worker = ResultIngester( + db, global_config['result_bundle_dir'], metadata['instance_id'], + lease_seconds=int(((config.get('supervisor') or {}).get('result_ingester') or {}).get('lease_seconds', 300)), + quarantine_max_items=int(global_config.get('pipeline_quarantine_max_items', 10000)), + quarantine_max_bytes=int(global_config.get('pipeline_quarantine_max_bytes', 1024 * 1024 * 1024)), + metadata_retention_days=int(global_config.get('pipeline_metadata_retention_days', 30)), + metadata_retirement_batch=int(global_config.get('pipeline_metadata_retirement_batch', 100)), + ) + error = '' + try: + worker.start() + idle = max(0.05, float(((config.get('supervisor') or {}).get('result_ingester') or {}).get('poll_sec', 0.2))) + next_heartbeat = time.monotonic() + worker.lease_seconds / 3 + while True: + worker.retire_terminal_metadata() + worker.reconcile_terminal_artifacts(max_pages=1) + worker.recover(max_pages=1) + processed = worker.process_one() + if time.monotonic() >= next_heartbeat: + if not worker.heartbeat('ready'): + raise RuntimeError('result ingester heartbeat fence was lost') + next_heartbeat = time.monotonic() + worker.lease_seconds / 3 + if not processed: + time.sleep(idle) + except KeyboardInterrupt: + pass + except BaseException as exc: + error = f'{type(exc).__name__}: {exc}' + raise + finally: + try: + worker.stop(error) + finally: + db.close() + + +if __name__ == '__main__': + main() diff --git a/app/result_spool.py b/app/result_spool.py new file mode 100644 index 0000000..add6c38 --- /dev/null +++ b/app/result_spool.py @@ -0,0 +1,1073 @@ +import contextlib +import hashlib +import json +import logging +import math +import os +import re +import shutil +import stat +import threading +import time +import uuid +from dataclasses import dataclass +from datetime import datetime, timezone +from itertools import islice + +from process_identity import current_process_identity +from runtime_security import ( + PrivateFileError, + durable_move, + durable_replace, + durable_unlink, + harden_private_file, + is_reparse_point, + private_directory_ready, + private_file_ready, + reject_reparse_components, + require_private_directory, +) + + +EVENT_ID_RE = re.compile(r'^[A-Za-z0-9][A-Za-z0-9._-]{15,127}$') +INTERNAL_TEMP_RE = re.compile( + r'^(?P.+\.json)\.[1-9][0-9]*\.[1-9][0-9]*\.[1-9][0-9]*\.tmp$' +) +RESERVATION_FINAL_RE = re.compile(r'^[0-9a-f]{32}\.json$') +LOCK_FILENAME = '.spool.lock' +QUARANTINE_DIRNAME = 'quarantine' +RESERVATION_DIRNAME = 'reservations' +MAX_DIAGNOSTIC_BYTES = 64 * 1024 +MAX_RESERVATION_BYTES = 64 * 1024 +DEFAULT_MAX_EVENT_BYTES = 192 * 1024 * 1024 + +logger = logging.getLogger(__name__) +_PROCESS_LOCKS = {} +_PROCESS_LOCKS_GUARD = threading.Lock() + + +class ResultSpoolError(RuntimeError): + pass + + +class SpoolCapacityError(ResultSpoolError): + pass + + +class SpoolTransientCapacityError(SpoolCapacityError): + def __init__(self, message, state): + self.state = dict(state or {}) + super().__init__(message) + + +class SpoolBlockedError(ResultSpoolError): + pass + + +class SpoolBackpressureError(SpoolBlockedError): + pass + + +class SpoolContentionError(ResultSpoolError): + pass + + +class SpoolCorruptionError(SpoolBlockedError): + pass + + +class SpoolHashConflictError(SpoolBlockedError): + pass + + +@dataclass(frozen=True) +class SpoolEvent: + path: str + envelope: dict + + @property + def event_id(self): + return self.envelope['scan_event_id'] + + @property + def event_hash(self): + return self.envelope['scan_event_hash'] + + +def canonical_event_bytes(envelope): + if not isinstance(envelope, dict): + raise ValueError('scan event envelope must be an object') + payload = {key: value for key, value in envelope.items() if key != 'scan_event_hash'} + return json.dumps(payload, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str).encode('utf-8') + + +def scan_event_hash(envelope): + return hashlib.sha256(canonical_event_bytes(envelope)).hexdigest() + + +def prepare_scan_event(envelope): + if not isinstance(envelope, dict): + raise ValueError('scan event envelope must be an object') + if envelope.get('version') != 1: + raise ValueError('unsupported scan event envelope version') + prepared = dict(envelope) + event_id = str(prepared.get('scan_event_id') or '') + if not EVENT_ID_RE.fullmatch(event_id): + raise ValueError('scan_event_id has an invalid format') + expected = scan_event_hash(prepared) + supplied = str(prepared.get('scan_event_hash') or '') + if supplied and supplied != expected: + raise SpoolHashConflictError(f'scan event {event_id} has a conflicting payload hash') + prepared['scan_event_hash'] = expected + return prepared + + +def _serialized_event(envelope): + return json.dumps(envelope, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str).encode('utf-8') + b'\n' + + +def _is_link_or_junction(path): + return is_reparse_point(path) + + +class ResultSpool: + def __init__( + self, + directory, + max_event_bytes=DEFAULT_MAX_EVENT_BYTES, + max_events=10000, + max_total_bytes=2 * 1024 * 1024 * 1024, + min_free_bytes=1024 * 1024 * 1024, + lock_timeout_sec=30, + ): + self.directory = os.path.abspath(os.fspath(directory)) + self.quarantine_directory = os.path.join(self.directory, QUARANTINE_DIRNAME) + self.reservation_directory = os.path.join(self.directory, RESERVATION_DIRNAME) + self.max_event_bytes = int(max_event_bytes) + self.max_events = int(max_events) + self.max_total_bytes = int(max_total_bytes) + self.min_free_bytes = int(min_free_bytes) + if self.max_event_bytes <= 0 or self.max_events <= 0 or self.max_total_bytes < self.max_event_bytes: + raise ValueError('result spool limits must be positive and total bytes must cover one event') + if self.min_free_bytes < 0: + raise ValueError('result spool minimum free bytes cannot be negative') + self.lock_timeout_sec = max(0.1, float(lock_timeout_sec)) + with _PROCESS_LOCKS_GUARD: + self._process_lock = _PROCESS_LOCKS.setdefault(os.path.normcase(self.directory), threading.Lock()) + require_private_directory(self.directory, create=True) + require_private_directory(self.quarantine_directory, create=True) + require_private_directory(self.reservation_directory, create=True) + self._ensure_lock_file() + + @property + def lock_path(self): + return os.path.join(self.directory, LOCK_FILENAME) + + def _ensure_lock_file(self): + if os.path.lexists(self.lock_path) and _is_link_or_junction(self.lock_path): + raise PrivateFileError(f'result spool lock cannot be a link: {self.lock_path}') + existed = os.path.exists(self.lock_path) + flags = os.O_RDWR | os.O_CREAT + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + descriptor = os.open(self.lock_path, flags, 0o600) + try: + if os.fstat(descriptor).st_size == 0: + os.write(descriptor, b'0') + os.fsync(descriptor) + finally: + os.close(descriptor) + if existed: + if not private_file_ready(self.lock_path): + raise PrivateFileError(f'result spool lock is not private: {self.lock_path}') + else: + harden_private_file(self.lock_path) + + @contextlib.contextmanager + def _locked(self): + self._verify_directories() + if not self._process_lock.acquire(timeout=self.lock_timeout_sec): + raise SpoolContentionError('timed out acquiring in-process result spool lock') + handle = None + try: + handle = open(self.lock_path, 'r+b', buffering=0) + deadline = time.monotonic() + self.lock_timeout_sec + while True: + try: + if os.name == 'nt': + import msvcrt + + handle.seek(0) + msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1) + else: + import fcntl + + fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + break + except (OSError, BlockingIOError): + if time.monotonic() >= deadline: + raise SpoolContentionError('timed out acquiring result spool lock') + time.sleep(0.05) + self._recover_internal_temporaries() + yield + finally: + if handle is not None: + try: + if os.name == 'nt': + import msvcrt + + handle.seek(0) + msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1) + else: + import fcntl + + fcntl.flock(handle.fileno(), fcntl.LOCK_UN) + except OSError: + pass + handle.close() + self._process_lock.release() + + def _verify_directories(self): + if _is_link_or_junction(self.directory) or not private_directory_ready(self.directory): + raise PrivateFileError(f'result spool directory is absent or not private: {self.directory}') + if _is_link_or_junction(self.quarantine_directory) or not private_directory_ready(self.quarantine_directory): + raise PrivateFileError(f'result spool quarantine is absent or not private: {self.quarantine_directory}') + if _is_link_or_junction(self.reservation_directory) or not private_directory_ready(self.reservation_directory): + raise PrivateFileError(f'result spool reservations are absent or not private: {self.reservation_directory}') + if _is_link_or_junction(self.lock_path) or not private_file_ready(self.lock_path): + raise PrivateFileError(f'result spool lock is absent or not private: {self.lock_path}') + + def _bounded_entries(self, directory, limit): + entries = [] + with os.scandir(directory) as iterator: + for entry in iterator: + if entry.name in ('.', '..'): + continue + entries.append(entry) + if len(entries) > limit: + raise SpoolBlockedError(f'result spool directory exceeds its bounded entry limit: {directory}') + return entries + + def _quarantine_entries(self): + return self._bounded_entries(self.quarantine_directory, self.max_events + 1) + + @staticmethod + def _valid_event_final_name(name): + return name.endswith('.json') and bool(EVENT_ID_RE.fullmatch(name[:-5])) + + @staticmethod + def _valid_quarantine_final_name(name): + for suffix in ('.conflict.json.reason.json', '.conflict.json'): + if name.endswith(suffix): + parts = name[:-len(suffix)].rsplit('.', 2) + return ( + len(parts) == 3 + and bool(EVENT_ID_RE.fullmatch(parts[0])) + and all(re.fullmatch(r'[1-9][0-9]*', value) for value in parts[1:]) + ) + suffix = '.quarantined.reason.json' + if name.endswith(suffix): + parts = name[:-len(suffix)].rsplit('.', 2) + return ( + len(parts) == 3 + and bool(parts[0]) + and all(re.fullmatch(r'[1-9][0-9]*', value) for value in parts[1:]) + ) + return False + + def _recover_internal_temporaries(self): + scans = ( + # Normal limits plus one crash-left temporary. A writer holds this + # same lock, so at most one internal temporary can be introduced + # between successful recovery passes. + (self.directory, self.max_events + 4, self._valid_event_final_name), + (self.reservation_directory, self.max_events + 2, RESERVATION_FINAL_RE.fullmatch), + (self.quarantine_directory, self.max_events + 2, self._valid_quarantine_final_name), + ) + for directory, limit, valid_final_name in scans: + for entry in self._bounded_entries(directory, limit): + match = INTERNAL_TEMP_RE.fullmatch(entry.name) + if not match or not valid_final_name(match.group('final')): + continue + if ( + entry.is_symlink() + or _is_link_or_junction(entry.path) + or not entry.is_file(follow_symlinks=False) + or not private_file_ready(entry.path) + ): + continue + durable_unlink(entry.path) + + def _pending_paths(self): + paths = [] + unexpected = [] + entries = self._bounded_entries(self.directory, self.max_events + 3) + for entry in entries: + if entry.name in (LOCK_FILENAME, QUARANTINE_DIRNAME, RESERVATION_DIRNAME): + continue + if entry.is_symlink() or _is_link_or_junction(entry.path): + raise SpoolCorruptionError(f'result spool contains a link or reparse point: {entry.path}') + if entry.is_file(follow_symlinks=False) and entry.name.endswith('.json'): + paths.append(entry.path) + else: + unexpected.append(entry.path) + return sorted(paths), sorted(unexpected) + + @staticmethod + def _regular_file_size(path): + reject_reparse_components(path) + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode): + raise SpoolBlockedError(f'result spool entry is not a regular file: {path}') + return int(details.st_size) + + def _load_reservations(self, recover=True): + now = int(time.time()) + records = [] + for entry in self._bounded_entries(self.reservation_directory, self.max_events + 1): + if entry.is_symlink() or _is_link_or_junction(entry.path) or not entry.is_file(follow_symlinks=False): + raise SpoolCorruptionError(f'invalid result spool reservation entry: {entry.path}') + if not entry.name.endswith('.json') or not private_file_ready(entry.path): + raise SpoolCorruptionError(f'invalid result spool reservation file: {entry.path}') + size = self._regular_file_size(entry.path) + if size <= 0 or size > MAX_RESERVATION_BYTES: + raise SpoolCorruptionError(f'result spool reservation has an invalid size: {entry.path}') + with open(entry.path, 'rb') as handle: + payload = handle.read(MAX_RESERVATION_BYTES + 1) + try: + record = json.loads(payload.decode('utf-8')) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise SpoolCorruptionError(f'invalid result spool reservation: {entry.path}') from exc + if ( + not isinstance(record, dict) + or record.get('schema') != 1 + or str(record.get('reservation_id') or '') + '.json' != entry.name + or not isinstance(record.get('owner'), str) + or not record.get('owner') + or len(record.get('owner')) > 500 + or int(record.get('slots') or 0) < 0 + or int(record.get('bytes_per_event') or 0) <= 0 + or int(record.get('slots') or 0) > self.max_events + or int(record.get('bytes_per_event') or 0) > self.max_event_bytes + ): + raise SpoolCorruptionError(f'invalid result spool reservation metadata: {entry.path}') + identity_values = ( + record.get('owner_creation_time'), record.get('owner_executable'), + ) + binding_protocol = int(record.get('event_binding_protocol') or 0) + if ( + bool(identity_values[0]) != bool(identity_values[1]) + or (identity_values[0] and len(str(identity_values[0])) > 200) + or (identity_values[1] and len(str(identity_values[1])) > 32768) + or binding_protocol not in (0, 1) + or (binding_protocol == 1 and not all(identity_values)) + ): + raise SpoolCorruptionError(f'invalid result spool reservation owner identity: {entry.path}') + claims = record.get('claims') or [] + try: + expires_at = float(record.get('expires_at')) + recover_after = float(record.get('recover_after')) + except (TypeError, ValueError) as exc: + raise SpoolCorruptionError(f'invalid result spool reservation lease: {entry.path}') from exc + if ( + not isinstance(claims, list) + or len(claims) > int(record['slots']) + or size > 4096 + (len(claims) * 512) + or not math.isfinite(expires_at) + or not math.isfinite(recover_after) + or recover_after < expires_at + or any( + not isinstance(claim, dict) + or claim.get('queue_id') is None + or not claim.get('lease_token') + or (claim.get('event_id') and not EVENT_ID_RE.fullmatch(str(claim.get('event_id')))) + or not isinstance(claim.get('event_id_padding', ''), str) + or len(claim.get('event_id_padding', '')) > 127 + or ( + binding_protocol == 1 + and len(str(claim.get('event_id') or '')) + len(claim.get('event_id_padding', '')) != 127 + ) + or (claim.get('claim_batch') and str(claim.get('claim_batch')) != str(record.get('reservation_id'))) + or (claim.get('lease_owner') and str(claim.get('lease_owner')) != str(record.get('owner'))) + for claim in claims + ) + ): + raise SpoolCorruptionError(f'unbounded result spool reservation metadata: {entry.path}') + if recover and now >= recover_after: + logger.warning('Recovering expired result-spool reservation %s', record['reservation_id']) + durable_unlink(entry.path) + continue + records.append((entry.path, record, size)) + return records + + def _usage(self): + paths, unexpected = self._pending_paths() + total = 0 + for path in paths: + try: + total += self._regular_file_size(path) + except OSError as exc: + raise SpoolBlockedError(f'unable to inspect result spool entry {path}: {exc}') from exc + quarantine = self._quarantine_entries() + quarantine_bytes = 0 + for entry in quarantine: + if entry.is_symlink() or _is_link_or_junction(entry.path) or not entry.is_file(follow_symlinks=False): + raise SpoolCorruptionError(f'invalid quarantine entry was left untouched: {entry.path}') + quarantine_bytes += self._regular_file_size(entry.path) + reservations = self._load_reservations(recover=True) + active_reserved_claims = [ + claim + for _, record, _ in reservations + for claim in (record.get('claims') or []) + if not claim.get('event_id') or not os.path.isfile(self._event_path(str(claim['event_id']))) + ] + unbound_reserved_slots = sum( + max(0, int(record['slots']) - len(record.get('claims') or [])) + for _, record, _ in reservations + ) + reserved_slots = len(active_reserved_claims) + unbound_reserved_slots + reserved_future_bytes = sum( + ( + sum( + 1 for claim in (record.get('claims') or []) + if not claim.get('event_id') or not os.path.isfile(self._event_path(str(claim['event_id']))) + ) + + max(0, int(record['slots']) - len(record.get('claims') or [])) + ) * int(record['bytes_per_event']) + for _, record, _ in reservations + ) + reserved_bytes = sum( + size for _, _, size in reservations + ) + reserved_future_bytes + return { + 'paths': paths, + 'unexpected': unexpected, + 'pending_count': len(paths), + 'pending_bytes': total, + 'quarantine_count': len(quarantine), + 'quarantine_bytes': quarantine_bytes, + 'reservations': reservations, + 'reserved_slots': reserved_slots, + 'reserved_future_bytes': reserved_future_bytes, + 'reserved_bytes': reserved_bytes, + 'count': len(paths) + len(quarantine) + len(reservations) + reserved_slots, + 'bytes': total + quarantine_bytes + reserved_bytes, + } + + def _reservation_future_slots(self, record): + claims = record.get('claims') or [] + active_claims = sum( + 1 for claim in claims + if not claim.get('event_id') or not os.path.isfile(self._event_path(str(claim['event_id']))) + ) + return active_claims + max(0, int(record['slots']) - len(claims)) + + def _assert_not_blocked(self): + quarantine = self._quarantine_entries() + if quarantine: + raise SpoolBlockedError( + f'result spool quarantine contains {len(quarantine)} item(s): {self.quarantine_directory}' + ) + + def _check_capacity( + self, additional_bytes=0, additional_events=0, disk_bytes=None, reserved_credit=0, + reserved_event_credit=0, existing_bytes_credit=0, quarantine_source=None, + ): + usage = self._usage() + paths = usage['paths'] + unexpected = list(usage['unexpected']) + if quarantine_source: + source_key = os.path.normcase(os.path.abspath(quarantine_source)) + matching = [path for path in unexpected if os.path.normcase(os.path.abspath(path)) == source_key] + if matching: + source_size = self._regular_file_size(matching[0]) + usage['count'] += 1 + usage['bytes'] += source_size + unexpected.remove(matching[0]) + if unexpected: + path = unexpected[0] + try: + regular = stat.S_ISREG(os.stat(path, follow_symlinks=False).st_mode) + except OSError: + regular = False + if _is_link_or_junction(path) or not regular: + raise SpoolCorruptionError(f'unexpected result spool entry was left untouched: {path}') + self._quarantine_path(path, 'unexpected result spool entry') + raise SpoolCorruptionError(f'unexpected result spool entry quarantined: {path}') + requested_events = max(0, int(additional_events)) + requested_bytes = max(0, int(additional_bytes)) + if requested_events > self.max_events or requested_bytes > self.max_total_bytes: + raise SpoolCapacityError('result spool reservation request exceeds its configured maximum') + effective_count = usage['count'] - max(0, int(reserved_event_credit)) + requested_events + effective_bytes = usage['bytes'] - max(0, int(existing_bytes_credit)) + requested_bytes + capacity_state = { + 'current_bytes': int(usage['bytes']), + 'current_count': int(usage['count']), + 'reserved_future_bytes': int(usage['reserved_future_bytes']), + 'reserved_slots': int(usage['reserved_slots']), + 'reservation_count': len(usage['reservations']), + 'requested_bytes': requested_bytes, + 'requested_events': requested_events, + 'max_total_bytes': self.max_total_bytes, + 'max_events': self.max_events, + 'headroom_bytes': self.max_total_bytes - int(usage['bytes']), + } + try: + free = shutil.disk_usage(self.directory).free + except OSError as exc: + raise SpoolCapacityError(f'unable to inspect result spool free space: {exc}') from exc + required_disk = ( + max(0, int(usage['reserved_future_bytes']) - max(0, int(reserved_credit))) + + max(0, int(additional_bytes if disk_bytes is None else disk_bytes)) + ) + if free - max(0, required_disk) < self.min_free_bytes: + raise SpoolCapacityError('result spool minimum free-space reserve would be violated') + if effective_count > self.max_events: + if usage['reserved_slots'] > 0: + raise SpoolTransientCapacityError('result spool event capacity is temporarily reserved', capacity_state) + raise SpoolCapacityError('result spool event-count limit reached') + if effective_bytes > self.max_total_bytes: + if usage['reserved_future_bytes'] > 0: + raise SpoolTransientCapacityError('result spool byte capacity is temporarily reserved', capacity_state) + raise SpoolCapacityError('result spool total-byte limit reached') + + def assert_claims_allowed(self): + with self._locked(): + self._assert_not_blocked() + self._check_capacity() + pending = self._pending_paths()[0] + if pending: + self._read_event(pending[0]) + if pending: + raise SpoolBackpressureError(f'result spool still contains {len(pending)} pending event(s)') + return True + + def backpressure_state(self): + """Return bounded metadata while validating the oldest pending event.""" + with self._locked(): + self._assert_not_blocked() + usage = self._usage() + paths = usage['paths'] + oldest_age_sec = 0 + if paths: + self._read_event(paths[0]) + oldest_mtime = min(os.stat(path, follow_symlinks=False).st_mtime for path in paths) + oldest_age_sec = max(0, int(time.time() - oldest_mtime)) + return { + 'pending_count': int(usage['pending_count']), + 'pending_bytes': int(usage['pending_bytes']), + 'oldest_pending_age_sec': oldest_age_sec, + 'reservation_count': len(usage['reservations']), + 'reserved_slots': int(usage['reserved_slots']), + 'reserved_future_bytes': int(usage['reserved_future_bytes']), + 'headroom_bytes': self.max_total_bytes - int(usage['bytes']), + } + + def reservation_snapshot(self): + """Return validated reservation metadata for exact DB ownership checks.""" + with self._locked(): + self._assert_not_blocked() + usage = self._usage() + return { + 'reserved_future_bytes': int(usage['reserved_future_bytes']), + 'reserved_slots': int(usage['reserved_slots']), + 'reservation_count': len(usage['reservations']), + 'reservations': [dict(record, claims=[dict(claim) for claim in record.get('claims') or []]) for _, record, _ in usage['reservations']], + } + + def _event_path(self, event_id): + return os.path.join(self.directory, f'{event_id}.json') + + def _reservation_path(self, reservation_id): + return os.path.join(self.reservation_directory, f'{reservation_id}.json') + + def _write_private_bytes(self, final_path, payload): + reject_reparse_components(os.path.dirname(os.path.abspath(final_path))) + if os.path.lexists(final_path): + reject_reparse_components(final_path) + temporary = f'{final_path}.{os.getpid()}.{threading.get_ident()}.{time.time_ns()}.tmp' + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + descriptor = os.open(temporary, flags, 0o600) + try: + harden_private_file(temporary) + with os.fdopen(descriptor, 'wb') as handle: + descriptor = None + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + if not private_file_ready(temporary): + raise PrivateFileError(f'result spool temporary ACL changed: {temporary}') + durable_replace(temporary, final_path) + if not private_file_ready(final_path): + raise PrivateFileError(f'result spool event ACL changed during replace: {final_path}') + finally: + if descriptor is not None: + os.close(descriptor) + try: + if os.path.lexists(temporary): + reject_reparse_components(temporary) + durable_unlink(temporary) + except OSError: + pass + + def _save_reservation(self, record): + payload = _serialized_event(record) + if len(payload) > MAX_RESERVATION_BYTES: + raise SpoolCapacityError('result spool reservation metadata is too large') + self._write_private_bytes(self._reservation_path(record['reservation_id']), payload) + + def _replace_reservation(self, path, previous, replacement): + previous_size = self._regular_file_size(path) + previous_future = self._reservation_future_slots(previous) * int(previous['bytes_per_event']) + previous_events = 1 + self._reservation_future_slots(previous) + replacement_payload = _serialized_event(replacement) + if len(replacement_payload) > MAX_RESERVATION_BYTES: + raise SpoolCapacityError('result spool reservation metadata is too large') + replacement_future = self._reservation_future_slots(replacement) * int(replacement['bytes_per_event']) + replacement_events = 1 + self._reservation_future_slots(replacement) + self._check_capacity( + len(replacement_payload) + replacement_future, + replacement_events, + disk_bytes=len(replacement_payload) + replacement_future, + reserved_credit=previous_future, + reserved_event_credit=previous_events, + existing_bytes_credit=previous_size + previous_future, + ) + self._write_private_bytes(path, replacement_payload) + + def _reservation_record(self, reservation_id): + expected = self._reservation_path(reservation_id) + for path, record, _ in self._load_reservations(recover=True): + if path == expected: + return path, record + raise SpoolCapacityError(f'result spool reservation is absent or expired: {reservation_id}') + + def reserve_claims(self, owner, count, lease_seconds, bytes_per_event=None): + count = max(0, int(count or 0)) + if not owner or count <= 0: + raise ValueError('claim reservation requires an owner and positive count') + bytes_per_event = int(bytes_per_event or self.max_event_bytes) + if bytes_per_event <= 0 or bytes_per_event > self.max_event_bytes: + raise ValueError('claim reservation per-event bound is invalid') + lease_seconds = max(60, int(lease_seconds or 0)) + now = int(time.time()) + identity = current_process_identity() + reservation_id = uuid.uuid4().hex + record = { + 'schema': 1, + 'reservation_id': reservation_id, + 'owner': str(owner)[:500], + 'owner_pid': os.getpid(), + 'owner_creation_time': str(identity.creation_time), + 'owner_executable': str(identity.executable), + 'event_binding_protocol': 1, + 'slots': count, + 'bytes_per_event': bytes_per_event, + 'claims': [], + 'created_at': now, + 'expires_at': now + lease_seconds, + 'recover_after': now + (lease_seconds * 2), + } + metadata_bytes = len(_serialized_event(record)) + with self._locked(): + self._assert_not_blocked() + if self._pending_paths()[0]: + raise SpoolBackpressureError('result spool contains pending events; drain them before new claims') + self._check_capacity( + count * bytes_per_event + metadata_bytes, + count + 1, + disk_bytes=count * bytes_per_event + metadata_bytes, + ) + self._save_reservation(record) + return reservation_id + + def bind_claims(self, reservation_id, claims): + normalized = [] + for claim in claims or []: + queue_id = claim.get('id') if isinstance(claim, dict) else claim['id'] + lease_token = claim.get('lease_token') if isinstance(claim, dict) else claim['lease_token'] + if queue_id is None or not lease_token: + raise ValueError('reserved claim requires queue ID and lease token') + claim_batch = claim.get('claim_batch') if isinstance(claim, dict) else claim['claim_batch'] if 'claim_batch' in claim.keys() else None + lease_owner = claim.get('lease_owner') if isinstance(claim, dict) else claim['lease_owner'] if 'lease_owner' in claim.keys() else None + if claim_batch and str(claim_batch) != str(reservation_id): + raise ValueError('database claim batch does not match its spool reservation') + normalized.append({ + 'queue_id': int(queue_id), 'lease_token': str(lease_token), + 'claim_batch': str(reservation_id), 'lease_owner': str(lease_owner or ''), + 'event_id': '', 'event_id_padding': '0' * 127, + }) + with self._locked(): + path, record = self._reservation_record(reservation_id) + if any(claim.get('lease_owner') and claim['lease_owner'] != record['owner'] for claim in normalized): + raise SpoolCapacityError('database claim owner does not match its spool reservation') + if len(normalized) > int(record['slots']): + raise SpoolCapacityError('database returned more claims than were reserved') + if not normalized: + durable_unlink(path) + return None + previous = dict(record, claims=[dict(claim) for claim in record.get('claims') or []]) + record['claims'] = normalized + record['slots'] = len(normalized) + self._replace_reservation(path, previous, record) + return reservation_id + + def renew_reservation(self, reservation_id, lease_seconds): + lease_seconds = max(60, int(lease_seconds or 0)) + with self._locked(): + try: + _, record = self._reservation_record(reservation_id) + except SpoolCapacityError: + return False + now = int(time.time()) + previous = dict(record, claims=[dict(claim) for claim in record.get('claims') or []]) + record['expires_at'] = now + lease_seconds + record['recover_after'] = now + (lease_seconds * 2) + self._replace_reservation(self._reservation_path(reservation_id), previous, record) + return True + + def release_reservation(self, reservation_id): + if not reservation_id: + return False + with self._locked(): + path = self._reservation_path(reservation_id) + if not os.path.lexists(path): + return False + reject_reparse_components(path) + durable_unlink(path) + return True + + def release_reserved_claim(self, reservation_id, queue_id, lease_token): + if not reservation_id: + return False + with self._locked(): + try: + path, record = self._reservation_record(reservation_id) + except SpoolCapacityError: + return False + remaining = [ + claim for claim in record.get('claims') or [] + if not ( + int(claim.get('queue_id') or -1) == int(queue_id) + and str(claim.get('lease_token') or '') == str(lease_token or '') + ) + ] + if len(remaining) == len(record.get('claims') or []): + return False + if not remaining: + durable_unlink(path) + else: + previous = dict(record, claims=[dict(claim) for claim in record.get('claims') or []]) + record['claims'] = remaining + record['slots'] = len(remaining) + self._replace_reservation(path, previous, record) + return True + + def stopped_owner_unused_reservations(self, source, process_identity): + """Return only exact stopped-owner reservations with no durable event.""" + source = str(source or '') + with self._locked(): + output = [] + for _, record, _ in self._load_reservations(recover=True): + if ( + int(record.get('event_binding_protocol') or 0) != 1 + or str(record.get('owner') or '').split(':', 1)[0] != source + or int(record.get('owner_pid') or 0) != int(process_identity.pid) + or str(record.get('owner_creation_time') or '') != str(process_identity.creation_time) + or os.path.normcase(str(record.get('owner_executable') or '')) + != os.path.normcase(str(process_identity.executable)) + ): + continue + claims = [dict(claim) for claim in record.get('claims') or []] + if any( + claim.get('event_id') + and os.path.isfile(self._event_path(str(claim['event_id']))) + for claim in claims + ): + continue + output.append(dict(record, claims=claims)) + return output + + def release_stopped_owner_reservation(self, reservation_id, source, process_identity): + with self._locked(): + try: + path, record = self._reservation_record(reservation_id) + except SpoolCapacityError: + return False + if ( + int(record.get('event_binding_protocol') or 0) != 1 + or str(record.get('owner') or '').split(':', 1)[0] != str(source or '') + or int(record.get('owner_pid') or 0) != int(process_identity.pid) + or str(record.get('owner_creation_time') or '') != str(process_identity.creation_time) + or os.path.normcase(str(record.get('owner_executable') or '')) + != os.path.normcase(str(process_identity.executable)) + or any( + claim.get('event_id') + and os.path.isfile(self._event_path(str(claim['event_id']))) + for claim in record.get('claims') or [] + ) + ): + return False + durable_unlink(path) + return True + + def _consume_event_reservations(self, event_id): + for path, record, _ in self._load_reservations(recover=False): + claims = [dict(claim) for claim in record.get('claims') or []] + remaining = [claim for claim in claims if str(claim.get('event_id') or '') != str(event_id)] + if len(remaining) == len(claims): + continue + if remaining: + previous = dict(record, claims=claims) + record['claims'] = remaining + record['slots'] = len(remaining) + self._replace_reservation(path, previous, record) + else: + durable_unlink(path) + + @staticmethod + def _quarantine_reason_payload(quarantined_path, reason): + return _serialized_event({ + 'quarantined_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), + 'reason': str(reason)[:2000], + 'quarantined_file': os.path.basename(quarantined_path), + }) + + def _quarantine_path(self, path, reason): + reject_reparse_components(path) + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode): + raise SpoolCorruptionError(f'refusing to move non-regular quarantine source: {path}') + name = os.path.basename(path) + destination = os.path.join( + self.quarantine_directory, + f'{name}.{time.time_ns()}.{os.getpid()}.quarantined', + ) + reason_path = destination + '.reason.json' + reason_payload = self._quarantine_reason_payload(destination, reason) + self._check_capacity( + len(reason_payload), 1, disk_bytes=len(reason_payload), quarantine_source=path, + ) + try: + harden_private_file(path) + except OSError: + pass + durable_move(path, destination) + harden_private_file(destination) + self._write_private_bytes(reason_path, reason_payload) + return destination + + def _quarantine_conflicting_envelope(self, envelope, reason): + envelope = envelope if isinstance(envelope, dict) else {} + event_id = str(envelope.get('scan_event_id') or 'unknown') + if not EVENT_ID_RE.fullmatch(event_id): + event_id = 'invalid-' + hashlib.sha256(event_id.encode('utf-8', errors='replace')).hexdigest()[:24] + path = os.path.join( + self.quarantine_directory, + f'{event_id}.{time.time_ns()}.{os.getpid()}.conflict.json', + ) + diagnostic = { + 'quarantined_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), + 'reason': str(reason)[:2000], + 'scan_event_id': event_id, + 'offered_type': type(envelope).__name__, + 'offered_keys': sorted(str(key)[:100] for key in islice(envelope, 100)), + 'payload_omitted': True, + } + payload = _serialized_event(diagnostic) + if len(payload) > MAX_DIAGNOSTIC_BYTES: + diagnostic['offered_keys'] = [] + payload = _serialized_event(diagnostic) + payload = payload[:MAX_DIAGNOSTIC_BYTES] + reason_path = path + '.reason.json' + reason_payload = self._quarantine_reason_payload(path, reason) + self._check_capacity( + len(payload) + len(reason_payload), 2, + disk_bytes=len(payload) + len(reason_payload), + ) + try: + self._write_private_bytes(path, payload) + self._write_private_bytes(reason_path, reason_payload) + except Exception: + for created in (reason_path, path): + try: + if os.path.lexists(created) and not _is_link_or_junction(created): + durable_unlink(created) + except OSError: + pass + raise + + def write_event(self, envelope, reservation_id=None, queue_id=None, lease_token=None): + try: + prepared = prepare_scan_event(envelope) + except (SpoolHashConflictError, ValueError) as exc: + visible = envelope if isinstance(envelope, dict) else {'invalid_envelope_type': type(envelope).__name__} + with self._locked(): + self._assert_not_blocked() + self._quarantine_conflicting_envelope(visible, f'invalid event offered to spool: {exc}') + if isinstance(exc, SpoolHashConflictError): + raise + raise SpoolCorruptionError(str(exc)) from exc + payload = _serialized_event(prepared) + if len(payload) > self.max_event_bytes: + with self._locked(): + self._assert_not_blocked() + self._quarantine_conflicting_envelope( + {'scan_event_id': prepared.get('scan_event_id')}, + f'offered scan event exceeds {self.max_event_bytes} byte limit; payload omitted', + ) + raise SpoolCapacityError('scan event exceeds result spool per-event byte limit') + event_id = prepared['scan_event_id'] + final_path = self._event_path(event_id) + with self._locked(): + self._assert_not_blocked() + if os.path.lexists(final_path): + reject_reparse_components(final_path) + existing = self._read_event(final_path) + if existing.envelope['scan_event_hash'] == prepared['scan_event_hash']: + return existing + self._quarantine_conflicting_envelope(prepared, 'same event ID has a different payload hash') + raise SpoolHashConflictError(f'scan event {event_id} conflicts with its pending envelope') + reservation_path = None + reservation = None + reserved_bytes = 0 + if reservation_id: + reservation_path, reservation = self._reservation_record(reservation_id) + matching = [ + claim for claim in reservation.get('claims') or [] + if int(claim.get('queue_id') or -1) == int(queue_id if queue_id is not None else -1) + and str(claim.get('lease_token') or '') == str(lease_token or '') + ] + if len(matching) != 1: + raise SpoolCapacityError('scan event does not match a live reserved claim') + bound_event_id = matching[0].get('event_id') + if bound_event_id and str(bound_event_id) != event_id: + raise SpoolHashConflictError('reserved claim is already bound to a different scan event') + reserved_bytes = int(reservation['bytes_per_event']) + if len(payload) > reserved_bytes: + raise SpoolCapacityError('scan event exceeds its reserved per-event byte limit') + previous = dict(reservation, claims=[dict(claim) for claim in reservation.get('claims') or []]) + reservation['claims'] = [ + dict( + claim, + event_id=event_id, + event_id_padding='0' * (127 - len(event_id)), + ) if claim is matching[0] else dict(claim) + for claim in reservation.get('claims') or [] + ] + self._replace_reservation(reservation_path, previous, reservation) + self._check_capacity( + len(payload), 1, + disk_bytes=len(payload), reserved_credit=reserved_bytes, + reserved_event_credit=1, + existing_bytes_credit=reserved_bytes, + ) + else: + self._check_capacity(len(payload), 1, disk_bytes=len(payload)) + self._write_private_bytes(final_path, payload) + record = self._read_event(final_path) + if reservation is not None: + remaining = [ + claim for claim in reservation.get('claims') or [] + if not ( + int(claim.get('queue_id') or -1) == int(queue_id) + and str(claim.get('lease_token') or '') == str(lease_token or '') + ) + ] + try: + if remaining: + previous = dict(reservation, claims=[dict(claim) for claim in reservation.get('claims') or []]) + reservation['claims'] = remaining + reservation['slots'] = len(remaining) + self._replace_reservation(reservation_path, previous, reservation) + else: + durable_unlink(reservation_path) + except Exception as exc: + logger.critical( + 'Result event is durable but its conservative spool reservation could not be consumed: %s', + exc, + ) + return record + + def _read_event(self, path): + reject_reparse_components(path) + if not private_file_ready(path): + try: + self._quarantine_path(path, 'event file ACL is not private') + finally: + raise SpoolCorruptionError(f'non-private result spool event quarantined: {path}') + try: + size = self._regular_file_size(path) + if size <= 0 or size > self.max_event_bytes: + raise ValueError('event file has an invalid size') + with open(path, 'rb') as handle: + payload = handle.read(self.max_event_bytes + 1) + if len(payload) > self.max_event_bytes: + raise ValueError('event file exceeds the per-event limit') + envelope = json.loads(payload.decode('utf-8')) + prepared = prepare_scan_event(envelope) + expected_name = f'{prepared["scan_event_id"]}.json' + if os.path.basename(path) != expected_name: + raise ValueError('event filename does not match scan_event_id') + if envelope.get('scan_event_hash') != prepared['scan_event_hash']: + raise ValueError('event payload hash does not match') + return SpoolEvent(path, prepared) + except (OSError, UnicodeDecodeError, ValueError, json.JSONDecodeError, SpoolHashConflictError) as exc: + if os.path.lexists(path) and not _is_link_or_junction(path): + self._quarantine_path(path, f'malformed result spool event: {exc}') + raise SpoolCorruptionError(f'malformed result spool event quarantined: {path}: {exc}') from exc + + def pending_events(self, limit=1): + limit = 1 + with self._locked(): + self._assert_not_blocked() + usage = self._usage() + paths = usage['paths'] + unexpected = usage['unexpected'] + if unexpected: + path = unexpected[0] + try: + regular = stat.S_ISREG(os.stat(path, follow_symlinks=False).st_mode) + except OSError: + regular = False + if _is_link_or_junction(path) or not regular: + raise SpoolCorruptionError(f'unexpected result spool entry was left untouched: {path}') + self._quarantine_path(path, 'unexpected result spool entry') + raise SpoolCorruptionError(f'unexpected result spool entry quarantined: {path}') + return [self._read_event(path) for path in paths[:limit]] + + def next_pending_event(self): + records = self.pending_events(limit=1) + return records[0] if records else None + + def acknowledge(self, event_id, event_hash): + if not EVENT_ID_RE.fullmatch(str(event_id or '')): + raise ValueError('scan_event_id has an invalid format') + with self._locked(): + path = self._event_path(event_id) + if not os.path.lexists(path): + return False + reject_reparse_components(path) + record = self._read_event(path) + if record.event_hash != event_hash: + self._quarantine_path(path, 'acknowledgement hash conflicts with pending event') + raise SpoolHashConflictError(f'acknowledgement hash conflicts for scan event {event_id}') + durable_unlink(path) + self._consume_event_reservations(event_id) + return True + + def quarantine_event(self, event_id, reason): + with self._locked(): + path = self._event_path(event_id) + if not os.path.lexists(path): + return None + reject_reparse_components(path) + return self._quarantine_path(path, reason) + + def quarantine_payload(self, envelope, reason): + prepared = dict(envelope or {}) + with self._locked(): + self._quarantine_conflicting_envelope(prepared, reason) + raise SpoolHashConflictError(str(reason)) diff --git a/app/runtime_bootstrap.py b/app/runtime_bootstrap.py new file mode 100644 index 0000000..d2d506e --- /dev/null +++ b/app/runtime_bootstrap.py @@ -0,0 +1,143 @@ +"""Stdlib-only pre-import boundary for canonical runtime entrypoints.""" + +import sys +import os + +if __name__ == '__main__': + if sys.platform != 'linux' or not os.path.isfile('/.dockerenv') or os.path.abspath(__file__) != '/opt/truf/app/runtime_bootstrap.py': + raise SystemExit('Docker development copy: runtime control is disabled outside the prepared container. See DOCKER_MIGRATION.md.') + import runpy + runpy.run_path('/opt/truf/app/container_runtime.py')['require_container']() + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('runtime bootstrap could not disable bytecode writes') + +import runpy +import stat + + +RUNTIME_BOOTSTRAP_ENV = 'TRUF_RUNTIME_BOOTSTRAP' +RUNTIME_BOOTSTRAP_VALUE = '1' +APPLICATION_IMPORT_SUFFIXES = ('.py', '.pyw', '.pyc', '.pyd') +SUPERVISOR_ENTRYPOINT_FLAG = '--runtime-bootstrap-entrypoint' +TARGETS = { + 'supervisor': 'supervisor.py', + 'postgres-runtime': 'postgres_runtime.py', + 'migrate-runtime-safety': 'migrate_runtime_safety.py', +} + + +def _require_isolated_startup(): + if not ( + sys.flags.isolated + and sys.flags.no_site + and sys.flags.dont_write_bytecode + and sys.dont_write_bytecode + ): + raise RuntimeError('runtime bootstrap requires isolated no-site bytecode-free startup (-I -S -B)') + + +def _canonical(path): + return os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(path)))) + + +def _is_reparse_point(path): + details = os.lstat(path) + if stat.S_ISLNK(details.st_mode): + return True + attributes = getattr(details, 'st_file_attributes', 0) + reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0) + return bool(attributes & reparse_attribute) or getattr(os.path, 'isjunction', lambda _path: False)(path) + + +def _reject_cached_bytecode(app_dir): + def raise_walk_error(exc): + raise RuntimeError(f'unable to inspect the application root: {exc}') from exc + + try: + root_details = os.lstat(app_dir) + except OSError as exc: + raise RuntimeError(f'application root is unavailable: {app_dir}') from exc + if _is_reparse_point(app_dir): + raise RuntimeError(f'application root reparse point is forbidden: {app_dir}') + if not stat.S_ISDIR(root_details.st_mode): + raise RuntimeError(f'application root is not a directory: {app_dir}') + canonical_root = _canonical(app_dir) + for current, directories, files in os.walk(app_dir, followlinks=False, onerror=raise_walk_error): + for name in directories: + candidate = os.path.join(current, name) + if _is_reparse_point(candidate): + relative = os.path.relpath(candidate, app_dir).replace(os.sep, '/') + if name.lower() == '__pycache__': + raise RuntimeError(f'application __pycache__ link is forbidden: {relative}') + raise RuntimeError(f'application directory reparse point is forbidden: {relative}') + relative_current = os.path.relpath(current, app_dir) + in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep)) + for name in files: + candidate = os.path.join(current, name) + relative = os.path.relpath(candidate, app_dir).replace(os.sep, '/') + if _is_reparse_point(candidate): + raise RuntimeError(f'application file reparse point is forbidden: {relative}') + if name.lower().endswith(APPLICATION_IMPORT_SUFFIXES): + try: + contained = os.path.commonpath((canonical_root, _canonical(candidate))) == canonical_root + except ValueError: + contained = False + if not contained: + raise RuntimeError(f'application Python authority escapes its root: {relative}') + if in_cache and name.lower().endswith('.pyc'): + raise RuntimeError(f'application __pycache__ bytecode is forbidden: {relative}') + + +def _require_supervisor_entrypoint_binding(arguments, entrypoint): + bindings = [] + for index, argument in enumerate(arguments): + text = str(argument) + if text == SUPERVISOR_ENTRYPOINT_FLAG: + if index + 1 >= len(arguments): + raise RuntimeError('supervisor runtime bootstrap entrypoint binding has no path') + bindings.append(str(arguments[index + 1])) + elif text.startswith(SUPERVISOR_ENTRYPOINT_FLAG + '='): + bindings.append(text.split('=', 1)[1]) + if len(bindings) != 1: + raise RuntimeError('supervisor runtime requires exactly one explicit bootstrap entrypoint binding') + binding = bindings[0] + if not os.path.isabs(binding) or _canonical(binding) != _canonical(entrypoint): + raise RuntimeError('supervisor runtime bootstrap entrypoint binding is not canonical supervisor.py') + + +def main(): + _require_isolated_startup() + if len(sys.argv) < 3 or sys.argv[2] != '--': + raise RuntimeError('usage: runtime_bootstrap.py -- ') + target_name = str(sys.argv[1]).strip().lower() + target_file = TARGETS.get(target_name) + if not target_file: + raise RuntimeError(f'unsupported canonical runtime target: {target_name}') + + app_dir = os.path.dirname(os.path.abspath(__file__)) + _reject_cached_bytecode(app_dir) + + entrypoint = _canonical(os.path.join(app_dir, target_file)) + arguments = list(sys.argv[3:]) + if target_name == 'supervisor': + _require_supervisor_entrypoint_binding(arguments, entrypoint) + + child_namespace = runpy.run_path(os.path.join(app_dir, 'child_bootstrap.py')) + enable_dependencies = child_namespace.get('_enable_dependency_paths') + if not callable(enable_dependencies): + raise RuntimeError('authenticated dependency path bootstrap is unavailable') + enable_dependencies(target_name) + + os.environ[RUNTIME_BOOTSTRAP_ENV] = RUNTIME_BOOTSTRAP_VALUE + sys.path.insert(0, app_dir) + sys.argv = [entrypoint, *arguments] + runpy.run_path(entrypoint, run_name='__main__') + + +if __name__ == '__main__': + try: + main() + except Exception as exc: + raise SystemExit(f'canonical runtime bootstrap rejected launch: {exc}') from exc diff --git a/app/runtime_document.py b/app/runtime_document.py new file mode 100644 index 0000000..3d0d24e --- /dev/null +++ b/app/runtime_document.py @@ -0,0 +1,1461 @@ +from collections.abc import Hashable +from dataclasses import dataclass +import hashlib +import ipaddress +import json +import math +import posixpath +import re +from urllib.parse import urlsplit + +import yaml + +from capacity_model import MAX_RESULT_BUNDLE_BYTES, validate_remote_assignment_capacity +from managed_files import ( + ManagedFileConfigurationError, + managed_file_root_registry_from_config, + normalize_managed_file_roots, +) +from query_policy import QueryPolicyError, validate_rejected_query_policy + + +_MESSAGES = { + 'invalid_input': 'YAML document input is invalid', + 'size': 'YAML document exceeds its byte bound', + 'encoding': 'YAML document is not valid UTF-8', + 'duplicate_key': 'YAML document contains a duplicate mapping key', + 'syntax': 'YAML document is invalid', + 'mapping_root': 'Runtime document root must be a mapping', + 'unknown_key': 'Runtime document contains an unsupported field', + 'type': 'Runtime document field has an invalid type', + 'bounds': 'Runtime document field exceeds its bound', + 'schema': 'Runtime document is incomplete', + 'core_profile': 'Runtime core profile is invalid', + 'auth_pool': 'Runtime credential pool is invalid', + 'reference': 'Runtime document reference is invalid', + 'capability': 'Runtime package capability is invalid', + 'deployment_path': 'Runtime deployment path is invalid', +} +_MERGE_KEY = object() +_MISSING = object() + +MAX_DOCUMENT_DEPTH = 64 +MAX_DOCUMENT_NODES = 100000 +MAX_CONFIG_DOCUMENT_BYTES = 4 * 1024 * 1024 +MAX_SECRETS_DOCUMENT_BYTES = 1024 * 1024 +MAX_MAPPING_ENTRIES = 10000 +MAX_SEQUENCE_ITEMS = 100000 +MAX_MAPPING_KEY_BYTES = 512 +MAX_SCALAR_BYTES = 1024 * 1024 +MAX_PATH_BYTES = 4096 +MAX_NAME_BYTES = 128 +MAX_AUTH_POOLS = 64 +MAX_AUTH_ENTRIES = 256 +MAX_COMPATIBILITY_PROFILES = 16 +MAX_PACKAGE_CAPABILITIES = 16 +_MIN_INTEGER = -(1 << 63) +_MAX_INTEGER = (1 << 63) - 1 + +_SAFE_DOCUMENTS = frozenset({'config', 'secrets', 'package_capabilities', 'schema'}) +_SAFE_FIELD_PATH = re.compile(r'^[a-z0-9_.\[\]-]{1,512}$') +_ENV_NAME = re.compile(r'^[A-Z_][A-Z0-9_]{0,127}$') +_PLACEHOLDER = re.compile(r'{([A-Za-z_][A-Za-z0-9_]*)}') +_CORE_SOURCES = ('gitlab', 'dockerhub', 'huggingface') +_KNOWN_CAPABILITIES = frozenset({ + ('github', 'github', 'exact_git_v1'), + ('gitlab', 'gitlab', 'exact_git_v1'), + ('dockerhub', 'docker', 'docker_direct_v1'), + ('huggingface', 'huggingface', 'huggingface_space_v1'), +}) +_SOURCE_CAPABILITIES = { + 'gitlab': ('gitlab', 'gitlab', 'exact_git_v1'), + 'dockerhub': ('dockerhub', 'docker', 'docker_direct_v1'), + 'huggingface': ('huggingface', 'huggingface', 'huggingface_space_v1'), +} +_REQUIRED_CONFIG_PATHS = ( + ('global',), + ('supervisor',), + ('sources',), + ('global', 'root_dir'), + ('global', 'project_dir'), + ('global', 'runtime_dir'), + ('global', 'postgres_data_dir'), + ('global', 'postgres_bin_dir'), + ('global', 'result_bundle_dir'), + ('global', 'work_dir'), + ('global', 'control_dir'), + ('global', 'secrets_file'), + ('supervisor', 'enabled_sources'), + ('supervisor', 'control_dir'), + ('supervisor', 'worker_api'), + ('supervisor', 'worker_api', 'enabled'), + ('supervisor', 'worker_api', 'sources'), + ('supervisor', 'worker_api', 'auth_entries'), + ('supervisor', 'worker_api', 'compatibility_profiles'), + ('sources', 'gitlab'), + ('sources', 'gitlab', 'enabled'), + ('sources', 'gitlab', 'auth_pool'), + ('sources', 'dockerhub'), + ('sources', 'dockerhub', 'enabled'), + ('sources', 'dockerhub', 'auth_pool'), + ('sources', 'huggingface'), + ('sources', 'huggingface', 'enabled'), + ('sources', 'huggingface', 'auth_pool'), +) +_FIXED_PATHS = { + 'root_dir': '/opt/truf', + 'project_dir': '/opt/truf/app', + 'runtime_dir': '/data/runtime-linux', + 'postgres_data_dir': '/data/postgres-linux', + 'postgres_bin_dir': '/usr/lib/postgresql/16/bin', + 'result_bundle_dir': '/data/scanner-result-bundles', + 'work_dir': '/data/scanner-work', + 'control_dir': '/run/truf/control', + 'secrets_file': '/data/config/secrets.yaml', +} +_GLOBAL_PATH_FIELDS = frozenset({ + 'result_bundle_dir', 'work_dir', 'root_dir', 'project_dir', 'runtime_dir', + 'postgres_data_dir', 'postgres_bin_dir', 'control_dir', 'secrets_file', + 'result_spool_dir', 'legacy_result_spool_dir', 'results_dir', 'queue_dir', + 'state_dir', 'log_dir', 'keycheck_dir', 'postman_cache_dir', + 'gharchive_cache_dir', 'database_path', 'state_file', 'api_proxy_file', + 'download_proxy_file', 'dashboard_db_path', 'scan_limiter_db', + 'dockerhub_tag_cache_path', 'proxy_file', 'trufflehog_path', + 'trufflehog_config', +}) +_SUPERVISOR_PATH_FIELDS = frozenset({ + 'log_dir', 'control_dir', 'instance_file', 'lock_file', 'supervisor_log', + 'status_file', 'dashboard_log', 'state_dir', +}) +_KEYCHECK_PATH_FIELDS = frozenset({ + 'input', 'proxy_file', 'keycheck_dir', 'summary_tsv', 'summary_json', + 'alive_summary_tsv', +}) +_SOURCE_PATH_FIELDS = frozenset({ + 'target_file', 'trufflehog_config', 'postman_cache_dir', + 'gharchive_cache_dir', +}) +_CONFIG_TEMPLATE_SCHEMA_SHA256 = ( + '45d957d8168c796a9d4081ab0936d83d797c4675c8d477baf63195117363a9d8' +) + + +class RuntimeDocumentError(ValueError): + def __init__(self, category, line=None, column=None, *, document=None, path=None): + self.category = str(category) + self.line = int(line) if line is not None else None + self.column = int(column) if column is not None else None + self.document = document if document in _SAFE_DOCUMENTS else None + self.path = ( + path if type(path) is str and _SAFE_FIELD_PATH.fullmatch(path) else None + ) + message = _MESSAGES[self.category] + if self.line is not None and self.column is not None: + message += f' at line {self.line}, column {self.column}' + if self.document is not None: + message += f' in {self.document}' + if self.path is not None: + message += f' at {self.path}' + super().__init__(message) + + +@dataclass(frozen=True) +class ValidatedRuntimeDocuments: + config: dict + secrets: dict + + +class _MarkedDocumentError(Exception): + def __init__(self, mark): + self.mark = mark + + +class _DuplicateKeyError(_MarkedDocumentError): + pass + + +class _DocumentSyntaxError(_MarkedDocumentError): + pass + + +class _StrictSafeLoader(yaml.SafeLoader): + def _mapping_key(self, key_node, deep): + key = None + try: + if key_node.tag == 'tag:yaml.org,2002:merge': + return _MERGE_KEY + key = self.construct_object(key_node, deep=deep) + if not isinstance(key, Hashable): + raise _DocumentSyntaxError(key_node.start_mark) + return key + finally: + key_node = deep = key = None + + def _reject_duplicate_keys(self, pairs, deep): + seen = key_node = _value_node = key = duplicate = None + try: + seen = {} + for key_node, _value_node in pairs: + key = self._mapping_key(key_node, deep) + try: + duplicate = key in seen + except Exception: + raise _DocumentSyntaxError(key_node.start_mark) from None + if duplicate: + raise _DuplicateKeyError(key_node.start_mark) + seen[key] = None + finally: + pairs = deep = seen = key_node = _value_node = key = duplicate = None + + def flatten_mapping(self, node): + try: + # SafeLoader recursively calls this method for anonymous merge sources. + self._reject_duplicate_keys(node.value, False) + return super().flatten_mapping(node) + finally: + node = None + + def construct_mapping(self, node, deep=False): + mapping = key_node = value_node = key = None + try: + if not isinstance(node, yaml.MappingNode): + raise _DocumentSyntaxError(node.start_mark) + + self.flatten_mapping(node) + self._reject_duplicate_keys(node.value, deep) + + mapping = {} + for key_node, value_node in node.value: + key = self._mapping_key(key_node, deep) + mapping[key] = self.construct_object(value_node, deep=deep) + return mapping + finally: + node = deep = mapping = key_node = value_node = key = None + + +def _location(mark): + if mark is None: + return None, None + line = getattr(mark, 'line', None) + column = getattr(mark, 'column', None) + if type(line) is not int or type(column) is not int: + return None, None + return line + 1, column + 1 + + +def _parse_yaml_document(payload): + text = None + try: + text = payload.decode('utf-8', errors='strict') + except UnicodeDecodeError: + pass + if text is None: + return None, ('encoding', None, None) + + value = None + failure = None + loader = None + try: + loader = _StrictSafeLoader(text) + value = loader.get_single_data() + except _DuplicateKeyError as exc: + failure = ('duplicate_key', *_location(exc.mark)) + except _DocumentSyntaxError as exc: + failure = ('syntax', *_location(exc.mark)) + except Exception as exc: + mark = getattr(exc, 'problem_mark', None) or getattr(exc, 'context_mark', None) + failure = ('syntax', *_location(mark)) + except BaseException: + if loader is not None: + try: + loader.dispose() + except Exception: + pass + loader = None + value = None + text = None + payload = None + raise + finally: + if loader is not None: + try: + loader.dispose() + except Exception: + failure = failure or ('syntax', None, None) + + if failure is not None: + value = None + return value, failure + + +def load_yaml_document(payload, *, max_bytes): + if type(payload) is not bytes or type(max_bytes) is not int or max_bytes <= 0: + payload = None + max_bytes = None + raise RuntimeDocumentError('invalid_input') + if len(payload) > max_bytes: + payload = None + raise RuntimeDocumentError('size') + + value = failure = None + try: + value, failure = _parse_yaml_document(payload) + finally: + payload = None + max_bytes = None + + if failure is not None: + raise RuntimeDocumentError(*failure) + return value + + +def _preview_document(payload, max_bytes, document): + try: + value = load_yaml_document(payload, max_bytes=max_bytes) + except RuntimeDocumentError as exc: + return None, ( + exc.category, exc.line, exc.column, document, exc.path, + ) + finally: + payload = None + return value, None + + +def _field_path(path): + return '.'.join(path) if path else 'root' + + +def _failure(category, document, path): + return category, document, _field_path(path) + + +def _consume(state, count=1): + state['nodes'] += count + return state['nodes'] <= MAX_DOCUMENT_NODES + + +def _bounded_text(value, limit=MAX_SCALAR_BYTES, *, nonempty=False): + try: + if type(value) is not str or '\x00' in value: + return False + try: + size = len(value.encode('utf-8', errors='strict')) + except UnicodeError: + return False + return size <= limit and (not nonempty or bool(value)) + finally: + value = None + + +def _bounded_name(value): + try: + return ( + _bounded_text(value, MAX_NAME_BYTES, nonempty=True) + and value == value.strip() + ) + finally: + value = None + + +def _normalize_string_list(value, path, state, *, max_items=MAX_SEQUENCE_ITEMS): + normalized = item = None + try: + if type(value) is not list: + return None, _failure('type', 'config', path) + if len(value) > max_items: + return None, _failure('bounds', 'config', path) + normalized = [] + for item in value: + if not _consume(state) or not _bounded_text(item, nonempty=True): + return None, _failure('bounds', 'config', path + ('item',)) + normalized.append(item) + return normalized, None + finally: + value = path = state = normalized = item = None + + +def _normalize_dynamic_config(value, path, state): + normalized = key = entry_name = profile_name = raw_profile = None + item_path = package_manifest = profile = sources = failure = item = None + query = raw_override = override = name = maximum = None + try: + if path == ('supervisor', 'worker_api', 'admin', 'managed_file_roots'): + try: + normalized, _registry = normalize_managed_file_roots(value) + except ManagedFileConfigurationError as exc: + return None, _failure(exc.category, 'config', path + exc.field) + return normalized, None + + if path in ( + ('supervisor', 'worker_api', 'sources'), + ('supervisor', 'defaults', 'extra_args'), + ): + return _normalize_string_list(value, path, state) + + if path == ('supervisor', 'worker_api', 'assignment_ttl_seconds_by_source'): + if type(value) is not dict: + return None, _failure('type', 'config', path) + if len(value) > len(_CORE_SOURCES): + return None, _failure('bounds', 'config', path) + normalized = {} + for key, item in value.items(): + item_path = path + (key,) if type(key) is str else path + ('field',) + if not _consume(state): + return None, _failure('bounds', 'config', item_path) + if key not in _CORE_SOURCES: + return None, _failure('unknown_key', 'config', item_path) + if type(item) is not int or not 60 <= item <= 7 * 24 * 60 * 60: + return None, _failure('bounds', 'config', item_path) + normalized[key] = item + return normalized, None + + if path == ('supervisor', 'worker_api', 'auth_entries'): + if type(value) is not dict: + return None, _failure('type', 'config', path) + if len(value) > len(_CORE_SOURCES) + 1: + return None, _failure('bounds', 'config', path) + normalized = {} + for key, entry_name in value.items(): + if not _consume(state) or not _bounded_name(key) or not _bounded_name(entry_name): + return None, _failure('bounds', 'config', path + ('item',)) + normalized[key] = entry_name + return normalized, None + + if path == ('supervisor', 'worker_api', 'compatibility_profiles'): + if type(value) is not dict: + return None, _failure('type', 'config', path) + if len(value) > MAX_COMPATIBILITY_PROFILES: + return None, _failure('bounds', 'config', path) + normalized = {} + for profile_name, raw_profile in value.items(): + item_path = path + ('profile',) + if not _consume(state) or not _bounded_name(profile_name): + return None, _failure('bounds', 'config', item_path) + if type(raw_profile) is not dict: + return None, _failure('type', 'config', item_path) + if set(raw_profile) - {'package_manifest', 'sources'}: + return None, _failure('unknown_key', 'config', item_path) + package_manifest = raw_profile.get('package_manifest', _MISSING) + if package_manifest is _MISSING: + return None, _failure('schema', 'config', item_path + ('package_manifest',)) + if not _consume(state) or not _bounded_text(package_manifest, MAX_PATH_BYTES, nonempty=True): + return None, _failure('bounds', 'config', item_path + ('package_manifest',)) + profile = {'package_manifest': package_manifest} + if 'sources' in raw_profile: + sources, failure = _normalize_string_list( + raw_profile['sources'], item_path + ('sources',), state, + max_items=len(_CORE_SOURCES), + ) + if failure is not None: + return None, failure + profile['sources'] = sources + normalized[profile_name] = profile + return normalized, None + + if ( + path == ('keychecks', 'env') + or len(path) == 4 + and path[0] == 'supervisor' + and path[1] == 'sources' + and path[3] == 'env' + ): + if type(value) is not dict: + return None, _failure('type', 'config', path) + if len(value) > 128: + return None, _failure('bounds', 'config', path) + normalized = {} + for key, item in value.items(): + if ( + not _consume(state) + or type(key) is not str + or _ENV_NAME.fullmatch(key) is None + or not _bounded_text(item) + ): + return None, _failure('bounds', 'config', path + ('item',)) + normalized[key] = item + return normalized, None + + if ( + len(path) == 3 + and path[0] == 'sources' + and path[2] == 'query_overrides' + ): + if type(value) is not dict: + return None, _failure('type', 'config', path) + if len(value) > MAX_MAPPING_ENTRIES: + return None, _failure('bounds', 'config', path) + normalized = {} + for query, raw_override in value.items(): + item_path = path + ('item',) + if not _consume(state) or not _bounded_text(query, 512, nonempty=True): + return None, _failure('bounds', 'config', item_path) + if type(raw_override) is not dict: + return None, _failure('type', 'config', item_path) + if set(raw_override) - {'pages', 'per_page', 'max_targets'}: + return None, _failure('unknown_key', 'config', item_path) + override = {} + for name, item in raw_override.items(): + maximum = 100000 if name == 'max_targets' else 1000 + if ( + not _consume(state) + or type(item) is not int + or not 1 <= item <= maximum + ): + return None, _failure('bounds', 'config', item_path + (name,)) + override[name] = item + normalized[query] = override + return normalized, None + + return _MISSING + finally: + value = path = state = normalized = key = entry_name = None + profile_name = raw_profile = item_path = package_manifest = profile = None + sources = failure = item = query = raw_override = override = None + name = maximum = None + + +def _normalize_scalar(value, template, path): + try: + if type(value) is not type(template): + return None, _failure('type', 'config', path) + if type(value) is int and not 0 <= value <= _MAX_INTEGER: + return None, _failure('bounds', 'config', path) + if type(value) is float and (not math.isfinite(value) or value < 0): + return None, _failure('bounds', 'config', path) + if type(value) is str and not _bounded_text(value): + return None, _failure('bounds', 'config', path) + if type(value) not in {type(None), bool, int, float, str}: + return None, _failure('type', 'config', path) + return value, None + finally: + value = template = path = None + + +def _normalize_config_shape(value, template, path, state, depth=0): + identity = dynamic = normalized = key = item = copied = failure = None + try: + if depth > MAX_DOCUMENT_DEPTH or not _consume(state): + return None, _failure('bounds', 'config', path) + + if type(value) in (dict, list): + identity = id(value) + if identity in state['active']: + return None, _failure('bounds', 'config', path) + state['active'].add(identity) + try: + dynamic = _normalize_dynamic_config(value, path, state) + if dynamic is not _MISSING: + return dynamic + + if type(template) is not type(value): + return None, _failure('type', 'config', path) + if type(value) is dict: + if len(value) > MAX_MAPPING_ENTRIES: + return None, _failure('bounds', 'config', path) + normalized = {} + for key, item in value.items(): + if ( + type(key) is not str + or not _bounded_text(key, MAX_MAPPING_KEY_BYTES, nonempty=True) + ): + return None, _failure('type', 'config', path + ('field',)) + if key not in template: + if ( + len(path) == 2 + and path[0] == 'sources' + and key in _SOURCE_PATH_FIELDS + ): + copied, failure = _normalize_scalar( + item, '', path + (key,), + ) + if failure is not None: + return None, failure + normalized[key] = copied + continue + if ( + len(path) == 3 + and path[:2] == ('supervisor', 'sources') + and key == 'enabled' + ): + copied, failure = _normalize_scalar( + item, False, path + (key,), + ) + if failure is not None: + return None, failure + normalized[key] = copied + continue + if ( + len(path) == 3 + and path[:2] == ('supervisor', 'sources') + and key == 'env' + ): + copied, failure = _normalize_dynamic_config( + item, path + (key,), state, + ) + if failure is not None: + return None, failure + normalized[key] = copied + continue + return None, _failure('unknown_key', 'config', path + ('unknown',)) + copied, failure = _normalize_config_shape( + item, template[key], path + (key,), state, depth + 1, + ) + if failure is not None: + return None, failure + normalized[key] = copied + return normalized, None + + if len(value) > MAX_SEQUENCE_ITEMS: + return None, _failure('bounds', 'config', path) + if not template: + if value: + return None, _failure('type', 'config', path) + return [], None + normalized = [] + for item in value: + copied, failure = _normalize_config_shape( + item, template[0], path + ('item',), state, depth + 1, + ) + if failure is not None: + return None, failure + normalized.append(copied) + return normalized, None + finally: + state['active'].remove(identity) + + return _normalize_scalar(value, template, path) + finally: + value = template = path = state = identity = dynamic = None + normalized = key = item = copied = failure = None + + +def _dynamic_schema_kind(path): + if path == ('supervisor', 'worker_api', 'admin', 'managed_file_roots'): + return 'managed_file_roots' + if path in ( + ('supervisor', 'worker_api', 'sources'), + ('supervisor', 'defaults', 'extra_args'), + ): + return 'string_list' + if path == ('supervisor', 'worker_api', 'assignment_ttl_seconds_by_source'): + return 'assignment_ttl_seconds_by_source' + if path == ('supervisor', 'worker_api', 'auth_entries'): + return 'auth_entries' + if path == ('supervisor', 'worker_api', 'compatibility_profiles'): + return 'compatibility_profiles' + if ( + path == ('keychecks', 'env') + or len(path) == 4 + and path[0] == 'supervisor' + and path[1] == 'sources' + and path[3] == 'env' + ): + return 'environment' + if ( + len(path) == 3 + and path[0] == 'sources' + and path[2] == 'query_overrides' + ): + return 'query_overrides' + return None + + +def _config_template_schema_identity(value, path=()): + dynamic_kind = scalar_names = None + try: + dynamic_kind = _dynamic_schema_kind(path) + if dynamic_kind is not None: + return {'dynamic': dynamic_kind} + if type(value) is dict: + return { + 'mapping': [ + [key, _config_template_schema_identity(value[key], path + (key,))] + for key in sorted(value) + ], + } + if type(value) is list: + return { + 'sequence': ( + _config_template_schema_identity(value[0], path + ('item',)) + if value else None + ), + } + scalar_names = { + type(None): 'null', bool: 'bool', int: 'int', float: 'float', str: 'str', + } + return {'scalar': scalar_names.get(type(value), 'unsupported')} + finally: + value = path = dynamic_kind = scalar_names = None + + +def _config_template_schema_hash(value): + encoded = None + try: + encoded = json.dumps( + _config_template_schema_identity(value), + ensure_ascii=True, + separators=(',', ':'), + sort_keys=True, + ).encode('ascii') + return hashlib.sha256(encoded).hexdigest() + finally: + value = encoded = None + + +def _normalize_secrets(value): + state = {'nodes': 0, 'active': set()} + pools = normalized = active = None + pool_name = raw_entries = raw_entry = entries = names = None + name = token = entry = username = None + try: + if type(value) is not dict: + return None, _failure('mapping_root', 'secrets', ()) + if set(value) - {'auth_pools'}: + return None, _failure('unknown_key', 'secrets', ('unknown',)) + pools = value.get('auth_pools', {}) + if type(pools) is not dict: + return None, _failure('type', 'secrets', ('auth_pools',)) + if len(pools) > MAX_AUTH_POOLS: + return None, _failure('bounds', 'secrets', ('auth_pools',)) + + normalized = {} + active = {id(value), id(pools)} + state['active'].update(active) + if not _consume(state, 2): + return None, _failure('bounds', 'secrets', ()) + for pool_name, raw_entries in pools.items(): + pool_path = ('auth_pools', 'pool') + if not _bounded_name(pool_name) or type(raw_entries) is not list: + return None, _failure('auth_pool', 'secrets', pool_path) + if id(raw_entries) in state['active'] or len(raw_entries) > MAX_AUTH_ENTRIES: + return None, _failure('bounds', 'secrets', pool_path) + state['active'].add(id(raw_entries)) + raw_entries_id = id(raw_entries) + entries = [] + names = set() + try: + for raw_entry in raw_entries: + entry_path = pool_path + ('entry',) + if not _consume(state) or type(raw_entry) is not dict: + return None, _failure('auth_pool', 'secrets', entry_path) + if id(raw_entry) in state['active']: + return None, _failure('bounds', 'secrets', entry_path) + state['active'].add(id(raw_entry)) + try: + if set(raw_entry) not in ({'name', 'token'}, {'name', 'username', 'token'}): + return None, _failure('auth_pool', 'secrets', entry_path) + name = raw_entry.get('name') + token = raw_entry.get('token') + if not _bounded_name(name) or not _bounded_text(token, nonempty=True): + return None, _failure('auth_pool', 'secrets', entry_path) + if name in names: + return None, _failure('auth_pool', 'secrets', entry_path + ('name',)) + names.add(name) + entry = {'name': name} + if 'username' in raw_entry: + username = raw_entry['username'] + if not _bounded_text(username, MAX_NAME_BYTES, nonempty=True): + return None, _failure('auth_pool', 'secrets', entry_path + ('username',)) + entry['username'] = username + entry['token'] = token + entries.append(entry) + finally: + state['active'].remove(id(raw_entry)) + finally: + state['active'].remove(raw_entries_id) + normalized[pool_name] = entries + return {'auth_pools': normalized}, None + finally: + value = pools = normalized = active = None + pool_name = raw_entries = raw_entry = entries = names = None + name = token = entry = username = None + state = None + + +def _get_path(value, path): + current = name = None + try: + current = value + for name in path: + if type(current) is not dict or name not in current: + return _MISSING + current = current[name] + return current + finally: + value = path = current = name = None + + +def _validate_required_fields(config): + path = None + try: + for path in _REQUIRED_CONFIG_PATHS: + if _get_path(config, path) is _MISSING: + return _failure('schema', 'config', path) + return None + finally: + config = path = None + + +def _validate_auth_references(config, secrets): + pools = sources = worker_config = raw_worker_sources = None + worker_sources = auth_entries = source_config = matches = None + docker_pool = source_name = pool_name = entry_name = None + try: + pools = secrets['auth_pools'] + sources = config['sources'] + docker_pool = sources['dockerhub'].get('auth_pool') or '' + + for source_name, source_config in sources.items(): + pool_name = source_config.get('auth_pool') + if pool_name and pool_name not in pools: + return _failure('reference', 'config', ('sources', source_name, 'auth_pool')) + + if docker_pool: + if any('username' not in entry for entry in pools[docker_pool]): + return _failure('auth_pool', 'secrets', ('auth_pools', 'pool', 'entry')) + + worker_config = config['supervisor']['worker_api'] + raw_worker_sources = worker_config['sources'] + if len(raw_worker_sources) != len(set(raw_worker_sources)): + return _failure('core_profile', 'config', ('supervisor', 'worker_api', 'sources')) + if raw_worker_sources: + if any(source not in _CORE_SOURCES for source in raw_worker_sources): + return _failure('core_profile', 'config', ('supervisor', 'worker_api', 'sources')) + worker_sources = tuple(source for source in _CORE_SOURCES if source in raw_worker_sources) + else: + worker_sources = _CORE_SOURCES + worker_config['sources'] = list(worker_sources) + + auth_entries = worker_config['auth_entries'] + if set(auth_entries) - (set(worker_sources) | {'github'}): + return _failure('reference', 'config', ('supervisor', 'worker_api', 'auth_entries')) + if set(auth_entries) - {'github', 'gitlab'}: + return _failure('reference', 'config', ('supervisor', 'worker_api', 'auth_entries')) + for source_name, entry_name in auth_entries.items(): + source_config = sources.get(source_name) + if type(source_config) is not dict or source_config.get('enabled') is not True: + return _failure('reference', 'config', ('supervisor', 'worker_api', 'auth_entries')) + pool_name = source_config.get('auth_pool') + if not pool_name or pool_name not in pools: + return _failure('reference', 'config', ('supervisor', 'worker_api', 'auth_entries')) + matches = [entry for entry in pools[pool_name] if entry['name'] == entry_name] + if len(matches) != 1: + return _failure('reference', 'config', ('supervisor', 'worker_api', 'auth_entries')) + return None + finally: + config = secrets = pools = sources = None + worker_config = raw_worker_sources = worker_sources = None + auth_entries = source_config = matches = None + docker_pool = source_name = pool_name = entry_name = None + + +def _normalize_capabilities(package_capabilities, profiles, worker_sources, enabled): + if package_capabilities is None: + package_capabilities = {} + if type(package_capabilities) is not dict: + return None, _failure('capability', 'package_capabilities', ()) + if set(package_capabilities) != set(profiles): + return None, _failure('capability', 'package_capabilities', ('profiles',)) + if enabled and not profiles: + return None, _failure( + 'capability', 'config', + ('supervisor', 'worker_api', 'compatibility_profiles'), + ) + + normalized_capabilities = {} + for profile_name, evidence in package_capabilities.items(): + path = ('profiles', 'profile') + if ( + not _bounded_name(profile_name) + or type(evidence) is not dict + or set(evidence) != {'package_manifest', 'capabilities'} + or evidence.get('package_manifest') != profiles[profile_name]['package_manifest'] + ): + return None, _failure('capability', 'package_capabilities', path) + raw_capabilities = evidence['capabilities'] + if type(raw_capabilities) is not list: + return None, _failure('capability', 'package_capabilities', path) + if not 1 <= len(raw_capabilities) <= MAX_PACKAGE_CAPABILITIES: + return None, _failure('capability', 'package_capabilities', path) + seen = set() + capabilities = [] + for raw_item in raw_capabilities: + item_path = path + ('capability',) + if type(raw_item) is not dict or set(raw_item) != { + 'source', 'platform', 'planning_kind', + }: + return None, _failure('capability', 'package_capabilities', item_path) + if any(type(raw_item[name]) is not str for name in raw_item): + return None, _failure('capability', 'package_capabilities', item_path) + capability = ( + raw_item['source'], raw_item['platform'], raw_item['planning_kind'], + ) + if capability not in _KNOWN_CAPABILITIES or capability in seen: + return None, _failure('capability', 'package_capabilities', item_path) + seen.add(capability) + capabilities.append(capability) + normalized_capabilities[profile_name] = frozenset(capabilities) + + covered = set() + for profile_name, profile in profiles.items(): + capabilities = normalized_capabilities[profile_name] + raw_sources = profile.get('sources') or [item[0] for item in capabilities] + if len(raw_sources) != len(set(raw_sources)): + return None, _failure( + 'capability', 'config', + ('supervisor', 'worker_api', 'compatibility_profiles', 'profile', 'sources'), + ) + profile_sources = tuple(source for source in _CORE_SOURCES if source in raw_sources) + if ( + not profile_sources + or set(raw_sources) != set(profile_sources) + or not set(profile_sources) <= set(worker_sources) + or any(_SOURCE_CAPABILITIES[source] not in capabilities for source in profile_sources) + ): + return None, _failure( + 'capability', 'config', + ('supervisor', 'worker_api', 'compatibility_profiles', 'profile', 'sources'), + ) + profile['sources'] = list(profile_sources) + covered.update(profile_sources) + if enabled and covered != set(worker_sources): + return None, _failure( + 'capability', 'config', + ('supervisor', 'worker_api', 'compatibility_profiles'), + ) + return normalized_capabilities, None + + +def _validate_worker_config(config, package_capabilities): + worker_config = enabled = address_text = address = integer_bounds = None + name = minimum = maximum = value = source = effective_ttl = minimum_ttl = None + ttl_overrides = None + admin_config = default = origin = marker = parsed = parsed_port = None + _normalized = failure = managed_file_roots = None + try: + worker_config = config['supervisor']['worker_api'] + enabled = worker_config['enabled'] + if type(enabled) is not bool: + return _failure('type', 'config', ('supervisor', 'worker_api', 'enabled')) + + address_text = worker_config.get('address', '127.0.0.1') + try: + address = ipaddress.ip_address(address_text) + except (TypeError, ValueError): + return _failure('bounds', 'config', ('supervisor', 'worker_api', 'address')) + if address.is_unspecified or address.is_multicast or not ( + address.is_loopback or address.is_private + ): + return _failure('bounds', 'config', ('supervisor', 'worker_api', 'address')) + + integer_bounds = { + 'port': (1024, 65535), + 'assignment_ttl_seconds': (60, 7 * 24 * 60 * 60), + 'max_bundle_bytes': ( + 1024 * 1024, + min(MAX_RESULT_BUNDLE_BYTES, config['global'].get( + 'result_bundle_max_event_bytes', MAX_RESULT_BUNDLE_BYTES, + )), + ), + 'reaper_interval_seconds': (5, 3600), + 'reaper_batch_size': (1, 1000), + 'limit_concurrency': (1, 1024), + 'body_idle_timeout_seconds': (1, 120), + 'json_body_timeout_seconds': (1, 300), + 'bundle_body_timeout_seconds': (30, 86400), + } + for name, (minimum, maximum) in integer_bounds.items(): + value = worker_config.get(name) + if type(value) is not int or not minimum <= value <= maximum: + return _failure('bounds', 'config', ('supervisor', 'worker_api', name)) + + ttl_overrides = worker_config.get('assignment_ttl_seconds_by_source', {}) + for source in worker_config['sources']: + effective_ttl = ttl_overrides.get( + source, worker_config['assignment_ttl_seconds'], + ) + minimum_ttl = ( + config['sources'][source]['timeout'] + + worker_config['bundle_body_timeout_seconds'] + + 60 + ) + if effective_ttl < minimum_ttl: + return _failure( + 'bounds', 'config', + ( + 'supervisor', 'worker_api', + 'assignment_ttl_seconds_by_source', source, + ) if source in ttl_overrides else ( + 'supervisor', 'worker_api', 'assignment_ttl_seconds', + ), + ) + + admin_config = worker_config.get('admin') or {} + if type(admin_config) is not dict or type(admin_config.get('enabled', False)) is not bool: + return _failure('type', 'config', ('supervisor', 'worker_api', 'admin')) + for name, default, minimum, maximum in ( + ('max_body_bytes', 8192, 1024, 65536), + ('snapshot_limit', 200, 1, 500), + ('requeue_limit', 100, 1, 500), + ): + value = admin_config.get(name, default) + if type(value) is not int or not minimum <= value <= maximum: + return _failure( + 'bounds', 'config', ('supervisor', 'worker_api', 'admin', name), + ) + try: + managed_file_roots = managed_file_root_registry_from_config(config) + except ManagedFileConfigurationError as exc: + return _failure( + exc.category, 'config', + ('supervisor', 'worker_api', 'admin', 'managed_file_roots') + + exc.field, + ) + if admin_config.get('enabled') is True: + origin = admin_config.get('origin') + marker = admin_config.get('edge_marker') + try: + parsed = urlsplit(origin) + parsed_port = parsed.port + except (TypeError, ValueError): + parsed = None + parsed_port = None + if ( + type(origin) is not str + or not 1 <= len(origin) <= 512 + or any(character.isspace() for character in origin) + or parsed is None + or parsed.scheme != 'https' + or not parsed.hostname + or parsed.username is not None + or parsed.password is not None + or parsed.path + or parsed.query + or parsed.fragment + or parsed_port is not None and not 1 <= parsed_port <= 65535 + ): + return _failure( + 'bounds', 'config', ('supervisor', 'worker_api', 'admin', 'origin'), + ) + if ( + type(marker) is not str + or not 32 <= len(marker) <= 512 + or any(character.isspace() for character in marker) + ): + return _failure( + 'bounds', 'config', ('supervisor', 'worker_api', 'admin', 'edge_marker'), + ) + + _normalized, failure = _normalize_capabilities( + package_capabilities, + worker_config['compatibility_profiles'], + worker_config['sources'], + enabled, + ) + return failure + finally: + config = package_capabilities = worker_config = enabled = address_text = None + address = integer_bounds = name = minimum = maximum = value = None + source = effective_ttl = minimum_ttl = ttl_overrides = None + admin_config = default = origin = marker = None + parsed = parsed_port = _normalized = failure = managed_file_roots = None + + +def _expand_path(raw_value, resolved_globals): + if not _bounded_text(raw_value, MAX_PATH_BYTES) or any( + marker in raw_value for marker in ('\\', '$', '~') + ): + return None + references = _PLACEHOLDER.findall(raw_value) + if len(references) > 32: + return None + if any(reference not in resolved_globals for reference in references): + return None + expanded = _PLACEHOLDER.sub(lambda match: resolved_globals[match.group(1)], raw_value) + if '{' in expanded or '}' in expanded or not expanded: + return None + if not _bounded_text(expanded, MAX_PATH_BYTES) or not expanded.startswith('/'): + return None + if posixpath.normpath(expanded) != expanded: + return None + return expanded + + +def _resolve_global_deployment_paths(config): + global_config = resolved = active = name = raw_value = None + try: + global_config = config['global'] + resolved = {} + active = set() + + def resolve(name): + raw_value = references = reference = value = None + try: + if name in resolved: + return resolved[name] + if name in active or name not in global_config or name not in _GLOBAL_PATH_FIELDS: + return None + raw_value = global_config[name] + if type(raw_value) is not str: + return None + active.add(name) + references = _PLACEHOLDER.findall(raw_value) + for reference in references: + if resolve(reference) is None: + active.remove(name) + return None + value = _expand_path(raw_value, resolved) + active.remove(name) + if value is not None: + resolved[name] = value + return value + finally: + name = raw_value = references = reference = value = None + + for name in _GLOBAL_PATH_FIELDS & set(global_config): + raw_value = global_config[name] + if raw_value == '': + resolved[name] = '' + continue + if resolve(name) is None: + return None, _failure('deployment_path', 'config', ('global', name)) + + return resolved, None + finally: + config = global_config = resolved = active = name = raw_value = None + + +def _resolve_package_manifest_path(config, raw_value): + resolved = failure = path = None + try: + try: + resolved, failure = _resolve_global_deployment_paths(config) + except Exception: + return None + if failure is not None: + return None + path = _expand_path(raw_value, resolved) + if path is None or not path.startswith('/data/worker-packages/'): + return None + return path + finally: + config = raw_value = resolved = failure = path = None + + +def _validate_deployment_paths(config): + global_config = resolved = failure = name = expected = runtime_root = None + value = trufflehog_config = supervisor = supervisor_paths = None + supervisor_roots = root = keychecks = keycheck_paths = None + source_name = source_config = source_paths = source_roots = None + profiles = profile = path = None + try: + global_config = config['global'] + resolved, failure = _resolve_global_deployment_paths(config) + if failure is not None: + return failure + + for name, expected in _FIXED_PATHS.items(): + if resolved.get(name) != expected: + return _failure('deployment_path', 'config', ('global', name)) + + def contained(path, root): + return path == root or path.startswith(root.rstrip('/') + '/') + + runtime_root = resolved['runtime_dir'] + for name in ( + 'result_spool_dir', 'legacy_result_spool_dir', 'results_dir', 'queue_dir', + 'state_dir', 'log_dir', 'keycheck_dir', 'postman_cache_dir', + 'gharchive_cache_dir', 'database_path', 'state_file', 'api_proxy_file', + 'download_proxy_file', 'dashboard_db_path', 'scan_limiter_db', + 'dockerhub_tag_cache_path', 'proxy_file', + ): + value = resolved.get(name) + if value and not contained(value, runtime_root): + return _failure('deployment_path', 'config', ('global', name)) + trufflehog_config = resolved.get('trufflehog_config') + if trufflehog_config and not contained(trufflehog_config, resolved['project_dir']): + return _failure('deployment_path', 'config', ('global', 'trufflehog_config')) + + def validate_section(section, fields, prefix): + values = name = raw_value = value = None + try: + values = {} + for name in fields & set(section): + raw_value = section[name] + if raw_value == '': + values[name] = '' + continue + value = _expand_path(raw_value, resolved) + if value is None: + return _failure('deployment_path', 'config', prefix + (name,)) + values[name] = value + return values + finally: + section = fields = prefix = values = name = raw_value = value = None + + supervisor = config['supervisor'] + supervisor_paths = validate_section( + supervisor, _SUPERVISOR_PATH_FIELDS, ('supervisor',), + ) + if type(supervisor_paths) is tuple: + return supervisor_paths + if supervisor_paths.get('control_dir') != _FIXED_PATHS['control_dir']: + return _failure('deployment_path', 'config', ('supervisor', 'control_dir')) + supervisor_roots = { + 'log_dir': resolved.get('log_dir'), + 'instance_file': resolved['control_dir'], + 'lock_file': resolved['control_dir'], + 'supervisor_log': resolved.get('log_dir'), + 'status_file': resolved.get('log_dir'), + 'dashboard_log': resolved.get('log_dir'), + 'state_dir': resolved.get('state_dir'), + } + for name, root in supervisor_roots.items(): + value = supervisor_paths.get(name) + if value and (not root or not contained(value, root)): + return _failure('deployment_path', 'config', ('supervisor', name)) + + keychecks = config.get('keychecks') + if type(keychecks) is dict: + keycheck_paths = validate_section(keychecks, _KEYCHECK_PATH_FIELDS, ('keychecks',)) + if type(keycheck_paths) is tuple: + return keycheck_paths + for name, value in keycheck_paths.items(): + root = resolved.get('proxy_file') if name == 'proxy_file' else resolved.get('keycheck_dir') + if value and (not root or not contained(value, root)): + return _failure('deployment_path', 'config', ('keychecks', name)) + for source_name, source_config in config['sources'].items(): + source_paths = validate_section( + source_config, _SOURCE_PATH_FIELDS, ('sources', source_name), + ) + if type(source_paths) is tuple: + return source_paths + source_roots = { + 'target_file': resolved.get('queue_dir'), + 'trufflehog_config': resolved.get('trufflehog_config'), + 'postman_cache_dir': resolved.get('postman_cache_dir'), + 'gharchive_cache_dir': resolved.get('gharchive_cache_dir'), + } + for name, value in source_paths.items(): + root = source_roots[name] + if value and (not root or not contained(value, root)): + return _failure( + 'deployment_path', 'config', ('sources', source_name, name), + ) + + profiles = supervisor['worker_api']['compatibility_profiles'] + for profile in profiles.values(): + path = _resolve_package_manifest_path(config, profile['package_manifest']) + if path is None: + return _failure( + 'deployment_path', 'config', + ('supervisor', 'worker_api', 'compatibility_profiles', 'profile', 'package_manifest'), + ) + return None + finally: + config = global_config = resolved = failure = name = expected = None + runtime_root = value = trufflehog_config = supervisor = supervisor_paths = None + supervisor_roots = root = keychecks = keycheck_paths = None + source_name = source_config = source_paths = source_roots = None + profiles = profile = path = None + + +def _validate_runtime_semantics(config, secrets, package_capabilities): + failure = enabled_sources = supervisor_sources = None + source_name = source_config = queries = overrides = None + try: + failure = _validate_required_fields(config) + if failure is not None: + return failure + + enabled_sources = config['supervisor']['enabled_sources'] + if ( + type(enabled_sources) is not list + or len(enabled_sources) != len(_CORE_SOURCES) + or set(enabled_sources) != set(_CORE_SOURCES) + ): + return _failure('core_profile', 'config', ('supervisor', 'enabled_sources')) + config['supervisor']['enabled_sources'] = list(_CORE_SOURCES) + if any(config['sources'][source]['enabled'] is not True for source in _CORE_SOURCES): + return _failure('core_profile', 'config', ('sources',)) + + try: + validate_remote_assignment_capacity(config['global']) + except ValueError as exc: + return _failure('bounds', 'config', ('global', str(exc))) + + supervisor_sources = config['supervisor'].get('sources', {}) + if any(source not in config['sources'] for source in supervisor_sources): + return _failure('reference', 'config', ('supervisor', 'sources')) + for source_name, source_config in config['sources'].items(): + queries = source_config.get('queries', []) + overrides = source_config.get('query_overrides', {}) + if any(query not in queries for query in overrides): + return _failure('reference', 'config', ('sources', source_name, 'query_overrides')) + if source_config.get('target_claim_order', 'oldest') not in ( + 'oldest', 'newest', 'balanced', + ): + return _failure( + 'core_profile', 'config', + ('sources', source_name, 'target_claim_order'), + ) + + failure = _validate_auth_references(config, secrets) + if failure is not None: + return failure + failure = _validate_worker_config(config, package_capabilities) + if failure is not None: + return failure + failure = _validate_deployment_paths(config) + if failure is not None: + return failure + + try: + validate_rejected_query_policy(config) + except QueryPolicyError: + return _failure('reference', 'config', ('query_policy',)) + return None + finally: + config = secrets = package_capabilities = None + failure = enabled_sources = supervisor_sources = None + source_name = source_config = queries = overrides = None + + +def _validate_runtime_documents_inner(config, secrets, config_template, package_capabilities): + template = normalized_config = normalized_secrets = None + template_state = state = failure = None + try: + if type(config) is not dict: + return None, _failure('mapping_root', 'config', ()) + if type(config_template) is not dict: + return None, _failure('mapping_root', 'schema', ()) + + template_state = {'nodes': 0, 'active': set()} + template, failure = _normalize_config_shape( + config_template, config_template, (), template_state, + ) + if failure is not None: + return None, _failure('schema', 'schema', ()) + if _config_template_schema_hash(template) != _CONFIG_TEMPLATE_SCHEMA_SHA256: + return None, _failure('schema', 'schema', ()) + state = {'nodes': 0, 'active': set()} + normalized_config, failure = _normalize_config_shape( + config, template, (), state, + ) + if failure is not None: + return None, failure + normalized_secrets, failure = _normalize_secrets(secrets) + if failure is not None: + return None, failure + failure = _validate_runtime_semantics( + normalized_config, normalized_secrets, package_capabilities, + ) + if failure is not None: + return None, failure + return ValidatedRuntimeDocuments( + config=normalized_config, + secrets=normalized_secrets, + ), None + except Exception: + return None, _failure('schema', 'schema', ()) + except BaseException: + config = secrets = config_template = package_capabilities = None + template = normalized_config = normalized_secrets = None + template_state = state = failure = None + raise + + +def validate_runtime_documents( + config, + secrets, + *, + config_template, + package_capabilities=None, +): + validated = failure = None + try: + validated, failure = _validate_runtime_documents_inner( + config, secrets, config_template, package_capabilities, + ) + finally: + config = None + secrets = None + config_template = None + package_capabilities = None + + if failure is not None: + raise RuntimeDocumentError( + failure[0], document=failure[1], path=failure[2], + ) + return validated + + +def _preview_runtime_documents_inner( + config_payload, + secrets_payload, + config_template_payload, + package_capabilities, +): + config = secrets = config_template = validated = None + try: + config, failure = _preview_document( + config_payload, MAX_CONFIG_DOCUMENT_BYTES, 'config', + ) + if failure is not None: + return None, failure + secrets, failure = _preview_document( + secrets_payload, MAX_SECRETS_DOCUMENT_BYTES, 'secrets', + ) + if failure is not None: + return None, failure + config_template, failure = _preview_document( + config_template_payload, MAX_CONFIG_DOCUMENT_BYTES, 'schema', + ) + if failure is not None: + return None, failure + validated, failure = _validate_runtime_documents_inner( + config, secrets, config_template, package_capabilities, + ) + if failure is not None: + return None, (failure[0], None, None, failure[1], failure[2]) + return validated, None + except Exception: + return None, ('schema', None, None, 'schema', 'root') + finally: + config_payload = None + secrets_payload = None + config_template_payload = None + package_capabilities = None + config = None + secrets = None + config_template = None + validated = None + + +def preview_runtime_documents( + config_payload, + secrets_payload, + *, + config_template_payload, + package_capabilities=None, +): + validated = failure = None + try: + validated, failure = _preview_runtime_documents_inner( + config_payload, + secrets_payload, + config_template_payload, + package_capabilities, + ) + finally: + config_payload = None + secrets_payload = None + config_template_payload = None + package_capabilities = None + + if failure is not None: + raise RuntimeDocumentError( + failure[0], failure[1], failure[2], + document=failure[3], path=failure[4], + ) + return validated diff --git a/app/runtime_document_io.py b/app/runtime_document_io.py new file mode 100644 index 0000000..f831f6b --- /dev/null +++ b/app/runtime_document_io.py @@ -0,0 +1,1442 @@ +from dataclasses import dataclass, field +import hashlib +import hmac +import json +import os +from pathlib import Path +import re +import stat + +from runtime_document import ( + MAX_CONFIG_DOCUMENT_BYTES, + MAX_SECRETS_DOCUMENT_BYTES, + RuntimeDocumentError, + _resolve_package_manifest_path, + load_yaml_document, + preview_runtime_documents, +) +from runtime_security import ( + PrivateFileLock, + durable_replace, + fsync_directory, + harden_private_file, + read_stable_root_file, + require_private_directory, + require_private_file, +) +from worker_package import ( + MAX_WORKER_PACKAGE_MANIFEST_BYTES, + load_worker_package_manifest_bytes, +) + + +MANAGED_TEMPLATE_PATH = Path(__file__).with_name('config.linux.yaml') +MANAGED_WORKER_PACKAGE_DIRECTORY = Path('/data/worker-packages') +MANAGED_SECRETS_PATH = Path('/data/config/secrets.yaml') +MANAGED_CANDIDATE_DIRECTORY = Path('/data/runtime-document-candidates') +MANAGED_CONFIG_CANDIDATE_PATH = MANAGED_CANDIDATE_DIRECTORY / 'config.yaml' +MANAGED_SECRETS_CANDIDATE_PATH = MANAGED_CANDIDATE_DIRECTORY / 'secrets.yaml' +MANAGED_CANDIDATE_LOCK_PATH = MANAGED_CANDIDATE_DIRECTORY / 'candidate.lock' + +MAX_CONFIG_DIFF_ENTRIES = 200 +MAX_CONFIG_DIFF_OUTPUT_BYTES = 65536 +MAX_CONFIG_DIFF_PATH_BYTES = 512 +MAX_CONFIG_DIFF_VALUE_BYTES = 512 +MAX_CONFIG_DIFF_ENTRY_BYTES = 1024 + +_HASH = re.compile(r'^[0-9a-f]{64}$') +_CANDIDATE_TEMPORARY = re.compile( + r'^\.(?:config|secrets)\.yaml\.[0-9a-f]{24}\.tmp$' +) +_SENSITIVE_PATH = re.compile( + r'(?:pass(?:word)?|token|secret|credential|auth|cookie)', re.IGNORECASE, +) +_SENSITIVE_CONFIG_FIELDS = frozenset({ + 'dashboard_db_url', 'database_url', 'edge_marker', 'postgres_dsn', +}) +_MISSING = object() + + +@dataclass(frozen=True) +class ManagedRuntimeDocuments: + config: dict + secrets: dict | None + config_sha256: str + secrets_sha256: str | None + + +@dataclass(frozen=True) +class RuntimeDocumentIdentity: + sha256: str + byte_count: int + present: bool + + +@dataclass(frozen=True) +class ManagedRuntimeDocumentState: + active_config: RuntimeDocumentIdentity + active_secrets: RuntimeDocumentIdentity + candidate_config: RuntimeDocumentIdentity + candidate_secrets: RuntimeDocumentIdentity + + +@dataclass(frozen=True) +class ConfigDiffEntry: + path: str + change: str + before: str | None + after: str | None + value_redacted: bool + value_truncated: bool + + +@dataclass(frozen=True) +class ConfigDiff: + entries: tuple[ConfigDiffEntry, ...] + truncated: bool + format_only_changed: bool + output_bytes: int + + +@dataclass(frozen=True) +class SecretsRedactedDiff: + document_changed: bool + semantic_changed: bool + pools_before: int + pools_after: int + pools_added: int + pools_removed: int + pools_changed: int + entries_before: int + entries_after: int + entries_added: int + entries_removed: int + pools_reordered: int + usernames_added: int + usernames_removed: int + usernames_changed: int + tokens_changed: int + + +@dataclass(frozen=True) +class ManagedRuntimeCandidatePreview: + document: str + state: ManagedRuntimeDocumentState + proposed: RuntimeDocumentIdentity + diff: ConfigDiff | SecretsRedactedDiff + + +@dataclass(frozen=True) +class ManagedRuntimeCandidateRevision: + document: str + before: ManagedRuntimeDocumentState + after: ManagedRuntimeDocumentState + proposed: RuntimeDocumentIdentity + created: bool + content_changed: bool + written: bool + diff: ConfigDiff | SecretsRedactedDiff + + +@dataclass(frozen=True) +class ManagedRuntimeApplyVerification: + action: str + state: ManagedRuntimeDocumentState + config_source: str + secrets_source: str + effective_config_sha256: str + effective_secrets_sha256: str + + +@dataclass(frozen=True) +class ManagedRuntimeEditorDocument: + document: str + source: str + text: str = field(repr=False) + state: ManagedRuntimeDocumentState + selected: RuntimeDocumentIdentity + + +@dataclass(frozen=True) +class _CandidateSnapshot: + active_config: bytes + active_secrets: bytes + candidate_config: bytes | None + candidate_secrets: bytes | None + state: ManagedRuntimeDocumentState + + +def _error_details(exc, default_document=None): + return ( + exc.category, + exc.line, + exc.column, + exc.document or default_document, + exc.path, + ) + + +def _read_private_document(path, max_bytes, document): + failed = False + payload = None + descriptor = None + try: + checked = require_private_file(os.fspath(path)) + flags = os.O_RDONLY + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + if hasattr(os, 'O_NOFOLLOW'): + flags |= os.O_NOFOLLOW + descriptor = os.open(checked, flags) + before = os.fstat(descriptor) + if not stat.S_ISREG(before.st_mode) or before.st_nlink != 1: + raise OSError('private document identity is invalid') + with os.fdopen(descriptor, 'rb') as handle: + descriptor = None + payload = handle.read(max_bytes + 1) + after = os.fstat(handle.fileno()) + require_private_file(checked) + current = os.stat(checked, follow_symlinks=False) + identity = lambda value: (value.st_dev, value.st_ino) + if ( + identity(before) != identity(after) + or identity(after) != identity(current) + or after.st_nlink != 1 + or current.st_nlink != 1 + or before.st_size != after.st_size + or after.st_size != current.st_size + or getattr(before, 'st_mtime_ns', None) != getattr(after, 'st_mtime_ns', None) + or getattr(after, 'st_mtime_ns', None) != getattr(current, 'st_mtime_ns', None) + or getattr(before, 'st_ctime_ns', None) != getattr(after, 'st_ctime_ns', None) + or ( + os.name != 'nt' + and getattr(after, 'st_ctime_ns', None) + != getattr(current, 'st_ctime_ns', None) + ) + ): + raise OSError('private document changed while it was being read') + except Exception: + failed = True + except BaseException: + payload = None + raise + finally: + if descriptor is not None: + os.close(descriptor) + if failed: + return None, ('schema', None, None, document, 'root') + if len(payload) > max_bytes: + return None, ('size', None, None, document, None) + return payload, None + + +def _read_package_manifest(path): + if os.name == 'nt': + return _read_private_document( + path, MAX_WORKER_PACKAGE_MANIFEST_BYTES, 'package_capabilities', + ) + try: + return read_stable_root_file( + path, MAX_WORKER_PACKAGE_MANIFEST_BYTES, + MANAGED_WORKER_PACKAGE_DIRECTORY, + ), None + except Exception: + return None, ('schema', None, None, 'package_capabilities', 'root') + + +def _package_capability_evidence(config): + profiles = evidence = profile_name = profile = reference = None + resolved = payload = failure = manifest = None + try: + try: + profiles = config['supervisor']['worker_api']['compatibility_profiles'] + except Exception: + return None, None + if type(profiles) is not dict: + return None, None + + evidence = {} + for profile_name, profile in profiles.items(): + if type(profile) is not dict: + return None, None + reference = profile.get('package_manifest') + resolved = _resolve_package_manifest_path(config, reference) + if resolved is None: + return None, None + try: + payload, failure = _read_package_manifest(resolved) + if failure is not None: + return None, None + manifest = load_worker_package_manifest_bytes(payload) + except Exception: + return None, None + evidence[profile_name] = { + 'package_manifest': reference, + 'capabilities': manifest['capabilities'], + } + return evidence, True + finally: + config = profiles = evidence = profile_name = profile = reference = None + resolved = payload = failure = manifest = None + + +def _validate_managed_runtime_files_inner(config_path, secrets_bytes): + config_payload = None + template_payload = None + active_secrets_payload = None + config = None + package_capabilities = None + validated = None + try: + config_payload, failure = _read_private_document( + config_path, MAX_CONFIG_DOCUMENT_BYTES, 'config', + ) + if failure is not None: + return None, failure + template_payload, failure = _read_private_document( + MANAGED_TEMPLATE_PATH, MAX_CONFIG_DOCUMENT_BYTES, 'schema', + ) + if failure is not None: + return None, failure + if secrets_bytes is None: + active_secrets_payload, failure = _read_private_document( + MANAGED_SECRETS_PATH, MAX_SECRETS_DOCUMENT_BYTES, 'secrets', + ) + if failure is not None: + return None, failure + else: + active_secrets_payload = secrets_bytes + + config = load_yaml_document( + config_payload, max_bytes=MAX_CONFIG_DOCUMENT_BYTES, + ) + package_capabilities, ready = _package_capability_evidence(config) + if ready is None: + return None, ('capability', None, None, 'package_capabilities', 'root') + validated = preview_runtime_documents( + config_payload, + active_secrets_payload, + config_template_payload=template_payload, + package_capabilities=package_capabilities, + ) + return ManagedRuntimeDocuments( + config=validated.config, + secrets=validated.secrets, + config_sha256=hashlib.sha256(config_payload).hexdigest(), + secrets_sha256=hashlib.sha256(active_secrets_payload).hexdigest(), + ), None + except RuntimeDocumentError as exc: + return None, _error_details(exc, 'config') + except Exception: + return None, ('schema', None, None, 'schema', 'root') + finally: + secrets_bytes = None + config_payload = None + template_payload = None + active_secrets_payload = None + config = None + package_capabilities = None + validated = None + + +def validate_managed_runtime_files(config_path, *, secrets_bytes=None): + validated = failure = None + try: + validated, failure = _validate_managed_runtime_files_inner( + config_path, secrets_bytes, + ) + finally: + config_path = None + secrets_bytes = None + if failure is not None: + raise RuntimeDocumentError( + failure[0], failure[1], failure[2], + document=failure[3], path=failure[4], + ) + return validated + + +def _load_managed_runtime_config_inner(config_path): + payload = config = failure = None + try: + payload, failure = _read_private_document( + config_path, MAX_CONFIG_DOCUMENT_BYTES, 'config', + ) + if failure is not None: + return None, failure + config = load_yaml_document(payload, max_bytes=MAX_CONFIG_DOCUMENT_BYTES) + if type(config) is not dict: + return None, ('mapping_root', None, None, 'config', 'root') + return ManagedRuntimeDocuments( + config=config, + secrets=None, + config_sha256=hashlib.sha256(payload).hexdigest(), + secrets_sha256=None, + ), None + except RuntimeDocumentError as exc: + return None, _error_details(exc, 'config') + except Exception: + return None, ('schema', None, None, 'config', 'root') + finally: + payload = config = failure = None + + +def load_managed_runtime_config(config_path): + loaded, failure = _load_managed_runtime_config_inner(config_path) + config_path = None + if failure is not None: + raise RuntimeDocumentError( + failure[0], failure[1], failure[2], + document=failure[3], path=failure[4], + ) + return loaded + + +def _candidate_failure(category='schema', document='schema', path='root'): + return category, None, None, document, path + + +def _document_identity(payload, present=True): + try: + return RuntimeDocumentIdentity( + sha256=hashlib.sha256(payload).hexdigest(), + byte_count=len(payload), + present=present, + ) + finally: + payload = None + + +def _read_optional_candidate(path, max_bytes, document): + if not os.path.lexists(path): + return None, None + return _read_private_document(path, max_bytes, document) + + +def _read_candidate_snapshot(config_path): + active_config = active_secrets = None + candidate_config = candidate_secrets = None + try: + active_config, failure = _read_private_document( + config_path, MAX_CONFIG_DOCUMENT_BYTES, 'config', + ) + if failure is not None: + return None, failure + active_secrets, failure = _read_private_document( + MANAGED_SECRETS_PATH, MAX_SECRETS_DOCUMENT_BYTES, 'secrets', + ) + if failure is not None: + return None, failure + candidate_config, failure = _read_optional_candidate( + MANAGED_CONFIG_CANDIDATE_PATH, MAX_CONFIG_DOCUMENT_BYTES, 'config', + ) + if failure is not None: + return None, failure + candidate_secrets, failure = _read_optional_candidate( + MANAGED_SECRETS_CANDIDATE_PATH, MAX_SECRETS_DOCUMENT_BYTES, 'secrets', + ) + if failure is not None: + return None, failure + + state = ManagedRuntimeDocumentState( + active_config=_document_identity(active_config), + active_secrets=_document_identity(active_secrets), + candidate_config=_document_identity( + candidate_config if candidate_config is not None else active_config, + candidate_config is not None, + ), + candidate_secrets=_document_identity( + candidate_secrets if candidate_secrets is not None else active_secrets, + candidate_secrets is not None, + ), + ) + return _CandidateSnapshot( + active_config=active_config, + active_secrets=active_secrets, + candidate_config=candidate_config, + candidate_secrets=candidate_secrets, + state=state, + ), None + except Exception: + return None, _candidate_failure() + except BaseException: + active_config = active_secrets = None + candidate_config = candidate_secrets = None + raise + + +def _validate_payload_pair(config_payload, secrets_payload): + template_payload = config = package_capabilities = validated = None + try: + template_payload, failure = _read_private_document( + MANAGED_TEMPLATE_PATH, MAX_CONFIG_DOCUMENT_BYTES, 'schema', + ) + if failure is not None: + return None, failure + config = load_yaml_document( + config_payload, max_bytes=MAX_CONFIG_DOCUMENT_BYTES, + ) + package_capabilities, ready = _package_capability_evidence(config) + if ready is None: + return None, _candidate_failure( + 'capability', 'package_capabilities', 'root', + ) + validated = preview_runtime_documents( + config_payload, + secrets_payload, + config_template_payload=template_payload, + package_capabilities=package_capabilities, + ) + return validated, None + except RuntimeDocumentError as exc: + return None, _error_details(exc, 'config') + except Exception: + return None, _candidate_failure() + finally: + config_payload = None + secrets_payload = None + template_payload = None + config = None + package_capabilities = None + validated = None + + +def _prepare_candidate_store(): + existed = os.path.lexists(MANAGED_CANDIDATE_DIRECTORY) + require_private_directory(MANAGED_CANDIDATE_DIRECTORY, create=True) + if not existed: + fsync_directory(Path(MANAGED_CANDIDATE_DIRECTORY).parent) + + +def _verify_candidate_lock(): + checked = require_private_file(MANAGED_CANDIDATE_LOCK_PATH) + details = os.stat(checked, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or details.st_nlink != 1: + raise OSError('candidate lock identity is invalid') + if os.name != 'nt' and stat.S_IMODE(details.st_mode) != 0o600: + raise OSError('candidate lock mode is invalid') + + +def _cleanup_candidate_temporaries(): + removed = False + with os.scandir(MANAGED_CANDIDATE_DIRECTORY) as entries: + for entry in entries: + if _CANDIDATE_TEMPORARY.fullmatch(entry.name) is None: + continue + details = os.stat(entry.path, follow_symlinks=False) + if ( + not stat.S_ISREG(details.st_mode) + or details.st_nlink != 1 + or (os.name != 'nt' and stat.S_IMODE(details.st_mode) != 0o600) + ): + raise OSError('candidate temporary file identity is invalid') + require_private_file(entry.path) + os.unlink(entry.path) + removed = True + if removed: + fsync_directory(MANAGED_CANDIDATE_DIRECTORY) + + +def _valid_expected_hash(value): + return type(value) is str and _HASH.fullmatch(value) is not None + + +def _same_hash(actual, expected): + return hmac.compare_digest(actual, expected) + + +def _truncate_text(value, max_bytes): + encoded = None + shortened = None + try: + encoded = value.encode('utf-8') + if len(encoded) <= max_bytes: + return value, False + shortened = encoded[:max(0, max_bytes - 3)].decode('utf-8', 'ignore') + '...' + return shortened, True + finally: + value = None + encoded = None + shortened = None + + +def _render_diff_value(value): + try: + return json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ) + finally: + value = None + + +def _config_diff(before, after, before_hash, after_hash): + entries = [] + output_bytes = 2 # JSON list delimiters. + truncated = False + stack = [('', before, after)] + old = new = old_text = new_text = entry = None + try: + while stack and not truncated: + path, old, new = stack.pop() + if old == new: + continue + old_empty_container = type(old) in (dict, list) and not old + new_empty_container = type(new) in (dict, list) and not new + leaf_container_change = ( + old_empty_container + and (new is _MISSING or type(new) is not type(old)) + ) or ( + new_empty_container + and (old is _MISSING or type(old) is not type(new)) + ) + if not leaf_container_change and type(old) is dict and type(new) is dict: + keys = sorted(set(old) | set(new), reverse=True) + for key in keys: + child = f'{path}.{key}' if path else str(key) + stack.append(( + child, old.get(key, _MISSING), new.get(key, _MISSING), + )) + continue + if not leaf_container_change and type(old) is list and type(new) is list: + for index in range(max(len(old), len(new)) - 1, -1, -1): + stack.append(( + f'{path}[{index}]', + old[index] if index < len(old) else _MISSING, + new[index] if index < len(new) else _MISSING, + )) + continue + if ( + not leaf_container_change + and old is _MISSING + and type(new) in (dict, list) + and new + ): + stack.append((path, {} if type(new) is dict else [], new)) + continue + if ( + not leaf_container_change + and new is _MISSING + and type(old) in (dict, list) + and old + ): + stack.append((path, old, {} if type(old) is dict else [])) + continue + if not leaf_container_change and ( + type(old) in (dict, list) or type(new) in (dict, list) + ): + if new is not _MISSING: + stack.append(( + path, + {} if type(new) is dict else [] if type(new) is list else _MISSING, + new, + )) + if old is not _MISSING: + stack.append(( + path, + old, + {} if type(old) is dict else [] if type(old) is list else _MISSING, + )) + continue + path = path or 'root' + if len(entries) >= MAX_CONFIG_DIFF_ENTRIES: + truncated = True + continue + if len(path.encode('utf-8')) > MAX_CONFIG_DIFF_PATH_BYTES: + truncated = True + continue + lowered = '.' + path.lower() + '.' + field_name = re.split(r'[.\[]', path)[-1].rstrip(']').lower() + redacted = ( + '.env.' in lowered + or _SENSITIVE_PATH.search(path) is not None + or field_name in _SENSITIVE_CONFIG_FIELDS + or type(old) is str + or type(new) is str + ) + value_truncated = False + if redacted: + old_text = None if old is _MISSING else '[redacted]' + new_text = None if new is _MISSING else '[redacted]' + else: + old_text = None if old is _MISSING else _render_diff_value(old) + new_text = None if new is _MISSING else _render_diff_value(new) + if old_text is not None: + old_text, cut = _truncate_text( + old_text, MAX_CONFIG_DIFF_VALUE_BYTES, + ) + value_truncated = value_truncated or cut + if new_text is not None: + new_text, cut = _truncate_text( + new_text, MAX_CONFIG_DIFF_VALUE_BYTES, + ) + value_truncated = value_truncated or cut + change = ( + 'added' if old is _MISSING + else 'removed' if new is _MISSING + else 'changed' + ) + entry = ConfigDiffEntry( + path=path, + change=change, + before=old_text, + after=new_text, + value_redacted=redacted, + value_truncated=value_truncated, + ) + entry_bytes = len(json.dumps( + entry.__dict__, ensure_ascii=True, sort_keys=True, + separators=(',', ':'), + ).encode('utf-8')) + entry_output_bytes = entry_bytes + (1 if entries else 0) + if ( + entry_bytes > MAX_CONFIG_DIFF_ENTRY_BYTES + or output_bytes + entry_output_bytes > MAX_CONFIG_DIFF_OUTPUT_BYTES + ): + truncated = True + continue + entries.append(entry) + output_bytes += entry_output_bytes + return ConfigDiff( + entries=tuple(entries), + truncated=truncated, + format_only_changed=( + before_hash != after_hash and before == after and not entries + ), + output_bytes=output_bytes, + ) + finally: + before = after = old = new = None + old_text = new_text = entry = None + stack = None + entries = None + + +def _secrets_diff(before, after, before_hash, after_hash): + before_pools = after_pools = None + old_entries = new_entries = None + old_username = new_username = None + entry = entries = None + before_names = after_names = common_pools = None + old_names = new_names = None + pool_name = name = None + try: + before_pools = before.get('auth_pools', {}) + after_pools = after.get('auth_pools', {}) + before_names = set(before_pools) + after_names = set(after_pools) + common_pools = before_names & after_names + entries_before = entries_after = 0 + for entries in before_pools.values(): + entries_before += len(entries) + for entries in after_pools.values(): + entries_after += len(entries) + entries_added = entries_removed = 0 + usernames_added = usernames_removed = usernames_changed = 0 + tokens_changed = 0 + + for pool_name in before_names | after_names: + old_entries = {} + for entry in before_pools.get(pool_name, []): + old_entries[entry['name']] = entry + new_entries = {} + for entry in after_pools.get(pool_name, []): + new_entries[entry['name']] = entry + old_names = set(old_entries) + new_names = set(new_entries) + entries_added += len(new_names - old_names) + entries_removed += len(old_names - new_names) + for name in new_names - old_names: + usernames_added += int('username' in new_entries[name]) + for name in old_names - new_names: + usernames_removed += int('username' in old_entries[name]) + for name in old_names & new_names: + old_username = old_entries[name].get('username', _MISSING) + new_username = new_entries[name].get('username', _MISSING) + if old_username is _MISSING and new_username is not _MISSING: + usernames_added += 1 + elif old_username is not _MISSING and new_username is _MISSING: + usernames_removed += 1 + elif old_username != new_username: + usernames_changed += 1 + if old_entries[name].get('token') != new_entries[name].get('token'): + tokens_changed += 1 + + pools_changed = 0 + for name in common_pools: + pools_changed += int(before_pools[name] != after_pools[name]) + return SecretsRedactedDiff( + document_changed=before_hash != after_hash, + semantic_changed=before != after, + pools_before=len(before_pools), + pools_after=len(after_pools), + pools_added=len(after_names - before_names), + pools_removed=len(before_names - after_names), + pools_changed=pools_changed, + entries_before=entries_before, + entries_after=entries_after, + entries_added=entries_added, + entries_removed=entries_removed, + pools_reordered=int( + before_names == after_names + and list(before_pools) != list(after_pools) + ), + usernames_added=usernames_added, + usernames_removed=usernames_removed, + usernames_changed=usernames_changed, + tokens_changed=tokens_changed, + ) + finally: + before = after = None + before_pools = after_pools = None + old_entries = new_entries = None + old_username = new_username = entry = entries = None + before_names = after_names = common_pools = None + old_names = new_names = None + pool_name = name = None + + +def _candidate_diff(document, active, proposed, active_payload, proposed_payload): + try: + active_hash = hashlib.sha256(active_payload).hexdigest() + proposed_hash = hashlib.sha256(proposed_payload).hexdigest() + if document == 'config': + return _config_diff( + active.config, proposed.config, active_hash, proposed_hash, + ) + return _secrets_diff( + active.secrets, proposed.secrets, active_hash, proposed_hash, + ) + finally: + active = proposed = None + active_payload = proposed_payload = None + + +def _proposed_pair(snapshot, document, candidate_bytes): + try: + if document == 'config': + return ( + candidate_bytes, + snapshot.candidate_secrets + if snapshot.candidate_secrets is not None + else snapshot.active_secrets, + ) + return ( + snapshot.candidate_config + if snapshot.candidate_config is not None + else snapshot.active_config, + candidate_bytes, + ) + finally: + snapshot = None + candidate_bytes = None + + +def _stage_candidate(path, payload, max_bytes): + temporary = None + descriptor = None + stored = None + try: + for _attempt in range(16): + temporary = Path(path).with_name( + f'.{Path(path).name}.{os.urandom(12).hex()}.tmp' + ) + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + if hasattr(os, 'O_NOFOLLOW'): + flags |= os.O_NOFOLLOW + try: + descriptor = os.open(temporary, flags, 0o600) + break + except FileExistsError: + temporary = None + if descriptor is None or temporary is None: + raise OSError('candidate temporary file could not be created') + with os.fdopen(descriptor, 'wb') as handle: + descriptor = None + before = os.fstat(handle.fileno()) + if not stat.S_ISREG(before.st_mode) or before.st_nlink != 1: + raise OSError('candidate temporary file identity is invalid') + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + after = os.fstat(handle.fileno()) + if ( + not stat.S_ISREG(after.st_mode) + or after.st_nlink != 1 + or after.st_size != len(payload) + or (os.name != 'nt' and stat.S_IMODE(after.st_mode) != 0o600) + ): + raise OSError('candidate temporary file metadata is invalid') + harden_private_file(temporary) + stored, failure = _read_private_document( + path=temporary, max_bytes=max_bytes, document='schema', + ) + if failure is not None or not hmac.compare_digest( + hashlib.sha256(stored).digest(), hashlib.sha256(payload).digest(), + ): + raise OSError('candidate temporary file verification failed') + return temporary + except BaseException: + if temporary is not None: + try: + os.unlink(temporary) + except OSError: + pass + raise + finally: + if descriptor is not None: + os.close(descriptor) + payload = None + stored = None + + +def _rollback_candidate(path, previous_payload, expected_payload, max_bytes, document): + current = temporary = None + try: + current, failure = _read_private_document(path, max_bytes, document) + if failure is not None or not hmac.compare_digest( + hashlib.sha256(current).digest(), + hashlib.sha256(expected_payload).digest(), + ): + return False + if previous_payload is None: + os.unlink(path) + fsync_directory(MANAGED_CANDIDATE_DIRECTORY) + return not os.path.lexists(path) + temporary = _stage_candidate(path, previous_payload, max_bytes) + durable_replace(temporary, path) + temporary = None + failure = _verify_final_candidate( + path, previous_payload, max_bytes, document, + ) + if failure is not None: + return False + fsync_directory(MANAGED_CANDIDATE_DIRECTORY) + return True + except Exception: + return False + finally: + try: + if temporary is not None: + try: + os.unlink(temporary) + except OSError: + pass + finally: + current = temporary = failure = None + previous_payload = expected_payload = None + + +def _rollback_committed_candidate( + path, previous_payload, expected_payload, max_bytes, document, +): + try: + with PrivateFileLock(MANAGED_CANDIDATE_LOCK_PATH): + _verify_candidate_lock() + return _rollback_candidate( + path, previous_payload, expected_payload, max_bytes, document, + ) + except BaseException: + return False + finally: + path = previous_payload = expected_payload = None + max_bytes = document = None + + +def _verify_final_candidate(path, expected_payload, max_bytes, document): + stored = None + try: + stored, failure = _read_private_document(path, max_bytes, document) + if failure is not None: + return failure + details = os.stat(path, follow_symlinks=False) + if ( + not stat.S_ISREG(details.st_mode) + or details.st_nlink != 1 + or (os.name != 'nt' and stat.S_IMODE(details.st_mode) != 0o600) + or len(stored) != len(expected_payload) + or not hmac.compare_digest( + hashlib.sha256(stored).digest(), + hashlib.sha256(expected_payload).digest(), + ) + ): + return _candidate_failure('schema', document, 'root') + return None + finally: + stored = None + expected_payload = None + + +def _load_editor_document_inner(config_path, document): + snapshot = payload = text = None + try: + if document not in ('config', 'secrets'): + return None, _candidate_failure('invalid_input') + _prepare_candidate_store() + with PrivateFileLock(MANAGED_CANDIDATE_LOCK_PATH): + _verify_candidate_lock() + _cleanup_candidate_temporaries() + snapshot, failure = _read_candidate_snapshot(config_path) + if failure is not None: + return None, failure + if document == 'config': + present = snapshot.state.candidate_config.present + payload = snapshot.candidate_config if present else snapshot.active_config + selected = snapshot.state.candidate_config + else: + present = snapshot.state.candidate_secrets.present + payload = snapshot.candidate_secrets if present else snapshot.active_secrets + selected = snapshot.state.candidate_secrets + try: + text = payload.decode('utf-8', errors='strict') + except UnicodeDecodeError: + return None, _candidate_failure('encoding', document, 'root') + return ManagedRuntimeEditorDocument( + document=document, + source='candidate' if present else 'active', + text=text, + state=snapshot.state, + selected=selected, + ), None + except Exception: + return None, _candidate_failure() + finally: + snapshot = payload = text = None + + +def load_managed_runtime_editor_document(config_path, document): + editor = failure = None + try: + editor, failure = _load_editor_document_inner(config_path, document) + finally: + config_path = document = None + if failure is not None: + raise RuntimeDocumentError( + failure[0], failure[1], failure[2], + document=failure[3], path=failure[4], + ) + return editor + + +def _preview_candidate_inner(config_path, document, candidate_bytes): + snapshot = active_validated = proposed_validated = None + proposed_config = proposed_secrets = None + active_payload = None + try: + if document not in ('config', 'secrets'): + return None, _candidate_failure('invalid_input') + if type(candidate_bytes) is not bytes: + return None, _candidate_failure('invalid_input', document, 'root') + _prepare_candidate_store() + with PrivateFileLock(MANAGED_CANDIDATE_LOCK_PATH): + _verify_candidate_lock() + _cleanup_candidate_temporaries() + snapshot, failure = _read_candidate_snapshot(config_path) + if failure is not None: + return None, failure + active_validated, failure = _validate_payload_pair( + snapshot.active_config, snapshot.active_secrets, + ) + if failure is not None: + return None, failure + proposed_config, proposed_secrets = _proposed_pair( + snapshot, document, candidate_bytes, + ) + proposed_validated, failure = _validate_payload_pair( + proposed_config, proposed_secrets, + ) + if failure is not None: + return None, failure + active_payload = ( + snapshot.active_config if document == 'config' + else snapshot.active_secrets + ) + return ManagedRuntimeCandidatePreview( + document=document, + state=snapshot.state, + proposed=_document_identity(candidate_bytes), + diff=_candidate_diff( + document, + active_validated, + proposed_validated, + active_payload, + candidate_bytes, + ), + ), None + except Exception: + return None, _candidate_failure() + finally: + candidate_bytes = None + snapshot = None + active_validated = None + proposed_validated = None + proposed_config = None + proposed_secrets = None + active_payload = None + + +def preview_managed_runtime_candidate(config_path, document, candidate_bytes): + preview = failure = None + try: + preview, failure = _preview_candidate_inner( + config_path, document, candidate_bytes, + ) + finally: + config_path = None + document = None + candidate_bytes = None + if failure is not None: + raise RuntimeDocumentError( + failure[0], failure[1], failure[2], + document=failure[3], path=failure[4], + ) + return preview + + +def _save_candidate_inner( + config_path, + document, + candidate_bytes, + expected_hashes, +): + snapshot = current = after = None + active_validated = proposed_validated = None + proposed_config = proposed_secrets = None + previous_payload = expected_after = None + temporary = None + committed = False + try: + if document not in ('config', 'secrets'): + return None, _candidate_failure('invalid_input') + if type(candidate_bytes) is not bytes: + return None, _candidate_failure('invalid_input', document, 'root') + if not all(_valid_expected_hash(value) for value in expected_hashes): + return None, _candidate_failure('reference', document, 'revision') + _prepare_candidate_store() + with PrivateFileLock(MANAGED_CANDIDATE_LOCK_PATH): + _verify_candidate_lock() + _cleanup_candidate_temporaries() + snapshot, failure = _read_candidate_snapshot(config_path) + if failure is not None: + return None, failure + actual_hashes = ( + snapshot.state.active_config.sha256, + snapshot.state.active_secrets.sha256, + snapshot.state.candidate_config.sha256, + snapshot.state.candidate_secrets.sha256, + ) + if not all( + _same_hash(actual, expected) + for actual, expected in zip(actual_hashes, expected_hashes) + ): + return None, _candidate_failure('reference', document, 'revision') + active_validated, failure = _validate_payload_pair( + snapshot.active_config, snapshot.active_secrets, + ) + if failure is not None: + return None, failure + proposed_config, proposed_secrets = _proposed_pair( + snapshot, document, candidate_bytes, + ) + proposed_validated, failure = _validate_payload_pair( + proposed_config, proposed_secrets, + ) + if failure is not None: + return None, failure + + selected_identity = ( + snapshot.state.candidate_config if document == 'config' + else snapshot.state.candidate_secrets + ) + proposed_identity = _document_identity(candidate_bytes) + selected_present = selected_identity.present + content_changed = not _same_hash( + selected_identity.sha256, proposed_identity.sha256, + ) or selected_identity.byte_count != proposed_identity.byte_count + diff = _candidate_diff( + document, + active_validated, + proposed_validated, + snapshot.active_config if document == 'config' else snapshot.active_secrets, + candidate_bytes, + ) + if selected_present and not content_changed: + return ManagedRuntimeCandidateRevision( + document=document, + before=snapshot.state, + after=snapshot.state, + proposed=proposed_identity, + created=False, + content_changed=False, + written=False, + diff=diff, + ), None + + path = ( + MANAGED_CONFIG_CANDIDATE_PATH + if document == 'config' + else MANAGED_SECRETS_CANDIDATE_PATH + ) + max_bytes = ( + MAX_CONFIG_DOCUMENT_BYTES + if document == 'config' + else MAX_SECRETS_DOCUMENT_BYTES + ) + previous_payload = ( + snapshot.candidate_config + if document == 'config' + else snapshot.candidate_secrets + ) + temporary = _stage_candidate(path, candidate_bytes, max_bytes) + current, failure = _read_candidate_snapshot(config_path) + if failure is not None: + return None, failure + if current.state != snapshot.state: + return None, _candidate_failure('reference', document, 'revision') + committed = True + durable_replace(temporary, path) + temporary = None + failure = _verify_final_candidate( + path, candidate_bytes, max_bytes, document, + ) + if failure is not None: + rolled_back = _rollback_candidate( + path, previous_payload, candidate_bytes, max_bytes, document, + ) + committed = not rolled_back + return None, failure if rolled_back else _candidate_failure() + try: + fsync_directory(MANAGED_CANDIDATE_DIRECTORY) + except Exception: + rolled_back = _rollback_candidate( + path, previous_payload, candidate_bytes, max_bytes, document, + ) + committed = not rolled_back + return None, _candidate_failure() + after, failure = _read_candidate_snapshot(config_path) + if failure is not None: + rolled_back = _rollback_candidate( + path, previous_payload, candidate_bytes, max_bytes, document, + ) + committed = not rolled_back + return None, failure if rolled_back else _candidate_failure() + expected_after = ManagedRuntimeDocumentState( + active_config=snapshot.state.active_config, + active_secrets=snapshot.state.active_secrets, + candidate_config=( + proposed_identity + if document == 'config' + else snapshot.state.candidate_config + ), + candidate_secrets=( + proposed_identity + if document == 'secrets' + else snapshot.state.candidate_secrets + ), + ) + if after.state != expected_after: + rolled_back = _rollback_candidate( + path, previous_payload, candidate_bytes, max_bytes, document, + ) + committed = not rolled_back + return None, _candidate_failure( + 'reference' if rolled_back else 'schema', + document, + 'revision' if rolled_back else 'root', + ) + committed = False + return ManagedRuntimeCandidateRevision( + document=document, + before=snapshot.state, + after=after.state, + proposed=proposed_identity, + created=not selected_present, + content_changed=content_changed, + written=True, + diff=diff, + ), None + except Exception: + if committed: + _rollback_committed_candidate( + path, previous_payload, candidate_bytes, max_bytes, document, + ) + return None, _candidate_failure() + except BaseException: + if committed: + _rollback_committed_candidate( + path, previous_payload, candidate_bytes, max_bytes, document, + ) + raise + finally: + if temporary is not None: + try: + os.unlink(temporary) + except OSError: + pass + candidate_bytes = None + snapshot = None + current = None + after = None + active_validated = None + proposed_validated = None + proposed_config = None + proposed_secrets = None + previous_payload = None + expected_after = None + + +def save_managed_runtime_candidate( + config_path, + document, + candidate_bytes, + *, + expected_active_config_sha256, + expected_active_secrets_sha256, + expected_candidate_config_sha256, + expected_candidate_secrets_sha256, +): + revision = failure = None + try: + revision, failure = _save_candidate_inner( + config_path, + document, + candidate_bytes, + ( + expected_active_config_sha256, + expected_active_secrets_sha256, + expected_candidate_config_sha256, + expected_candidate_secrets_sha256, + ), + ) + finally: + config_path = None + document = None + candidate_bytes = None + expected_active_config_sha256 = None + expected_active_secrets_sha256 = None + expected_candidate_config_sha256 = None + expected_candidate_secrets_sha256 = None + if failure is not None: + raise RuntimeDocumentError( + failure[0], failure[1], failure[2], + document=failure[3], path=failure[4], + ) + return revision + + +def _verify_candidates_inner( + config_path, + action, + expected_active_config_sha256, + expected_active_secrets_sha256, + expected_candidate_config_sha256, + expected_candidate_secrets_sha256, +): + snapshot = validated = None + config_payload = secrets_payload = None + try: + if action not in ('apply-config', 'apply-secrets', 'apply-both'): + return None, _candidate_failure('invalid_input') + if not _valid_expected_hash(expected_active_config_sha256) or not _valid_expected_hash( + expected_active_secrets_sha256 + ): + return None, _candidate_failure('reference', 'schema', 'revision') + required_config = action in ('apply-config', 'apply-both') + required_secrets = action in ('apply-secrets', 'apply-both') + if ( + required_config != (expected_candidate_config_sha256 is not None) + or required_secrets != (expected_candidate_secrets_sha256 is not None) + or ( + expected_candidate_config_sha256 is not None + and not _valid_expected_hash(expected_candidate_config_sha256) + ) + or ( + expected_candidate_secrets_sha256 is not None + and not _valid_expected_hash(expected_candidate_secrets_sha256) + ) + ): + return None, _candidate_failure('reference', 'schema', 'revision') + + _prepare_candidate_store() + with PrivateFileLock(MANAGED_CANDIDATE_LOCK_PATH): + _verify_candidate_lock() + _cleanup_candidate_temporaries() + snapshot, failure = _read_candidate_snapshot(config_path) + if failure is not None: + return None, failure + if not _same_hash( + snapshot.state.active_config.sha256, + expected_active_config_sha256, + ) or not _same_hash( + snapshot.state.active_secrets.sha256, + expected_active_secrets_sha256, + ): + return None, _candidate_failure('reference', 'schema', 'revision') + if required_config: + if not snapshot.state.candidate_config.present or not _same_hash( + snapshot.state.candidate_config.sha256, + expected_candidate_config_sha256, + ): + return None, _candidate_failure('reference', 'config', 'revision') + config_payload = snapshot.candidate_config + else: + config_payload = snapshot.active_config + if required_secrets: + if not snapshot.state.candidate_secrets.present or not _same_hash( + snapshot.state.candidate_secrets.sha256, + expected_candidate_secrets_sha256, + ): + return None, _candidate_failure('reference', 'secrets', 'revision') + secrets_payload = snapshot.candidate_secrets + else: + secrets_payload = snapshot.active_secrets + validated, failure = _validate_payload_pair(config_payload, secrets_payload) + if failure is not None: + return None, failure + return ManagedRuntimeApplyVerification( + action=action, + state=snapshot.state, + config_source='candidate' if required_config else 'active', + secrets_source='candidate' if required_secrets else 'active', + effective_config_sha256=hashlib.sha256(config_payload).hexdigest(), + effective_secrets_sha256=hashlib.sha256(secrets_payload).hexdigest(), + ), None + except Exception: + return None, _candidate_failure() + finally: + snapshot = None + validated = None + config_payload = None + secrets_payload = None + + +def verify_managed_runtime_candidates( + config_path, + action, + *, + expected_active_config_sha256, + expected_active_secrets_sha256, + expected_candidate_config_sha256=None, + expected_candidate_secrets_sha256=None, +): + verified, failure = _verify_candidates_inner( + config_path, + action, + expected_active_config_sha256, + expected_active_secrets_sha256, + expected_candidate_config_sha256, + expected_candidate_secrets_sha256, + ) + config_path = None + action = None + expected_active_config_sha256 = None + expected_active_secrets_sha256 = None + expected_candidate_config_sha256 = None + expected_candidate_secrets_sha256 = None + if failure is not None: + raise RuntimeDocumentError( + failure[0], failure[1], failure[2], + document=failure[3], path=failure[4], + ) + return verified diff --git a/app/runtime_security.py b/app/runtime_security.py new file mode 100644 index 0000000..151c37e --- /dev/null +++ b/app/runtime_security.py @@ -0,0 +1,1338 @@ +import ctypes +import errno +import hashlib +import json +import os +import re +import stat +import threading +from dataclasses import dataclass +from enum import Enum + + +MAX_PRIVATE_JSON_BYTES = 256 * 1024 +MAX_EXTENDED_PRIVATE_JSON_BYTES = 64 * 1024 * 1024 +_MACHINE_MUTEX_GUARD = threading.Lock() +_MACHINE_MUTEX_NAMES = set() + + +if os.name == 'nt': + import msvcrt + from ctypes import wintypes + + _ULONG_PTR = ctypes.c_ulonglong if ctypes.sizeof(ctypes.c_void_p) == 8 else wintypes.DWORD + + class _OVERLAPPED(ctypes.Structure): + _fields_ = [ + ('Internal', _ULONG_PTR), + ('InternalHigh', _ULONG_PTR), + ('Offset', wintypes.DWORD), + ('OffsetHigh', wintypes.DWORD), + ('hEvent', wintypes.HANDLE), + ] + + class _SECURITY_ATTRIBUTES(ctypes.Structure): + _fields_ = [ + ('nLength', wintypes.DWORD), + ('lpSecurityDescriptor', ctypes.c_void_p), + ('bInheritHandle', wintypes.BOOL), + ] + + class _SID_AND_ATTRIBUTES(ctypes.Structure): + _fields_ = [('Sid', ctypes.c_void_p), ('Attributes', wintypes.DWORD)] + + class _TOKEN_USER(ctypes.Structure): + _fields_ = [('User', _SID_AND_ATTRIBUTES)] + + class _FILETIME(ctypes.Structure): + _fields_ = [('dwLowDateTime', wintypes.DWORD), ('dwHighDateTime', wintypes.DWORD)] + + class _BY_HANDLE_FILE_INFORMATION(ctypes.Structure): + _fields_ = [ + ('dwFileAttributes', wintypes.DWORD), + ('ftCreationTime', _FILETIME), + ('ftLastAccessTime', _FILETIME), + ('ftLastWriteTime', _FILETIME), + ('dwVolumeSerialNumber', wintypes.DWORD), + ('nFileSizeHigh', wintypes.DWORD), + ('nFileSizeLow', wintypes.DWORD), + ('nNumberOfLinks', wintypes.DWORD), + ('nFileIndexHigh', wintypes.DWORD), + ('nFileIndexLow', wintypes.DWORD), + ] + + _P_OVERLAPPED = ctypes.POINTER(_OVERLAPPED) + _P_SECURITY_ATTRIBUTES = ctypes.POINTER(_SECURITY_ATTRIBUTES) + _P_VOID_P = ctypes.POINTER(ctypes.c_void_p) + _P_DWORD = ctypes.POINTER(wintypes.DWORD) + _P_HANDLE = ctypes.POINTER(wintypes.HANDLE) + _P_LPWSTR = ctypes.POINTER(wintypes.LPWSTR) + _P_TOKEN_USER = ctypes.POINTER(_TOKEN_USER) + _P_BY_HANDLE_FILE_INFORMATION = ctypes.POINTER(_BY_HANDLE_FILE_INFORMATION) + + _KERNEL32 = ctypes.WinDLL('kernel32', use_last_error=True) + _ADVAPI32 = ctypes.WinDLL('advapi32', use_last_error=True) + _SHELL32 = ctypes.WinDLL('shell32', use_last_error=True) + + _LOCK_FILE_EX = _KERNEL32.LockFileEx + _LOCK_FILE_EX.argtypes = [ + wintypes.HANDLE, wintypes.DWORD, wintypes.DWORD, wintypes.DWORD, + wintypes.DWORD, _P_OVERLAPPED, + ] + _LOCK_FILE_EX.restype = wintypes.BOOL + _UNLOCK_FILE_EX = _KERNEL32.UnlockFileEx + _UNLOCK_FILE_EX.argtypes = [ + wintypes.HANDLE, wintypes.DWORD, wintypes.DWORD, wintypes.DWORD, + _P_OVERLAPPED, + ] + _UNLOCK_FILE_EX.restype = wintypes.BOOL + _CREATE_MUTEX = _KERNEL32.CreateMutexW + _CREATE_MUTEX.argtypes = [_P_SECURITY_ATTRIBUTES, wintypes.BOOL, wintypes.LPCWSTR] + _CREATE_MUTEX.restype = wintypes.HANDLE + _WAIT_FOR_SINGLE_OBJECT = _KERNEL32.WaitForSingleObject + _WAIT_FOR_SINGLE_OBJECT.argtypes = [wintypes.HANDLE, wintypes.DWORD] + _WAIT_FOR_SINGLE_OBJECT.restype = wintypes.DWORD + _RELEASE_MUTEX = _KERNEL32.ReleaseMutex + _RELEASE_MUTEX.argtypes = [wintypes.HANDLE] + _RELEASE_MUTEX.restype = wintypes.BOOL + _CLOSE_HANDLE = _KERNEL32.CloseHandle + _CLOSE_HANDLE.argtypes = [wintypes.HANDLE] + _CLOSE_HANDLE.restype = wintypes.BOOL + _LOCAL_FREE = _KERNEL32.LocalFree + _LOCAL_FREE.argtypes = [wintypes.HLOCAL] + _LOCAL_FREE.restype = wintypes.HLOCAL + _GET_CURRENT_PROCESS = _KERNEL32.GetCurrentProcess + _GET_CURRENT_PROCESS.argtypes = [] + _GET_CURRENT_PROCESS.restype = wintypes.HANDLE + _CREATE_FILE = _KERNEL32.CreateFileW + _CREATE_FILE.argtypes = [ + wintypes.LPCWSTR, wintypes.DWORD, wintypes.DWORD, ctypes.c_void_p, + wintypes.DWORD, wintypes.DWORD, wintypes.HANDLE, + ] + _CREATE_FILE.restype = wintypes.HANDLE + _GET_FILE_INFORMATION = _KERNEL32.GetFileInformationByHandle + _GET_FILE_INFORMATION.argtypes = [wintypes.HANDLE, _P_BY_HANDLE_FILE_INFORMATION] + _GET_FILE_INFORMATION.restype = wintypes.BOOL + _MOVE_FILE_EX = _KERNEL32.MoveFileExW + _MOVE_FILE_EX.argtypes = [wintypes.LPCWSTR, wintypes.LPCWSTR, wintypes.DWORD] + _MOVE_FILE_EX.restype = wintypes.BOOL + + _CONVERT_SDDL = _ADVAPI32.ConvertStringSecurityDescriptorToSecurityDescriptorW + _CONVERT_SDDL.argtypes = [wintypes.LPCWSTR, wintypes.DWORD, _P_VOID_P, _P_DWORD] + _CONVERT_SDDL.restype = wintypes.BOOL + _SET_KERNEL_OBJECT_SECURITY = _ADVAPI32.SetKernelObjectSecurity + _SET_KERNEL_OBJECT_SECURITY.argtypes = [wintypes.HANDLE, wintypes.DWORD, ctypes.c_void_p] + _SET_KERNEL_OBJECT_SECURITY.restype = wintypes.BOOL + _OPEN_PROCESS_TOKEN = _ADVAPI32.OpenProcessToken + _OPEN_PROCESS_TOKEN.argtypes = [wintypes.HANDLE, wintypes.DWORD, _P_HANDLE] + _OPEN_PROCESS_TOKEN.restype = wintypes.BOOL + _GET_TOKEN_INFORMATION = _ADVAPI32.GetTokenInformation + _GET_TOKEN_INFORMATION.argtypes = [ + wintypes.HANDLE, wintypes.DWORD, ctypes.c_void_p, wintypes.DWORD, _P_DWORD, + ] + _GET_TOKEN_INFORMATION.restype = wintypes.BOOL + _CONVERT_SID = _ADVAPI32.ConvertSidToStringSidW + _CONVERT_SID.argtypes = [ctypes.c_void_p, _P_LPWSTR] + _CONVERT_SID.restype = wintypes.BOOL + _GET_FILE_SECURITY = _ADVAPI32.GetFileSecurityW + _GET_FILE_SECURITY.argtypes = [ + wintypes.LPCWSTR, wintypes.DWORD, ctypes.c_void_p, wintypes.DWORD, _P_DWORD, + ] + _GET_FILE_SECURITY.restype = wintypes.BOOL + _SECURITY_DESCRIPTOR_TO_SDDL = _ADVAPI32.ConvertSecurityDescriptorToStringSecurityDescriptorW + _SECURITY_DESCRIPTOR_TO_SDDL.argtypes = [ + ctypes.c_void_p, wintypes.DWORD, wintypes.DWORD, _P_LPWSTR, _P_DWORD, + ] + _SECURITY_DESCRIPTOR_TO_SDDL.restype = wintypes.BOOL + _SH_GET_FOLDER_PATH = _SHELL32.SHGetFolderPathW + _SH_GET_FOLDER_PATH.argtypes = [ + wintypes.HWND, ctypes.c_int, wintypes.HANDLE, wintypes.DWORD, wintypes.LPWSTR, + ] + _SH_GET_FOLDER_PATH.restype = ctypes.c_long +else: + _OVERLAPPED = _SECURITY_ATTRIBUTES = _SID_AND_ATTRIBUTES = None + _TOKEN_USER = _FILETIME = _BY_HANDLE_FILE_INFORMATION = None + _KERNEL32 = _ADVAPI32 = _SHELL32 = None + + +class PrivateFileError(OSError): + pass + + +class PrivatePathState(str, Enum): + PRESENT = 'present' + ABSENT = 'absent' + UNKNOWN = 'unknown' + + +@dataclass(frozen=True) +class PrivatePathInspection: + state: PrivatePathState + path: str + detail: str = '' + stat_result: object = None + + +def inspect_private_relative_path(root, relative_path): + """Inspect an indexed path without treating inaccessible storage as absence.""" + try: + root = require_private_directory(os.path.abspath(root), create=False) + relative = str(relative_path or '').replace('/', os.sep) + if not relative or os.path.isabs(relative): + raise ValueError('private relative path is empty or absolute') + path = os.path.abspath(os.path.join(root, relative)) + if path == root or os.path.commonpath((root, path)) != root: + raise ValueError('private relative path escapes its root') + parent = os.path.dirname(path) + require_private_directory(parent, create=False) + # Opening the shard proves traversal/access independently of target lstat. + with os.scandir(parent): + pass + try: + details = os.lstat(path) + except FileNotFoundError: + return PrivatePathInspection(PrivatePathState.ABSENT, path) + except OSError as exc: + return PrivatePathInspection( + PrivatePathState.UNKNOWN, path, f'{type(exc).__name__}: {exc}', + ) + return PrivatePathInspection(PrivatePathState.PRESENT, path, stat_result=details) + except (OSError, ValueError) as exc: + candidate = os.path.abspath(os.path.join(os.path.abspath(root), str(relative_path or ''))) + return PrivatePathInspection( + PrivatePathState.UNKNOWN, candidate, f'{type(exc).__name__}: {exc}', + ) + + +class PrivateFileLock: + """Cross-process lifetime lock backed by one byte of a private file.""" + + def __init__(self, path): + self.path = os.path.normcase(os.path.abspath(os.fspath(path))) + self._file = None + self._overlapped = None + self._acquired = False + + @property + def acquired(self): + return self._acquired + + def acquire(self): + if self._acquired: + return self + parent = os.path.dirname(self.path) + require_private_directory(parent, create=False) + reject_reparse_components(self.path) + flags = os.O_RDWR | os.O_CREAT + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + if hasattr(os, 'O_NOFOLLOW'): + flags |= os.O_NOFOLLOW + existed = os.path.lexists(self.path) + if existed: + descriptor = os.open(self.path, flags, 0o600) + else: + try: + descriptor = os.open(self.path, flags | os.O_EXCL, 0o600) + except FileExistsError: + existed = True + reject_reparse_components(self.path) + descriptor = os.open(self.path, flags, 0o600) + try: + if existed: + if not private_file_ready(self.path): + raise PrivateFileError(f'private lock file ACL is not ready: {self.path}') + else: + harden_private_file(self.path) + file_handle = os.fdopen(descriptor, 'r+b', buffering=0) + descriptor = None + if os.name == 'nt': + overlapped = _OVERLAPPED() + handle = wintypes.HANDLE(msvcrt.get_osfhandle(file_handle.fileno())) + flags_value = 0x00000002 | 0x00000001 # EXCLUSIVE_LOCK | FAIL_IMMEDIATELY + if not _LOCK_FILE_EX(handle, flags_value, 0, 1, 0, ctypes.byref(overlapped)): + error = ctypes.get_last_error() + file_handle.close() + raise BlockingIOError(error, 'private lifecycle lock is already held', self.path) + self._overlapped = overlapped + else: + import fcntl + + try: + fcntl.flock(file_handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + except OSError as exc: + file_handle.close() + raise BlockingIOError(exc.errno, 'private lifecycle lock is already held', self.path) from exc + self._file = file_handle + self._acquired = True + return self + finally: + if descriptor is not None: + os.close(descriptor) + + def release(self): + if not self._acquired: + return + self._acquired = False + file_handle = self._file + self._file = None + try: + if os.name == 'nt': + handle = wintypes.HANDLE(msvcrt.get_osfhandle(file_handle.fileno())) + _UNLOCK_FILE_EX(handle, 0, 1, 0, ctypes.byref(self._overlapped)) + self._overlapped = None + else: + import fcntl + + fcntl.flock(file_handle.fileno(), fcntl.LOCK_UN) + finally: + file_handle.close() + + def __enter__(self): + return self.acquire() + + def __exit__(self, exc_type, value, traceback): + self.release() + + +def canonical_cluster_data_directory(config): + global_config = (config or {}).get('global') or {} + runtime_dir = global_config.get('runtime_dir') + if not runtime_dir: + root_dir = global_config.get('root_dir') or os.path.dirname(os.path.dirname(os.path.abspath(__file__))) + runtime_dir = os.path.join(root_dir, 'runtime') + from paths import resolve_postgres_data_dir + + data_dir = reject_reparse_components( + resolve_postgres_data_dir(global_config, runtime_dir), + ) + drive, _ = os.path.splitdrive(data_dir) + root = drive + os.sep if drive else os.sep + if os.path.normcase(os.path.normpath(data_dir)) == os.path.normcase(os.path.normpath(root)): + raise PrivateFileError('PostgreSQL data directory cannot be a filesystem or volume root') + return canonical_path(data_dir) + + +def cluster_authority_lock_path(config): + data_dir = canonical_cluster_data_directory(config) + digest = hashlib.sha256(data_dir.encode('utf-8', errors='strict')).hexdigest() + global_config = (config or {}).get('global') or {} + runtime_dir = global_config.get('runtime_dir') + if not runtime_dir: + root_dir = global_config.get('root_dir') or os.path.dirname(os.path.dirname(os.path.abspath(__file__))) + runtime_dir = os.path.join(root_dir, 'runtime') + return os.path.join(canonical_path(os.path.join(runtime_dir, 'postgres')), f'.cluster-authority-{digest}.lock') + + +def _postgres_environment_values(config): + keys = { + 'TRUF_MANAGED_POSTGRES_DSN', 'SCANNER_DB_URL', 'DATABASE_URL', + 'TRUF_POSTGRES_DB', 'TRUF_POSTGRES_USER', 'TRUF_POSTGRES_PORT', + } + values = {key: os.getenv(key) for key in keys if os.getenv(key)} + global_config = (config or {}).get('global') or {} + candidates = [] + for parent in (global_config.get('root_dir'), global_config.get('project_dir')): + if parent: + candidates.append(os.path.join(parent, '.env.postgres')) + seen = set() + for candidate in candidates: + candidate = os.path.abspath(candidate) + normalized = os.path.normcase(candidate) + if normalized in seen or not os.path.isfile(candidate): + continue + seen.add(normalized) + try: + with open(candidate, 'r', encoding='utf-8') as handle: + for line in handle: + text = line.strip() + if not text or text.startswith('#') or '=' not in text: + continue + key, value = text.split('=', 1) + key = key.strip() + value = value.strip().strip('"').strip("'") + if key in keys and value and key not in values: + values[key] = value + except OSError as exc: + raise PrivateFileError(f'unable to determine PostgreSQL endpoint authority from {candidate}') from exc + if any(values.get(key) for key in ('TRUF_MANAGED_POSTGRES_DSN', 'SCANNER_DB_URL', 'DATABASE_URL')): + break + return values + + +def canonical_cluster_endpoint_identity(config=None, database_url=None): + """Return the credential-free identity of the PostgreSQL mutation endpoint.""" + if database_url is None: + raise PrivateFileError('canonical managed PostgreSQL DSN must be passed explicitly for endpoint authority') + from db_backend import POSTGRES_APPLICATION_SCHEMA, parse_postgres_url + + parsed = parse_postgres_url(database_url) + host = parsed['host'] + port = parsed['port'] + database = parsed['database'] + host = str(host or '').strip().lower().rstrip('.') + if host in ('localhost', '::1') or re.fullmatch(r'127(?:\.\d{1,3}){3}', host): + host = '127.0.0.1' + return json.dumps( + {'database': database, 'host': host, 'port': int(port), 'schema': POSTGRES_APPLICATION_SCHEMA}, + ensure_ascii=True, + sort_keys=True, + separators=(',', ':'), + ) + + +def _windows_common_appdata(): + buffer = ctypes.create_unicode_buffer(32768) + result = _SH_GET_FOLDER_PATH(None, 0x0023, None, 0, buffer) # CSIDL_COMMON_APPDATA + if result != 0 or not buffer.value: + raise PrivateFileError('unable to resolve the machine authority directory') + return os.path.abspath(buffer.value) + + +def _cluster_endpoint_lock_root(): + if os.name == 'nt': + return os.path.join(_windows_common_appdata(), 'Truf', 'authority') + return '/run/truf/authority' + + +def cluster_endpoint_authority_lock_path(config=None, endpoint_identity=None, endpoint_dsn=None): + endpoint = endpoint_identity or canonical_cluster_endpoint_identity(config, database_url=endpoint_dsn) + digest = hashlib.sha256(endpoint.encode('utf-8', errors='strict')).hexdigest() + return os.path.join(_cluster_endpoint_lock_root(), f'endpoint-{digest}.lock') + + +def cluster_endpoint_mutex_name(config=None, endpoint_identity=None, endpoint_dsn=None): + endpoint = endpoint_identity or canonical_cluster_endpoint_identity(config, database_url=endpoint_dsn) + digest = hashlib.sha256(endpoint.encode('utf-8', errors='strict')).hexdigest() + return f'Global\\Truf.ClusterEndpoint.{digest}' + + +class _WindowsGlobalMutex: + """Non-recursive process wrapper around a private cross-session mutex.""" + + def __init__(self, name): + self.name = str(name) + self._handle = None + self._acquired = False + + @property + def acquired(self): + return self._acquired + + def acquire(self): + if self._acquired: + return self + with _MACHINE_MUTEX_GUARD: + if self.name in _MACHINE_MUTEX_NAMES: + raise BlockingIOError(0, 'private machine endpoint authority is already held', self.name) + _MACHINE_MUTEX_NAMES.add(self.name) + + descriptor = ctypes.c_void_p() + sid = _windows_current_user_sid() + sddl = f'D:P(A;;GA;;;{sid})(A;;GA;;;SY)(A;;GA;;;BA)' + try: + if not _CONVERT_SDDL(sddl, 1, ctypes.byref(descriptor), None): + raise ctypes.WinError(ctypes.get_last_error()) + attributes = _SECURITY_ATTRIBUTES(ctypes.sizeof(_SECURITY_ATTRIBUTES), descriptor, False) + handle = _CREATE_MUTEX(ctypes.byref(attributes), False, self.name) + if not handle: + raise ctypes.WinError(ctypes.get_last_error()) + result = _WAIT_FOR_SINGLE_OBJECT(handle, 0) + if result not in (0, 0x00000080): # WAIT_OBJECT_0 / WAIT_ABANDONED + _CLOSE_HANDLE(handle) + if result == 0x00000102: # WAIT_TIMEOUT + raise BlockingIOError(0, 'private machine endpoint authority is already held', self.name) + raise ctypes.WinError(ctypes.get_last_error()) + self._handle = handle + self._acquired = True + return self + except BaseException: + with _MACHINE_MUTEX_GUARD: + _MACHINE_MUTEX_NAMES.discard(self.name) + raise + finally: + if descriptor: + _LOCAL_FREE(descriptor) + + def release(self): + if not self._acquired: + return + handle = self._handle + self._handle = None + self._acquired = False + try: + if handle and not _RELEASE_MUTEX(handle): + raise ctypes.WinError(ctypes.get_last_error()) + finally: + if handle: + _CLOSE_HANDLE(handle) + with _MACHINE_MUTEX_GUARD: + _MACHINE_MUTEX_NAMES.discard(self.name) + + +class ClusterAuthorityLock: + """Cross-process authority over both cluster storage and its SQL endpoint.""" + + def __init__(self, config, create_parent=False, endpoint_dsn=None): + if endpoint_dsn is None: + raise PrivateFileError('cluster endpoint authority requires the caller-selected canonical DSN') + self.data_directory = canonical_cluster_data_directory(config) + self.path = cluster_authority_lock_path(config) + self.endpoint_identity = canonical_cluster_endpoint_identity(config, database_url=endpoint_dsn) + self.endpoint_path = cluster_endpoint_authority_lock_path(config, self.endpoint_identity) + self.endpoint_mutex_name = cluster_endpoint_mutex_name(endpoint_identity=self.endpoint_identity) + self.paths = (self.endpoint_path, self.path) + self.create_parent = bool(create_parent) + self._locks = [] + + @property + def acquired(self): + return len(self._locks) == len(self.paths) and all(lock.acquired for lock in self._locks) + + def acquire(self): + data_parent = os.path.dirname(self.path) + if self.create_parent: + reject_reparse_components(os.path.dirname(data_parent)) + os.makedirs(data_parent, mode=0o700, exist_ok=True) + reject_reparse_components(data_parent) + harden_private_directory(data_parent) + try: + if os.name == 'nt': + self._locks.append(_WindowsGlobalMutex(self.endpoint_mutex_name).acquire()) + else: + endpoint_parent = os.path.dirname(self.endpoint_path) + if not os.path.exists(endpoint_parent): + os.makedirs(endpoint_parent, mode=0o700, exist_ok=False) + harden_private_directory(endpoint_parent) + require_private_directory(endpoint_parent, create=False) + self._locks.append(PrivateFileLock(self.endpoint_path).acquire()) + self._locks.append(PrivateFileLock(self.path).acquire()) + except BlockingIOError as exc: + self.release() + raise BlockingIOError( + getattr(exc, 'errno', 0), + 'another runtime or maintenance process owns this PostgreSQL data directory or endpoint authority', + getattr(exc, 'filename', None) or self.path, + ) from exc + except BaseException: + self.release() + raise + return self + + def release(self): + while self._locks: + self._locks.pop().release() + + def __enter__(self): + return self.acquire() + + def __exit__(self, exc_type, value, traceback): + self.release() + + +def canonical_path(path): + return os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(path)))) + + +def sha256_file(path): + digest = hashlib.sha256() + with open(path, 'rb') as handle: + while True: + block = handle.read(1024 * 1024) + if not block: + break + digest.update(block) + return digest.hexdigest() + + +def is_reparse_point(path): + try: + info = os.lstat(path) + except OSError: + return False + if stat.S_ISLNK(info.st_mode): + return True + return bool(getattr(info, 'st_file_attributes', 0) & 0x00000400) + + +def reject_reparse_components(path): + absolute = os.path.abspath(os.fspath(path)) + drive, tail = os.path.splitdrive(absolute) + current = drive + os.sep if drive else os.sep + for part in tail.strip(os.sep).split(os.sep): + if not part: + continue + current = os.path.join(current, part) + if os.path.lexists(current) and is_reparse_point(current): + raise PrivateFileError(f'private path contains a link or reparse point: {current}') + return absolute + + +def _windows_current_user_sid(): + token = wintypes.HANDLE() + if not _OPEN_PROCESS_TOKEN(_GET_CURRENT_PROCESS(), 0x0008, ctypes.byref(token)): + raise ctypes.WinError(ctypes.get_last_error()) + try: + needed = wintypes.DWORD() + _GET_TOKEN_INFORMATION(token, 1, None, 0, ctypes.byref(needed)) + if not needed.value: + raise ctypes.WinError(ctypes.get_last_error()) + buffer = ctypes.create_string_buffer(needed.value) + if not _GET_TOKEN_INFORMATION(token, 1, buffer, needed.value, ctypes.byref(needed)): + raise ctypes.WinError(ctypes.get_last_error()) + user = ctypes.cast(buffer, _P_TOKEN_USER).contents + output = wintypes.LPWSTR() + if not _CONVERT_SID(user.User.Sid, ctypes.byref(output)): + raise ctypes.WinError(ctypes.get_last_error()) + try: + return output.value + finally: + _LOCAL_FREE(output) + finally: + _CLOSE_HANDLE(token) + + +def _windows_private_sddl(path): + security_information = 0x00000001 | 0x00000004 + needed = wintypes.DWORD() + _GET_FILE_SECURITY(path, security_information, None, 0, ctypes.byref(needed)) + error = ctypes.get_last_error() + if not needed.value or error not in (0, 122): + raise ctypes.WinError(error) + buffer = ctypes.create_string_buffer(needed.value) + if not _GET_FILE_SECURITY(path, security_information, buffer, needed.value, ctypes.byref(needed)): + raise ctypes.WinError(ctypes.get_last_error()) + output = wintypes.LPWSTR() + if not _SECURITY_DESCRIPTOR_TO_SDDL(buffer, 1, security_information, ctypes.byref(output), None): + raise ctypes.WinError(ctypes.get_last_error()) + try: + return output.value or '' + finally: + _LOCAL_FREE(output) + + +def _harden_windows_path(path, directory=False): + # Protected DACL: the exact current owner SID, LocalSystem, and local Administrators only. + # Directories propagate the same protected allowlist to newly-created children. + descriptor = ctypes.c_void_p() + ace_flags = 'OICI' if directory else '' + owner_sid = _windows_current_user_sid() + current_sddl = _windows_private_sddl(path).upper() + owner_match = re.search(r'O:([^:()]+?)(?=[GDS]:|$)', current_sddl) + if not owner_match or owner_match.group(1) != owner_sid.upper(): + raise PrivateFileError(f'refusing to harden a path not owned by the current user: {path}') + trustees = [] + for trustee in (owner_sid, 'SY', 'BA'): + if trustee not in trustees: + trustees.append(trustee) + sddl = 'D:P' + ''.join(f'(A;{ace_flags};FA;;;{trustee})' for trustee in trustees) + if not _CONVERT_SDDL(sddl, 1, ctypes.byref(descriptor), None): + raise ctypes.WinError(ctypes.get_last_error()) + access = 0x00020000 | 0x00040000 # READ_CONTROL | WRITE_DAC + sharing = 0x00000001 | 0x00000002 | 0x00000004 + flags = 0x00200000 | 0x02000000 # OPEN_REPARSE_POINT | BACKUP_SEMANTICS + handle = _CREATE_FILE(path, access, sharing, None, 3, flags, None) + if handle == wintypes.HANDLE(-1).value: + _LOCAL_FREE(descriptor) + raise ctypes.WinError(ctypes.get_last_error()) + try: + information = _BY_HANDLE_FILE_INFORMATION() + if not _GET_FILE_INFORMATION(handle, ctypes.byref(information)): + raise ctypes.WinError(ctypes.get_last_error()) + if information.dwFileAttributes & 0x00000400: + raise PrivateFileError(f'refusing to harden a reparse point: {path}') + security_information = 0x00000004 | 0x80000000 # DACL | PROTECTED_DACL + if not _SET_KERNEL_OBJECT_SECURITY(handle, security_information, descriptor): + raise ctypes.WinError(ctypes.get_last_error()) + finally: + _CLOSE_HANDLE(handle) + _LOCAL_FREE(descriptor) + + +def harden_private_file(path): + reject_reparse_components(path) + if os.name == 'nt': + _harden_windows_path(path, directory=False) + else: + os.chmod(path, stat.S_IRUSR | stat.S_IWUSR) + if not private_file_ready(path): + raise PrivateFileError(f'private-file ACL verification failed: {path}') + + +def harden_private_directory(path): + reject_reparse_components(path) + if os.name == 'nt': + _harden_windows_path(path, directory=True) + else: + os.chmod(path, stat.S_IRWXU) + if not private_directory_ready(path): + raise PrivateFileError(f'private-directory ACL verification failed: {path}') + + +def ensure_private_directory(path, reject_reparse=False): + if reject_reparse: + reject_reparse_components(os.path.dirname(os.path.abspath(path)) or path) + os.makedirs(path, mode=0o700, exist_ok=True) + if reject_reparse: + reject_reparse_components(path) + harden_private_directory(path) + return path + + +def _private_acl_ready(path, directory=False): + if os.name != 'nt': + info = os.stat(path, follow_symlinks=False) + owner_ok = not hasattr(os, 'geteuid') or info.st_uid == os.geteuid() + return owner_ok and stat.S_IMODE(info.st_mode) & 0o077 == 0 + reject_reparse_components(path) + sddl = _windows_private_sddl(path).upper() + owner_match = re.search(r'O:([^:()]+?)(?=[GDS]:|$)', sddl) + if not owner_match or 'D:P' not in sddl: + return False + aliases = {'S-1-3-4': 'OW', 'S-1-5-18': 'SY', 'S-1-5-32-544': 'BA'} + owner = aliases.get(owner_match.group(1), owner_match.group(1)) + current_owner = _windows_current_user_sid().upper() + current_owner = aliases.get(current_owner, current_owner) + if owner != current_owner: + return False + aces = re.findall(r'\(([^()]*)\)', sddl) + trustees = set() + expected_trustees = {current_owner, 'SY', 'BA'} + if len(aces) != len(expected_trustees): + return False + expected_flags = {'OI', 'CI'} if directory else set() + for ace in aces: + fields = ace.split(';') + flags = set(re.findall(r'OI|CI|IO|NP|ID', fields[1])) if len(fields) == 6 else set() + if ( + len(fields) != 6 + or fields[0] != 'A' + or flags != expected_flags + or fields[2] != 'FA' + or fields[3] + or fields[4] + ): + return False + trustees.add(aliases.get(fields[5], fields[5])) + return trustees == expected_trustees + + +def private_file_ready(path): + try: + reject_reparse_components(path) + if not stat.S_ISREG(os.stat(path, follow_symlinks=False).st_mode): + return False + return _private_acl_ready(path, directory=False) + except (OSError, ValueError): + return False + + +def private_directory_ready(path): + try: + reject_reparse_components(path) + return stat.S_ISDIR(os.stat(path, follow_symlinks=False).st_mode) and _private_acl_ready(path, directory=True) + except (OSError, ValueError): + return False + + +def read_stable_root_file(path, max_bytes, trusted_root): + """Read one immutable, public root-owned file without following links.""" + if os.name == 'nt' or type(max_bytes) is not int or max_bytes < 1: + raise PrivateFileError('stable root file is unavailable') + descriptor = None + directories = [] + try: + absolute = os.path.abspath(os.fsdecode(path)) + root = os.path.abspath(os.fsdecode(trusted_root)) + if os.path.commonpath((absolute, root)) != root or absolute == root: + raise PrivateFileError('stable root file is outside its authority') + relative = os.path.relpath(absolute, root) + parts = relative.split(os.sep) + if not parts or any(part in ('', '.', '..') for part in parts): + raise PrivateFileError('stable root file path is invalid') + + directory_flags = ( + os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) + | getattr(os, 'O_DIRECTORY', 0) | getattr(os, 'O_NOFOLLOW', 0) + ) + directory = os.open(root, directory_flags) + directories.append(directory) + for part in parts[:-1]: + details = os.fstat(directory) + if ( + not stat.S_ISDIR(details.st_mode) + or details.st_uid != 0 + or details.st_gid != 0 + or stat.S_IMODE(details.st_mode) & 0o022 + ): + raise PrivateFileError('stable root directory metadata is invalid') + directory = os.open(part, directory_flags, dir_fd=directory) + directories.append(directory) + details = os.fstat(directory) + if ( + not stat.S_ISDIR(details.st_mode) + or details.st_uid != 0 + or details.st_gid != 0 + or stat.S_IMODE(details.st_mode) & 0o022 + ): + raise PrivateFileError('stable root directory metadata is invalid') + + flags = os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) | getattr(os, 'O_NOFOLLOW', 0) + descriptor = os.open(parts[-1], flags, dir_fd=directory) + before = os.fstat(descriptor) + + def identity(details): + return ( + details.st_dev, + details.st_ino, + details.st_size, + details.st_uid, + details.st_gid, + stat.S_IMODE(details.st_mode), + details.st_nlink, + getattr(details, 'st_mtime_ns', None), + getattr(details, 'st_ctime_ns', None), + ) + + if ( + not stat.S_ISREG(before.st_mode) + or before.st_nlink != 1 + or before.st_uid != 0 + or before.st_gid != 0 + or stat.S_IMODE(before.st_mode) != 0o644 + ): + raise PrivateFileError('stable root file metadata is invalid') + with os.fdopen(descriptor, 'rb') as handle: + descriptor = None + payload = handle.read(max_bytes + 1) + after = os.fstat(handle.fileno()) + current = os.stat(parts[-1], dir_fd=directory, follow_symlinks=False) + if ( + len(payload) > max_bytes + or identity(before) != identity(after) + or identity(after) != identity(current) + ): + raise PrivateFileError('stable root file changed while being read') + return payload + except PrivateFileError: + raise + except (OSError, TypeError, ValueError) as exc: + raise PrivateFileError('stable root file is not ready') from exc + finally: + if descriptor is not None: + os.close(descriptor) + for directory in reversed(directories): + os.close(directory) + + +def fsync_directory(path): + if os.name == 'nt' or not path: + return + flags = os.O_RDONLY + if hasattr(os, 'O_DIRECTORY'): + flags |= os.O_DIRECTORY + descriptor = os.open(path, flags) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + + +def durable_replace(source, destination): + source = os.path.abspath(source) + destination = os.path.abspath(destination) + if os.path.dirname(source) != os.path.dirname(destination): + raise ValueError('durable replacement must remain in the same directory') + durable_move(source, destination) + + +def durable_move(source, destination): + source = os.path.abspath(source) + destination = os.path.abspath(destination) + reject_reparse_components(source) + reject_reparse_components(os.path.dirname(destination)) + if os.path.lexists(destination): + reject_reparse_components(destination) + source_parent = os.path.dirname(source) + destination_parent = os.path.dirname(destination) + if os.name == 'nt': + movefile_replace_existing = 0x00000001 + movefile_write_through = 0x00000008 + if not _MOVE_FILE_EX(source, destination, movefile_replace_existing | movefile_write_through): + raise ctypes.WinError(ctypes.get_last_error()) + else: + os.replace(source, destination) + fsync_directory(destination_parent) + if source_parent != destination_parent: + fsync_directory(source_parent) + + +def durable_publish(source, destination): + """Atomically publish a new name without replacing an existing object.""" + source = os.path.abspath(source) + destination = os.path.abspath(destination) + reject_reparse_components(source) + reject_reparse_components(os.path.dirname(destination)) + if os.path.lexists(destination): + raise FileExistsError(destination) + source_parent = os.path.dirname(source) + destination_parent = os.path.dirname(destination) + if os.path.splitdrive(source)[0].lower() != os.path.splitdrive(destination)[0].lower(): + raise ValueError('durable publication must remain on one volume') + if os.name == 'nt': + movefile_write_through = 0x00000008 + if not _MOVE_FILE_EX(source, destination, movefile_write_through): + error = ctypes.get_last_error() + if error in (80, 183): + raise FileExistsError(destination) + raise ctypes.WinError(error) + else: + os.link(source, destination, follow_symlinks=False) + try: + os.unlink(source) + except BaseException: + try: + os.unlink(destination) + except OSError: + pass + raise + fsync_directory(destination_parent) + if source_parent != destination_parent: + fsync_directory(source_parent) + + +def durable_publish_directory(source, destination): + """Atomically move one exact directory to a new, non-existing name.""" + source = os.path.abspath(source) + destination = os.path.abspath(destination) + source_parent = os.path.dirname(source) + destination_parent = os.path.dirname(destination) + if source == destination or not source_parent or not destination_parent: + raise ValueError('durable directory publication paths are invalid') + reject_reparse_components(source) + reject_reparse_components(destination_parent) + source_details = os.stat(source, follow_symlinks=False) + if not stat.S_ISDIR(source_details.st_mode) or is_reparse_point(source): + raise ValueError('durable directory publication source is not an exact directory') + source_parent_details = os.stat(source_parent, follow_symlinks=False) + destination_parent_details = os.stat(destination_parent, follow_symlinks=False) + if source_parent_details.st_dev != destination_parent_details.st_dev: + raise OSError(errno.EXDEV, 'durable directory publication cannot cross devices') + if os.path.lexists(destination): + raise FileExistsError(destination) + if os.name == 'nt': + movefile_write_through = 0x00000008 + if not _MOVE_FILE_EX(source, destination, movefile_write_through): + error = ctypes.get_last_error() + if error in (80, 183): + raise FileExistsError(destination) + raise ctypes.WinError(error) + return + library = ctypes.CDLL(None, use_errno=True) + renameat2 = getattr(library, 'renameat2', None) + if renameat2 is None: + raise OSError( + errno.ENOSYS, + 'atomic non-replacing directory publication requires renameat2', + ) + renameat2.argtypes = [ + ctypes.c_int, ctypes.c_char_p, + ctypes.c_int, ctypes.c_char_p, + ctypes.c_uint, + ] + renameat2.restype = ctypes.c_int + if renameat2( + -100, os.fsencode(source), -100, os.fsencode(destination), 1, + ) != 0: + error = ctypes.get_errno() + if error == errno.EEXIST: + raise FileExistsError(destination) + raise OSError(error, os.strerror(error), destination) + fsync_directory(destination_parent) + if source_parent != destination_parent: + fsync_directory(source_parent) + + +def durable_unlink(path): + path = os.path.abspath(path) + reject_reparse_components(path) + parent = os.path.dirname(path) + os.remove(path) + fsync_directory(parent) + + +def require_private_directory(path, create=False): + """Require an exact private directory without silently repairing an existing ACL.""" + absolute = reject_reparse_components(path) + if not os.path.exists(absolute): + if not create: + raise PrivateFileError(f'private directory is absent: {absolute}') + os.makedirs(absolute, mode=0o700, exist_ok=False) + harden_private_directory(absolute) + if not private_directory_ready(absolute): + raise PrivateFileError(f'private directory ACL is not ready: {absolute}') + return absolute + + +def require_private_file(path): + """Require an existing exact private regular file without changing it.""" + absolute = reject_reparse_components(path) + if not private_file_ready(absolute): + raise PrivateFileError(f'private file ACL or owner is not ready: {absolute}') + return absolute + + +def require_protected_sensitive_file_parent(path): + """Accept private runtime directories or protected root-owned read-only authorities.""" + absolute = reject_reparse_components(path) + if private_directory_ready(absolute): + return absolute + if os.name != 'nt': + try: + details = os.stat(absolute, follow_symlinks=False) + mode = stat.S_IMODE(details.st_mode) + effective_uid = os.geteuid() + effective_gid = os.getegid() + groups = set(os.getgroups()) + if effective_uid == details.st_uid: + search_bit = stat.S_IXUSR + elif details.st_gid == effective_gid or details.st_gid in groups: + search_bit = stat.S_IXGRP + else: + search_bit = stat.S_IXOTH + if ( + stat.S_ISDIR(details.st_mode) + and details.st_uid == 0 + and details.st_gid == 0 + and mode & 0o022 == 0 + and mode & search_bit + ): + return absolute + except (OSError, ValueError): + pass + raise PrivateFileError(f'sensitive file parent ACL is not ready: {absolute}') + + +def require_trusted_native_executable(path): + """Return canonical native code trusted by the effective runtime user, without repairs. + + POSIX requires immutable root-owned code and parents; Windows retains the + private-file policy. Verification failures raise PrivateFileError. + """ + try: + if os.name == 'nt': + return canonical_path(require_private_file(path)) + raw_path = os.fsdecode(path) + if not os.path.isabs(raw_path) or '\x00' in raw_path: + raise PrivateFileError(f'trusted native executable must be an absolute path: {raw_path}') + # Inspect before normalization so a link followed by /.. cannot disappear. + current = raw_path + while True: + details = os.lstat(current) + if stat.S_ISLNK(details.st_mode): + raise PrivateFileError(f'trusted native path contains a link: {current}') + if current == raw_path: + if not stat.S_ISREG(details.st_mode): + raise PrivateFileError(f'trusted native executable is not a regular file: {current}') + elif not stat.S_ISDIR(details.st_mode): + raise PrivateFileError(f'trusted native parent is not a directory: {current}') + if details.st_uid != 0: + raise PrivateFileError(f'trusted native path is not root-owned: {current}') + if details.st_mode & (stat.S_ISUID | stat.S_ISGID | stat.S_IWGRP | stat.S_IWOTH): + raise PrivateFileError(f'trusted native path has unsafe permissions: {current}') + if os.access(current, os.W_OK, effective_ids=True): + raise PrivateFileError(f'trusted native path is writable by the runtime user: {current}') + if not os.access(current, os.X_OK, effective_ids=True): + raise PrivateFileError(f'trusted native path is not executable/searchable by the runtime user: {current}') + parent = os.path.dirname(current) + if parent == current: + break + current = parent + return canonical_path(raw_path) + except PrivateFileError: + raise + except (OSError, TypeError, ValueError, NotImplementedError) as exc: + raise PrivateFileError(f'trusted native executable verification failed: {path}: {exc}') from exc + + +def preflight_lifecycle_paths(config_path, config, *, authority_profile='full'): + """Read-only verification required before lifecycle secrets or logs are opened.""" + if authority_profile not in ('full', 'discovery-producer', 'server'): + raise PrivateFileError('unsupported lifecycle authority profile') + discovery_producer = authority_profile in ('discovery-producer', 'server') + global_config = (config or {}).get('global') or {} + supervisor_config = (config or {}).get('supervisor') or {} + runtime_dir = global_config.get('runtime_dir') + directory_values = [ + global_config.get('root_dir'), + global_config.get('project_dir'), + runtime_dir, + global_config.get('control_dir'), + supervisor_config.get('control_dir'), + global_config.get('log_dir'), + supervisor_config.get('log_dir'), + global_config.get('work_dir'), + global_config.get('results_dir'), + global_config.get('queue_dir'), + global_config.get('state_dir'), + global_config.get('keycheck_dir'), + global_config.get('postman_cache_dir'), + global_config.get('gharchive_cache_dir'), + global_config.get('result_spool_dir'), + global_config.get('result_bundle_dir'), + os.path.join(global_config.get('result_bundle_dir'), 'tmp') if global_config.get('result_bundle_dir') else None, + os.path.join(global_config.get('result_bundle_dir'), 'ready') if global_config.get('result_bundle_dir') else None, + os.path.join(global_config.get('result_bundle_dir'), 'quarantine') if global_config.get('result_bundle_dir') else None, + os.path.join(runtime_dir, 'postgres') if runtime_dir else None, + canonical_cluster_data_directory(config) if global_config.get('postgres_data_dir') else None, + _cluster_endpoint_lock_root() if os.name != 'nt' else None, + ] + seen = set() + try: + require_private_file(config_path) + for value in directory_values: + if not value: + continue + absolute = os.path.abspath(value) + normalized = os.path.normcase(absolute) + if normalized in seen: + continue + seen.add(normalized) + require_private_directory(absolute, create=False) + + sensitive_files = [ + global_config.get('secrets_file'), + global_config.get('proxy_file'), + global_config.get('api_proxy_file'), + global_config.get('download_proxy_file'), + supervisor_config.get('instance_file'), + supervisor_config.get('lock_file'), + supervisor_config.get('supervisor_log'), + supervisor_config.get('status_file'), + supervisor_config.get('dashboard_log'), + ] + if not discovery_producer: + sensitive_files.append(global_config.get('trufflehog_config')) + policy_paths = [] if discovery_producer else [global_config.get('trufflehog_config')] + from paths import resolve_optional_path + + if not discovery_producer: + for source in ((config or {}).get('sources') or {}).values(): + if isinstance(source, dict) and source.get('trufflehog_config'): + policy_paths.append(resolve_optional_path(source['trufflehog_config'], global_config)) + sensitive_files.extend(policy_paths) + root_dir = global_config.get('root_dir') + project_dir = global_config.get('project_dir') + config_dir = os.path.dirname(os.path.abspath(config_path)) + for parent in (root_dir, project_dir, config_dir, os.path.dirname(config_dir)): + if parent: + sensitive_files.append(os.path.join(parent, '.env.postgres')) + for value in sensitive_files: + if value and os.path.lexists(value): + require_private_file(value) + + # Import lazily: lifecycle authority itself depends on this module. + from lifecycle_authority import ( + GIT_MANIFEST_NAME, + TRUFFLEHOG_MANIFEST_NAME, + manifest_authority_paths, + resolve_manifest_executable, + ) + + for path in manifest_authority_paths( + global_config.get('project_dir'), + global_config.get('trufflehog_path'), + policy_paths=policy_paths, + existing_only=True, + include_executables=False, + ): + require_private_file(path) + if authority_profile != 'discovery-producer': + executable_values = [(GIT_MANIFEST_NAME, None)] + if authority_profile == 'full': + executable_values.insert( + 0, + (TRUFFLEHOG_MANIFEST_NAME, global_config.get('trufflehog_path')), + ) + for name, value in executable_values: + require_trusted_native_executable(resolve_manifest_executable( + value, name=name, app_dir=project_dir, + )) + except (OSError, ValueError) as exc: + raise PrivateFileError( + f'lifecycle sensitive-path preflight failed: {exc}. ' + 'Run the offline hardening command before startup; runtime startup will not repair ACLs.' + ) from exc + return True + + +def harden_private_tree(path): + """Offline recursive hardening that never traverses a reparse point.""" + absolute = reject_reparse_components(path) + if not os.path.lexists(absolute): + return 0 + if is_reparse_point(absolute): + raise PrivateFileError(f'refusing to harden a link or reparse point: {absolute}') + if os.path.isfile(absolute): + harden_private_file(absolute) + return 1 + if not os.path.isdir(absolute): + raise PrivateFileError(f'unsupported private-tree entry: {absolute}') + count = 0 + stack = [absolute] + directories = [] + while stack: + current = stack.pop() + reject_reparse_components(current) + directories.append(current) + with os.scandir(current) as entries: + for entry in entries: + if entry.is_symlink() or is_reparse_point(entry.path): + raise PrivateFileError(f'private tree contains a link or reparse point: {entry.path}') + if entry.is_dir(follow_symlinks=False): + stack.append(entry.path) + elif entry.is_file(follow_symlinks=False): + harden_private_file(entry.path) + count += 1 + else: + raise PrivateFileError(f'unsupported private-tree entry: {entry.path}') + for directory in reversed(directories): + harden_private_directory(directory) + count += 1 + return count + + +def require_sensitive_runtime_paths(global_config, create=False): + """Verify sensitive runtime parents before any credential or event write.""" + config = global_config or {} + directories = [] + for key in ( + 'runtime_dir', 'results_dir', 'result_spool_dir', 'result_bundle_dir', 'queue_dir', 'state_dir', + 'log_dir', 'control_dir', 'keycheck_dir', 'postman_cache_dir', 'gharchive_cache_dir', 'work_dir', + ): + value = config.get(key) + if value: + directories.append(os.path.abspath(value)) + if config.get('postgres_data_dir'): + directories.append(canonical_cluster_data_directory({'global': config})) + seen = set() + for directory in sorted(directories, key=lambda value: (value.count(os.sep), value)): + normalized = os.path.normcase(directory) + if normalized in seen: + continue + seen.add(normalized) + require_private_directory(directory, create=create) + sensitive_files = [config.get('secrets_file')] + root_dir = config.get('root_dir') + project_dir = config.get('project_dir') + if root_dir: + sensitive_files.append(os.path.join(root_dir, '.env.postgres')) + if project_dir: + sensitive_files.append(os.path.join(project_dir, '.env.postgres')) + for path in sensitive_files: + if not path or not os.path.lexists(path): + continue + require_protected_sensitive_file_parent( + os.path.dirname(os.path.abspath(path)) + ) + if not private_file_ready(path): + raise PrivateFileError(f'sensitive file is not private: {path}') + return True + + +def atomic_write_private_json(path, value, max_bytes=MAX_PRIVATE_JSON_BYTES): + max_bytes = int(max_bytes) + if max_bytes < 1 or max_bytes > MAX_EXTENDED_PRIVATE_JSON_BYTES: + raise ValueError('private JSON write bound is invalid') + parent = os.path.dirname(os.path.abspath(path)) + if parent: + require_private_directory(parent, create=True) + if os.path.lexists(path): + reject_reparse_components(path) + payload = json.dumps(value, ensure_ascii=True, sort_keys=True, separators=(',', ':')).encode('utf-8') + b'\n' + if len(payload) > max_bytes: + raise ValueError('private JSON payload is too large') + temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.tmp' + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + descriptor = os.open(temporary, flags, 0o600) + try: + os.close(descriptor) + descriptor = None + harden_private_file(temporary) + with open(temporary, 'wb') as handle: + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + if not private_file_ready(temporary): + raise PrivateFileError(f'temporary private-file ACL changed: {temporary}') + durable_replace(temporary, path) + if not private_file_ready(path): + raise PrivateFileError(f'private-file ACL changed during replace: {path}') + finally: + if descriptor is not None: + os.close(descriptor) + try: + if os.path.exists(temporary): + os.remove(temporary) + except OSError: + pass + + +def write_private_json_exclusive(path, value, max_bytes=MAX_PRIVATE_JSON_BYTES): + """Atomically publish a private JSON file without replacing an existing name.""" + max_bytes = int(max_bytes) + if max_bytes < 1 or max_bytes > MAX_EXTENDED_PRIVATE_JSON_BYTES: + raise ValueError('private JSON write bound is invalid') + parent = os.path.dirname(os.path.abspath(path)) + if not parent or not private_directory_ready(parent): + raise PrivateFileError(f'private parent directory is not ready: {parent}') + reject_reparse_components(parent) + payload = json.dumps(value, ensure_ascii=True, sort_keys=True, separators=(',', ':')).encode('utf-8') + b'\n' + if len(payload) > max_bytes: + raise ValueError('private JSON payload is too large') + temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.tmp' + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + descriptor = os.open(temporary, flags, 0o600) + try: + with os.fdopen(descriptor, 'wb') as handle: + descriptor = None + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + harden_private_file(temporary) + if not private_file_ready(temporary): + raise PrivateFileError(f'temporary private-file ACL changed: {temporary}') + os.link(temporary, path) + fsync_directory(parent) + if not private_file_ready(path): + raise PrivateFileError(f'private-file ACL changed during publication: {path}') + finally: + if descriptor is not None: + os.close(descriptor) + try: + os.remove(temporary) + except OSError: + pass + + +def read_private_json(path, max_bytes=MAX_PRIVATE_JSON_BYTES): + if not private_file_ready(path): + raise PrivateFileError(f'file is absent or not private: {path}') + size = os.stat(path, follow_symlinks=False).st_size + if size <= 0 or size > max_bytes: + raise PrivateFileError(f'private JSON file has invalid size: {path}') + with open(path, 'rb') as handle: + payload = handle.read(max_bytes + 1) + if len(payload) > max_bytes: + raise PrivateFileError(f'private JSON file is too large: {path}') + try: + value = json.loads(payload.decode('utf-8')) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise PrivateFileError(f'invalid private JSON file: {path}') from exc + if not isinstance(value, dict): + raise PrivateFileError(f'private JSON root must be an object: {path}') + return value diff --git a/app/scan_execution.py b/app/scan_execution.py new file mode 100644 index 0000000..a92d5f2 --- /dev/null +++ b/app/scan_execution.py @@ -0,0 +1,941 @@ +import hashlib +import hmac +import json +import platform as host_platform +import sys +from contextlib import nullcontext +from dataclasses import dataclass +from datetime import datetime, timedelta, timezone +from typing import Mapping + +from result_bundle import BundleReservation, FORMAT_VERSION +from scanner import ( + cleanup_assignment_work_dir, + client_remote_execution_binding, + client_scan_phase_events, + client_scan_execution_policy, + scan_slot_scope, + scan_target_result, + stage_result_bundle, +) +from scanner_db import normalize_target +from target_identity import normalize_huggingface_space_id, parse_dockerhub_digest_target + + +PROTOCOL_VERSION = 2 +REMOTE_EXECUTION_SNAPSHOT_SCHEMA = 1 +PACKAGE_DETECTOR_POLICY = '@package/detector_policy' +MAX_REMOTE_EXECUTION_SNAPSHOT_BYTES = 64 * 1024 +_REMOTE_SCAN_POLICY_BOUNDS = { + 'trufflehog_stdout_max_mb': (1, 4096), + 'trufflehog_stderr_max_mb': (1, 4096), + 'result_bundle_max_event_bytes': (1024, 4 * 1024 * 1024 * 1024), + 'trufflehog_max_findings_per_target': (1, 1000000), + 'trufflehog_job_memory_limit_bytes': (0, 1 << 50), + 'trufflehog_windows_job_cpu_weight': (0, 10000), + 'trufflehog_windows_memory_priority': (0, 5), + 'trufflehog_diagnostic_max_lines': (1, 2000), + 'trufflehog_diagnostic_max_line_chars': (1, 8192), + 'trufflehog_diagnostic_max_line_bytes': (1, 8192), + 'trufflehog_diagnostic_max_errors': (1, 200), + 'trufflehog_diagnostic_max_warnings': (1, 200), + 'trufflehog_diagnostic_max_unclassified': (1, 20), +} + + +class ScanExecutionError(RuntimeError): + pass + + +@dataclass(frozen=True) +class QueueDispositionPolicy: + target_retry_max_attempts: int = 3 + target_retry_base_delay_sec: int = 3600 + target_retry_max_delay_sec: int = 86400 + target_timeout_retry_delay_sec: int = 21600 + docker_layer_checkpoint_delay_sec: int = 60 + ci_soft_cooldown_days: int = 7 + soft_skip_reasons: tuple[str, ...] = () + + +@dataclass(frozen=True) +class ScanCompatibility: + protocol_version: int + bundle_format_version: int + platform_tag: str + code_manifest_sha256: str + effective_config_sha256: str + detector_policy_sha256: str = '' + + @classmethod + def from_mapping(cls, value): + value = dict(value or {}) + return cls( + protocol_version=int(value.get('protocol_version') or 0), + bundle_format_version=int(value.get('bundle_format_version') or 0), + platform_tag=str(value.get('platform_tag') or ''), + code_manifest_sha256=_digest(value.get('code_manifest_sha256'), 'code manifest'), + effective_config_sha256=_digest( + value.get('effective_config_sha256'), 'effective config', + ), + detector_policy_sha256=_digest( + value.get('detector_policy_sha256'), 'detector policy', optional=True, + ), + ) + + def as_dict(self): + return dict(self.__dict__) + + +@dataclass(frozen=True) +class WorkerBuildCompatibility: + protocol_version: int + bundle_format_version: int + platform_tag: str + code_manifest_sha256: str + detector_policy_sha256: str + + @classmethod + def from_mapping(cls, value): + value = dict(value or {}) + if set(value) != { + 'protocol_version', 'bundle_format_version', 'platform_tag', + 'code_manifest_sha256', 'detector_policy_sha256', + }: + raise ValueError('worker build compatibility shape is invalid') + return cls( + protocol_version=int(value.get('protocol_version') or 0), + bundle_format_version=int(value.get('bundle_format_version') or 0), + platform_tag=str(value.get('platform_tag') or ''), + code_manifest_sha256=_digest(value.get('code_manifest_sha256'), 'code manifest'), + detector_policy_sha256=_digest( + value.get('detector_policy_sha256'), 'detector policy', + ), + ) + + def as_dict(self): + return dict(self.__dict__) + + +def _digest(value, label, optional=False): + value = str(value or '') + if optional and not value: + return '' + if len(value) != 64 or any(char not in '0123456789abcdef' for char in value): + raise ValueError(f'invalid {label} digest') + return value + + +def local_platform_tag(): + machine = host_platform.machine().strip().lower().replace('amd64', 'x86_64') + system = 'windows' if sys.platform == 'win32' else 'linux' if sys.platform.startswith('linux') else '' + if not system or machine not in {'x86_64', 'aarch64', 'arm64'}: + raise ScanExecutionError('unsupported worker platform') + return f'{system}-{machine.replace("arm64", "aarch64")}' + + +def validate_scan_compatibility(required, local): + required = required if isinstance(required, ScanCompatibility) else ScanCompatibility.from_mapping(required) + local = local if isinstance(local, ScanCompatibility) else ScanCompatibility.from_mapping(local) + if required.protocol_version != PROTOCOL_VERSION or local.protocol_version != PROTOCOL_VERSION: + raise ScanExecutionError('worker protocol is incompatible') + if required.bundle_format_version != FORMAT_VERSION or local.bundle_format_version != FORMAT_VERSION: + raise ScanExecutionError('result bundle format is incompatible') + for name in ( + 'platform_tag', 'code_manifest_sha256', 'effective_config_sha256', + 'detector_policy_sha256', + ): + if not hmac.compare_digest(str(getattr(required, name)), str(getattr(local, name))): + raise ScanExecutionError(f'worker {name.replace("_", " ")} is incompatible') + return required + + +def validate_worker_build_compatibility( + required, local, *, expected_protocol_version=PROTOCOL_VERSION, +): + required = ( + required if isinstance(required, WorkerBuildCompatibility) + else WorkerBuildCompatibility.from_mapping(required) + ) + local = ( + local if isinstance(local, WorkerBuildCompatibility) + else WorkerBuildCompatibility.from_mapping(local) + ) + if ( + required.protocol_version != expected_protocol_version + or local.protocol_version != expected_protocol_version + ): + raise ScanExecutionError('worker protocol is incompatible') + if required.bundle_format_version != FORMAT_VERSION or local.bundle_format_version != FORMAT_VERSION: + raise ScanExecutionError('result bundle format is incompatible') + for name in ('platform_tag', 'code_manifest_sha256', 'detector_policy_sha256'): + if not hmac.compare_digest(str(getattr(required, name)), str(getattr(local, name))): + raise ScanExecutionError(f'worker {name.replace("_", " ")} is incompatible') + return required + + +_COMMON_SCAN_KWARGS = { + 'timeout_sec', 'detectors', 'exclude_detectors', 'no_verification', + 'trufflehog_config', 'token', +} +_SOURCE_SCAN_KWARGS = { + 'git': {'git_plan'}, + 'github': {'git_plan', 'max_depth', 'max_commit_age_days', 'commit_lookup_pages', + 'skip_if_commit_lookup_fails'}, + 'github_archive': {'max_depth', 'max_commit_age_days', 'commit_lookup_pages', + 'skip_if_commit_lookup_fails'}, + 'gitlab': {'git_plan', 'external_trufflehog_lifecycle', 'max_depth', + 'max_commit_age_days', 'commit_lookup_pages', 'skip_if_commit_lookup_fails'}, + 'docker': {'docker_layer_work', 'trufflehog_concurrency', 'docker_recovery_limits', + 'docker_recovery_min_free_bytes'}, + 'huggingface': set(), + 'npm': {'max_artifact_size_mb'}, + 'pypi': {'max_artifact_size_mb'}, + 'package_git': {'max_depth', 'max_commit_age_days', 'commit_lookup_pages', + 'skip_if_commit_lookup_fails'}, + 'postman': {'max_artifact_size_mb'}, + 'github_gists': {'max_artifact_size_mb'}, + 'github_archive_files': {'max_artifact_size_mb'}, + 'github_actions': { + 'ci_runs_per_repo', 'ci_lookback_days', 'ci_max_log_archive_mb', + 'ci_max_log_file_mb', 'ci_failed_first', 'ci_scan_artifacts', + 'ci_max_artifacts_per_run', 'ci_max_artifact_archive_mb', + 'ci_max_artifact_file_mb', 'ci_max_artifact_files', + 'ci_target_max_download_mb', 'fetch_timeout', + }, + 'gitlab_ci': { + 'ci_pipelines_per_project', 'ci_jobs_per_pipeline', 'ci_lookback_days', + 'ci_max_trace_mb', 'ci_scan_artifacts', 'ci_max_artifacts_per_pipeline', + 'ci_max_artifact_archive_mb', 'ci_max_artifact_file_mb', + 'ci_max_artifact_files', 'ci_target_max_download_mb', 'fetch_timeout', + }, +} + + +def validate_scan_kwargs(platform, scan_kwargs): + platform = str(platform or '').strip().lower() + if platform not in _SOURCE_SCAN_KWARGS: + raise ScanExecutionError('unsupported scan platform') + values = dict(scan_kwargs or {}) + unknown = set(values) - _COMMON_SCAN_KWARGS - _SOURCE_SCAN_KWARGS[platform] + if unknown: + raise ScanExecutionError('scan settings contain unsupported fields') + timeout = values.get('timeout_sec') + if isinstance(timeout, bool): + raise ScanExecutionError('scan timeout is invalid') + try: + timeout = float(timeout) + except (TypeError, ValueError, OverflowError): + raise ScanExecutionError('scan timeout is invalid') from None + if not 1 <= timeout <= 86400: + raise ScanExecutionError('scan timeout is outside the worker bound') + values['timeout_sec'] = timeout + return values + + +def normalize_remote_scan_policy(value): + values = dict(value or {}) + expected = { + 'drop_detectors', 'strict_git_provider_token_filter', + *_REMOTE_SCAN_POLICY_BOUNDS, + } + if set(values) != expected: + raise ScanExecutionError('remote scan policy shape is invalid') + raw_drop = values['drop_detectors'] + if isinstance(raw_drop, str): + raw_drop = raw_drop.split(',') + if not isinstance(raw_drop, (list, tuple)) or len(raw_drop) > 256: + raise ScanExecutionError('remote detector drop policy is invalid') + drop_detectors = [] + for item in raw_drop: + if not isinstance(item, str): + raise ScanExecutionError('remote detector drop policy is invalid') + item = item.strip().lower() + if not item: + continue + if len(item) > 128 or any(ord(char) < 32 or ord(char) == 127 for char in item): + raise ScanExecutionError('remote detector drop policy is invalid') + drop_detectors.append(item) + strict = values['strict_git_provider_token_filter'] + if not isinstance(strict, bool): + raise ScanExecutionError('remote Git provider token policy is invalid') + normalized = { + 'drop_detectors': sorted(set(drop_detectors)), + 'strict_git_provider_token_filter': strict, + } + for name, (minimum, maximum) in _REMOTE_SCAN_POLICY_BOUNDS.items(): + raw = values[name] + if not isinstance(raw, int) or isinstance(raw, bool): + raise ScanExecutionError('remote scan policy limit is invalid') + number = raw + if number < minimum or number > maximum: + raise ScanExecutionError('remote scan policy limit is outside its bounds') + normalized[name] = number + return normalized + + +def remote_execution_identity( + platform, scan_kwargs, event_scan_options, queue_policy, limits, scan_policy, +): + normalized_scan = validate_scan_kwargs(platform, scan_kwargs) + event_options = dict(event_scan_options or {}) + if 'token' in event_options or 'git_plan' in event_options: + raise ScanExecutionError('event scan settings contain private or planned fields') + expected_event = { + name: value for name, value in normalized_scan.items() + if name not in {'token', 'git_plan'} + } + if event_options != expected_event: + raise ScanExecutionError('event scan settings do not match execution settings') + try: + policy = ( + queue_policy if isinstance(queue_policy, QueueDispositionPolicy) + else QueueDispositionPolicy(**dict(queue_policy or {})) + ) + except (TypeError, ValueError) as exc: + raise ScanExecutionError('queue disposition policy is invalid') from exc + policy_value = { + 'target_retry_max_attempts': int(policy.target_retry_max_attempts), + 'target_retry_base_delay_sec': int(policy.target_retry_base_delay_sec), + 'target_retry_max_delay_sec': int(policy.target_retry_max_delay_sec), + 'target_timeout_retry_delay_sec': int(policy.target_timeout_retry_delay_sec), + 'docker_layer_checkpoint_delay_sec': int(policy.docker_layer_checkpoint_delay_sec), + 'ci_soft_cooldown_days': int(policy.ci_soft_cooldown_days), + 'soft_skip_reasons': list(policy.soft_skip_reasons), + } + limit_values = dict(limits or {}) + if set(limit_values) != {'candidate_max_items', 'candidate_max_bytes'}: + raise ScanExecutionError('worker assignment limits are invalid') + normalized_limits = { + 'candidate_max_items': int(limit_values['candidate_max_items']), + 'candidate_max_bytes': int(limit_values['candidate_max_bytes']), + } + if ( + not 1 <= normalized_limits['candidate_max_items'] <= 100000 + or not 1024 <= normalized_limits['candidate_max_bytes'] <= 64 * 1024 * 1024 + ): + raise ScanExecutionError('worker assignment limits are outside their bounds') + execution = { + 'source': str(platform or '').strip().lower(), + 'scan_kwargs': event_options, + 'scan_policy': normalize_remote_scan_policy(scan_policy), + 'queue_policy': policy_value, + 'limits': normalized_limits, + } + return canonical_json_sha256(execution), execution + + +def validate_remote_assignment_compatibility( + required, local_build, platform, scan_kwargs, event_scan_options, queue_policy, limits, + scan_policy, *, expected_protocol_version=PROTOCOL_VERSION, +): + required = required if isinstance(required, ScanCompatibility) else ScanCompatibility.from_mapping(required) + validate_worker_build_compatibility({ + 'protocol_version': required.protocol_version, + 'bundle_format_version': required.bundle_format_version, + 'platform_tag': required.platform_tag, + 'code_manifest_sha256': required.code_manifest_sha256, + 'detector_policy_sha256': required.detector_policy_sha256, + }, local_build, expected_protocol_version=expected_protocol_version) + effective, _ = remote_execution_identity( + platform, scan_kwargs, event_scan_options, queue_policy, limits, scan_policy, + ) + if not hmac.compare_digest(effective, required.effective_config_sha256): + raise ScanExecutionError('worker effective config is incompatible') + return required + + +def _remote_snapshot_envelope(value): + if not isinstance(value, dict): + raise ScanExecutionError('remote execution snapshot must be an object') + value = dict(value) + if set(value) != { + 'schema', 'compatibility', 'execution', 'planning', 'credential_ref', + } or value.get('schema') != REMOTE_EXECUTION_SNAPSHOT_SCHEMA: + raise ScanExecutionError('remote execution snapshot shape is invalid') + compatibility = ScanCompatibility.from_mapping(value.get('compatibility')) + execution = dict(value.get('execution') or {}) + if set(execution) != { + 'source', 'scan_kwargs', 'scan_policy', 'queue_policy', 'limits', + }: + raise ScanExecutionError('remote execution snapshot settings are invalid') + source = str(execution.get('source') or '').strip().lower() + effective, normalized_execution = remote_execution_identity( + source, execution.get('scan_kwargs'), execution.get('scan_kwargs'), + execution.get('queue_policy'), execution.get('limits'), + execution.get('scan_policy'), + ) + if normalized_execution['scan_kwargs'].get('trufflehog_config') != PACKAGE_DETECTOR_POLICY: + raise ScanExecutionError('remote execution snapshot policy path is invalid') + if not hmac.compare_digest(effective, compatibility.effective_config_sha256): + raise ScanExecutionError('remote execution snapshot effective config is invalid') + planning = dict(value.get('planning') or {}) + credential_ref = dict(value.get('credential_ref') or {}) + if set(credential_ref) != {'source', 'auth_entry'}: + raise ScanExecutionError('remote execution snapshot credential reference is invalid') + queue_source = str(credential_ref.get('source') or '').strip().lower() + auth_entry = str(credential_ref.get('auth_entry') or '') + if len(auth_entry) > 128 or '\x00' in auth_entry: + raise ScanExecutionError('remote execution snapshot credential reference is invalid') + return compatibility, normalized_execution, planning, queue_source, auth_entry + + +def _normalize_exact_git_v1_planning(planning): + planning = dict(planning or {}) + if set(planning) != { + 'kind', 'git_baseline_depth', 'git_ref_resolution_attempts', + 'git_ref_resolution_timeout_sec', 'git_ref_resolution_max_bytes', + } or planning.get('kind') != 'exact_git_v1': + raise ScanExecutionError('remote execution snapshot planning is invalid') + + try: + baseline_depth = int(planning['git_baseline_depth']) + attempts = int(planning['git_ref_resolution_attempts']) + timeout = float(planning['git_ref_resolution_timeout_sec']) + max_bytes = int(planning['git_ref_resolution_max_bytes']) + except (TypeError, ValueError, OverflowError) as exc: + raise ScanExecutionError('remote execution snapshot planning is invalid') from exc + if ( + isinstance(planning['git_baseline_depth'], bool) + or isinstance(planning['git_ref_resolution_attempts'], bool) + or isinstance(planning['git_ref_resolution_timeout_sec'], bool) + or isinstance(planning['git_ref_resolution_max_bytes'], bool) + or not 1 <= baseline_depth <= 1000000 + or not 1 <= attempts <= 20 + or not 0.1 <= timeout <= 300 + or not 1024 <= max_bytes <= 64 * 1024 * 1024 + ): + raise ScanExecutionError('remote execution snapshot planning is outside its bounds') + return { + 'kind': 'exact_git_v1', + 'git_baseline_depth': baseline_depth, + 'git_ref_resolution_attempts': attempts, + 'git_ref_resolution_timeout_sec': timeout, + 'git_ref_resolution_max_bytes': max_bytes, + } + + +def _normalize_kind_only_planning(planning, kind): + planning = dict(planning or {}) + if planning != {'kind': kind}: + raise ScanExecutionError('remote execution snapshot planning is invalid') + return {'kind': kind} + + +def _normalized_remote_snapshot( + value, *, queue_sources, worker_platform, planning_kind, + planning_normalizer, public_credential, +): + compatibility, execution, planning, queue_source, auth_entry = ( + _remote_snapshot_envelope(value) + ) + platform = execution['source'] + if ( + queue_source not in queue_sources + or (worker_platform is None and platform != queue_source) + or (worker_platform is not None and platform != worker_platform) + or (public_credential and auth_entry) + ): + raise ScanExecutionError('remote execution snapshot source capability is invalid') + normalized_planning = planning_normalizer(planning) + if normalized_planning.get('kind') != planning_kind: + raise ScanExecutionError('remote execution snapshot planning kind is invalid') + normalized = { + 'schema': REMOTE_EXECUTION_SNAPSHOT_SCHEMA, + 'compatibility': compatibility.as_dict(), + 'execution': execution, + 'planning': normalized_planning, + 'credential_ref': {'source': queue_source, 'auth_entry': auth_entry}, + } + encoded = json.dumps( + normalized, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + allow_nan=False, + ).encode('utf-8') + if len(encoded) > MAX_REMOTE_EXECUTION_SNAPSHOT_BYTES: + raise ScanExecutionError('remote execution snapshot exceeds its byte bound') + return normalized + + +def normalize_exact_git_execution_snapshot(value): + return _normalized_remote_snapshot( + value, + queue_sources=frozenset(('github', 'gitlab')), + worker_platform=None, + planning_kind='exact_git_v1', + planning_normalizer=_normalize_exact_git_v1_planning, + public_credential=False, + ) + + +def normalize_docker_direct_execution_snapshot(value): + return _normalized_remote_snapshot( + value, + queue_sources=frozenset(('dockerhub',)), + worker_platform='docker', + planning_kind='docker_direct_v1', + planning_normalizer=lambda planning: _normalize_kind_only_planning( + planning, 'docker_direct_v1', + ), + public_credential=True, + ) + + +def normalize_huggingface_space_execution_snapshot(value): + return _normalized_remote_snapshot( + value, + queue_sources=frozenset(('huggingface',)), + worker_platform='huggingface', + planning_kind='huggingface_space_v1', + planning_normalizer=lambda planning: _normalize_kind_only_planning( + planning, 'huggingface_space_v1', + ), + public_credential=True, + ) + + +def normalize_remote_execution_snapshot(value): + if not isinstance(value, dict) or not isinstance(value.get('planning'), dict): + raise ScanExecutionError('remote execution snapshot planning is invalid') + kind = value['planning'].get('kind') + normalizer = { + 'exact_git_v1': normalize_exact_git_execution_snapshot, + 'docker_direct_v1': normalize_docker_direct_execution_snapshot, + 'huggingface_space_v1': normalize_huggingface_space_execution_snapshot, + }.get(kind) + if normalizer is None: + raise ScanExecutionError('remote execution snapshot planning kind is unsupported') + return normalizer(value) + + +def normalize_docker_direct_execution_target(value): + try: + return parse_dockerhub_digest_target(value) + except (TypeError, ValueError) as exc: + raise ScanExecutionError('Docker direct target is invalid') from exc + + +def normalize_huggingface_space_execution_target(value): + try: + return normalize_huggingface_space_id(value) + except (TypeError, ValueError) as exc: + raise ScanExecutionError('HuggingFace Space target is invalid') from exc + + +def remote_execution_snapshot_sha256(value): + normalized = normalize_remote_execution_snapshot(value) + encoded = json.dumps( + normalized, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + allow_nan=False, + ).encode('utf-8') + return hashlib.sha256(encoded).hexdigest() + + +def _normalize_remote_assignment_deadlines(value, reservation, scan_kwargs): + if not isinstance(value, dict) or set(value) != { + 'target_scan_timeout_seconds', 'result_upload_body_timeout_seconds', + 'assignment_ttl_seconds', 'assignment_issued_at', + 'assignment_deadline_at', + }: + raise ScanExecutionError('worker assignment deadlines shape is invalid') + deadlines = dict(value) + for name in ( + 'target_scan_timeout_seconds', 'result_upload_body_timeout_seconds', + 'assignment_ttl_seconds', + ): + if type(deadlines[name]) is not int or deadlines[name] <= 0: + raise ScanExecutionError('worker assignment deadline value is invalid') + issued_at = deadlines['assignment_issued_at'] + deadline_at = deadlines['assignment_deadline_at'] + if ( + type(issued_at) is not str + or type(deadline_at) is not str + or issued_at != reservation.get('remote_issued_at') + or deadline_at != reservation.get('remote_expires_at') + ): + raise ScanExecutionError('worker assignment deadline changed after reservation') + try: + issued = datetime.fromisoformat(issued_at) + deadline = datetime.fromisoformat(deadline_at) + except ValueError as exc: + raise ScanExecutionError('worker assignment deadline timestamp is invalid') from exc + if ( + issued.tzinfo is None + or deadline.tzinfo is None + or issued.utcoffset() != timedelta(0) + or deadline.utcoffset() != timedelta(0) + or issued.isoformat(timespec='seconds') != issued_at + or deadline.isoformat(timespec='seconds') != deadline_at + or deadline - issued != timedelta(seconds=deadlines['assignment_ttl_seconds']) + ): + raise ScanExecutionError('worker assignment deadline timestamp is invalid') + if deadlines['target_scan_timeout_seconds'] != scan_kwargs.get('timeout_sec'): + raise ScanExecutionError('worker assignment target scan timeout changed') + return deadlines + + +def _validate_remote_assignment( + assignment, local_build, expected_protocol_version, + package_capabilities=None, +): + if not isinstance(assignment, dict) or set(assignment) != { + 'reservation', 'deadlines', 'compatibility', 'scan_kwargs', 'event_scan_options', + 'queue_policy', 'limits', 'scan_policy', 'execution_snapshot', + 'execution_snapshot_sha256', 'execution_plan', + }: + raise ScanExecutionError('worker assignment shape is invalid') + reservation_value = dict(assignment.get('reservation') or {}) + try: + reservation = BundleReservation.from_mapping(reservation_value) + except (KeyError, TypeError, ValueError) as exc: + raise ScanExecutionError('worker assignment reservation is invalid') from exc + source = str(reservation.source or '').strip().lower() + platform = str(reservation.platform or '').strip().lower() + if ( + reservation_value.get('assignment_kind') != 'remote' + or not source or not platform + ): + raise ScanExecutionError('worker assignment reservation is not remote') + + snapshot = normalize_remote_execution_snapshot( + assignment.get('execution_snapshot'), + ) + snapshot_sha256 = _digest( + assignment.get('execution_snapshot_sha256'), 'execution snapshot', + ) + if not hmac.compare_digest( + snapshot_sha256, remote_execution_snapshot_sha256(snapshot), + ): + raise ScanExecutionError('worker assignment execution snapshot hash changed') + if snapshot['compatibility']['protocol_version'] != expected_protocol_version: + raise ScanExecutionError('worker assignment snapshot protocol is incompatible') + required = ScanCompatibility.from_mapping(assignment.get('compatibility')) + if required.as_dict() != snapshot['compatibility']: + raise ScanExecutionError('worker assignment compatibility changed after admission') + validate_remote_assignment_compatibility( + required, local_build, platform, assignment.get('scan_kwargs'), + assignment.get('event_scan_options'), assignment.get('queue_policy'), + assignment.get('limits'), assignment.get('scan_policy'), + expected_protocol_version=expected_protocol_version, + ) + effective, execution = remote_execution_identity( + platform, assignment.get('scan_kwargs'), + assignment.get('event_scan_options'), assignment.get('queue_policy'), + assignment.get('limits'), assignment.get('scan_policy'), + ) + if execution != snapshot['execution']: + raise ScanExecutionError('worker assignment settings changed after admission') + if ( + snapshot['credential_ref']['source'] != source + or snapshot['execution']['source'] != platform + or not hmac.compare_digest(effective, required.effective_config_sha256) + or not hmac.compare_digest( + str(reservation_value.get('remote_effective_config_sha256') or ''), + required.effective_config_sha256, + ) + ): + raise ScanExecutionError('worker assignment identity changed after admission') + + planning_kind = snapshot['planning']['kind'] + if expected_protocol_version == 1 and planning_kind != 'exact_git_v1': + raise ScanExecutionError('legacy worker assignment planning kind is invalid') + capability = (source, platform, planning_kind) + if package_capabilities is not None: + capabilities = { + tuple(value) for value in package_capabilities + if isinstance(value, (list, tuple)) and len(value) == 3 + } + if capability not in capabilities: + raise ScanExecutionError( + 'worker assignment capability is not supported by this package' + ) + + plan = assignment.get('execution_plan') + if not isinstance(plan, dict) or set(plan) != { + 'kind', 'execution_target', 'bound_plan', + } or plan.get('kind') != planning_kind: + raise ScanExecutionError('worker assignment execution plan is invalid') + scan_kwargs = dict(assignment.get('scan_kwargs') or {}) + event_scan_options = dict(assignment.get('event_scan_options') or {}) + deadlines = _normalize_remote_assignment_deadlines( + assignment.get('deadlines'), reservation_value, scan_kwargs, + ) + if planning_kind == 'exact_git_v1': + if ( + source not in {'github', 'gitlab'} or platform != source + or str(plan.get('execution_target') or '') != reservation.target + or not isinstance(plan.get('bound_plan'), dict) + or scan_kwargs.get('git_plan') != plan['bound_plan'] + or scan_kwargs.get('docker_layer_work') is not None + ): + raise ScanExecutionError('worker assignment exact Git plan is invalid') + execution_target = reservation.target + elif planning_kind == 'docker_direct_v1': + if source != 'dockerhub' or platform != 'docker': + raise ScanExecutionError('worker assignment Docker capability is invalid') + parsed = normalize_docker_direct_execution_target(reservation.target) + execution_target = parsed['image'] + if parsed['normalized_target'] != reservation.normalized_target: + raise ScanExecutionError('worker assignment Docker identity is invalid') + elif planning_kind == 'huggingface_space_v1': + if source != 'huggingface' or platform != 'huggingface': + raise ScanExecutionError('worker assignment HuggingFace capability is invalid') + execution_target = normalize_huggingface_space_execution_target( + reservation.target, + ) + if normalize_target(execution_target, platform) != reservation.normalized_target: + raise ScanExecutionError('worker assignment HuggingFace identity is invalid') + else: + raise ScanExecutionError('worker assignment planning kind is unsupported') + if planning_kind != 'exact_git_v1' and ( + plan.get('bound_plan') is not None + or str(plan.get('execution_target') or '') != execution_target + or scan_kwargs != event_scan_options + or any(name in scan_kwargs for name in ( + 'token', 'git_plan', 'docker_layer_work', + )) + ): + raise ScanExecutionError('worker assignment direct plan is not credential-free') + return { + 'reservation': reservation, + 'snapshot': snapshot, + 'snapshot_sha256': snapshot_sha256, + 'compatibility': required, + 'planning_kind': planning_kind, + 'execution_target': execution_target, + 'deadlines': deadlines, + 'execution_plan': { + 'kind': planning_kind, + 'execution_target': execution_target, + 'bound_plan': plan.get('bound_plan'), + }, + } + + +def validate_protocol1_remote_assignment(assignment, local_build): + return _validate_remote_assignment(assignment, local_build, 1) + + +def validate_protocol2_remote_assignment( + assignment, local_build, package_capabilities=None, +): + return _validate_remote_assignment( + assignment, local_build, PROTOCOL_VERSION, package_capabilities, + ) + + +def _first_error_line(result): + for error in result.get('errors') or (): + for line in str(error).splitlines(): + line = line.strip() + if not line: + continue + try: + payload = json.loads(line) + except (TypeError, ValueError): + return line[:300] + return str(payload.get('error') or payload.get('msg') or line)[:300] + return '' + + +def _docker_result_resets_attempts(result): + if result.get('docker_layer_plan') is None or not result.get('retryable', False): + return False + execution = result.get('docker_layer_execution') + records = execution.get('blobs') if isinstance(execution, dict) else None + descriptors = result['docker_layer_plan'].get('descriptors') + if not isinstance(records, list) or not isinstance(descriptors, list): + return False + if not any( + item.get('coverage_state') in ('selected', 'shared_pending') + for item in descriptors if isinstance(item, dict) + ): + return False + return not any( + item.get('status') in ('retryable_failed', 'terminal_failed') + for item in records if isinstance(item, dict) + ) + + +def queue_disposition_for_result(result, platform, attempts, policy, *, now=None): + policy = policy if isinstance(policy, QueueDispositionPolicy) else QueueDispositionPolicy(**policy) + attempts = max(0, int(attempts or 0)) + max_attempts = max(1, int(policy.target_retry_max_attempts or 1)) + now = now or datetime.now(timezone.utc) + skipped = str(result.get('skipped') or '') + if skipped in set(policy.soft_skip_reasons): + return { + 'queue_status': 'deferred', 'queue_error': skipped, + 'available_after': (now + timedelta(days=max(1, policy.ci_soft_cooldown_days))).isoformat(timespec='seconds'), + 'reset_attempts': True, + } + if result.get('docker_layer_plan') is not None: + if not result.get('errors'): + status, available_after, reset = 'done', None, False + elif not bool(result.get('retryable', False)): + status, available_after, reset = 'failed', None, False + else: + reset = _docker_result_resets_attempts(result) + if not reset and attempts >= max_attempts: + status, available_after = 'failed', None + else: + status = 'deferred' + available_after = (now + timedelta(seconds=max( + 1, int(policy.docker_layer_checkpoint_delay_sec or 1), + ))).isoformat(timespec='seconds') + return { + 'queue_status': status, + 'queue_error': _first_error_line(result) if result.get('errors') else None, + 'available_after': available_after, 'reset_attempts': reset, + } + if not result.get('errors'): + return { + 'queue_status': 'done', 'queue_error': None, + 'available_after': None, 'reset_attempts': False, + } + timed_out = bool((result.get('scan_meta') or {}).get('command_timed_out')) \ + or result.get('error_class') == 'timeout' + if timed_out: + status = 'failed' if attempts >= max_attempts else 'deferred' + delay = max(60, int(policy.target_timeout_retry_delay_sec or 60)) + elif result.get('source_failure'): + status = 'failed' if not result.get('retryable', True) and attempts >= max_attempts else 'deferred' + delay = max(1, int(policy.target_retry_max_delay_sec or 1)) + elif not result.get('retryable', True) or attempts >= max_attempts: + status, delay = 'failed', 0 + else: + status = 'deferred' + base = max(1, int(policy.target_retry_base_delay_sec or 1)) + maximum = max(base, int(policy.target_retry_max_delay_sec or base)) + delay = min(maximum, base * (2 ** max(0, attempts - 1))) + return { + 'queue_status': status, + 'queue_error': _first_error_line(result), + 'available_after': ( + (now + timedelta(seconds=delay)).isoformat(timespec='seconds') + if status == 'deferred' else None + ), + 'reset_attempts': bool( + (result.get('source_failure') and result.get('retryable', True)) + or _docker_result_resets_attempts(result) + ), + } + + +def stage_scan_result_in_scope( + result, reservation, bundle_root, event_scan_options, queue_policy, *, attempts, + candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024, + require_s_drive=False, fault=None, diagnostic_slot_id=0, +): + reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation) + disposition = queue_disposition_for_result( + result, reservation.platform, attempts, queue_policy, + ) + return stage_result_bundle( + result, reservation, bundle_root, event_scan_options, disposition, + candidate_max_items=candidate_max_items, + candidate_max_bytes=candidate_max_bytes, + require_s_drive=require_s_drive, fault=fault, + diagnostic_slot_id=diagnostic_slot_id, + diagnostic_attempt=max(1, int(attempts or 1)), + ) + + +def execute_planned_result_in_scope( + reservation, bundle_root, scan_kwargs, event_scan_options, queue_policy, *, + attempts, candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024, + execution_target=None, scan_meta_defaults=None, require_s_drive=False, + phase_callback=None, bundle_fault=None, diagnostic_slot_id=0, +): + reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation) + scan_kwargs = validate_scan_kwargs(reservation.platform, scan_kwargs) + target = reservation.target if execution_target is None else execution_target + expected = reservation.normalized_target or normalize_target(reservation.target, reservation.platform) + if normalize_target(target, reservation.platform) != expected: + raise ScanExecutionError('execution target does not match the reservation') + with client_scan_phase_events(phase_callback): + result = scan_target_result( + target, reservation.platform, reservation.scan_event_id, scan_kwargs, + ) + result['target'] = reservation.target + result['scan_type'] = reservation.platform + if scan_meta_defaults: + metadata = result.setdefault('scan_meta', {}) + if not isinstance(metadata, dict): + raise ScanExecutionError('scanner metadata is invalid') + for name, value in dict(scan_meta_defaults).items(): + metadata.setdefault(name, value) + if phase_callback is not None: + phase_callback('cleaning') + cleanup = cleanup_assignment_work_dir() + phase_callback('cleaning', cleanup) + phase_callback('bundling') + return stage_scan_result_in_scope( + result, reservation, bundle_root, event_scan_options, queue_policy, + attempts=attempts, candidate_max_items=candidate_max_items, + candidate_max_bytes=candidate_max_bytes, require_s_drive=require_s_drive, + fault=bundle_fault, diagnostic_slot_id=diagnostic_slot_id, + ) + + +def execute_planned_claim( + reservation, bundle_root, scan_kwargs, event_scan_options, queue_policy, scan_policy, *, + attempts, candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024, + lease=None, execution_target=None, scan_meta_defaults=None, + require_s_drive=False, phase_callback=None, bundle_fault=None, + diagnostic_slot_id=0, +): + timeout = validate_scan_kwargs( + reservation.platform if isinstance(reservation, BundleReservation) else reservation.get('platform'), + scan_kwargs, + )['timeout_sec'] + platform = reservation.platform if isinstance(reservation, BundleReservation) else reservation.get('platform') + policy = normalize_remote_scan_policy(scan_policy) + if phase_callback is not None: + phase_callback('waiting_permit', {'boundary': 'scan_slot_scope'}) + with client_scan_execution_policy(policy): + with scan_slot_scope(['scan-target', platform], timeout, lease=lease): + return execute_planned_result_in_scope( + reservation, bundle_root, scan_kwargs, event_scan_options, queue_policy, + attempts=attempts, candidate_max_items=candidate_max_items, + candidate_max_bytes=candidate_max_bytes, execution_target=execution_target, + scan_meta_defaults=scan_meta_defaults, require_s_drive=require_s_drive, + phase_callback=phase_callback, bundle_fault=bundle_fault, + diagnostic_slot_id=diagnostic_slot_id, + ) + + +def execute_protocol2_remote_claim( + validated_assignment, bundle_root, scan_kwargs, event_scan_options, + queue_policy, scan_policy, *, attempts, candidate_max_items=2000, + candidate_max_bytes=2 * 1024 * 1024, lease=None, + scan_meta_defaults=None, require_s_drive=False, phase_callback=None, + bundle_fault=None, diagnostic_slot_id=0, +): + kind = str(validated_assignment.get('planning_kind') or '') + authority = ( + client_remote_execution_binding(kind) + if kind in {'docker_direct_v1', 'huggingface_space_v1'} + else nullcontext() + ) + with authority: + return execute_planned_claim( + validated_assignment['reservation'], bundle_root, scan_kwargs, + event_scan_options, queue_policy, scan_policy, + attempts=attempts, + candidate_max_items=candidate_max_items, + candidate_max_bytes=candidate_max_bytes, + lease=lease, + execution_target=validated_assignment['execution_target'], + scan_meta_defaults={ + **dict(scan_meta_defaults or {}), 'planning_kind': kind, + }, + require_s_drive=require_s_drive, + phase_callback=phase_callback, bundle_fault=bundle_fault, + diagnostic_slot_id=diagnostic_slot_id, + ) + + +def canonical_json_sha256(value): + return hashlib.sha256(json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('utf-8')).hexdigest() diff --git a/app/scan_manager.py b/app/scan_manager.py new file mode 100644 index 0000000..6c878ac --- /dev/null +++ b/app/scan_manager.py @@ -0,0 +1,71 @@ +import json +import threading + + +RETIRED_MESSAGE = 'Legacy ScanManager mutation controls are retired; use the authenticated supervisor.' + + +class ScanManager: + """Read-only compatibility shell for the retired legacy scanner UI.""" + + def __init__(self): + self.current_scan = None + self.scan_thread = None + self.results = [] + self.progress = { + 'total': 0, + 'completed': 0, + 'failed': 0, + 'secrets_found': 0, + 'current_target': None, + 'status': 'retired', + } + self.scan_lock = threading.Lock() + self.scanned_targets = set() + self.scan_history = [] + + def start_scan(self, scan_type, targets, scan_options): + return False, RETIRED_MESSAGE + + def pause_scan(self): + return False + + def resume_scan(self): + return False + + def stop_scan(self): + return False + + def is_scanning(self): + return False + + def get_progress(self): + with self.scan_lock: + return self.progress.copy() + + def get_results(self): + with self.scan_lock: + return self.results.copy() + + def clear_results(self): + return False + + def get_scan_info(self): + with self.scan_lock: + return self.current_scan.copy() if self.current_scan else None + + def export_results(self, format='json'): + with self.scan_lock: + if str(format).lower() == 'json': + return json.dumps(self.results, indent=2, default=str) + return None + + def clear_scan_history(self): + return False + + def get_scan_history(self): + with self.scan_lock: + return self.scan_history.copy() + + def get_scanned_targets_count(self): + return len(self.scanned_targets) diff --git a/app/scanner.py b/app/scanner.py new file mode 100644 index 0000000..568a7f9 --- /dev/null +++ b/app/scanner.py @@ -0,0 +1,17045 @@ +import os +import sys +import socket +import sqlite3 +import requests +import json +import re +import subprocess +import concurrent.futures +import threading +import time +import multiprocessing +import tempfile +import shutil +import tarfile +import zipfile +import gzip +import bz2 +import codecs +import lzma +import base64 +import binascii +import copy +import hashlib +import io +import logging +import math +import stat +import uuid +import ctypes +import ipaddress +from email.utils import parsedate_to_datetime +from html.parser import HTMLParser +from collections import Counter +from contextlib import contextmanager +from contextvars import ContextVar +from dataclasses import dataclass, replace +from urllib.parse import parse_qsl, quote, urlencode, urljoin, urlsplit, urlunsplit +from datetime import datetime, timedelta, timezone + +import urllib3.util.connection as urllib3_connection +import urllib3.exceptions as urllib3_exceptions + +from paths import default_project_paths +from scanner_db import ( + DOCKER_ADAPTIVE_PAYLOAD_CLASSES, + canonical_docker_layer_plan_bytes, + canonical_git_scan_plan_bytes, + normalize_target, + sanitize_endpoint, + sanitize_endpoint_host, + sanitize_postman_context, + target_status, + validate_docker_layer_plan, + validate_git_resolution, +) +from owned_process import OwnedProcess, run_owned +from process_identity import ( + current_process_identity, + exact_process_identity_state, + open_process, + serialize_process_identity, +) +from janitor import JanitorBudget, bounded_remove_tree +from runtime_security import ( + atomic_write_private_json, + canonical_path, + durable_replace, + durable_unlink, + ensure_private_directory, + harden_private_directory, + harden_private_file, + harden_private_tree, + is_reparse_point, + PrivateFileLock, + private_directory_ready, + private_file_ready, + read_private_json, + reject_reparse_components, + require_private_directory, + require_private_file, + sha256_file, +) +from target_identity import ( + normalize_docker_digest, + normalize_huggingface_space_id, + parse_dockerhub_digest_target, + parse_docker_target, + postman_target_identity as semantic_postman_target_identity, + validate_docker_image_reference, +) +from docker_depth_experiment import ( + DOCKER_DEPTH_SELECTOR_VERSION, + canonical_docker_depth_selection_evidence_hash, + canonical_selector_hash, + select_docker_layer_graphs, + validate_docker_images_per_repository, +) +from lifecycle_authority import ( + CHILD_KIND_ENV, + REMOTE_WORKER_CODE_AUTHORITY_FILES, + verify_code_manifest, + require_active_supervisor_child, + resolve_manifest_executable, + strip_supervisor_credentials, +) +from keycheck_candidates import extract_candidates, extract_structured_candidates +from result_bundle import BundleReservation, ResultBundleWriter +from worker_contracts import ( + AssignmentOutcome, + DiagnosticCategory, + DiagnosticExceptionContext, + DiagnosticHTTPContext, + DiagnosticKind, + DiagnosticProcessContext, + MAX_DIAGNOSTIC_BODY_BYTES, + MAX_DIAGNOSTIC_LOG_BYTES, + ScanOutcome, + WorkerPhase, + build_diagnostic_envelope, + diagnostic_material_bytes, + make_body_material, + make_log_material, +) + + +def configure_requests_networking(): + if os.getenv('SCANNER_FORCE_IPV4', '1').strip().lower() in ('0', 'false', 'no', 'off'): + return + urllib3_connection.allowed_gai_family = lambda: socket.AF_INET + + +logger = logging.getLogger(__name__) +_runtime_initialized = False +_cleanup_registered = False +_client_scan_manifest = ContextVar('client_scan_manifest', default=None) +_client_scan_policy = ContextVar('client_scan_policy', default=None) +_client_remote_execution_kind = ContextVar( + 'client_remote_execution_kind', default=None, +) +_client_scan_phase_callback = ContextVar( + 'client_scan_phase_callback', default=None, +) + + +@contextmanager +def client_scan_launch_authority(manifest, expected_sha256=None): + verified = verify_code_manifest( + manifest, + expected_sha256=expected_sha256, + required_names=REMOTE_WORKER_CODE_AUTHORITY_FILES, + external_names=(), + ) + token = _client_scan_manifest.set(verified) + try: + yield verified + finally: + _client_scan_manifest.reset(token) + + +@contextmanager +def client_scan_execution_policy(policy): + if not isinstance(policy, dict): + raise RuntimeError('remote scan execution policy is invalid') + token = _client_scan_policy.set(dict(policy)) + try: + yield + finally: + _client_scan_policy.reset(token) + + +@contextmanager +def client_remote_execution_binding(planning_kind): + if _client_scan_manifest.get() is None: + raise RuntimeError('remote direct execution requires client launch authority') + kind = str(planning_kind or '') + if kind not in {'docker_direct_v1', 'huggingface_space_v1'}: + raise RuntimeError('remote direct execution kind is invalid') + token = _client_remote_execution_kind.set(kind) + try: + yield + finally: + _client_remote_execution_kind.reset(token) + + +@contextmanager +def client_scan_phase_events(callback): + if callback is not None and not callable(callback): + raise TypeError('scan phase callback must be callable') + token = _client_scan_phase_callback.set(callback) + try: + yield + finally: + _client_scan_phase_callback.reset(token) + + +def emit_client_scan_phase(phase, progress=None): + callback = _client_scan_phase_callback.get() + if callback is not None: + callback(phase, dict(progress or {})) + + +def _scan_policy_value(name, default): + policy = _client_scan_policy.get() + if policy is not None: + if name not in policy: + raise RuntimeError('remote scan execution policy is incomplete') + return policy[name] + return getattr(scan_config, name, default) + + +def initialize_scanner_runtime(*, preflight_complete=False, register_cleanup=True): + """Apply process-global scanner setup only after lifecycle preflight.""" + global _runtime_initialized, _cleanup_registered + if not preflight_complete: + raise RuntimeError('scanner runtime initialization requires completed lifecycle preflight') + if not _runtime_initialized: + configure_requests_networking() + logging.basicConfig( + level=logging.INFO, + format='%(asctime)s - %(levelname)s - %(message)s', + ) + _runtime_initialized = True + # Stale tree ownership belongs to the isolated janitor, never an atexit hook. + _cleanup_registered = False + + +def require_scanner_runtime_initialized(): + if not _runtime_initialized: + raise RuntimeError('scanner runtime is not initialized after lifecycle preflight') + + +def bool_setting(value, default=False): + if value is None: + return default + if isinstance(value, bool): + return value + return str(value).strip().lower() in ('1', 'true', 'yes', 'on') + + +def int_setting(value, default): + try: + return int(value) + except (TypeError, ValueError): + return default + + +def float_setting(value, default): + try: + return float(value) + except (TypeError, ValueError): + return default + + +def csv_items(value): + if not value: + return [] + if isinstance(value, str): + return [item.strip() for item in value.split(',') if item.strip()] + return [str(item).strip() for item in value if str(item).strip()] + + +DEFAULT_DROP_DETECTORS = ( + 'Privacy', + 'URI', + 'JDBC', + 'Postgres', + 'MongoDB', + 'SQLServer', + 'Box', + 'ZohoCRM', + 'Accuweather', + 'Roaring', + 'Flatio', + 'LinkPreview', + 'RailwayApp', +) + + +class RateLimitError(Exception): + def __init__( + self, source, message, reset_at=None, category='rate_limit', + retryable=True, auth_related=True, diagnostic_http=None, + ): + super().__init__(message) + self.source = source + self.reset_at = reset_at + self.category = category + self.retryable = retryable + self.auth_related = auth_related + self.diagnostic_http = ( + dict(diagnostic_http) if isinstance(diagnostic_http, dict) else None + ) + + +def response_message(response): + if response is None: + return '' + payload_bytes = bytearray() + try: + digest = hashlib.sha256() + original_size = 0 + for chunk in response.iter_content(chunk_size=4096): + if not chunk: + continue + digest.update(chunk) + original_size += len(chunk) + remaining = MAX_DIAGNOSTIC_BODY_BYTES - len(payload_bytes) + if remaining > 0: + payload_bytes.extend(chunk[:remaining]) + captured = bytes(payload_bytes) + material = make_body_material(captured) + if original_size > len(captured): + material = replace( + material, + original_size=original_size, + sha256=digest.hexdigest(), + truncated=True, + ) + response._truf_diagnostic_body_material = material + response._truf_diagnostic_body = captured + response._truf_diagnostic_body_truncated = material.truncated + payload = json.loads(captured.decode('utf-8', errors='strict')) + if isinstance(payload, dict): + message = str(payload.get('message') or payload.get('error') or payload) + else: + message = str(payload) + return message.replace('\x00', '\\u0000') + except Exception: + captured = bytes(payload_bytes[:MAX_DIAGNOSTIC_BODY_BYTES]) + response._truf_diagnostic_body = captured + return captured[:500].decode( + 'utf-8', errors='replace' + ).replace('\x00', '\\u0000') + +def retry_after_reset(response): + if response is None: + return None + retry_after = response.headers.get('Retry-After') + if retry_after: + try: + return (datetime.now(timezone.utc) + timedelta(seconds=int(retry_after))).isoformat(timespec='seconds') + except (TypeError, ValueError): + return None + return None + +def build_api_error(source, category, message, response=None, reset_at=None, retryable=True, auth_related=False): + if reset_at is None: + reset_at = retry_after_reset(response) + diagnostic_http = None + if response is not None and type(getattr(response, 'status_code', None)) is int: + headers = getattr(response, 'headers', {}) or {} + body_material = getattr(response, '_truf_diagnostic_body_material', None) + if body_material is None and ( + not getattr(response, 'raw', None) + or getattr(response, '_content_consumed', False) + ): + captured = getattr(response, 'content', b'') + body = captured if isinstance(captured, bytes) else str(captured).encode('utf-8') + body_material = make_body_material(body) + diagnostic_http = { + 'operation': f'{source}-api', + 'status_code': int(response.status_code), + 'content_type': str(headers.get('Content-Type') or '') or None, + 'request_id': str( + headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or '' + ) or None, + 'headers_b64': base64.b64encode(json.dumps( + {str(key): str(value) for key, value in headers.items()}, + ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).decode('ascii'), + } + if body_material is not None: + body = diagnostic_material_bytes(body_material) + diagnostic_http.update({ + 'body_b64': base64.b64encode(body).decode('ascii'), + 'body_original_size': body_material.original_size, + 'body_stored_size': body_material.stored_size, + 'body_sha256': body_material.sha256, + 'body_capture_truncated': body_material.truncated, + }) + else: + diagnostic_http['body_capture_truncated'] = False + return RateLimitError( + source, + message, + reset_at=reset_at, + category=category, + retryable=retryable, + auth_related=auth_related, + diagnostic_http=diagnostic_http, + ) + +def github_api_error(response): + status = response.status_code if response is not None else None + message = response_message(response) + lower_message = message.lower() + reset_at = github_rate_limit_reset(response) or retry_after_reset(response) + if status == 401: + return build_api_error('github', 'auth_invalid', f'GitHub API authentication failed (HTTP 401): {message}', response, retryable=False, auth_related=True) + if status == 403: + remaining = response.headers.get('X-RateLimit-Remaining') if response is not None else None + if remaining == '0': + return build_api_error('github', 'rate_limit', f'GitHub API rate limit hit: {message}', response, reset_at, auth_related=True) + if 'secondary rate limit' in lower_message or 'abuse' in lower_message: + return build_api_error('github', 'secondary_rate_limit', f'GitHub secondary rate limit hit: {message}', response, reset_at, auth_related=True) + return build_api_error('github', 'auth_forbidden', f'GitHub API access forbidden (HTTP 403): {message}', response, retryable=False, auth_related=True) + if status == 429: + return build_api_error('github', 'rate_limit', f'GitHub API returned HTTP 429: {message}', response, reset_at, auth_related=True) + if status == 422: + return build_api_error('github', 'query_invalid', f'GitHub search query invalid (HTTP 422): {message}', response, retryable=False, auth_related=False) + if status == 404: + return build_api_error('github', 'not_found', f'GitHub API resource not found (HTTP 404): {message}', response, retryable=False, auth_related=False) + if status and status >= 500: + return build_api_error('github', 'server_error', f'GitHub API server error (HTTP {status}): {message}', response, auth_related=False) + return build_api_error('github', 'api', f'GitHub API error (HTTP {status}): {message}', response, auth_related=False) + +def gitlab_api_error(response): + status = response.status_code if response is not None else None + message = response_message(response) + reset_at = gitlab_rate_limit_reset(response) or retry_after_reset(response) + if status == 401: + return build_api_error('gitlab', 'auth_invalid', f'GitLab API authentication failed (HTTP 401): {message}', response, retryable=False, auth_related=True) + if status == 403: + return build_api_error('gitlab', 'auth_forbidden', f'GitLab API access forbidden (HTTP 403): {message}', response, retryable=False, auth_related=True) + if status == 429: + return build_api_error('gitlab', 'rate_limit', f'GitLab API returned HTTP 429: {message}', response, reset_at, auth_related=True) + if status == 404: + return build_api_error('gitlab', 'not_found', f'GitLab API resource not found (HTTP 404): {message}', response, retryable=False, auth_related=False) + if status and status >= 500: + return build_api_error('gitlab', 'server_error', f'GitLab API server error (HTTP {status}): {message}', response, auth_related=False) + return build_api_error('gitlab', 'api', f'GitLab API error (HTTP {status}): {message}', response, auth_related=False) + +# ===================== +# GLOBAL CONFIGURATION +# ===================== +class ScanConfig: + def __init__(self): + defaults = default_project_paths() + self.git_timeout = 900 + self.docker_timeout = 1800 + self.detectors = "" + self.exclude_detectors = os.getenv("TRUFFLEHOG_EXCLUDE_DETECTORS", "github.v1,gitlab.v1,GitHubOauth2") + self.no_verification = os.getenv("TRUFFLEHOG_NO_VERIFICATION", "0").strip().lower() in ("1", "true", "yes", "on") + self.strict_git_provider_token_filter = os.getenv( + "STRICT_GIT_PROVIDER_TOKEN_FILTER", + "1", + ).strip().lower() not in ("0", "false", "no", "off") + self.drop_detectors = csv_items(os.getenv("SCANNER_DROP_DETECTORS")) + self.webhook_url = None + self.max_concurrent = min(10, max(1, multiprocessing.cpu_count() * 2)) + self.trufflehog_path = os.getenv( + "TRUFFLEHOG_PATH", + defaults['trufflehog_path'] + ) + self.trufflehog_config = os.getenv("TRUFFLEHOG_CONFIG", "") + self.trufflehog_job_memory_limit_bytes = int_setting( + os.getenv("TRUFFLEHOG_JOB_MEMORY_LIMIT_BYTES"), + 4096 * 1024 * 1024, + ) + self.trufflehog_windows_job_cpu_weight = int_setting( + os.getenv("TRUFFLEHOG_WINDOWS_JOB_CPU_WEIGHT"), 0, + ) + self.trufflehog_windows_memory_priority = int_setting( + os.getenv("TRUFFLEHOG_WINDOWS_MEMORY_PRIORITY"), 0, + ) + self.trufflehog_stdout_max_mb = int_setting(os.getenv('TRUFFLEHOG_STDOUT_MAX_MB'), 32) + self.trufflehog_stderr_max_mb = int_setting(os.getenv('TRUFFLEHOG_STDERR_MAX_MB'), 8) + self.trufflehog_max_findings_per_target = int_setting( + os.getenv('TRUFFLEHOG_MAX_FINDINGS_PER_TARGET'), 20000, + ) + self.work_dir = os.getenv("TRUFFLEHOG_WORK_DIR", defaults['work_dir']) + self.runtime_dir = defaults['runtime_dir'] + self.results_dir = os.getenv("SCAN_RESULTS_DIR", defaults['results_dir']) + self.result_spool_dir = os.getenv("SCANNER_RESULT_SPOOL_DIR", defaults['result_spool_dir']) + self.result_spool_max_event_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_EVENT_BYTES"), 192 * 1024 * 1024) + self.result_spool_max_events = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_EVENTS"), 10000) + self.result_spool_max_total_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MAX_TOTAL_BYTES"), 2 * 1024 * 1024 * 1024) + self.result_spool_min_free_bytes = int_setting(os.getenv("SCANNER_RESULT_SPOOL_MIN_FREE_BYTES"), 1024 * 1024 * 1024) + self.result_bundle_dir = os.getenv('SCANNER_RESULT_BUNDLE_DIR', defaults['result_bundle_dir']) + self.result_bundle_max_event_bytes = int_setting( + os.getenv('SCANNER_RESULT_BUNDLE_MAX_EVENT_BYTES'), 64 * 1024 * 1024, + ) + self.scan_outbox_max_pending_items = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_ITEMS'), 10000) + self.scan_outbox_max_pending_bytes = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_BYTES'), 1024 * 1024 * 1024) + self.scan_outbox_max_pending_age_sec = int_setting(os.getenv('SCAN_OUTBOX_MAX_PENDING_AGE_SEC'), 24 * 60 * 60) + self.queue_dir = defaults['queue_dir'] + self.keycheck_dir = defaults['keycheck_dir'] + self.postman_cache_dir = defaults['postman_cache_dir'] + self.postman_cache_max_items = int_setting(os.getenv('POSTMAN_CACHE_MAX_ITEMS'), 100000) + self.postman_cache_max_bytes = int_setting(os.getenv('POSTMAN_CACHE_MAX_BYTES'), 20 * 1024 * 1024 * 1024) + self.postman_cache_min_free_bytes = int_setting(os.getenv('POSTMAN_CACHE_MIN_FREE_BYTES'), 20 * 1024 * 1024 * 1024) + self.postman_cache_lock_timeout_sec = int_setting(os.getenv('POSTMAN_CACHE_LOCK_TIMEOUT_SEC'), 30) + self.postman_discovery_max_artifacts_per_cycle = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ARTIFACTS_PER_CYCLE'), 1000) + self.postman_discovery_max_artifacts_per_page = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ARTIFACTS_PER_PAGE'), 100) + self.postman_discovery_max_bytes_per_cycle = int_setting(os.getenv('POSTMAN_DISCOVERY_MAX_BYTES_PER_CYCLE'), 1024 * 1024 * 1024) + self.postman_discovery_max_elapsed_sec = float_setting(os.getenv('POSTMAN_DISCOVERY_MAX_ELAPSED_SEC'), 300.0) + self.postman_package_harvest_max_artifacts = int_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_ARTIFACTS'), 100) + self.postman_package_harvest_max_bytes = int_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_BYTES'), 128 * 1024 * 1024) + self.postman_package_harvest_max_elapsed_sec = float_setting(os.getenv('POSTMAN_PACKAGE_HARVEST_MAX_ELAPSED_SEC'), 30.0) + self.postman_context_max_input_bytes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_INPUT_BYTES'), 16 * 1024 * 1024) + self.postman_context_max_nodes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_NODES'), 100000) + self.postman_context_max_depth = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_DEPTH'), 64) + self.postman_context_max_scalar_bytes = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_SCALAR_BYTES'), 16 * 1024 * 1024) + self.postman_context_max_items = int_setting(os.getenv('POSTMAN_CONTEXT_MAX_ITEMS'), 50000) + self.context_enrichment_max_source_bytes = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_SOURCE_BYTES'), 16 * 1024 * 1024) + self.context_enrichment_max_findings = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_FINDINGS'), 2000) + self.context_enrichment_max_postman_comparisons = int_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_POSTMAN_COMPARISONS'), 200000) + self.context_enrichment_max_elapsed_sec = float_setting(os.getenv('SCANNER_CONTEXT_ENRICHMENT_MAX_ELAPSED_SEC'), 5.0) + self.trufflehog_diagnostic_max_lines = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINES'), 2000) + self.trufflehog_diagnostic_max_line_chars = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINE_CHARS'), 8192) + self.trufflehog_diagnostic_max_line_bytes = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_LINE_BYTES'), 8192) + self.trufflehog_diagnostic_max_errors = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_ERRORS'), 200) + self.trufflehog_diagnostic_max_warnings = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_WARNINGS'), 200) + self.trufflehog_diagnostic_max_unclassified = int_setting(os.getenv('TRUFFLEHOG_DIAGNOSTIC_MAX_UNCLASSIFIED'), 20) + self.keycheck_input_max_line_bytes = int_setting(os.getenv('KEYCHECK_INPUT_MAX_LINE_BYTES'), 16 * 1024 * 1024) + self.keycheck_candidate_artifact_max_items = int_setting(os.getenv('KEYCHECK_CANDIDATE_ARTIFACT_MAX_ITEMS'), 2000) + self.keycheck_candidate_artifact_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_ARTIFACT_MAX_BYTES'), 2 * 1024 * 1024) + self.keycheck_candidate_file_max_items = int_setting(os.getenv('KEYCHECK_CANDIDATE_FILE_MAX_ITEMS'), 100000) + self.keycheck_candidate_file_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_FILE_MAX_BYTES'), 32 * 1024 * 1024) + self.keycheck_candidate_line_max_bytes = int_setting(os.getenv('KEYCHECK_CANDIDATE_LINE_MAX_BYTES'), 8192) + self.gharchive_cache_dir = defaults['gharchive_cache_dir'] + self.gharchive_cache_max_items = int_setting(os.getenv('GHARCHIVE_CACHE_MAX_ITEMS'), 48) + self.gharchive_cache_max_bytes = int_setting(os.getenv('GHARCHIVE_CACHE_MAX_BYTES'), 8 * 1024 * 1024 * 1024) + self.gharchive_cache_min_free_bytes = int_setting(os.getenv('GHARCHIVE_CACHE_MIN_FREE_BYTES'), 5 * 1024 * 1024 * 1024) + self.gharchive_download_max_bytes = int_setting(os.getenv('GHARCHIVE_DOWNLOAD_MAX_BYTES'), 512 * 1024 * 1024) + self.gharchive_decompressed_max_bytes = int_setting(os.getenv('GHARCHIVE_DECOMPRESSED_MAX_BYTES'), 8 * 1024 * 1024 * 1024) + self.gharchive_max_events = int_setting(os.getenv('GHARCHIVE_MAX_EVENTS'), 5000000) + self.gharchive_max_line_bytes = int_setting(os.getenv('GHARCHIVE_MAX_LINE_BYTES'), 8 * 1024 * 1024) + self.gharchive_cache_lock_timeout_sec = int_setting(os.getenv('GHARCHIVE_CACHE_LOCK_TIMEOUT_SEC'), 600) + self.proxy_file = defaults['proxy_file'] + self.api_proxy_enabled = bool_setting(os.getenv("SCANNER_API_PROXY_ENABLED"), False) + self.api_proxy_file = os.getenv("SCANNER_API_PROXY_FILE", self.proxy_file) + self.api_proxy_timeout = int_setting(os.getenv("SCANNER_API_PROXY_TIMEOUT"), 5) + self.api_proxy_max_retries = int_setting(os.getenv("SCANNER_API_PROXY_MAX_RETRIES"), 100) + self.api_proxy_retry_delay = int_setting(os.getenv("SCANNER_API_PROXY_RETRY_DELAY"), 5) + self.download_proxy_enabled = bool_setting(os.getenv("SCANNER_DOWNLOAD_PROXY_ENABLED"), False) + self.download_proxy_file = os.getenv("SCANNER_DOWNLOAD_PROXY_FILE", "") + self.max_active_scans = int_setting(os.getenv("SCANNER_MAX_ACTIVE_SCANS"), 0) + self.opportunistic_scan_slots = int_setting( + os.getenv("SCANNER_OPPORTUNISTIC_SCAN_SLOTS"), 0, + ) + self.opportunistic_scan_sources = csv_items( + os.getenv("SCANNER_OPPORTUNISTIC_SCAN_SOURCES") + ) + self.opportunistic_scan_reserve_overhead_bytes = int_setting( + os.getenv("SCANNER_OPPORTUNISTIC_SCAN_RESERVE_OVERHEAD_BYTES"), + 1024 * 1024 * 1024, + ) + self.opportunistic_scan_min_available_after_reserve_bytes = int_setting( + os.getenv("SCANNER_OPPORTUNISTIC_SCAN_MIN_AVAILABLE_AFTER_RESERVE_BYTES"), + 4 * 1024 * 1024 * 1024, + ) + self.opportunistic_scan_min_commit_after_reserve_bytes = int_setting( + os.getenv("SCANNER_OPPORTUNISTIC_SCAN_MIN_COMMIT_AFTER_RESERVE_BYTES"), + 6 * 1024 * 1024 * 1024, + ) + self.scan_limiter_db = os.getenv("SCANNER_SCAN_LIMITER_DB", os.path.join(defaults['state_dir'], 'scan_limiter.db')) + self.dockerhub_tag_cache_path = os.getenv("DOCKERHUB_TAG_CACHE_PATH", os.path.join(defaults['state_dir'], 'dockerhub_tag_cache.sqlite')) + self.dockerhub_tag_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_TTL_SEC"), 21600) + self.dockerhub_tag_negative_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_NEGATIVE_CACHE_TTL_SEC"), 3600) + self.dockerhub_tag_rate_limit_cache_ttl_sec = int_setting(os.getenv("DOCKERHUB_TAG_RATE_LIMIT_CACHE_TTL_SEC"), 1800) + self.dockerhub_tag_cache_max_rows = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_ROWS"), 50000) + self.dockerhub_tag_cache_max_age_sec = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_AGE_SEC"), 7 * 86400) + self.dockerhub_tag_cache_max_bytes = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MAX_BYTES"), 256 * 1024 * 1024) + self.dockerhub_tag_cache_min_free_bytes = int_setting(os.getenv("DOCKERHUB_TAG_CACHE_MIN_FREE_BYTES"), 512 * 1024 * 1024) + self.scan_slot_wait_sec = float_setting(os.getenv("SCANNER_SCAN_SLOT_WAIT_SEC"), 0.5) + self.scan_slot_wait_log_sec = int_setting(os.getenv("SCANNER_SCAN_SLOT_WAIT_LOG_SEC"), 30) + self.scan_slot_stale_sec = int_setting(os.getenv("SCANNER_SCAN_SLOT_STALE_SEC"), 7200) + self.low_space_cleanup_max_items = int_setting(os.getenv("SCANNER_LOW_SPACE_CLEANUP_MAX_ITEMS"), 50) + self.min_free_gb = float(os.getenv("TRUFFLEHOG_MIN_FREE_GB", "5")) + self.jsonl_rotation_enabled = bool_setting(os.getenv("SCANNER_JSONL_ROTATION_ENABLED"), False) + self.found_secrets_max_mb = int_setting(os.getenv("SCANNER_FOUND_SECRETS_MAX_MB"), 512) + self.scan_results_max_mb = int_setting(os.getenv("SCANNER_SCAN_RESULTS_MAX_MB"), 1024) + self.scan_errors_max_mb = int_setting(os.getenv("SCANNER_SCAN_ERRORS_MAX_MB"), 64) + self.scan_errors_keep = int_setting(os.getenv("SCANNER_SCAN_ERRORS_KEEP"), 5) + self.jsonl_lock_stale_sec = int_setting(os.getenv("SCANNER_JSONL_LOCK_STALE_SEC"), 300) + self.jsonl_max_segments = int_setting(os.getenv("SCANNER_JSONL_MAX_SEGMENTS"), 16) + self.jsonl_ledger_max_rows = int_setting(os.getenv("SCANNER_JSONL_LEDGER_MAX_ROWS"), 1000000) + self.jsonl_ledger_max_bytes = int_setting(os.getenv("SCANNER_JSONL_LEDGER_MAX_BYTES"), 512 * 1024 * 1024) + self.jsonl_legacy_index_max_bytes = int_setting(os.getenv("SCANNER_JSONL_LEGACY_INDEX_MAX_BYTES"), 16 * 1024 * 1024) + self.jsonl_tail_scan_max_bytes = int_setting(os.getenv("SCANNER_JSONL_TAIL_SCAN_MAX_BYTES"), 8 * 1024 * 1024) + self.jsonl_torn_quarantine_max_bytes = int_setting(os.getenv("SCANNER_JSONL_TORN_QUARANTINE_MAX_BYTES"), 64 * 1024) + +scan_config = ScanConfig() +pending_temp_dirs = set() +pending_temp_lock = threading.Lock() +TEMP_OWNER_FILE = '.scanner-owner.json' +TEMP_OWNER_SCHEMA = 2 +PENDING_TEMP_SCHEMA = 1 +APPROVED_TEMP_PREFIXES = ('trufflehog-', 'trufflehog-run-', 'trufflehog-probe-', 'hg-run-', 'docker-config-', 'tmp-', 'worker-assignment-') + + +class ApiRequestError(Exception): + def __init__( + self, message, *, response=None, operation='provider-api', + capture_body=True, + ): + super().__init__(message) + self.operation = str(operation) + self.status_code = None + self.content_type = None + self.request_id = None + self.body = None + self.body_original_size = None + self.body_stored_size = None + self.body_sha256 = None + self.body_capture_truncated = False + self.headers = None + if response is not None and type(getattr(response, 'status_code', None)) is int: + self.status_code = int(response.status_code) + headers = getattr(response, 'headers', {}) or {} + self.headers = { + str(key): str(value) for key, value in headers.items() + } + self.content_type = str(headers.get('Content-Type') or '') or None + self.request_id = str( + headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or '' + ) or None + body_material = getattr(response, '_truf_diagnostic_body_material', None) + if body_material is not None: + self.body = diagnostic_material_bytes(body_material) + elif capture_body and ( + not getattr(response, 'raw', None) + or getattr(response, '_content_consumed', False) + ): + body = getattr(response, 'content', b'') + body = body if isinstance(body, bytes) else str(body).encode('utf-8') + body_material = make_body_material(body) + self.body = diagnostic_material_bytes(body_material) + if body_material is not None: + self.body_original_size = body_material.original_size + self.body_stored_size = body_material.stored_size + self.body_sha256 = body_material.sha256 + self.body_capture_truncated = body_material.truncated + + +class GitLabDiscoveryTransportError(ApiRequestError): + pass + + +class DockerHubDiscoveryTransportError(ApiRequestError): + def __init__( + self, message, *, category='page_unavailable', retry_at=None, + remote_attempted=True, retryable=True, + ): + super().__init__(message) + self.category = str(category or 'page_unavailable') + self.retry_at = retry_at + self.remote_attempted = bool(remote_attempted) + self.retryable = bool(retryable) + + +API_RETRY_STATUSES = {408, 500, 502, 503, 504} +API_RETRY_EXCEPTIONS = ( + requests.exceptions.ProxyError, + requests.exceptions.ConnectionError, + requests.exceptions.ConnectTimeout, + requests.exceptions.ReadTimeout, + requests.exceptions.Timeout, + requests.exceptions.SSLError, + requests.exceptions.ChunkedEncodingError, +) +_api_proxy_lock = threading.Lock() +_api_proxy_cache_path = None +_api_proxy_cache_mtime = None +_api_proxy_cache_entries = [] +_api_proxy_cache_index = 0 + + +def _redacted_proxy_url(proxy_url): + try: + parsed = urlsplit(proxy_url) + if '@' not in parsed.netloc: + return proxy_url + host = parsed.hostname or '' + port = f':{parsed.port}' if parsed.port else '' + return urlunsplit((parsed.scheme, f'***:***@{host}{port}', parsed.path, parsed.query, parsed.fragment)) + except Exception: + return '' + + +def parse_proxy_line(line): + line = str(line or '').strip() + if not line or line.startswith('#'): + return None + if '://' in line: + proxy_url = line + else: + parts = line.split(':', 3) + if len(parts) == 2: + host, port = parts + proxy_url = f'http://{host}:{port}' + elif len(parts) == 4: + host, port, username, password = parts + credentials = f'{quote(username, safe="")}:{quote(password, safe="")}' + proxy_url = f'http://{credentials}@{host}:{port}' + else: + raise ValueError('expected host:port or host:port:username:password') + return {'http': proxy_url, 'https': proxy_url} + + +def load_proxy_entries(proxy_file): + entries = [] + if not proxy_file or not os.path.exists(proxy_file): + return entries + with open(proxy_file, 'r', encoding='utf-8') as f: + for line_number, line in enumerate(f, 1): + try: + proxy = parse_proxy_line(line) + except ValueError as e: + logger.warning(f'Ignoring bad proxy line {proxy_file}:{line_number}: {e}') + continue + if proxy: + entries.append(proxy) + return entries + + +def next_api_proxy(): + global _api_proxy_cache_path, _api_proxy_cache_mtime, _api_proxy_cache_entries, _api_proxy_cache_index + if not scan_config.api_proxy_enabled: + return None + + proxy_file = scan_config.api_proxy_file or scan_config.proxy_file + try: + mtime = os.path.getmtime(proxy_file) if proxy_file else None + except OSError: + mtime = None + + with _api_proxy_lock: + if proxy_file != _api_proxy_cache_path or mtime != _api_proxy_cache_mtime: + _api_proxy_cache_path = proxy_file + _api_proxy_cache_mtime = mtime + _api_proxy_cache_entries = load_proxy_entries(proxy_file) + _api_proxy_cache_index = 0 + if _api_proxy_cache_entries: + logger.info(f'Loaded {len(_api_proxy_cache_entries)} API proxy entry(ies) from {proxy_file}') + if not _api_proxy_cache_entries: + raise ApiRequestError(f'API proxy is enabled but no valid proxies are loaded from {proxy_file}') + proxy = _api_proxy_cache_entries[_api_proxy_cache_index % len(_api_proxy_cache_entries)] + _api_proxy_cache_index += 1 + return proxy + + +def _short_url(url): + try: + parsed = urlsplit(url) + return urlunsplit((parsed.scheme, parsed.netloc, parsed.path, '', '')) + except Exception: + return str(url) + + +def _log_api_retry(method, url, attempt, attempts, error, proxy): + if attempt != 1 and attempt % 10 != 0 and attempt != attempts: + return + proxy_url = None + if proxy: + proxy_url = proxy.get('https') or proxy.get('http') + proxy_part = f' via {_redacted_proxy_url(proxy_url)}' if proxy_url else '' + logger.warning(f'API {method} {_short_url(url)} failed ({attempt}/{attempts}){proxy_part}: {str(error)[:300]}') + + +def _direct_request(method, url, **kwargs): + # None removes merged proxy routes; no_proxy also blocks environment rebuilds + # on redirects. Keep Requests' certificate and streamed-response behavior. + kwargs['proxies'] = {'http': None, 'https': None, 'all': None, 'no_proxy': '*'} + return requests.request(method, url, **kwargs) + + +def api_request( + method, url, *, timeout=None, max_retries=None, retry_delay=None, + retry_statuses=None, deadline=None, use_proxy=None, **kwargs, +): + # False opts out; other values retain the configured discovery policy. + use_proxy = use_proxy is not False and scan_config.api_proxy_enabled + attempts = max(1, int(max_retries if max_retries is not None else (scan_config.api_proxy_max_retries if use_proxy else 1))) + delay = max(0, int(retry_delay if retry_delay is not None else scan_config.api_proxy_retry_delay)) + retry_statuses = set(API_RETRY_STATUSES if retry_statuses is None else retry_statuses) + request_timeout = timeout + if use_proxy and scan_config.api_proxy_timeout is not None: + if isinstance(timeout, (tuple, list)): + connect_timeout, read_timeout = timeout + else: + connect_timeout = read_timeout = timeout + proxy_timeout = float(scan_config.api_proxy_timeout) + # Proxy connection limits must not replace the caller's read budget. + request_timeout = ( + proxy_timeout if connect_timeout is None else min(proxy_timeout, float(connect_timeout)), + proxy_timeout if timeout is None else read_timeout, + ) + + last_error = None + for attempt in range(1, attempts + 1): + _raise_if_scan_slot_fatal() + remaining = None if deadline is None else float(deadline) - time.monotonic() + if remaining is not None and remaining <= 0: + raise ApiRequestError(f'API request deadline expired before attempt {attempt}: {method} {_short_url(url)}') + proxy = next_api_proxy() if use_proxy else None + request_kwargs = dict(kwargs) + if proxy: + request_kwargs['proxies'] = proxy + effective_timeout = request_timeout + if remaining is not None and effective_timeout is None: + effective_timeout = max(0.001, remaining) + elif remaining is not None and isinstance(effective_timeout, (int, float)): + effective_timeout = max(0.001, min(float(effective_timeout), remaining)) + elif remaining is not None and isinstance(effective_timeout, (tuple, list)): + effective_timeout = tuple( + max(0.001, remaining if value is None else min(float(value), remaining)) + for value in effective_timeout + ) + try: + request = requests.request if use_proxy else _direct_request + response = request(method, url, timeout=effective_timeout, **request_kwargs) + if _scan_slot_fatal_event.is_set(): + response.close() + _raise_if_scan_slot_fatal() + if deadline is not None and time.monotonic() >= float(deadline): + response.close() + raise ApiRequestError(f'API request deadline expired after response: {method} {_short_url(url)}') + if response.status_code in retry_statuses: + detail = '' + if not request_kwargs.get('stream'): + detail = response.text[:300] if response.text else '' + last_error = f'HTTP {response.status_code}' + (f': {detail}' if detail else '') + if attempt >= attempts: + failure = ApiRequestError( + f'API request failed after {attempts} attempt(s): ' + f'{method} {_short_url(url)}: {last_error}', + response=response, + operation='provider-api-request', + capture_body=not bool(request_kwargs.get('stream')), + ) + response.close() + raise failure + _log_api_retry(method, url, attempt, attempts, last_error, proxy) + response.close() + if deadline is not None and time.monotonic() + delay >= float(deadline): + raise ApiRequestError(f'API request deadline expired during retry: {method} {_short_url(url)}') + _wait_or_raise_scan_slot_fatal(delay) + continue + return response + except API_RETRY_EXCEPTIONS as e: + try: + url_has_query = bool(urlsplit(str(url)).query) + except ValueError: + url_has_query = True + safe_error = ( + type(e).__name__ + if request_kwargs.get('stream') or url_has_query + else str(e)[:300] + ) + last_error = safe_error + _log_api_retry(method, url, attempt, attempts, safe_error, proxy) + if attempt >= attempts: + raise ApiRequestError( + f'API request failed after {attempts} attempt(s): ' + f'{method} {_short_url(url)}: {safe_error}' + ) from e + if deadline is not None and time.monotonic() + delay >= float(deadline): + raise ApiRequestError(f'API request deadline expired during retry: {method} {_short_url(url)}') from e + _wait_or_raise_scan_slot_fatal(delay) + raise ApiRequestError(f'API request failed after {attempts} attempt(s): {method} {_short_url(url)}: {last_error}') + + +SCAN_SLOT_SCHEMA = """ +CREATE TABLE IF NOT EXISTS scan_slots ( + slot_id TEXT PRIMARY KEY, + owner_pid INTEGER NOT NULL, + owner_thread INTEGER NOT NULL, + owner_source TEXT, + owner_creation_time TEXT, + owner_executable TEXT, + child_pid INTEGER, + child_creation_time TEXT, + child_executable TEXT, + slot_kind TEXT NOT NULL DEFAULT 'base' CHECK(slot_kind IN ('base', 'bonus')), + command TEXT, + acquired_at REAL NOT NULL, + updated_at REAL NOT NULL +); +CREATE TABLE IF NOT EXISTS scan_waiters ( + waiter_id TEXT PRIMARY KEY, + owner_pid INTEGER NOT NULL, + owner_thread INTEGER NOT NULL, + owner_source TEXT NOT NULL, + owner_creation_time TEXT NOT NULL, + owner_executable TEXT NOT NULL, + enqueued_at REAL NOT NULL +); +CREATE INDEX IF NOT EXISTS idx_scan_waiters_fair +ON scan_waiters(owner_source, enqueued_at, waiter_id); +CREATE TABLE IF NOT EXISTS scan_source_fairness ( + owner_source TEXT PRIMARY KEY, + last_granted_at REAL NOT NULL +); +""" +_scan_limiter_init_lock = threading.Lock() +_scan_limiter_initialized_paths = set() +_scan_slot_scope_local = threading.local() +_scan_slot_fatal_event = threading.Event() +_scan_slot_fatal_lock = threading.Lock() +_scan_slot_fatal_detail = None + + +class _ScanSlotScope: + def __init__(self, lease): + self.lease = lease + + +class ScanSlotFatalError(RuntimeError): + pass + + +def _set_scan_slot_fatal(detail): + global _scan_slot_fatal_detail + bounded = str(detail or 'scan-slot durability failure')[:1000] + with _scan_slot_fatal_lock: + if not _scan_slot_fatal_event.is_set(): + _scan_slot_fatal_detail = bounded + _scan_slot_fatal_event.set() + + +def _raise_if_scan_slot_fatal(): + if not _scan_slot_fatal_event.is_set(): + return + with _scan_slot_fatal_lock: + detail = _scan_slot_fatal_detail + raise ScanSlotFatalError(detail or 'FATAL: scan-slot limiter is in an indeterminate state') + + +def _wait_or_raise_scan_slot_fatal(delay): + remaining = max(0.0, float(delay or 0)) + while remaining > 0: + _raise_if_scan_slot_fatal() + interval = min(0.2, remaining) + time.sleep(interval) + remaining -= interval + _raise_if_scan_slot_fatal() + + +class ScanSlotLease: + DB_RETRY_ATTEMPTS = 5 + RELEASE_PENDING_DB_ATTEMPTS = 2 + DB_RETRY_DELAY_SEC = 0.1 + MIN_HEARTBEAT_INTERVAL_SEC = 5.0 + RELEASE_PENDING_INTERVAL_SEC = 1.0 + HEARTBEAT_JOIN_TIMEOUT_SEC = 2.0 + + def __init__(self, slot_id, db_path, owner_pid=None, owner_thread=None): + self.slot_id = slot_id + self.db_path = db_path + self.owner_pid = int(os.getpid() if owner_pid is None else owner_pid) + self.owner_thread = int(threading.get_ident() if owner_thread is None else owner_thread) + self.child_pid = None + self.heartbeat_stop = threading.Event() + self.heartbeat_thread = None + self._heartbeat_wake = threading.Event() + self._state_lock = threading.Lock() + self._db_lock = threading.Lock() + self._release_call_lock = threading.Lock() + self._released = False + self._releasable = True + self._release_requested = False + self._release_pending = False + self._heartbeat_interval = 30.0 + self._release_pending_interval = self.RELEASE_PENDING_INTERVAL_SEC + self._last_release_error_log_at = 0.0 + + @property + def released(self): + with self._state_lock: + return self._released + + @property + def releasable(self): + with self._state_lock: + return self._releasable + + @property + def release_pending(self): + with self._state_lock: + return self._release_pending + + def _connect(self): + return sqlite3.connect(self.db_path, timeout=30) + + def _close_connection(self, conn): + try: + conn.close() + except Exception as exc: + logger.warning('Unable to close scan-slot DB connection for %s: %s', self.slot_id, exc) + + def _execute_update_once(self, sql, params): + conn = None + try: + conn = self._connect() + conn.execute('PRAGMA busy_timeout=30000') + cursor = conn.execute(sql, params) + conn.commit() + return cursor.rowcount != 0 + finally: + if conn is not None: + self._close_connection(conn) + + def _delete_slot_once(self): + conn = None + try: + conn = self._connect() + conn.execute('PRAGMA busy_timeout=30000') + cursor = conn.execute( + '''DELETE FROM scan_slots + WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''', + (self.slot_id, self.owner_pid, self.owner_thread), + ) + if cursor.rowcount != 1: + existing = conn.execute( + 'SELECT owner_pid, owner_thread FROM scan_slots WHERE slot_id = ?', + (self.slot_id,), + ).fetchone() + if existing is not None: + raise RuntimeError( + f'scan slot identity changed from pid/thread ' + f'{self.owner_pid}/{self.owner_thread} to {existing[0]}/{existing[1]}' + ) + conn.commit() + return True + finally: + if conn is not None: + self._close_connection(conn) + + def _retry_db_operation(self, operation, attempts): + attempts = max(1, int(attempts)) + last_error = None + for attempt in range(1, attempts + 1): + try: + with self._db_lock: + return bool(operation()), None + except (sqlite3.Error, OSError) as exc: + last_error = exc + except Exception as exc: + last_error = exc + break + if attempt < attempts: + time.sleep(max(0.0, float(self.DB_RETRY_DELAY_SEC)) * attempt) + return False, last_error + + def update(self, sql, params, attempts=None, log_failure=True): + success, last_error = self._retry_db_operation( + lambda: self._execute_update_once(sql, params), + self.DB_RETRY_ATTEMPTS if attempts is None else attempts, + ) + if last_error is not None and log_failure: + logger.warning('Unable to update scan slot %s: %s', self.slot_id, last_error) + return success + + def set_child_pid(self, child_pid): + if not child_pid: + return False + child_pid = int(child_pid) + with self._release_call_lock: + with self._state_lock: + if ( + self._released or not self._releasable or self._release_requested + or not self.slot_id or not self.db_path + ): + return False + identity = capture_process_identity(child_pid) + updated = self.update( + '''UPDATE scan_slots SET child_pid = ?, child_creation_time = ?, child_executable = ?, updated_at = ? + WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''', + ( + child_pid, + identity.get('creation_time') if identity else None, + identity.get('executable') if identity else None, + time.time(), + self.slot_id, + self.owner_pid, + self.owner_thread, + ), + ) + if updated: + with self._state_lock: + self.child_pid = child_pid + return updated + + def mark_non_releasable(self): + with self._release_call_lock: + with self._state_lock: + if self._released: + return + self._releasable = False + self._release_pending = False + self._heartbeat_wake.set() + + def _heartbeat_update(self, attempts=None, log_failure=True): + return self.update( + '''UPDATE scan_slots SET updated_at = ? + WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''', + (time.time(), self.slot_id, self.owner_pid, self.owner_thread), + attempts=attempts, + log_failure=log_failure, + ) + + def start_heartbeat(self): + interval = max( + float(self.MIN_HEARTBEAT_INTERVAL_SEC), + float_setting(getattr(scan_config, 'scan_slot_heartbeat_sec', 30), 30), + ) + release_interval = max(0.01, min(interval, float(self.RELEASE_PENDING_INTERVAL_SEC))) + with self._state_lock: + if self._released: + return False + if self.heartbeat_thread is not None and self.heartbeat_thread.is_alive(): + return True + self._heartbeat_interval = interval + self._release_pending_interval = release_interval + thread = threading.Thread( + target=self._heartbeat_loop, + name=f'scan-slot-heartbeat-{self.slot_id[:8]}', + daemon=True, + ) + self.heartbeat_thread = thread + try: + thread.start() + except Exception: + self.heartbeat_thread = None + raise + return True + + def transfer_to_current_thread(self): + new_thread = int(threading.get_ident()) + with self._release_call_lock: + with self._state_lock: + if self._released or self._release_requested or not self._releasable: + raise RuntimeError(f'scan slot {self.slot_id} cannot be transferred after release started') + if self.heartbeat_thread is not None: + raise RuntimeError(f'scan slot {self.slot_id} cannot be transferred after heartbeat start') + old_thread = self.owner_thread + if old_thread == new_thread: + return True + updated = self.update( + '''UPDATE scan_slots SET owner_thread = ?, updated_at = ? + WHERE slot_id = ? AND owner_pid = ? AND owner_thread = ?''', + (new_thread, time.time(), self.slot_id, self.owner_pid, old_thread), + ) + if not updated: + raise RuntimeError(f'scan slot {self.slot_id} ownership transfer was not confirmed') + with self._state_lock: + self.owner_thread = new_thread + return True + + def _heartbeat_loop(self): + while True: + with self._state_lock: + if self._released: + return + pending = self._release_pending and self._releasable + interval = self._release_pending_interval if pending else self._heartbeat_interval + + self._heartbeat_wake.wait(interval) + self._heartbeat_wake.clear() + + with self._state_lock: + if self._released or self.heartbeat_stop.is_set(): + return + pending = self._release_pending and self._releasable + + if not pending: + self._heartbeat_update() + continue + + with self._release_call_lock: + with self._state_lock: + pending = self._release_pending and self._releasable and not self._released + if not pending: + continue + success, last_error = self._retry_db_operation( + self._delete_slot_once, + self.RELEASE_PENDING_DB_ATTEMPTS, + ) + if success: + completed = self._complete_release() + if completed: + logger.info('Released scan slot %s after background retry', self.slot_id) + return + + self._log_pending_release_failure(last_error) + # Keep the exact owner row fresh when DELETE itself is temporarily unavailable. + self._heartbeat_update(attempts=1, log_failure=False) + + def _log_pending_release_failure(self, error): + now = time.monotonic() + with self._state_lock: + if now - self._last_release_error_log_at < 30: + return + self._last_release_error_log_at = now + logger.error('Scan slot %s release remains pending and fail-closed: %s', self.slot_id, error) + + def _complete_release(self): + with self._state_lock: + if self._released: + return False + self._released = True + self._release_pending = False + self.heartbeat_stop.set() + self._heartbeat_wake.set() + return True + + def _join_heartbeat(self): + with self._state_lock: + thread = self.heartbeat_thread + if thread is None or thread is threading.current_thread() or not thread.is_alive(): + return + thread.join(timeout=max(0.0, float(self.HEARTBEAT_JOIN_TIMEOUT_SEC))) + if thread.is_alive(): + logger.warning('Scan-slot heartbeat did not stop promptly for %s', self.slot_id) + + def release(self): + should_join = False + success = False + with self._release_call_lock: + with self._state_lock: + if self._released: + success = True + should_join = True + elif ( + not self._releasable or self._release_requested + or not self.slot_id or not self.db_path + ): + return + else: + self._release_requested = True + + if not success: + success, last_error = self._retry_db_operation( + self._delete_slot_once, + self.DB_RETRY_ATTEMPTS, + ) + if success: + self._complete_release() + should_join = True + else: + with self._state_lock: + pending = not self._released and self._releasable + if pending: + self._release_pending = True + self._last_release_error_log_at = time.monotonic() + if pending: + try: + self.start_heartbeat() + except Exception as exc: + logger.error('Unable to start pending-release heartbeat for scan slot %s: %s', self.slot_id, exc) + self._heartbeat_wake.set() + logger.error( + 'Unable to release scan slot %s after %d attempts; ' + 'lease remains live and background retries will continue: %s', + self.slot_id, self.DB_RETRY_ATTEMPTS, last_error, + ) + + if should_join: + self._join_heartbeat() + + +def process_exists(pid): + try: + pid = int(pid) + except (TypeError, ValueError): + return False + if pid <= 0: + return False + if pid == os.getpid(): + return True + if os.name == 'nt': + try: + import ctypes + process_query_limited_information = 0x1000 + handle = ctypes.windll.kernel32.OpenProcess(process_query_limited_information, False, pid) + if handle: + ctypes.windll.kernel32.CloseHandle(handle) + return True + return False + except Exception: + return True + try: + os.kill(pid, 0) + return True + except ProcessLookupError: + return False + except PermissionError: + return True + except Exception: + return True + + +def capture_process_identity(pid): + try: + process = open_process(int(pid)) + except Exception: + return None + try: + return { + 'creation_time': str(process.identity.creation_time), + 'executable': canonical_path(process.identity.executable), + } + finally: + process.close() + + +def exact_process_identity_live(pid, creation_time, executable): + if not pid: + return False + if not creation_time or not executable: + return None if process_exists(pid) else False + try: + process = open_process(int(pid)) + except Exception: + return None if process_exists(pid) else False + try: + return bool( + process.is_running() + and str(process.identity.creation_time) == str(creation_time) + and canonical_path(process.identity.executable) == canonical_path(executable) + ) + finally: + process.close() + + +def scan_limiter_enabled(): + return int_setting(getattr(scan_config, 'max_active_scans', 0), 0) > 0 + + +def scan_limiter_db_path(): + path = getattr(scan_config, 'scan_limiter_db', '') or '' + if not path: + defaults = default_project_paths() + path = os.path.join(defaults['state_dir'], 'scan_limiter.db') + return path + + +def ensure_scan_limiter_db(path): + parent = os.path.dirname(path) + if parent: + os.makedirs(parent, exist_ok=True) + with _scan_limiter_init_lock: + if path in _scan_limiter_initialized_paths: + return + conn = sqlite3.connect(path, timeout=30) + try: + conn.execute('PRAGMA busy_timeout=30000') + conn.execute('PRAGMA journal_mode=WAL') + conn.executescript(SCAN_SLOT_SCHEMA) + conn.execute('BEGIN IMMEDIATE') + existing = {row[1] for row in conn.execute('PRAGMA table_info(scan_slots)').fetchall()} + if 'child_pid' not in existing: + conn.execute('ALTER TABLE scan_slots ADD COLUMN child_pid INTEGER') + for name, declaration in ( + ('owner_creation_time', 'TEXT'), + ('owner_executable', 'TEXT'), + ('child_creation_time', 'TEXT'), + ('child_executable', 'TEXT'), + ): + if name not in existing: + conn.execute(f'ALTER TABLE scan_slots ADD COLUMN {name} {declaration}') + if 'slot_kind' not in existing: + conn.execute( + "ALTER TABLE scan_slots ADD COLUMN slot_kind TEXT NOT NULL DEFAULT 'base'" + ) + conn.execute( + "CREATE UNIQUE INDEX IF NOT EXISTS idx_scan_slots_single_bonus " + "ON scan_slots(slot_kind) WHERE slot_kind = 'bonus'" + ) + conn.commit() + finally: + conn.close() + _scan_limiter_initialized_paths.add(path) + + +def connect_scan_limiter_db(path): + ensure_scan_limiter_db(path) + conn = sqlite3.connect(path, timeout=30) + conn.execute('PRAGMA busy_timeout=30000') + return conn + + +def cleanup_stale_scan_slots(conn, now, stale_sec): + columns = {row[1] for row in conn.execute('PRAGMA table_info(scan_slots)').fetchall()} + child_expr = 'child_pid' if 'child_pid' in columns else 'NULL AS child_pid' + owner_creation_expr = 'owner_creation_time' if 'owner_creation_time' in columns else 'NULL AS owner_creation_time' + owner_executable_expr = 'owner_executable' if 'owner_executable' in columns else 'NULL AS owner_executable' + child_creation_expr = 'child_creation_time' if 'child_creation_time' in columns else 'NULL AS child_creation_time' + child_executable_expr = 'child_executable' if 'child_executable' in columns else 'NULL AS child_executable' + rows = conn.execute( + f'''SELECT slot_id, owner_pid, {owner_creation_expr}, {owner_executable_expr}, + {child_expr}, {child_creation_expr}, {child_executable_expr}, acquired_at, updated_at + FROM scan_slots''' + ).fetchall() + for slot_id, owner_pid, owner_creation, owner_executable, child_pid, child_creation, child_executable, acquired_at, updated_at in rows: + heartbeat_age = now - float(updated_at or acquired_at or 0) + owner_live = exact_process_identity_live(owner_pid, owner_creation, owner_executable) + child_live = exact_process_identity_live(child_pid, child_creation, child_executable) + if owner_live is False and child_live is False: + conn.execute('DELETE FROM scan_slots WHERE slot_id = ?', (slot_id,)) + elif heartbeat_age > stale_sec: + logger.warning( + 'Stale scan-slot heartbeat remains capacity-blocking: slot=%s owner_live=%s child_live=%s age=%.0fs', + slot_id, owner_live, child_live, heartbeat_age, + ) + + if 'scan_waiters' in { + row[0] for row in conn.execute("SELECT name FROM sqlite_master WHERE type = 'table'").fetchall() + }: + waiters = conn.execute( + '''SELECT waiter_id, owner_pid, owner_creation_time, owner_executable + FROM scan_waiters''' + ).fetchall() + for waiter_id, owner_pid, owner_creation, owner_executable in waiters: + if exact_process_identity_live(owner_pid, owner_creation, owner_executable) is False: + conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,)) + + +def redact_scan_command_text(cmd): + text = ' '.join(str(part) for part in cmd) + text = re.sub(r'(https?://)[^\s/@:]+:[^\s/@]+@', r'\1***:***@', text) + text = re.sub(r'github_pat_[A-Za-z0-9_]+', 'github_pat_***', text) + text = re.sub(r'gh[pousr]_[A-Za-z0-9_]+', 'ghp_***', text) + text = re.sub(r'hf_[A-Za-z0-9]+', 'hf_***', text) + return text[:1000] + + +def windows_scan_capacity_snapshot(): + if os.name != 'nt': + raise OSError('opportunistic scan capacity is supported only on Windows') + + from ctypes import wintypes + + class PerformanceInformation(ctypes.Structure): + _fields_ = [ + ('cb', wintypes.DWORD), + ('CommitTotal', ctypes.c_size_t), + ('CommitLimit', ctypes.c_size_t), + ('CommitPeak', ctypes.c_size_t), + ('PhysicalTotal', ctypes.c_size_t), + ('PhysicalAvailable', ctypes.c_size_t), + ('SystemCache', ctypes.c_size_t), + ('KernelTotal', ctypes.c_size_t), + ('KernelPaged', ctypes.c_size_t), + ('KernelNonpaged', ctypes.c_size_t), + ('PageSize', ctypes.c_size_t), + ('HandleCount', wintypes.DWORD), + ('ProcessCount', wintypes.DWORD), + ('ThreadCount', wintypes.DWORD), + ] + + get_performance_info = ctypes.WinDLL('psapi', use_last_error=True).GetPerformanceInfo + get_performance_info.argtypes = [ctypes.POINTER(PerformanceInformation), wintypes.DWORD] + get_performance_info.restype = wintypes.BOOL + info = PerformanceInformation() + info.cb = ctypes.sizeof(info) + if not get_performance_info(ctypes.byref(info), info.cb): + raise ctypes.WinError(ctypes.get_last_error()) + page_size = int(info.PageSize) + if page_size <= 0 or int(info.CommitLimit) < int(info.CommitTotal): + raise OSError('Windows returned invalid scan-capacity counters') + return { + 'available_physical_bytes': int(info.PhysicalAvailable) * page_size, + 'commit_headroom_bytes': (int(info.CommitLimit) - int(info.CommitTotal)) * page_size, + } + + +def opportunistic_scan_slot_allowed(source): + if max(0, min(1, int_setting(getattr(scan_config, 'opportunistic_scan_slots', 0), 0))) <= 0: + return False + eligible = { + item.lower() for item in csv_items( + getattr(scan_config, 'opportunistic_scan_sources', []) + ) + } + if str(source or '').lower() not in eligible: + return False + try: + job_limit = int(getattr(scan_config, 'trufflehog_job_memory_limit_bytes', 0)) + overhead = max(0, int(getattr( + scan_config, 'opportunistic_scan_reserve_overhead_bytes', 0, + ))) + reserve = job_limit + overhead + if job_limit <= 0 or reserve <= 0: + return False + capacity = windows_scan_capacity_snapshot() + available_after = int(capacity['available_physical_bytes']) - reserve + commit_after = int(capacity['commit_headroom_bytes']) - reserve + minimum_available = max(0, int(getattr( + scan_config, 'opportunistic_scan_min_available_after_reserve_bytes', 0, + ))) + minimum_commit = max(0, int(getattr( + scan_config, 'opportunistic_scan_min_commit_after_reserve_bytes', 0, + ))) + return available_after >= minimum_available and commit_after >= minimum_commit + except (OSError, TypeError, ValueError, OverflowError, KeyError): + return False + + +def acquire_scan_slot(cmd, timeout_sec=None, wait=True, start_heartbeat=True): + _raise_if_scan_slot_fatal() + max_active = int_setting(getattr(scan_config, 'max_active_scans', 0), 0) + if max_active <= 0: + return None + + db_path = scan_limiter_db_path() + wait_sec = max(0.1, float_setting(getattr(scan_config, 'scan_slot_wait_sec', 0.5), 0.5)) + wait_log_sec = max(1, int_setting(getattr(scan_config, 'scan_slot_wait_log_sec', 30), 30)) + stale_sec = max( + int_setting(getattr(scan_config, 'scan_slot_stale_sec', 7200), 7200), + int(timeout_sec or 0) + 300, + ) + source = os.getenv('SCANNER_SOURCE') or (cmd[1] if len(cmd) > 1 else 'unknown') + command_text = redact_scan_command_text(cmd) + owner_pid = os.getpid() + owner_thread = threading.get_ident() + slot_id = f'{owner_pid}-{owner_thread}-{uuid.uuid4().hex}' + waiter_id = f'wait-{owner_pid}-{owner_thread}-{uuid.uuid4().hex}' + owner_identity = current_process_identity() + started_waiting = time.monotonic() + enqueued_at = time.time() + last_log_at = 0.0 + waiter_registered = False + + while True: + _raise_if_scan_slot_fatal() + now = time.time() + conn = None + slot_committed = False + try: + conn = connect_scan_limiter_db(db_path) + conn.execute('BEGIN IMMEDIATE') + _raise_if_scan_slot_fatal() + cleanup_stale_scan_slots(conn, now, stale_sec) + conn.execute( + '''INSERT OR IGNORE INTO scan_waiters( + waiter_id, owner_pid, owner_thread, owner_source, + owner_creation_time, owner_executable, enqueued_at + ) VALUES (?, ?, ?, ?, ?, ?, ?)''', + ( + waiter_id, owner_pid, owner_thread, str(source), + owner_identity.creation_time, canonical_path(owner_identity.executable), + enqueued_at, + ), + ) + waiter_registered = True + base_active, bonus_active = conn.execute( + "SELECT " + "SUM(CASE WHEN slot_kind = 'base' THEN 1 ELSE 0 END), " + "SUM(CASE WHEN slot_kind = 'bonus' THEN 1 ELSE 0 END) " + "FROM scan_slots" + ).fetchone() + base_active = int(base_active or 0) + bonus_active = int(bonus_active or 0) + active = base_active + bonus_active + next_waiter = conn.execute( + '''SELECT w.waiter_id + FROM scan_waiters w + LEFT JOIN scan_source_fairness f ON f.owner_source = w.owner_source + ORDER BY COALESCE(f.last_granted_at, 0), w.enqueued_at, w.waiter_id + LIMIT 1''' + ).fetchone() + slot_kind = None + if next_waiter and next_waiter[0] == waiter_id: + if base_active < max_active: + slot_kind = 'base' + elif bonus_active < max(0, min(1, int_setting( + getattr(scan_config, 'opportunistic_scan_slots', 0), 0, + ))) and opportunistic_scan_slot_allowed(source): + slot_kind = 'bonus' + if slot_kind is not None: + conn.execute( + '''INSERT INTO scan_slots( + slot_id, owner_pid, owner_thread, owner_source, owner_creation_time, + owner_executable, slot_kind, command, acquired_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + slot_id, owner_pid, owner_thread, str(source), + owner_identity.creation_time, canonical_path(owner_identity.executable), + slot_kind, command_text, now, now, + ), + ) + conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,)) + conn.execute( + '''INSERT INTO scan_source_fairness(owner_source, last_granted_at) + VALUES (?, ?) + ON CONFLICT(owner_source) DO UPDATE SET last_granted_at = excluded.last_granted_at''', + (str(source), now), + ) + conn.commit() + slot_committed = True + waiter_registered = False + waited = time.monotonic() - started_waiting + if waited >= wait_log_sec: + hard_limit = max_active + max(0, min(1, int_setting( + getattr(scan_config, 'opportunistic_scan_slots', 0), 0, + ))) + logger.info( + f'Acquired {slot_kind} scan slot after waiting {waited:.0f}s ' + f'({active + 1}/{hard_limit})' + ) + lease = ScanSlotLease(slot_id, db_path, owner_pid=owner_pid, owner_thread=owner_thread) + if start_heartbeat: + try: + if not lease.start_heartbeat(): + raise RuntimeError('scan-slot heartbeat did not start') + except BaseException as start_error: + logger.error( + 'Scan-slot heartbeat failed to start for %s; synchronously removing the exact owner row: %s', + slot_id, start_error, + ) + lease.release() + if not lease.released: + fatal_detail = ( + 'FATAL: scan-slot heartbeat startup failed and exact-owner rollback ' + f'could not be confirmed for slot {slot_id}' + ) + _set_scan_slot_fatal(fatal_detail) + logger.critical( + 'FATAL scan-slot acquisition rollback is unconfirmed for %s; capacity remains fail-closed', + slot_id, + ) + raise ScanSlotFatalError(fatal_detail) from start_error + raise + return lease + if not wait: + conn.execute('DELETE FROM scan_waiters WHERE waiter_id = ?', (waiter_id,)) + conn.commit() + waiter_registered = False + return None + conn.commit() + if time.monotonic() - last_log_at >= wait_log_sec: + logger.info(f'Waiting for scan slot ({active}/{max_active} active)') + last_log_at = time.monotonic() + except sqlite3.OperationalError as e: + if slot_committed: + raise + if not wait: + return None + if time.monotonic() - last_log_at >= wait_log_sec: + logger.warning(f'Waiting for scan limiter DB lock: {str(e)}') + last_log_at = time.monotonic() + finally: + if conn is not None: + conn.close() + if _scan_slot_fatal_event.wait(wait_sec): + _raise_if_scan_slot_fatal() + + +def acquire_scan_slot_leases(cmd, count, timeout_sec=None): + count = max(0, int(count or 0)) + if count <= 0 or not scan_limiter_enabled(): + return [] + leases = [] + try: + first = acquire_scan_slot(cmd, timeout_sec, wait=True, start_heartbeat=False) + if first is not None: + leases.append(first) + while len(leases) < count: + lease = acquire_scan_slot(cmd, timeout_sec, wait=False, start_heartbeat=False) + if lease is None: + break + leases.append(lease) + return leases + except BaseException: + for lease in leases: + lease.release() + raise + + +@contextmanager +def scan_slot_scope(cmd, timeout_sec=None, lease=None): + """Own one physical lease through the target's durable bundle handoff.""" + if getattr(_scan_slot_scope_local, 'scope', None) is not None: + raise RuntimeError('scan slot scopes cannot be nested on one worker thread') + if lease is None: + lease = acquire_scan_slot(cmd, timeout_sec) + else: + try: + lease.transfer_to_current_thread() + if not lease.start_heartbeat(): + raise RuntimeError('transferred scan-slot heartbeat did not start') + except BaseException: + lease.release() + raise + scope = _ScanSlotScope(lease) + _scan_slot_scope_local.scope = scope + try: + yield lease + finally: + try: + if lease and lease.releasable: + lease.release() + finally: + if getattr(_scan_slot_scope_local, 'scope', None) is scope: + del _scan_slot_scope_local.scope + + +def scoped_scan_slot_lease(): + scope = getattr(_scan_slot_scope_local, 'scope', None) + return (scope is not None, scope.lease if scope is not None else None) + +def get_pending_temp_file(): + work_dir = get_work_dir() + if not work_dir: + return None + return os.path.join(work_dir, 'pending_cleanup.json') + + +def get_pending_temp_lock_file(): + path = get_pending_temp_file() + return path + '.lock' if path else None + + +def _load_persisted_pending_temp_dirs_unlocked(path): + if not path or not os.path.exists(path): + return set() + try: + value = read_private_json(path) + except OSError: + return set() + if value.get('schema') != PENDING_TEMP_SCHEMA or not isinstance(value.get('paths'), list): + return set() + return {str(item) for item in value['paths'] if isinstance(item, str) and item.strip()} + + +def _persist_pending_temp_dirs_unlocked(path, paths): + normalized = sorted({str(item) for item in paths if str(item).strip()}) + if normalized: + atomic_write_private_json(path, {'schema': PENDING_TEMP_SCHEMA, 'paths': normalized}) + elif os.path.exists(path): + if not private_file_ready(path): + raise OSError(f'refusing to remove non-private pending cleanup list: {path}') + durable_unlink(path) + +def load_persisted_pending_temp_dirs(): + path = get_pending_temp_file() + lock_path = get_pending_temp_lock_file() + if not path or not lock_path: + return set() + with PrivateFileLock(lock_path): + return _load_persisted_pending_temp_dirs_unlocked(path) + +def persist_pending_temp_dirs(paths): + path = get_pending_temp_file() + lock_path = get_pending_temp_lock_file() + if not path or not lock_path: + return + try: + with PrivateFileLock(lock_path): + _persist_pending_temp_dirs_unlocked(path, paths) + except OSError: + pass + +def redact_secrets(text, secrets): + if not text: + return text + + redacted = text + for secret in secrets: + if secret: + redacted = redacted.replace(secret, '***REDACTED***') + redacted = redacted.replace(quote(secret, safe=''), '***REDACTED***') + return redacted + +def build_authenticated_git_url(repo_url, provider=None, token=None): + if not token: + return repo_url, [] + + parsed = urlsplit(repo_url) + if parsed.scheme != 'https' or not parsed.netloc: + return repo_url, [] + + hostname = (parsed.hostname or '').lower() + detected_provider = 'gitlab' if hostname == 'gitlab.com' else 'github' if hostname == 'github.com' else None + if provider and provider != detected_provider: + return repo_url, [] + provider = detected_provider + if provider not in ('github', 'gitlab'): + return repo_url, [] + + return repo_url, [token] + +def get_git_provider_and_path(repo_url, provider=None): + parsed = urlsplit(repo_url) + hostname = (parsed.hostname or '').lower() + path = parsed.path.strip('/') + if path.endswith('.git'): + path = path[:-4] + provider = provider or ('gitlab' if 'gitlab.' in hostname or hostname == 'gitlab.com' else 'github' if 'github.' in hostname or hostname == 'github.com' else None) + return provider, path + +def recent_commit_boundary(repo_url, provider=None, token=None, max_age_days=None, lookup_pages=3): + if not max_age_days or max_age_days <= 0: + return {'since_commit': None, 'skip': False, 'reason': ''} + + provider, repo_path = get_git_provider_and_path(repo_url, provider) + if provider not in ('github', 'gitlab') or not repo_path: + return {'since_commit': None, 'skip': True, 'reason': 'unsupported provider for commit age lookup'} + + cutoff = datetime.now(timezone.utc) - timedelta(days=max_age_days) + since = cutoff.isoformat().replace('+00:00', 'Z') + headers = {'User-Agent': 'GitSecretsScanner/2.0'} + if token: + headers['Authorization'] = f'Bearer {token}' + + commits = [] + lookup_cap_reached = False + for page in range(1, max(1, lookup_pages) + 1): + try: + if provider == 'github': + url = f'https://api.github.com/repos/{repo_path}/commits' + params = {'since': since, 'per_page': 100, 'page': page} + else: + url = f'https://gitlab.com/api/v4/projects/{quote(repo_path, safe="")}/repository/commits' + params = {'since': since, 'per_page': 100, 'page': page} + response = api_request('GET', url, headers=headers, params=params, timeout=30) + if token and response.status_code in (401, 403): + anonymous = api_request( + 'GET', url, headers={'User-Agent': 'GitSecretsScanner/2.0'}, + params=params, timeout=30, + ) + if anonymous.status_code < 400: + response = anonymous + response.raise_for_status() + page_commits = response.json() + if not page_commits: + break + commits.extend(page_commits) + if len(page_commits) < 100: + break + if page == max(1, lookup_pages): + lookup_cap_reached = True + except requests.exceptions.HTTPError as e: + api_error = github_api_error(e.response) if provider == 'github' else gitlab_api_error(e.response) + if api_error.category == 'not_found': + return { + 'since_commit': None, + 'skip': True, + 'permanent': True, + 'reason': f'commit age lookup found no repository: {str(api_error)[:300]}', + 'error_category': api_error.category, + 'auth_related': False, + } + return { + 'since_commit': None, + 'skip': False, + 'error': True, + 'reason': f'commit age lookup failed ({api_error.category}): {str(api_error)[:300]}', + 'error_category': api_error.category, + 'auth_related': bool(getattr(api_error, 'auth_related', True)), + } + except Exception as e: + if isinstance(e, ApiRequestError): + return { + 'since_commit': None, 'skip': False, 'error': True, + 'reason': f'commit age lookup transport failed: {str(e)[:300]}', + 'error_category': 'network', 'auth_related': False, + } + return { + 'since_commit': None, + 'skip': False, + 'error': True, + 'reason': f'commit age lookup failed: {str(e)[:300]}', + 'error_category': 'unknown', + 'auth_related': False, + } + + if not commits: + return { + 'since_commit': None, + 'skip': True, + 'reason': f'no commits newer than {max_age_days} days' + } + if lookup_cap_reached: + return { + 'since_commit': None, + 'skip': False, + 'reason': 'commit lookup cap reached; scanning without since-commit boundary', + 'recent_commit_count': len(commits), + 'cutoff': since, + } + + oldest = commits[-1] + if provider == 'github': + parents = oldest.get('parents') or [] + parent_sha = parents[0].get('sha') if parents else None + oldest_sha = oldest.get('sha') + else: + parents = oldest.get('parent_ids') or [] + parent_sha = parents[0] if parents else None + oldest_sha = oldest.get('id') + + return { + 'since_commit': parent_sha, + 'skip': False, + 'reason': '', + 'recent_commit_count': len(commits), + 'cutoff': since, + 'boundary_commit': oldest_sha, + } + +def get_trufflehog_cmd(): + """Return configured TruffleHog executable path.""" + return scan_config.trufflehog_path or "trufflehog" + + +def require_trufflehog_launch_authority(command=None): + child_kind = str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() + if child_kind not in {'scanner', 'docker-shadow'}: + raise RuntimeError('TruffleHog launch requires scanner or Docker shadow authority') + manifest = _client_scan_manifest.get() + metadata = {'code_manifest': manifest, 'authority': 'remote-worker'} if manifest else None + if metadata is None: + metadata = require_active_supervisor_child(child_kind=child_kind, require_dsn=True) + manifest = metadata.get('code_manifest') or {} + expected = (manifest.get('executables') or {}).get('trufflehog') or {} + candidate = resolve_manifest_executable((command or [get_trufflehog_cmd()])[0]) + if canonical_path(candidate) != canonical_path(expected.get('path') or ''): + raise RuntimeError('TruffleHog command does not match immutable supervisor authority') + values = list(command or []) + if '--config' in values: + try: + policy_path = canonical_path(values[values.index('--config') + 1]) + except (IndexError, TypeError, ValueError) as exc: + raise RuntimeError('TruffleHog policy argument is incomplete') from exc + assets = manifest.get('assets') or {} + if policy_path not in {canonical_path(item.get('path') or '') for item in assets.values() if isinstance(item, dict)}: + raise RuntimeError('TruffleHog policy does not match immutable supervisor authority') + return metadata + +def get_git_cmd(): + """Return manifested Git in runtime; uninitialized tests may resolve PATH.""" + manifest = _client_scan_manifest.get() + if manifest: + return str(((manifest.get('executables') or {}).get('git') or {}).get('path') or '') + if not _runtime_initialized: + return shutil.which('git') or 'git' + if str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() != 'scanner': + raise RuntimeError('Git clone launch requires scanner authority') + manifest = _client_scan_manifest.get() + if manifest: + metadata = {'code_manifest': manifest, 'authority': 'remote-worker'} + else: + metadata = require_active_supervisor_child(child_kind='scanner', require_dsn=True) + manifest = metadata.get('code_manifest') or {} + expected = (manifest.get('executables') or {}).get('git') or {} + path = expected.get('path') + if not isinstance(path, str) or not os.path.isabs(path): + raise RuntimeError('Git executable is absent from immutable supervisor authority') + return path + + +def prepend_client_git_environment(env): + manifest = _client_scan_manifest.get() + if manifest is None: + return env + path = str(((manifest.get('executables') or {}).get('git') or {}).get('path') or '') + if not os.path.isabs(path): + raise RuntimeError('remote worker Git executable is absent from immutable authority') + directory = os.path.dirname(path) + env['PATH'] = os.pathsep.join((directory, env.get('PATH', ''))) + return env + + +def require_git_clone_launch_authority(cmd): + """Authorize only checkout-free HTTPS clones, never general Git commands.""" + if ( + not isinstance(cmd, (list, tuple)) or len(cmd) != 7 + or any(not isinstance(value, str) or not value or any(ord(ch) < 32 or ord(ch) == 127 for ch in value) for value in cmd) + or list(cmd[1:5]) != ['clone', '--no-checkout', '--no-recurse-submodules', '--'] + ): + raise RuntimeError('Git clone command does not match the allowed argv contract') + source, destination = cmd[5:] + try: + parsed = urlsplit(source) + valid_source = ( + source.startswith('https://') and bool(parsed.hostname) + and parsed.username is None and parsed.password is None + and bool(parsed.path) and parsed.path.startswith('/') + and not parsed.query and not parsed.fragment + and not any(ch.isspace() for ch in source) and '\\' not in source + and '%' not in parsed.netloc and parsed.port != 0 + ) + except ValueError: + valid_source = False + if not valid_source: + raise RuntimeError('Git clone source must be credential-free absolute HTTPS without query or fragment') + if ( + not os.path.isabs(destination) or destination.startswith('-') + or (os.name == 'nt' and not os.path.splitdrive(destination)[0]) + ): + raise RuntimeError('Git clone destination must be an absolute path') + if not os.path.isabs(cmd[0]): + raise RuntimeError('Git command does not match immutable supervisor authority') + if str(os.getenv(CHILD_KIND_ENV) or 'scanner').strip().lower() != 'scanner': + raise RuntimeError('Git clone launch requires scanner authority') + manifest = _client_scan_manifest.get() + if manifest: + metadata = {'code_manifest': manifest, 'authority': 'remote-worker'} + else: + metadata = require_active_supervisor_child(child_kind='scanner', require_dsn=True) + manifest = metadata.get('code_manifest') or {} + expected = (manifest.get('executables') or {}).get('git') or {} + # Authentication rehashes manifest contents, not just path/mtime identity. + if cmd[0] != expected.get('path'): + raise RuntimeError('Git command does not match immutable supervisor authority') + return metadata + + +def get_trufflehog_config(config_path=None): + """Return configured TruffleHog custom detector config path, if any.""" + return str(config_path if config_path is not None else getattr(scan_config, 'trufflehog_config', '') or '').strip() + +def append_trufflehog_scan_args(cmd, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None): + """Append common TruffleHog scan flags in one place.""" + config_path = get_trufflehog_config(trufflehog_config) + if config_path: + cmd.extend(['--config', config_path]) + if detectors: + cmd.extend(['--include-detectors', detectors]) + if exclude_detectors: + cmd.extend(['--exclude-detectors', exclude_detectors]) + if no_verification: + cmd.append('--no-verification') + return cmd + +def get_work_dir(): + """Prepare and return the directory used for TruffleHog temporary data.""" + require_scanner_runtime_initialized() + if not scan_config.work_dir: + raise RuntimeError('TruffleHog work_dir is required') + try: + return require_private_directory(scan_config.work_dir, create=False) + except OSError as exc: + raise RuntimeError(f'Unable to use private TruffleHog work_dir {scan_config.work_dir}: {exc}') from exc + +def create_command_work_dir(): + """Create an isolated temporary directory for one TruffleHog subprocess.""" + work_dir = get_work_dir() + ensure_work_dir_space() + path = tempfile.mkdtemp(prefix='trufflehog-run-', dir=work_dir) + try: + harden_private_directory(path) + if not write_temp_owner(path, ['scanner-workdir'], os.getpid(), required=True): + raise RuntimeError(f'Unable to write required temp owner marker for {path}') + return path + except Exception: + try_remove_tree(path, attempts=2, delay=0.2) + raise + + +def write_temp_owner(path, cmd=None, owner_pid=None, required=False, owner_identity=None): + require_scanner_runtime_initialized() + if not path: + return + try: + owner_pid = int((owner_identity or {}).get('pid') if isinstance(owner_identity, dict) else owner_pid or os.getpid()) + parent_identity = current_process_identity() + if isinstance(owner_identity, dict): + owner = dict(owner_identity) + elif owner_pid == parent_identity.pid: + owner = serialize_process_identity(parent_identity) + else: + with open_process(owner_pid) as retained: + owner = serialize_process_identity(retained.identity) + work_root = canonical_path(get_work_dir()) + candidate = canonical_path(path) + relative = os.path.relpath(candidate, work_root) + if relative == '.' or relative.startswith('..' + os.sep) or os.path.isabs(relative): + raise RuntimeError('temp owner marker path escapes configured work_dir') + parent = serialize_process_identity(parent_identity) + payload = { + 'schema': TEMP_OWNER_SCHEMA, + 'owner_pid': owner['pid'], + 'owner_creation_time': owner['creation_time'], + 'owner_executable': owner['executable'], + 'parent_pid': parent['pid'], + 'parent_creation_time': parent['creation_time'], + 'parent_executable': parent['executable'], + 'created_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), + 'root_kind': 'work', + 'relative_path': relative.replace(os.sep, '/'), + 'command': redact_command_args(cmd or [])[:64], + } + atomic_write_private_json(os.path.join(path, TEMP_OWNER_FILE), payload) + return True + except (OSError, ValueError) as e: + if required: + raise RuntimeError(f'Unable to write required temp owner marker for {path}: {e}') from e + logger.warning(f'Unable to write temp owner marker for {path}: {str(e)[:200]}') + return False + + +def redact_command_args(args): + sensitive_flags = {'--token', '--docker-token', '--password', '--api-key', '--secret'} + redacted = [] + hide_next = False + for value in args or []: + text = str(value) + if hide_next: + redacted.append('***REDACTED***') + hide_next = False + continue + flag = text.split('=', 1)[0].lower() + if flag in sensitive_flags: + if '=' in text: + redacted.append(text.split('=', 1)[0] + '=***REDACTED***') + else: + redacted.append(text) + hide_next = True + continue + try: + parsed = urlsplit(text) + if parsed.scheme in ('http', 'https') and parsed.hostname and ('@' in parsed.netloc or parsed.password): + netloc = parsed.hostname + if parsed.port: + netloc += f':{parsed.port}' + else: + netloc = parsed.netloc + if parsed.scheme in ('http', 'https') and parsed.hostname: + query = [] + for key, query_value in parse_qsl(parsed.query, keep_blank_values=True): + sensitive = any(part in key.lower() for part in ('password', 'passwd', 'pwd', 'token', 'secret', 'credential')) + query.append((key, '***REDACTED***' if sensitive else query_value)) + text = urlunsplit((parsed.scheme, netloc, parsed.path, urlencode(query), parsed.fragment)) + except ValueError: + pass + redacted.append(text) + return redacted + + +def read_temp_owner(path): + marker = os.path.join(path, TEMP_OWNER_FILE) + try: + if not private_file_ready(marker): + return {} + data = read_private_json(marker) + return data if isinstance(data, dict) else {} + except (OSError, ValueError): + return {} + + +def temp_dir_active(path): + owner = read_temp_owner(path) + if owner.get('schema') != TEMP_OWNER_SCHEMA: + return True + states = [ + exact_process_identity_state( + owner.get(f'{prefix}_pid'), + owner.get(f'{prefix}_creation_time'), + owner.get(f'{prefix}_executable'), + ) + for prefix in ('owner', 'parent') + ] + return any(state in ('alive', 'unknown') for state in states) + + +def create_docker_config_dir(): + work_dir = get_work_dir() + docker_config_root = os.path.join(work_dir, 'docker-config') + require_private_directory(docker_config_root, create=True) + path = tempfile.mkdtemp(prefix='docker-config-', dir=docker_config_root) + harden_private_directory(path) + write_temp_owner(path, ['docker-auth-config'], os.getpid()) + return path + +def force_remove_readonly(function, path, exc_info): + try: + reject_reparse_components(path) + os.chmod(path, stat.S_IWRITE) + function(path) + except Exception: + pass + +def try_remove_tree(path, attempts=1, delay=0.0): + if not path: + return True + + for attempt in range(attempts): + try: + budget = JanitorBudget( + max_candidates=1, + max_entries=10000, + max_bytes=1024 * 1024 * 1024, + max_seconds=5.0, + max_depth=64, + ) + return bounded_remove_tree(path, budget) + except FileNotFoundError: + return True + except KeyboardInterrupt: + raise + except Exception: + if delay and attempt + 1 < attempts: + time.sleep(delay) + return False + +def _shared_staging_owners(roots): + """Resolve only private trees already owned by this scanner; never adopt input data.""" + work_root = canonical_path(get_work_dir()) + current = serialize_process_identity(current_process_identity()) + owners = {} + for value in roots: + root = canonical_path(require_private_directory(value, create=False)) + if root == work_root or os.path.commonpath((root, work_root)) != work_root: + raise RuntimeError('shared staging root escapes configured work_dir') + while root != work_root and not os.path.lexists(os.path.join(root, TEMP_OWNER_FILE)): + root = os.path.dirname(root) + if root == work_root: + raise RuntimeError('shared staging root has no authenticated owner') + if root in owners: + continue + require_private_directory(root, create=False) + marker = read_private_json(require_private_file(os.path.join(root, TEMP_OWNER_FILE)), max_bytes=65536) + relative = os.path.relpath(root, work_root).replace(os.sep, '/') + if marker.get('schema') != TEMP_OWNER_SCHEMA or marker.get('root_kind') != 'work' or marker.get('relative_path') != relative: + raise RuntimeError('shared staging owner marker does not match its private root') + if any(marker.get(f'{prefix}_{field}') != current[field] + for prefix in ('owner', 'parent') for field in ('pid', 'creation_time', 'executable')): + raise RuntimeError('shared staging root belongs to another owner') + if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')): + if any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')) or ( + exact_process_identity_state(marker.get('child_pid'), marker.get('child_creation_time'), + marker.get('child_executable')) != 'dead' + ): + raise RuntimeError('shared staging root has an unconfirmed child') + for field in ('pid', 'creation_time', 'executable'): + marker.pop(f'child_{field}', None) + owners[root] = marker + return list(owners.items()) + + +def cleanup_command_work_dir(path): + if not path: + return + + marker_path = os.path.join(path, TEMP_OWNER_FILE) + if os.path.lexists(marker_path): + try: + marker = read_private_json(require_private_file(marker_path), max_bytes=65536) + if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')): + if any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')) or ( + exact_process_identity_state(marker.get('child_pid'), marker.get('child_creation_time'), + marker.get('child_executable')) != 'dead' + ): + logger.warning('Retaining private command tree with an unconfirmed child') + return + except (OSError, TypeError, ValueError): + logger.warning('Retaining private command tree with unreadable ownership evidence') + return + if try_remove_tree(path, attempts=2, delay=0.2): + return + + logger.info('Bounded immediate cleanup deferred to authenticated janitor: %s', path) + + +def approved_pending_temp_path(path, work_dir=None): + """Validate scanner-owned placement without following an external path.""" + if not path or not os.path.isabs(path): + return False + try: + work_root = work_dir or get_work_dir() + reject_reparse_components(work_root) + reject_reparse_components(path) + if is_reparse_point(path) or not os.path.isdir(path) or not private_directory_ready(path): + return False + work_root = canonical_path(work_root) + candidate = canonical_path(path) + if os.path.commonpath((work_root, candidate)) != work_root or candidate == work_root: + return False + relative = os.path.relpath(candidate, work_root) + except (OSError, ValueError): + return False + parts = relative.split(os.sep) + if len(parts) == 1: + approved_name = parts[0].startswith(APPROVED_TEMP_PREFIXES) + elif len(parts) == 2 and parts[0] == 'docker-config': + approved_name = parts[1].startswith('docker-config-') + elif len(parts) == 2 and parts[0] == 'hg': + approved_name = parts[1].startswith('hg-run-') + elif len(parts) == 2 and parts[0] == 'tmp': + approved_name = parts[1].startswith(APPROVED_TEMP_PREFIXES) + elif len(parts) == 3 and parts[:2] == ['tmp', 'docker-config']: + approved_name = parts[2].startswith('docker-config-') + else: + approved_name = False + if not approved_name: + return False + owner = read_temp_owner(candidate) + owner_pid = owner.get('owner_pid') or owner.get('parent_pid') + return bool(owner_pid) and not temp_dir_active(candidate) + +def cleanup_pending_command_work_dirs(max_items=None, attempts=1, delay=0.0, log_failures=False): + if log_failures: + logger.info('Source-side pending temp cleanup is retired; the authenticated janitor owns recovery') + return 0 + + +def cleanup_assignment_work_dir(max_items=256): + """Bound one cooperative cleanup pass to this runner's private work root.""" + work_root = get_work_dir() + report = {'enumerated': 0, 'removed': 0, 'retained': 0} + with os.scandir(work_root) as entries: + for entry in entries: + report['enumerated'] += 1 + if report['enumerated'] > max(1, int(max_items)): + report['retained'] += 1 + break + if ( + entry.is_symlink() + or not entry.is_dir(follow_symlinks=False) + or not entry.name.startswith(APPROVED_TEMP_PREFIXES) + ): + continue + if cleanup_command_work_dir(entry.path) is None and not os.path.exists(entry.path): + report['removed'] += 1 + else: + report['retained'] += 1 + return report + +def ensure_work_dir_space(): + work_dir = get_work_dir() + if not work_dir or scan_config.min_free_gb <= 0: + return + + min_free_bytes = scan_config.min_free_gb * 1024 * 1024 * 1024 + free_bytes = shutil.disk_usage(work_dir).free + if free_bytes >= min_free_bytes: + return + + raise RuntimeError( + f"Not enough free space on {work_dir}: {free_bytes / (1024 ** 3):.2f} GB free, " + f"minimum is {scan_config.min_free_gb:.2f} GB; admission is closed without cleanup" + ) + +def cleanup_stale_temp_dirs(age_minutes=120, log=True, max_items=None): + """Compatibility no-op; stale recovery is isolated in janitor.py.""" + if log: + logger.info('Source-side stale temp cleanup is retired; the authenticated janitor owns recovery') + return 0 + +def get_results_dir(): + """Prepare and return the directory used for persisted scan output.""" + require_scanner_runtime_initialized() + if not scan_config.results_dir: + raise RuntimeError('scan results directory is required') + try: + return require_private_directory(scan_config.results_dir, create=False) + except OSError as exc: + raise RuntimeError(f'Unable to use private scan results directory {scan_config.results_dir}: {exc}') from exc + +def append_jsonl(path, payload): + lock = None + lock_path = f'{path}.lock' + try: + require_private_directory(os.path.dirname(os.path.abspath(path)), create=True) + if os.path.lexists(path): + reject_reparse_components(path) + lock = acquire_file_lock(lock_path, timeout_sec=30) + repair_jsonl_tail(path) + serialized = (json.dumps(payload, ensure_ascii=False, default=str) + '\n').encode('utf-8') + with open(path, 'ab') as f: + f.write(serialized) + f.flush() + os.fsync(f.fileno()) + harden_private_file(path) + return True + except OSError as e: + if getattr(e, 'errno', None) == 28: + logger.error(f"No space left while writing {path}. Result was not persisted.") + else: + logger.error(f"Unable to write {path}: {str(e)}") + return False + finally: + if lock is not None: + release_file_lock(lock, lock_path) + + +def jsonl_manifest_path(path): + base, ext = os.path.splitext(path) + return f'{base}.manifest.json' + + +def load_jsonl_manifest(path): + manifest_path = jsonl_manifest_path(path) + try: + if os.path.getsize(manifest_path) > 1024 * 1024: + raise ValueError(f'JSONL manifest exceeds its bounded size: {manifest_path}') + with open(manifest_path, 'r', encoding='utf-8') as f: + data = json.load(f) + return data if isinstance(data, dict) else {} + except FileNotFoundError: + return {} + + +def write_jsonl_manifest(path, manifest): + manifest_path = jsonl_manifest_path(path) + tmp_path = f'{manifest_path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp' + with open(tmp_path, 'w', encoding='utf-8') as f: + json.dump(manifest, f, ensure_ascii=False, indent=2, sort_keys=True) + f.flush() + os.fsync(f.fileno()) + harden_private_file(tmp_path) + os.replace(tmp_path, manifest_path) + harden_private_file(manifest_path) + + +def next_jsonl_segment_path(path, manifest): + base, ext = os.path.splitext(path) + seq = int(manifest.get('next_sequence') or 1) + while True: + segment = f'{base}.{seq:06d}{ext or ".jsonl"}' + if not os.path.exists(segment): + return segment, seq + seq += 1 + + +def acquire_file_lock(lock_path, stale_sec=300, timeout_sec=30): + require_private_directory(os.path.dirname(os.path.abspath(lock_path)), create=True) + reject_reparse_components(os.path.dirname(os.path.abspath(lock_path))) + deadline = time.monotonic() + max(0.01, float(timeout_sec)) + while True: + lock = PrivateFileLock(lock_path) + try: + return lock.acquire() + except BlockingIOError: + if time.monotonic() >= deadline: + raise TimeoutError(f'timed out acquiring lock {lock_path}') + time.sleep(min(0.05, max(0.0, deadline - time.monotonic()))) + + +def release_file_lock(lock, lock_path): + try: + lock.release() + except (AttributeError, OSError): + return + + +class JsonlProjectionReconciliationRequired(RuntimeError): + pass + + +def jsonl_ledger_path(path): + base, _ = os.path.splitext(path) + return f'{base}.publication-ledger.sqlite3' + + +def _projection_segment_sequence(path): + parent = os.path.dirname(os.path.abspath(path)) + base, extension = os.path.splitext(os.path.basename(path)) + pattern = re.compile(rf'^{re.escape(base)}\.(\d{{6}}){re.escape(extension)}$') + output = [] + inspect_limit = max(2, int(getattr(scan_config, 'jsonl_max_segments', 16)) + 1) + try: + with os.scandir(parent) as entries: + for entry in entries: + match = pattern.fullmatch(entry.name) + if not match: + continue + if entry.is_symlink() or is_reparse_point(entry.path) or not entry.is_file(follow_symlinks=False): + raise JsonlProjectionReconciliationRequired(f'unsafe JSONL segment entry: {entry.path}') + output.append((int(match.group(1)), entry.path)) + if len(output) > inspect_limit: + raise JsonlProjectionReconciliationRequired( + f'JSONL physical segment count exceeds its bound for {path}; use offline reconciliation' + ) + except FileNotFoundError: + return [] + return sorted(output) + + +def _write_torn_tail_quarantine(path, payload): + limit = max(1, int(getattr(scan_config, 'jsonl_torn_quarantine_max_bytes', 64 * 1024))) + sample = bytes(payload[:limit]) + quarantine = f'{path}.torn-tail.bin' + temporary = f'{quarantine}.{os.getpid()}.{threading.get_ident()}.tmp' + with open(temporary, 'wb') as handle: + handle.write(sample) + handle.flush() + os.fsync(handle.fileno()) + harden_private_file(temporary) + durable_replace(temporary, quarantine) + harden_private_file(quarantine) + + +def repair_jsonl_tail(path): + """Quarantine and remove one bounded unterminated tail before appending.""" + if not os.path.exists(path): + return 0 + reject_reparse_components(path) + size = os.path.getsize(path) + if size <= 0: + return 0 + scan_limit = max(1, int(getattr(scan_config, 'jsonl_tail_scan_max_bytes', 8 * 1024 * 1024))) + with open(path, 'r+b') as handle: + handle.seek(-1, os.SEEK_END) + if handle.read(1) == b'\n': + return 0 + start = max(0, size - scan_limit) + handle.seek(start) + tail = handle.read(size - start) + newline = tail.rfind(b'\n') + if newline < 0 and start: + raise JsonlProjectionReconciliationRequired( + f'JSONL tail exceeds the bounded repair window for {path}; use offline reconciliation' + ) + truncate_at = start + newline + 1 if newline >= 0 else 0 + torn = tail[newline + 1:] if newline >= 0 else tail + _write_torn_tail_quarantine(path, torn) + handle.truncate(truncate_at) + handle.flush() + os.fsync(handle.fileno()) + logger.error('Quarantined and truncated %s torn byte(s) from %s', size - truncate_at, path) + return size - truncate_at + + +def _open_projection_ledger(path): + ledger_path = jsonl_ledger_path(path) + reject_reparse_components(os.path.dirname(os.path.abspath(ledger_path))) + if os.path.lexists(ledger_path): + reject_reparse_components(ledger_path) + if not private_file_ready(ledger_path): + raise JsonlProjectionReconciliationRequired(f'JSONL publication ledger is not private: {ledger_path}') + connection = sqlite3.connect(ledger_path, timeout=30) + try: + connection.execute('PRAGMA busy_timeout=30000') + connection.execute('PRAGMA journal_mode=DELETE') + connection.execute('PRAGMA synchronous=FULL') + connection.executescript(''' + CREATE TABLE IF NOT EXISTS publication_identity ( + identity_key TEXT NOT NULL, + identity_value TEXT NOT NULL, + payload_sha256 TEXT NOT NULL, + state TEXT NOT NULL, + file_name TEXT NOT NULL, + byte_offset INTEGER NOT NULL, + byte_length INTEGER NOT NULL, + created_at REAL NOT NULL, + updated_at REAL NOT NULL, + PRIMARY KEY(identity_key, identity_value) + ); + CREATE TABLE IF NOT EXISTS publication_meta ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL + ); + CREATE TABLE IF NOT EXISTS publication_identity_variant ( + identity_key TEXT NOT NULL, + identity_value TEXT NOT NULL, + payload_sha256 TEXT NOT NULL, + file_name TEXT NOT NULL, + byte_offset INTEGER NOT NULL, + byte_length INTEGER NOT NULL, + created_at REAL NOT NULL, + PRIMARY KEY(identity_key, identity_value, payload_sha256) + ); + CREATE TABLE IF NOT EXISTS reconciliation_issue ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + identity_key TEXT NOT NULL, + file_name TEXT NOT NULL, + file_device TEXT NOT NULL, + file_inode TEXT NOT NULL, + file_size INTEGER NOT NULL, + file_mtime_ns TEXT NOT NULL, + byte_offset INTEGER NOT NULL, + byte_length INTEGER NOT NULL, + record_sha256 TEXT NOT NULL, + classification TEXT NOT NULL, + status TEXT NOT NULL, + created_at REAL NOT NULL, + resolved_at REAL, + UNIQUE(identity_key, file_name, byte_offset, record_sha256) + ); + CREATE TABLE IF NOT EXISTS reconciliation_variant_issue ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + identity_key TEXT NOT NULL, + identity_sha256 TEXT NOT NULL, + file_name TEXT NOT NULL, + file_device TEXT NOT NULL, + file_inode TEXT NOT NULL, + file_size INTEGER NOT NULL, + file_mtime_ns TEXT NOT NULL, + byte_offset INTEGER NOT NULL, + byte_length INTEGER NOT NULL, + payload_sha256 TEXT NOT NULL, + field_name_set_sha256 TEXT NOT NULL, + status TEXT NOT NULL, + created_at REAL NOT NULL, + resolved_at REAL, + UNIQUE(identity_key, identity_sha256, file_name, byte_offset, payload_sha256) + ); + CREATE INDEX IF NOT EXISTS idx_publication_identity_state_created + ON publication_identity(state, created_at); + CREATE INDEX IF NOT EXISTS idx_publication_identity_file_state + ON publication_identity(file_name, state); + CREATE INDEX IF NOT EXISTS idx_publication_identity_variant_identity + ON publication_identity_variant(identity_key, identity_value); + CREATE INDEX IF NOT EXISTS idx_reconciliation_issue_status + ON reconciliation_issue(status, id); + CREATE INDEX IF NOT EXISTS idx_reconciliation_variant_issue_status + ON reconciliation_variant_issue(status, id); + ''') + connection.execute( + '''INSERT OR IGNORE INTO publication_identity_variant ( + identity_key, identity_value, payload_sha256, file_name, + byte_offset, byte_length, created_at + ) + SELECT identity_key, identity_value, payload_sha256, file_name, + byte_offset, byte_length, created_at + FROM publication_identity WHERE state = 'appended' ''' + ) + row = connection.execute("SELECT value FROM publication_meta WHERE key = 'row_count'").fetchone() + if row is None: + count = int(connection.execute('SELECT COUNT(*) FROM publication_identity').fetchone()[0]) + connection.execute( + "INSERT INTO publication_meta(key, value) VALUES ('row_count', ?)", + (str(count),), + ) + connection.commit() + harden_private_file(ledger_path) + return connection + except BaseException: + connection.close() + raise + + +def _ledger_row_count(connection): + row = connection.execute("SELECT value FROM publication_meta WHERE key = 'row_count'").fetchone() + return max(0, int(row[0] if row else 0)) + + +def _set_ledger_row_count(connection, count): + connection.execute( + "INSERT OR REPLACE INTO publication_meta(key, value) VALUES ('row_count', ?)", + (str(max(0, int(count))),), + ) + + +def _bounded_projection_bytes(path, offset, length): + max_record = max( + 1024 * 1024, + int(getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)) + 1, + ) + if offset < 0 or length <= 0 or length > max_record: + return b'' + try: + with open(path, 'rb') as handle: + handle.seek(offset) + return handle.read(length) + except OSError: + return b'' + + +def _recover_prepared_publications(connection, path): + rows = connection.execute( + '''SELECT identity_key, identity_value, payload_sha256, file_name, byte_offset, byte_length + FROM publication_identity WHERE state = 'prepared' ORDER BY created_at LIMIT 2''' + ).fetchall() + if len(rows) > 1: + raise JsonlProjectionReconciliationRequired('publication ledger contains multiple unresolved append states') + for identity_key, identity_value, digest, file_name, offset, length in rows: + candidate = os.path.join(os.path.dirname(os.path.abspath(path)), os.path.basename(file_name)) + payload = _bounded_projection_bytes(candidate, int(offset), int(length)) + if payload.endswith(b'\n') and hashlib.sha256(payload).hexdigest() == digest: + connection.execute( + '''UPDATE publication_identity SET state = 'appended', updated_at = ? + WHERE identity_key = ? AND identity_value = ? AND state = 'prepared' ''', + (time.time(), identity_key, identity_value), + ) + connection.execute( + '''INSERT OR IGNORE INTO publication_identity_variant ( + identity_key, identity_value, payload_sha256, file_name, + byte_offset, byte_length, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?)''', + (identity_key, identity_value, digest, file_name, offset, length, time.time()), + ) + else: + connection.execute( + 'DELETE FROM publication_identity WHERE identity_key = ? AND identity_value = ? AND state = ?', + (identity_key, identity_value, 'prepared'), + ) + _set_ledger_row_count(connection, _ledger_row_count(connection) - 1) + connection.commit() + + +def _ensure_projection_ledger_bootstrapped(connection, path, identity_key): + marker = f'bootstrapped:{identity_key}' + if connection.execute('SELECT 1 FROM publication_meta WHERE key = ?', (marker,)).fetchone(): + return + candidates = projection_segment_paths(path) + total_bytes = sum(os.path.getsize(candidate) for candidate in candidates) + max_bytes = max(0, int(getattr(scan_config, 'jsonl_legacy_index_max_bytes', 16 * 1024 * 1024))) + if total_bytes > max_bytes: + raise JsonlProjectionReconciliationRequired( + f'existing JSONL history for {path} is unindexed ({total_bytes} bytes); ' + 'run the offline JSONL reconciliation procedure before publication' + ) + row_count = _ledger_row_count(connection) + row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000))) + for candidate in candidates: + offset = 0 + with open(candidate, 'rb') as handle: + for raw_line in handle: + if not raw_line.endswith(b'\n'): + raise JsonlProjectionReconciliationRequired(f'unterminated closed JSONL record in {candidate}') + identity = _projection_identity_from_line(raw_line, identity_key, candidate) + if identity: + digest = hashlib.sha256(raw_line).hexdigest() + existing = connection.execute( + '''SELECT payload_sha256 FROM publication_identity + WHERE identity_key = ? AND identity_value = ?''', + (identity_key, identity), + ).fetchone() + if existing and existing[0] != digest: + raise JsonlProjectionReconciliationRequired( + 'conflicting identity in existing JSONL history: ' + f'identity_key={identity_key} ' + f'identity_sha256={hashlib.sha256(identity.encode("utf-8")).hexdigest()}' + ) + if not existing: + if row_count >= row_limit: + raise JsonlProjectionReconciliationRequired('existing JSONL identities exceed the ledger row bound') + now = time.time() + connection.execute( + '''INSERT INTO publication_identity ( + identity_key, identity_value, payload_sha256, state, file_name, + byte_offset, byte_length, created_at, updated_at + ) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''', + ( + identity_key, identity, digest, os.path.basename(candidate), + offset, len(raw_line), now, now, + ), + ) + connection.execute( + '''INSERT INTO publication_identity_variant ( + identity_key, identity_value, payload_sha256, file_name, + byte_offset, byte_length, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?)''', + ( + identity_key, identity, digest, os.path.basename(candidate), + offset, len(raw_line), now, + ), + ) + row_count += 1 + offset += len(raw_line) + _set_ledger_row_count(connection, row_count) + connection.execute('INSERT INTO publication_meta(key, value) VALUES (?, ?)', (marker, '1')) + connection.commit() + + +def _projection_identity_from_line(raw_line, identity_key, candidate): + if not raw_line.endswith(b'\n'): + raise JsonlProjectionReconciliationRequired(f'unterminated projection record in {candidate}') + if identity_key == 'error_row_id': + try: + identity, separator, _ = raw_line.partition(b'\t') + if not separator or not identity: + raise ValueError('missing error projection identity separator') + return identity.decode('utf-8') + except UnicodeDecodeError as exc: + raise JsonlProjectionReconciliationRequired( + f'invalid existing projection record in {candidate}' + ) from exc + except ValueError as exc: + raise JsonlProjectionReconciliationRequired( + f'invalid existing projection record in {candidate}' + ) from exc + try: + payload = json.loads(raw_line.decode('utf-8')) + except (UnicodeDecodeError, ValueError) as exc: + raise JsonlProjectionReconciliationRequired( + f'invalid existing JSONL record in {candidate}' + ) from exc + return str(payload.get(identity_key) or '') if isinstance(payload, dict) else '' + + +def _bounded_projection_json_values(raw_line): + body = raw_line[:-1] if raw_line.endswith(b'\n') else raw_line + if not raw_line.endswith(b'\n'): + return None, 'unterminated_record' + try: + text = body.decode('utf-8') + except UnicodeDecodeError: + return None, 'invalid_utf8' + if text.startswith('\ufeff'): + return None, 'utf8_bom_prefix' + try: + value = json.loads(text) + return [{ + 'value': value, + 'relative_offset': 0, + 'byte_length': len(raw_line), + 'payload_sha256': hashlib.sha256(raw_line).hexdigest(), + }], 'single_json' + except ValueError: + pass + decoder = json.JSONDecoder() + position = 0 + parsed = [] + while True: + while position < len(text) and text[position].isspace(): + position += 1 + if position >= len(text): + break + start = position + try: + value, position = decoder.raw_decode(text, position) + except json.JSONDecodeError: + parsed = [] + break + parsed.append((start, position, value)) + if len(parsed) > 1 and position >= len(text) and all(isinstance(item[2], dict) for item in parsed): + values = [] + for start, end, value in parsed: + prefix_bytes = len(text[:start].encode('utf-8')) + serialized = text[start:end].encode('utf-8') + b'\n' + values.append({ + 'value': value, + 'relative_offset': prefix_bytes, + 'byte_length': len(serialized), + 'payload_sha256': hashlib.sha256(serialized).hexdigest(), + }) + return values, 'concatenated_json_objects' + if re.match(r'^[0-9]+,\s*', text): + return None, 'legacy_numeric_prefix_corrupt_json' + return None, 'invalid_json' + + +def _projection_field_name_set_sha256(value): + paths = [] + + def visit(item, prefix=''): + if isinstance(item, dict): + for key in sorted(map(str, item.keys())): + path = prefix + key + paths.append(path) + visit(item.get(key), path + '.') + elif isinstance(item, list): + paths.append(prefix + '[]') + for child in item[:32]: + visit(child, prefix + '[].') + + visit(value) + return hashlib.sha256('\x00'.join(sorted(set(paths))).encode('utf-8')).hexdigest() + + +def _error_projection_field_name_set_sha256(raw_line): + try: + text = raw_line[:-1].decode('utf-8') if raw_line.endswith(b'\n') else raw_line.decode('utf-8') + tail = text.rsplit('\t', 1)[-1] + value = json.loads(tail) + except (UnicodeDecodeError, ValueError): + value = {} + return _projection_field_name_set_sha256(value) + + +def _save_projection_reconciliation_progress(ledger, prefix, file_index, offset, indexed_rows): + _set_ledger_row_count(ledger, indexed_rows) + for key, value in ( + (prefix + 'file_index', str(file_index)), + (prefix + 'offset', str(offset)), + ): + ledger.execute('INSERT OR REPLACE INTO publication_meta(key, value) VALUES (?, ?)', (key, value)) + + +def _record_projection_reconciliation_issue( + ledger, + identity_key, + plan_entry, + byte_offset, + raw_line, + classification, + resolved=False, + byte_length=None, + record_sha256=None, +): + if raw_line is not None: + byte_length = len(raw_line) + record_sha256 = hashlib.sha256(raw_line).hexdigest() + byte_length = int(byte_length or 0) + digest = str(record_sha256 or '').strip().lower() + if byte_length <= 0 or not re.fullmatch(r'[a-f0-9]{64}', digest): + raise ValueError('projection reconciliation issue metadata is invalid') + now = time.time() + ledger.execute( + '''INSERT OR IGNORE INTO reconciliation_issue ( + identity_key, file_name, file_device, file_inode, file_size, file_mtime_ns, + byte_offset, byte_length, record_sha256, classification, status, created_at, resolved_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + identity_key, + os.path.basename(plan_entry['path']), + str(plan_entry['device']), + str(plan_entry['inode']), + int(plan_entry['size']), + str(plan_entry['mtime_ns']), + int(byte_offset), + byte_length, + digest, + classification, + 'resolved' if resolved else 'pending', + now, + now if resolved else None, + ), + ) + if resolved: + ledger.execute( + '''UPDATE reconciliation_issue SET status = 'resolved', resolved_at = COALESCE(resolved_at, ?) + WHERE identity_key = ? AND file_name = ? AND byte_offset = ? AND record_sha256 = ?''', + (now, identity_key, os.path.basename(plan_entry['path']), int(byte_offset), digest), + ) + return ledger.execute( + '''SELECT id, status FROM reconciliation_issue + WHERE identity_key = ? AND file_name = ? AND byte_offset = ? AND record_sha256 = ?''', + (identity_key, os.path.basename(plan_entry['path']), int(byte_offset), digest), + ).fetchone() + + +def _record_projection_variant_issue( + ledger, + identity_key, + identity, + plan_entry, + byte_offset, + byte_length, + payload_sha256, + field_name_set_sha256, + resolved=False, +): + identity_sha256 = hashlib.sha256(identity.encode('utf-8')).hexdigest() + now = time.time() + ledger.execute( + '''INSERT OR IGNORE INTO reconciliation_variant_issue ( + identity_key, identity_sha256, file_name, file_device, file_inode, + file_size, file_mtime_ns, byte_offset, byte_length, payload_sha256, + field_name_set_sha256, status, created_at, resolved_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + identity_key, + identity_sha256, + os.path.basename(plan_entry['path']), + str(plan_entry['device']), + str(plan_entry['inode']), + int(plan_entry['size']), + str(plan_entry['mtime_ns']), + int(byte_offset), + int(byte_length), + payload_sha256, + field_name_set_sha256, + 'resolved' if resolved else 'pending', + now, + now if resolved else None, + ), + ) + if resolved: + ledger.execute( + '''UPDATE reconciliation_variant_issue + SET status = 'resolved', resolved_at = COALESCE(resolved_at, ?) + WHERE identity_key = ? AND identity_sha256 = ? AND file_name = ? + AND byte_offset = ? AND payload_sha256 = ?''', + ( + now, identity_key, identity_sha256, os.path.basename(plan_entry['path']), + int(byte_offset), payload_sha256, + ), + ) + return ledger.execute( + '''SELECT id, status FROM reconciliation_variant_issue + WHERE identity_key = ? AND identity_sha256 = ? AND file_name = ? + AND byte_offset = ? AND payload_sha256 = ?''', + ( + identity_key, identity_sha256, os.path.basename(plan_entry['path']), + int(byte_offset), payload_sha256, + ), + ).fetchone() + + +def _stream_projection_record(handle, first_chunk, chunk_bytes=1024 * 1024): + digest = hashlib.sha256() + digest.update(first_chunk) + total = len(first_chunk) + newline_terminated = first_chunk.endswith(b'\n') + while not newline_terminated: + chunk = handle.readline(max(1, int(chunk_bytes))) + if not chunk: + break + digest.update(chunk) + total += len(chunk) + newline_terminated = chunk.endswith(b'\n') + return total, digest.hexdigest(), newline_terminated + + +def _projection_reconciliation_plan(path): + candidates = projection_segment_paths(path) + if not candidates: + _publish_empty_jsonl_generation(path) + candidates = [os.path.abspath(path)] + plan = [] + for candidate in candidates: + require_private_file(candidate) + details = os.stat(candidate, follow_symlinks=False) + plan.append({ + 'path': os.path.abspath(candidate), + 'device': int(getattr(details, 'st_dev', 0) or 0), + 'inode': int(getattr(details, 'st_ino', 0) or 0), + 'size': int(details.st_size), + 'mtime_ns': int(getattr(details, 'st_mtime_ns', int(details.st_mtime * 1_000_000_000))), + }) + return plan + + +def reconcile_projection_ledger_batch( + path, + identity_key, + max_rows=10000, + max_bytes=64 * 1024 * 1024, + max_seconds=30.0, + row_limit=None, + ledger_byte_limit=None, + max_record_bytes=None, +): + """Build one bounded, resumable ledger batch without modifying JSONL history.""" + if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'): + raise ValueError('unsupported projection reconciliation identity') + path = os.path.abspath(path) + require_private_directory(os.path.dirname(path), create=False) + max_rows = max(1, int(max_rows)) + max_bytes = max(1, int(max_bytes)) + max_seconds = max(0.01, float(max_seconds)) + row_limit = max(1, int(row_limit or getattr(scan_config, 'jsonl_ledger_max_rows', 1000000))) + ledger_byte_limit = max( + 1024 * 1024, + int(ledger_byte_limit or getattr(scan_config, 'jsonl_ledger_max_bytes', 512 * 1024 * 1024)), + ) + max_record_bytes = max( + 1024 * 1024, + int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)), + ) + lock_path = f'{path}.lock' + lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) + ledger = None + try: + prefix = f'offline-reconcile:{identity_key}:' + marker = f'bootstrapped:{identity_key}' + ledger = _open_projection_ledger(path) + if ledger.execute('SELECT 1 FROM publication_meta WHERE key = ?', (marker,)).fetchone(): + return { + 'path': path, 'identity_key': identity_key, 'complete': True, + 'batch_rows': 0, 'batch_bytes': 0, 'indexed_rows': _ledger_row_count(ledger), + } + plan = _projection_reconciliation_plan(path) + plan_json = json.dumps(plan, ensure_ascii=True, sort_keys=True, separators=(',', ':')) + existing_plan = ledger.execute( + 'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'plan',), + ).fetchone() + if existing_plan and existing_plan[0] != plan_json: + raise JsonlProjectionReconciliationRequired( + f'projection history changed during offline reconciliation: {path}' + ) + if not existing_plan: + ledger.execute( + 'INSERT INTO publication_meta(key, value) VALUES (?, ?)', + (prefix + 'plan', plan_json), + ) + file_index_row = ledger.execute( + 'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'file_index',), + ).fetchone() + offset_row = ledger.execute( + 'SELECT value FROM publication_meta WHERE key = ?', (prefix + 'offset',), + ).fetchone() + file_index = max(0, int(file_index_row[0] if file_index_row else 0)) + offset = max(0, int(offset_row[0] if offset_row else 0)) + indexed_rows = _ledger_row_count(ledger) + batch_rows = 0 + batch_bytes = 0 + started = time.monotonic() + while file_index < len(plan): + candidate = plan[file_index]['path'] + with open(candidate, 'rb') as handle: + handle.seek(offset) + while True: + raw_line = handle.readline(max_record_bytes + 1) + if not raw_line: + file_index += 1 + offset = 0 + break + if len(raw_line) > max_record_bytes: + byte_length, record_sha256, newline_terminated = _stream_projection_record( + handle, raw_line, + ) + classification = 'oversized_record' if newline_terminated else 'unterminated_record' + issue = _record_projection_reconciliation_issue( + ledger, + identity_key, + plan[file_index], + offset, + None, + classification, + byte_length=byte_length, + record_sha256=record_sha256, + ) + _save_projection_reconciliation_progress( + ledger, prefix, file_index, offset, indexed_rows, + ) + ledger.commit() + if issue[1] != 'resolved': + raise JsonlProjectionReconciliationRequired( + 'projection reconciliation issue requires explicit review: ' + f'id={issue[0]} file={os.path.basename(candidate)} offset={offset} ' + f'length={byte_length} sha256={record_sha256} ' + f'classification={classification}' + ) + offset += byte_length + batch_rows += 1 + batch_bytes += byte_length + if ( + batch_rows >= max_rows + or batch_bytes >= max_bytes + or time.monotonic() - started >= max_seconds + ): + break + continue + if identity_key == 'error_row_id': + try: + identity = _projection_identity_from_line(raw_line, identity_key, candidate) + if len(identity) > 256 or any( + character in identity for character in ('\x00', '\r', '\n') + ): + values = None + classification = 'invalid_error_projection' + else: + values = [{ + 'identity': identity, + 'relative_offset': 0, + 'byte_length': len(raw_line), + 'payload_sha256': hashlib.sha256(raw_line).hexdigest(), + 'field_name_set_sha256': _error_projection_field_name_set_sha256(raw_line), + }] + classification = 'error_projection' + except JsonlProjectionReconciliationRequired: + values = None + classification = ( + 'unterminated_record' if not raw_line.endswith(b'\n') + else 'invalid_error_projection' + ) + else: + parsed_values, classification = _bounded_projection_json_values(raw_line) + values = None if parsed_values is None else [{ + 'identity': ( + str(item['value'].get(identity_key) or '') + if isinstance(item['value'], dict) else '' + ), + 'relative_offset': item['relative_offset'], + 'byte_length': item['byte_length'], + 'payload_sha256': item['payload_sha256'], + 'field_name_set_sha256': _projection_field_name_set_sha256(item['value']), + } for item in parsed_values] + if values is None: + issue = _record_projection_reconciliation_issue( + ledger, + identity_key, + plan[file_index], + offset, + raw_line, + classification, + ) + _save_projection_reconciliation_progress( + ledger, prefix, file_index, offset, indexed_rows, + ) + ledger.commit() + if issue[1] != 'resolved': + raise JsonlProjectionReconciliationRequired( + 'projection reconciliation issue requires explicit review: ' + f'id={issue[0]} file={os.path.basename(candidate)} offset={offset} ' + f'length={len(raw_line)} sha256={hashlib.sha256(raw_line).hexdigest()} ' + f'classification={classification}' + ) + offset += len(raw_line) + batch_rows += 1 + batch_bytes += len(raw_line) + if ( + batch_rows >= max_rows + or batch_bytes >= max_bytes + or time.monotonic() - started >= max_seconds + ): + break + continue + for value in values: + identity = value['identity'] + if not identity: + continue + if len(identity) > 256 or any(character in identity for character in ('\x00', '\r', '\n')): + raise JsonlProjectionReconciliationRequired( + f'invalid {identity_key} in existing projection history: {candidate}:{offset}' + ) + digest = value['payload_sha256'] + existing = ledger.execute( + '''SELECT payload_sha256, state FROM publication_identity + WHERE identity_key = ? AND identity_value = ?''', + (identity_key, identity), + ).fetchone() + variant = ledger.execute( + '''SELECT 1 FROM publication_identity_variant + WHERE identity_key = ? AND identity_value = ? AND payload_sha256 = ?''', + (identity_key, identity, digest), + ).fetchone() + if variant: + continue + if existing: + if existing[1] != 'appended': + raise JsonlProjectionReconciliationRequired( + 'publication ledger retained an unresolved historical identity state' + ) + variant_offset = offset + int(value['relative_offset']) + issue = _record_projection_variant_issue( + ledger, + identity_key, + identity, + plan[file_index], + variant_offset, + int(value['byte_length']), + digest, + value['field_name_set_sha256'], + ) + _save_projection_reconciliation_progress( + ledger, prefix, file_index, offset, indexed_rows, + ) + ledger.commit() + raise JsonlProjectionReconciliationRequired( + 'historical projection payload variant requires explicit review: ' + f'id={issue[0]} file={os.path.basename(candidate)} ' + f'offset={variant_offset} length={int(value["byte_length"])} ' + f'identity_sha256={hashlib.sha256(identity.encode("utf-8")).hexdigest()} ' + f'payload_sha256={digest}' + ) + if not existing: + if indexed_rows >= row_limit: + raise JsonlProjectionReconciliationRequired( + f'existing JSONL identities exceed the ledger row bound: {row_limit}' + ) + now = time.time() + ledger.execute( + '''INSERT INTO publication_identity ( + identity_key, identity_value, payload_sha256, state, file_name, + byte_offset, byte_length, created_at, updated_at + ) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''', + ( + identity_key, identity, digest, os.path.basename(candidate), + offset + int(value['relative_offset']), int(value['byte_length']), now, now, + ), + ) + ledger.execute( + '''INSERT INTO publication_identity_variant ( + identity_key, identity_value, payload_sha256, file_name, + byte_offset, byte_length, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?)''', + ( + identity_key, identity, digest, os.path.basename(candidate), + offset + int(value['relative_offset']), int(value['byte_length']), now, + ), + ) + indexed_rows += 1 + offset += len(raw_line) + batch_rows += 1 + batch_bytes += len(raw_line) + if ( + batch_rows >= max_rows + or batch_bytes >= max_bytes + or time.monotonic() - started >= max_seconds + ): + break + if batch_rows and ( + batch_rows >= max_rows + or batch_bytes >= max_bytes + or time.monotonic() - started >= max_seconds + ): + break + _save_projection_reconciliation_progress(ledger, prefix, file_index, offset, indexed_rows) + complete = file_index >= len(plan) + if complete: + ledger.execute('INSERT OR REPLACE INTO publication_meta(key, value) VALUES (?, ?)', (marker, '1')) + if os.path.getsize(jsonl_ledger_path(path)) > ledger_byte_limit: + raise JsonlProjectionReconciliationRequired( + f'JSONL identity ledger exceeds its {ledger_byte_limit} byte bound' + ) + ledger.commit() + return { + 'path': path, + 'identity_key': identity_key, + 'complete': complete, + 'batch_rows': batch_rows, + 'batch_bytes': batch_bytes, + 'indexed_rows': indexed_rows, + 'file_index': file_index, + 'file_count': len(plan), + 'byte_offset': offset, + } + except BaseException: + if ledger is not None: + ledger.rollback() + raise + finally: + if ledger is not None: + ledger.close() + if os.path.exists(jsonl_ledger_path(path)): + harden_private_file(jsonl_ledger_path(path)) + release_file_lock(lock, lock_path) + + +def _review_projection_issue_record( + plan_entry, + identity_key, + byte_offset, + expected_sha256, + max_record_bytes, + expected_length=None, + expected_classification=None, +): + if byte_offset >= int(plan_entry['size']): + raise JsonlProjectionReconciliationRequired('projection issue offset is outside the immutable file') + with open(plan_entry['path'], 'rb') as handle: + if byte_offset: + handle.seek(byte_offset - 1) + if handle.read(1) != b'\n': + raise JsonlProjectionReconciliationRequired('projection issue offset is not a record boundary') + handle.seek(byte_offset) + raw_line = handle.readline(max_record_bytes + 1) + if not raw_line: + raise JsonlProjectionReconciliationRequired('projection issue record is absent') + if len(raw_line) > max_record_bytes: + byte_length, actual_sha256, newline_terminated = _stream_projection_record(handle, raw_line) + classification = 'oversized_record' if newline_terminated else 'unterminated_record' + raw_line = None + else: + byte_length = len(raw_line) + actual_sha256 = hashlib.sha256(raw_line).hexdigest() + if identity_key == 'error_row_id': + try: + identity = _projection_identity_from_line(raw_line, identity_key, plan_entry['path']) + if len(identity) > 256 or any(character in identity for character in ('\x00', '\r', '\n')): + raise ValueError('error projection identity exceeds its bounded format') + except (JsonlProjectionReconciliationRequired, ValueError): + classification = ( + 'unterminated_record' if not raw_line.endswith(b'\n') + else 'invalid_error_projection' + ) + else: + raise JsonlProjectionReconciliationRequired( + 'reviewed projection issue is a parseable error record and must be indexed' + ) + else: + parsed_values, classification = _bounded_projection_json_values(raw_line) + if parsed_values is not None: + raise JsonlProjectionReconciliationRequired( + 'reviewed projection issue contains bounded parseable JSON and must be indexed' + ) + if actual_sha256 != expected_sha256: + raise JsonlProjectionReconciliationRequired('projection issue record SHA-256 does not match review') + if expected_length is not None and byte_length != int(expected_length): + raise JsonlProjectionReconciliationRequired('projection issue record length does not match review') + if expected_classification is not None and classification != str(expected_classification): + raise JsonlProjectionReconciliationRequired('projection issue classification does not match review') + return raw_line, byte_length, actual_sha256, classification + + +def approve_projection_reconciliation_issues( + path, identity_key, reviewed_issues, max_record_bytes=None, return_details=False, +): + """Approve exact reviewed corrupt records without storing or changing their payloads.""" + if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'): + raise ValueError('unsupported projection reconciliation identity') + reviewed_issues = list(reviewed_issues or []) + if not reviewed_issues or len(reviewed_issues) > 100000: + raise ValueError('projection issue review count is outside its bound') + path = os.path.abspath(path) + max_record_bytes = max( + 1024 * 1024, + int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)), + ) + lock_path = f'{path}.lock' + lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) + ledger = None + try: + plan = _projection_reconciliation_plan(path) + plan_by_name = {os.path.basename(item['path']): (index, item) for index, item in enumerate(plan)} + if len(plan_by_name) != len(plan): + raise JsonlProjectionReconciliationRequired('projection history contains ambiguous basenames') + normalized = [] + seen = set() + for value in reviewed_issues: + requested_file = str(value.get('file') or value.get('basename') or '') + physical_file = os.path.basename(requested_file) + byte_offset = int(value.get('offset', value.get('byte_offset', -1))) + expected_sha256 = str(value.get('sha256') or '').strip().lower() + if ( + not physical_file or physical_file != requested_file + or physical_file not in plan_by_name + or byte_offset < 0 + or not re.fullmatch(r'[a-f0-9]{64}', expected_sha256) + ): + raise ValueError('projection issue review metadata is invalid') + identity = (physical_file, byte_offset, expected_sha256) + if identity in seen: + raise ValueError('projection issue review contains duplicate metadata') + seen.add(identity) + normalized.append(( + plan_by_name[physical_file][0], + physical_file, + byte_offset, + expected_sha256, + value.get('length', value.get('byte_length')), + value.get('classification'), + )) + ledger = _open_projection_ledger(path) + details = [] + classifications = Counter() + for sequence, (_, physical_file, byte_offset, expected_sha256, expected_length, expected_classification) in enumerate(sorted(normalized), 1): + plan_entry = plan_by_name[physical_file][1] + raw_line, byte_length, actual_sha256, classification = _review_projection_issue_record( + plan_entry, + identity_key, + byte_offset, + expected_sha256, + max_record_bytes, + expected_length=expected_length, + expected_classification=expected_classification, + ) + issue = _record_projection_reconciliation_issue( + ledger, + identity_key, + plan_entry, + byte_offset, + raw_line, + classification, + resolved=True, + byte_length=byte_length, + record_sha256=actual_sha256, + ) + classifications[classification] += 1 + if return_details: + details.append({ + 'id': int(issue[0]), 'status': issue[1], 'file': physical_file, + 'offset': byte_offset, 'length': byte_length, + 'sha256': actual_sha256, 'classification': classification, + }) + if sequence % 100 == 0: + ledger.commit() + ledger.commit() + report = { + 'resolved_count': len(normalized), + 'classifications': dict(sorted(classifications.items())), + } + if return_details: + report['details'] = details + return report + finally: + if ledger is not None: + ledger.close() + harden_private_file(jsonl_ledger_path(path)) + release_file_lock(lock, lock_path) + + +def approve_projection_reconciliation_issue( + path, identity_key, physical_file, byte_offset, expected_sha256, max_record_bytes=None, +): + report = approve_projection_reconciliation_issues( + path, + identity_key, + [{ + 'file': physical_file, + 'offset': byte_offset, + 'sha256': expected_sha256, + }], + max_record_bytes=max_record_bytes, + return_details=True, + ) + return report['details'][0] + + +def _apply_reviewed_projection_conflict_variant( + ledger, + handle, + plan_entry, + identity_key, + reviewed, + max_record_bytes, +): + file_name, offset, length, identity_sha256, payload_sha256, field_set_sha256 = reviewed + if offset >= int(plan_entry['size']): + raise JsonlProjectionReconciliationRequired('projection conflict offset is outside the immutable file') + if offset: + handle.seek(offset - 1) + if handle.read(1) != b'\n': + raise JsonlProjectionReconciliationRequired( + 'projection conflict offset is not a record boundary' + ) + handle.seek(offset) + raw_line = handle.readline(max_record_bytes + 1) + if not raw_line or len(raw_line) > max_record_bytes or len(raw_line) != length: + raise JsonlProjectionReconciliationRequired( + 'projection conflict record is absent or does not match its reviewed length' + ) + actual_payload_sha256 = hashlib.sha256(raw_line).hexdigest() + if identity_key == 'error_row_id': + identity = _projection_identity_from_line(raw_line, identity_key, plan_entry['path']) + actual_field_set_sha256 = _error_projection_field_name_set_sha256(raw_line) + else: + parsed_values, _ = _bounded_projection_json_values(raw_line) + if not parsed_values or len(parsed_values) != 1 or not isinstance(parsed_values[0]['value'], dict): + raise JsonlProjectionReconciliationRequired( + 'projection conflict record is not one bounded JSON identity record' + ) + identity = str(parsed_values[0]['value'].get(identity_key) or '') + actual_field_set_sha256 = _projection_field_name_set_sha256(parsed_values[0]['value']) + if ( + not identity + or hashlib.sha256(identity.encode('utf-8')).hexdigest() != identity_sha256 + or actual_payload_sha256 != payload_sha256 + or actual_field_set_sha256 != field_set_sha256 + ): + raise JsonlProjectionReconciliationRequired( + 'projection conflict record does not match reviewed identity/payload/schema hashes' + ) + primary = ledger.execute( + '''SELECT state FROM publication_identity + WHERE identity_key = ? AND identity_value = ?''', + (identity_key, identity), + ).fetchone() + if primary and primary[0] != 'appended': + raise JsonlProjectionReconciliationRequired( + 'projection conflict review primary identity is unresolved' + ) + if not primary: + row_count = _ledger_row_count(ledger) + row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000))) + if row_count >= row_limit: + raise JsonlProjectionReconciliationRequired( + f'JSONL identity ledger reached its {row_limit} row bound' + ) + now = time.time() + ledger.execute( + '''INSERT INTO publication_identity ( + identity_key, identity_value, payload_sha256, state, file_name, + byte_offset, byte_length, created_at, updated_at + ) VALUES (?, ?, ?, 'appended', ?, ?, ?, ?, ?)''', + ( + identity_key, identity, payload_sha256, file_name, + offset, length, now, now, + ), + ) + _set_ledger_row_count(ledger, row_count + 1) + ledger.execute( + '''INSERT OR IGNORE INTO publication_identity_variant ( + identity_key, identity_value, payload_sha256, file_name, + byte_offset, byte_length, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?)''', + (identity_key, identity, payload_sha256, file_name, offset, length, time.time()), + ) + _record_projection_variant_issue( + ledger, + identity_key, + identity, + plan_entry, + offset, + length, + payload_sha256, + field_set_sha256, + resolved=True, + ) + + +def approve_projection_conflict_variants( + path, + identity_key, + reviewed_variants, + max_record_bytes=None, + commit_batch_size=250, + progress_callback=None, +): + if identity_key not in ('scan_event_id', 'finding_uid', 'error_row_id'): + raise ValueError('unsupported projection conflict identity') + reviewed_variants = list(reviewed_variants or []) + if not reviewed_variants or len(reviewed_variants) > 100000: + raise ValueError('projection conflict review count is outside its bound') + path = os.path.abspath(path) + max_record_bytes = max( + 1024 * 1024, + int(max_record_bytes or getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024)), + ) + commit_batch_size = min(1000, max(1, int(commit_batch_size or 250))) + lock_path = f'{path}.lock' + lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) + ledger = None + try: + plan = _projection_reconciliation_plan(path) + plan_by_name = {os.path.basename(item['path']): (index, item) for index, item in enumerate(plan)} + if len(plan_by_name) != len(plan): + raise JsonlProjectionReconciliationRequired('projection history contains ambiguous basenames') + normalized = [] + seen = set() + for value in reviewed_variants: + file_name = str(value.get('file') or '') + offset = int(value.get('offset', -1)) + length = int(value.get('length', 0)) + identity_sha256 = str(value.get('identity_sha256') or '').lower() + payload_sha256 = str(value.get('payload_sha256') or '').lower() + field_set_sha256 = str(value.get('field_name_set_sha256') or '').lower() + if ( + not file_name or os.path.basename(file_name) != file_name + or file_name not in plan_by_name or offset < 0 or length <= 0 + or not re.fullmatch(r'[a-f0-9]{64}', identity_sha256) + or not re.fullmatch(r'[a-f0-9]{64}', payload_sha256) + or not re.fullmatch(r'[a-f0-9]{64}', field_set_sha256) + or value.get('classification') != 'historical_payload_variant' + ): + raise ValueError('projection conflict review metadata is invalid') + key = (file_name, offset, identity_sha256, payload_sha256) + if key in seen: + raise ValueError('projection conflict review contains duplicate metadata') + seen.add(key) + normalized.append(( + plan_by_name[file_name][0], file_name, + (file_name, offset, length, identity_sha256, payload_sha256, field_set_sha256), + )) + ledger = _open_projection_ledger(path) + resolved_rows = ledger.execute( + '''SELECT identity_sha256, file_name, byte_offset, byte_length, + payload_sha256, field_name_set_sha256 + FROM reconciliation_variant_issue + WHERE identity_key = ? AND status = 'resolved' ''', + (identity_key,), + ).fetchall() + resolved = { + (row[1], int(row[2]), row[0], row[4]): (int(row[3]), row[5]) + for row in resolved_rows + } + pending = [] + already_resolved = 0 + for plan_index, file_name, reviewed in sorted(normalized): + key = (file_name, reviewed[1], reviewed[3], reviewed[4]) + expected = resolved.get(key) + if expected is not None: + if expected != (reviewed[2], reviewed[5]): + raise JsonlProjectionReconciliationRequired( + 'resolved projection conflict metadata does not match this review manifest' + ) + already_resolved += 1 + continue + pending.append((plan_index, file_name, reviewed)) + if progress_callback is not None: + progress_callback({ + 'reviewed': len(normalized), + 'already_resolved': already_resolved, + 'newly_resolved': 0, + }) + newly_resolved = 0 + grouped = {} + for plan_index, file_name, reviewed in pending: + grouped.setdefault((plan_index, file_name), []).append(reviewed) + for plan_index, file_name in sorted(grouped): + plan_entry = plan_by_name[file_name][1] + with open(plan_entry['path'], 'rb') as handle: + for reviewed in sorted(grouped[(plan_index, file_name)], key=lambda item: item[1]): + _apply_reviewed_projection_conflict_variant( + ledger, handle, plan_entry, identity_key, reviewed, max_record_bytes, + ) + newly_resolved += 1 + if newly_resolved % commit_batch_size == 0: + ledger.commit() + if progress_callback is not None: + progress_callback({ + 'reviewed': len(normalized), + 'already_resolved': already_resolved, + 'newly_resolved': newly_resolved, + }) + ledger.commit() + if progress_callback is not None and newly_resolved % commit_batch_size: + progress_callback({ + 'reviewed': len(normalized), + 'already_resolved': already_resolved, + 'newly_resolved': newly_resolved, + }) + return { + 'resolved_variant_count': len(normalized), + 'already_resolved': already_resolved, + 'newly_resolved': newly_resolved, + } + finally: + if ledger is not None: + ledger.close() + harden_private_file(jsonl_ledger_path(path)) + release_file_lock(lock, lock_path) + + +def _files_share_prefix(segment_path, current_path, size): + if size <= 0 or not os.path.isfile(current_path) or os.path.getsize(current_path) < size: + return False + remaining = size + with open(segment_path, 'rb') as segment, open(current_path, 'rb') as current: + while remaining: + amount = min(1024 * 1024, remaining) + left = segment.read(amount) + right = current.read(amount) + if left != right or not left: + return False + remaining -= len(left) + return remaining == 0 + + +def _publish_empty_jsonl_generation(path): + temporary = ( + f'{path}.{os.getpid()}.{threading.get_ident()}.' + f'{uuid.uuid4().hex}.empty.tmp' + ) + try: + with open(temporary, 'xb') as handle: + handle.flush() + os.fsync(handle.fileno()) + harden_private_file(temporary) + durable_replace(temporary, path) + harden_private_file(path) + finally: + if os.path.exists(temporary): + os.remove(temporary) + + +def _remove_file_prefix(path, size): + current_size = os.path.getsize(path) + if size <= 0 or current_size < size: + return + if current_size == size: + _publish_empty_jsonl_generation(path) + return + temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.prefix.tmp' + with open(path, 'rb') as source, open(temporary, 'wb') as destination: + source.seek(size) + shutil.copyfileobj(source, destination, 1024 * 1024) + destination.flush() + os.fsync(destination.fileno()) + harden_private_file(temporary) + durable_replace(temporary, path) + harden_private_file(path) + + +def _segment_manifest_entry(path, sequence): + return { + 'name': os.path.basename(path), + 'path': os.path.abspath(path), + 'bytes': int(os.path.getsize(path)), + 'closed_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), + 'sequence': int(sequence), + } + + +def _jsonl_file_generation(path): + details = os.stat(path, follow_symlinks=False) + return { + 'device': int(getattr(details, 'st_dev', 0) or 0), + 'size': int(details.st_size), + 'mtime_ns': int(getattr(details, 'st_mtime_ns', int(details.st_mtime * 1_000_000_000))), + 'inode': int(getattr(details, 'st_ino', 0) or 0), + } + + +def _manifest_skip_matches_generation(path, manifest, skip_bytes): + signature = manifest.get('current_skip_signature') if isinstance(manifest, dict) else None + if not isinstance(signature, dict) or not os.path.isfile(path): + return False + try: + current = _jsonl_file_generation(path) + return ( + int(skip_bytes) <= current['size'] + and all(int(current[key]) == int(signature.get(key, -1)) for key in ('device', 'size', 'mtime_ns', 'inode')) + ) + except (OSError, TypeError, ValueError): + return False + + +def reconcile_jsonl_segments(path): + """Publish physical orphan segments before any active-file mutation.""" + manifest = load_jsonl_manifest(path) + physical = _projection_segment_sequence(path) + existing = { + str(item.get('name') or os.path.basename(str(item.get('path') or ''))): item + for item in (manifest.get('segments') or []) if isinstance(item, dict) + } + normalized = [] + missing = [] + for sequence, segment_path in physical: + item = existing.get(os.path.basename(segment_path)) + if item is None: + item = _segment_manifest_entry(segment_path, sequence) + missing.append((sequence, segment_path)) + else: + item = dict(item, path=os.path.abspath(segment_path), name=os.path.basename(segment_path), sequence=sequence) + normalized.append(item) + + skip_bytes = max(0, int(manifest.get('current_skip_bytes') or 0)) + duplicate_path = None + duplicate_size = 0 + candidates = list(reversed(missing)) + if skip_bytes and physical: + candidates.insert(0, physical[-1]) + for _, segment_path in candidates: + size = os.path.getsize(segment_path) + if _files_share_prefix(segment_path, path, size): + duplicate_path = segment_path + duplicate_size = size + break + + changed = bool(missing) or normalized != (manifest.get('segments') or []) + effective_skip = duplicate_size if duplicate_path else 0 + if changed or skip_bytes or manifest.get('current_skip_signature'): + next_sequence = max([sequence for sequence, _ in physical] or [0]) + 1 + manifest.update({ + 'current': os.path.basename(path), + 'current_path': os.path.abspath(path), + 'next_sequence': next_sequence, + 'segments': normalized, + 'current_skip_bytes': effective_skip, + 'current_skip_signature': _jsonl_file_generation(path) if effective_skip and os.path.isfile(path) else None, + 'updated_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), + }) + write_jsonl_manifest(path, manifest) + if duplicate_path: + _remove_file_prefix(path, duplicate_size) + manifest['current_skip_bytes'] = 0 + manifest['current_skip_signature'] = None + manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds') + write_jsonl_manifest(path, manifest) + return manifest + + +def _segment_consumed_by_registered_keychecks(segment_path): + if os.path.basename(segment_path).lower().startswith('scan_results.'): + return True + keycheck_root = getattr(scan_config, 'keycheck_dir', '') or '' + if not os.path.isdir(keycheck_root): + return False + states = [] + consumer_directories = 0 + with os.scandir(keycheck_root) as entries: + for entry in entries: + if not entry.is_dir(follow_symlinks=False): + continue + consumer_directories += 1 + if consumer_directories > 128: + return False + state_path = os.path.join(entry.path, 'input_state.json') + if not os.path.isfile(state_path): + return False + try: + if os.path.getsize(state_path) > 1024 * 1024: + return False + with open(state_path, 'r', encoding='utf-8') as handle: + states.append(json.load(handle)) + except (OSError, ValueError): + return False + if not states or len(states) != consumer_directories: + return False + absolute = os.path.abspath(segment_path) + size = os.path.getsize(segment_path) + for state in states: + files = state.get('files') if isinstance(state, dict) and isinstance(state.get('files'), dict) else {} + record = files.get(absolute) or files.get(segment_path) + if not isinstance(record, dict) or int(record.get('offset', 0) or 0) < size: + return False + return True + + +def _prune_consumed_jsonl_segments(path, manifest, keep_below): + physical = _projection_segment_sequence(path) + removed = set() + while len(physical) >= keep_below and physical: + _, candidate = physical[0] + if not _segment_consumed_by_registered_keychecks(candidate): + break + reject_reparse_components(candidate) + os.remove(candidate) + removed.add(os.path.basename(candidate)) + physical.pop(0) + if removed: + manifest['segments'] = [ + item for item in (manifest.get('segments') or []) + if str(item.get('name') or os.path.basename(str(item.get('path') or ''))) not in removed + ] + manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds') + write_jsonl_manifest(path, manifest) + return physical + + +def rotate_jsonl_if_needed(path, max_bytes, ledger=None): + if not max_bytes or max_bytes <= 0: + return + if not os.path.lexists(path): + return + reject_reparse_components(path) + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or details.st_size < max_bytes: + return + repair_jsonl_tail(path) + manifest = reconcile_jsonl_segments(path) + max_segments = max(1, int(getattr(scan_config, 'jsonl_max_segments', 16))) + physical = _prune_consumed_jsonl_segments(path, manifest, max_segments) + if len(physical) >= max_segments: + raise JsonlProjectionReconciliationRequired( + f'JSONL segment retention bound reached for {path}; registered consumers must catch up ' + 'or offline reconciliation must retire acknowledged segments' + ) + segment_path, seq = next_jsonl_segment_path(path, manifest) + size = os.path.getsize(path) + temporary = f'{segment_path}.{os.getpid()}.{threading.get_ident()}.tmp' + try: + with open(path, 'rb') as source, open(temporary, 'xb') as destination: + shutil.copyfileobj(source, destination, 1024 * 1024) + destination.flush() + os.fsync(destination.fileno()) + harden_private_file(temporary) + os.replace(temporary, segment_path) + finally: + if os.path.exists(temporary): + os.remove(temporary) + harden_private_file(segment_path) + segments = manifest.get('segments') if isinstance(manifest.get('segments'), list) else [] + segments.append(_segment_manifest_entry(segment_path, seq)) + manifest.update({ + 'current': os.path.basename(path), + 'current_path': path, + 'next_sequence': seq + 1, + 'max_bytes': int(max_bytes), + 'segments': segments, + 'current_skip_bytes': int(size), + 'current_skip_signature': _jsonl_file_generation(path), + 'updated_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), + }) + write_jsonl_manifest(path, manifest) + if ledger is not None: + ledger.execute( + "UPDATE publication_identity SET file_name = ? WHERE file_name = ? AND state = 'appended'", + (os.path.basename(segment_path), os.path.basename(path)), + ) + ledger.execute( + "UPDATE publication_identity_variant SET file_name = ? WHERE file_name = ?", + (os.path.basename(segment_path), os.path.basename(path)), + ) + ledger.commit() + _publish_empty_jsonl_generation(path) + manifest['current_skip_bytes'] = 0 + manifest['current_skip_signature'] = None + manifest['updated_at'] = datetime.now(timezone.utc).isoformat(timespec='seconds') + write_jsonl_manifest(path, manifest) + logger.info(f'Rotated JSONL {path} -> {segment_path} ({size} bytes)') + + +def append_rotating_jsonl(path, payload, max_mb=None): + if not scan_config.jsonl_rotation_enabled: + return append_jsonl(path, payload) + max_bytes = int(max_mb or 0) * 1024 * 1024 + serialized = json.dumps(payload, ensure_ascii=False, default=str) + '\n' + lock_path = f'{path}.lock' + fd = None + try: + parent = os.path.dirname(path) + if parent: + require_private_directory(parent, create=True) + fd = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) + repair_jsonl_tail(path) + rotate_jsonl_if_needed(path, max_bytes) + with open(path, 'ab') as f: + f.write(serialized.encode('utf-8')) + f.flush() + os.fsync(f.fileno()) + harden_private_file(path) + return True + except Exception as e: + logger.error(f'Unable to rotating-write {path}: {str(e)}') + return False + finally: + if fd is not None: + release_file_lock(fd, lock_path) + + +def projection_segment_paths(path): + paths = [segment_path for _, segment_path in _projection_segment_sequence(path)] + if os.path.isfile(path): + paths.append(path) + return paths + + +def jsonl_projection_contains(path, identity_key, identity_value): + expected = str(identity_value or '') + if not expected: + return False + ledger_path = jsonl_ledger_path(path) + if not os.path.isfile(ledger_path) or not private_file_ready(ledger_path): + return False + connection = sqlite3.connect(f'file:{ledger_path}?mode=ro', uri=True, timeout=5) + try: + row = connection.execute( + '''SELECT state FROM publication_identity + WHERE identity_key = ? AND identity_value = ?''', + (identity_key, expected), + ).fetchone() + return bool(row and row[0] == 'appended') + finally: + connection.close() + + +def _append_projection_once_locked( + path, serialized, identity_key, identity_value, max_bytes, rotation_enabled=None, +): + repair_jsonl_tail(path) + reconcile_jsonl_segments(path) + if len(identity_key) > 64 or len(identity_value) > 256 or any( + character in identity_value for character in ('\x00', '\r', '\n') + ): + raise JsonlProjectionReconciliationRequired('projection identity exceeds its bounded format') + max_record_bytes = max(1, int(getattr(scan_config, 'result_spool_max_event_bytes', 192 * 1024 * 1024))) + if len(serialized) > max_record_bytes: + raise JsonlProjectionReconciliationRequired('projection record exceeds its per-record byte bound') + ledger = _open_projection_ledger(path) + try: + _ensure_projection_ledger_bootstrapped(ledger, path, identity_key) + _recover_prepared_publications(ledger, path) + digest = hashlib.sha256(serialized).hexdigest() + row = ledger.execute( + '''SELECT payload_sha256, state FROM publication_identity + WHERE identity_key = ? AND identity_value = ?''', + (identity_key, identity_value), + ).fetchone() + if row: + if row[1] != 'appended': + raise JsonlProjectionReconciliationRequired('publication ledger retained an unresolved append state') + variant = ledger.execute( + '''SELECT 1 FROM publication_identity_variant + WHERE identity_key = ? AND identity_value = ? AND payload_sha256 = ?''', + (identity_key, identity_value, digest), + ).fetchone() + if variant: + return True + raise JsonlProjectionReconciliationRequired( + f'{identity_key} {identity_value} has a conflicting projection payload' + ) + + row_limit = max(1, int(getattr(scan_config, 'jsonl_ledger_max_rows', 1000000))) + ledger_byte_limit = max(1024 * 1024, int(getattr(scan_config, 'jsonl_ledger_max_bytes', 512 * 1024 * 1024))) + row_count = _ledger_row_count(ledger) + if row_count >= row_limit: + raise JsonlProjectionReconciliationRequired( + f'JSONL identity ledger reached its {row_limit} row bound; run offline reconciliation' + ) + ledger_file = jsonl_ledger_path(path) + if os.path.getsize(ledger_file) + 8192 > ledger_byte_limit: + raise JsonlProjectionReconciliationRequired( + f'JSONL identity ledger reached its {ledger_byte_limit} byte bound; run offline reconciliation' + ) + should_rotate = scan_config.jsonl_rotation_enabled if rotation_enabled is None else bool(rotation_enabled) + if should_rotate: + rotate_jsonl_if_needed(path, max_bytes, ledger=ledger) + if not os.path.exists(path): + with open(path, 'ab'): + pass + harden_private_file(path) + offset = os.path.getsize(path) + now = time.time() + ledger.execute( + '''INSERT INTO publication_identity ( + identity_key, identity_value, payload_sha256, state, file_name, + byte_offset, byte_length, created_at, updated_at + ) VALUES (?, ?, ?, 'prepared', ?, ?, ?, ?, ?)''', + ( + identity_key, identity_value, digest, os.path.basename(path), + offset, len(serialized), now, now, + ), + ) + _set_ledger_row_count(ledger, row_count + 1) + ledger.commit() + with open(path, 'ab') as handle: + handle.write(serialized) + handle.flush() + os.fsync(handle.fileno()) + harden_private_file(path) + ledger.execute( + '''UPDATE publication_identity SET state = 'appended', updated_at = ? + WHERE identity_key = ? AND identity_value = ? AND state = 'prepared' ''', + (time.time(), identity_key, identity_value), + ) + ledger.execute( + '''INSERT INTO publication_identity_variant ( + identity_key, identity_value, payload_sha256, file_name, + byte_offset, byte_length, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?)''', + ( + identity_key, identity_value, digest, os.path.basename(path), + offset, len(serialized), time.time(), + ), + ) + ledger.commit() + return True + finally: + ledger.close() + if os.path.exists(jsonl_ledger_path(path)): + harden_private_file(jsonl_ledger_path(path)) + + +def append_rotating_jsonl_once(path, payload, identity_key, identity_value, max_mb=None): + if not identity_value: + logger.error(f'Unable to publish {path}: missing {identity_key}') + return False + max_bytes = int(max_mb or 0) * 1024 * 1024 + lock_path = f'{path}.lock' + lock = None + try: + require_private_directory(os.path.dirname(os.path.abspath(path)), create=True) + lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) + serialized = (json.dumps(payload, ensure_ascii=False, default=str) + '\n').encode('utf-8') + return _append_projection_once_locked( + path, serialized, identity_key, str(identity_value), max_bytes, + ) + except Exception as exc: + logger.error(f'Unable to idempotently publish {path}: {exc}') + return False + finally: + if lock is not None: + release_file_lock(lock, lock_path) + + +def _rotate_bounded_text_log(path, keep, max_bytes=None): + if not os.path.exists(path) or os.path.getsize(path) <= 0: + return + if max_bytes and os.path.getsize(path) > int(max_bytes): + with open(path, 'r+b') as handle: + handle.truncate(int(max_bytes)) + handle.flush() + os.fsync(handle.fileno()) + segment, _ = next_jsonl_segment_path(path, {}) + os.replace(path, segment) + harden_private_file(segment) + with open(path, 'ab'): + pass + harden_private_file(path) + segments = [candidate for candidate in projection_segment_paths(path) if candidate != path] + segments.sort(key=lambda candidate: os.path.getmtime(candidate), reverse=True) + for candidate in segments[max(0, int(keep or 0)):]: + reject_reparse_components(candidate) + os.remove(candidate) + + +def append_scan_errors_once(path, result, max_mb=None, keep=5): + event_id = str(result.get('scan_event_id') or '') + if not event_id: + logger.error(f'Unable to publish {path}: missing scan_event_id') + return False + errors = list(result.get('errors') or []) + if not errors: + return True + max_bytes = max(1, int(max_mb or 0) * 1024 * 1024) + timestamp = result.get('timestamp') or result.get('scan_started_at') or event_id + scan_type = result.get('scan_type', '') + target = result.get('target', '') + lock_path = f'{path}.lock' + lock = None + try: + require_private_directory(os.path.dirname(os.path.abspath(path)), create=True) + lock = acquire_file_lock(lock_path, scan_config.jsonl_lock_stale_sec) + repair_jsonl_tail(path) + for index, error in enumerate(errors, 1): + row_id = f'{event_id}:{index}' + line = f'{row_id}\t{timestamp}\t{scan_type}\t{target}\t{error}\n'.encode('utf-8', errors='replace') + if len(line) > max_bytes: + suffix = b'...[truncated]\n' + line = line[:max(0, max_bytes - len(suffix))] + suffix + if os.path.exists(path) and os.path.getsize(path) + len(line) > max_bytes: + _rotate_bounded_text_log(path, keep, max_bytes) + _append_projection_once_locked( + path, line, 'error_row_id', row_id, 0, rotation_enabled=False, + ) + return True + except Exception as exc: + logger.error(f'Unable to idempotently publish {path}: {exc}') + return False + finally: + if lock is not None: + release_file_lock(lock, lock_path) + +def parse_finding_datetime(value): + if not value: + return None + value = str(value).strip() + for date_format in ("%Y-%m-%d %H:%M:%S %z", "%Y-%m-%dT%H:%M:%SZ", "%Y-%m-%dT%H:%M:%S%z"): + try: + return datetime.strptime(value, date_format) + except ValueError: + continue + return None + +def get_finding_timestamp(finding): + data = finding.get('SourceMetadata', {}).get('Data', {}) + if not isinstance(data, dict): + return None + for source in data.values(): + if isinstance(source, dict) and source.get('timestamp'): + return parse_finding_datetime(source.get('timestamp')) + return None + +def filter_findings_by_age(findings, max_age_days=None): + if not max_age_days or max_age_days <= 0: + return findings, 0, 0 + + cutoff = datetime.now().astimezone() - timedelta(days=max_age_days) + kept = [] + skipped_old = 0 + skipped_missing = 0 + + for finding in findings: + timestamp = get_finding_timestamp(finding) + if timestamp is None: + skipped_missing += 1 + continue + if timestamp >= cutoff: + kept.append(finding) + else: + skipped_old += 1 + + return kept, skipped_old, skipped_missing + + +def filter_dropped_detectors(findings): + drop = { + item.strip().lower() + for item in csv_items(_scan_policy_value('drop_detectors', [])) + if item.strip() + } + if not drop: + return findings, 0, Counter() + kept = [] + counts = Counter() + skipped = 0 + for finding in findings or []: + detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '') + if detector.lower() in drop: + skipped += 1 + counts[detector or '(unknown)'] += 1 + continue + kept.append(finding) + return kept, skipped, counts + +GITHUB_TOKEN_PREFIXES = ('ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_') +GITLAB_TOKEN_PREFIXES = ('glpat-', 'gloas-', 'glcbt-', 'glimt-', 'glrt-', 'glft-', 'glsoat-') + +def finding_raw_values(finding): + values = [] + for key in ('Raw', 'RawV2'): + value = finding.get(key) + if value: + values.append(str(value)) + return values + +def is_known_provider_token_shape(detector_name, finding): + values = finding_raw_values(finding) + detector = str(detector_name or '').lower() + if detector in ('github', 'githuboauth2'): + return any(value.startswith(GITHUB_TOKEN_PREFIXES) for value in values) + if detector == 'gitlab': + return any(value.startswith(GITLAB_TOKEN_PREFIXES) for value in values) + return True + +def filter_noisy_findings(findings): + if not _scan_policy_value('strict_git_provider_token_filter', True): + return findings, 0 + + kept = [] + skipped = 0 + for finding in findings: + detector = finding.get('DetectorName') or '' + detector_key = str(detector).lower() + if detector_key in ('github', 'githuboauth2', 'gitlab') and not finding.get('Verified', False): + if not is_known_provider_token_shape(detector, finding): + skipped += 1 + continue + kept.append(finding) + return kept, skipped + +CUSTOM_DETECTOR_NAME_ALIASES = { + 'xaicontextafter': 'Xai', + 'zaiglmcontextafter': 'ZaiGLM', +} + + +def normalize_custom_detector_names(findings): + for finding in findings or []: + detector = str(finding.get('DetectorName') or '') + extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {} + custom_name = str(extra.get('name') or '').strip() + if detector.lower() == 'customregex' and custom_name: + finding.setdefault('OriginalDetectorName', detector) + finding['DetectorName'] = CUSTOM_DETECTOR_NAME_ALIASES.get( + custom_name.lower(), custom_name, + ) + return findings + +def apply_finding_filters(results, target_label='target', *, log_target=True): + emit_client_scan_phase('filtering') + findings = results.get('findings') or [] + normalize_custom_detector_names(findings) + findings, dropped, dropped_counts = filter_dropped_detectors(findings) + if dropped: + results['findings'] = findings + results['dropped_detectors_count'] = int(results.get('dropped_detectors_count', 0) or 0) + dropped + results['dropped_detectors'] = dict(dropped_counts) + target_context = f' for {target_label}' if log_target else '' + logger.info( + f"Dropped {dropped} configured noise detector finding(s)" + f"{target_context}: {dict(dropped_counts)}" + ) + filtered, skipped = filter_noisy_findings(findings) + if skipped: + results['findings'] = filtered + results['filtered_findings_count'] = int(results.get('filtered_findings_count', 0) or 0) + skipped + target_context = f' for {target_label}' if log_target else '' + logger.info( + f"Filtered {skipped} noisy unverified Git provider finding(s)" + f"{target_context}" + ) + return results + +def finding_source_location(finding): + metadata = finding.get('SourceMetadata') or {} + data = metadata.get('Data') if isinstance(metadata, dict) else {} + if not isinstance(data, dict): + return None, None + for source in data.values(): + if not isinstance(source, dict): + continue + file_path = source.get('file') or source.get('path') or source.get('File') + line = source.get('line') or source.get('Line') + if file_path: + try: + line = int(line) if line else None + except (TypeError, ValueError): + line = None + return file_path, line + return None, None + + +def context_enrichment_budget(): + return { + 'started_at': time.monotonic(), + 'max_source_bytes': max(0, int(getattr(scan_config, 'context_enrichment_max_source_bytes', 16 * 1024 * 1024))), + 'max_findings': max(0, int(getattr(scan_config, 'context_enrichment_max_findings', 2000))), + 'max_postman_comparisons': max(0, int(getattr(scan_config, 'context_enrichment_max_postman_comparisons', 200000))), + 'max_elapsed_sec': max(0.0, float(getattr(scan_config, 'context_enrichment_max_elapsed_sec', 5.0))), + 'source_bytes': 0, + 'postman_comparisons': 0, + 'finding_ids': set(), + 'files': {}, + } + + +def _context_budget_expired(budget): + return time.monotonic() - budget['started_at'] >= budget['max_elapsed_sec'] + + +def _context_budget_claim_finding(budget, finding): + identity = id(finding) + if identity in budget['finding_ids']: + return True + if len(budget['finding_ids']) >= budget['max_findings']: + return False + budget['finding_ids'].add(identity) + return True + + +def _add_context_warning(results, warning_class, detail): + warning_class = f'context_enrichment_{warning_class}' + classes = list(results.get('warning_classes') or []) + if warning_class not in classes: + warnings = list(results.get('warnings') or []) + warnings.append(f'Optional context enrichment degraded: {detail}'[:400]) + results['warnings'] = warnings + classes.append(warning_class) + results['warning_classes'] = sorted(set(classes)) + results['context_enrichment_degraded'] = True + results['degraded'] = True + + +def _read_context_source(file_path, budget): + key = os.path.normcase(os.path.abspath(os.fspath(file_path))) + if key in budget['files']: + return budget['files'][key] + if _context_budget_expired(budget): + return None, False, 'elapsed' + try: + size = max(0, int(os.path.getsize(file_path))) + remaining = max(0, budget['max_source_bytes'] - budget['source_bytes']) + if remaining <= 0 and size: + return None, False, 'source_bytes' + amount = min(size, remaining) + with open(file_path, 'rb') as handle: + payload = handle.read(amount) + budget['source_bytes'] += len(payload) + complete = len(payload) == size + reason = None if complete else 'source_bytes' if size > remaining else 'read' + budget['files'][key] = (payload, complete, reason) + return payload, complete, reason + except (OSError, TypeError, ValueError): + return None, False, 'read' + + +def _nearby_context_from_lines(lines, file_path, line_number=None, radius=20, max_chars=12000): + if not lines: + return None + if line_number and line_number > 0: + start = max(0, line_number - radius - 1) + requested_end = line_number + radius + else: + start = 0 + requested_end = radius * 2 + 1 + if start >= len(lines): + return None + end = min(len(lines), requested_end) + return { + 'file': file_path, + 'line': line_number, + 'start_line': start + 1, + 'end_line': end, + 'nearby': ''.join(lines[start:end])[:max_chars], + } + + +def read_nearby_context(file_path, line_number=None, radius=20, max_chars=12000): + if not file_path or not os.path.exists(file_path): + return None + if line_number and line_number > 0: + start = max(0, line_number - radius - 1) + requested_end = line_number + radius + else: + start = 0 + requested_end = radius * 2 + 1 + selected = [] + actual_end = 0 + try: + with open(file_path, 'r', encoding='utf-8', errors='replace') as f: + for index in range(requested_end): + line = f.readline(max_chars + 1) + if not line: + break + if not line.endswith('\n') and len(line) > max_chars: + while True: + remainder = f.readline(64 * 1024) + if not remainder or remainder.endswith('\n'): + break + actual_end = index + 1 + if index >= start and sum(len(value) for value in selected) < max_chars: + selected.append(line[:max_chars]) + except OSError: + return None + if actual_end == 0: + return None + snippet = ''.join(selected)[:max_chars] + return { + 'file': file_path, + 'line': line_number, + 'start_line': start + 1, + 'end_line': actual_end, + 'nearby': snippet, + } + +def attach_nearby_context(results, budget=None): + budget = budget or context_enrichment_budget() + grouped = {} + finding_budget_exhausted = False + try: + for finding in results.get('findings') or []: + if _context_budget_expired(budget): + _add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained') + return results + if not isinstance(finding, dict): + continue + file_path, line_number = finding_source_location(finding) + if not file_path: + continue + if not _context_budget_claim_finding(budget, finding): + finding_budget_exhausted = True + break + key = os.path.normcase(os.path.abspath(os.fspath(file_path))) + grouped.setdefault(key, {'path': file_path, 'findings': []})['findings'].append((finding, line_number)) + + for group in grouped.values(): + if _context_budget_expired(budget): + _add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained') + return results + payload, complete, reason = _read_context_source(group['path'], budget) + if payload is None: + if reason in ('elapsed', 'source_bytes'): + dimension = 'elapsed-time' if reason == 'elapsed' else 'source-byte' + _add_context_warning(results, 'budget', f'{dimension} budget was exhausted; remaining findings were retained') + return results + _add_context_warning(results, 'failure', 'a nearby source file could not be read; findings were retained') + continue + lines = payload.decode('utf-8', errors='replace').splitlines(keepends=True) + for finding, line_number in group['findings']: + if _context_budget_expired(budget): + _add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained') + return results + context = _nearby_context_from_lines(lines, group['path'], line_number) + if context: + finding['ScannerContext'] = context + if not complete: + if reason == 'source_bytes': + _add_context_warning(results, 'budget', 'source-byte budget was exhausted; remaining findings were retained') + return results + _add_context_warning(results, 'failure', 'a nearby source file changed during its bounded read; findings were retained') + if finding_budget_exhausted: + _add_context_warning(results, 'budget', 'finding-count budget was exhausted; remaining findings were retained') + except Exception: + logger.warning('Optional nearby context enrichment failed; parsed findings were retained') + _add_context_warning(results, 'failure', 'nearby context parsing failed; findings were retained') + return results + + +QWEN_ROUTING_CONTEXT_RE = re.compile( + r'(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)', + re.IGNORECASE, +) +DEEPSEEK_ROUTING_CONTEXT_RE = re.compile( + r'(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)', + re.IGNORECASE, +) +KIMI_ROUTING_CONTEXT_RE = re.compile( + r'(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))', + re.IGNORECASE, +) +ZAI_ROUTING_CONTEXT_RE = re.compile( + r'(?:ZAI_API_KEY|GLM_API_KEY|ZHIPUAI_API_KEY|BIGMODEL_API_KEY|api\.z\.ai|' + r'open\.bigmodel\.cn|zhipuai|chatglm)', + re.IGNORECASE, +) +QWEN_ROUTING_DETECTORS = {'qwendashscope', 'qwen_dashscope', 'dashscope', 'qwen'} +QWEN_EXPLICIT_ROUTING_DETECTORS = {'qwendashscope', 'qwen_dashscope'} +DEEPSEEK_ROUTING_DETECTORS = {'deepseek', 'deepseekapikey', 'deepseek_api_key'} +DEEPSEEK_EXPLICIT_ROUTING_DETECTORS = {'deepseekapikey', 'deepseek_api_key'} +KIMI_ROUTING_DETECTORS = {'kimimoonshot', 'moonshotai', 'moonshot', 'kimi'} +KIMI_EXPLICIT_ROUTING_DETECTORS = {'kimimoonshot', 'moonshotai'} +ZAI_ROUTING_DETECTORS = {'zaiglm'} +ZAI_EXPLICIT_ROUTING_DETECTORS = {'zaiglm'} +AMBIGUOUS_QWEN_DEEPSEEK_HINT = 'ambiguous_qwen_deepseek' +AMBIGUOUS_GENERIC_SK_HINT = 'ambiguous_generic_sk' +GENERIC_SK_PROVIDERS = {'qwen', 'deepseek', 'kimi', 'zai'} +GENERIC_SK_PROVIDER_ORDER = ('deepseek', 'zai', 'qwen', 'kimi') +GENERIC_SK_PROVIDER_HINTS = { + *GENERIC_SK_PROVIDERS, AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT, +} +EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE = 'explicit_assignment' +GENERIC_SK_ROUTING_RE = re.compile(r'^sk-[A-Za-z0-9][A-Za-z0-9_-]{20,505}$') +QWEN_SPECIFIC_ROUTING_RE = re.compile(r'^sk-sp-[A-Za-z0-9][A-Za-z0-9_-]{20,505}$') +ZAI_SPECIFIC_ROUTING_RE = re.compile( + r'^(?:zai-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|' + r'[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{20,505})$' +) + + +def provider_routing_context_text(finding, max_context_chars=65536): + if not isinstance(finding, dict): + return '' + context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {} + parts = [] + remaining = max(0, int(max_context_chars)) + + def add(value): + nonlocal remaining + if not value or remaining <= 0: + return + text = str(value)[:remaining] + parts.append(text) + remaining -= len(text) + + for key in ('nearby', 'file'): + add(context.get(key)) + metadata = finding.get('SourceMetadata') if isinstance(finding.get('SourceMetadata'), dict) else {} + data = metadata.get('Data') if isinstance(metadata.get('Data'), dict) else {} + for details in data.values(): + if not isinstance(details, dict): + continue + for key in ('file', 'repository', 'repo', 'link', 'image'): + add(details.get(key)) + return '\n'.join(parts) + + +def provider_routing_detector_names(finding): + if not isinstance(finding, dict): + return set() + detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '').strip().lower() + extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {} + custom_name = str(extra.get('name') or '').strip().lower() + return {name for name in (detector, custom_name) if name} + + +def is_qwen_routing_detector(finding): + return bool(provider_routing_detector_names(finding) & QWEN_ROUTING_DETECTORS) + + +def is_explicit_qwen_routing_detector(finding): + return bool(provider_routing_detector_names(finding) & QWEN_EXPLICIT_ROUTING_DETECTORS) + + +def is_deepseek_routing_detector(finding): + return bool(provider_routing_detector_names(finding) & DEEPSEEK_ROUTING_DETECTORS) + + +def is_explicit_deepseek_routing_detector(finding): + return bool(provider_routing_detector_names(finding) & DEEPSEEK_EXPLICIT_ROUTING_DETECTORS) + + +def is_kimi_routing_detector(finding): + return bool(provider_routing_detector_names(finding) & KIMI_ROUTING_DETECTORS) + + +def is_explicit_kimi_routing_detector(finding): + return bool(provider_routing_detector_names(finding) & KIMI_EXPLICIT_ROUTING_DETECTORS) + + +def is_zai_routing_detector(finding): + return bool(provider_routing_detector_names(finding) & ZAI_ROUTING_DETECTORS) + + +def is_explicit_zai_routing_detector(finding): + return bool(provider_routing_detector_names(finding) & ZAI_EXPLICIT_ROUTING_DETECTORS) + + +def is_generic_sk_routing_detector(finding): + return bool( + is_qwen_routing_detector(finding) + or is_deepseek_routing_detector(finding) + or is_kimi_routing_detector(finding) + or is_zai_routing_detector(finding) + ) + + +def provider_routing_hint_evidence(hint): + if hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT: + return {'qwen', 'deepseek'} + if hint == AMBIGUOUS_GENERIC_SK_HINT: + return set(GENERIC_SK_PROVIDERS) + return {hint} if hint in GENERIC_SK_PROVIDERS else set() + + +def ambiguous_provider_routing_hint(evidence): + evidence = set(evidence) + if evidence == {'qwen', 'deepseek'}: + return AMBIGUOUS_QWEN_DEEPSEEK_HINT + return AMBIGUOUS_GENERIC_SK_HINT + + +def explicit_provider_routing_evidence(finding): + evidence = set() + if is_explicit_qwen_routing_detector(finding): + evidence.add('qwen') + if is_explicit_deepseek_routing_detector(finding): + evidence.add('deepseek') + if is_explicit_kimi_routing_detector(finding): + evidence.add('kimi') + if is_explicit_zai_routing_detector(finding): + evidence.add('zai') + context = finding.get('ScannerContext') if isinstance(finding, dict) else None + if isinstance(context, dict) and context.get('provider_hint_source') == EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE: + evidence.update(provider_routing_hint_evidence(context.get('provider_hint'))) + return evidence + + +def provider_routing_evidence(finding, max_context_chars=65536): + if not isinstance(finding, dict): + return set() + context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {} + text = provider_routing_context_text(finding, max_context_chars) + evidence = explicit_provider_routing_evidence(finding) + if QWEN_ROUTING_CONTEXT_RE.search(text): + evidence.add('qwen') + if DEEPSEEK_ROUTING_CONTEXT_RE.search(text): + evidence.add('deepseek') + if KIMI_ROUTING_CONTEXT_RE.search(text): + evidence.add('kimi') + if ZAI_ROUTING_CONTEXT_RE.search(text): + evidence.add('zai') + evidence.update(provider_routing_hint_evidence(context.get('provider_hint'))) + raw_values = finding_raw_values(finding) + if any(QWEN_SPECIFIC_ROUTING_RE.fullmatch(value) for value in raw_values): + evidence.add('qwen') + if any(ZAI_SPECIFIC_ROUTING_RE.fullmatch(value) for value in raw_values): + evidence.add('zai') + if ( + not evidence + and is_generic_sk_routing_detector(finding) + and any(GENERIC_SK_ROUTING_RE.fullmatch(value) for value in raw_values) + ): + evidence.update(GENERIC_SK_PROVIDERS) + return evidence + + +def derive_provider_routing_hint( + finding, max_context_chars=65536, evidence=None, explicit_evidence=None, +): + if not isinstance(finding, dict): + return '' + context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {} + evidence = set(evidence) if evidence is not None else provider_routing_evidence(finding, max_context_chars) + explicit_evidence = ( + set(explicit_evidence) + if explicit_evidence is not None + else explicit_provider_routing_evidence(finding) + ) + if len(explicit_evidence) > 1: + provider_hint = ambiguous_provider_routing_hint(explicit_evidence) + elif explicit_evidence: + provider_hint = next(iter(explicit_evidence)) + elif len(evidence) > 1: + provider_hint = ambiguous_provider_routing_hint(evidence) + elif evidence: + provider_hint = next(iter(evidence)) + else: + provider_hint = '' + + persisted_context = dict(context) + if provider_hint: + persisted_context['provider_hint'] = provider_hint + else: + persisted_context.pop('provider_hint', None) + if explicit_evidence: + persisted_context['provider_hint_source'] = EXPLICIT_ASSIGNMENT_PROVIDER_HINT_SOURCE + else: + persisted_context.pop('provider_hint_source', None) + if provider_hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT): + persisted_context['provider_candidates'] = [ + provider for provider in GENERIC_SK_PROVIDER_ORDER if provider in evidence + ] + else: + persisted_context.pop('provider_candidates', None) + if persisted_context or 'ScannerContext' in finding: + finding['ScannerContext'] = persisted_context + return provider_hint + + +def strip_nearby_context_for_persistence(result): + findings = result.get('findings') or [] + evidence_by_value = {} + explicit_evidence_by_value = {} + for finding in findings: + if not ( + is_qwen_routing_detector(finding) + or is_deepseek_routing_detector(finding) + or is_kimi_routing_detector(finding) + or is_zai_routing_detector(finding) + ): + continue + evidence = provider_routing_evidence(finding) + explicit_evidence = explicit_provider_routing_evidence(finding) + for value in finding_raw_values(finding): + evidence_by_value.setdefault(value, set()).update(evidence) + explicit_evidence_by_value.setdefault(value, set()).update(explicit_evidence) + + for finding in findings: + is_routed_detector = ( + is_qwen_routing_detector(finding) + or is_deepseek_routing_detector(finding) + or is_kimi_routing_detector(finding) + or is_zai_routing_detector(finding) + ) + evidence = provider_routing_evidence(finding) + explicit_evidence = explicit_provider_routing_evidence(finding) + if is_routed_detector: + for value in finding_raw_values(finding): + evidence.update(evidence_by_value.get(value, ())) + explicit_evidence.update(explicit_evidence_by_value.get(value, ())) + derive_provider_routing_hint( + finding, evidence=evidence, explicit_evidence=explicit_evidence, + ) + + for finding in findings: + postman_context = finding.get('PostmanContext') if isinstance(finding, dict) else None + if isinstance(postman_context, dict): + finding['PostmanContext'] = sanitize_postman_context(postman_context) + context = finding.get('ScannerContext') if isinstance(finding, dict) else None + if isinstance(context, dict) and 'nearby' in context: + finding['ScannerContext'] = {key: value for key, value in context.items() if key != 'nearby'} + if is_foundry_detector(finding): + text = foundry_finding_text(finding) + endpoints = [normalize_foundry_endpoint(match) for match in POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text)] + keys = foundry_candidate_keys(finding, text) + endpoint = next((item for item in endpoints if item), '') + key = keys[0] if keys else '' + if endpoint and key: + finding['Raw'] = key + finding['RawV2'] = f'{endpoint}:{key}' + finding['Redacted'] = f'{endpoint}:***REDACTED***' + return result + + +def finding_raw_secret_for_uid(finding): + for key in ('RawV2', 'Raw'): + value = finding.get(key) + if value: + return str(value) + structured = finding.get('StructuredData') + if isinstance(structured, dict): + for value in structured.values(): + if isinstance(value, str) and value: + return value + return '' + + +def finding_location_for_uid(finding): + metadata = finding.get('SourceMetadata') if isinstance(finding, dict) else {} + data = metadata.get('Data') if isinstance(metadata, dict) else {} + if not isinstance(data, dict): + return '', '', '' + for details in data.values(): + if not isinstance(details, dict): + continue + file_path = details.get('file') or details.get('path') or details.get('File') or '' + line_number = details.get('line') or details.get('Line') or '' + commit_hash = details.get('commit') or details.get('commitHash') or details.get('commit_hash') or '' + return str(file_path or ''), str(line_number or ''), str(commit_hash or '') + return '', '', '' + + +def sha256_json(value): + return hashlib.sha256(json.dumps(value, ensure_ascii=False, default=str, sort_keys=True).encode('utf-8', errors='replace')).hexdigest() + + +def _bounded_utf8(value, max_bytes): + text = ''.join(' ' if ord(character) < 32 else character for character in str(value or '')) + return text.encode('utf-8', errors='replace')[:max_bytes].decode('utf-8', errors='ignore') + + +def keycheck_input_line_limit(): + return max(1024, int(getattr(scan_config, 'keycheck_input_max_line_bytes', 16 * 1024 * 1024))) + + +def finding_projection_payload(finding): + payload_bytes = json.dumps(finding, ensure_ascii=False, default=str).encode('utf-8') + line_limit = keycheck_input_line_limit() + if len(payload_bytes) + 1 <= line_limit: + return finding, False + + raw_secret = finding_raw_secret_for_uid(finding) + metadata = finding.get('SourceMetadata') if isinstance(finding.get('SourceMetadata'), dict) else {} + data = metadata.get('Data') if isinstance(metadata.get('Data'), dict) else {} + source_type = next(iter(data), '') + file_path, line_number, commit_hash = finding_location_for_uid(finding) + marker = { + 'finding_uid': _bounded_utf8(finding.get('finding_uid'), 256), + 'DetectorName': _bounded_utf8(finding.get('DetectorName') or finding.get('DetectorType'), 64), + 'finding_omitted': True, + 'keycheck_uncheckable': True, + 'omission_reason': 'oversized_finding', + 'secret_sha256': hashlib.sha256(raw_secret.encode('utf-8', errors='replace')).hexdigest() if raw_secret else '', + 'payload_sha256': hashlib.sha256(payload_bytes).hexdigest(), + 'SourceIdentity': { + 'type': _bounded_utf8(source_type, 32), + 'file': _bounded_utf8(file_path, 128), + 'line': _bounded_utf8(line_number, 16), + 'commit': _bounded_utf8(commit_hash, 64), + 'sha256': sha256_json(metadata), + }, + } + marker_bytes = (json.dumps(marker, ensure_ascii=False, default=str) + '\n').encode('utf-8') + if len(marker_bytes) > line_limit: + marker = { + 'finding_uid': _bounded_utf8(finding.get('finding_uid'), 256), + 'DetectorName': _bounded_utf8(finding.get('DetectorName') or finding.get('DetectorType'), 32), + 'finding_omitted': True, + 'keycheck_uncheckable': True, + 'secret_sha256': marker['secret_sha256'], + 'payload_sha256': marker['payload_sha256'], + 'source_identity_sha256': marker['SourceIdentity']['sha256'], + } + marker_bytes = (json.dumps(marker, ensure_ascii=False, default=str) + '\n').encode('utf-8') + if len(marker_bytes) > line_limit: + raise RuntimeError('bounded oversized-finding marker exceeds the keycheck input line limit') + return marker, True + + +def assign_finding_uids(result): + findings = result.get('findings') or [] + scan_event_id = str(result.get('scan_event_id') or '') + if not scan_event_id: + raise ValueError('event-based finding identities require scan_event_id') + for index, finding in enumerate(findings, 1): + if not isinstance(finding, dict) or finding.get('finding_uid'): + continue + detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '') + raw_secret = finding_raw_secret_for_uid(finding) + secret_hash = hashlib.sha256(raw_secret.encode('utf-8', errors='replace')).hexdigest() if raw_secret else '' + fallback_hash = sha256_json(finding) if not secret_hash else '' + file_path, line_number, commit_hash = finding_location_for_uid(finding) + finding['finding_uid'] = hashlib.sha256('|'.join([ + 'truf-finding-v2', + scan_event_id, + str(index), + detector, + secret_hash or fallback_hash, + file_path, + line_number, + commit_hash, + ]).encode('utf-8', errors='replace')).hexdigest() + + +def save_scan_result(result): + """Persist findings like raw TruffleHog JSONL and keep errors separately.""" + results_dir = get_results_dir() + if not results_dir: + return False + success = True + + event_id = str(result.get('scan_event_id') or '') + if not event_id: + logger.error('Unable to persist scan result without scan_event_id') + return False + target = result.get('target', '') + scan_type = result.get('scan_type', '') + timestamp = result.get('timestamp', datetime.now().isoformat()) + assign_finding_uids(result) + try: + foundry_candidates = write_foundry_keycheck_candidates_from_findings(copy.deepcopy(result)) + if foundry_candidates: + logger.info(f"Queued {foundry_candidates} Azure Foundry keycheck candidate(s) from {scan_type}:{target}") + except Exception as e: + logger.warning(f"Unable to queue Azure Foundry keycheck candidate(s) for {scan_type}:{target}: {str(e)}") + success = False + projection_result = copy.deepcopy(result) + strip_nearby_context_for_persistence(projection_result) + projected_findings = [] + oversized_count = 0 + for finding in projection_result.get('findings') or []: + projected, oversized = finding_projection_payload(finding) + projected_findings.append(projected) + oversized_count += int(oversized) + projection_result['findings'] = projected_findings + if oversized_count: + projection_result.setdefault('warnings', []).append( + f'{oversized_count} oversized finding(s) were projected as uncheckable metadata markers; ' + 'PostgreSQL retains the authoritative findings' + ) + projection_result['degraded'] = True + projection_result['oversized_findings_omitted'] = oversized_count + + if projected_findings or projection_result.get('errors') or projection_result.get('warnings') or projection_result.get('skipped'): + success = append_rotating_jsonl_once( + os.path.join(results_dir, 'scan_results.jsonl'), + projection_result, + 'scan_event_id', + event_id, + scan_config.scan_results_max_mb, + ) and success + + for finding in projected_findings: + success = append_rotating_jsonl_once( + os.path.join(results_dir, 'found_secrets.jsonl'), + finding, + 'finding_uid', + finding.get('finding_uid'), + scan_config.found_secrets_max_mb, + ) and success + + if projection_result.get('errors'): + path = os.path.join(results_dir, 'scan_errors.log') + success = append_scan_errors_once( + path, + projection_result, + scan_config.scan_errors_max_mb, + scan_config.scan_errors_keep, + ) and success + return success + +# ====================== +# DOCKER TOKEN MANAGEMENT +# ====================== +@dataclass(frozen=True, repr=False) +class DockerAccount: + name: str + username: str + token: str + config_dir: str + + def __repr__(self): + return f'DockerAccount(name={self.name!r})' + + +@dataclass(frozen=True, repr=False) +class DockerRegistryAuth: + token: str + account_name: str = '' + challenge: str = '' + + +class DockerTokenManager: + def __init__(self): + self.tokens = [] + self.token_dirs = [] + self.accounts = [] + self.current_index = 0 + self.request_index = 0 + self.lock = threading.RLock() + self.config_key = None + self.cooldown_sec = 1800 + self.cooldown_until = {} + self.cooldown_categories = {} + self.hub_tokens = {} + self.hub_token_locks = {} + self.invalid_accounts = set() + self.status_events = {} + self.explicit_pool = False + + @staticmethod + def _account_name(entry, index): + return str(entry.get('name') or f'docker_{index + 1}').strip() + + def setup_accounts( + self, entries, cooldown_sec=1800, explicit_pool=False, + create_config_dirs=True, + ): + normalized = [] + seen_names = set() + for index, entry in enumerate(entries or []): + if not isinstance(entry, dict): + continue + name = self._account_name(entry, index) + username = str(entry.get('username') or '').strip() + token = str(entry.get('token') or '').strip() + if not name or not username or not token or name in seen_names: + continue + seen_names.add(name) + normalized.append((name, username, token)) + config_key = (bool(create_config_dirs),) + tuple( + (name, username, hashlib.sha256(token.encode('utf-8')).hexdigest()) + for name, username, token in normalized + ) + + with self.lock: + self.cooldown_sec = max(60, min(86400, int(cooldown_sec or 1800))) + if ( + config_key == self.config_key + and self.explicit_pool == bool(explicit_pool) + and ( + not normalized or not create_config_dirs + or all( + os.path.isfile(os.path.join(account.config_dir, 'config.json')) + for account in self.accounts + ) + ) + ): + return + self._cleanup_locked() + self.config_key = config_key + self.current_index = 0 + self.request_index = 0 + self.cooldown_until = {} + self.cooldown_categories = {} + self.hub_tokens = {} + self.hub_token_locks = {} + self.invalid_accounts = set() + self.status_events = {} + self.explicit_pool = bool(explicit_pool) + + for name, username, token in normalized: + temp_dir = '' + if create_config_dirs: + temp_dir = create_docker_config_dir() + auth = base64.b64encode(f'{username}:{token}'.encode()).decode() + docker_auth = {'auth': auth} + config = { + 'auths': { + 'https://index.docker.io/v1/': docker_auth, + 'index.docker.io': docker_auth, + 'registry-1.docker.io': docker_auth, + 'https://registry-1.docker.io': docker_auth, + 'docker.io': docker_auth, + } + } + config_path = os.path.join(temp_dir, 'config.json') + with open(config_path, 'w') as f: + json.dump(config, f) + harden_private_file(config_path) + account = DockerAccount(name, username, token, temp_dir) + self.accounts.append(account) + self.tokens.append(f'{username}:{token}') + if temp_dir: + self.token_dirs.append(temp_dir) + logger.info('Configured Docker account: %s', name) + + def setup_tokens(self, tokens_str=None, username=None, create_config_dirs=True): + """Initialize Docker tokens from environment variable""" + username = (username or os.getenv('DOCKERHUB_USERNAME') or os.getenv('DOCKER_USERNAME') or '').strip() + tokens_str = tokens_str if tokens_str is not None else os.getenv('DOCKER_TOKENS', '') + + if not tokens_str: + single_token = os.getenv('DOCKERHUB_TOKEN') or os.getenv('DOCKER_TOKEN') + if single_token: + tokens_str = single_token + + entries = [] + if tokens_str: + for index, raw_token in enumerate(tokens_str.replace('\n', ',').split(',')): + raw_token = raw_token.strip() + if not raw_token: + continue + if ':' in raw_token: + account_username, token = raw_token.split(':', 1) + elif username: + account_username, token = username, raw_token + else: + logger.warning('Docker token without username ignored. Use username:token or set DOCKERHUB_USERNAME.') + continue + account_username = str(account_username).strip() + token = str(token).strip() + if account_username and token: + entries.append({ + 'name': f'docker_{index + 1}', + 'username': account_username, + 'token': token, + }) + self.setup_accounts( + entries, explicit_pool=False, create_config_dirs=create_config_dirs, + ) + + def has_accounts(self): + with self.lock: + return bool(self.accounts) + + def account_count(self): + with self.lock: + return len(self.accounts) + + def next_account(self, endpoint, excluded_names=None): + excluded_names = set(excluded_names or ()) + with self.lock: + if not self.accounts: + return None + now = time.time() + for offset in range(len(self.accounts)): + position = (self.request_index + offset) % len(self.accounts) + account = self.accounts[position] + if account.name in excluded_names: + continue + if account.name in self.invalid_accounts: + continue + if float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now: + continue + self.request_index = (position + 1) % len(self.accounts) + return account + return None + + def report_http_status(self, account, endpoint, status, response=None, category=None): + account_name = account.name if isinstance(account, DockerAccount) else str(account or '') + if not account_name: + return False + status = int(status or 0) + category = str(category or ('rate_limit' if status == 429 else 'auth_forbidden')) + with self.lock: + if account_name in self.invalid_accounts and category != 'auth_invalid': + return self._all_unavailable_locked(endpoint) + if category == 'auth_invalid': + self.invalid_accounts.add(account_name) + retry_at = float('inf') + reset_at = 'manual' + else: + delay = dockerhub_retry_after_seconds(response) if status == 429 else self.cooldown_sec + retry_at = time.time() + max(60, int(delay)) + reset_at = datetime.fromtimestamp(retry_at, timezone.utc).isoformat(timespec='seconds') + self.cooldown_until[(account_name, endpoint)] = retry_at + self.cooldown_categories[(account_name, endpoint)] = category + if status == 401 or category == 'auth_invalid': + self.hub_tokens.pop(account_name, None) + self.status_events[(account_name, endpoint)] = { + 'name': account_name, + 'endpoint': endpoint, + 'category': category, + 'reset_at': reset_at, + 'message': f'Docker {endpoint} HTTP {status}', + } + return self._all_unavailable_locked(endpoint) + + def report_success(self, account, endpoint): + account_name = account.name if isinstance(account, DockerAccount) else str(account or '') + if not account_name: + return + with self.lock: + if account_name in self.invalid_accounts: + return + expires_at = float(self.cooldown_until.get((account_name, endpoint), 0) or 0) + if expires_at <= time.time(): + self.cooldown_until.pop((account_name, endpoint), None) + self.cooldown_categories.pop((account_name, endpoint), None) + event_key = (account_name, endpoint) + if event_key not in self.status_events: + self.status_events[event_key] = { + 'name': account_name, + 'endpoint': endpoint, + 'category': 'ok', + 'reset_at': None, + 'message': '', + } + + def _all_unavailable_locked(self, endpoint): + if not self.accounts: + return False + now = time.time() + return all( + account.name in self.invalid_accounts + or float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now + for account in self.accounts + ) + + def all_unavailable(self, endpoint): + with self.lock: + return self._all_unavailable_locked(endpoint) + + def rate_limit_contributes_to_exhaustion(self, endpoint): + with self.lock: + if not self._all_unavailable_locked(endpoint): + return False + now = time.time() + return any( + float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now + and self.cooldown_categories.get((account.name, endpoint)) == 'rate_limit' + for account in self.accounts + ) + + def uses_explicit_pool(self): + with self.lock: + return bool(self.explicit_pool) + + def seconds_until_available(self, endpoint): + with self.lock: + now = time.time() + deadlines = [ + float(self.cooldown_until.get((account.name, endpoint), 0) or 0) + for account in self.accounts + if float(self.cooldown_until.get((account.name, endpoint), 0) or 0) != float('inf') + and float(self.cooldown_until.get((account.name, endpoint), 0) or 0) > now + ] + if not deadlines: + return self.cooldown_sec + return max(60, math.ceil(min(deadlines) - now)) + + def account_available(self, account_name, endpoint): + with self.lock: + return account_name not in self.invalid_accounts and float( + self.cooldown_until.get((account_name, endpoint), 0) or 0 + ) <= time.time() + + def hub_token_lock(self, account_name): + with self.lock: + lock = self.hub_token_locks.get(account_name) + if lock is None: + lock = threading.Lock() + self.hub_token_locks[account_name] = lock + return lock + + def restore_endpoint_cooldowns(self, endpoint_status): + if not isinstance(endpoint_status, dict): + return + with self.lock: + account_names = {account.name for account in self.accounts} + now = time.time() + for endpoint, accounts in endpoint_status.items(): + if endpoint not in {'hub_search', 'hub_tags', 'registry'} or not isinstance(accounts, dict): + continue + for account_name, status in accounts.items(): + if account_name not in account_names or not isinstance(status, dict): + continue + disabled_until = status.get('disabled_until') + if disabled_until == 'manual': + retry_at = float('inf') + else: + try: + retry_at = datetime.fromisoformat( + str(disabled_until).replace('Z', '+00:00') + ).timestamp() + except (TypeError, ValueError, OverflowError): + continue + if retry_at <= now: + continue + key = (account_name, endpoint) + category = str(status.get('disabled_reason') or 'rate_limit') + if category == 'auth_invalid': + self.invalid_accounts.add(account_name) + if retry_at > float(self.cooldown_until.get(key, 0) or 0): + self.cooldown_until[key] = retry_at + self.cooldown_categories[key] = category + + def cached_hub_token(self, account_name): + with self.lock: + token, expires_at = self.hub_tokens.get(account_name, ('', 0)) + if token and float(expires_at or 0) > time.time() + 30: + return token + self.hub_tokens.pop(account_name, None) + return '' + + def cache_hub_token(self, account_name, token, expires_in): + with self.lock: + lifetime = max(60, min(600, int(expires_in or 600))) + self.hub_tokens[account_name] = (token, time.time() + lifetime) + + def invalidate_hub_token(self, account_name): + with self.lock: + self.hub_tokens.pop(account_name, None) + + def drain_status_events(self): + with self.lock: + events = list(self.status_events.values()) + self.status_events = {} + return events + + def get_next_config(self): + """Get the next Docker config directory in rotation""" + with self.lock: + if not self.accounts: + return None + for offset in range(len(self.accounts)): + position = (self.current_index + offset) % len(self.accounts) + account = self.accounts[position] + if account.name in self.invalid_accounts: + continue + self.current_index = (position + 1) % len(self.accounts) + return account.config_dir + return None + + def _cleanup_locked(self): + for directory in self.token_dirs: + try: + cleanup_command_work_dir(directory) + except Exception as exc: + logger.error('Error cleaning Docker config: %s', str(exc)) + self.tokens = [] + self.token_dirs = [] + self.accounts = [] + + def cleanup(self): + """Clean up temporary Docker config directories""" + with self.lock: + self._cleanup_locked() + self.config_key = None + self.cooldown_until = {} + self.cooldown_categories = {} + self.hub_tokens = {} + self.hub_token_locks = {} + self.invalid_accounts = set() + self.status_events = {} + self.explicit_pool = False + +docker_token_manager = DockerTokenManager() + +def configure_docker_tokens(tokens_str=None, username=None): + """Reconfigure Docker auth tokens after UI or CLI input.""" + require_scanner_runtime_initialized() + docker_token_manager.setup_tokens(tokens_str, username) + + +def configure_docker_accounts(entries, cooldown_sec=1800): + require_scanner_runtime_initialized() + docker_token_manager.setup_accounts( + entries, cooldown_sec=cooldown_sec, explicit_pool=True, + ) + + +def configure_docker_discovery_tokens(tokens_str=None, username=None): + docker_token_manager.setup_tokens( + tokens_str, username, create_config_dirs=False, + ) + + +def configure_docker_discovery_accounts(entries, cooldown_sec=1800): + docker_token_manager.setup_accounts( + entries, cooldown_sec=cooldown_sec, explicit_pool=True, + create_config_dirs=False, + ) + + +def restore_docker_endpoint_cooldowns(endpoint_status): + docker_token_manager.restore_endpoint_cooldowns(endpoint_status) + + +def drain_docker_auth_events(): + return docker_token_manager.drain_status_events() + +# =================== +# API FETCH FUNCTIONS +# =================== +def github_repo_to_target(item): + return { + 'url': item.get('clone_url', ''), + 'name': item.get('full_name', ''), + 'created_at': item.get('created_at', ''), + 'updated_at': item.get('updated_at', ''), + 'pushed_at': item.get('pushed_at', ''), + } + + +def github_headers(token=None): + headers = {'Accept': 'application/vnd.github+json', 'User-Agent': 'GitSecretsScanner/2.0'} + if token: + headers['Authorization'] = f'Bearer {token}' + return headers + + +def gitlab_headers(token=None): + headers = {'User-Agent': 'GitSecretsScanner/2.0'} + if token: + headers['PRIVATE-TOKEN'] = token + return headers + + +def parse_github_repo_target(target): + if isinstance(target, dict): + repo = target.get('repo') or target.get('full_name') or '' + if not repo and target.get('repo_url'): + candidate = normalize_git_repo_candidate(target.get('repo_url')) + if candidate and candidate.get('provider') == 'github': + repo = candidate.get('repo_path') or '' + url = target.get('url') or (f'https://github.com/{repo}' if repo else '') + return repo.strip('/'), url + text = str(target or '').strip() + if text.startswith('{'): + try: + return parse_github_repo_target(json.loads(text)) + except Exception: + pass + lowered = text.lower().rstrip('/') + if lowered.endswith('.git'): + lowered = lowered[:-4] + candidate = normalize_git_repo_candidate(text) + if candidate and candidate.get('provider') == 'github': + repo = candidate.get('repo_path') or '' + return repo, f'https://github.com/{repo}' + match = re.search(r'github\.com[:/]([^/\s]+/[^/\s]+)', lowered, re.IGNORECASE) + if match: + repo = match.group(1).strip('/') + return repo, f'https://github.com/{repo}' + if re.match(r'^[^/\s]+/[^/\s]+$', text): + repo = text.strip('/') + return repo, f'https://github.com/{repo}' + return '', text + + +def parse_gitlab_project_target(target): + if isinstance(target, dict): + project = target.get('project') or target.get('path_with_namespace') or target.get('repo') or '' + if not project and target.get('repo_url'): + candidate = normalize_git_repo_candidate(target.get('repo_url')) + if candidate and candidate.get('provider') == 'gitlab': + project = candidate.get('repo_path') or '' + url = target.get('url') or (f'https://gitlab.com/{project}' if project else '') + return project.strip('/'), url + text = str(target or '').strip() + if text.startswith('{'): + try: + return parse_gitlab_project_target(json.loads(text)) + except Exception: + pass + lowered = text.lower().rstrip('/') + if lowered.endswith('.git'): + lowered = lowered[:-4] + candidate = normalize_git_repo_candidate(text) + if candidate and candidate.get('provider') == 'gitlab': + project = candidate.get('repo_path') or '' + return project, f'https://gitlab.com/{project}' + match = re.search(r'gitlab\.com[:/](.+)$', lowered, re.IGNORECASE) + if match: + project = match.group(1).strip('/') + return project, f'https://gitlab.com/{project}' + if '/' in text and '://' not in text: + project = text.strip('/') + return project, f'https://gitlab.com/{project}' + return '', text + +def github_rate_limit_reset(response): + if not response: + return None + reset = response.headers.get('X-RateLimit-Reset') + if not reset: + return None + try: + return datetime.fromtimestamp(int(reset), timezone.utc).isoformat(timespec='seconds') + except (TypeError, ValueError): + return None + +def page_is_known(targets, known_targets=None, normalize_target=None, known_target_lookup=None): + if not normalize_target or not targets: + return False + offered = [target for target in targets if target] + normalized = [normalize_target(target) for target in offered] + if known_target_lookup: + try: + known_targets = set(known_target_lookup(offered) or ()) + except Exception as exc: + logger.warning(f'Known-target page lookup failed open: {str(exc)[:300]}') + return False + if not known_targets: + return False + return bool(normalized) and all(target in known_targets for target in normalized) + + +POSTMAN_COLLECTION_SUFFIX = 'postman_collection.json' +POSTMAN_ENVIRONMENT_SUFFIX = 'postman_environment.json' +API_ARTIFACT_PATTERNS = { + 'collection': ('postman_collection.json',), + 'environment': ('postman_environment.json',), + 'postman': ('postman.json', '.postman.json'), + 'insomnia': ('insomnia.json', '.insomnia.json'), + 'bruno': ('.bru', 'bruno.json'), + 'thunder_collection': ('thunder-collection.json',), + 'thunder_environment': ('thunder-environment.json',), + 'hoppscotch': ('hoppscotch.json',), + 'generic': ('collection.json', 'environment.json'), +} +API_ARTIFACT_SEARCHES = { + 'collection': ('filename:postman_collection.json {query}',), + 'environment': ('filename:postman_environment.json {query}',), + 'postman': ('filename:postman.json {query}', 'filename:.postman.json {query}'), + 'insomnia': ('filename:insomnia.json {query}', 'filename:.insomnia.json {query}'), + 'bruno': ('extension:bru {query}', 'filename:bruno.json {query}'), + 'thunder_collection': ('filename:thunder-collection.json {query}', 'filename:thunder-collection_ {query}'), + 'thunder_environment': ('filename:thunder-environment.json {query}', 'filename:thunder-environment_ {query}'), + 'hoppscotch': ('filename:hoppscotch.json {query}',), + 'generic': ('filename:collection.json postman {query}', 'filename:environment.json postman {query}'), + 'signature': ( + '"currentValue" "api_key" {query}', + '"pm.collectionVariables" {query}', + '"pm.environment.set" {query}', + '"openai.azure.com" "api-key" {query}', + '"services.ai.azure.com" "api-key" {query}', + '"generativelanguage.googleapis.com" "key" {query}', + ), +} +POSTMAN_PLACEHOLDER_RE = re.compile( + r'^(?:\{\{[^}]+\}\}|<[^>]+>|your[_ -]?[a-z0-9_-]+|replace[_ -]?me|change[_ -]?me|changeme|example|dummy|test)$', + re.IGNORECASE, +) +POSTMAN_JSON_HARD_MAX_INPUT_BYTES = 16 * 1024 * 1024 +POSTMAN_HARVEST_MAX_WARNINGS = 5 +POSTMAN_HARVEST_WARNING_MAX_CHARS = 400 + + +class PostmanCacheValidationError(ValueError): + pass + + +class PostmanCacheTooLarge(PostmanCacheValidationError): + pass + + +class PostmanCacheCapacityError(PostmanCacheValidationError): + pass + + +class _PostmanHarvestDeadlineReached(RuntimeError): + pass + + +def _postman_discovery_limits(max_artifacts=None, max_page_artifacts=None, max_total_bytes=None, max_elapsed_sec=None): + artifacts = int( + getattr(scan_config, 'postman_discovery_max_artifacts_per_cycle', 1000) + if max_artifacts is None else max_artifacts + ) + page_artifacts = int( + getattr(scan_config, 'postman_discovery_max_artifacts_per_page', 100) + if max_page_artifacts is None else max_page_artifacts + ) + total_bytes = int( + getattr(scan_config, 'postman_discovery_max_bytes_per_cycle', 1024 * 1024 * 1024) + if max_total_bytes is None else max_total_bytes + ) + elapsed_sec = float( + getattr(scan_config, 'postman_discovery_max_elapsed_sec', 300.0) + if max_elapsed_sec is None else max_elapsed_sec + ) + if artifacts <= 0 or page_artifacts <= 0 or total_bytes <= 0 or elapsed_sec <= 0 or not math.isfinite(elapsed_sec): + raise ValueError('Postman discovery limits must be finite and positive') + return artifacts, page_artifacts, total_bytes, elapsed_sec + + +class _PostmanDiscoveryBudget: + MAX_WARNINGS = 5 + + def __init__(self, source, max_artifacts=None, max_page_artifacts=None, max_total_bytes=None, max_elapsed_sec=None): + ( + self.max_artifacts, + self.max_page_artifacts, + self.max_total_bytes, + elapsed_sec, + ) = _postman_discovery_limits(max_artifacts, max_page_artifacts, max_total_bytes, max_elapsed_sec) + self.source = str(source or 'non-package') + self.elapsed_sec = elapsed_sec + self.deadline = time.monotonic() + elapsed_sec + self.artifacts = 0 + self.total_bytes = 0 + self.stop_reason = '' + self._warnings = set() + + def warn(self, detail): + detail = str(detail or '')[:400] + if detail in self._warnings or len(self._warnings) >= self.MAX_WARNINGS: + return + self._warnings.add(detail) + logger.warning('Optional Postman %s discovery bounded: %s', self.source, detail) + + def stop(self, detail): + if not self.stop_reason: + self.stop_reason = str(detail or 'bounded discovery limit reached')[:400] + self.warn(self.stop_reason) + return False + + def check_deadline(self, context='during discovery'): + if self.stop_reason: + return False + if time.monotonic() >= self.deadline: + return self.stop(f'elapsed deadline of {self.elapsed_sec:g}s reached {context}') + return True + + def request_timeout(self, configured_timeout): + if not self.check_deadline('before a network request'): + raise _PostmanHarvestDeadlineReached(self.stop_reason) + remaining = self.deadline - time.monotonic() + if remaining <= 0: + self.stop(f'elapsed deadline of {self.elapsed_sec:g}s reached before a network request') + raise _PostmanHarvestDeadlineReached(self.stop_reason) + return max(0.01, min(float(configured_timeout or remaining), remaining)) + + def admit(self, page_artifacts): + if not self.check_deadline(): + return False, 'cycle' + if page_artifacts >= self.max_page_artifacts: + self.warn(f'per-page artifact limit of {self.max_page_artifacts} reached') + return False, 'page' + if self.artifacts >= self.max_artifacts: + self.stop(f'per-cycle artifact limit of {self.max_artifacts} reached') + return False, 'cycle' + self.artifacts += 1 + return True, '' + + def account_bytes(self, size): + size = max(0, int(size or 0)) + if self.total_bytes + size > self.max_total_bytes: + return self.stop( + f'aggregate byte limit of {self.max_total_bytes} reached after {self.total_bytes} byte(s)' + ) + self.total_bytes += size + return True + + def exhausted(self): + if self.stop_reason: + return True + if self.artifacts >= self.max_artifacts: + return not self.stop(f'per-cycle artifact limit of {self.max_artifacts} reached') + if self.total_bytes >= self.max_total_bytes: + return not self.stop(f'aggregate byte limit of {self.max_total_bytes} reached') + return not self.check_deadline() + + +def _publish_postman_discovery_batch(prepared, cache_dir, budget): + if not prepared: + return [], False + published, deadline_reached = _publish_postman_cache_entries( + [item['entry'] for item in prepared], + cache_dir, + deadline=budget.deadline, + ) + if deadline_reached: + budget.stop(f'elapsed deadline of {budget.elapsed_sec:g}s reached while scanning or publishing the cache') + return list(zip(prepared, published)), deadline_reached + + +def utc_now_for_postman(): + return datetime.now(timezone.utc) + + +def parse_postman_time(value): + if not value: + return None + try: + parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.astimezone(timezone.utc) + except ValueError: + return None + + +def postman_kind_for_path(path): + name = os.path.basename(str(path or '')).lower() + normalized = str(path or '').replace('\\', '/').lower() + for kind, suffixes in API_ARTIFACT_PATTERNS.items(): + if any(name.endswith(suffix) or normalized.endswith('/bruno/' + suffix) for suffix in suffixes): + return kind + return 'artifact' + + +def normalize_postman_search_kinds(value): + if not value: + return ['collection', 'environment'] + if isinstance(value, str): + parts = [item.strip().lower() for item in value.split(',')] + else: + parts = [str(item).strip().lower() for item in value] + aliases = { + 'collections': 'collection', 'env': 'environment', 'environments': 'environment', + 'api_artifacts': 'all', 'api-artifacts': 'all', 'thunder': 'thunder_collection', + } + kinds = [] + for item in parts: + item = aliases.get(item, item) + if item == 'all': + for kind in API_ARTIFACT_SEARCHES: + if kind not in kinds: + kinds.append(kind) + continue + if item in API_ARTIFACT_SEARCHES and item not in kinds: + kinds.append(item) + return kinds or ['collection', 'environment'] + + +def parse_postman_target(target): + if isinstance(target, dict): + return dict(target) + text = str(target or '').strip() + if not text: + return {} + if text.startswith('{'): + return json.loads(text) + if text.lower().startswith('file:'): + path = text[5:] + return {'source': 'local_file', 'kind': postman_kind_for_path(path), 'local_path': path, 'path': path} + if os.path.exists(text): + return {'source': 'local_file', 'kind': postman_kind_for_path(text), 'local_path': text, 'path': text} + return {'source': 'url', 'kind': postman_kind_for_path(text), 'url': text} + + +def postman_target_identity(target): + return semantic_postman_target_identity(target) + + +def get_postman_cache_dir(cache_dir=None): + path = cache_dir or getattr(scan_config, 'postman_cache_dir', None) + if not path: + raise PostmanCacheValidationError('configured Postman cache directory is required') + runtime_dir = getattr(scan_config, 'runtime_dir', None) + if not runtime_dir: + raise PostmanCacheValidationError('configured runtime directory is required for Postman cache containment') + runtime_root = canonical_path(runtime_dir) + cache_root = canonical_path(path) + try: + contained = os.path.commonpath((runtime_root, cache_root)) == runtime_root and cache_root != runtime_root + except ValueError: + contained = False + if not contained: + raise PostmanCacheValidationError('Postman cache must remain under the configured runtime directory') + return require_private_directory(cache_root, create=True) + + +def sha256_bytes(value): + return hashlib.sha256(value or b'').hexdigest() + + +def postman_cache_file_path(digest, cache_dir=None): + root = get_postman_cache_dir(cache_dir) + prefix = str(digest or '')[:2] or 'xx' + directory = os.path.join(root, prefix) + ensure_private_directory(directory, reject_reparse=True) + return os.path.join(directory, f'{digest}.json') + + +def postman_cache_usage(root, deadline=None): + usage = {'items': 0, 'files': 0, 'bytes': 0} + stack = [require_private_directory(root, create=False)] + while stack: + if deadline is not None and time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached() + current = stack.pop() + with os.scandir(current) as entries: + for entry in entries: + if deadline is not None and time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached() + if entry.is_symlink() or is_reparse_point(entry.path): + raise PostmanCacheCapacityError(f'Postman cache contains a linked entry: {entry.path}') + if entry.is_dir(follow_symlinks=False): + if not private_directory_ready(entry.path): + raise PostmanCacheCapacityError(f'Postman cache directory is not private: {entry.path}') + stack.append(entry.path) + continue + if not entry.is_file(follow_symlinks=False): + raise PostmanCacheCapacityError(f'Postman cache contains an unsupported entry: {entry.path}') + if entry.name == '.postman-cache.lock': + continue + details = entry.stat(follow_symlinks=False) + usage['files'] += 1 + usage['bytes'] += int(details.st_size) + if not entry.name.endswith('.meta.json'): + usage['items'] += 1 + return usage + + +def _postman_cache_limits(max_items=None, max_bytes=None, min_free_bytes=None): + values = { + 'items': int(getattr(scan_config, 'postman_cache_max_items', 0) if max_items is None else max_items), + 'bytes': int(getattr(scan_config, 'postman_cache_max_bytes', 0) if max_bytes is None else max_bytes), + 'min_free_bytes': int(getattr(scan_config, 'postman_cache_min_free_bytes', 0) if min_free_bytes is None else min_free_bytes), + } + if values['items'] <= 0 or values['bytes'] <= 0 or values['min_free_bytes'] < 0: + raise PostmanCacheCapacityError('Postman cache aggregate limits must be finite positive values') + return values + + +def _write_private_cache_file(path, payload): + if os.path.lexists(path): + raise FileExistsError(path) + temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp' + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0) + descriptor = os.open(temporary, flags, 0o600) + try: + os.close(descriptor) + descriptor = None + harden_private_file(temporary) + with open(temporary, 'wb') as handle: + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + if not private_file_ready(temporary): + raise PostmanCacheCapacityError(f'Postman cache temporary file is not private: {temporary}') + if os.path.lexists(path): + raise FileExistsError(path) + durable_replace(temporary, path) + if not private_file_ready(path): + raise PostmanCacheCapacityError(f'Postman cache publication is not private: {path}') + finally: + if descriptor is not None: + os.close(descriptor) + try: + if os.path.exists(temporary): + os.remove(temporary) + except OSError: + pass + + +def _bounded_json_bytes(value, max_bytes, *, indent=None, sort_keys=False, newline=False): + output = bytearray() + encoder = json.JSONEncoder(ensure_ascii=False, indent=indent, sort_keys=sort_keys) + for chunk in encoder.iterencode(value): + encoded = chunk.encode('utf-8') + if len(output) + len(encoded) + (1 if newline else 0) > max_bytes: + raise ValueError('serialized JSON exceeds its bounded byte limit') + output.extend(encoded) + if newline: + output.extend(b'\n') + return bytes(output) + + +def validate_postman_cache_artifact(target_data, max_artifact_size_mb=20, expected_size=None): + _raise_if_scan_slot_fatal() + if not isinstance(target_data, dict): + raise PostmanCacheValidationError('Postman cache metadata must be an object') + offered_path = target_data.get('cache_path') or target_data.get('local_path') + if not offered_path: + raise PostmanCacheValidationError('Postman target missing cached artifact') + cache_root = canonical_path(get_postman_cache_dir()) + reject_reparse_components(offered_path) + cache_path = canonical_path(offered_path) + try: + contained = os.path.commonpath((cache_root, cache_path)) == cache_root and cache_path != cache_root + except ValueError: + contained = False + if not contained: + raise PostmanCacheValidationError('Postman cache path escapes the configured cache root') + reject_reparse_components(cache_path) + if is_reparse_point(cache_path) or not private_file_ready(cache_path): + raise PostmanCacheValidationError('Postman cached artifact is absent, linked, or not private') + declared_hash = str(target_data.get('sha256') or '').strip().lower() + if not re.fullmatch(r'[0-9a-f]{64}', declared_hash): + raise PostmanCacheValidationError('Postman target has no valid declared SHA-256') + if postman_target_identity(target_data) != f'postman:sha256:{declared_hash}': + raise PostmanCacheValidationError('Postman semantic identity does not match its declared SHA-256') + + flags = os.O_RDONLY | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0) + descriptor = os.open(cache_path, flags) + try: + opened = os.fstat(descriptor) + if not stat.S_ISREG(opened.st_mode): + raise PostmanCacheValidationError('Postman cached artifact is not a regular file') + size = int(opened.st_size) + if size <= 0: + raise PostmanCacheValidationError('Postman cached artifact is empty') + declared_sizes = [expected_size] + declared_sizes.extend(target_data.get(name) for name in ('size', 'bytes')) + for declared_size in declared_sizes: + if declared_size in (None, ''): + continue + try: + parsed_size = int(declared_size) + except (TypeError, ValueError) as exc: + raise PostmanCacheValidationError('Postman cached artifact has invalid declared size') from exc + if parsed_size != size: + raise PostmanCacheValidationError('Postman cached artifact size mismatch') + max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 + if max_bytes <= 0: + raise PostmanCacheValidationError('Postman artifact limit must be finite and positive') + if size > max_bytes: + raise PostmanCacheTooLarge(f'artifact exceeds {max_artifact_size_mb} MB') + digest = hashlib.sha256() + while True: + _raise_if_scan_slot_fatal() + block = os.read(descriptor, 1024 * 1024) + if not block: + break + digest.update(block) + if digest.hexdigest() != declared_hash: + raise PostmanCacheValidationError('Postman cached artifact SHA-256 mismatch') + _raise_if_scan_slot_fatal() + current = os.stat(cache_path, follow_symlinks=False) + opened_identity = (opened.st_dev, opened.st_ino, opened.st_size, getattr(opened, 'st_mtime_ns', None)) + current_identity = (current.st_dev, current.st_ino, current.st_size, getattr(current, 'st_mtime_ns', None)) + if opened_identity != current_identity: + raise PostmanCacheValidationError('Postman cached artifact changed during validation') + finally: + os.close(descriptor) + return cache_path, size + + +def _prepare_postman_cache_entry(content, kind, origin, max_artifact_size_mb): + max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 + if max_bytes <= 0: + raise ValueError('Postman artifact limit must be finite and positive') + if isinstance(content, str) and len(content) > max_bytes: + raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB') + content_bytes = content.encode('utf-8', errors='replace') if isinstance(content, str) else bytes(content or b'') + if len(content_bytes) > max_bytes: + raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB') + if not content_bytes: + raise ValueError('Postman artifact is empty') + stripped = content_bytes.lstrip() + if stripped.startswith((b'{', b'[')): + if len(content_bytes) > POSTMAN_JSON_HARD_MAX_INPUT_BYTES: + raise ValueError('Postman JSON artifact exceeds the hard pre-parse byte limit') + try: + json.loads(content_bytes.decode('utf-8-sig')) + except (UnicodeDecodeError, ValueError, RecursionError) as exc: + raise ValueError('Postman JSON artifact is invalid') from exc + raw_content = content_bytes + digest = sha256_bytes(raw_content) + metadata = { + 'sha256': digest, + 'kind': kind, + 'bytes': len(raw_content), + 'origin': origin or {}, + 'cached_at': utc_now_for_postman().isoformat(timespec='seconds'), + } + try: + metadata_bytes = _bounded_json_bytes(metadata, max_bytes, indent=2, sort_keys=True, newline=True) + except ValueError as exc: + raise PostmanCacheCapacityError('Postman cache metadata exceeds the per-artifact byte limit') from exc + return { + 'content': raw_content, + 'digest': digest, + 'metadata': metadata_bytes, + 'size': len(raw_content), + } + + +def _publish_postman_cache_entries( + entries, + cache_dir=None, + cache_max_items=None, + cache_max_bytes=None, + cache_min_free_bytes=None, + deadline=None, +): + if not entries: + return [], False + root = get_postman_cache_dir(cache_dir) + limits = _postman_cache_limits(cache_max_items, cache_max_bytes, cache_min_free_bytes) + lock_path = os.path.join(root, '.postman-cache.lock') + lock_timeout = max(1.0, float(getattr(scan_config, 'postman_cache_lock_timeout_sec', 30))) + if deadline is not None: + remaining = deadline - time.monotonic() + if remaining <= 0: + return [], True + lock_timeout = min(lock_timeout, max(0.01, remaining)) + try: + lock = acquire_file_lock( + lock_path, + timeout_sec=lock_timeout, + ) + except TimeoutError: + if deadline is not None and time.monotonic() >= deadline: + return [], True + raise + try: + if deadline is not None and time.monotonic() >= deadline: + return [], True + try: + usage = ( + postman_cache_usage(root) + if deadline is None + else postman_cache_usage(root, deadline=deadline) + ) + except _PostmanHarvestDeadlineReached: + return [], True + + plans = [] + reserved_bytes = 0 + free_bytes = None + seen = set() + for entry in entries: + if deadline is not None and time.monotonic() >= deadline: + return [], True + digest = entry['digest'] + if digest in seen: + continue + seen.add(digest) + if sha256_bytes(entry['content']) != digest: + raise PostmanCacheValidationError('Postman cache entry SHA-256 changed before publication') + + prefix_dir = os.path.join(root, digest[:2]) + path = os.path.join(prefix_dir, f'{digest}.json') + meta_path = f'{path}.meta.json' + if os.path.lexists(prefix_dir): + reject_reparse_components(prefix_dir) + if not os.path.isdir(prefix_dir) or not private_directory_ready(prefix_dir): + raise PostmanCacheCapacityError(f'Postman cache prefix directory is not private: {prefix_dir}') + artifact_exists = os.path.lexists(path) + metadata_exists = os.path.lexists(meta_path) + for existing in (path, meta_path): + if os.path.lexists(existing): + reject_reparse_components(existing) + if not private_file_ready(existing): + raise PostmanCacheCapacityError(f'Postman cache file is not private: {existing}') + + added_items = 0 if artifact_exists else 1 + added_files = int(not artifact_exists) + int(not metadata_exists) + added_bytes = ( + (0 if artifact_exists else entry['size']) + + (0 if metadata_exists else len(entry['metadata'])) + ) + if usage['items'] + added_items > limits['items']: + raise PostmanCacheCapacityError( + f'Postman cache item capacity reached ({usage["items"]}/{limits["items"]})' + ) + if usage['bytes'] + added_bytes > limits['bytes']: + raise PostmanCacheCapacityError( + f'Postman cache byte capacity reached ({usage["bytes"]}/{limits["bytes"]})' + ) + if added_bytes and free_bytes is None: + free_bytes = int(shutil.disk_usage(root).free) + if added_bytes and free_bytes - reserved_bytes - added_bytes < limits['min_free_bytes']: + raise PostmanCacheCapacityError( + f'Postman cache free-space reserve would be crossed ({free_bytes} bytes free)' + ) + usage['items'] += added_items + usage['files'] += added_files + usage['bytes'] += added_bytes + reserved_bytes += added_bytes + plans.append((entry, prefix_dir, path, meta_path, artifact_exists, metadata_exists)) + + published = [] + for entry, prefix_dir, path, meta_path, artifact_exists, metadata_exists in plans: + if deadline is not None and time.monotonic() >= deadline: + return published, True + if not os.path.isdir(prefix_dir): + ensure_private_directory(prefix_dir, reject_reparse=True) + elif not private_directory_ready(prefix_dir): + raise PostmanCacheCapacityError(f'Postman cache prefix directory is not private: {prefix_dir}') + if not artifact_exists: + _write_private_cache_file(path, entry['content']) + if not metadata_exists: + _write_private_cache_file(meta_path, entry['metadata']) + published.append((path, entry['digest'], entry['size'])) + return published, False + finally: + release_file_lock(lock, lock_path) + + +def write_postman_cache( + content, + kind='artifact', + origin=None, + cache_dir=None, + max_artifact_size_mb=20, + cache_max_items=None, + cache_max_bytes=None, + cache_min_free_bytes=None, +): + entry = _prepare_postman_cache_entry(content, kind, origin, max_artifact_size_mb) + published, deadline_reached = _publish_postman_cache_entries( + [entry], + cache_dir, + cache_max_items, + cache_max_bytes, + cache_min_free_bytes, + ) + if deadline_reached or not published: + raise PostmanCacheCapacityError('Postman cache publication did not complete') + return published[0] + + +def postman_target_from_cached_artifact(source, kind, cache_path, digest, origin=None, **extra): + payload = {'source': source, 'kind': kind, 'cache_path': cache_path, 'sha256': digest, 'origin': origin or {}} + payload.update({key: value for key, value in extra.items() if value not in (None, '')}) + return json.dumps(payload, separators=(',', ':'), ensure_ascii=False, sort_keys=True) + + +def cache_package_postman_artifact(file_path, package_source, package, relative_path, cache_dir=None, max_artifact_size_mb=20): + kind = postman_kind_for_path(file_path) + max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 + with open(file_path, 'rb') as f: + content = f.read(max_bytes + 1) + if max_bytes <= 0 or len(content) > max_bytes: + raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB') + origin = { + 'package_source': package_source, + 'package_name': package.get('name'), + 'package_version': package.get('version'), + 'package_artifact': package.get('artifact') or package.get('tarball'), + 'path': relative_path, + } + cache_path, digest, size = write_postman_cache(content, kind, origin, cache_dir, max_artifact_size_mb) + return postman_target_from_cached_artifact( + f'{package_source}_package', + kind, + cache_path, + digest, + origin, + path=relative_path, + package_name=package.get('name'), + package_version=package.get('version'), + package_artifact=package.get('artifact') or package.get('tarball'), + size=size, + ) + + +def _postman_package_harvest_limits(max_artifacts=None, max_total_bytes=None, max_elapsed_sec=None): + artifacts = int( + getattr(scan_config, 'postman_package_harvest_max_artifacts', 100) + if max_artifacts is None else max_artifacts + ) + total_bytes = int( + getattr(scan_config, 'postman_package_harvest_max_bytes', 128 * 1024 * 1024) + if max_total_bytes is None else max_total_bytes + ) + elapsed_sec = float( + getattr(scan_config, 'postman_package_harvest_max_elapsed_sec', 30.0) + if max_elapsed_sec is None else max_elapsed_sec + ) + if artifacts <= 0 or total_bytes <= 0 or elapsed_sec <= 0 or not math.isfinite(elapsed_sec): + raise ValueError('Postman package harvest limits must be finite and positive') + return artifacts, total_bytes, elapsed_sec + + +def _add_postman_harvest_warning(warnings, detail, force=False): + message = f'Optional Postman package harvesting degraded: {detail}'[:POSTMAN_HARVEST_WARNING_MAX_CHARS] + if message in warnings: + return + if len(warnings) < POSTMAN_HARVEST_MAX_WARNINGS: + warnings.append(message) + logger.warning(message) + elif force: + warnings[-1] = message + logger.warning(message) + + +def _attach_postman_harvest_warnings(results, warnings): + if not warnings: + return results + merged = list(results.get('warnings') or []) + for warning in warnings: + if warning not in merged: + merged.append(warning) + results['warnings'] = merged + results['warning_classes'] = sorted(set(list(results.get('warning_classes') or []) + ['postman_package_harvest'])) + results['degraded'] = True + return results + + +def _bounded_postman_walk(root_dir, deadline): + stack = [root_dir] + while stack: + _raise_if_scan_slot_fatal() + if time.monotonic() >= deadline: + yield None, (), () + return + directory = stack.pop() + try: + entries = os.scandir(directory) + except OSError: + continue + try: + for entry in entries: + _raise_if_scan_slot_fatal() + if time.monotonic() >= deadline: + yield None, (), () + return + try: + if entry.is_dir(follow_symlinks=False): + if not entry.is_symlink() and not is_reparse_point(entry.path): + stack.append(entry.path) + continue + except OSError: + continue + yield directory, (), (entry.name,) + finally: + entries.close() + + +def find_postman_artifacts( + root_dir, + package_source, + package, + cache_dir=None, + max_artifact_size_mb=20, + warnings=None, + max_artifacts=None, + max_total_bytes=None, + max_elapsed_sec=None, +): + targets = [] + if not root_dir or not os.path.isdir(root_dir): + return targets + warning_sink = warnings if warnings is not None else [] + artifact_limit, total_byte_limit, elapsed_limit = _postman_package_harvest_limits( + max_artifacts, + max_total_bytes, + max_elapsed_sec, + ) + max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 + if max_bytes <= 0: + raise ValueError('Postman artifact limit must be finite and positive') + deadline = time.monotonic() + elapsed_limit + suffixes = tuple(suffix for values in API_ARTIFACT_PATTERNS.values() for suffix in values) + prepared = [] + seen = set() + examined = 0 + examined_bytes = 0 + stopped = False + + walk = _bounded_postman_walk(root_dir, deadline) + for directory, _, files in walk: + _raise_if_scan_slot_fatal() + if directory is None or time.monotonic() >= deadline: + _add_postman_harvest_warning( + warning_sink, + f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)', + force=True, + ) + stopped = True + break + for name in files: + _raise_if_scan_slot_fatal() + if time.monotonic() >= deadline: + _add_postman_harvest_warning( + warning_sink, + f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)', + force=True, + ) + stopped = True + break + lower_name = name.lower() + if not lower_name.endswith(suffixes): + continue + if examined >= artifact_limit: + _add_postman_harvest_warning( + warning_sink, + f'artifact limit of {artifact_limit} reached; remaining matching files were not examined', + force=True, + ) + stopped = True + break + examined += 1 + path = os.path.join(directory, name) + try: + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or os.path.islink(path) or is_reparse_point(path): + raise ValueError('artifact is not a regular unlinked file') + size = int(details.st_size) + relative_path = os.path.relpath(path, root_dir).replace('\\', '/') + if size > max_bytes: + _add_postman_harvest_warning( + warning_sink, + f'skipped {relative_path[:180]} because it exceeds the {max_artifact_size_mb} MB artifact limit', + ) + continue + if examined_bytes + size > total_byte_limit: + _add_postman_harvest_warning( + warning_sink, + f'aggregate byte limit of {total_byte_limit} reached after {examined_bytes} byte(s)', + force=True, + ) + stopped = True + break + examined_bytes += size + + flags = os.O_RDONLY | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0) + descriptor = os.open(path, flags) + try: + opened = os.fstat(descriptor) + opened_identity = (opened.st_dev, opened.st_ino, opened.st_size, getattr(opened, 'st_mtime_ns', None)) + expected_identity = (details.st_dev, details.st_ino, details.st_size, getattr(details, 'st_mtime_ns', None)) + if not stat.S_ISREG(opened.st_mode) or opened_identity != expected_identity: + raise ValueError('artifact changed before harvesting') + content = bytearray() + while len(content) < size: + _raise_if_scan_slot_fatal() + if time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached() + block = os.read(descriptor, min(1024 * 1024, size - len(content))) + if not block: + raise ValueError('artifact changed while harvesting') + content.extend(block) + if os.read(descriptor, 1): + raise ValueError('artifact grew while harvesting') + current = os.stat(path, follow_symlinks=False) + current_identity = (current.st_dev, current.st_ino, current.st_size, getattr(current, 'st_mtime_ns', None)) + if current_identity != opened_identity: + raise ValueError('artifact changed while harvesting') + finally: + os.close(descriptor) + + origin = { + 'package_source': package_source, + 'package_name': package.get('name'), + 'package_version': package.get('version'), + 'package_artifact': package.get('artifact') or package.get('tarball'), + 'path': relative_path, + } + entry = _prepare_postman_cache_entry(bytes(content), postman_kind_for_path(path), origin, max_artifact_size_mb) + if entry['digest'] not in seen: + entry['origin'] = origin + entry['kind'] = postman_kind_for_path(path) + entry['relative_path'] = relative_path + prepared.append(entry) + seen.add(entry['digest']) + except _PostmanHarvestDeadlineReached: + _add_postman_harvest_warning( + warning_sink, + f'elapsed deadline of {elapsed_limit:g}s reached after examining {examined} artifact(s)', + force=True, + ) + stopped = True + break + except PostmanCacheValidationError: + raise + except ScanSlotFatalError: + raise + except Exception as e: + relative_path = os.path.relpath(path, root_dir).replace('\\', '/') + _add_postman_harvest_warning( + warning_sink, + f'unable to harvest {relative_path[:180]}: {str(e)[:160]}', + ) + if stopped: + break + walk.close() + + if prepared and not (stopped and time.monotonic() >= deadline): + _raise_if_scan_slot_fatal() + published, deadline_reached = _publish_postman_cache_entries(prepared, cache_dir, deadline=deadline) + for entry, (cache_path, digest, size) in zip(prepared, published): + targets.append(postman_target_from_cached_artifact( + f'{package_source}_package', + entry['kind'], + cache_path, + digest, + entry['origin'], + path=entry['relative_path'], + package_name=package.get('name'), + package_version=package.get('version'), + package_artifact=package.get('artifact') or package.get('tarball'), + size=size, + )) + if deadline_reached: + _add_postman_harvest_warning( + warning_sink, + f'elapsed deadline of {elapsed_limit:g}s reached while publishing {len(published)} artifact(s)', + force=True, + ) + return targets + + +class GitHubTokenPool: + def __init__(self, token_entries=None, token=None, status=None, code_search_rpm_per_token=8, fallback_cooldown=1800): + entries = [] + for index, entry in enumerate(token_entries or []): + if isinstance(entry, str): + item = {'name': f'github_{index + 1}', 'token': entry} + elif isinstance(entry, dict): + item = dict(entry) + item.setdefault('name', f'github_{index + 1}') + else: + continue + if item.get('token'): + entries.append(item) + if token and not any(item.get('token') == token for item in entries): + entries.append({'name': 'token', 'token': token}) + self.entries = entries + self.status = status if isinstance(status, dict) else {} + self.index = 0 + self.last_code_search_at = {} + self.code_search_interval = 60.0 / max(1, int(code_search_rpm_per_token or 8)) + self.fallback_cooldown = int(fallback_cooldown or 1800) + + def _status_for(self, entry): + return self.status.setdefault(entry.get('name'), {}) + + def _disabled_until(self, entry): + value = self._status_for(entry).get('disabled_until') + if value == 'manual': + return 'manual' + return parse_postman_time(value) + + def _available_entries(self): + now = utc_now_for_postman() + available = [] + for entry in self.entries: + disabled_until = self._disabled_until(entry) + if disabled_until == 'manual': + continue + if disabled_until and disabled_until > now: + continue + available.append(entry) + return available + + def _mark_unavailable(self, entry, category, message, reset_at=None): + item = self._status_for(entry) + if category in ('auth_invalid', 'auth_forbidden'): + item['disabled_until'] = 'manual' + else: + item['disabled_until'] = reset_at or (utc_now_for_postman() + timedelta(seconds=self.fallback_cooldown)).isoformat(timespec='seconds') + item['disabled_reason'] = category + item['last_error'] = str(message or '')[:500] + item['last_failure_at'] = utc_now_for_postman().isoformat(timespec='seconds') + item['failures'] = int(item.get('failures', 0) or 0) + 1 + + def _next_entry(self): + available = self._available_entries() + if not available: + return None + for _ in range(len(self.entries)): + entry = self.entries[self.index % len(self.entries)] + self.index = (self.index + 1) % len(self.entries) + if entry in available: + return entry + return available[0] + + def _sleep_for_resource(self, entry, resource, deadline=None): + if resource != 'code_search': + return + name = entry.get('name') + last = self.last_code_search_at.get(name) + now = time.monotonic() + if last is not None: + delay = self.code_search_interval - (now - last) + if delay > 0: + if deadline is not None and now + delay >= deadline: + remaining = max(0.0, deadline - now) + if remaining: + time.sleep(remaining) + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during code-search pacing') + time.sleep(delay) + self.last_code_search_at[name] = time.monotonic() + + def _wait_until_any_available(self, deadline=None): + waits = [] + now = utc_now_for_postman() + manual_count = 0 + for entry in self.entries: + disabled_until = self._disabled_until(entry) + if disabled_until == 'manual': + manual_count += 1 + if disabled_until and disabled_until != 'manual' and disabled_until > now: + waits.append((disabled_until - now).total_seconds()) + if manual_count >= len(self.entries): + raise RateLimitError('postman', 'All GitHub tokens are manually disabled for Postman discovery', category='auth_invalid', retryable=False, auth_related=True) + wait_for = min(waits) if waits else self.fallback_cooldown + wait_for = max(1, min(wait_for, self.fallback_cooldown)) + if deadline is not None and time.monotonic() + wait_for >= deadline: + remaining = max(0.0, deadline - time.monotonic()) + logger.warning(f'All GitHub tokens unavailable for Postman discovery; deadline in {remaining:.1f}s') + if remaining: + time.sleep(remaining) + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached while all GitHub tokens were unavailable') + logger.warning(f'All GitHub tokens unavailable for Postman discovery; sleeping {wait_for:.0f}s') + time.sleep(wait_for) + + def _token_core_status(self, entry, timeout=10): + request_headers = { + 'Accept': 'application/vnd.github+json', + 'User-Agent': 'GitSecretsScanner/2.0', + 'Authorization': f"Bearer {entry.get('token')}", + } + try: + response = api_request('GET', 'https://api.github.com/user', headers=request_headers, timeout=timeout, max_retries=1) + except Exception: + return 'unknown' + try: + if response.status_code == 200: + return 'valid' + if response.status_code == 401: + return 'invalid' + if response.status_code in (403, 429) and response.headers.get('X-RateLimit-Remaining') == '0': + return 'rate_limited' + return f'http_{response.status_code}' + finally: + response.close() + + def _mark_github_api_error(self, entry, api_error): + category = getattr(api_error, 'category', 'api') + if category not in ('rate_limit', 'secondary_rate_limit', 'auth_invalid', 'auth_forbidden'): + return False + # Code search may return auth-like 401/403 even when the token is valid for core GitHub API. + # Confirm against /user before permanently disabling the token as dead. + if category in ('auth_invalid', 'auth_forbidden'): + core_status = self._token_core_status(entry) + if core_status == 'valid': + self._mark_unavailable(entry, f'{category}_resource', str(api_error), api_error.reset_at) + return True + if core_status in ('unknown', 'rate_limited'): + self._mark_unavailable(entry, f'{category}_unconfirmed', str(api_error), api_error.reset_at) + return True + self._mark_unavailable(entry, category, str(api_error), api_error.reset_at) + return True + + def request(self, method, url, params=None, headers=None, timeout=30, resource='core', deadline=None, use_proxy=None): + if not self.entries: + raise RateLimitError('postman', 'GitHub token is required for Postman GitHub code search', category='auth_invalid', retryable=False, auth_related=True) + attempts = 0 + last_error = None + while attempts < max(1, len(self.entries) * 2): + if deadline is not None and time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GitHub request') + entry = self._next_entry() + if not entry: + self._wait_until_any_available(deadline) + attempts += 1 + continue + self._sleep_for_resource(entry, resource, deadline) + request_headers = {'Accept': 'application/vnd.github+json', 'User-Agent': 'GitSecretsScanner/2.0'} + request_headers.update(headers or {}) + request_headers['Authorization'] = f"Bearer {entry.get('token')}" + try: + response = api_request( + method, url, headers=request_headers, params=params, timeout=timeout, deadline=deadline, + use_proxy=use_proxy, + ) + if response.status_code in (401, 403, 429): + api_error = github_api_error(response) + response.close() + if self._mark_github_api_error(entry, api_error): + last_error = api_error + attempts += 1 + continue + try: + response.raise_for_status() + except requests.exceptions.HTTPError as e: + raise github_api_error(e.response) from e + return response + except (requests.exceptions.RequestException, ApiRequestError) as e: + last_error = e + logger.warning(f'GitHub request failed for Postman discovery with token {entry.get("name")}: {str(e)[:300]}') + attempts += 1 + delay = min(2, attempts) + if deadline is not None and time.monotonic() + delay >= deadline: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GitHub request retry') from e + time.sleep(delay) + continue + if last_error: + raise RateLimitError('postman', f'GitHub Postman discovery request failed after retries: {last_error}', category='network', retryable=True, auth_related=False) + raise RateLimitError('postman', 'All GitHub tokens are unavailable for Postman discovery', category='rate_limit', retryable=True, auth_related=True) + + +def latest_github_path_commit(repo, path, pool, request_timeout=20, deadline=None): + response = pool.request( + 'GET', + f'https://api.github.com/repos/{repo}/commits', + params={'path': path, 'per_page': 1}, + timeout=request_timeout, + resource='core', + deadline=deadline, + ) + data = response.json() + if not data: + return None + commit = data[0].get('commit') or {} + committer = commit.get('committer') or {} + author = commit.get('author') or {} + return committer.get('date') or author.get('date') + + +def github_content_bytes(item, pool, request_timeout=20, max_artifact_size_mb=20, deadline=None): + response = pool.request( + 'GET', item.get('url'), timeout=request_timeout, resource='core', deadline=deadline, + use_proxy=False, + ) + data = response.json() + size = int(data.get('size') or 0) + max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 + if max_bytes and size > max_bytes: + raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB') + content = str(data.get('content') or '') + if str(data.get('encoding') or '').lower() == 'base64': + decoded = base64.b64decode(re.sub(r'\s+', '', content)) + else: + decoded = content.encode('utf-8', errors='replace') + if max_bytes <= 0 or len(decoded) > max_bytes: + raise ValueError(f'Postman artifact exceeds {max_artifact_size_mb} MB') + return decoded + + +def github_postman_item_to_target(item, kind, cache_path, digest, size=None, commit_date=None): + repo = (item.get('repository') or {}).get('full_name') or '' + path = item.get('path') or '' + sha = item.get('sha') or digest or '' + origin = { + 'provider': 'github', + 'repo': repo, + 'path': path, + 'sha': sha, + 'html_url': item.get('html_url'), + 'api_url': item.get('url'), + 'commit_date': commit_date, + } + return postman_target_from_cached_artifact( + 'github_code', + kind, + cache_path, + digest, + origin, + repo=repo, + path=path, + sha=sha, + html_url=item.get('html_url'), + api_url=item.get('url'), + commit_date=commit_date, + size=size, + ) + + +def fetch_github_postman_targets(query, pages=1, per_page=100, token_entries=None, token=None, search_kinds=None, cache_dir=None, max_file_age_days=365, max_artifact_size_mb=20, request_timeout=20, code_search_rpm_per_token=8, all_tokens_cooldown=1800, auth_status=None, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None): + if not query: + return [] + budget = _PostmanDiscoveryBudget( + 'GitHub code-search', + discovery_max_artifacts, + discovery_max_artifacts_per_page, + discovery_max_bytes, + discovery_max_elapsed_sec, + ) + pool = GitHubTokenPool(token_entries, token, auth_status, code_search_rpm_per_token, all_tokens_cooldown) + kinds = normalize_postman_search_kinds(search_kinds) + per_page = max(1, min(int(per_page or 100), 100)) + max_pages = min(max(1, int(pages or 1)), max(1, (1000 + per_page - 1) // per_page)) + cutoff = utc_now_for_postman() - timedelta(days=int(max_file_age_days or 0)) if int(max_file_age_days or 0) > 0 else None + targets = [] + seen_targets = set() + seen_digests = set() + stop_cycle = False + for kind in kinds: + if stop_cycle or not budget.check_deadline(): + break + templates = API_ARTIFACT_SEARCHES.get(kind) or API_ARTIFACT_SEARCHES['generic'] + for template in templates: + if stop_cycle or not budget.check_deadline(): + break + search_query = template.format(query=query).strip() + logger.info(f"Fetching GitHub API artifact {kind} targets for: {search_query!r}") + known_pages = 0 + for page in range(1, max_pages + 1): + if not budget.check_deadline(): + stop_cycle = True + break + try: + response = pool.request( + 'GET', + 'https://api.github.com/search/code', + params={'q': search_query, 'per_page': per_page, 'page': page, 'sort': 'indexed', 'order': 'desc'}, + timeout=budget.request_timeout(request_timeout), + resource='code_search', + deadline=budget.deadline, + ) + except _PostmanHarvestDeadlineReached as e: + budget.stop(str(e)) + stop_cycle = True + break + except RateLimitError as e: + if getattr(e, 'category', '') == 'network': + logger.warning(f'GitHub API artifact {kind} network failure on page {page}: {str(e)[:300]}') + raise + payload = response.json() + if not isinstance(payload, dict) or 'items' not in payload or not isinstance(payload.get('items'), list): + raise ApiRequestError('invalid GitHub code search payload') + items = payload.get('items') or [] + if not items: + logger.info(f'GitHub API artifact {kind} page {page}: no results') + break + page_targets = [] + prepared = [] + page_artifacts = 0 + for item in items: + admitted, scope = budget.admit(page_artifacts) + if not admitted: + stop_cycle = scope == 'cycle' + break + page_artifacts += 1 + repo = (item.get('repository') or {}).get('full_name') or '' + path = item.get('path') or '' + commit_date = None + if cutoff: + try: + commit_date = latest_github_path_commit( + repo, path, pool, budget.request_timeout(request_timeout), budget.deadline, + ) + parsed_commit = parse_postman_time(commit_date) + if not parsed_commit or parsed_commit < cutoff: + continue + except _PostmanHarvestDeadlineReached as e: + budget.stop(str(e)) + stop_cycle = True + break + except RateLimitError as e: + if getattr(e, 'category', '') in ('network', 'not_found'): + logger.warning(f'Skipping API artifact freshness check for {repo}:{path}: {str(e)[:300]}') + continue + raise + except Exception as e: + logger.warning(f'Unable to check API artifact freshness for {repo}:{path}: {str(e)}') + continue + try: + content = github_content_bytes( + item, + pool, + budget.request_timeout(request_timeout), + max_artifact_size_mb, + budget.deadline, + ) + if not budget.account_bytes(len(content)): + stop_cycle = True + break + origin = {'provider': 'github', 'repo': repo, 'path': path, 'sha': item.get('sha'), 'html_url': item.get('html_url'), 'api_url': item.get('url'), 'commit_date': commit_date} + detected_kind = postman_kind_for_path(path) + entry = _prepare_postman_cache_entry(content, detected_kind or kind, origin, max_artifact_size_mb) + if entry['digest'] in seen_digests: + continue + seen_digests.add(entry['digest']) + prepared.append({ + 'entry': entry, + 'item': item, + 'kind': detected_kind or kind, + 'commit_date': commit_date, + }) + except _PostmanHarvestDeadlineReached as e: + budget.stop(str(e)) + stop_cycle = True + break + except RateLimitError as e: + if getattr(e, 'category', '') in ('network', 'not_found'): + logger.warning(f'Skipping GitHub API artifact for {repo}:{path}: {str(e)[:300]}') + continue + raise + except Exception as e: + if isinstance(e, PostmanCacheCapacityError): + raise + logger.warning(f'Unable to cache GitHub API artifact {repo}:{path}: {str(e)}') + published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget) + for record, (cache_path, digest, size) in published: + target = github_postman_item_to_target( + record['item'], record['kind'], cache_path, digest, size, record['commit_date'], + ) + identity = postman_target_identity(target) + if identity not in seen_targets: + targets.append(target) + page_targets.append(target) + seen_targets.add(identity) + logger.info(f'GitHub API artifact {kind} page {page}: fetched {len(items)}, queued candidates {len(page_targets)}') + if stop_on_seen_pages and page >= max(1, int(min_pages_before_stop or 1)): + if page_targets and page_is_known( + page_targets, known_targets, normalize_target, known_target_lookup, + ): + known_pages += 1 + logger.info(f'GitHub API artifact {kind} page {page}: all targets known ({known_pages}/{seen_page_threshold})') + if known_pages >= max(1, int(seen_page_threshold or 1)): + logger.info(f'Stopping API artifact {kind} pagination early after {known_pages} known page(s)') + break + else: + known_pages = 0 + if deadline_reached or stop_cycle or budget.exhausted(): + stop_cycle = True + break + return targets + + +def fetch_github_repo_items(query, pages, per_page=100, token=None, sort_by="updated", sort_order="desc", created_filter="any", raise_rate_limit=False, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None): + """Fetch GitHub repositories with metadata for filtering.""" + repos = [] + headers = {'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/vnd.github.v3+json'} + if token: + headers['Authorization'] = f'Bearer {token}' + + # Build search query with filters + search_query = query if query else "*" + + # Add created date filter + if created_filter != "any": + date_filters = { + "today": "created:>{}".format((datetime.now() - timedelta(days=1)).strftime("%Y-%m-%d")), + "week": "created:>{}".format((datetime.now() - timedelta(weeks=1)).strftime("%Y-%m-%d")), + "month": "created:>{}".format((datetime.now() - timedelta(days=30)).strftime("%Y-%m-%d")), + "year": "created:>{}".format((datetime.now() - timedelta(days=365)).strftime("%Y-%m-%d")) + } + if created_filter in date_filters: + search_query += f" {date_filters[created_filter]}" + + logger.info(f"Fetching GitHub repositories for: '{search_query}' sorted by {sort_by} ({sort_order})...") + + seen_pages = 0 + successful_pages = 0 + for page in range(1, pages + 1): + url = "https://api.github.com/search/repositories" + params = { + 'q': search_query, + 'sort': sort_by, + 'order': sort_order, + 'per_page': per_page, + 'page': page, + } + + try: + response = api_request('GET', url, headers=headers, params=params, timeout=30) + response.raise_for_status() + + # Handle rate limits + if response.status_code == 403 and 'X-RateLimit-Remaining' in response.headers: + if int(response.headers['X-RateLimit-Remaining']) == 0: + reset_time = datetime.fromtimestamp(int(response.headers['X-RateLimit-Reset'])) + wait_seconds = (reset_time - datetime.now()).total_seconds() + 10 + logger.warning(f"Rate limit exceeded. Resuming at {reset_time}. Waiting {wait_seconds:.0f} seconds...") + time.sleep(wait_seconds) + continue + + data = response.json() + if not isinstance(data, dict) or 'items' not in data or not isinstance(data.get('items'), list): + raise ValueError('invalid GitHub repository search payload') + successful_pages += 1 + if not data.get('items'): + logger.info(f"Page {page} returned no results. Stopping.") + break + + page_repos = [github_repo_to_target(item) for item in data['items'] if item.get('clone_url')] + repos.extend(page_repos) + logger.info(f"Page {page}: Fetched {len(data['items'])} repositories") + if stop_on_seen_pages and page >= max(1, min_pages_before_stop): + if page_is_known( + [item.get('url') for item in page_repos], known_targets, + normalize_target, known_target_lookup, + ): + seen_pages += 1 + logger.info(f"Page {page}: all GitHub repositories are already queued/checked ({seen_pages}/{seen_page_threshold})") + if seen_pages >= max(1, seen_page_threshold): + logger.info(f"Stopping GitHub pagination early after {seen_pages} all-known page(s)") + break + else: + seen_pages = 0 + + except requests.exceptions.HTTPError as e: + api_error = github_api_error(e.response) + if raise_rate_limit: + raise api_error from e + logger.error(str(api_error)) + raise api_error from e + except Exception as e: + if isinstance(e, ApiRequestError): + raise + logger.error(f"Error fetching page {page}: {str(e)}") + raise ApiRequestError(f'GitHub discovery payload failed: {e}') from e + + return repos + +def fetch_github_repos(query, pages, per_page=100, token=None, sort_by="updated", sort_order="desc", created_filter="any", raise_rate_limit=False, **kwargs): + """Fetch GitHub repository clone URLs with pagination and authentication.""" + return [ + item['url'] for item in fetch_github_repo_items( + query, pages, per_page, token, sort_by, sort_order, created_filter, raise_rate_limit, **kwargs + ) + if item.get('url') + ] + + +GHARCHIVE_REPO_TERMS = ( + 'ai', 'agent', 'assistant', 'bot', 'chat', 'chatbot', 'gpt', 'llm', 'rag', 'mcp', + 'model', 'inference', 'embedding', 'vector', 'semantic', 'prompt', 'workflow', + 'copilot', 'codegen', 'langchain', 'llamaindex', 'litellm', 'ollama', 'vllm', + 'claude', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'grok', 'xai', 'openrouter', 'replicate', + 'deepseek', 'zai', 'glm', 'zhipu', 'dashscope', 'bedrock', 'vertex', 'foundry', + 'aiplatform', 'boto3', 'terraform', 'cloudbuild', 'service-account', 'credentials', +) +GHARCHIVE_PATH_TERMS = ( + '.env', 'env.', 'secret', 'secrets', 'credential', 'credentials', 'apikey', 'api_key', + 'token', 'tokens', 'key', 'keys', 'config', 'settings', 'appsettings', 'docker-compose', + 'compose.yaml', 'compose.yml', '.github/workflows', 'workflow', 'deploy', 'deployment', + 'kubernetes', 'k8s', 'helm', 'terraform', 'tfvars', 'notebook', '.ipynb', 'postman', + 'collection.json', 'environment.json', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', + 'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope', + 'aws_access_key_id', 'aws_secret_access_key', 'bedrock-runtime', 'boto3', + 'google_application_credentials', 'service-account', 'service_account', 'credentials.json', + 'application_default_credentials', 'vertexai', 'aiplatform', 'provider.tf', 'cloudbuild.yaml', +) +GHARCHIVE_FILE_FETCH_TERMS = ( + '.env', 'env.', '.env.', '.env-', 'secret', 'secrets', 'credential', 'credentials', + 'credentials.json', 'service-account', 'service_account', 'application_default_credentials', + 'google_application_credentials', 'terraform.tfvars', '.tfvars', 'provider.tf', + 'cloudbuild.yaml', 'cloudbuild.yml', '.github/workflows', 'docker-compose', + 'compose.yml', 'compose.yaml', 'appsettings', '.ipynb', 'notebook', + 'bedrock', 'vertex', 'aiplatform', 'boto3', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', + 'groq', 'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope', +) +GHARCHIVE_FILE_SKIP_SUFFIXES = ( + '.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg', '.ico', '.pdf', '.zip', '.gz', '.tgz', + '.tar', '.rar', '.7z', '.bin', '.safetensors', '.pt', '.pth', '.onnx', '.parquet', '.arrow', + '.mp4', '.mov', '.avi', '.mp3', '.wav', '.lock', '.sum', '.min.js', '.map', '.pyc', +) +GHARCHIVE_FILE_SKIP_PARTS = ( + '/__pycache__/', '/node_modules/', '/vendor/', '/.git/', '/dist/', '/build/', '/target/', +) +GITHUB_GIST_FILE_TERMS = ( + '.env', 'env', 'secret', 'secrets', 'credential', 'credentials', 'apikey', 'api_key', + 'token', 'tokens', 'key', 'keys', 'config', 'settings', 'appsettings', 'credentials.json', + 'service-account', 'service_account', 'application_default_credentials', 'terraform', + 'tfvars', 'provider.tf', 'docker-compose', 'compose.yml', 'compose.yaml', 'workflow', + 'cloudbuild', 'ipynb', 'notebook', 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', + 'xai', 'grok', 'openrouter', 'replicate', 'deepseek', 'zai', 'glm', 'dashscope', + 'bedrock', 'vertex', 'aiplatform', 'postman', +) +GHARCHIVE_MESSAGE_TERMS = ( + 'api key', 'apikey', 'token', 'secret', 'credential', '.env', 'config', 'settings', + 'openai', 'anthropic', 'gemini', 'qwen', 'kimi', 'moonshot', 'groq', 'xai', 'grok', 'openrouter', + 'replicate', 'deepseek', 'zai', 'glm', 'dashscope', 'bedrock', 'vertex', + 'aws_access_key_id', 'aws_secret_access_key', 'google_application_credentials', + 'service account', 'credentials.json', 'vertexai', 'aiplatform', 'bedrock-runtime', + 'terraform', 'tfvars', 'cloudbuild', +) +GHARCHIVE_MAX_HOURS_BACK = 168 + + +def _contains_term(text, terms): + text = str(text or '').lower() + return any(term in text for term in terms) + + +def github_archive_event_score(event): + repo = event.get('repo') if isinstance(event.get('repo'), dict) else {} + repo_name = str(repo.get('name') or '').lower() + score = 0 + if _contains_term(repo_name.replace('-', ' ').replace('_', ' '), GHARCHIVE_REPO_TERMS): + score += 8 + payload = event.get('payload') if isinstance(event.get('payload'), dict) else {} + ref = str(payload.get('ref') or '').lower() + if _contains_term(ref, GHARCHIVE_REPO_TERMS): + score += 3 + commits = payload.get('commits') if isinstance(payload.get('commits'), list) else [] + for commit in commits[:20]: + if not isinstance(commit, dict): + continue + message = str(commit.get('message') or '') + if _contains_term(message, GHARCHIVE_MESSAGE_TERMS): + score += 5 + for key in ('added', 'modified', 'removed'): + paths = commit.get(key) if isinstance(commit.get(key), list) else [] + for path in paths[:50]: + if _contains_term(path, GHARCHIVE_PATH_TERMS): + score += 4 + if event.get('type') == 'PushEvent': + score += 1 + return score + + +def github_archive_event_branch(event): + payload = event.get('payload') if isinstance(event.get('payload'), dict) else {} + ref = str(payload.get('ref') or '').strip() + if ref.startswith('refs/heads/'): + return ref[len('refs/heads/'):], ref + if payload.get('ref_type') == 'branch' and ref: + return ref, f'refs/heads/{ref}' + return '', ref + + +def github_archive_target_payload(repo_name, event, score, archive_hour): + payload = event.get('payload') if isinstance(event.get('payload'), dict) else {} + branch, ref = github_archive_event_branch(event) + data = { + 'url': f'https://github.com/{repo_name}.git', + 'repo': repo_name, + 'source': 'gharchive', + 'event_type': event.get('type') or '', + 'archive_hour': archive_hour.isoformat(), + 'score': int(score or 0), + 'ref': ref, + 'branch': branch, + 'head_sha': payload.get('head') or payload.get('after') or '', + } + return json.dumps(data, separators=(',', ':'), ensure_ascii=False, sort_keys=True) + + +def gharchive_changed_paths(event): + payload = event.get('payload') if isinstance(event.get('payload'), dict) else {} + commits = payload.get('commits') if isinstance(payload.get('commits'), list) else [] + for commit in commits[:20]: + if not isinstance(commit, dict): + continue + sha = commit.get('sha') or commit.get('id') or payload.get('head') or payload.get('after') or '' + for key in ('added', 'modified'): + paths = commit.get(key) if isinstance(commit.get(key), list) else [] + for path in paths[:80]: + path = str(path or '').strip().replace('\\', '/') + if path: + yield sha, path + + +def gharchive_commit_urls(event, repo_name): + payload = event.get('payload') if isinstance(event.get('payload'), dict) else {} + commits = payload.get('commits') if isinstance(payload.get('commits'), list) else [] + seen = set() + for commit in commits[:5]: + if not isinstance(commit, dict): + continue + sha = commit.get('sha') or commit.get('id') or '' + url = commit.get('url') or (f'https://api.github.com/repos/{repo_name}/commits/{sha}' if sha else '') + if sha and url and sha not in seen: + seen.add(sha) + yield sha, url + head = payload.get('head') or payload.get('after') or '' + if head and head not in seen: + yield head, f'https://api.github.com/repos/{repo_name}/commits/{head}' + + +def fetch_github_commit_files(event, repo_name, headers, request_timeout=20, deadline=None): + for sha, url in gharchive_commit_urls(event, repo_name): + if deadline is not None and time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during commit lookup') + timeout = request_timeout + if deadline is not None: + timeout = max(0.01, min(float(request_timeout or 20), deadline - time.monotonic())) + try: + response = api_request( + 'GET', url, headers=headers, timeout=timeout, max_retries=2, + retry_delay=1, deadline=deadline, + ) + if response.status_code == 404: + continue + if response.status_code in (401, 403, 429): + raise github_api_error(response) + response.raise_for_status() + data = response.json() + except (requests.exceptions.RequestException, ApiRequestError) as e: + logger.warning(f'Unable to fetch commit files {repo_name}@{sha}: {str(e)[:300]}') + continue + files = data.get('files') if isinstance(data.get('files'), list) else [] + for item in files[:100]: + if not isinstance(item, dict): + continue + status = str(item.get('status') or '') + if status not in ('added', 'modified', 'renamed'): + continue + path = item.get('filename') or item.get('previous_filename') or '' + raw_url = item.get('raw_url') or github_raw_url(repo_name, sha, path) + if path and raw_url: + yield sha, str(path).replace('\\', '/'), raw_url + + +def gharchive_path_interesting(path): + lowered = str(path or '').lower() + if not lowered or lowered.endswith(GHARCHIVE_FILE_SKIP_SUFFIXES): + return False + normalized = '/' + lowered.strip('/') + if any(part in normalized for part in GHARCHIVE_FILE_SKIP_PARTS): + return False + return _contains_term(lowered, GHARCHIVE_FILE_FETCH_TERMS) + + +def github_raw_url(repo_name, sha, path): + if not repo_name or not sha or not path: + return '' + return f'https://raw.githubusercontent.com/{repo_name}/{sha}/{quote(path)}' + + +def gharchive_file_target(source, kind, cache_path, digest, origin=None, **extra): + return postman_target_from_cached_artifact(source, kind, cache_path, digest, origin, **extra) + + +class GHArchiveBoundsError(ValueError): + pass + + +class GHArchiveCacheCapacityError(RuntimeError): + pass + + +def gharchive_item_lock_path(cache_dir, artifact_path): + digest = hashlib.sha256(canonical_path(artifact_path).encode('utf-8', errors='strict')).hexdigest() + return os.path.join(cache_dir, f'.item-lock-{int(digest[:8], 16) % 64:02d}.lock') + + +def _gharchive_limits(): + return { + 'max_items': max(1, int(scan_config.gharchive_cache_max_items)), + 'max_bytes': max(1, int(scan_config.gharchive_cache_max_bytes)), + 'min_free_bytes': max(0, int(scan_config.gharchive_cache_min_free_bytes)), + 'download_max_bytes': max(1, int(scan_config.gharchive_download_max_bytes)), + 'decompressed_max_bytes': max(1, int(scan_config.gharchive_decompressed_max_bytes)), + 'max_events': max(1, int(scan_config.gharchive_max_events)), + 'max_line_bytes': max(1, int(scan_config.gharchive_max_line_bytes)), + 'lock_timeout_sec': max(1, int(scan_config.gharchive_cache_lock_timeout_sec)), + } + + +def require_gharchive_cache_dir(cache_dir=None): + cache_dir = canonical_path(cache_dir or scan_config.gharchive_cache_dir) + runtime_dir = canonical_path(scan_config.runtime_dir) + require_private_directory(runtime_dir, create=False) + try: + contained = os.path.commonpath((runtime_dir, cache_dir)) == runtime_dir + except ValueError: + contained = False + if not contained or cache_dir == runtime_dir: + raise RuntimeError('GHArchive cache must be a dedicated private directory under runtime_dir') + return require_private_directory(cache_dir, create=True) + + +def iter_gharchive_lines(path, decompressed_max_bytes=None, max_events=None, max_line_bytes=None, deadline=None): + _raise_if_scan_slot_fatal() + limits = _gharchive_limits() + decompressed_max_bytes = max(1, int(decompressed_max_bytes or limits['decompressed_max_bytes'])) + max_events = max(1, int(max_events or limits['max_events'])) + max_line_bytes = max(1, int(max_line_bytes or limits['max_line_bytes'])) + total = 0 + events = 0 + with gzip.open(path, 'rb') as archive: + while True: + _raise_if_scan_slot_fatal() + if deadline is not None and time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached while reading GHArchive data') + raw_line = archive.readline(max_line_bytes + 1) + if not raw_line: + break + if len(raw_line) > max_line_bytes: + raise GHArchiveBoundsError('GHArchive line exceeds configured byte limit') + total += len(raw_line) + if total > decompressed_max_bytes: + raise GHArchiveBoundsError('GHArchive decompressed bytes exceed configured limit') + events += 1 + if events > max_events: + raise GHArchiveBoundsError('GHArchive event count exceeds configured limit') + yield raw_line + + +def validate_gharchive_gzip(path, decompressed_max_bytes=None, max_events=None, max_line_bytes=None, deadline=None): + for _ in iter_gharchive_lines(path, decompressed_max_bytes, max_events, max_line_bytes, deadline): + pass + return True + + +def gharchive_cache_usage(cache_dir): + usage = {'items': 0, 'files': 0, 'bytes': 0} + with os.scandir(cache_dir) as entries: + for entry in entries: + if entry.is_symlink() or is_reparse_point(entry.path) or not entry.is_file(follow_symlinks=False): + continue + size = entry.stat(follow_symlinks=False).st_size + usage['files'] += 1 + usage['bytes'] += max(0, int(size)) + if entry.name.endswith('.json.gz'): + usage['items'] += 1 + return usage + + +def _gharchive_artifacts_oldest(cache_dir, protected): + candidates = [] + with os.scandir(cache_dir) as entries: + for entry in entries: + canonical = canonical_path(entry.path) + if ( + canonical in protected + or not entry.name.endswith('.json.gz') + or entry.is_symlink() + or is_reparse_point(entry.path) + or not entry.is_file(follow_symlinks=False) + ): + continue + details = entry.stat(follow_symlinks=False) + candidates.append((details.st_mtime_ns, canonical)) + return [path for _, path in sorted(candidates)] + + +def _gharchive_evict_for_capacity(cache_dir, required_items=0, required_bytes=0, protected=None): + limits = _gharchive_limits() + protected = {canonical_path(path) for path in (protected or set())} + while True: + usage = gharchive_cache_usage(cache_dir) + try: + free_bytes = shutil.disk_usage(cache_dir).free + except OSError as exc: + raise GHArchiveCacheCapacityError(f'unable to inspect GHArchive cache free space: {exc}') from exc + if ( + usage['items'] + int(required_items) <= limits['max_items'] + and usage['bytes'] + int(required_bytes) <= limits['max_bytes'] + and free_bytes - int(required_bytes) >= limits['min_free_bytes'] + ): + return usage + evicted = False + for candidate in _gharchive_artifacts_oldest(cache_dir, protected): + item_lock = PrivateFileLock(gharchive_item_lock_path(cache_dir, candidate)) + try: + item_lock.acquire() + except BlockingIOError: + continue + try: + if os.path.isfile(candidate): + reject_reparse_components(candidate) + os.remove(candidate) + evicted = True + break + finally: + item_lock.release() + if not evicted: + raise GHArchiveCacheCapacityError('GHArchive cache quota or free-space reserve cannot be satisfied without evicting an active entry') + + +def _acquire_cached_gharchive_hour(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None): + cache_dir = require_gharchive_cache_dir(cache_dir) + limits = _gharchive_limits() + name = f'{hour:%Y-%m-%d-%H}.json.gz' + path = canonical_path(os.path.join(cache_dir, name)) + global_lock_path = os.path.join(cache_dir, '.cache.lock') + if deadline is not None and time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive cache access') + lock_timeout = limits['lock_timeout_sec'] + if deadline is not None: + lock_timeout = min(lock_timeout, max(0.01, deadline - time.monotonic())) + try: + global_lock = acquire_file_lock(global_lock_path, timeout_sec=lock_timeout) + except TimeoutError as exc: + if deadline is not None and time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached waiting for GHArchive cache lock') from exc + raise + item_lock = None + try: + item_lock_timeout = limits['lock_timeout_sec'] + if deadline is not None: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive item lock') + item_lock_timeout = min(item_lock_timeout, max(0.01, remaining)) + try: + item_lock = acquire_file_lock( + gharchive_item_lock_path(cache_dir, path), + timeout_sec=item_lock_timeout, + ) + except TimeoutError as exc: + if deadline is not None and time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached waiting for GHArchive item lock') from exc + raise + for entry in os.scandir(cache_dir): + if entry.name.startswith(name + '.part.') and entry.is_file(follow_symlinks=False): + remove_file_quiet(entry.path) + if os.path.lexists(path): + reject_reparse_components(path) + if not private_file_ready(path): + raise RuntimeError(f'GHArchive cache file is not private: {path}') + try: + validate_gharchive_gzip(path, deadline=deadline) + os.utime(path, None) + return path, item_lock + except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError): + remove_file_quiet(path) + + _gharchive_evict_for_capacity(cache_dir, required_items=1, protected={path}) + url = f'https://data.gharchive.org/{name}' + last_error = None + attempts = max(1, int(retries or 1)) + for attempt in range(attempts): + _raise_if_scan_slot_fatal() + if deadline is not None and time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached before GHArchive download') + temp_path = f'{path}.part.{os.getpid()}.{attempt}' + remove_file_quiet(temp_path) + response = None + try: + logger.info(f'Downloading GHArchive {name} attempt {attempt + 1}/{attempts}') + request_deadline = None if deadline is None else max(0.01, deadline - time.monotonic()) + response = _direct_request( + 'GET', url, + headers={'User-Agent': 'GitSecretsScanner/2.0'}, + stream=True, + timeout=( + min(10, request_deadline) if request_deadline is not None else 10, + min(max(30, int(request_timeout or 120)), request_deadline) if request_deadline is not None else max(30, int(request_timeout or 120)), + ), + ) + if _scan_slot_fatal_event.is_set(): + _raise_if_scan_slot_fatal() + if response.status_code == 404: + return None, item_lock + response.raise_for_status() + declared = response.headers.get('Content-Length') + if declared and int(declared) > limits['download_max_bytes']: + raise GHArchiveBoundsError('GHArchive compressed response exceeds configured byte limit') + total = 0 + with open(temp_path, 'xb') as output: + for chunk in response.iter_content(chunk_size=1024 * 1024): + _raise_if_scan_slot_fatal() + if deadline is not None and time.monotonic() >= deadline: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GHArchive download') + if not chunk: + continue + total += len(chunk) + if total > limits['download_max_bytes']: + raise GHArchiveBoundsError('GHArchive compressed response exceeds configured byte limit') + _gharchive_evict_for_capacity( + cache_dir, + required_bytes=len(chunk), + protected={path, temp_path}, + ) + output.write(chunk) + output.flush() + os.fsync(output.fileno()) + _raise_if_scan_slot_fatal() + harden_private_file(temp_path) + validate_gharchive_gzip(temp_path, deadline=deadline) + _raise_if_scan_slot_fatal() + os.replace(temp_path, path) + harden_private_file(path) + return path, item_lock + except _PostmanHarvestDeadlineReached: + remove_file_quiet(temp_path) + raise + except ( + requests.RequestException, OSError, EOFError, ValueError, gzip.BadGzipFile, + GHArchiveBoundsError, GHArchiveCacheCapacityError, + urllib3_exceptions.ProtocolError, urllib3_exceptions.ReadTimeoutError, + ) as exc: + last_error = exc + remove_file_quiet(temp_path) + if attempt + 1 < attempts: + delay = min(30, 2 ** attempt) + if deadline is not None and time.monotonic() + delay >= deadline: + raise _PostmanHarvestDeadlineReached('Postman discovery deadline reached during GHArchive retry') from exc + _wait_or_raise_scan_slot_fatal(delay) + finally: + if response is not None: + response.close() + raise ApiRequestError(f'GHArchive download failed after {attempts} attempt(s): {name}: {last_error}') + except BaseException: + if item_lock is not None: + release_file_lock(item_lock, item_lock.path) + raise + finally: + release_file_lock(global_lock, global_lock_path) + + +def cached_gharchive_hour(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None): + path, item_lock = _acquire_cached_gharchive_hour(hour, cache_dir, request_timeout, retries, deadline) + if item_lock is not None: + release_file_lock(item_lock, item_lock.path) + return path + + +@contextmanager +def cached_gharchive_hour_reader(hour, cache_dir=None, request_timeout=120, retries=4, deadline=None): + path = None + item_lock = None + try: + path, item_lock = _acquire_cached_gharchive_hour(hour, cache_dir, request_timeout, retries, deadline) + yield path + finally: + if item_lock is not None: + release_file_lock(item_lock, item_lock.path) + + +def fetch_github_archive_repos(hours_back=6, max_repos=200, event_types=None, request_timeout=120, archive_cache_dir=None): + """Fetch recently active public GitHub repositories from GHArchive hourly dumps.""" + event_types = {str(item).strip() for item in (event_types or []) if str(item).strip()} + if not event_types: + event_types = {'PushEvent', 'CreateEvent', 'PublicEvent'} + hours_back = min(GHARCHIVE_MAX_HOURS_BACK, max(1, int(hours_back or 1))) + max_repos = max(1, int(max_repos or 1)) + now = datetime.now(timezone.utc).replace(minute=0, second=0, microsecond=0) + candidates = {} + order = 0 + for offset in range(1, hours_back + 1): + hour = now - timedelta(hours=offset) + url = f'https://data.gharchive.org/{hour:%Y-%m-%d-%H}.json.gz' + logger.info(f'Fetching GHArchive hour {hour.isoformat()} from {url}') + with cached_gharchive_hour_reader(hour, archive_cache_dir, request_timeout) as archive_path: + if not archive_path: + logger.info(f'GHArchive hour unavailable yet: {url}') + continue + try: + for raw_line in iter_gharchive_lines(archive_path): + try: + event = json.loads(raw_line.decode('utf-8', errors='replace')) + except (ValueError, UnicodeDecodeError): + continue + if event.get('type') not in event_types: + continue + repo = event.get('repo') if isinstance(event.get('repo'), dict) else {} + name = str(repo.get('name') or '').strip() + if not name or '/' not in name: + continue + if not re.match(r'^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$', name): + continue + key = name.lower() + order += 1 + score = github_archive_event_score(event) + existing = candidates.get(key) + if existing and score > existing['score']: + candidates[key] = { + 'name': name, + 'score': score, + 'order': existing['order'], + 'event': event, + 'archive_hour': hour, + } + elif not existing: + candidate = { + 'name': name, + 'score': score, + 'order': order, + 'event': event, + 'archive_hour': hour, + } + if len(candidates) < max_repos: + candidates[key] = candidate + else: + worst_key = min( + candidates, + key=lambda value: (candidates[value]['score'], -candidates[value]['order']), + ) + worst = candidates[worst_key] + if (score, -order) > (worst['score'], -worst['order']): + del candidates[worst_key] + candidates[key] = candidate + except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError, urllib3_exceptions.ProtocolError) as e: + remove_file_quiet(archive_path) + raise ApiRequestError(f'Unable to read cached GHArchive gzip {url}: {e}') from e + ranked = sorted(candidates.values(), key=lambda item: (-item['score'], item['order'])) + selected = ranked[:max_repos] + positive = sum(1 for item in selected if item['score'] > 0) + logger.info(f'GHArchive selected {len(selected)} repos from {len(candidates)} candidates; positive_score={positive}') + return [github_archive_target_payload(item['name'], item['event'], item['score'], item['archive_hour']) for item in selected] + + +def fetch_github_archive_file_targets(hours_back=6, max_files=300, event_types=None, request_timeout=120, cache_dir=None, max_file_size_mb=2, token=None, max_commit_lookups=200, archive_cache_dir=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None): + event_types = {str(item).strip() for item in (event_types or []) if str(item).strip()} or {'PushEvent'} + hours_back = min(GHARCHIVE_MAX_HOURS_BACK, max(1, int(hours_back or 1))) + max_files = max(1, int(max_files or 1)) + max_bytes = int(max_file_size_mb or 2) * 1024 * 1024 + budget = _PostmanDiscoveryBudget( + 'GHArchive-file', + discovery_max_artifacts, + discovery_max_artifacts_per_page, + discovery_max_bytes, + discovery_max_elapsed_sec, + ) + now = datetime.now(timezone.utc).replace(minute=0, second=0, microsecond=0) + targets = [] + seen = set() + seen_digests = set() + headers = github_headers(token) + commit_lookups = 0 + stop_cycle = False + for offset in range(1, hours_back + 1): + if stop_cycle or not budget.check_deadline(): + break + hour = now - timedelta(hours=offset) + url = f'https://data.gharchive.org/{hour:%Y-%m-%d-%H}.json.gz' + logger.info(f'Fetching GHArchive file candidates hour {hour.isoformat()} from {url}') + prepared = [] + pending_keys = set() + page_artifacts = 0 + stop_hour = False + try: + with cached_gharchive_hour_reader( + hour, archive_cache_dir, budget.request_timeout(request_timeout), deadline=budget.deadline, + ) as archive_path: + if not archive_path: + continue + for raw_line in iter_gharchive_lines(archive_path, deadline=budget.deadline): + if not budget.check_deadline('while reading GHArchive events'): + stop_cycle = True + break + try: + event = json.loads(raw_line.decode('utf-8', errors='replace')) + except (ValueError, UnicodeDecodeError): + continue + if event.get('type') not in event_types: + continue + repo = event.get('repo') if isinstance(event.get('repo'), dict) else {} + repo_name = str(repo.get('name') or '').strip() + if not repo_name or '/' not in repo_name: + continue + path_items = [(sha, path, github_raw_url(repo_name, sha, path)) for sha, path in gharchive_changed_paths(event)] + if not path_items and github_archive_event_score(event) <= 0: + continue + if not path_items: + if commit_lookups >= int(max_commit_lookups or 0): + continue + commit_lookups += 1 + try: + path_items = list(fetch_github_commit_files( + event, + repo_name, + headers, + budget.request_timeout(request_timeout), + budget.deadline, + )) + except _PostmanHarvestDeadlineReached as e: + budget.stop(str(e)) + stop_cycle = True + break + if not path_items and commit_lookups >= int(max_commit_lookups or 0): + logger.info(f'GHArchive file fetch reached commit lookup cap: {max_commit_lookups}') + for sha, path, raw_url in path_items: + if not budget.check_deadline('while examining GHArchive paths'): + stop_cycle = True + break + if len(targets) + len(prepared) >= max_files: + stop_cycle = True + break + if not gharchive_path_interesting(path): + continue + key = f'{repo_name.lower()}@{sha}:{path.lower()}' + if key in seen or key in pending_keys: + continue + if not raw_url: + continue + admitted, scope = budget.admit(page_artifacts) + if not admitted: + stop_cycle = scope == 'cycle' + stop_hour = True + break + page_artifacts += 1 + pending_keys.add(key) + try: + raw = api_request( + 'GET', raw_url, timeout=budget.request_timeout(request_timeout), + use_proxy=False, + max_retries=2, retry_delay=1, + retry_statuses={408, 500, 502, 503, 504}, + deadline=budget.deadline, + ) + if raw.status_code == 404: + continue + raw.raise_for_status() + content = raw.content + except _PostmanHarvestDeadlineReached as e: + budget.stop(str(e)) + stop_cycle = True + break + except (requests.exceptions.RequestException, ApiRequestError) as e: + logger.warning(f'Unable to fetch GHArchive raw file {repo_name}:{path}: {str(e)[:300]}') + continue + if max_bytes and len(content) > max_bytes: + continue + if not budget.account_bytes(len(content)): + stop_cycle = True + break + origin = { + 'provider': 'gharchive_file', + 'repo': repo_name, + 'path': path, + 'sha': sha, + 'raw_url': raw_url, + 'archive_hour': hour.isoformat(), + 'event_type': event.get('type') or '', + } + kind = postman_kind_for_path(path) or 'gharchive_file' + try: + entry = _prepare_postman_cache_entry(content, kind, origin, max_file_size_mb) + except Exception as e: + if isinstance(e, PostmanCacheCapacityError): + raise + logger.warning(f'Unable to cache GHArchive file {repo_name}:{path}: {str(e)[:300]}') + continue + if entry['digest'] in seen_digests: + continue + seen_digests.add(entry['digest']) + prepared.append({ + 'entry': entry, + 'key': key, + 'kind': kind, + 'origin': origin, + 'repo': repo_name, + 'path': path, + 'sha': sha, + 'raw_url': raw_url, + }) + if stop_cycle or stop_hour: + break + except _PostmanHarvestDeadlineReached as e: + budget.stop(str(e)) + stop_cycle = True + except (OSError, EOFError, gzip.BadGzipFile, GHArchiveBoundsError, urllib3_exceptions.ProtocolError) as e: + if 'archive_path' in locals() and archive_path: + remove_file_quiet(archive_path) + raise ApiRequestError(f'Unable to read cached GHArchive gzip {url}: {e}') from e + + published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget) + for record, (cache_path, digest, size) in published: + target = gharchive_file_target( + 'github_archive_file', record['kind'], cache_path, digest, record['origin'], + repo=record['repo'], path=record['path'], sha=record['sha'], + raw_url=record['raw_url'], size=size, + ) + targets.append(target) + seen.add(record['key']) + if deadline_reached or stop_cycle or budget.exhausted(): + stop_cycle = True + break + logger.info(f'GHArchive file fetch produced {len(targets)} targets; commit_lookups={commit_lookups}') + return targets + + +def gist_file_interesting(filename, file_meta): + text = ' '.join([ + str(filename or '').lower(), + str((file_meta or {}).get('type') or '').lower(), + str((file_meta or {}).get('language') or '').lower(), + ]) + return _contains_term(text, GITHUB_GIST_FILE_TERMS) + + +def fetch_github_gist_targets(pages=2, per_page=100, since=None, token=None, cache_dir=None, max_file_size_mb=2, request_timeout=20, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, discovery_max_artifacts=None, discovery_max_artifacts_per_page=None, discovery_max_bytes=None, discovery_max_elapsed_sec=None): + headers = github_headers(token) + per_page = max(1, min(int(per_page or 100), 100)) + max_pages = max(1, int(pages or 1)) + max_bytes = int(max_file_size_mb or 2) * 1024 * 1024 + budget = _PostmanDiscoveryBudget( + 'GitHub Gist', + discovery_max_artifacts, + discovery_max_artifacts_per_page, + discovery_max_bytes, + discovery_max_elapsed_sec, + ) + targets = [] + seen_candidates = set() + seen_targets = set() + seen_digests = set() + known_pages = 0 + stop_cycle = False + params_base = {'per_page': per_page} + if since: + params_base['since'] = since + for page in range(1, max_pages + 1): + if stop_cycle or not budget.check_deadline(): + break + params = dict(params_base) + params['page'] = page + try: + response = api_request( + 'GET', 'https://api.github.com/gists/public', headers=headers, params=params, + timeout=budget.request_timeout(request_timeout), + deadline=budget.deadline, + ) + response.raise_for_status() + except _PostmanHarvestDeadlineReached as e: + budget.stop(str(e)) + break + except requests.exceptions.HTTPError as e: + raise github_api_error(e.response) from e + except (requests.exceptions.RequestException, ApiRequestError) as e: + logger.warning(f'GitHub Gists request failed on page {page}: {str(e)[:300]}') + break + gists = response.json() or [] + if not gists: + break + page_targets = [] + prepared = [] + pending_candidates = set() + page_artifacts = 0 + stop_page = False + for gist in gists: + if not budget.check_deadline('while examining a Gist page'): + stop_cycle = True + break + gist_id = str(gist.get('id') or '') + files = gist.get('files') if isinstance(gist.get('files'), dict) else {} + for filename, meta in files.items(): + if not budget.check_deadline('while examining Gist files'): + stop_cycle = True + break + if not isinstance(meta, dict): + continue + size = int(meta.get('size') or 0) + raw_url = meta.get('raw_url') or '' + if not raw_url or (max_bytes and size > max_bytes): + continue + if not gist_file_interesting(filename, meta): + continue + identity = f'gist:{gist_id}:{filename}:{meta.get("raw_url")}' + if identity in seen_candidates or identity in pending_candidates: + continue + admitted, scope = budget.admit(page_artifacts) + if not admitted: + stop_cycle = scope == 'cycle' + stop_page = True + break + page_artifacts += 1 + pending_candidates.add(identity) + try: + raw = api_request( + 'GET', raw_url, headers=headers, + use_proxy=False, + timeout=budget.request_timeout(request_timeout), + deadline=budget.deadline, + ) + raw.raise_for_status() + content = raw.content + except _PostmanHarvestDeadlineReached as e: + budget.stop(str(e)) + stop_cycle = True + break + except (requests.exceptions.RequestException, ApiRequestError) as e: + logger.warning(f'Unable to fetch gist raw {gist_id}/{filename}: {str(e)[:300]}') + continue + if max_bytes and len(content) > max_bytes: + continue + if not budget.account_bytes(len(content)): + stop_cycle = True + break + origin = { + 'provider': 'github_gist', + 'gist_id': gist_id, + 'filename': filename, + 'raw_url': raw_url, + 'html_url': gist.get('html_url'), + 'created_at': gist.get('created_at'), + 'updated_at': gist.get('updated_at'), + 'owner': ((gist.get('owner') or {}).get('login') if isinstance(gist.get('owner'), dict) else ''), + } + kind = postman_kind_for_path(filename) or 'gist_file' + try: + entry = _prepare_postman_cache_entry(content, kind, origin, max_file_size_mb) + except Exception as e: + if isinstance(e, PostmanCacheCapacityError): + raise + logger.warning(f'Unable to cache gist {gist_id}/{filename}: {str(e)[:300]}') + continue + if entry['digest'] in seen_digests: + continue + seen_digests.add(entry['digest']) + prepared.append({ + 'entry': entry, + 'candidate_identity': identity, + 'kind': kind, + 'origin': origin, + 'gist_id': gist_id, + 'filename': filename, + 'raw_url': raw_url, + 'html_url': gist.get('html_url'), + }) + if stop_cycle or stop_page: + break + + published, deadline_reached = _publish_postman_discovery_batch(prepared, cache_dir, budget) + for record, (cache_path, digest, cached_size) in published: + target = postman_target_from_cached_artifact( + 'github_gist', record['kind'], cache_path, digest, record['origin'], + gist_id=record['gist_id'], filename=record['filename'], + raw_url=record['raw_url'], html_url=record['html_url'], size=cached_size, + ) + target_identity = postman_target_identity(target) + seen_candidates.add(record['candidate_identity']) + if target_identity in seen_targets: + continue + targets.append(target) + page_targets.append(target) + seen_targets.add(target_identity) + logger.info(f'GitHub Gists page {page}: gists={len(gists)}, targets={len(page_targets)}') + if stop_on_seen_pages and page >= max(1, int(min_pages_before_stop or 1)): + if page_targets and page_is_known( + page_targets, known_targets, normalize_target, known_target_lookup, + ): + known_pages += 1 + if known_pages >= max(1, int(seen_page_threshold or 1)): + break + else: + known_pages = 0 + if deadline_reached or stop_cycle or budget.exhausted(): + stop_cycle = True + break + return targets + +def gitlab_project_to_target(project): + return { + 'url': project.get('http_url_to_repo', ''), + 'name': project.get('path_with_namespace', ''), + 'created_at': project.get('created_at', ''), + 'updated_at': project.get('updated_at') or project.get('last_activity_at', ''), + 'last_activity_at': project.get('last_activity_at', ''), + } + +def gitlab_rate_limit_reset(response): + if not response: + return None + reset = response.headers.get('RateLimit-Reset') or response.headers.get('X-RateLimit-Reset') + if not reset: + retry_after = response.headers.get('Retry-After') + if retry_after: + try: + return (datetime.now(timezone.utc) + timedelta(seconds=int(retry_after))).isoformat(timespec='seconds') + except (TypeError, ValueError): + return None + return None + try: + if str(reset).isdigit(): + return datetime.fromtimestamp(int(reset), timezone.utc).isoformat(timespec='seconds') + return str(reset) + except (TypeError, ValueError): + return None + +def fetch_gitlab_repo_items(query, pages, per_page=100, token=None, sort_by="last_activity_at", sort_order="desc", visibility="public", last_activity_after=None, raise_rate_limit=False, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, request_attempts=1, retry_delay=0): + """Fetch GitLab repositories with metadata for filtering.""" + repos = [] + headers = {} + if token: + headers['Authorization'] = f'Bearer {token}' + + logger.info(f"Fetching GitLab repositories for: '{query if query else 'all repositories'}' sorted by {sort_by} ({sort_order})...") + page = 1 + + seen_pages = 0 + successful_pages = 0 + request_attempts = max(1, int(request_attempts or 1)) + retry_delay = max(0, int(retry_delay or 0)) + request_budget = 30 * request_attempts + retry_delay * (request_attempts - 1) + while page <= pages: + url = "https://gitlab.com/api/v4/projects" + params = { + 'visibility': visibility, + 'per_page': per_page, + 'page': page, + 'order_by': sort_by, + 'sort': sort_order, + } + if query: + params['search'] = query + if last_activity_after: + params['last_activity_after'] = last_activity_after + + try: + response = api_request( + 'GET', url, headers=headers, params=params, timeout=30, + max_retries=request_attempts, retry_delay=retry_delay, + deadline=time.monotonic() + request_budget, + ) + response.raise_for_status() + + # Handle rate limits + if response.status_code == 429: + retry_after = int(response.headers.get('Retry-After', 60)) + logger.warning(f"Rate limit exceeded. Waiting {retry_after} seconds...") + time.sleep(retry_after) + continue + + data = response.json() + if not isinstance(data, list): + raise ValueError('invalid GitLab project search payload') + successful_pages += 1 + if not data: + logger.info(f"Page {page} returned no results. Stopping.") + break + + page_repos = [gitlab_project_to_target(project) for project in data if project.get('http_url_to_repo')] + repos.extend(page_repos) + logger.info(f"Page {page}: Fetched {len(data)} repositories") + if stop_on_seen_pages and page >= max(1, min_pages_before_stop): + if page_is_known( + [item.get('url') for item in page_repos], known_targets, + normalize_target, known_target_lookup, + ): + seen_pages += 1 + logger.info(f"Page {page}: all GitLab repositories are already queued/checked ({seen_pages}/{seen_page_threshold})") + if seen_pages >= max(1, seen_page_threshold): + logger.info(f"Stopping GitLab pagination early after {seen_pages} all-known page(s)") + break + else: + seen_pages = 0 + page += 1 + + except requests.exceptions.HTTPError as e: + if e.response.status_code == 401 and not token: + logger.warning("GitLab API authentication would improve results. Consider adding a GitLab token.") + raise gitlab_api_error(e.response) from e + else: + api_error = gitlab_api_error(e.response) + if raise_rate_limit: + raise api_error from e + logger.error(str(api_error)) + raise api_error from e + except Exception as e: + if isinstance(e, ApiRequestError): + raise GitLabDiscoveryTransportError(str(e)) from e + logger.error(f"Error fetching page {page}: {str(e)}") + raise ApiRequestError(f'GitLab discovery payload failed: {e}') from e + + return repos + +def fetch_gitlab_repos(query, pages, per_page=100, token=None, sort_by="last_activity_at", sort_order="desc", visibility="public", raise_rate_limit=False, **kwargs): + """Fetch GitLab repository clone URLs with pagination and authentication.""" + return [ + item['url'] for item in fetch_gitlab_repo_items( + query, pages, per_page, token, sort_by, sort_order, visibility, raise_rate_limit=raise_rate_limit, **kwargs + ) + if item.get('url') + ] + +def github_recent_query(query, since): + since_str = since.strftime("%Y-%m-%d") + query = (query or '').strip() + if query and ' in:' not in f' {query.lower()} ': + query = f'{query} in:name,description,readme' + return f"{query} updated:>={since_str}" if query else f"updated:>={since_str}" + +def fetch_recent_github_repos(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, **kwargs): + """Fetch GitHub repositories updated since a specific timestamp""" + return fetch_github_repos(github_recent_query(query, since), pages=pages, per_page=per_page, token=token, raise_rate_limit=raise_rate_limit, **kwargs) + +def fetch_recent_github_repo_items(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, **kwargs): + """Fetch recent GitHub repositories with metadata.""" + return fetch_github_repo_items(github_recent_query(query, since), pages=pages, per_page=per_page, token=token, raise_rate_limit=raise_rate_limit, **kwargs) + +def fetch_recent_gitlab_repos(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, visibility="public", **kwargs): + """Fetch GitLab repositories updated since a specific timestamp""" + since_str = since.strftime("%Y-%m-%dT%H:%M:%SZ") + return [ + item['url'] for item in fetch_gitlab_repo_items( + query, pages=pages, per_page=per_page, token=token, + sort_by="last_activity_at", sort_order="desc", visibility=visibility, + last_activity_after=since_str, + raise_rate_limit=raise_rate_limit, + **kwargs, + ) + if item.get('url') + ] + +def fetch_recent_gitlab_repo_items(query, since, token=None, raise_rate_limit=False, pages=1, per_page=100, visibility="public", **kwargs): + """Fetch recent GitLab repositories with metadata.""" + since_str = since.strftime("%Y-%m-%dT%H:%M:%SZ") + return fetch_gitlab_repo_items( + query, pages=pages, per_page=per_page, token=token, + sort_by="last_activity_at", sort_order="desc", visibility=visibility, + last_activity_after=since_str, + raise_rate_limit=raise_rate_limit, + **kwargs, + ) + +DOCKERHUB_SEARCH_MAX_PAGES = 30 +DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC = 60 +DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC = 3600 +DOCKER_REGISTRY_MANIFEST_MAX_BYTES = 8 * 1024 * 1024 +DOCKER_REGISTRY_MAX_DESCRIPTORS = 1000 +DOCKER_REGISTRY_MAX_LAYERS = 2048 +DOCKER_REGISTRY_TOKEN_MAX_BYTES = 1024 * 1024 +DOCKER_CONFIG_MEDIA_TYPES = frozenset(( + 'application/vnd.oci.image.config.v1+json', + 'application/vnd.docker.container.image.v1+json', +)) +DOCKER_LAYER_MEDIA_TYPES = frozenset(( + 'application/vnd.oci.image.layer.v1.tar', + 'application/vnd.oci.image.layer.v1.tar+gzip', + 'application/vnd.oci.image.layer.v1.tar+zstd', + 'application/vnd.docker.image.rootfs.diff.tar', + 'application/vnd.docker.image.rootfs.diff.tar.gzip', +)) +DOCKER_LAYER_GZIP_MEDIA_TYPES = frozenset(( + 'application/vnd.oci.image.layer.v1.tar+gzip', + 'application/vnd.docker.image.rootfs.diff.tar.gzip', +)) +DOCKER_CONFIG_HISTORY_MAX_ENTRIES = DOCKER_REGISTRY_MAX_LAYERS * 2 +DOCKER_CONFIG_HISTORY_COMMAND_MAX_CHARS = 64 * 1024 + + +def docker_history_payload_class(created_by): + if not isinstance(created_by, str) or not created_by: + return 'unknown' + if len(created_by) > DOCKER_CONFIG_HISTORY_COMMAND_MAX_CHARS: + return 'unknown' + command = created_by.casefold() + if re.search(r'(^|#\(nop\)\s+)(copy|add)\s', command.strip()): + return 'copy_add' + if not re.search(r'(^|[\s#])run(\s|$)|/bin/(ba)?sh\s+-c', command): + return 'unknown' + if any(token in command for token in ( + '/models/', '/model/', 'model_weights', 'checkpoint.', '.safetensors', + '.gguf', '.onnx', '.pt ', '.pth ', 'huggingface-cli download', + )): + return 'bulk_data' + if re.search( + r'\b(apt-get|apt|apk|yum|dnf|microdnf|pip|pip3|poetry|npm|pnpm|yarn|' + r'bundle|gem|cargo)\s+(install|add|sync)\b', + command, + ): + return 'package_run' + if any(token in command for token in ( + '/app', '/srv', '/workspace', '/opt/app', 'config', '.env', + 'requirements.txt', 'package.json', 'pyproject.toml', + )): + return 'app_config_run' + return 'other_run' + + +def docker_config_payload_classes(config, layer_count): + try: + layer_count = int(layer_count) + except (TypeError, ValueError): + layer_count = -1 + fallback = ['unknown'] * max(0, layer_count) + if ( + not isinstance(config, dict) + or layer_count < 0 + or layer_count > DOCKER_REGISTRY_MAX_LAYERS + ): + return fallback + history = config.get('history') + if not isinstance(history, list) or len(history) > DOCKER_CONFIG_HISTORY_MAX_ENTRIES: + return fallback + rootfs = config.get('rootfs') + if rootfs is not None: + if not isinstance(rootfs, dict): + return fallback + diff_ids = rootfs.get('diff_ids') + if not isinstance(diff_ids, list) or len(diff_ids) != layer_count: + return fallback + commands = [] + for entry in history: + if not isinstance(entry, dict): + return fallback + empty_layer = entry.get('empty_layer', False) + if not isinstance(empty_layer, bool): + return fallback + if empty_layer: + continue + created_by = entry.get('created_by') + if not isinstance(created_by, str): + return fallback + commands.append(created_by) + if len(commands) != layer_count: + return fallback + classes = [docker_history_payload_class(command) for command in commands] + if any(payload_class not in DOCKER_ADAPTIVE_PAYLOAD_CLASSES for payload_class in classes): + return fallback + return classes + + +class DockerRegistryResolutionError(ValueError): + pass + + +class DockerResolverLeaseLostError(RuntimeError): + pass + + +class DockerRemoteAccessError(DockerRegistryResolutionError): + def __init__(self, message, status='unknown', retry_at=None, remote_attempted=True): + super().__init__(message) + self.status = str(status or 'unknown') + self.retry_at = retry_at + self.remote_attempted = bool(remote_attempted) + + +class DockerContentTransferError(RuntimeError): + def __init__(self, error_code, message, retryable=True): + super().__init__(message) + self.error_code = str(error_code or 'transfer_failed') + self.retryable = bool(retryable) + self.source_failure = False + self.transfer_bytes = 0 + self.duration_ms = 0 + + +class DockerLayerInfrastructureError(DockerContentTransferError): + def __init__( + self, error_code, message, category='remote_transient', auth_related=False, + ): + super().__init__(error_code, message, retryable=True) + self.category = str(category or 'remote_transient') + self.auth_related = bool(auth_related) + self.source_failure = True + + +class DockerContentScanError(RuntimeError): + def __init__(self, error_code, message, retryable=False): + super().__init__(message) + self.error_code = str(error_code or 'invalid_content') + self.retryable = bool(retryable) + + +@dataclass(frozen=True, repr=False) +class DockerBlobDownloadOutcome: + path: str + verified_bytes: int + transfer_bytes: int + duration_ms: int + bearer_auth: DockerRegistryAuth + + +@dataclass(frozen=True) +class DockerTagResolutionOutcome: + tags: tuple + status: str + remote_attempted: bool + retry_at: str = None + error: str = '' + selection_records: tuple = () + candidate_records: tuple = () + selector_version: str = '' + selector_hash: str = '' + candidate_distinct_graph_count: int = 0 + fresh_graph_evidence: bool = False + cache_bypassed: bool = False + + @property + def selections(self): + return self.selection_records + + @property + def selector_sha256(self): + return self.selector_hash + + +DOCKER_TAG_CONCLUSIVE_STATUSES = frozenset({'ok', 'empty', 'unsupported', 'not_found'}) + + +def docker_tag_resolution_is_conclusive(status): + return str(status or '') in DOCKER_TAG_CONCLUSIVE_STATUSES + + +def docker_images_per_repository_limit(value): + return validate_docker_images_per_repository(value) + + +def dockerhub_search_page_window(pages): + try: + requested = max(1, int(pages or 1)) + except (TypeError, ValueError): + requested = 1 + return requested, min(requested, DOCKERHUB_SEARCH_MAX_PAGES) + + +def fetch_dockerhub_search_page( + query, page, per_page=100, sort_by='updated_at', sort_order='desc', + request_timeout=15, +): + """Fetch and validate one bounded Docker Hub repository-search page.""" + try: + page = int(page) + per_page = max(1, int(per_page or 1)) + except (TypeError, ValueError, OverflowError): + raise DockerHubDiscoveryTransportError( + 'Docker Hub search page request is invalid', + category='invalid_payload', remote_attempted=False, retryable=False, + ) from None + if page < 1 or page > DOCKERHUB_SEARCH_MAX_PAGES: + raise DockerHubDiscoveryTransportError( + 'Docker Hub search page request is invalid', + category='invalid_payload', remote_attempted=False, retryable=False, + ) + + params = { + 'query': query, + 'page': page, + 'page_size': per_page, + 'sort': sort_by, + 'order': sort_order, + } + started = time.perf_counter() + try: + response = dockerhub_search_response( + 'https://hub.docker.com/v2/search/repositories', params, + request_timeout=request_timeout, + ) + except DockerRemoteAccessError as error: + category = { + 'rate_limited': 'rate_limit', + 'auth_failed': 'auth_unavailable', + 'remote_transient': 'remote_transient', + }.get(error.status, 'page_unavailable') + if error.retry_at and not error.remote_attempted: + category = 'provider_cooldown' + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} failed after bounded attempts', + category=category, retry_at=error.retry_at, + remote_attempted=error.remote_attempted, + ) from None + except ApiRequestError: + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} failed after bounded attempts', + category='network', remote_attempted=True, + ) from None + except Exception: + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} failed after bounded attempts', + category='page_unavailable', remote_attempted=True, + ) from None + + try: + response.raise_for_status() + except Exception: + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} failed after bounded attempts', + category='page_unavailable', remote_attempted=True, + ) from None + + try: + data = response.json() + except Exception: + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} returned an invalid payload', + category='invalid_payload', remote_attempted=True, retryable=False, + ) from None + + if not isinstance(data, dict) or not isinstance(data.get('results'), list): + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} returned an invalid payload', + category='invalid_payload', remote_attempted=True, retryable=False, + ) + + raw_count = data.get('count') + try: + total_count = int(raw_count) + except (TypeError, ValueError, OverflowError): + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} returned an invalid result count', + category='invalid_payload', remote_attempted=True, retryable=False, + ) from None + if ( + isinstance(raw_count, bool) + or (isinstance(raw_count, float) and not raw_count.is_integer()) + or total_count < 0 + ): + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} returned an invalid result count', + category='invalid_payload', remote_attempted=True, retryable=False, + ) + + result_count = len(data['results']) + absolute_start = (page - 1) * per_page + if ( + result_count > per_page + or total_count < absolute_start + result_count + or (result_count == 0 and total_count > absolute_start) + ): + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} returned incoherent pagination evidence', + category='invalid_payload', remote_attempted=True, retryable=False, + ) + + repositories = [] + seen_repo_names = set() + for repository in data['results']: + if not isinstance(repository, dict): + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} returned an invalid payload', + category='invalid_payload', remote_attempted=True, retryable=False, + ) + repo_name = repository.get('repo_name') + if not isinstance(repo_name, str) or not repo_name.strip(): + raise DockerHubDiscoveryTransportError( + f'Docker Hub search page {page} returned an invalid payload', + category='invalid_payload', remote_attempted=True, retryable=False, + ) + repo_name = repo_name.strip() + if repo_name in seen_repo_names: + continue + seen_repo_names.add(repo_name) + safe_repository = {'repo_name': repo_name} + for field in ('last_updated', 'last_modified'): + if field in repository: + safe_repository[field] = repository[field] + repositories.append(safe_repository) + + return { + 'page': page, + 'repositories': repositories, + 'total_count': total_count, + 'elapsed': time.perf_counter() - started, + } + + +def fetch_dockerhub_images(query, pages, per_page=100, sort_by="updated_at", sort_order="desc", fetch_workers=8, request_timeout=15, resolve_tags=True, tag_fetch_workers=None, tag_retry_count=2, tag_retry_delay=5, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64', platform_candidate_tags=20, images_per_repository=1): + """Fetch Docker Hub images with pagination and sorting""" + images = [] + per_page = max(1, int(per_page or 1)) + requested_pages, pages = dockerhub_search_page_window(pages) + if requested_pages > pages: + logger.info( + f'Docker Hub search is limited to {pages} accessible page(s); ' + f'capping requested pages from {requested_pages}' + ) + logger.info(f"Fetching Docker Hub images for: '{query}' sorted by {sort_by} ({sort_order})...") + + def fetch_page(page): + started = time.perf_counter() + try: + return fetch_dockerhub_search_page( + query, page, per_page=per_page, sort_by=sort_by, + sort_order=sort_order, request_timeout=request_timeout, + ), None + except DockerHubDiscoveryTransportError as error: + return { + 'page': page, + 'repositories': [], + 'total_count': 0, + 'elapsed': time.perf_counter() - started, + }, error + + first_page, first_error = fetch_page(1) + if first_error is not None: + raise first_error + + total_count = first_page['total_count'] + expected_pages = min( + pages, + max(1, (total_count + per_page - 1) // per_page), + ) + page_results = [(first_page, None)] + if expected_pages > 1: + max_workers = max(1, min(fetch_workers, expected_pages - 1)) + with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor: + futures = [ + executor.submit(fetch_page, page) + for page in range(2, expected_pages + 1) + ] + page_results.extend( + future.result() for future in concurrent.futures.as_completed(futures) + ) + + repo_names = [] + failed_pages = [] + for page_result, error in sorted( + page_results, key=lambda item: item[0]['page'], + ): + page = page_result['page'] + elapsed = page_result['elapsed'] + if error: + logger.error( + f"Page {page}: Docker Hub fetch failed after bounded attempts " + f"({elapsed:.1f}s)" + ) + failed_pages.append(page) + continue + repositories = page_result['repositories'] + if not repositories: + logger.info( + f"Page {page}: no results after {elapsed:.1f}s " + f"(total matches: {page_result['total_count']})" + ) + continue + + repo_names.extend(repository['repo_name'] for repository in repositories) + logger.info(f"Page {page}: fetched {len(repositories)} images in {elapsed:.1f}s") + + if failed_pages: + raise DockerHubDiscoveryTransportError( + f'Docker Hub search pagination incomplete: {len(failed_pages)} ' + f'of {expected_pages} expected page(s) failed' + ) + + if not resolve_tags: + return repo_names + + tag_fetch_workers = tag_fetch_workers if tag_fetch_workers is not None else min(4, fetch_workers) + logger.info(f"Resolving Docker tags for {len(repo_names)} repositories with {tag_fetch_workers} worker(s)...") + max_workers = max(1, min(tag_fetch_workers, len(repo_names) or 1)) + with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor: + futures = { + executor.submit( + fetch_dockerhub_tags, repo_name, None, + docker_images_per_repository_limit(images_per_repository), + tag_retry_count, tag_retry_delay, + platform_filter_enabled, platform_os, platform_arch, platform_candidate_tags, + True, + ): repo_name + for repo_name in repo_names + } + for future in concurrent.futures.as_completed(futures): + repo_name = futures[future] + tags, status = future.result() + if tags: + images.extend(tags) + if status != 'ok': + images.append(repo_name) + elif not docker_tag_resolution_is_conclusive(status): + images.append(repo_name) + logger.info(f"Deferring Docker tag resolution for {repo_name}: metadata lookup unavailable") + else: + logger.info(f"Skipping {repo_name}: no tags found") + + logger.info(f"Resolved {len(images)} tagged Docker images from {len(repo_names)} repositories") + + return images + +def parse_dockerhub_datetime(value): + if not value: + return None + + value = value.rstrip('Z') + for date_format in ("%Y-%m-%dT%H:%M:%S.%f", "%Y-%m-%dT%H:%M:%S"): + try: + return datetime.strptime(value, date_format) + except ValueError: + continue + return None + +def fetch_dockerhub_last_updated(repo_name): + if '/' in repo_name: + namespace, name = repo_name.split('/', 1) + else: + namespace, name = 'library', repo_name + + url = f"https://hub.docker.com/v2/repositories/{namespace}/{name}/" + try: + response = api_request('GET', url, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=8) + response.raise_for_status() + data = response.json() + return parse_dockerhub_datetime(data.get('last_updated') or data.get('last_modified')) + except Exception as e: + if isinstance(e, ApiRequestError): + raise + logger.warning(f"Unable to fetch Docker Hub metadata for {repo_name}: {str(e)}") + return None + + +DOCKERHUB_TAG_CACHE_SCHEMA = """ +CREATE TABLE IF NOT EXISTS dockerhub_tag_cache ( + cache_key TEXT PRIMARY KEY, + repo_key TEXT NOT NULL, + status TEXT NOT NULL, + tags_json TEXT, + checked_at REAL NOT NULL, + expires_at REAL NOT NULL, + since_at REAL, + message TEXT +); +CREATE TABLE IF NOT EXISTS dockerhub_tag_cache_meta ( + key TEXT PRIMARY KEY, + value TEXT, + expires_at REAL +); +""" +_dockerhub_tag_cache_init_lock = threading.Lock() +_dockerhub_tag_cache_write_lock = threading.Lock() +_dockerhub_tag_cache_initialized = set() + + +def dockerhub_repo_key(repo_name): + repo_name = str(repo_name or '').strip().lower() + if ':' in repo_name: + repo_name = repo_name.split(':', 1)[0] + if '/' not in repo_name: + repo_name = 'library/' + repo_name + return repo_name.strip('/') + + +def dockerhub_tag_cache_path(): + return getattr(scan_config, 'dockerhub_tag_cache_path', '') or '' + + +def dockerhub_tag_cache_limits(): + return { + 'rows': max(1, int(getattr(scan_config, 'dockerhub_tag_cache_max_rows', 50000))), + 'age': max(60, int(getattr(scan_config, 'dockerhub_tag_cache_max_age_sec', 7 * 86400))), + 'bytes': max(4096, int(getattr(scan_config, 'dockerhub_tag_cache_max_bytes', 256 * 1024 * 1024))), + 'min_free': max(0, int(getattr(scan_config, 'dockerhub_tag_cache_min_free_bytes', 512 * 1024 * 1024))), + } + + +def dockerhub_tag_cache_disk_bytes(path): + return sum( + os.path.getsize(candidate) for candidate in (path, path + '-wal', path + '-shm') + if os.path.isfile(candidate) + ) + + +def _dockerhub_since_epoch(since): + if since is None: + return None + try: + if isinstance(since, datetime): + value = since + else: + value = datetime.fromisoformat(str(since).replace('Z', '+00:00')) + if value.tzinfo is None: + value = value.replace(tzinfo=timezone.utc) + return float(value.timestamp()) + except (TypeError, ValueError, OverflowError): + return None + + +def maintain_dockerhub_tag_cache(conn, path): + limits = dockerhub_tag_cache_limits() + now = time.time() + conn.execute('DELETE FROM dockerhub_tag_cache WHERE expires_at <= ? OR checked_at < ?', (now, now - limits['age'])) + conn.execute('DELETE FROM dockerhub_tag_cache_meta WHERE expires_at IS NOT NULL AND expires_at <= ?', (now,)) + conn.execute( + '''DELETE FROM dockerhub_tag_cache WHERE cache_key NOT IN ( + SELECT cache_key FROM dockerhub_tag_cache ORDER BY checked_at DESC LIMIT ? + )''', + (limits['rows'],), + ) + conn.commit() + if dockerhub_tag_cache_disk_bytes(path) > limits['bytes']: + # The cache is disposable. Clearing it under SQLite's full auto-vacuum + # is safer than allowing stale pages to consume an unbounded volume. + conn.execute('DELETE FROM dockerhub_tag_cache') + conn.execute('DELETE FROM dockerhub_tag_cache_meta') + conn.commit() + try: + conn.execute('PRAGMA incremental_vacuum') + except sqlite3.DatabaseError: + pass + disk_bytes = dockerhub_tag_cache_disk_bytes(path) + parent = os.path.dirname(path) or os.getcwd() + if int(shutil.disk_usage(parent).free) - disk_bytes >= limits['min_free']: + try: + conn.execute('PRAGMA wal_checkpoint(TRUNCATE)') + conn.execute('PRAGMA journal_mode=DELETE') + conn.execute('VACUUM') + conn.execute('PRAGMA journal_mode=WAL') + except sqlite3.DatabaseError: + pass + return dockerhub_tag_cache_disk_bytes(path) <= limits['bytes'] + + +def connect_dockerhub_tag_cache(): + path = dockerhub_tag_cache_path() + if not path: + return None + parent = os.path.dirname(path) + if parent: + os.makedirs(parent, exist_ok=True) + with _dockerhub_tag_cache_init_lock: + if path not in _dockerhub_tag_cache_initialized: + new_database = not os.path.exists(path) + conn = sqlite3.connect(path, timeout=10) + try: + conn.execute('PRAGMA busy_timeout=10000') + if new_database: + conn.execute('PRAGMA auto_vacuum=FULL') + conn.execute('PRAGMA journal_mode=WAL') + conn.executescript(DOCKERHUB_TAG_CACHE_SCHEMA) + columns = {row[1] for row in conn.execute('PRAGMA table_info(dockerhub_tag_cache)').fetchall()} + if 'since_at' not in columns: + conn.execute('ALTER TABLE dockerhub_tag_cache ADD COLUMN since_at REAL') + conn.commit() + if not maintain_dockerhub_tag_cache(conn, path): + return None + finally: + conn.close() + _dockerhub_tag_cache_initialized.add(path) + conn = sqlite3.connect(path, timeout=10) + conn.execute('PRAGMA busy_timeout=10000') + return conn + + +def dockerhub_cache_key(repo_name, since, limit, platform_variant=''): + return f"{dockerhub_repo_key(repo_name)}|{int(limit or 1)}|{platform_variant}" + + +def get_dockerhub_tag_cache(repo_name, since, limit, platform_variant='', return_status=False): + conn = connect_dockerhub_tag_cache() + if not conn: + return None + try: + row = conn.execute( + 'SELECT status, tags_json, expires_at, since_at FROM dockerhub_tag_cache WHERE cache_key = ?', + (dockerhub_cache_key(repo_name, since, limit, platform_variant),), + ).fetchone() + now = time.time() + if not row or float(row[2] or 0) <= now: + return None + status, tags_json, _, stored_since = row + requested_since = _dockerhub_since_epoch(since) + if requested_since is not None and stored_since is None: + return None + if requested_since is not None and float(stored_since or 0) > requested_since: + return None + if status == 'ok': + records = json.loads(tags_json or '[]') + targets = [] + for record in records: + if isinstance(record, str): + return None + if not isinstance(record, dict) or not record.get('name'): + continue + updated_at = record.get('updated_at') + if requested_since is not None and updated_at is not None and float(updated_at) < requested_since: + continue + target = record.get('target') + if not target: + return None + try: + targets.append(parse_docker_target(target)['target']) + except (TypeError, ValueError): + return None + if len(targets) >= max(1, int(limit or 1)): + break + if not targets: + return None + return (targets, status) if return_status else targets + return ([], status) if return_status else [] + except Exception: + return None + finally: + conn.close() + + +def put_dockerhub_tag_cache( + repo_name, since, limit, status, tags=None, ttl=None, message='', platform_variant='', tag_records=None, +): + with _dockerhub_tag_cache_write_lock: + conn = connect_dockerhub_tag_cache() + if not conn: + return + path = dockerhub_tag_cache_path() + try: + now = time.time() + ttl = int(ttl if ttl is not None else getattr(scan_config, 'dockerhub_tag_cache_ttl_sec', 21600)) + records = [] + offered_records = list(tag_records or []) + for index, tag in enumerate(tags or []): + text = str(tag) + record = offered_records[index] if index < len(offered_records) and isinstance(offered_records[index], dict) else {} + name = str(record.get('name') or '') + if not name: + name = text.rsplit(':', 1)[1] if ':' in text and '@' not in text and not text.startswith('{') else text + target = record.get('target') or text + try: + target = parse_docker_target(target)['target'] + except (TypeError, ValueError): + continue + records.append({'name': name, 'target': target, 'updated_at': record.get('updated_at')}) + if status == 'ok' and not records: + return False + payload = json.dumps(records, ensure_ascii=True, sort_keys=True, separators=(',', ':')) + limits = dockerhub_tag_cache_limits() + projected_bytes = len(payload.encode('utf-8')) + len(str(message or '').encode('utf-8')) + 1024 + parent = os.path.dirname(path) or os.getcwd() + if ( + not maintain_dockerhub_tag_cache(conn, path) + or dockerhub_tag_cache_disk_bytes(path) + projected_bytes > limits['bytes'] + or int(shutil.disk_usage(parent).free) - projected_bytes < limits['min_free'] + ): + return False + conn.execute( + '''INSERT OR REPLACE INTO dockerhub_tag_cache( + cache_key, repo_key, status, tags_json, checked_at, expires_at, since_at, message + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?)''', + ( + dockerhub_cache_key(repo_name, since, limit, platform_variant), + dockerhub_repo_key(repo_name), + status, + payload, + now, + now + max(1, ttl), + _dockerhub_since_epoch(since), + str(message or '')[:500], + ), + ) + conn.commit() + maintain_dockerhub_tag_cache(conn, path) + return True + except Exception: + return False + finally: + conn.close() + + +def dockerhub_tag_rate_limit_state(): + conn = connect_dockerhub_tag_cache() + if not conn: + return {'active': False, 'retry_at': None} + try: + row = conn.execute("SELECT expires_at FROM dockerhub_tag_cache_meta WHERE key = 'rate_limited'").fetchone() + expires_at = float(row[0] or 0) if row else 0 + active = expires_at > time.time() + return { + 'active': active, + 'retry_at': ( + datetime.fromtimestamp(expires_at, timezone.utc).isoformat(timespec='seconds') + if active else None + ), + } + except Exception: + return {'active': False, 'retry_at': None} + finally: + conn.close() + + +def dockerhub_tags_rate_limited(): + return dockerhub_tag_rate_limit_state()['active'] + + +def dockerhub_retry_after_seconds(response=None): + fallback = int(getattr(scan_config, 'dockerhub_tag_rate_limit_cache_ttl_sec', 1800) or 1800) + seconds = None + headers = (getattr(response, 'headers', None) or {}) if response is not None else {} + retry_after = headers.get('Retry-After') + if retry_after: + try: + seconds = int(retry_after) + except (TypeError, ValueError): + try: + retry_at = parsedate_to_datetime(str(retry_after)) + if retry_at.tzinfo is None: + retry_at = retry_at.replace(tzinfo=timezone.utc) + seconds = math.ceil((retry_at - datetime.now(timezone.utc)).total_seconds()) + except (IndexError, TypeError, ValueError, OverflowError): + seconds = None + seconds = fallback if seconds is None else seconds + return max( + DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC, + min(DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC, int(seconds)), + ) + + +def put_dockerhub_tags_rate_limit(response=None, retry_seconds=None, return_retry_at=False): + ttl = dockerhub_retry_after_seconds(response) if retry_seconds is None else max( + DOCKERHUB_TAG_RATE_LIMIT_MIN_BACKOFF_SEC, + min(DOCKERHUB_TAG_RATE_LIMIT_MAX_BACKOFF_SEC, int(retry_seconds)), + ) + with _dockerhub_tag_cache_write_lock: + conn = connect_dockerhub_tag_cache() + if not conn: + return + try: + path = dockerhub_tag_cache_path() + limits = dockerhub_tag_cache_limits() + parent = os.path.dirname(path) or os.getcwd() + if ( + dockerhub_tag_cache_disk_bytes(path) + 8192 > limits['bytes'] + or int(shutil.disk_usage(parent).free) - 8192 < limits['min_free'] + ): + return False + expires_at = math.ceil(time.time() + ttl) + conn.execute( + '''INSERT INTO dockerhub_tag_cache_meta(key, value, expires_at) + VALUES ('rate_limited', '1', ?) + ON CONFLICT(key) DO UPDATE SET value = '1', + expires_at = MAX(dockerhub_tag_cache_meta.expires_at, excluded.expires_at)''', + (expires_at,), + ) + conn.commit() + stored = conn.execute( + "SELECT expires_at FROM dockerhub_tag_cache_meta WHERE key = 'rate_limited'" + ).fetchone() + maintain_dockerhub_tag_cache(conn, path) + if return_retry_at: + return datetime.fromtimestamp(float(stored[0]), timezone.utc).isoformat(timespec='seconds') + return True + except Exception: + return False + finally: + conn.close() + + +def put_dockerhub_exhausted_rate_limit(endpoint, response=None): + if endpoint == 'hub_search': + if not docker_token_manager.has_accounts(): + if ( + docker_token_manager.uses_explicit_pool() + or int(getattr(response, 'status_code', 0) or 0) != 429 + ): + return None + retry_seconds = dockerhub_retry_after_seconds(response) + else: + if not docker_token_manager.rate_limit_contributes_to_exhaustion(endpoint): + return None + retry_seconds = docker_token_manager.seconds_until_available(endpoint) + return datetime.fromtimestamp( + time.time() + retry_seconds, timezone.utc, + ).isoformat(timespec='seconds') + if not docker_token_manager.has_accounts(): + if docker_token_manager.uses_explicit_pool(): + return None + if int(getattr(response, 'status_code', 0) or 0) != 429: + return None + return put_dockerhub_tags_rate_limit(response, return_retry_at=True) + if not docker_token_manager.rate_limit_contributes_to_exhaustion(endpoint): + return None + return put_dockerhub_tags_rate_limit( + response, + retry_seconds=docker_token_manager.seconds_until_available(endpoint), + return_retry_at=True, + ) + + +def docker_tag_platform_support(tag, platform_os='linux', platform_arch='amd64'): + images = tag.get('images') if isinstance(tag, dict) else None + if not isinstance(images, list) or not images: + return None + platforms = [] + for image in images: + if not isinstance(image, dict): + return None + image_os = str(image.get('os') or '').strip().lower() + image_arch = str(image.get('architecture') or '').strip().lower() + if not image_os or not image_arch or image_os == 'unknown' or image_arch == 'unknown': + return None + platforms.append((image_os, image_arch)) + wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower()) + return wanted in platforms + + +def _bounded_docker_registry_json( + response, label, max_bytes=DOCKER_REGISTRY_MANIFEST_MAX_BYTES, *, + deadline=None, return_raw=False, +): + max_bytes = max(1, int(max_bytes)) + content_length = (getattr(response, 'headers', None) or {}).get('Content-Length') + if content_length is not None: + try: + content_length = int(content_length) + except (TypeError, ValueError) as exc: + raise DockerRegistryResolutionError(f'{label} has an invalid Content-Length') from exc + if content_length < 0: + raise DockerRegistryResolutionError(f'{label} has an invalid Content-Length') + if content_length > max_bytes: + raise DockerRegistryResolutionError(f'{label} exceeds the response size limit') + + iterator = None + iter_content = getattr(response, 'iter_content', None) + if callable(iter_content): + try: + iterator = iter(iter_content(chunk_size=min(64 * 1024, max_bytes + 1))) + except TypeError: + if isinstance(response, requests.Response): + raise DockerRegistryResolutionError(f'{label} cannot be streamed safely') + except (requests.RequestException, OSError) as exc: + raise DockerRemoteAccessError( + f'{label} stream is unavailable', status='remote_transient', + remote_attempted=True, + ) from exc + + raw_content = None + if iterator is not None: + content = bytearray() + try: + for chunk in iterator: + if deadline is not None and time.monotonic() >= float(deadline): + raise DockerRemoteAccessError( + f'{label} deadline expired', status='remote_transient', + remote_attempted=True, + ) + if not chunk: + continue + if not isinstance(chunk, (bytes, bytearray)): + raise DockerRegistryResolutionError(f'{label} returned invalid bytes') + if len(content) + len(chunk) > max_bytes: + raise DockerRegistryResolutionError(f'{label} exceeds the response size limit') + content.extend(chunk) + except (DockerRegistryResolutionError, DockerRemoteAccessError): + raise + except (requests.RequestException, OSError) as exc: + raise DockerRemoteAccessError( + f'{label} stream failed', status='remote_transient', + remote_attempted=True, + ) from exc + if deadline is not None and time.monotonic() >= float(deadline): + raise DockerRemoteAccessError( + f'{label} deadline expired', status='remote_transient', + remote_attempted=True, + ) + raw_content = bytes(content) + if content_length is not None and len(raw_content) != content_length: + raise DockerRegistryResolutionError(f'{label} Content-Length is inconsistent') + try: + payload = json.loads(raw_content.decode('utf-8')) + except (RecursionError, UnicodeDecodeError, TypeError, ValueError) as exc: + raise DockerRegistryResolutionError(f'{label} is not valid JSON') from exc + else: + # Lightweight response doubles used by unit tests may not implement streaming. + content = getattr(response, 'content', None) + if isinstance(content, (bytes, bytearray)): + raw_content = bytes(content) + if len(raw_content) > max_bytes: + raise DockerRegistryResolutionError(f'{label} exceeds the response size limit') + try: + payload = response.json() + except (RecursionError, TypeError, ValueError) as exc: + raise DockerRegistryResolutionError(f'{label} is not valid JSON') from exc + if deadline is not None and time.monotonic() >= float(deadline): + raise DockerRemoteAccessError( + f'{label} deadline expired', status='remote_transient', + remote_attempted=True, + ) + if not isinstance(payload, dict): + raise DockerRegistryResolutionError(f'{label} must be a JSON object') + if raw_content is None: + encoded = json.dumps(payload, ensure_ascii=True, separators=(',', ':')).encode('utf-8') + if len(encoded) > max_bytes: + raise DockerRegistryResolutionError(f'{label} exceeds the response size limit') + if return_raw: + if raw_content is None: + raise DockerRegistryResolutionError(f'{label} raw bytes are unavailable') + return payload, raw_content + return payload + + +def _docker_access_exhausted(endpoint, response=None, remote_attempted=False): + retry_at = put_dockerhub_exhausted_rate_limit(endpoint, response) + if retry_at: + raise DockerRemoteAccessError( + f'Docker {endpoint} accounts are rate-limited', + status='rate_limited', retry_at=retry_at, + remote_attempted=remote_attempted, + ) + raise DockerRemoteAccessError( + f'Docker {endpoint} authentication is unavailable', + status='auth_failed', remote_attempted=remote_attempted, + ) + + +def _docker_hub_access_token( + account, force_refresh=False, endpoint='hub_tags', stale_token='', +): + with docker_token_manager.hub_token_lock(account.name): + if not docker_token_manager.account_available(account.name, endpoint): + raise DockerRemoteAccessError( + f'Docker {endpoint} authentication is unavailable', + status='auth_failed', remote_attempted=False, + ) + cached = docker_token_manager.cached_hub_token(account.name) + if force_refresh: + if stale_token and cached and cached != stale_token: + return cached + elif cached: + return cached + docker_token_manager.invalidate_hub_token(account.name) + response = api_request( + 'POST', 'https://hub.docker.com/v2/auth/token', + json={'identifier': account.username, 'secret': account.token}, + headers={'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'}, + timeout=(5, 15), max_retries=1, + retry_statuses={408, 500, 502, 503, 504}, + allow_redirects=False, + ) + if response.status_code in (401, 403, 429): + category = 'rate_limit' if response.status_code == 429 else ( + 'auth_invalid' if response.status_code == 401 else 'auth_forbidden' + ) + docker_token_manager.report_http_status( + account, endpoint, response.status_code, response, category, + ) + raise DockerRemoteAccessError( + f'Docker Hub token endpoint returned HTTP {response.status_code}', + status='rate_limited' if response.status_code == 429 else 'auth_failed', + remote_attempted=True, + ) + if response.status_code >= 400: + raise DockerRemoteAccessError( + f'Docker Hub token endpoint returned HTTP {response.status_code}', + remote_attempted=True, + ) + try: + payload = _bounded_docker_registry_json( + response, 'Docker Hub token response', max_bytes=1024 * 1024, + ) + except DockerRegistryResolutionError as exc: + raise DockerRemoteAccessError( + 'Docker Hub returned an invalid token response', remote_attempted=True, + ) from exc + token = payload.get('access_token') or payload.get('token') + if not isinstance(token, str) or not token or len(token) > 16384: + raise DockerRemoteAccessError( + 'Docker Hub returned an invalid access token', remote_attempted=True, + ) + docker_token_manager.cache_hub_token( + account.name, token, payload.get('expires_in') or 600, + ) + docker_token_manager.report_success(account, endpoint) + return token + + +def dockerhub_search_response(url, params, request_timeout=15): + endpoint = 'hub_search' + + def request(token='', attempts=1): + headers = { + 'User-Agent': 'GitSecretsScanner/2.0', + 'Accept': 'application/json', + } + if token: + headers['Authorization'] = f'Bearer {token}' + return api_request( + 'GET', url, params=params, headers=headers, + timeout=(5, request_timeout), max_retries=attempts, retry_delay=1, + retry_statuses={408, 500, 502, 503, 504}, + allow_redirects=False, + ) + + if not docker_token_manager.has_accounts(): + if docker_token_manager.uses_explicit_pool(): + _docker_access_exhausted(endpoint, remote_attempted=False) + response = request(attempts=2) + if response.status_code == 429: + _docker_access_exhausted(endpoint, response, remote_attempted=True) + return response + + excluded = set() + last_response = None + remote_attempted = False + account = None + token = '' + refreshed_accounts = set() + page_attempts = 0 + while page_attempts < 2: + if account is None: + account = docker_token_manager.next_account(endpoint, excluded) + if account is None: + break + try: + token = _docker_hub_access_token(account, endpoint=endpoint) + except (ApiRequestError, DockerRemoteAccessError): + remote_attempted = True + excluded.add(account.name) + account = None + continue + + try: + response = request(token) + except ApiRequestError: + remote_attempted = True + page_attempts += 1 + if page_attempts < 2: + _wait_or_raise_scan_slot_fatal(1) + continue + raise + + page_attempts += 1 + remote_attempted = True + last_response = response + if ( + response.status_code == 401 + and account.name not in refreshed_accounts + and page_attempts < 2 + ): + refreshed_accounts.add(account.name) + try: + token = _docker_hub_access_token( + account, force_refresh=True, endpoint=endpoint, + stale_token=token, + ) + continue + except (ApiRequestError, DockerRemoteAccessError): + excluded.add(account.name) + account = None + continue + if response.status_code not in (401, 403, 429): + docker_token_manager.report_success(account, endpoint) + return response + category = 'rate_limit' if response.status_code == 429 else 'auth_forbidden' + docker_token_manager.report_http_status( + account, endpoint, response.status_code, response, category, + ) + excluded.add(account.name) + account = None + token = '' + _docker_access_exhausted( + endpoint, last_response, remote_attempted=remote_attempted, + ) + + +def dockerhub_tags_response(url, params): + endpoint = 'hub_tags' + if not docker_token_manager.has_accounts(): + if docker_token_manager.uses_explicit_pool(): + _docker_access_exhausted(endpoint, remote_attempted=False) + response = api_request( + 'GET', url, params=params, + headers={'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'}, + timeout=8, max_retries=1, + retry_statuses={408, 500, 502, 503, 504}, + allow_redirects=False, + ) + if response.status_code == 429: + _docker_access_exhausted(endpoint, response, remote_attempted=True) + return response + + excluded = set() + last_response = None + remote_attempted = False + for _ in range(docker_token_manager.account_count()): + account = docker_token_manager.next_account(endpoint, excluded) + if account is None: + break + excluded.add(account.name) + try: + token = _docker_hub_access_token(account) + except DockerRemoteAccessError: + remote_attempted = True + continue + + def request(): + return api_request( + 'GET', url, params=params, + headers={ + 'User-Agent': 'GitSecretsScanner/2.0', + 'Accept': 'application/json', + 'Authorization': f'Bearer {token}', + }, + timeout=8, max_retries=1, + retry_statuses={408, 500, 502, 503, 504}, + allow_redirects=False, + ) + + response = request() + remote_attempted = True + last_response = response + if response.status_code == 401: + try: + token = _docker_hub_access_token( + account, force_refresh=True, stale_token=token, + ) + response = request() + last_response = response + except DockerRemoteAccessError: + continue + if response.status_code not in (401, 403, 429): + docker_token_manager.report_success(account, endpoint) + return response + if response.status_code != 403: + category = 'rate_limit' if response.status_code == 429 else 'auth_forbidden' + docker_token_manager.report_http_status( + account, endpoint, response.status_code, response, category, + ) + _docker_access_exhausted( + endpoint, last_response, remote_attempted=remote_attempted, + ) + + +def docker_registry_bearer_token( + challenge, repo_key, excluded_accounts=None, *, deadline=None, + anonymous_only=False, +): + text = str(challenge or '').strip() + if not text.lower().startswith('bearer '): + raise DockerRegistryResolutionError('Docker registry did not provide a bearer challenge') + values = { + key.lower(): value + for key, value in re.findall(r'([A-Za-z][A-Za-z0-9_-]*)="([^"\\]*)"', text[7:]) + } + realm = values.get('realm', '') + parsed = urlsplit(realm) + if ( + parsed.scheme.lower() != 'https' + or (parsed.hostname or '').lower() != 'auth.docker.io' + or parsed.username is not None + or parsed.password is not None + or parsed.port not in (None, 443) + ): + raise DockerRegistryResolutionError('Docker registry bearer realm is not trusted') + excluded = set(excluded_accounts or ()) + authenticated = not anonymous_only and docker_token_manager.has_accounts() + if ( + not authenticated and not anonymous_only + and docker_token_manager.uses_explicit_pool() + ): + _docker_access_exhausted('registry', remote_attempted=False) + attempts = docker_token_manager.account_count() if authenticated else 1 + last_response = None + remote_attempted = False + saw_auth_failure = False + saw_rate_limit = False + saw_target_forbidden = False + saw_invalid_response = False + for _ in range(max(1, attempts)): + if deadline is not None and time.monotonic() >= float(deadline): + raise DockerRemoteAccessError( + 'Docker registry token deadline expired', + status='remote_transient', remote_attempted=remote_attempted, + ) + account = docker_token_manager.next_account('registry', excluded) if authenticated else None + if authenticated and account is None: + break + if account is not None: + excluded.add(account.name) + try: + response = api_request( + 'GET', realm, + params={ + 'service': 'registry.docker.io', + 'scope': f'repository:{repo_key}:pull', + }, + headers={ + 'User-Agent': 'GitSecretsScanner/2.0', + 'Accept': 'application/json', + 'Accept-Encoding': 'identity', + }, + auth=(account.username, account.token) if account is not None else None, + timeout=(5, 15), max_retries=1, + retry_statuses={408, 500, 502, 503, 504}, + allow_redirects=False, stream=True, deadline=deadline, + ) + except ApiRequestError as exc: + raise DockerRemoteAccessError( + 'Docker registry token endpoint is temporarily unavailable', + status='remote_transient', remote_attempted=True, + ) from exc + remote_attempted = True + last_response = response + try: + status_code = int(response.status_code) + if status_code in (401, 403, 429): + if status_code == 429: + saw_rate_limit = True + if account is not None: + docker_token_manager.report_http_status( + account, 'registry', status_code, response, 'rate_limit', + ) + elif status_code == 401: + saw_auth_failure = True + if account is not None: + docker_token_manager.report_http_status( + account, 'registry', status_code, response, 'auth_invalid', + ) + else: + saw_target_forbidden = True + continue + if status_code in (408, 425) or status_code >= 500: + raise DockerRemoteAccessError( + 'Docker registry token endpoint is temporarily unavailable', + status='remote_transient', remote_attempted=True, + ) + if status_code >= 400: + raise DockerRemoteAccessError( + 'Docker registry denied access to the requested target', + status='target_forbidden', remote_attempted=True, + ) + try: + payload = _bounded_docker_registry_json( + response, 'Docker registry token response', + max_bytes=DOCKER_REGISTRY_TOKEN_MAX_BYTES, deadline=deadline, + ) + except DockerRegistryResolutionError: + saw_invalid_response = True + continue + finally: + response.close() + token = payload.get('token') or payload.get('access_token') + if not isinstance(token, str) or not token or len(token) > 16384: + saw_invalid_response = True + continue + if account is not None: + docker_token_manager.report_success(account, 'registry') + return DockerRegistryAuth( + token=token, + account_name=account.name if account is not None else '', + challenge=text, + ) + if saw_rate_limit: + retry_at = put_dockerhub_exhausted_rate_limit('registry', last_response) + raise DockerRemoteAccessError( + 'Docker registry accounts are rate-limited', status='rate_limited', + retry_at=retry_at, remote_attempted=remote_attempted, + ) + if saw_target_forbidden: + raise DockerRemoteAccessError( + 'Docker registry denied access to the requested target', + status='target_forbidden', remote_attempted=remote_attempted, + ) + if saw_invalid_response and not saw_auth_failure: + raise DockerRemoteAccessError( + 'Docker registry returned an invalid token response', + status='remote_transient', remote_attempted=remote_attempted, + ) + _docker_access_exhausted( + 'registry', last_response, remote_attempted=remote_attempted, + ) + + +def docker_registry_manifest( + repo_name, digest, bearer_auth=None, *, verify_content_digest=False, + return_raw=False, deadline=None, lease_renewal_callback=None, + anonymous_only=False, +): + repo_key = dockerhub_repo_key(repo_name) + digest = normalize_docker_digest(digest) + if not digest: + raise DockerRegistryResolutionError('Docker manifest digest is invalid') + url = f"https://registry-1.docker.io/v2/{quote(repo_key, safe='/')}/manifests/{digest}" + accept = ', '.join(( + 'application/vnd.oci.image.index.v1+json', + 'application/vnd.docker.distribution.manifest.list.v2+json', + 'application/vnd.oci.image.manifest.v1+json', + 'application/vnd.docker.distribution.manifest.v2+json', + )) + + if isinstance(bearer_auth, str): + bearer_auth = DockerRegistryAuth(token=bearer_auth) + bearer_auth = bearer_auth or DockerRegistryAuth(token='') + + def request(auth): + headers = { + 'User-Agent': 'GitSecretsScanner/2.0', + 'Accept': accept, + 'Accept-Encoding': 'identity', + } + if auth.token: + headers['Authorization'] = f'Bearer {auth.token}' + return api_request( + 'GET', url, headers=headers, timeout=(5, 15), max_retries=1, + retry_statuses={408, 500, 502, 503, 504}, + allow_redirects=False, stream=True, deadline=deadline, + ) + + excluded = set() + attempts = 2 if anonymous_only else max( + 2, docker_token_manager.account_count() + 1, + ) + response = None + last_response = None + last_status = None + saw_bearer_unauthorized = False + for _ in range(attempts): + if lease_renewal_callback is not None: + if not callable(lease_renewal_callback): + raise ValueError('Docker resolver lease renewal callback is invalid') + if lease_renewal_callback() is False: + raise DockerResolverLeaseLostError('Docker resolver lease renewal was rejected') + if deadline is not None and time.monotonic() >= float(deadline): + raise DockerRemoteAccessError( + 'Docker registry manifest deadline expired', + status='remote_transient', remote_attempted=response is not None, + ) + response = request(bearer_auth) + last_response = response + last_status = int(response.status_code) + if response.status_code not in (401, 403, 429): + break + challenge = ( + (getattr(response, 'headers', None) or {}).get('WWW-Authenticate') + or bearer_auth.challenge + ) + status_code = int(response.status_code) + if status_code == 401 and bearer_auth.token: + saw_bearer_unauthorized = True + if bearer_auth.account_name: + if status_code == 429: + docker_token_manager.report_http_status( + bearer_auth.account_name, 'registry', 429, response, 'rate_limit', + ) + excluded.add(bearer_auth.account_name) + if status_code == 403: + response.close() + raise DockerRemoteAccessError( + 'Docker registry denied access to the requested manifest', + status='target_forbidden', remote_attempted=True, + ) + if status_code == 429 and ( + anonymous_only or not docker_token_manager.has_accounts() + ): + try: + _docker_access_exhausted('registry', response, remote_attempted=True) + finally: + response.close() + response.close() + response = None + try: + if lease_renewal_callback is not None and lease_renewal_callback() is False: + raise DockerResolverLeaseLostError('Docker resolver lease renewal was rejected') + bearer_auth = docker_registry_bearer_token( + challenge, repo_key, excluded_accounts=excluded, deadline=deadline, + anonymous_only=anonymous_only, + ) + except DockerRemoteAccessError as exc: + if saw_bearer_unauthorized and exc.status == 'auth_failed': + raise DockerRemoteAccessError( + 'Docker registry denied access to the requested manifest', + status='target_forbidden', remote_attempted=True, + ) from exc + raise + except DockerRegistryResolutionError as exc: + raise DockerRemoteAccessError( + 'Docker registry authentication challenge is invalid', + status='auth_failed' if status_code == 401 else 'remote_transient', + remote_attempted=True, + ) from exc + if response is None: + if last_status == 429: + retry_at = put_dockerhub_exhausted_rate_limit('registry', last_response) + raise DockerRemoteAccessError( + 'Docker registry accounts are rate-limited', status='rate_limited', + retry_at=retry_at, remote_attempted=True, + ) + if last_status == 401: + raise DockerRemoteAccessError( + 'Docker registry denied access to the requested manifest' + if saw_bearer_unauthorized + else 'Docker registry authentication is unavailable', + status='target_forbidden' if saw_bearer_unauthorized else 'auth_failed', + remote_attempted=True, + ) + raise DockerRemoteAccessError( + 'Docker registry manifest request did not run', + status='remote_transient', remote_attempted=last_response is not None, + ) + if response.status_code in (401, 429): + try: + _docker_access_exhausted('registry', response, remote_attempted=True) + finally: + response.close() + status_code = int(response.status_code) + if 300 <= status_code < 400: + response.close() + raise DockerRegistryResolutionError('Docker registry manifest redirect was rejected') + try: + if status_code == 404: + raise DockerRegistryResolutionError('Docker registry manifest was not found') + if status_code in (408, 425) or status_code >= 500: + raise DockerRemoteAccessError( + 'Docker registry manifest endpoint is temporarily unavailable', + status='remote_transient', remote_attempted=True, + ) + if status_code >= 400: + raise DockerRegistryResolutionError('Docker registry rejected the manifest request') + returned_digest = normalize_docker_digest( + (getattr(response, 'headers', None) or {}).get('Docker-Content-Digest') + ) + if returned_digest and returned_digest != digest: + raise DockerRegistryResolutionError('Docker registry returned a different manifest digest') + if verify_content_digest or return_raw: + payload, raw_content = _bounded_docker_registry_json( + response, 'Docker manifest', deadline=deadline, return_raw=True, + ) + else: + payload = _bounded_docker_registry_json( + response, 'Docker manifest', deadline=deadline, + ) + raw_content = None + if verify_content_digest: + calculated = 'sha256:' + hashlib.sha256(raw_content).hexdigest() + if calculated != digest: + raise DockerRegistryResolutionError('Docker manifest payload digest is invalid') + if bearer_auth.account_name: + docker_token_manager.report_success(bearer_auth.account_name, 'registry') + if return_raw: + return payload, bearer_auth, raw_content + return payload, bearer_auth + finally: + response.close() + + +def resolve_docker_layer_graph( + repo_name, digest, platform_os='linux', platform_arch='amd64', + bearer_auth=None, *, deadline=None, include_descriptors=False, + lease_renewal_callback=None, +): + manifest_digest = normalize_docker_digest(digest) + if include_descriptors: + payload, bearer_auth, raw_content = docker_registry_manifest( + repo_name, manifest_digest, bearer_auth, deadline=deadline, + verify_content_digest=True, return_raw=True, + lease_renewal_callback=lease_renewal_callback, + ) + else: + payload, bearer_auth = docker_registry_manifest( + repo_name, manifest_digest, bearer_auth, deadline=deadline, + lease_renewal_callback=lease_renewal_callback, + ) + raw_content = None + descriptors = payload.get('manifests') + if descriptors is not None: + if not isinstance(descriptors, list) or len(descriptors) > DOCKER_REGISTRY_MAX_DESCRIPTORS: + raise DockerRegistryResolutionError('Docker manifest index has invalid descriptor bounds') + wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower()) + descriptor = next(( + item for item in descriptors + if isinstance(item, dict) + and isinstance(item.get('platform'), dict) + and ( + str(item['platform'].get('os') or '').lower(), + str(item['platform'].get('architecture') or '').lower(), + ) == wanted + ), None) + if descriptor is None: + return None, bearer_auth + manifest_digest = normalize_docker_digest(descriptor.get('digest')) + if not manifest_digest: + raise DockerRegistryResolutionError('Docker platform descriptor has an invalid digest') + if include_descriptors: + payload, bearer_auth, raw_content = docker_registry_manifest( + repo_name, manifest_digest, bearer_auth, deadline=deadline, + verify_content_digest=True, return_raw=True, + lease_renewal_callback=lease_renewal_callback, + ) + else: + payload, bearer_auth = docker_registry_manifest( + repo_name, manifest_digest, bearer_auth, deadline=deadline, + lease_renewal_callback=lease_renewal_callback, + ) + + layers = payload.get('layers') + if not isinstance(layers, list) or len(layers) > DOCKER_REGISTRY_MAX_LAYERS: + raise DockerRegistryResolutionError('Docker manifest has invalid layer bounds') + ordered_layers = [] + layer_descriptors = [] + for position, layer in enumerate(layers, 1): + layer_digest = normalize_docker_digest(layer.get('digest')) if isinstance(layer, dict) else '' + if not layer_digest: + raise DockerRegistryResolutionError('Docker manifest contains an invalid layer digest') + ordered_layers.append(layer_digest) + if include_descriptors: + layer_descriptors.append(_docker_content_descriptor(layer, 'layer', position)) + graph = { + 'manifest_digest': manifest_digest, + 'layers': tuple(ordered_layers), + } + if include_descriptors: + config = _docker_content_descriptor(payload.get('config'), 'config', 0) + manifest_media_type = str( + payload.get('mediaType') + or 'application/vnd.docker.distribution.manifest.v2+json' + ).strip().lower() + if not manifest_media_type or len(manifest_media_type) > 256: + raise DockerRegistryResolutionError('Docker manifest media type is invalid') + graph.update({ + 'manifest_media_type': manifest_media_type, + 'manifest_size_bytes': len(raw_content), + 'config_digest': config['digest'], + 'layer_descriptors': tuple(layer_descriptors), + }) + return graph, bearer_auth + + +def _dockerhub_manifest_target_parts(target): + parsed = parse_docker_target(target) + image = str(parsed['image']).lower() + image_name, manifest_digest = image.rsplit('@', 1) + manifest_digest = normalize_docker_digest(manifest_digest) + if not manifest_digest: + raise DockerRegistryResolutionError('Docker manifest target digest is invalid') + parts = image_name.split('/') + if len(parts) > 1 and ('.' in parts[0] or ':' in parts[0] or parts[0] == 'localhost'): + registry = parts.pop(0) + if registry not in ('docker.io', 'index.docker.io', 'registry-1.docker.io'): + raise DockerRegistryResolutionError( + 'Docker layer scanning only supports Docker Hub targets' + ) + if not parts or any(not part for part in parts): + raise DockerRegistryResolutionError('Docker Hub repository is invalid') + repository = '/'.join(parts) + registry_repository = repository if '/' in repository else f'library/{repository}' + return image, repository, registry_repository, manifest_digest + + +def _docker_content_descriptor(value, kind, position): + if not isinstance(value, dict): + raise DockerRegistryResolutionError(f'Docker {kind} descriptor is invalid') + digest = normalize_docker_digest(value.get('digest')) + size = value.get('size') + media_type = str(value.get('mediaType') or '').strip().lower() + if ( + not digest + or isinstance(size, bool) + or not isinstance(size, int) + or size < 0 + or size > 1024 * 1024 * 1024 * 1024 + or not media_type + or len(media_type) > 256 + ): + raise DockerRegistryResolutionError(f'Docker {kind} descriptor has invalid bounds') + return { + 'digest': digest, + 'size': size, + 'media_type': media_type, + } + + +def resolve_docker_content_manifest( + target, platform_os='linux', platform_arch='amd64', bearer_auth=None, + *, deadline=None, anonymous_only=False, +): + image, repository, registry_repository, manifest_digest = ( + _dockerhub_manifest_target_parts(target) + ) + target_manifest_digest = manifest_digest + payload, bearer_auth = docker_registry_manifest( + registry_repository, manifest_digest, bearer_auth, + verify_content_digest=True, deadline=deadline, + anonymous_only=anonymous_only, + ) + descriptors = payload.get('manifests') + if descriptors is not None: + if not isinstance(descriptors, list) or len(descriptors) > DOCKER_REGISTRY_MAX_DESCRIPTORS: + raise DockerRegistryResolutionError('Docker manifest index has invalid descriptor bounds') + wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower()) + child = next(( + item for item in descriptors + if isinstance(item, dict) + and isinstance(item.get('platform'), dict) + and ( + str(item['platform'].get('os') or '').lower(), + str(item['platform'].get('architecture') or '').lower(), + ) == wanted + ), None) + if child is None: + raise DockerRegistryResolutionError('Docker target platform manifest is unavailable') + manifest_digest = normalize_docker_digest(child.get('digest')) + if not manifest_digest: + raise DockerRegistryResolutionError('Docker platform descriptor digest is invalid') + payload, bearer_auth = docker_registry_manifest( + registry_repository, manifest_digest, bearer_auth, + verify_content_digest=True, deadline=deadline, + anonymous_only=anonymous_only, + ) + + config = _docker_content_descriptor(payload.get('config'), 'config', 0) + layers = payload.get('layers') + if not isinstance(layers, list) or len(layers) > DOCKER_REGISTRY_MAX_LAYERS: + raise DockerRegistryResolutionError('Docker manifest has invalid layer bounds') + normalized_layers = [ + _docker_content_descriptor(layer, 'layer', position) + for position, layer in enumerate(layers, 1) + ] + manifest_media_type = str( + payload.get('mediaType') + or 'application/vnd.docker.distribution.manifest.v2+json' + ).strip().lower() + if not manifest_media_type or len(manifest_media_type) > 256: + raise DockerRegistryResolutionError('Docker manifest media type is invalid') + if deadline is not None and time.monotonic() >= float(deadline): + raise DockerRemoteAccessError( + 'Docker registry manifest deadline expired', + status='remote_transient', remote_attempted=True, + ) + return { + 'version': 1, + 'image': image, + 'repository': registry_repository, + 'manifest_digest': target_manifest_digest, + 'platform_os': str(platform_os or 'linux').lower(), + 'platform_arch': str(platform_arch or 'amd64').lower(), + 'manifest_media_type': manifest_media_type, + 'config': config, + 'layers': normalized_layers, + }, bearer_auth + + +DOCKER_BLOB_REDIRECT_SUFFIXES = ( + '.docker.com', + '.docker.io', + '.cloudfront.net', + '.cloudflarestorage.com', + '.amazonaws.com', +) + + +def _docker_blob_url_validation_error(url, *, registry_origin=False): + try: + parsed = urlsplit(str(url or '')) + hostname = (parsed.hostname or '').lower().rstrip('.') + port = parsed.port + except ValueError: + return 'invalid_url' + if ( + parsed.scheme.lower() != 'https' + or not hostname + or parsed.username is not None + or parsed.password is not None + or port not in (None, 443) + or parsed.fragment + ): + return 'invalid_url' + if registry_origin: + return '' if hostname == 'registry-1.docker.io' and not parsed.query else 'invalid_registry' + try: + address = ipaddress.ip_address(hostname) + except ValueError: + address = None + if address is not None and not address.is_global: + return 'non_global_address' + if not any(hostname.endswith(suffix) for suffix in DOCKER_BLOB_REDIRECT_SUFFIXES): + return 'untrusted_host' + try: + answers = socket.getaddrinfo( + hostname, 443, type=socket.SOCK_STREAM, proto=socket.IPPROTO_TCP, + ) + except OSError: + return 'dns_unavailable' + if not answers: + return 'dns_unavailable' + for answer in answers: + try: + resolved = ipaddress.ip_address(str(answer[4][0]).split('%', 1)[0]) + except (IndexError, TypeError, ValueError): + return 'invalid_dns_answer' + if not resolved.is_global: + return 'non_global_address' + return '' + + +def _docker_blob_url_allowed(url, *, registry_origin=False): + return not _docker_blob_url_validation_error(url, registry_origin=registry_origin) + + +def _require_docker_blob_url(url, *, registry_origin=False): + error = _docker_blob_url_validation_error(url, registry_origin=registry_origin) + if error == 'dns_unavailable': + raise DockerLayerInfrastructureError( + 'remote_dns', 'Docker blob redirect DNS is temporarily unavailable', + category='remote_transient', + ) + if error: + raise DockerContentTransferError( + 'unsafe_redirect' if not registry_origin else 'unsafe_registry_url', + 'Docker blob URL is not trusted', False, + ) + + +def _docker_registry_blob_response( + repository, digest, bearer_auth, deadline, *, anonymous_only=False, +): + repo_key = dockerhub_repo_key(repository) + digest = normalize_docker_digest(digest) + if not digest: + raise DockerContentTransferError('invalid_descriptor', 'Docker blob digest is invalid', False) + url = f'https://registry-1.docker.io/v2/{quote(repo_key, safe="/")}/blobs/{digest}' + _require_docker_blob_url(url, registry_origin=True) + if isinstance(bearer_auth, str): + bearer_auth = DockerRegistryAuth(token=bearer_auth) + bearer_auth = bearer_auth or DockerRegistryAuth(token='') + excluded = set() + attempts = 2 if anonymous_only else max( + 2, docker_token_manager.account_count() + 1, + ) + response = None + saw_rate_limit = False + saw_bearer_unauthorized = False + for _ in range(attempts): + remaining = max(0.0, float(deadline) - time.monotonic()) + if remaining <= 0: + raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired') + headers = { + 'User-Agent': 'GitSecretsScanner/2.0', + 'Accept': 'application/octet-stream', + 'Accept-Encoding': 'identity', + } + if bearer_auth.token: + headers['Authorization'] = f'Bearer {bearer_auth.token}' + try: + response = api_request( + 'GET', url, headers=headers, timeout=(5, min(30, remaining)), + use_proxy=False, + max_retries=1, retry_statuses={408, 500, 502, 503, 504}, + allow_redirects=False, stream=True, deadline=deadline, + ) + except ApiRequestError as exc: + if time.monotonic() >= float(deadline): + raise DockerContentTransferError( + 'transfer_timeout', 'Docker blob deadline expired', + ) from exc + raise DockerLayerInfrastructureError( + 'remote_transient', 'Docker blob endpoint is temporarily unavailable', + category='remote_transient', + ) from exc + if response.status_code not in (401, 429): + return response, bearer_auth + challenge = ( + (getattr(response, 'headers', None) or {}).get('WWW-Authenticate') + or bearer_auth.challenge + ) + if response.status_code == 401 and bearer_auth.token: + saw_bearer_unauthorized = True + if bearer_auth.account_name: + if response.status_code == 429: + saw_rate_limit = True + docker_token_manager.report_http_status( + bearer_auth.account_name, 'registry', 429, response, 'rate_limit', + ) + excluded.add(bearer_auth.account_name) + elif response.status_code == 429: + saw_rate_limit = True + response.close() + response = None + try: + bearer_auth = docker_registry_bearer_token( + challenge, repo_key, excluded_accounts=excluded, deadline=deadline, + anonymous_only=anonymous_only, + ) + except DockerRemoteAccessError as exc: + if exc.status == 'target_forbidden': + raise DockerContentTransferError( + 'target_forbidden', 'Docker blob target is forbidden', False, + ) from exc + if exc.status == 'rate_limited': + raise DockerLayerInfrastructureError( + 'remote_rate_limit', 'Docker blob authorization is rate-limited', + category='docker_rate_limit', + ) from exc + if exc.status == 'auth_failed': + if saw_bearer_unauthorized: + raise DockerContentTransferError( + 'target_forbidden', 'Docker blob target is forbidden', False, + ) from exc + raise DockerLayerInfrastructureError( + 'remote_auth', 'Docker blob authorization is unavailable', + category='docker_auth', auth_related=True, + ) from exc + raise DockerLayerInfrastructureError( + 'remote_transient', 'Docker blob authorization is temporarily unavailable', + category='remote_transient', + ) from exc + except DockerRegistryResolutionError as exc: + raise DockerLayerInfrastructureError( + 'remote_auth', 'Docker blob authentication challenge is invalid', + category='docker_auth', auth_related=True, + ) from exc + if response is not None: + response.close() + if saw_rate_limit: + raise DockerLayerInfrastructureError( + 'remote_rate_limit', 'Docker blob authorization is rate-limited', + category='docker_rate_limit', + ) + if saw_bearer_unauthorized: + raise DockerContentTransferError( + 'target_forbidden', 'Docker blob target is forbidden', False, + ) + raise DockerLayerInfrastructureError( + 'remote_auth', 'Docker blob authorization is unavailable', + category='docker_auth', auth_related=True, + ) + + +def stream_docker_registry_blob( + repository, descriptor, destination, bearer_auth=None, *, deadline, + min_free_bytes=0, redirect_limit=5, anonymous_only=False, +): + digest = normalize_docker_digest((descriptor or {}).get('digest')) + declared_bytes = (descriptor or {}).get('size') + kind = str((descriptor or {}).get('kind') or '') + media_type = str((descriptor or {}).get('media_type') or '').strip().lower() + if ( + not digest + or isinstance(declared_bytes, bool) + or not isinstance(declared_bytes, int) + or declared_bytes < 0 + or declared_bytes > 1024 * 1024 * 1024 * 1024 + or kind not in ('config', 'layer') + or not media_type + ): + raise DockerContentTransferError('invalid_descriptor', 'Docker blob descriptor is invalid', False) + supported_media = DOCKER_CONFIG_MEDIA_TYPES if kind == 'config' else DOCKER_LAYER_MEDIA_TYPES + if media_type not in supported_media: + raise DockerContentTransferError( + 'unsupported_media_type', 'Docker blob media type is unsupported', False, + ) + deadline = float(deadline) + if deadline <= time.monotonic(): + raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired') + destination = os.path.abspath(destination) + try: + parent = require_private_directory(os.path.dirname(destination), create=False) + reject_reparse_components(parent) + except (OSError, ValueError) as exc: + raise DockerLayerInfrastructureError( + 'private_storage', 'Docker blob private storage is unavailable', + category='source_resource', + ) from exc + if os.path.lexists(destination): + raise DockerLayerInfrastructureError( + 'destination_exists', 'Docker blob destination is not available', + category='source_resource', + ) + required_free = max(0, int(min_free_bytes)) + declared_bytes + try: + free_bytes = shutil.disk_usage(parent).free + except OSError as exc: + raise DockerLayerInfrastructureError( + 'disk_reserve', 'Docker blob free space cannot be verified', + category='source_resource', + ) from exc + if free_bytes < required_free: + raise DockerLayerInfrastructureError( + 'disk_reserve', 'Docker blob would violate the free-space reserve', + category='source_resource', + ) + + started = time.monotonic() + response = None + total = 0 + temporary = ( + f'{destination}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.partial' + ) + published = False + try: + response, bearer_auth = _docker_registry_blob_response( + repository, digest, bearer_auth, deadline, + anonymous_only=anonymous_only, + ) + redirects = 0 + while response.status_code in (301, 302, 303, 307, 308): + location = (getattr(response, 'headers', None) or {}).get('Location') + current_url = str(getattr(response, 'url', '') or '') + response.close() + response = None + redirects += 1 + if not location or redirects > max(0, min(5, int(redirect_limit))): + raise DockerContentTransferError('unsafe_redirect', 'Docker blob redirect limit exceeded', False) + next_url = urljoin(current_url, location) + _require_docker_blob_url(next_url) + remaining = max(0.0, deadline - time.monotonic()) + if remaining <= 0: + raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired') + response = api_request( + 'GET', next_url, + use_proxy=False, + headers={ + 'User-Agent': 'GitSecretsScanner/2.0', + 'Accept': 'application/octet-stream', + 'Accept-Encoding': 'identity', + }, + timeout=(5, min(30, remaining)), max_retries=1, + retry_statuses={408, 500, 502, 503, 504}, + allow_redirects=False, stream=True, deadline=deadline, + ) + status_code = int(response.status_code) + if status_code == 401: + raise DockerContentTransferError( + 'target_forbidden', 'Docker blob target is forbidden', False, + ) + if status_code == 403: + raise DockerContentTransferError( + 'target_forbidden', 'Docker blob target is forbidden', False, + ) + if status_code == 404: + raise DockerContentTransferError( + 'blob_not_found', 'Docker blob target is unavailable', False, + ) + if status_code == 429: + raise DockerLayerInfrastructureError( + 'remote_rate_limit', 'Docker blob endpoint is rate-limited', + category='docker_rate_limit', + ) + if status_code in (408, 425) or status_code >= 500: + raise DockerLayerInfrastructureError( + 'remote_transient', 'Docker blob endpoint is temporarily unavailable', + category='remote_transient', + ) + if status_code >= 400: + raise DockerContentTransferError( + 'target_rejected', 'Docker blob target was rejected', False, + ) + content_encoding = str( + (getattr(response, 'headers', None) or {}).get('Content-Encoding') or '' + ).strip().lower() + if content_encoding not in ('', 'identity'): + raise DockerContentTransferError( + 'content_encoding', 'Docker blob response changed the content encoding', + ) + content_length = (getattr(response, 'headers', None) or {}).get('Content-Length') + try: + content_length = int(content_length) + except (TypeError, ValueError) as exc: + raise DockerContentTransferError( + 'size_mismatch', 'Docker blob response lacks an exact Content-Length', False, + ) from exc + if content_length != declared_bytes: + raise DockerContentTransferError( + 'size_mismatch', 'Docker blob Content-Length differs from its descriptor', False, + ) + + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) + descriptor_fd = os.open(temporary, flags, 0o600) + os.close(descriptor_fd) + harden_private_file(temporary) + digest_hash = hashlib.sha256() + with open(temporary, 'wb', buffering=0) as output: + for chunk in response.iter_content(chunk_size=1024 * 1024): + _raise_if_scan_slot_fatal() + if time.monotonic() >= deadline: + raise DockerContentTransferError('transfer_timeout', 'Docker blob deadline expired') + if not chunk: + continue + total += len(chunk) + if total > declared_bytes: + raise DockerContentTransferError( + 'size_mismatch', 'Docker blob exceeded its declared size', False, + ) + if shutil.disk_usage(parent).free < max(0, int(min_free_bytes)): + raise DockerLayerInfrastructureError( + 'disk_reserve', 'Docker blob transfer reached the free-space reserve', + category='source_resource', + ) + digest_hash.update(chunk) + output.write(chunk) + output.flush() + os.fsync(output.fileno()) + if time.monotonic() >= deadline: + raise DockerContentTransferError( + 'transfer_timeout', 'Docker blob deadline expired', + ) + if total != declared_bytes: + raise DockerContentTransferError( + 'size_mismatch', 'Docker blob byte count differs from its descriptor', False, + ) + if f'sha256:{digest_hash.hexdigest()}' != digest: + raise DockerContentTransferError( + 'digest_mismatch', 'Docker blob SHA-256 differs from its descriptor', False, + ) + harden_private_file(temporary) + durable_replace(temporary, destination) + if not private_file_ready(destination): + raise DockerLayerInfrastructureError( + 'private_file_lost', 'Docker blob lost its private file identity', + category='source_resource', + ) + if time.monotonic() >= deadline: + raise DockerContentTransferError( + 'transfer_timeout', 'Docker blob deadline expired', + ) + published = True + duration_ms = max(0, int((time.monotonic() - started) * 1000)) + return DockerBlobDownloadOutcome( + path=destination, + verified_bytes=total, + transfer_bytes=total, + duration_ms=duration_ms, + bearer_auth=bearer_auth, + ) + except DockerContentTransferError as exc: + exc.transfer_bytes = min(declared_bytes, max(0, int(total))) + exc.duration_ms = max(0, int((time.monotonic() - started) * 1000)) + raise + except (ApiRequestError, requests.RequestException) as exc: + if time.monotonic() >= deadline: + raise DockerContentTransferError( + 'transfer_timeout', 'Docker blob deadline expired', + ) from exc + raise DockerLayerInfrastructureError( + 'remote_transient', 'Docker blob transfer is temporarily unavailable', + category='remote_transient', + ) from exc + except OSError as exc: + raise DockerLayerInfrastructureError( + 'private_storage', 'Docker blob private storage failed', + category='source_resource', + ) from exc + finally: + if response is not None: + response.close() + if os.path.lexists(temporary): + durable_unlink(temporary) + if not published and os.path.lexists(destination): + durable_unlink(destination) + + +def fetch_docker_config_payload_classes( + resolved, bearer_auth=None, *, deadline, min_free_bytes=0, +): + layers = list((resolved or {}).get('layers') or ()) + fallback = ['unknown'] * len(layers) + config = dict((resolved or {}).get('config') or {}) + config.update({'kind': 'config', 'position': 0}) + if ( + config.get('media_type') not in DOCKER_CONFIG_MEDIA_TYPES + or not isinstance(config.get('size'), int) + or config['size'] < 0 + or config['size'] > DOCKER_REGISTRY_MANIFEST_MAX_BYTES + ): + return fallback, bearer_auth + work_root = None + destination = None + try: + work_root = tempfile.mkdtemp(prefix='docker-history-', dir=get_work_dir()) + harden_private_directory(work_root) + write_temp_owner(work_root, ['docker-config-history'], os.getpid(), required=True) + destination = os.path.join(work_root, 'config.json') + outcome = stream_docker_registry_blob( + resolved['repository'], config, destination, bearer_auth, + deadline=float(deadline), min_free_bytes=max(0, int(min_free_bytes or 0)), + ) + try: + parsed = validate_docker_content_artifact(destination, config) + except DockerContentScanError: + return fallback, outcome.bearer_auth + return docker_config_payload_classes(parsed, len(layers)), outcome.bearer_auth + except DockerContentScanError: + return fallback, bearer_auth + finally: + if destination and os.path.lexists(destination): + durable_unlink(destination) + if work_root: + cleanup_command_work_dir(work_root) + + +def _docker_depth_selection_evidence( + record, candidate_count, selector_version=DOCKER_DEPTH_SELECTOR_VERSION, +): + evidence = { + 'schema': 1, + 'type': 'docker-depth-selection-evidence-v1', + 'selector_version': selector_version, + 'selector_sha256': canonical_selector_hash(selector_version), + 'candidate_distinct_graph_count': int(candidate_count), + 'image_rank': int(record['image_rank']), + 'selection_reason': str(record['selection_reason']), + 'target': str(record['target']), + 'repository': str(record['repository']), + 'manifest_digest': str(record['manifest_digest']), + 'manifest_media_type': str(record['manifest_media_type']), + 'manifest_size_bytes': int(record['manifest_size_bytes']), + 'config_digest': str(record['config_digest']), + 'graph_sha256': str(record['graph_sha256']), + 'layers': [dict(layer) for layer in record['layer_metadata']], + } + return { + **record, + 'candidate_distinct_graph_count': int(candidate_count), + 'selection_evidence_sha256': canonical_docker_depth_selection_evidence_hash( + evidence + ), + } + + +def dockerhub_tag_digest(tag, platform_filter_enabled=False, platform_os='linux', platform_arch='amd64'): + if not isinstance(tag, dict): + return '' + images = tag.get('images') if isinstance(tag.get('images'), list) else [] + wanted = (str(platform_os or 'linux').lower(), str(platform_arch or 'amd64').lower()) + candidates = ( + image.get('digest') for image in images + if isinstance(image, dict) + and (str(image.get('os') or '').lower(), str(image.get('architecture') or '').lower()) == wanted + ) + platform_digest = next(( + digest for digest in (normalize_docker_digest(value) for value in candidates) if digest + ), '') + return platform_digest or normalize_docker_digest(tag.get('digest')) + + +def fetch_dockerhub_tags( + repo_name, since=None, limit=1, retry_count=2, retry_delay=5, + platform_filter_enabled=False, platform_os='linux', platform_arch='amd64', + platform_candidate_tags=20, return_status=False, *, return_outcome=False, + fresh_graph_evidence=False, lease_renewal_callback=None, + selector_version=DOCKER_DEPTH_SELECTOR_VERSION, +): + def output( + tags, status, remote_attempted=False, retry_at=None, error='', + selection_records=(), candidate_records=(), candidate_distinct_graph_count=0, + ): + tags = list(tags or []) + if return_outcome: + return DockerTagResolutionOutcome( + tags=tuple(tags), status=str(status), + remote_attempted=bool(remote_attempted), + retry_at=retry_at, error=str(error or '')[:500], + selection_records=tuple(selection_records or ()), + candidate_records=tuple(candidate_records or selection_records or ()), + selector_version=selector_version, + selector_hash=canonical_selector_hash(selector_version), + candidate_distinct_graph_count=int(candidate_distinct_graph_count or 0), + fresh_graph_evidence=bool( + fresh_graph_evidence + and remote_attempted + and docker_tag_resolution_is_conclusive(status) + ), + cache_bypassed=bool(fresh_graph_evidence), + ) + return (tags, status) if return_status else tags + + def cache_write(*values, **kwargs): + if not fresh_graph_evidence: + put_dockerhub_tag_cache(*values, **kwargs) + + try: + retry_count = max(0, min(5, int(retry_count or 0))) + except (TypeError, ValueError): + retry_count = 2 + try: + retry_delay = max(0, min(60, int(retry_delay or 0))) + except (TypeError, ValueError): + retry_delay = 5 + + limit = docker_images_per_repository_limit(limit) + repo_name = str(repo_name or '').strip() + if '@' in repo_name: + try: + return output([parse_docker_target(repo_name)['target']], 'ok') + except (TypeError, ValueError): + return output([], 'unknown') + if ':' in repo_name.rsplit('/', 1)[-1]: + return output([], 'unknown') + + platform_variant = ( + f'{selector_version}:{str(platform_os).lower()}/{str(platform_arch).lower()}:' + f'filter={int(bool(platform_filter_enabled))}:candidates={int(platform_candidate_tags or 0)}' + ) + cached = None if fresh_graph_evidence else get_dockerhub_tag_cache( + repo_name, since, limit, platform_variant, return_status=True, + ) + if cached is not None: + return output(cached[0], cached[1]) + + rate_limit_state = dockerhub_tag_rate_limit_state() + if rate_limit_state['active']: + logger.info(f"Docker Hub tag API is rate-limited; deferring tag fetch for {repo_name} from cache state") + return output( + [], 'global_cooldown', remote_attempted=False, + retry_at=rate_limit_state['retry_at'], + error='Docker Hub shared rate-limit cooldown is active', + ) + + if '/' in repo_name: + namespace, name = repo_name.split('/', 1) + else: + namespace, name = 'library', repo_name + + url = ( + f"https://hub.docker.com/v2/namespaces/{quote(namespace, safe='')}" + f"/repositories/{quote(name, safe='')}/tags" + ) + last_error = None + for attempt in range(retry_count + 1): + _raise_if_scan_slot_fatal() + try: + if lease_renewal_callback is not None: + if not callable(lease_renewal_callback): + raise ValueError('Docker resolver lease renewal callback is invalid') + if lease_renewal_callback() is False: + raise DockerResolverLeaseLostError( + 'Docker resolver lease renewal was rejected' + ) + response = dockerhub_tags_response( + url, { + 'page_size': max( + 1, min(max(limit, int(platform_candidate_tags or 0)), 100), + ), + }, + ) + if response.status_code == 404: + cache_write( + repo_name, since, limit, 'not_found', [], + getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600), + 'Docker Hub repository not found', platform_variant, + ) + return output([], 'not_found', remote_attempted=True) + if response.status_code in (401, 403): + raise DockerRemoteAccessError( + f'Docker Hub tags endpoint returned HTTP {response.status_code}', + status='auth_failed', remote_attempted=True, + ) + response.raise_for_status() + graph_candidates = [] + supported_or_unknown = 0 + unsupported = 0 + unresolved_digest = 0 + resolution_status = '' + resolution_retry_at = None + resolution_error = '' + tag_payload = _bounded_docker_registry_json( + response, 'Docker Hub tags response', + ) + if not isinstance(tag_payload, dict) or not isinstance(tag_payload.get('results'), list): + raise DockerRegistryResolutionError('Docker Hub tags response is malformed') + registry_auth = DockerRegistryAuth(token='') + for source_index, tag in enumerate(tag_payload.get('results', [])): + _raise_if_scan_slot_fatal() + tag_name = tag.get('name') + if not tag_name: + continue + + revision = str(tag.get('last_updated') or tag.get('tag_last_pushed') or '').strip() + tag_updated = parse_dockerhub_datetime( + revision + ) + if since and tag_updated and tag_updated < since: + continue + platform_support = docker_tag_platform_support(tag, platform_os, platform_arch) if platform_filter_enabled else True + if platform_support is False: + unsupported += 1 + logger.info( + f"Skipping Docker tag {repo_name}:{tag_name}: no {platform_os}/{platform_arch} image" + ) + continue + supported_or_unknown += 1 + tagged_image = f"{repo_name}:{tag_name}" + try: + validate_docker_image_reference(tagged_image, require_digest=False) + except ValueError: + logger.warning('Skipping invalid Docker Hub image reference for %s tag %s', repo_name, tag_name) + continue + digest = dockerhub_tag_digest(tag, platform_filter_enabled, platform_os, platform_arch) + if not digest: + unresolved_digest += 1 + logger.warning('Deferring Docker tag without a valid content digest: %s', tagged_image) + continue + try: + graph_kwargs = ( + {'include_descriptors': True} if fresh_graph_evidence else {} + ) + if lease_renewal_callback is not None: + graph_kwargs['lease_renewal_callback'] = lease_renewal_callback + graph, registry_auth = resolve_docker_layer_graph( + repo_name, digest, platform_os, platform_arch, registry_auth, + **graph_kwargs, + ) + except DockerResolverLeaseLostError: + raise + except ScanSlotFatalError: + raise + except DockerRemoteAccessError as exc: + unresolved_digest += 1 + resolution_status = exc.status + resolution_retry_at = exc.retry_at + resolution_error = str(exc) + logger.warning( + 'Deferring Docker manifest graph for %s: %s', + tagged_image, str(exc)[:300], + ) + break + except Exception as exc: + unresolved_digest += 1 + logger.warning( + 'Deferring Docker manifest graph for %s: %s', tagged_image, str(exc)[:300], + ) + if isinstance(exc, ApiRequestError) or 'rate-limit' in str(exc).lower(): + break + continue + if graph is None: + unsupported += 1 + continue + target = validate_docker_image_reference( + f"{repo_name}@{graph['manifest_digest']}" + ) + graph_candidates.append({ + 'name': tag_name, + 'target': target, + 'repository': repo_name.lower(), + 'manifest_digest': graph['manifest_digest'], + **({ + 'manifest_media_type': graph['manifest_media_type'], + 'manifest_size_bytes': graph['manifest_size_bytes'], + 'config_digest': graph['config_digest'], + 'layer_descriptors': graph['layer_descriptors'], + } if fresh_graph_evidence else {}), + 'layers': graph['layers'], + 'source_index': source_index, + 'updated_at': ( + tag_updated.replace(tzinfo=timezone.utc).timestamp() + if tag_updated is not None and tag_updated.tzinfo is None + else tag_updated.timestamp() if tag_updated is not None else None + ), + }) + candidate_distinct_graph_count = len({ + tuple(candidate['layers']) for candidate in graph_candidates + }) + candidate_records = ( + select_docker_layer_graphs( + graph_candidates, + min(100, candidate_distinct_graph_count), + replacement_pool=True, + ) + if fresh_graph_evidence and candidate_distinct_graph_count + else () + ) + selected = ( + candidate_records[:limit] + if fresh_graph_evidence + else select_docker_layer_graphs(graph_candidates, limit) + ) + if fresh_graph_evidence: + candidate_records = [ + _docker_depth_selection_evidence( + record, candidate_distinct_graph_count, selector_version, + ) + for record in candidate_records + ] + selected = candidate_records[:limit] + tags = [record['target'] for record in selected] + ttl = getattr(scan_config, 'dockerhub_tag_cache_ttl_sec', 21600) if tags else getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600) + status = ( + 'partial' if tags and unresolved_digest + else resolution_status if resolution_status + else 'ok' if tags + else 'unknown' if unresolved_digest + else 'unsupported' if unsupported and not graph_candidates + else 'empty' + ) + if status != 'unknown': + if status == 'ok': + cache_write( + repo_name, since, limit, status, tags, ttl, + platform_variant=platform_variant, tag_records=selected, + ) + elif status not in ('partial',): + cache_write( + repo_name, since, limit, status, tags, ttl, + platform_variant=platform_variant, + ) + return output( + tags, status, remote_attempted=True, + retry_at=resolution_retry_at, error=resolution_error, + selection_records=selected, candidate_records=candidate_records, + candidate_distinct_graph_count=candidate_distinct_graph_count, + ) + except DockerResolverLeaseLostError: + raise + except ScanSlotFatalError: + raise + except DockerRemoteAccessError as exc: + logger.warning('Deferring Docker tag resolution for %s: %s', repo_name, str(exc)) + return output( + [], exc.status, remote_attempted=exc.remote_attempted, + retry_at=exc.retry_at, error=str(exc), + ) + except Exception as e: + last_error = str(e) + if '404' in last_error or 'not found' in last_error.lower(): + cache_write(repo_name, since, limit, 'not_found', [], getattr(scan_config, 'dockerhub_tag_negative_cache_ttl_sec', 3600), last_error, platform_variant) + return output([], 'not_found', remote_attempted=True) + if attempt < retry_count and isinstance(e, ApiRequestError): + if lease_renewal_callback is not None and lease_renewal_callback() is False: + raise DockerResolverLeaseLostError( + 'Docker resolver lease renewal was rejected' + ) + _wait_or_raise_scan_slot_fatal(min(300, retry_delay * (attempt + 1))) + continue + break + logger.warning(f"Unable to fetch Docker Hub tags for {repo_name}: {last_error}") + return output( + [], 'unknown', remote_attempted=True, + error=last_error or 'Docker tag resolution failed', + ) + +def resolve_recent_dockerhub_image( + image, since, platform_filter_enabled=False, platform_os='linux', + platform_arch='amd64', platform_candidate_tags=20, + images_per_repository=1, resolve_tags=True, +): + repo_name = image.get('repo_name') + if not repo_name: + return [], 'missing_date' + + last_updated = parse_dockerhub_datetime( + image.get('last_updated') or image.get('last_modified') + ) + + if last_updated and last_updated < since: + return [], 'old' + + if last_updated is None: + last_updated = fetch_dockerhub_last_updated(repo_name) + if last_updated is None: + return [repo_name], 'recent' + if last_updated < since: + return [], 'old' + + if not resolve_tags: + return [repo_name], 'recent' + + tags, tag_status = fetch_dockerhub_tags( + repo_name, since=since, + limit=docker_images_per_repository_limit(images_per_repository), + platform_filter_enabled=platform_filter_enabled, + platform_os=platform_os, + platform_arch=platform_arch, + platform_candidate_tags=platform_candidate_tags, + return_status=True, + ) + if tags: + if tag_status != 'ok': + tags.append(repo_name) + return tags, 'recent' + + if not docker_tag_resolution_is_conclusive(tag_status): + return [repo_name], 'recent' + return [], 'missing_date' + +def fetch_recent_dockerhub_images( + query, since, per_page=100, pages=1, platform_filter_enabled=False, + platform_os='linux', platform_arch='amd64', platform_candidate_tags=20, + images_per_repository=1, resolve_tags=True, +): + """Fetch Docker Hub images updated since a specific timestamp""" + images = [] + page = 1 + per_page = max(1, int(per_page or 1)) + requested_pages, pages = dockerhub_search_page_window(pages) + if requested_pages > pages: + logger.info( + f'Docker Hub search is limited to {pages} accessible page(s); ' + f'capping requested pages from {requested_pages}' + ) + + logger.info(f"Fetching recent Docker Hub images updated since {since.strftime('%Y-%m-%d')}...") + + expected_pages = pages + while page <= expected_pages: + try: + logger.info(f"Docker Hub query '{query}': fetching page {page}/{pages}...") + page_result = fetch_dockerhub_search_page( + query, page, per_page=per_page, request_timeout=30, + ) + if page == 1: + expected_pages = min( + pages, + max(1, (page_result['total_count'] + per_page - 1) // per_page), + ) + repositories = page_result['repositories'] + + if not repositories: + logger.info( + f"Docker Hub query '{query}' returned no results " + f"(total matches: {page_result['total_count']})." + ) + break + + logger.info(f"Page {page}: checking dates/tags for {len(repositories)} Docker Hub repositories...") + new_images = [] + old_images = 0 + missing_dates = 0 + with concurrent.futures.ThreadPoolExecutor(max_workers=min(8, len(repositories))) as executor: + futures = [ + executor.submit( + resolve_recent_dockerhub_image, image, since, + platform_filter_enabled, platform_os, platform_arch, platform_candidate_tags, + docker_images_per_repository_limit(images_per_repository), + resolve_tags, + ) + for image in repositories + ] + for future in concurrent.futures.as_completed(futures): + tags, status = future.result() + if status == 'recent': + new_images.extend(tags) + elif status == 'old': + old_images += 1 + else: + missing_dates += 1 + + images.extend(new_images) + logger.info(f"Page {page}: fetched {len(new_images)} recent images, skipped {old_images} older images, skipped {missing_dates} without dates") + page += 1 + + except DockerHubDiscoveryTransportError: + raise + except Exception as e: + logger.error(f"Docker Hub recent discovery page {page} failed after bounded attempts") + raise DockerHubDiscoveryTransportError( + f'Docker Hub recent discovery page {page} failed after bounded attempts' + ) from e + + return images + + +def huggingface_space_to_target(space): + target = { + 'url': space.get('id') or space.get('name') or '', + 'name': space.get('id') or space.get('name') or '', + 'created_at': space.get('createdAt') or space.get('created_at') or '', + 'updated_at': space.get('lastModified') or space.get('updatedAt') or space.get('updated_at') or '', + } + for field in ('private', 'protected', 'gated', 'disabled'): + if field in space: + target[field] = space[field] + return target + + +def fetch_huggingface_spaces(pages=1, token=None, request_timeout=15, known_targets=None, normalize_target=None, stop_on_seen_pages=False, seen_page_threshold=2, min_pages_before_stop=1, known_target_lookup=None, return_metadata=False, request_attempts=1, retry_delay=0): + spaces = [] + headers = {'User-Agent': 'GitSecretsScanner/2.0', 'Accept': 'application/json'} + if token: + headers['Authorization'] = f'Bearer {token}' + seen_pages = 0 + request_attempts = max(1, int(request_attempts or 1)) + retry_delay = max(0, int(retry_delay or 0)) + request_budget = float(request_timeout) * request_attempts + retry_delay * (request_attempts - 1) + + if return_metadata: + next_url = 'https://huggingface.co/api/spaces' + next_params = {'sort': 'lastModified', 'direction': '-1', 'limit': 100} + logger.info(f"Fetching newest-modified HuggingFace Spaces for {pages} page(s)...") + for page in range(max(1, int(pages or 1))): + try: + response = api_request( + 'GET', next_url, headers=headers, params=next_params, + timeout=request_timeout, + max_retries=request_attempts, retry_delay=retry_delay, + deadline=time.monotonic() + request_budget, + ) + if response.status_code >= 400: + status = response.status_code + category = 'auth_invalid' if status == 401 else 'auth_forbidden' if status == 403 else 'rate_limit' if status == 429 else 'server_error' if status >= 500 else 'api' + reset_at = retry_after_reset(response) + response.close() + raise RateLimitError( + 'huggingface', f'HuggingFace discovery HTTP {status}', + reset_at=reset_at, category=category, + auth_related=status in (401, 403, 429), + ) + response.raise_for_status() + data = response.json() + if not isinstance(data, list): + raise ValueError('invalid HuggingFace spaces payload') + page_spaces = [ + huggingface_space_to_target(space) for space in data + if isinstance(space, dict) and space.get('id') + ] + if not page_spaces: + logger.info(f"HuggingFace page {page}: no spaces returned. Stopping.") + break + spaces.extend(page_spaces) + logger.info(f"HuggingFace page {page}: fetched {len(page_spaces)} newest-modified spaces") + next_link = (getattr(response, 'links', {}) or {}).get('next') or {} + next_url = str(next_link.get('url') or '') + next_params = None + if not next_url: + break + except Exception as e: + if isinstance(e, (ApiRequestError, RateLimitError)): + raise + logger.error(f"Error fetching HuggingFace page {page}: {str(e)}") + raise ApiRequestError(f'HuggingFace discovery payload failed: {e}') from e + return spaces + + logger.info(f"Fetching HuggingFace Spaces for {pages} page(s)...") + for page in range(max(1, int(pages or 1))): + url = 'https://huggingface.co/spaces-json' + params = {'p': page, 'withCount': 'false', 'sort': 'created'} + try: + response = api_request( + 'GET', url, headers=headers, params=params, timeout=request_timeout, + max_retries=request_attempts, retry_delay=retry_delay, + deadline=time.monotonic() + request_budget, + ) + if response.status_code >= 400: + status = response.status_code + category = 'auth_invalid' if status == 401 else 'auth_forbidden' if status == 403 else 'rate_limit' if status == 429 else 'server_error' if status >= 500 else 'api' + reset_at = retry_after_reset(response) + response.close() + raise RateLimitError( + 'huggingface', f'HuggingFace discovery HTTP {status}', + reset_at=reset_at, category=category, + auth_related=status in (401, 403, 429), + ) + response.raise_for_status() + data = response.json() + if not isinstance(data, dict) or 'spaces' not in data or not isinstance(data.get('spaces'), list): + raise ValueError('invalid HuggingFace spaces payload') + page_spaces = [huggingface_space_to_target(space) for space in data.get('spaces', []) if space.get('id')] + if not page_spaces: + logger.info(f"HuggingFace page {page}: no spaces returned. Stopping.") + break + spaces.extend(item['url'] for item in page_spaces if item.get('url')) + logger.info(f"HuggingFace page {page}: fetched {len(page_spaces)} spaces") + page_number = page + 1 + if stop_on_seen_pages and page_number >= max(1, min_pages_before_stop): + if page_is_known( + [item.get('url') for item in page_spaces], known_targets, + normalize_target, known_target_lookup, + ): + seen_pages += 1 + logger.info(f"HuggingFace page {page}: all spaces are already queued/checked ({seen_pages}/{seen_page_threshold})") + if seen_pages >= max(1, seen_page_threshold): + logger.info(f"Stopping HuggingFace pagination early after {seen_pages} all-known page(s)") + break + else: + seen_pages = 0 + except Exception as e: + if isinstance(e, (ApiRequestError, RateLimitError)): + raise + logger.error(f"Error fetching HuggingFace page {page}: {str(e)}") + raise ApiRequestError(f'HuggingFace discovery payload failed: {e}') from e + + return spaces + +# ===================== +# NPM FETCH FUNCTIONS +# ===================== +def parse_iso_datetime(value): + if not value: + return None + try: + parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.astimezone(timezone.utc) + except ValueError: + return None + +def npm_package_target(name, version, tarball_url, date=None): + return json.dumps({ + 'name': name, + 'version': version, + 'tarball': tarball_url, + 'date': date or '', + }, separators=(',', ':'), ensure_ascii=False) + +def parse_npm_target(target): + if isinstance(target, dict): + return target + target = str(target).strip() + if target.startswith('{'): + return json.loads(target) + package_id, tarball = target.split('|', 1) + name, version = package_id.rsplit('@', 1) + return {'name': name, 'version': version, 'tarball': tarball, 'date': ''} + +def npm_package_id(target): + data = parse_npm_target(target) + return f"npm:{data.get('name')}@{data.get('version')}" + +def select_npm_release_targets(name, data, max_versions=1, cutoff=None): + targets = [] + version_times = data.get('time') or {} + for version, version_data in (data.get('versions') or {}).items(): + tarball = (version_data.get('dist') or {}).get('tarball') + if not tarball: + continue + version_date = version_times.get(version) or '' + parsed_date = parse_iso_datetime(version_date) + if cutoff and (not parsed_date or parsed_date < cutoff): + continue + targets.append({ + 'name': name, + 'version': version, + 'tarball': tarball, + 'date': version_date, + 'parsed_date': parsed_date or datetime.min.replace(tzinfo=timezone.utc), + }) + targets.sort(key=lambda item: item['parsed_date'], reverse=True) + return targets[:max(1, int(max_versions or 1))] + +def fetch_npm_packages(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1, repo_candidate_callback=None): + """Fetch npm package tarball targets for recent versions matching a query.""" + if not query: + return [] + + targets = [] + seen_packages = set() + seen_versions = set() + cutoff = None + if max_version_age_days and max_version_age_days > 0: + cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days) + + logger.info(f"Fetching npm packages for query: '{query}'...") + for page in range(max(1, pages)): + params = {'text': query, 'size': per_page, 'from': page * per_page} + try: + response = api_request( + 'GET', + 'https://registry.npmjs.org/-/v1/search', + params=params, + headers={'User-Agent': 'GitSecretsScanner/2.0'}, + timeout=request_timeout, + ) + response.raise_for_status() + payload = response.json() + if not isinstance(payload, dict) or 'objects' not in payload or not isinstance(payload.get('objects'), list): + raise ValueError('invalid npm search payload') + objects = payload.get('objects', []) + if objects and not any(isinstance(item, dict) and isinstance(item.get('package'), dict) and item['package'].get('name') for item in objects): + raise ValueError('npm search payload contains no valid package entries') + if not objects: + logger.info(f"npm page {page + 1}: no results") + break + + logger.info(f"npm page {page + 1}: fetched {len(objects)} packages") + for item in objects: + package = item.get('package', {}) + name = package.get('name') + version = package.get('version') + if not name or not version: + continue + package_name_key = name.lower() + if package_name_key in seen_packages: + continue + + metadata_url = f"https://registry.npmjs.org/{quote(name, safe='')}" + metadata = api_request( + 'GET', + metadata_url, + headers={'User-Agent': 'GitSecretsScanner/2.0'}, + timeout=request_timeout, + ) + metadata.raise_for_status() + data = metadata.json() + selected_versions = select_npm_release_targets(name, data, versions_per_package, cutoff) + if repo_candidate_callback: + repo_candidates = [] + for version_item in selected_versions or [{'version': version}]: + item_version = version_item.get('version') or version + for candidate in extract_npm_git_candidates(data, item_version): + repo_candidates.append({ + 'package_source': 'npm', + 'name': name, + 'version': item_version, + 'repo_url': candidate['repo_url'], + 'provider': candidate['provider'], + 'evidence': candidate.get('evidence') or [], + 'confidence': 'high', + }) + repo_candidate_callback(repo_candidates) + for item in selected_versions: + package_key = f"{item['name']}@{item['version']}".lower() + if package_key in seen_versions: + continue + targets.append(npm_package_target(item['name'], item['version'], item['tarball'], item['date'])) + seen_versions.add(package_key) + seen_packages.add(package_name_key) + except Exception as e: + if isinstance(e, ApiRequestError): + raise + logger.error(f"Error fetching npm page {page + 1}: {str(e)}") + raise ApiRequestError(f'npm discovery failed: {e}') from e + + logger.info(f"npm query '{query}': prepared {len(targets)} package targets") + return targets + +# ====================== +# PYPI FETCH FUNCTIONS +# ====================== +pypi_project_index_cache_path = None + +def pypi_package_target(name, version, artifact_url, date=None, filename=None, packagetype=None, size=None): + return json.dumps({ + 'source': 'pypi', + 'name': name, + 'version': version, + 'artifact': artifact_url, + 'date': date or '', + 'filename': filename or '', + 'packagetype': packagetype or '', + 'size': size or 0, + }, separators=(',', ':'), ensure_ascii=False) + +def parse_pypi_target(target): + if isinstance(target, dict): + return target + target = str(target).strip() + if target.startswith('{'): + return json.loads(target) + package_id, artifact = target.split('|', 1) + name, version = package_id.rsplit('@', 1) + return {'source': 'pypi', 'name': name, 'version': version, 'artifact': artifact, 'date': ''} + +def pypi_package_id(target): + data = parse_pypi_target(target) + return f"pypi:{data.get('name')}@{data.get('version')}" + +# ============================= +# PACKAGE -> GIT FETCH HELPERS +# ============================= +GIT_PATH_STOP_SEGMENTS = { + '-', 'issues', 'issue', 'pull', 'pulls', 'merge_requests', 'merge_request', + 'tree', 'blob', 'commit', 'commits', 'releases', 'tags', 'branches', 'wiki', +} + + +def canonical_git_path_parts(host, parts): + if host == 'github.com': + if len(parts) < 2: + return [] + return parts[:2] + + cleaned = [] + for part in parts: + lowered = part.lower() + if lowered in GIT_PATH_STOP_SEGMENTS: + break + cleaned.append(part) + if len(cleaned) < 2: + return [] + return cleaned + + +def normalize_git_repo_candidate(value): + if not value: + return None + raw = str(value).strip().strip('"\'') + if not raw: + return None + + if raw.startswith('git+'): + raw = raw[4:] + if raw.startswith('github:'): + raw = 'https://github.com/' + raw.split(':', 1)[1] + elif raw.startswith('gitlab:'): + raw = 'https://gitlab.com/' + raw.split(':', 1)[1] + elif raw.startswith('git@github.com:'): + raw = 'https://github.com/' + raw.split(':', 1)[1] + elif raw.startswith('git@gitlab.com:'): + raw = 'https://gitlab.com/' + raw.split(':', 1)[1] + elif raw.startswith('git://'): + raw = 'https://' + raw[6:] + + try: + parsed = urlsplit(raw) + except ValueError: + return None + if parsed.scheme not in ('http', 'https') or not parsed.netloc: + return None + host = (parsed.hostname or '').lower() + if host in ('www.github.com',): + host = 'github.com' + if host in ('www.gitlab.com',): + host = 'gitlab.com' + if host not in ('github.com', 'gitlab.com'): + return None + + parts = [part for part in parsed.path.strip('/').split('/') if part] + repo_parts = canonical_git_path_parts(host, parts) + if not repo_parts: + return None + repo_parts[-1] = repo_parts[-1][:-4] if repo_parts[-1].endswith('.git') else repo_parts[-1] + if any(not part for part in repo_parts): + return None + provider = 'github' if host == 'github.com' else 'gitlab' + repo_path = '/'.join(repo_parts) + return { + 'provider': provider, + 'repo_url': f'https://{host}/{repo_path}.git', + 'repo_path': repo_path, + } + +def package_git_target(package_source, name, version, repo_url, provider, evidence=None, confidence='medium'): + return json.dumps({ + 'source': 'package_git', + 'package_source': package_source, + 'name': name, + 'version': version or '', + 'repo_url': repo_url, + 'provider': provider, + 'evidence': evidence or [], + 'confidence': confidence, + }, separators=(',', ':'), ensure_ascii=False) + +def parse_package_git_target(target): + if isinstance(target, dict): + return target + text = str(target).strip() + if text.startswith('{'): + return json.loads(text) + candidate = normalize_git_repo_candidate(text) + if not candidate: + raise ValueError(f'Unsupported package_git target: {text}') + return { + 'source': 'package_git', + 'package_source': 'custom', + 'name': '', + 'version': '', + 'repo_url': candidate['repo_url'], + 'provider': candidate['provider'], + 'evidence': ['custom'], + 'confidence': 'high', + } + +def package_git_id(target): + data = parse_package_git_target(target) + return f"package_git:{data.get('provider')}:{data.get('repo_url')}".lower() + +def collect_git_candidates(values): + candidates = [] + seen = set() + for evidence, value in values: + candidate = normalize_git_repo_candidate(value) + if not candidate: + continue + key = candidate['repo_url'].lower() + if key in seen: + continue + seen.add(key) + candidate['evidence'] = [evidence] + candidates.append(candidate) + return candidates + + +GIT_URL_TEXT_RE = re.compile( + r'(?:https?://|git\+https?://|git://|git@)(?:github\.com[:/]|gitlab\.com[:/])' + r'[A-Za-z0-9_.-]+(?:/[A-Za-z0-9_.-]+){1,8}(?:\.git)?(?:/[A-Za-z0-9_.~/-]+)?', + re.IGNORECASE, +) + + +def git_candidates_from_text(label, text, max_urls=8): + if not text: + return [] + values = [] + seen = set() + for match in GIT_URL_TEXT_RE.finditer(str(text)): + raw = match.group(0).rstrip(').,;\'"<>') + if raw.startswith('git@github.com/'): + raw = raw.replace('git@github.com/', 'git@github.com:', 1) + if raw.startswith('git@gitlab.com/'): + raw = raw.replace('git@gitlab.com/', 'git@gitlab.com:', 1) + key = raw.lower() + if key in seen: + continue + seen.add(key) + values.append((label, raw)) + if len(values) >= max_urls: + break + return collect_git_candidates(values) + +def extract_npm_git_candidates(metadata, version=None): + values = [] + repository = metadata.get('repository') + if isinstance(repository, dict): + values.append(('repository.url', repository.get('url'))) + elif isinstance(repository, str): + values.append(('repository', repository)) + bugs = metadata.get('bugs') + if isinstance(bugs, dict): + values.append(('bugs.url', bugs.get('url'))) + values.append(('homepage', metadata.get('homepage'))) + + version_data = (metadata.get('versions') or {}).get(version or '', {}) + version_repository = version_data.get('repository') if isinstance(version_data, dict) else None + if isinstance(version_repository, dict): + values.append(('version.repository.url', version_repository.get('url'))) + elif isinstance(version_repository, str): + values.append(('version.repository', version_repository)) + if isinstance(version_data, dict): + version_bugs = version_data.get('bugs') + if isinstance(version_bugs, dict): + values.append(('version.bugs.url', version_bugs.get('url'))) + values.append(('version.homepage', version_data.get('homepage'))) + candidates = collect_git_candidates(values) + candidates.extend(git_candidates_from_text('readme.github_url', metadata.get('readme'))) + candidates.extend(git_candidates_from_text('description.github_url', metadata.get('description'))) + deduped = [] + seen = set() + for candidate in candidates: + key = candidate['repo_url'].lower() + if key in seen: + continue + seen.add(key) + deduped.append(candidate) + return deduped + +def extract_pypi_git_candidates(metadata): + info = metadata.get('info') or {} + values = [] + project_urls = info.get('project_urls') or {} + if isinstance(project_urls, dict): + for key, value in project_urls.items(): + label = str(key).lower() + if any(item in label for item in ('source', 'repository', 'repo', 'code', 'homepage', 'home', 'bug', 'issue', 'tracker')): + values.append((f'project_urls.{key}', value)) + values.append(('home_page', info.get('home_page'))) + values.append(('project_url', info.get('project_url'))) + candidates = collect_git_candidates(values) + candidates.extend(git_candidates_from_text('description.github_url', info.get('description'))) + candidates.extend(git_candidates_from_text('summary.github_url', info.get('summary'))) + deduped = [] + seen = set() + for candidate in candidates: + key = candidate['repo_url'].lower() + if key in seen: + continue + seen.add(key) + deduped.append(candidate) + return deduped + +def fetch_npm_package_git_repos(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1): + if not query: + return [] + targets = [] + seen_repos = set() + cutoff = None + if max_version_age_days and max_version_age_days > 0: + cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days) + + logger.info(f"Fetching npm package git repos for query: '{query}'...") + for page in range(max(1, pages)): + params = {'text': query, 'size': per_page, 'from': page * per_page} + try: + response = api_request( + 'GET', + 'https://registry.npmjs.org/-/v1/search', + params=params, + headers={'User-Agent': 'GitSecretsScanner/2.0'}, + timeout=request_timeout, + ) + response.raise_for_status() + payload = response.json() + if not isinstance(payload, dict) or 'objects' not in payload or not isinstance(payload.get('objects'), list): + raise ValueError('invalid npm package_git search payload') + objects = payload.get('objects', []) + if objects and not any(isinstance(item, dict) and isinstance(item.get('package'), dict) and item['package'].get('name') for item in objects): + raise ValueError('npm package_git payload contains no valid package entries') + if not objects: + logger.info(f"npm package_git page {page + 1}: no results") + break + logger.info(f"npm package_git page {page + 1}: fetched {len(objects)} packages") + for item in objects: + package = item.get('package', {}) + name = package.get('name') + if not name: + continue + metadata = api_request( + 'GET', + f"https://registry.npmjs.org/{quote(name, safe='')}", + headers={'User-Agent': 'GitSecretsScanner/2.0'}, + timeout=request_timeout, + ) + metadata.raise_for_status() + data = metadata.json() + selected_versions = select_npm_release_targets(name, data, versions_per_package, cutoff) or [{'version': package.get('version') or ''}] + for version_item in selected_versions: + version = version_item.get('version') or package.get('version') or '' + for candidate in extract_npm_git_candidates(data, version): + key = candidate['repo_url'].lower() + if key in seen_repos: + continue + seen_repos.add(key) + targets.append(package_git_target('npm', name, version, candidate['repo_url'], candidate['provider'], candidate['evidence'], 'high')) + except Exception as e: + if isinstance(e, ApiRequestError): + raise + logger.error(f"Error fetching npm package_git page {page + 1}: {str(e)}") + raise ApiRequestError(f'npm package_git discovery failed: {e}') from e + logger.info(f"npm package_git query '{query}': prepared {len(targets)} git repo targets") + return targets + +def fetch_pypi_package_git_repos(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1): + if not query: + return [] + targets = [] + seen_repos = set() + cutoff = None + if max_version_age_days and max_version_age_days > 0: + cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days) + + logger.info(f"Fetching PyPI package git repos for query: '{query}'...") + for page in range(1, max(1, pages) + 1): + try: + names = fetch_pypi_package_names(query, page, per_page, request_timeout) + if not names: + logger.info(f"PyPI package_git page {page}: no results") + break + logger.info(f"PyPI package_git page {page}: fetched {len(names)} package names") + for name in names: + try: + metadata = api_request( + 'GET', + f"https://pypi.org/pypi/{quote(name, safe='')}/json", + headers={'User-Agent': 'GitSecretsScanner/2.0'}, + timeout=request_timeout, + ) + metadata.raise_for_status() + data = metadata.json() + except requests.exceptions.HTTPError as e: + if e.response is not None and e.response.status_code == 404: + logger.info(f"Skipping PyPI package_git project {name}: metadata not found") + continue + raise ApiRequestError(f'PyPI package_git metadata failed for {name}: {e}') from e + except ApiRequestError: + raise + if not isinstance(data, dict) or not isinstance(data.get('info', {}), dict): + raise ApiRequestError(f'invalid PyPI package_git metadata payload for {name}') + release_files = select_pypi_release_files(data, cutoff, versions_per_package) + version = release_files[0]['version'] if release_files else (data.get('info') or {}).get('version') or '' + package_name = (data.get('info') or {}).get('name') or name + for candidate in extract_pypi_git_candidates(data): + key = candidate['repo_url'].lower() + if key in seen_repos: + continue + seen_repos.add(key) + targets.append(package_git_target('pypi', package_name, version, candidate['repo_url'], candidate['provider'], candidate['evidence'], 'medium')) + except Exception as e: + if isinstance(e, ApiRequestError): + raise + logger.error(f"Error fetching PyPI package_git page {page}: {str(e)}") + raise ApiRequestError(f'PyPI package_git discovery failed: {e}') from e + logger.info(f"PyPI package_git query '{query}': prepared {len(targets)} git repo targets") + return targets + +class _PyPIProjectParser(HTMLParser): + def __init__(self, sink): + super().__init__(convert_charrefs=True) + self.sink = sink + self.in_anchor = False + self.parts = [] + + def handle_starttag(self, tag, attrs): + if tag.lower() == 'a': + self.in_anchor = True + self.parts = [] + + def handle_endtag(self, tag): + if tag.lower() == 'a': + name = ''.join(self.parts).strip() + if name: + self.sink(name) + self.in_anchor = False + self.parts = [] + + def handle_data(self, data): + if self.in_anchor: + text = str(data or '') + if sum(len(part) for part in self.parts) + len(text) <= 512: + self.parts.append(text) + + +def _pypi_index_path(): + global pypi_project_index_cache_path + if pypi_project_index_cache_path: + return pypi_project_index_cache_path + state_dir = os.path.dirname(scan_limiter_db_path()) + require_private_directory(state_dir, create=False) + pypi_project_index_cache_path = os.path.join(state_dir, 'pypi_project_index.sqlite3') + return pypi_project_index_cache_path + + +def load_pypi_project_index(request_timeout=20): + path = _pypi_index_path() + refresh_sec = max(3600, int(os.getenv('PYPI_PROJECT_INDEX_REFRESH_SEC', '86400'))) + if not os.path.exists(path): + descriptor = os.open( + path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), 0o600, + ) + os.close(descriptor) + harden_private_file(path) + connection = sqlite3.connect(path, timeout=30) + try: + connection.execute('PRAGMA journal_mode=DELETE') + connection.execute('PRAGMA synchronous=FULL') + connection.executescript(''' + CREATE TABLE IF NOT EXISTS pypi_projects ( + normalized_name TEXT PRIMARY KEY, + name TEXT NOT NULL + ); + CREATE TABLE IF NOT EXISTS pypi_index_meta ( + id INTEGER PRIMARY KEY CHECK(id = 1), + refreshed_at REAL NOT NULL, + project_count INTEGER NOT NULL + ); + ''') + current = connection.execute( + 'SELECT refreshed_at, project_count FROM pypi_index_meta WHERE id = 1' + ).fetchone() + if current and time.time() - float(current[0]) < refresh_sec and int(current[1]) > 0: + return path + + response = _direct_request( + 'GET', 'https://pypi.org/simple/', + headers={'Accept': 'text/html', 'User-Agent': 'GitSecretsScanner/2.0'}, + timeout=request_timeout, + stream=True, + ) + response.raise_for_status() + connection.execute('BEGIN IMMEDIATE') + connection.execute('DELETE FROM pypi_projects') + batch = [] + count = 0 + + def accept(name): + nonlocal count + normalized = re.sub(r'[-_.]+', '-', name).lower() + if not normalized or len(normalized) > 512: + return + batch.append((normalized, name[:512])) + if len(batch) >= 1000: + connection.executemany( + 'INSERT OR IGNORE INTO pypi_projects(normalized_name, name) VALUES (?, ?)', + batch, + ) + count += len(batch) + batch.clear() + + parser = _PyPIProjectParser(accept) + decoder = codecs.getincrementaldecoder('utf-8')('strict') + response_bytes = 0 + response_max_bytes = max( + 1024 * 1024, int(os.getenv('PYPI_PROJECT_INDEX_MAX_BYTES', str(512 * 1024 * 1024))), + ) + try: + for chunk in response.iter_content(chunk_size=256 * 1024): + _raise_if_scan_slot_fatal() + if chunk: + response_bytes += len(chunk) + if response_bytes > response_max_bytes: + raise ApiRequestError('PyPI simple index exceeds its streamed byte bound') + parser.feed(decoder.decode(chunk)) + parser.feed(decoder.decode(b'', final=True)) + parser.close() + finally: + response.close() + if batch: + connection.executemany( + 'INSERT OR IGNORE INTO pypi_projects(normalized_name, name) VALUES (?, ?)', + batch, + ) + count += len(batch) + actual = int(connection.execute('SELECT COUNT(*) FROM pypi_projects').fetchone()[0]) + if actual <= 0: + raise ApiRequestError('PyPI simple index contains no valid project entries') + connection.execute( + '''INSERT INTO pypi_index_meta(id, refreshed_at, project_count) VALUES (1, ?, ?) + ON CONFLICT(id) DO UPDATE SET refreshed_at = excluded.refreshed_at, + project_count = excluded.project_count''', + (time.time(), actual), + ) + connection.commit() + logger.info('Streamed %d PyPI project names into the on-disk index', actual) + return path + except Exception: + connection.rollback() + raise + finally: + connection.close() + if os.path.exists(path): + harden_private_file(path) + +def pypi_name_rank(name, query): + normalized = name.lower() + query = query.lower() + parts = [part for part in re.split(r'[-_.]+', normalized) if part] + if normalized == query: + rank = 0 + elif normalized.startswith(query): + rank = 1 + elif query in parts: + rank = 2 + else: + rank = 3 + return rank, len(normalized), normalized + +def fetch_pypi_package_names(query, page=1, per_page=50, request_timeout=20): + query = query.strip().lower() + if not query: + return [] + + tokens = [token for token in re.split(r'\s+', query) if token] + path = load_pypi_project_index(request_timeout) + page_size = max(1, int(per_page or 50)) + requested_end = max(1, int(page)) * page_size + candidate_limit = min(100000, max(1000, requested_end * 20)) + connection = sqlite3.connect(f'file:{path.replace(os.sep, "/")}?mode=ro', uri=True, timeout=30) + try: + clauses = ' AND '.join('normalized_name LIKE ?' for _ in tokens) + rows = connection.execute( + f'''SELECT name FROM pypi_projects WHERE {clauses} + ORDER BY normalized_name LIMIT ?''', + (*[f'%{token}%' for token in tokens], candidate_limit), + ) + matches = [row[0] for row in rows] + finally: + connection.close() + matches.sort(key=lambda name: pypi_name_rank(name, query)) + + start = (max(1, int(page)) - 1) * page_size + return matches[start:start + page_size] + +def select_pypi_release_files(data, cutoff=None, max_versions=1): + release_candidates = [] + priority_by_type = {'sdist': 2, 'bdist_wheel': 1} + + for version, files in (data.get('releases') or {}).items(): + file_candidates = [] + for file_info in files or []: + if file_info.get('yanked'): + continue + artifact_url = file_info.get('url') + if not artifact_url: + continue + uploaded = file_info.get('upload_time_iso_8601') or file_info.get('upload_time') or '' + parsed_date = parse_iso_datetime(uploaded) + if cutoff and (not parsed_date or parsed_date < cutoff): + continue + file_candidates.append({ + 'version': version, + 'url': artifact_url, + 'date': uploaded, + 'parsed_date': parsed_date or datetime.min.replace(tzinfo=timezone.utc), + 'filename': file_info.get('filename') or '', + 'packagetype': file_info.get('packagetype') or '', + 'size': file_info.get('size') or 0, + 'priority': priority_by_type.get(file_info.get('packagetype'), 0), + }) + + if file_candidates: + release_date = max(item['parsed_date'] for item in file_candidates) + file_candidates.sort(key=lambda item: (item['priority'], item['parsed_date']), reverse=True) + release_candidates.append((release_date, file_candidates[0])) + + if not release_candidates: + return [] + release_candidates.sort(key=lambda item: item[0], reverse=True) + return [item[1] for item in release_candidates[:max(1, int(max_versions or 1))]] + +def select_pypi_release_file(data, cutoff=None): + files = select_pypi_release_files(data, cutoff, 1) + return files[0] if files else None + +def fetch_pypi_packages(query, pages=1, per_page=50, max_version_age_days=0, request_timeout=20, versions_per_package=1, repo_candidate_callback=None): + """Fetch PyPI package artifact targets for recent matching releases.""" + if not query: + return [] + + targets = [] + seen = set() + cutoff = None + if max_version_age_days and max_version_age_days > 0: + cutoff = datetime.now(timezone.utc) - timedelta(days=max_version_age_days) + + logger.info(f"Fetching PyPI packages for query: '{query}'...") + for page in range(1, max(1, pages) + 1): + try: + names = fetch_pypi_package_names(query, page, per_page, request_timeout) + if not names: + logger.info(f"PyPI page {page}: no results") + break + + logger.info(f"PyPI page {page}: fetched {len(names)} package names") + + for name in names: + normalized_name = name.lower() + if normalized_name in seen: + continue + + metadata_url = f"https://pypi.org/pypi/{quote(name, safe='')}/json" + try: + metadata = api_request( + 'GET', + metadata_url, + headers={'User-Agent': 'GitSecretsScanner/2.0'}, + timeout=request_timeout, + ) + metadata.raise_for_status() + data = metadata.json() + except requests.exceptions.HTTPError as e: + if e.response is not None and e.response.status_code == 404: + logger.info(f"Skipping PyPI project {name}: metadata not found") + seen.add(normalized_name) + continue + logger.warning(f"Skipping PyPI project {name}: metadata fetch failed: {str(e)}") + raise ApiRequestError(f'PyPI metadata failed for {name}: {e}') from e + except ApiRequestError: + raise + except Exception as e: + raise ApiRequestError(f'PyPI metadata payload failed for {name}: {e}') from e + if not isinstance(data, dict) or not isinstance(data.get('info', {}), dict): + raise ApiRequestError(f'invalid PyPI metadata payload for {name}') + + release_files = select_pypi_release_files(data, cutoff, versions_per_package) + if not release_files: + continue + + if repo_candidate_callback: + package_name = data.get('info', {}).get('name') or name + repo_candidates = [] + for release_file in release_files: + for candidate in extract_pypi_git_candidates(data): + repo_candidates.append({ + 'package_source': 'pypi', + 'name': package_name, + 'version': release_file['version'], + 'repo_url': candidate['repo_url'], + 'provider': candidate['provider'], + 'evidence': candidate.get('evidence') or [], + 'confidence': 'medium', + }) + repo_candidate_callback(repo_candidates) + + for release_file in release_files: + targets.append(pypi_package_target( + data.get('info', {}).get('name') or name, + release_file['version'], + release_file['url'], + release_file['date'], + release_file['filename'], + release_file['packagetype'], + release_file['size'], + )) + seen.add(normalized_name) + except Exception as e: + if isinstance(e, ApiRequestError): + raise + logger.error(f"Error fetching PyPI page {page}: {str(e)}") + raise ApiRequestError(f'PyPI discovery failed: {e}') from e + + logger.info(f"PyPI query '{query}': prepared {len(targets)} package targets") + return targets + +# ===================== +# SCANNING FUNCTIONS +# ===================== +def check_dependencies(): + """Verify required dependencies are installed""" + probe_dir = None + try: + work_dir = get_work_dir() + probe_dir = tempfile.mkdtemp(prefix='trufflehog-probe-', dir=work_dir) + harden_private_directory(probe_dir) + command = [get_trufflehog_cmd(), '--version', '--no-update'] + write_temp_owner(probe_dir, command, os.getpid()) + env = os.environ.copy() + strip_supervisor_credentials(env) + env['PATH'] = os.pathsep.join([ + os.path.expanduser('~/bin'), + os.path.expanduser('~/.local/bin'), + env.get('PATH', '') + ]) + prepend_client_git_environment(env) + env['TEMP'] = probe_dir + env['TMP'] = probe_dir + env['TMPDIR'] = probe_dir + require_trufflehog_launch_authority(command) + result = run_owned( + command, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + timeout=30, + env=env, + cwd=probe_dir, + creationflags=subprocess.CREATE_NO_WINDOW if os.name == 'nt' else 0, + ) + if result.returncode != 0: + stderr = (result.stderr or b'') if isinstance(result.stderr, bytes) else str(result.stderr or '').encode('utf-8', errors='replace') + detail = stderr.decode('utf-8', errors='replace').strip()[:500] + raise RuntimeError(f"TruffleHog version probe exited with code {result.returncode}: {detail or 'no stderr'}") + logger.info("trufflehog is installed and working") + return True + except Exception as e: + if isinstance(e, subprocess.TimeoutExpired): + detail = f"timed out after {e.timeout}s" + else: + detail = f"{type(e).__name__}: {e}" + logger.error("TruffleHog dependency probe failed: %s", redact_scan_command_text([detail])[:500]) + if isinstance(e, FileNotFoundError): + logger.error("TruffleHog executable was not found. Please install it.") + logger.error("Visit https://github.com/trufflesecurity/trufflehog for installation instructions.") + else: + logger.error("TruffleHog dependency probe failed closed; executable authority was not bypassed.") + return False + finally: + if probe_dir: + cleanup_command_work_dir(probe_dir) + + +def command_output_limits(): + if _client_scan_policy.get() is None: + stdout_mb = int_setting( + os.getenv('TRUFFLEHOG_STDOUT_MAX_MB'), + getattr(scan_config, 'trufflehog_stdout_max_mb', 32), + ) + stderr_mb = int_setting( + os.getenv('TRUFFLEHOG_STDERR_MAX_MB'), + getattr(scan_config, 'trufflehog_stderr_max_mb', 8), + ) + else: + stdout_mb = int(_scan_policy_value('trufflehog_stdout_max_mb', 32)) + stderr_mb = int(_scan_policy_value('trufflehog_stderr_max_mb', 8)) + requested_stdout = max(1, int_setting( + stdout_mb, 32, + )) * 1024 * 1024 + requested_stderr = max(1, int_setting( + stderr_mb, 8, + )) * 1024 * 1024 + event_limit = max(1, int(_scan_policy_value( + 'result_bundle_max_event_bytes', 64 * 1024 * 1024, + ))) + reserve = min(32 * 1024 * 1024, max(64 * 1024, event_limit // 4)) + output_budget = max(2, event_limit - reserve) + requested_total = requested_stdout + requested_stderr + if requested_total <= output_budget: + return requested_stdout, requested_stderr + stdout = max(1, (output_budget * requested_stdout) // requested_total) + stderr = max(1, output_budget - stdout) + return stdout, stderr + + +class CommandOutputLimitError(RuntimeError): + pass + + +class StreamedCommandOutput: + def __init__(self, stdout_file, stderr_file, returncode, max_stdout, max_stderr, synthetic_stderr=''): + self._stdout = stdout_file + self._stderr = stderr_file + self.returncode = int(returncode) + self.max_stdout = int(max_stdout) + self.max_stderr = int(max_stderr) + self.synthetic_stderr = str(synthetic_stderr or '') + + @staticmethod + def _lines(handle, byte_limit, max_line_bytes, max_lines, redactions=()): + handle.seek(0) + consumed = 0 + count = 0 + while consumed < byte_limit and count < max_lines: + raw = handle.readline(min(max_line_bytes + 1, byte_limit - consumed + 1)) + if not raw: + return + consumed += len(raw) + count += 1 + if len(raw) > max_line_bytes and not raw.endswith((b'\n', b'\r')): + raise CommandOutputLimitError(f'command output line exceeded {max_line_bytes} bytes') + line = raw.decode('utf-8', errors='replace') + if redactions: + line = redact_secrets(line, redactions) + yield line + if handle.read(1): + raise CommandOutputLimitError('command output exceeded its line or byte bound') + + def stdout_lines(self, max_line_bytes=16 * 1024 * 1024, max_lines=20000, redactions=()): + return self._lines( + self._stdout, self.max_stdout, max(1, int(max_line_bytes)), + max(1, int(max_lines)), redactions, + ) + + def stderr_lines(self, max_line_bytes=8192, max_lines=2000, redactions=()): + for line in self._lines( + self._stderr, self.max_stderr, max(1, int(max_line_bytes)), + max(1, int(max_lines)), redactions, + ): + yield line + if self.synthetic_stderr: + yield self.synthetic_stderr.rstrip('\r\n') + '\n' + + @staticmethod + def _raw_bytes(handle): + position = handle.tell() + try: + handle.seek(0) + return handle.read() + finally: + handle.seek(position) + + def raw_stdout_bytes(self): + return self._raw_bytes(self._stdout) + + def raw_stderr_bytes(self): + return self._raw_bytes(self._stderr) + + +@contextmanager +def streamed_output_from_text(stdout='', stderr='', returncode=0): + stdout_file = io.BytesIO(str(stdout).encode('utf-8')) + stderr_file = io.BytesIO(str(stderr).encode('utf-8')) + yield StreamedCommandOutput( + stdout_file, stderr_file, returncode, + max(1, len(stdout_file.getvalue())), max(1, len(stderr_file.getvalue())), + ) + + +def _check_command_staging(roots, deadline, output_files=()): + """Best-effort live staging watchdog, not a filesystem quota or atomic snapshot.""" + exceeded = 'TruffleHog staging limit exceeded' + unavailable = 'Unable to monitor TruffleHog staging' + total_bytes = 0 + entries = 0 + try: + unique_roots = [] + for root in sorted({os.path.normcase(os.path.abspath(root)) for root in roots}, key=len): + if not any(root == parent or root.startswith(os.path.join(parent, '')) for parent in unique_roots): + unique_roots.append(root) + # Check ancestors too: lstat on a child alone would follow a linked parent. + for root in unique_roots: + ancestor = root + while True: + if time.monotonic() >= deadline: + return unavailable + try: + info = os.lstat(ancestor) + except FileNotFoundError: + pass + else: + if stat.S_ISLNK(info.st_mode) or getattr(info, 'st_file_attributes', 0) & stat.FILE_ATTRIBUTE_REPARSE_POINT: + return unavailable + parent = os.path.dirname(ancestor) + if parent == ancestor: + break + ancestor = parent + # POSIX TemporaryFile output may be unlinked and thus absent from scandir. + for handle in output_files: + if time.monotonic() >= deadline: + return unavailable + info = os.fstat(handle.fileno()) + if info.st_nlink == 0: + total_bytes += info.st_size + entries += 1 + pending = list(unique_roots) + while pending: + if time.monotonic() >= deadline: + return unavailable + path = pending.pop() + try: + info = os.lstat(path) + if stat.S_ISLNK(info.st_mode) or getattr(info, 'st_file_attributes', 0) & stat.FILE_ATTRIBUTE_REPARSE_POINT: + return unavailable + if stat.S_ISREG(info.st_mode): + total_bytes += info.st_size + elif not stat.S_ISDIR(info.st_mode): + return unavailable + if total_bytes > 2 * 1024 ** 3: + return exceeded + if stat.S_ISDIR(info.st_mode): + with os.scandir(path) as children: + for child in children: + if time.monotonic() >= deadline: + return unavailable + entries += 1 + if entries > 100000: + return exceeded + pending.append(child.path) + except FileNotFoundError: + continue + if time.monotonic() >= deadline: + return unavailable + if total_bytes > 2 * 1024 ** 3 or entries > 100000: + return exceeded + except (OSError, ValueError): + return unavailable + return '' + + +@contextmanager +def run_command_streamed(cmd, timeout_sec, env=None, *, deadline=None, staging_roots=None, native_git_clone=False): + """Run one owned command; optional staging roots also monitor its private temp tree.""" + if type(native_git_clone) is not bool: + raise ValueError('native_git_clone must be an explicit boolean') + if native_git_clone and not staging_roots: + raise ValueError('native Git clone requires private staging roots') + command_work_dir = None + process = None + scan_slot = None + owns_scan_slot = False + release_scan_slot = True + stdout_file = None + stderr_file = None + max_stdout, max_stderr = command_output_limits() + fatal_slot_error = None + propagating_fatal_error = None + returncode = -1 + synthetic_stderr = '' + armed_owners = [] + command_owner_published = False + requested_timeout = ( + max(1, int(timeout_sec or 1)) + if deadline is None + else max(0.001, float(timeout_sec or 0.001)) + ) + command_deadline = time.monotonic() + requested_timeout + if deadline is not None: + deadline = float(deadline) + if not math.isfinite(deadline): + raise ValueError('command deadline must be finite') + command_deadline = min(command_deadline, deadline) + try: + _raise_if_scan_slot_fatal() + env = dict(os.environ if env is None else env) + for key in list(env): + if key.lower() in ('http_proxy', 'https_proxy', 'all_proxy', 'no_proxy'): + del env[key] + env['NO_PROXY'] = '*' + strip_supervisor_credentials(env) + if os.name == 'nt': + env['PATH'] = os.pathsep.join([ + os.path.expanduser('~/bin'), os.path.expanduser('~/.local/bin'), env.get('PATH', ''), + ]) + prepend_client_git_environment(env) + env['GIT_TERMINAL_PROMPT'] = '0' + env['GIT_ASKPASS'] = 'true' + min_free_gb = max(0.0, float(getattr(scan_config, 'min_free_gb', 0) or 0)) + min_free_bytes = int(min_free_gb * 1024 * 1024 * 1024) + + borrowed_scan_slot, scan_slot = scoped_scan_slot_lease() + if not borrowed_scan_slot: + scan_slot = acquire_scan_slot( + cmd, max(0.001, command_deadline - time.monotonic()), + ) + owns_scan_slot = True + if scan_slot and not scan_slot.releasable: + raise RuntimeError('scan slot is fail-closed after unconfirmed child termination') + + command_work_dir = create_command_work_dir() + shared_owners = _shared_staging_owners((command_work_dir, *(staging_roots or ()))) + staging_roots = (command_work_dir, *staging_roots) if staging_roots is not None else None + env['TEMP'] = command_work_dir + env['TMP'] = command_work_dir + env['TMPDIR'] = command_work_dir + if env.get('TRUF_GIT_TOKEN'): + if os.name == 'nt': + askpass_path = os.path.join(command_work_dir, 'git-askpass.cmd') + with open(askpass_path, 'w', encoding='ascii') as askpass: + askpass.write('@echo off\r\n') + askpass.write('echo %~1 | findstr /I "username" >nul\r\n') + askpass.write('if %errorlevel%==0 (echo %TRUF_GIT_USERNAME%) else (echo %TRUF_GIT_TOKEN%)\r\n') + else: + askpass_path = os.path.join(command_work_dir, 'git-askpass.sh') + with open(askpass_path, 'x', encoding='ascii', newline='\n') as askpass: + askpass.write( + '#!/bin/sh\n' + 'case "$1" in\n' + ' *[Uu][Ss][Ee][Rr][Nn][Aa][Mm][Ee]*) printf \'%s\\n\' "$TRUF_GIT_USERNAME" ;;\n' + ' *) printf \'%s\\n\' "$TRUF_GIT_TOKEN" ;;\n' + 'esac\n' + ) + harden_private_file(askpass_path) + if os.name != 'nt': + os.chmod(askpass_path, stat.S_IRWXU) + env['GIT_ASKPASS'] = askpass_path + + creationflags = ( + subprocess.CREATE_NEW_PROCESS_GROUP + | subprocess.CREATE_NO_WINDOW + ) if os.name == 'nt' else 0 + stdout_file = tempfile.TemporaryFile(mode='w+b', dir=command_work_dir) + stderr_file = tempfile.TemporaryFile(mode='w+b', dir=command_work_dir) + if native_git_clone: + require_git_clone_launch_authority(cmd) + if time.monotonic() >= command_deadline: + raise subprocess.TimeoutExpired(cmd, timeout_sec) + destination_parent = os.path.normcase(os.path.abspath(os.path.dirname(cmd[-1]))) + if destination_parent not in {os.path.normcase(os.path.abspath(root)) for root in staging_roots}: + raise RuntimeError('Git clone destination is outside its private staging parent') + staging_error = _check_command_staging(staging_roots, command_deadline, (stdout_file, stderr_file)) + if staging_error: + raise RuntimeError(staging_error) + else: + require_trufflehog_launch_authority(cmd) + if time.monotonic() >= command_deadline: + raise subprocess.TimeoutExpired(cmd, timeout_sec) + process_options = { + 'stdout': stdout_file, 'stderr': stderr_file, 'env': env, + 'cwd': command_work_dir, 'stdin': subprocess.DEVNULL, + 'creationflags': creationflags, + } + if os.name == 'nt': + job_memory_limit_bytes = _scan_policy_value( + 'trufflehog_job_memory_limit_bytes', 0, + ) + if isinstance(job_memory_limit_bytes, bool) or not isinstance(job_memory_limit_bytes, int) or job_memory_limit_bytes <= 0: + raise RuntimeError('trufflehog_job_memory_limit_bytes must be a positive integer on Windows') + process_options['job_memory_limit_bytes'] = job_memory_limit_bytes + job_cpu_weight = int(_scan_policy_value('trufflehog_windows_job_cpu_weight', 0)) + memory_priority = int(_scan_policy_value('trufflehog_windows_memory_priority', 0)) + if job_cpu_weight < 0 or job_cpu_weight > 9: + raise RuntimeError('trufflehog_windows_job_cpu_weight must be between 0 and 9') + if memory_priority < 0 or memory_priority > 5: + raise RuntimeError('trufflehog_windows_memory_priority must be between 0 and 5') + process_options['job_cpu_weight'] = job_cpu_weight + process_options['process_memory_priority'] = memory_priority + for root, marker in shared_owners: + armed_owners.append((root, marker)) + pending = dict(marker, child_pid=None, child_creation_time=None, child_executable=None) + atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), pending) + if time.monotonic() >= command_deadline: + raise subprocess.TimeoutExpired(cmd, timeout_sec) + process = OwnedProcess(cmd, **process_options) + if not process.job_membership_verified: + raise RuntimeError('TruffleHog exact Job membership was not verified') + if scan_slot and not scan_slot.set_child_pid(process.pid): + raise RuntimeError('unable to publish TruffleHog child identity to the scan slot') + write_temp_owner( + command_work_dir, cmd, process.pid, required=True, + owner_identity=getattr(process, 'payload_identity', None), + ) + command_owner_published = True + child_identity = getattr(process, 'payload_identity', None) + for root, marker in armed_owners: + if root == canonical_path(command_work_dir): + continue + if not isinstance(child_identity, dict) or any(not child_identity.get(field) for field in ('pid', 'creation_time', 'executable')): + raise RuntimeError('shared staging child identity is unavailable') + active = dict(marker, **{f'child_{field}': child_identity[field] for field in ('pid', 'creation_time', 'executable')}) + atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), active) + limit_error = '' + next_staging_check = 0.0 + while True: + completed = process.poll() is not None + if completed and staging_roots is None: + break + _raise_if_scan_slot_fatal() + stdout_size = os.fstat(stdout_file.fileno()).st_size + stderr_size = os.fstat(stderr_file.fileno()).st_size + if stdout_size > max_stdout: + limit_error = f'TruffleHog stdout exceeded {max_stdout} bytes' + elif stderr_size > max_stderr: + limit_error = f'TruffleHog stderr exceeded {max_stderr} bytes' + elif min_free_bytes: + try: + free_bytes = shutil.disk_usage(command_work_dir).free + except OSError as exc: + limit_error = ( + 'Unable to monitor TruffleHog staging' if staging_roots is not None + else f'Unable to monitor configured TruffleHog work volume free space: {exc}' + ) + else: + if free_bytes <= min_free_bytes: + limit_error = ( + f'Not enough free space on configured TruffleHog work volume {command_work_dir}: ' + f'{free_bytes / (1024 ** 3):.2f} GB free, minimum is {min_free_gb:.2f} GB' + ) + if not limit_error and staging_roots is not None and ( + completed or time.monotonic() >= next_staging_check + ): + limit_error = _check_command_staging( + staging_roots, command_deadline, (stdout_file, stderr_file), + ) + next_staging_check = time.monotonic() + 1.0 + if limit_error: + process.kill() + try: + process.wait(timeout=10) + except subprocess.TimeoutExpired: + release_scan_slot = False + limit_error += '; process tree termination failed' + synthetic_stderr = f'Error: {limit_error}' + returncode = -1 + break + if completed: + break + if time.monotonic() >= command_deadline: + raise subprocess.TimeoutExpired(cmd, timeout_sec) + if _scan_slot_fatal_event.wait(0.2): + _raise_if_scan_slot_fatal() + if not limit_error: + returncode = process.returncode + stdout_file.seek(0) + stderr_file.seek(0) + yield StreamedCommandOutput( + stdout_file, stderr_file, returncode, max_stdout, max_stderr, synthetic_stderr, + ) + except subprocess.TimeoutExpired: + if process and process.poll() is None: + process.kill() + try: + process.wait(timeout=5) + except subprocess.TimeoutExpired: + release_scan_slot = False + if stdout_file is None or stderr_file is None: + raise + timeout_error = f'Command timed out after {timeout_sec} seconds' + if staging_roots is not None: + staging_error = _check_command_staging( + staging_roots, command_deadline, (stdout_file, stderr_file), + ) + if staging_error: + timeout_error = f'Error: {staging_error}' + stdout_file.seek(0) + stderr_file.seek(0) + yield StreamedCommandOutput( + stdout_file, stderr_file, -1, max_stdout, max_stderr, + timeout_error, + ) + except ScanSlotFatalError as exc: + propagating_fatal_error = exc + raise + finally: + if process: + try: + child_running = process.poll() is None + except Exception: + child_running = True + if child_running: + try: + process.kill() + process.wait(timeout=5) + except Exception: + release_scan_slot = False + if not release_scan_slot: + if scan_slot: + scan_slot.mark_non_releasable() + detail = 'FATAL: TruffleHog child termination was not confirmed; scan capacity remains fail-closed' + _set_scan_slot_fatal(detail) + fatal_slot_error = propagating_fatal_error or ScanSlotFatalError(detail) + restore_error = None + if release_scan_slot: + for root, marker in armed_owners: + if command_owner_published and root == canonical_path(command_work_dir): + continue + try: + atomic_write_private_json(os.path.join(root, TEMP_OWNER_FILE), marker) + except Exception as exc: + restore_error = exc + if scan_slot and owns_scan_slot and release_scan_slot: + scan_slot.release() + if stdout_file: + stdout_file.close() + if stderr_file: + stderr_file.close() + if release_scan_slot and fatal_slot_error is None and propagating_fatal_error is None: + cleanup_command_work_dir(command_work_dir) + if fatal_slot_error is not None: + raise fatal_slot_error + if restore_error is not None: + raise RuntimeError('Unable to restore shared staging ownership after child termination') from restore_error + + +def run_command(cmd, timeout_sec, env=None): + """Bounded compatibility adapter; production parsers use run_command_streamed.""" + try: + with run_command_streamed(cmd, timeout_sec, env) as output: + stdout = ''.join(output.stdout_lines(max_line_bytes=output.max_stdout, max_lines=20000)) + stderr = ''.join(output.stderr_lines(max_line_bytes=output.max_stderr, max_lines=2000)) + return stdout, stderr, output.returncode + except ScanSlotFatalError: + raise + except Exception as exc: + return '', f'Error running command: {exc}', -1 + + +def _trufflehog_diagnostic_policy(line, source_type, returncode): + text = str(line or '').strip() + payload = None + try: + parsed = json.loads(text) + if isinstance(parsed, dict): + payload = parsed + except (TypeError, ValueError): + pass + + if payload and payload.get('errors'): + causes = payload['errors'] + limits = _trufflehog_diagnostic_limits() + if not isinstance(causes, list): + return 'error', 'trufflehog', True + if len(causes) > limits['errors'] or any( + isinstance(cause, str) and ( + len(cause) > limits['line_chars'] + or len(cause.encode('utf-8', errors='replace')) > limits['line_bytes'] + ) for cause in causes + ): + return 'error', 'output_limit', False + envelope = dict(payload) + del envelope['errors'] + envelope_policy = _trufflehog_diagnostic_policy(json.dumps(envelope), source_type, returncode) + if envelope_policy[1] == 'source_auth' and not (payload.get('error') or payload.get('message')): + envelope_policy = ('error', 'auth_or_permission', envelope_policy[2]) + policies = [] + for cause in causes: + if not isinstance(cause, str) or not cause.strip(): + continue + cause_payload = {'level': 'error', 'error': cause} + policy = _trufflehog_diagnostic_policy(json.dumps(cause_payload), source_type, returncode) + cause_payload['msg'] = payload.get('msg') or '' + contextual_policy = _trufflehog_diagnostic_policy(json.dumps(cause_payload), source_type, returncode) + # Keep known warnings, but not at the expense of an independent fatal cause. + if contextual_policy[0] == 'warning' and ( + policy[1] == 'trufflehog' + or ( + contextual_policy[1] == 'detector_timeout' and policy[1] == 'timeout' + and cause.strip().lower() == 'context deadline exceeded' + ) + ): + policy = contextual_policy + if policy[1] == 'source_auth': + # A nested diagnostic can be detector verification, not the selected source credential. + policy = ('error', 'auth_or_permission', policy[2]) + policies.append(policy) + # Envelope summaries are not additional causes, but explicit fatal details are. + if envelope_policy[0] == 'error' and ( + envelope_policy[1] != 'trufflehog' or payload.get('error') or payload.get('message') + ): + policies.append(envelope_policy) + fatal = [policy for policy in policies if policy[0] == 'error'] + if not fatal: + if policies and all(policy[0] == 'warning' for policy in policies): + return 'warning', policies[0][1], all(policy[2] for policy in policies) + return 'error', 'trufflehog', True + retryable = all(policy[2] for policy in fatal) + classes = {policy[1] for policy in fatal} + for error_class in ( + 'memory_limit', 'source_configuration', 'output_limit', 'source_resource', + 'source_auth', 'docker_registry_access', 'auth_or_permission', + ): + if error_class in classes: + return 'error', error_class, retryable + return 'error', next(iter(classes)) if len(classes) == 1 else 'mixed', retryable + + if ( + source_type == 'docker' and payload + and payload.get('error') and payload.get('message') + and payload['error'] != payload['message'] + ): + # Independent detail channels must survive codec recovery; reuse fatal priority and bounds. + policy = _trufflehog_diagnostic_policy(json.dumps({ + 'level': 'error', 'msg': payload.get('msg'), + 'errors': [str(payload['error']), str(payload['message'])], + }), source_type, returncode) + return policy if policy[0] == 'error' else ('error', 'trufflehog', True) + + message = str((payload or {}).get('msg') or '') + detail = str((payload or {}).get('error') or (payload or {}).get('message') or '') + level = str((payload or {}).get('level') or '').lower() + message_lower = message.lower() + detail_lower = detail.lower() + combined = f'{message_lower} {detail_lower}' if payload else text.lower() + direct_kind = _client_remote_execution_kind.get() + + if any(token in combined for token in ('virtualalloc', 'out of memory', 'cannot allocate memory')): + return 'error', 'memory_limit', False + if any(token in combined for token in ('unknown flag', 'unknown command', 'invalid detector', 'failed to load config')): + return 'error', 'source_configuration', False + if any(token in combined for token in ( + 'trufflehog stdout exceeded', 'trufflehog stderr exceeded', + )): + return 'error', 'output_limit', False + if any(token in combined for token in ( + 'no space left', 'not enough free space', 'disk quota', + 'trufflehog work_dir', 'configured command work directory', + 'temporary file', 'temporaryfile', 'disk fsync', + )): + return 'error', 'source_resource', True + if message_lower == 'a detector ignored the context timeout': + return 'warning', 'detector_timeout', False + if source_type == 'huggingface' and detail_lower == 'no repo found for repo': + return 'permanent', 'huggingface_no_repo', False + if source_type == 'docker' and 'no child with platform linux/amd64' in detail_lower: + return 'permanent', 'docker_no_linux_amd64', False + + if any(token in combined for token in ('timed out', 'timeout', 'deadline exceeded')): + return 'error', 'timeout', True + if ( + source_type == 'docker' and payload + and payload.get('msg') == 'error processing layer' + and payload.get('error') == 'unexpected EOF' + ): + # Keep the persisted retry class; layer EOF alone does not prove a network cause. + return 'error', 'network', True + if any(token in combined for token in ( + 'connection reset', 'connection aborted', 'connection refused', 'could not resolve host', + 'temporary failure', 'tls', 'ssl', 'proxy error', 'network is unreachable', 'unexpected eof', + )): + return 'error', 'network', True + if any(token in combined for token in (' 408', ' 429', ' 500', ' 502', ' 503', ' 504', 'too many requests', 'rate limit')): + return 'error', 'remote_transient', True + auth_error = any(token in combined for token in ( + 'authentication failed', 'unauthorized', 'invalid username or token', 'invalid api key', + 'bad credentials', + )) + if source_type == 'huggingface' and direct_kind == 'huggingface_space_v1' and ( + auth_error or any(token in combined for token in ( + 'permission denied', 'repository not found', 'private repository', + 'gated repo', ' 401', ' 403', + )) + ): + return 'permanent', 'huggingface_inaccessible', False + if source_type == 'docker' and direct_kind == 'docker_direct_v1' and ( + auth_error or any(token in combined for token in ( + 'permission denied', 'pull access denied', + 'requested access to the resource is denied', 'insufficient scope', + 'manifest unknown', 'name unknown', 'repository does not exist', + ' 401', ' 403', ' 404', + )) + ): + return 'permanent', 'docker_registry_access', False + if source_type == 'docker' and auth_error: + # A registry manifest may be private or stale; the rotating credential cannot be + # attributed from this target diagnostic, so it must not disable the whole pool. + return 'error', 'docker_registry_access', False + if auth_error: + return 'error', 'source_auth', True + if 'permission denied' in combined: + return 'error', 'auth_or_permission', True + if any(token in combined for token in ('killed by signal', 'terminated by signal', 'process terminated', 'segmentation fault')): + return 'error', 'command_exit', True + + if ( + source_type in ('docker', 'filesystem') and payload + and message == 'skipping file: size exceeds max allowed' + and not any(key in payload for key in ('error', 'errors', 'message')) + ): + return 'warning' if returncode == 0 else 'error', 'archive_member_size', False + + if returncode == 0: + if message_lower == 'error cleaning temporary artifacts': + return 'warning', 'cleanup', False + if message_lower == 'skipping result: invalid' and detail_lower == 'empty raw': + return 'warning', 'invalid_empty_result', False + if message_lower == 'non-critical error processing chunk': + return 'warning', 'chunk_processing', False + if source_type == 'git' and message_lower == 'error reading chunk' and detail_lower == 'brotli: excessive input': + return 'warning', 'chunk_read', False + if source_type in ('npm', 'pypi', 'postman', 'filesystem') and message_lower == 'error reading chunk' and any( + token in detail_lower for token in ('brotli:', 'flate: corrupt input', 'error identifying archive', 'invalid header') + ): + return 'warning', 'chunk_read', False + if source_type == 'docker' and message_lower == 'error processing layer' and detail_lower == 'gzip: invalid header': + return 'warning', 'docker_layer_gzip', False + + if payload and ('error' in level or 'error' in message_lower or payload.get('error')): + return 'error', 'trufflehog', True + if not payload and re.search(r'\b(error|failed|fatal|panic)\b', combined): + return 'error', 'command', True + # Only observed structured progress is exempt from retention, never error details. + if ( + source_type in ('docker', 'filesystem') and payload + and payload.get('logger') == 'trufflehog' + and not any(key in payload for key in ('error', 'errors', 'message')) + and ( + (level == 'info-0' and message in ('running source', 'finished scanning')) + or (level == 'info-2' and message in ( + 'trufflehog dev', 'starting scanner workers', 'starting detector workers', + 'starting verificationOverlap workers', 'starting notifier workers', 'enumerating source', + )) + or (source_type == 'docker' and level == 'info-2' and message in ( + 'scanning image', 'scanning image history', 'scanning image history entry', + 'scanning image layers', 'scanning layer', + )) + ) + ): + return 'routine', '', False + return 'info', '', False + + +def _trufflehog_diagnostic_limits(): + return { + 'lines': min(2000, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_lines', 2000), 2000))), + 'line_chars': min(8192, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_line_chars', 8192), 8192))), + 'line_bytes': min(8192, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_line_bytes', 8192), 8192))), + 'errors': min(200, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_errors', 200), 200))), + 'warnings': min(200, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_warnings', 200), 200))), + 'unclassified': min(20, max(1, int_setting(_scan_policy_value('trufflehog_diagnostic_max_unclassified', 20), 20))), + } + + +def _iter_output_lines(value): + if isinstance(value, str): + yield from io.StringIO(value) + return + if isinstance(value, bytes): + for raw_line in io.BytesIO(value): + yield raw_line.decode('utf-8', errors='replace') + return + if value is None: + return + yield from value + + +def apply_trufflehog_diagnostics( + results, stderr, returncode, source_type, require_completion=False, + redactions=(), +): + errors = [] + warnings = [] + warning_classes = [] + warning_retryability = [] + permanent = [] + unclassified = [] + limits = _trufflehog_diagnostic_limits() + line_count = 0 + timed_out_seen = False + finished_seen = False + output_limit = '' + + captured_stdout = captured_stderr = None + captured_transformation = None + if isinstance(stderr, StreamedCommandOutput): + captured_stdout = stderr.raw_stdout_bytes() + captured_stderr = stderr.raw_stderr_bytes() + captured_transformation = ( + 'legacy E-frame classification parsed raw process stderr with any ' + 'pre-existing configured parser redactions; ' + 'canonical process material preserves the pre-parse captured bytes' + ) + stderr = stderr.stderr_lines(redactions=redactions) + + try: + for raw_line in _iter_output_lines(stderr): + if line_count >= limits['lines']: + output_limit = f'total line limit of {limits["lines"]} exceeded' + break + line_count += 1 + if len(raw_line) > limits['line_chars']: + output_limit = f'line character limit of {limits["line_chars"]} exceeded' + break + if len(raw_line.encode('utf-8', errors='replace')) > limits['line_bytes']: + output_limit = f'line byte limit of {limits["line_bytes"]} exceeded' + break + line = raw_line.strip() + if not line: + continue + try: + payload = json.loads(line) + if isinstance(payload, dict) and str(payload.get('msg') or '').strip().lower() == 'finished scanning': + finished_seen = True + except (TypeError, ValueError): + pass + if 'timed out' in line.lower(): + timed_out_seen = True + severity, error_class, retryable = _trufflehog_diagnostic_policy(line, source_type, returncode) + if severity == 'warning': + if len(warnings) + len(permanent) >= limits['warnings']: + output_limit = f'retained warning limit of {limits["warnings"]} exceeded' + break + warnings.append(line) + warning_retryability.append(bool(retryable)) + if error_class: + warning_classes.append(error_class) + elif severity == 'permanent': + if len(warnings) + len(permanent) >= limits['warnings']: + output_limit = f'retained warning limit of {limits["warnings"]} exceeded' + break + permanent.append((line, error_class)) + elif severity == 'error': + if len(errors) >= limits['errors']: + output_limit = f'retained error limit of {limits["errors"]} exceeded' + break + errors.append((line, error_class, retryable)) + if error_class == 'memory_limit': + break + elif severity != 'routine': # Routine records still count against the hard limits above. + if len(unclassified) >= limits['unclassified']: + output_limit = f'retained unclassified limit of {limits["unclassified"]} exceeded' + break + unclassified.append(line) + except CommandOutputLimitError as exc: + output_limit = str(exc) + + if output_limit: + synthetic = ( + f'TruffleHog diagnostic output_limit reached: {output_limit}; ' + 'remaining diagnostic output was not retained' + )[:limits['line_chars']] + errors = errors[:max(0, limits['errors'] - 1)] + errors.append((synthetic, 'source_resource', True)) + + if permanent and not errors: + warnings.extend(line for line, _ in permanent) + warning_classes.extend(error_class for _, error_class in permanent if error_class) + classes = {error_class for _, error_class in permanent} + if classes in ({'huggingface_no_repo'}, {'huggingface_inaccessible'}): + results['skipped'] = 'HuggingFace Space repository is unavailable' + elif classes == {'docker_no_linux_amd64'}: + results['skipped'] = 'Docker image has no linux/amd64 manifest' + elif classes == {'docker_registry_access'}: + results['skipped'] = 'Docker image is unavailable to the worker' + else: + results['skipped'] = 'target is permanently unavailable' + results['error_class'] = next(iter(classes), 'permanent') + results['retryable'] = False + else: + warnings.extend(line for line, _ in permanent) + warning_classes.extend(error_class for _, error_class in permanent if error_class) + + completion_required = ( + source_type == 'docker' or bool(require_completion) + or (source_type == 'git' and 'chunk_read' in warning_classes) + ) + if completion_required and not errors and not permanent and not results.get('skipped'): + if returncode != 0 and finished_seen: + errors.append((f'TruffleHog exited with code {returncode} after the completion marker', 'wrapper_exit', True)) + elif returncode != 0 or not finished_seen: + errors.append((f'TruffleHog exited with code {returncode} before the completion marker', 'command_incomplete', True)) + elif returncode != 0 and not errors and not permanent: + errors.append((f'TruffleHog exited with code {returncode} without a fatal diagnostic', 'command_exit', True)) + elif returncode != 0 and warnings and not errors and not results.get('skipped'): + errors.append((f'TruffleHog exited with code {returncode} after non-fatal diagnostics', 'command_exit', True)) + + if errors: + results['errors'] = [line for line, _, _ in errors] + classes = [error_class for _, error_class, _ in errors if error_class] + results['error_class'] = classes[0] if len(set(classes)) <= 1 else 'mixed' + results['retryable'] = all(retryable for _, _, retryable in errors) + results['source_failure'] = any(error_class in classes for error_class in ('source_configuration', 'source_resource', 'source_auth')) + if results['source_failure']: + results['source_failure_category'] = 'source_auth' if 'source_auth' in classes else 'source_resource' if 'source_resource' in classes else 'source_configuration' + results['source_failure_auth_related'] = 'source_auth' in classes + if output_limit: + results['error_class'] = 'source_resource' + results['retryable'] = True + results['source_failure'] = True + results['source_failure_category'] = 'source_resource' + results['source_failure_auth_related'] = False + if warnings: + results['warnings'] = warnings + results['warning_classes'] = sorted(set(warning_classes)) + results['degraded'] = not bool(results.get('skipped')) + # Nonfatal coverage warnings must not suppress retries of fatal errors. + if warning_retryability and not errors: + warnings_retryable = all(warning_retryability) + results['retryable'] = bool( + results.get('retryable', True) + ) and warnings_retryable + + scan_meta = results.setdefault('scan_meta', {}) + scan_meta['trufflehog_returncode'] = returncode + scan_meta['trufflehog_finished'] = finished_seen + scan_meta['command_timed_out'] = returncode == -1 and timed_out_seen + scan_meta['diagnostic_lines_processed'] = line_count + if warning_retryability: + scan_meta['trufflehog_warnings_retryable'] = all(warning_retryability) + if output_limit: + scan_meta['diagnostic_output_limited'] = True + scan_meta['diagnostic_output_limit_reason'] = output_limit + if unclassified: + scan_meta['stderr_unclassified'] = unclassified + if results.get('errors') and captured_stderr is not None: + results['_diagnostic_raw_stdout_b64'] = base64.b64encode( + captured_stdout or b'' + ).decode('ascii') + results['_diagnostic_raw_stderr_b64'] = base64.b64encode( + captured_stderr + ).decode('ascii') + results['_diagnostic_stderr_transformation'] = captured_transformation + return results + + +def convert_package_git_unavailable_to_skip(result): + errors = result.get('errors') or [] + if not errors: + return result + for error in errors: + text = str(error).lower() + if not ( + ('repository not found' in text or 'project not found' in text) + and ('failed to clone' in text or 'remote:' in text or 'error preparing repo' in text) + ): + return result + result['warnings'] = list(result.get('warnings') or []) + list(errors) + result['warning_classes'] = sorted(set(list(result.get('warning_classes') or []) + ['package_git_repo_unavailable'])) + result['errors'] = [] + result['skipped'] = 'package_git repository is unavailable or private' + result['error_class'] = 'package_git_repo_unavailable' + result['retryable'] = False + result['degraded'] = False + return result + + +def apply_result_error_scope(result): + errors = result.get('errors') or [] + if not errors or result.get('error_class'): + return result + text = '\n'.join(str(error) for error in errors).lower() + if any(token in text for token in ( + 'no space left', 'not enough free space', 'disk quota', + 'unable to create npm work dir', 'unable to create pypi work dir', + 'unable to create postman work dir', 'unable to create github actions work dir', + 'unable to create gitlab ci work dir', + )): + result['error_class'] = 'source_resource' + result['retryable'] = True + result['source_failure'] = True + result['source_failure_category'] = 'source_resource' + return result + + +def append_trufflehog_findings(results, stdout): + invalid = [] + if _client_scan_policy.get() is None: + max_findings = max(1, int(os.getenv('TRUFFLEHOG_MAX_FINDINGS_PER_TARGET', '20000'))) + else: + max_findings = max(1, int(_scan_policy_value( + 'trufflehog_max_findings_per_target', 20000, + ))) + try: + lines = _iter_output_lines(stdout) + for line in lines: + if not line.strip(): + continue + try: + finding = json.loads(line) + if not isinstance(finding, dict): + raise ValueError('finding JSON is not an object') + if len(results.setdefault('findings', [])) >= max_findings: + results.setdefault('errors', []).append(f'TruffleHog findings exceeded {max_findings} per target') + results['error_class'] = 'output_limit' + results['retryable'] = False + break + results['findings'].append(finding) + except (json.JSONDecodeError, ValueError) as exc: + if len(invalid) < 5: + invalid.append(f'{str(exc)}: {line[:300]}') + except CommandOutputLimitError as exc: + invalid.append(str(exc)) + if invalid: + results.setdefault('errors', []).append('Malformed TruffleHog JSON output: ' + '; '.join(invalid)) + results['error_class'] = 'output_parse' + results['retryable'] = True + return results + +def parse_git_scan_target(target): + text = str(target or '').strip() + if not text.startswith('{'): + return {'url': text, 'branch': '', 'metadata': {}} + try: + data = json.loads(text) + except (TypeError, ValueError): + return {'url': text, 'branch': '', 'metadata': {}} + if not isinstance(data, dict): + return {'url': text, 'branch': '', 'metadata': {}} + return { + 'url': str(data.get('url') or data.get('repo_url') or text).strip(), + 'branch': str(data.get('branch') or '').strip(), + 'metadata': data, + } + + +def git_branch_ref(branch): + branch = str(branch or '') + resolution = { + 'provider': 'github', + 'repo_url': 'https://github.com/a/b.git', + 'repo_path': 'a/b', + 'branch': branch, + 'ref': f'refs/heads/{branch}', + 'head_sha': '0' * 40, + 'ref_source': 'explicit', + } + validate_git_resolution(resolution) + return resolution['ref'] + + +def normalize_git_scan_resolution_target(target, provider=None): + target_info = parse_git_scan_target(target) + raw_url = str(target_info['url'] or '').strip() + if not raw_url or re.search(r'[\x00-\x20\x7f]', raw_url): + raise ValueError('Git target URL is empty or contains control characters') + if re.search(r'%(?:2f|5c)', raw_url, flags=re.IGNORECASE): + raise ValueError('Git target URL contains an encoded path separator') + parse_url = raw_url[4:] if raw_url.startswith('git+') else raw_url + if parse_url.startswith(('github:', 'gitlab:')): + if any(marker in parse_url for marker in ('?', '#', '@')): + raise ValueError('Git target shorthand contains unsafe URL components') + else: + try: + parsed = urlsplit(parse_url) + port = parsed.port + except ValueError as exc: + raise ValueError('Git target URL is malformed') from exc + if ( + parsed.scheme not in ('http', 'https', 'git') or not parsed.netloc + or parsed.username is not None or parsed.password is not None + or parsed.query or parsed.fragment or port is not None + ): + raise ValueError('Git target URL contains unsupported or unsafe components') + normalized = normalize_git_repo_candidate(raw_url) + if not normalized: + raise ValueError('Git target is not a supported GitHub or GitLab repository') + expected_provider = str(provider or '').strip().lower() + if expected_provider and normalized['provider'] != expected_provider: + raise ValueError('Git target provider does not match the scan source') + + metadata = target_info['metadata'] + branch = str(target_info.get('branch') or '').strip() + raw_ref = str(metadata.get('ref') or '').strip() if isinstance(metadata, dict) else '' + ref_branch = '' + if raw_ref: + prefix = 'refs/heads/' + if not raw_ref.startswith(prefix): + raise ValueError('Git target ref must be a branch ref') + ref_branch = raw_ref[len(prefix):] + git_branch_ref(ref_branch) + if branch: + git_branch_ref(branch) + if branch and ref_branch and branch != ref_branch: + raise ValueError('Git target branch and ref hints conflict') + branch = branch or ref_branch + return normalized, branch + + +def git_ref_resolution_api_json( + provider, url, token, deadline, request_attempts, timeout_sec, max_response_bytes, +): + response = api_request( + 'GET', url, + headers=github_headers(token) if provider == 'github' else gitlab_headers(token), + timeout=max(0.001, float(timeout_sec)), max_retries=max(1, int(request_attempts)), + retry_delay=1, deadline=deadline, stream=True, allow_redirects=False, + ) + if 300 <= response.status_code < 400: + response.close() + raise ApiRequestError(f'{provider} ref resolution refused an HTTP redirect') + if response.status_code >= 400: + try: + error = github_api_error(response) if provider == 'github' else gitlab_api_error(response) + finally: + response.close() + raise error + payload = bounded_response_json(response, max_bytes=max_response_bytes) + if not isinstance(payload, dict): + raise ApiRequestError(f'{provider} ref resolution response is not an object') + return payload + + +def redacted_git_resolution_error(exc, token): + message = redact_secrets(str(exc), [token]) + if isinstance(exc, RateLimitError): + return RateLimitError( + exc.source, message, reset_at=exc.reset_at, category=exc.category, + retryable=exc.retryable, auth_related=exc.auth_related, + ) + if isinstance(exc, ApiRequestError): + return ApiRequestError(message) + return exc + + +def resolve_github_ref_head( + repo_path, token=None, ref_hint=None, *, request_attempts=2, timeout_sec=10, + deadline=None, max_response_bytes=1 << 20, +): + deadline = float(deadline) if deadline is not None else time.monotonic() + max(0.001, float(timeout_sec)) + encoded_repo = quote(str(repo_path), safe='/') + branch = str(ref_hint or '') + ref_source = 'explicit' if branch else 'provider_default' + try: + if not branch: + payload = git_ref_resolution_api_json( + 'github', f'https://api.github.com/repos/{encoded_repo}', token, deadline, + request_attempts, timeout_sec, max_response_bytes, + ) + branch = str(payload.get('default_branch') or '') + git_branch_ref(branch) + ref = f'refs/heads/{branch}' + payload = git_ref_resolution_api_json( + 'github', f'https://api.github.com/repos/{encoded_repo}/git/ref/{quote("heads/" + branch, safe="")}', + token, deadline, request_attempts, timeout_sec, max_response_bytes, + ) + obj = payload.get('object') + if payload.get('ref') != ref or not isinstance(obj, dict) or obj.get('type') != 'commit': + raise ApiRequestError('GitHub ref resolution returned a mismatched commit ref') + resolved = { + 'provider': 'github', 'repo_url': f'https://github.com/{repo_path}.git', + 'repo_path': str(repo_path), 'branch': branch, 'ref': ref, + 'head_sha': str(obj.get('sha') or '').lower(), 'ref_source': ref_source, + } + return validate_git_resolution(resolved) + except (RateLimitError, ApiRequestError) as exc: + sanitized = redacted_git_resolution_error(exc, token) + if sanitized is exc: + raise + raise sanitized from exc + + +def resolve_gitlab_ref_head( + repo_path, token=None, ref_hint=None, *, request_attempts=2, timeout_sec=10, + deadline=None, max_response_bytes=1 << 20, +): + deadline = float(deadline) if deadline is not None else time.monotonic() + max(0.001, float(timeout_sec)) + project_url = f'https://gitlab.com/api/v4/projects/{quote(str(repo_path), safe="")}' + branch = str(ref_hint or '') + ref_source = 'explicit' if branch else 'provider_default' + try: + if not branch: + payload = git_ref_resolution_api_json( + 'gitlab', project_url, token, deadline, request_attempts, timeout_sec, + max_response_bytes, + ) + branch = str(payload.get('default_branch') or '') + git_branch_ref(branch) + payload = git_ref_resolution_api_json( + 'gitlab', f'{project_url}/repository/branches/{quote(branch, safe="")}', + token, deadline, request_attempts, timeout_sec, max_response_bytes, + ) + commit = payload.get('commit') + if payload.get('name') != branch or not isinstance(commit, dict): + raise ApiRequestError('GitLab ref resolution returned a mismatched branch') + resolved = { + 'provider': 'gitlab', 'repo_url': f'https://gitlab.com/{repo_path}.git', + 'repo_path': str(repo_path), 'branch': branch, 'ref': f'refs/heads/{branch}', + 'head_sha': str(commit.get('id') or '').lower(), 'ref_source': ref_source, + } + return validate_git_resolution(resolved) + except (RateLimitError, ApiRequestError) as exc: + sanitized = redacted_git_resolution_error(exc, token) + if sanitized is exc: + raise + raise sanitized from exc + + +def resolve_git_scan_target( + target, provider, token=None, *, request_attempts=2, timeout_sec=10, + max_response_bytes=1 << 20, +): + normalized, branch = normalize_git_scan_resolution_target(target, provider) + resolver = resolve_github_ref_head if normalized['provider'] == 'github' else resolve_gitlab_ref_head + return resolver( + normalized['repo_path'], token, branch or None, request_attempts=request_attempts, + timeout_sec=timeout_sec, max_response_bytes=max_response_bytes, + ) + + +def validate_bound_git_scan_plan(plan, target, provider=None): + if not isinstance(plan, dict): + raise ValueError('exact Git scan requires a bound plan object') + required = { + 'version', 'provider', 'repo_url', 'repo_path', 'branch', 'ref', 'head_sha', + 'ref_source', 'base_sha', 'mode', 'baseline_depth', + } + if set(plan) != required or plan.get('version') != 1: + raise ValueError('bound Git scan plan has an unsupported shape') + resolution = validate_git_resolution(plan) + normalized, _ = normalize_git_scan_resolution_target(target, provider or resolution['provider']) + if normalized['repo_url'] != resolution['repo_url'] or normalized['repo_path'] != resolution['repo_path']: + raise ValueError('bound Git scan plan repository conflicts with the target') + mode = str(plan.get('mode') or '') + base_sha = plan.get('base_sha') + if base_sha is not None: + base_sha = str(base_sha).lower() + if not re.fullmatch(r'[a-f0-9]{40}|[a-f0-9]{64}', base_sha): + raise ValueError('bound Git scan plan has an invalid base SHA') + try: + baseline_depth = int(plan.get('baseline_depth')) + except (TypeError, ValueError) as exc: + raise ValueError('bound Git scan plan has an invalid baseline depth') from exc + if not 1 <= baseline_depth <= 1000000: + raise ValueError('bound Git scan plan baseline depth is out of range') + if ( + (mode == 'baseline' and base_sha is not None) + or (mode == 'delta' and (not base_sha or base_sha == resolution['head_sha'])) + or (mode == 'noop' and base_sha != resolution['head_sha']) + or mode not in ('baseline', 'delta', 'noop') + ): + raise ValueError('bound Git scan plan mode and base are inconsistent') + normalized_plan = { + 'version': 1, **resolution, 'base_sha': base_sha, + 'mode': mode, 'baseline_depth': baseline_depth, + } + if canonical_git_scan_plan_bytes(normalized_plan) != canonical_git_scan_plan_bytes(plan): + raise ValueError('bound Git scan plan is not normalized') + return normalized_plan, hashlib.sha256(canonical_git_scan_plan_bytes(plan)).hexdigest() + + +def git_delta_base_unavailable(result, base_sha): + diagnostics = '\n'.join(str(item) for item in ( + list(result.get('errors') or []) + list(result.get('warnings') or []) + )).lower() + if not diagnostics: + return False + base_markers = ( + 'bad object', 'unknown revision', 'invalid object', 'object not found', + 'reference not found', 'could not find commit', 'unable to resolve commit', + 'invalid since commit', 'since-commit', 'since commit', + ) + return any(marker in diagnostics for marker in base_markers) and ( + str(base_sha or '').lower()[:12] in diagnostics + or 'since' in diagnostics + or 'commit' in diagnostics + or 'revision' in diagnostics + or 'object' in diagnostics + ) + + +def git_checkout_recovery_allowed(result): + meta = result.get('scan_meta') or {} + errors = result.get('errors') or [] + if ( + os.name != 'nt' or len(errors) not in (1, 2) or result.get('error_class') != 'trufflehog' + or result.get('source_failure') or result.get('warnings') or result.get('degraded') or result.get('skipped') + or meta.get('trufflehog_returncode') != 1 or meta.get('trufflehog_finished') is not False + or meta.get('diagnostic_output_limited') or meta.get('command_timed_out') + ): + return False + limits = _trufflehog_diagnostic_limits() + companion = None + if len(errors) == 2: + companion_line = errors[0] + if ( + not isinstance(companion_line, str) + or len(companion_line) > limits['line_chars'] + or len(companion_line.encode('utf-8', errors='replace')) > limits['line_bytes'] + ): + return False + try: + companion = json.loads(companion_line) + except (TypeError, ValueError): + return False + if ( + not isinstance(companion, dict) + or set(companion) != { + 'level', 'ts', 'logger', 'msg', 'subcommand', 'repo', 'path', + 'args', 'error', + } + or companion.get('level') != 'info-0' + or companion.get('logger') != 'trufflehog' + or companion.get('msg') != 'git clone failed' + or companion.get('subcommand') != 'git clone' + or companion.get('args') != [] + or any(not isinstance(companion.get(key), str) or not companion.get(key) for key in ( + 'ts', 'repo', 'path', 'error', + )) + ): + return False + line = errors[-1] + if not isinstance(line, str) or len(line) > limits['line_chars'] or len(line.encode('utf-8', errors='replace')) > limits['line_bytes']: + return False + try: + payload = json.loads(line) + except (TypeError, ValueError): + return False + if ( + not isinstance(payload, dict) or payload.get('msg') != 'error running scan' + or payload.get('level') != 'error' or payload.get('errors') + ): + return False + detail = payload.get('error') + if not isinstance(detail, str): + return False + if companion is not None and companion['error'] not in detail: + return False + detail = detail.lower() + if not all(marker in detail for marker in ( + 'error preparing repo', 'error executing git clone: exit status 128', + 'clone succeeded, but checkout failed', + )): + return False + prefix, _, git_stderr = detail.partition('error executing git clone: exit status 128') + if re.search(r'\b(?:fatal|error):', prefix): + return False + quoted_path = r"(?:'(?:[^'\\\r\n]|\\.)+'|\"(?:[^\"\\\r\n]|\\.)+\")" + path_failure = False + for physical_line in git_stderr.lstrip(' ,').splitlines(): + line_match = re.match(r'^(?:remote:\s*)?(?:fatal|error):\s*(.*)$', physical_line.strip()) + if not line_match: + if re.search(r'\b(?:fatal|error):', physical_line): + return False + continue + cause = line_match.group(1) + if re.fullmatch(r'invalid path ' + quoted_path, cause): + path_failure = True + continue + long_path = re.fullmatch(r'(?:unable to create file |cannot create directory (?:at )?)(.+): filename too long', cause) + if long_path: + path = long_path.group(1) + if path.startswith(("'", '"')): + if not re.fullmatch(quoted_path, path): + return False + elif re.search(r'\b(?:fatal|error):', path): + return False + path_failure = True + elif cause != 'unable to checkout working tree': + return False + return path_failure + + +def scan_exact_git_plan( + target, plan, plan_sha256, timeout_sec, detectors, exclude_detectors, + no_verification, trufflehog_config, token, external_trufflehog_lifecycle, +): + deadline = time.monotonic() + max(0.001, float(timeout_sec or 1)) + provider = plan['provider'] + scan_url, secrets_to_redact = build_authenticated_git_url(plan['repo_url'], provider, token) + command_env = os.environ.copy() + _append_windows_git_longpaths(command_env, 'Exact Git') + if secrets_to_redact: + command_env['TRUF_GIT_TOKEN'] = token + command_env['TRUF_GIT_USERNAME'] = 'oauth2' if provider == 'gitlab' else 'x-access-token' + recovery_root = None + local_url = None + checkout_errors = [] + findings = [] + cleanup_safe = True + recovery = {'attempted': False, 'clone_succeeded': False, 'coverage_complete': False} + + def remaining(): + seconds = deadline - time.monotonic() + if seconds <= 0: + raise subprocess.TimeoutExpired('exact Git scan', timeout_sec) + return seconds + + def run_mode(mode): + nonlocal recovery_root, local_url + remaining() + cmd = [ + get_trufflehog_cmd(), 'git', local_url or scan_url, '--json', '--no-update', + '--branch', plan['head_sha'], + ] + if external_trufflehog_lifecycle: + cmd.append('--local-dev') + append_trufflehog_scan_args( + cmd, detectors, exclude_detectors, no_verification, trufflehog_config, + ) + if mode in ('baseline', 'baseline_reset'): + cmd.extend(['--max-depth', str(plan['baseline_depth'])]) + elif mode == 'delta': + cmd.extend(['--since-commit', plan['base_sha']]) + result = {'findings': findings, 'errors': []} + emit_client_scan_phase('scanning', { + 'integrated_operation': 'git_acquisition_and_scan', + 'execution_mode': mode, + }) + with run_command_streamed( + cmd, remaining(), command_env, deadline=deadline, + staging_roots=(recovery_root,) if recovery_root else None, + ) as output: + apply_trufflehog_diagnostics( + result, output, + output.returncode, 'git', require_completion=True, + redactions=secrets_to_redact, + ) + append_trufflehog_findings( + result, output.stdout_lines(redactions=secrets_to_redact), + ) + checkout_candidate = not recovery['attempted'] and git_checkout_recovery_allowed(result) + if checkout_candidate: + checkout_errors.extend(result['errors']) + if checkout_candidate: + remaining() + recovery['attempted'] = True + emit_client_scan_phase('cloning', { + 'operation': 'git_clone_recovery', + }) + recovery_root = create_command_work_dir() + destination = os.path.join(recovery_root, 'repo') + clone_cmd = [get_git_cmd(), 'clone', '--no-checkout', '--no-recurse-submodules', '--', scan_url, destination] + clone_result = {'errors': []} + with run_command_streamed( + clone_cmd, remaining(), command_env, deadline=deadline, + staging_roots=(recovery_root,), native_git_clone=True, + ) as output: + apply_trufflehog_diagnostics( + clone_result, output, output.returncode, 'git', + redactions=secrets_to_redact, + ) + try: + for _ in output.stdout_lines(max_line_bytes=8192, max_lines=2000, redactions=secrets_to_redact): + pass + except CommandOutputLimitError: + clone_result.setdefault('errors', []).append('Git clone output exceeded its diagnostic bounds') + clone_result.update(error_class='output_limit', retryable=False) + if clone_result.get('errors') or clone_result.get('skipped') or clone_result.get('degraded'): + return clone_result + recovery['clone_succeeded'] = True + # Native Windows TH expects the drive in the file URI authority, not /C:/. + from pathlib import Path + local_url = Path(destination).as_uri() + if os.name == 'nt': + local_url = local_url.replace('file:///', 'file://', 1) + return run_mode(mode) + return result + + execution_mode = plan['mode'] + continuity_reset = False + result = {'findings': [], 'errors': []} + if execution_mode == 'noop': + emit_client_scan_phase('scanning', { + 'operation': 'exact_git_noop', + 'execution_mode': 'noop', + }) + result = {'findings': [], 'errors': []} + else: + try: + result = run_mode(execution_mode) + if ( + execution_mode == 'delta' and (not recovery['attempted'] or recovery['clone_succeeded']) + and git_delta_base_unavailable(result, plan['base_sha']) + ): + execution_mode = 'baseline_reset' + continuity_reset = True + result = run_mode(execution_mode) + result.setdefault('scan_meta', {})['git_continuity_reset_reason'] = 'covered base unavailable' + except ScanSlotFatalError: + cleanup_safe = False + raise + except subprocess.TimeoutExpired: + result.setdefault('errors', []).append('Exact Git scan exhausted its absolute deadline') + result.update(error_class='timeout', retryable=True) + except Exception as exc: + message = redact_secrets(str(exc), [token]) + logger.error('Error scanning pinned Git repository %s: %s', plan['repo_url'], message) + result.setdefault('errors', []).append(f'Scan failed: {message}') + result.update(retryable=True, error_class='remote_transient') + finally: + if recovery_root and cleanup_safe: + try: + cleanup_command_work_dir(recovery_root) + except ScanSlotFatalError: + raise + except Exception as exc: + result.setdefault('errors', []).append('Git recovery cleanup failed: ' + redact_secrets(str(exc), [token])) + result.update(error_class='source_resource', retryable=True, source_failure=True, + source_failure_category='source_resource', source_failure_auth_related=False) + + result['findings'] = findings + # Freeze scan coverage before optional filtering or candidate staging adds warnings. + success = not result.get('errors') and not result.get('skipped') and not result.get('degraded') + result['git_scan_plan'] = plan + result['git_scan_execution'] = { + 'mode': execution_mode, + 'pinned': True, + 'success': bool(success), + 'coverage_complete': bool(success), + 'continuity_reset': continuity_reset, + 'plan_sha256': plan_sha256, + } + result.setdefault('scan_meta', {})['exact_git_scope'] = { + 'provider': plan['provider'], 'ref': plan['ref'], 'head_sha': plan['head_sha'], + 'base_sha': plan['base_sha'], 'mode': execution_mode, + 'baseline_depth': plan['baseline_depth'], 'ref_source': plan['ref_source'], + 'pinned': True, 'continuity_reset': continuity_reset, + } + result = apply_finding_filters(result, target) + if execution_mode != 'noop' and time.monotonic() >= deadline: + if not result.get('errors'): + result.update(error_class='timeout', retryable=True) + result.setdefault('errors', []).append('Exact Git scan exceeded its absolute deadline including cleanup and filtering') + result.setdefault('scan_meta', {})['git_deadline_exceeded'] = True + result['git_scan_execution'].update(success=False, coverage_complete=False) + if result.get('errors'): + result['git_scan_execution'].update(success=False, coverage_complete=False) + if recovery['attempted']: + recovery['coverage_complete'] = result['git_scan_execution']['coverage_complete'] + result.setdefault('scan_meta', {})['git_checkout_recovery'] = recovery + if checkout_errors and not result['git_scan_execution']['coverage_complete']: + result['errors'] = checkout_errors + list(result.get('errors') or []) + result.setdefault('error_class', 'trufflehog') + result.setdefault('retryable', True) + return result + + +def scan_git_repo(repo_url, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, provider=None, max_depth=None, max_commit_age_days=None, commit_lookup_pages=3, skip_if_commit_lookup_fails=True, external_trufflehog_lifecycle=False, git_plan=None): + """Scan a single Git repository for secrets""" + target_info = parse_git_scan_target(repo_url) + original_target = repo_url + repo_url = target_info['url'] + parsed_repo_url = urlsplit(repo_url) + if parsed_repo_url.username or parsed_repo_url.password: + return {"findings": [], "errors": ["Git target URL must not contain userinfo credentials"], "retryable": False, "error_class": "invalid_target"} + target_branch = target_info['branch'] + target_metadata = target_info['metadata'] + if git_plan is not None: + try: + plan, plan_sha256 = validate_bound_git_scan_plan(git_plan, original_target, provider) + except (TypeError, ValueError) as exc: + return { + 'findings': [], 'errors': [f'Bound Git plan rejected: {exc}'], + 'retryable': False, 'error_class': 'invalid_target', + } + return scan_exact_git_plan( + original_target, plan, plan_sha256, timeout_sec, detectors, exclude_detectors, + no_verification, trufflehog_config, token, external_trufflehog_lifecycle, + ) + branch_label = f" branch={target_branch}" if target_branch else '' + logger.info(f"Scanning Git repository: {repo_url}{branch_label}") + + emit_client_scan_phase('resolving', { + 'operation': 'recent_commit_boundary', + 'provider': str(provider or 'git'), + }) + boundary = recent_commit_boundary(repo_url, provider, token, max_commit_age_days, commit_lookup_pages) + if boundary.get('error'): + category = str(boundary.get('error_category') or 'unknown') + auth_related = bool(boundary.get('auth_related')) + return { + "findings": [], "errors": [boundary.get('reason') or 'commit age lookup failed'], + "error_class": 'source_auth' if auth_related else 'remote_transient', + "retryable": True, "source_failure": True, + "source_failure_category": category, + "source_failure_auth_related": auth_related, + "scan_meta": {**boundary, 'target_metadata': target_metadata}, + } + if boundary.get('skip'): + reason = boundary.get('reason', 'skipped by commit age filter') + logger.info(f"Skipping {repo_url}: {reason}") + if boundary.get('permanent') or skip_if_commit_lookup_fails: + return {"findings": [], "errors": [], "skipped": reason, "scan_meta": {**boundary, 'target_metadata': target_metadata}} + + effective_provider, _ = get_git_provider_and_path(repo_url, provider) + scan_url, secrets_to_redact = build_authenticated_git_url(repo_url, effective_provider, token) + cmd = [get_trufflehog_cmd(), 'git', scan_url, '--json', '--no-update'] + if external_trufflehog_lifecycle: + cmd.append('--local-dev') + append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) + if max_depth: + cmd.extend(['--max-depth', str(max_depth)]) + if target_branch: + cmd.extend(['--branch', target_branch]) + if boundary.get('since_commit'): + cmd.extend(['--since-commit', boundary['since_commit']]) + logger.info( + f"Scanning {repo_url} since commit {boundary['since_commit']} " + f"({boundary.get('recent_commit_count')} commits after {boundary.get('cutoff')})" + ) + + try: + command_env = os.environ.copy() + if secrets_to_redact: + command_env['TRUF_GIT_TOKEN'] = token + command_env['TRUF_GIT_USERNAME'] = 'oauth2' if effective_provider == 'gitlab' else 'x-access-token' + results = {"findings": [], "errors": []} + emit_client_scan_phase('scanning', { + 'integrated_operation': 'git_acquisition_and_scan', + }) + with run_command_streamed(cmd, timeout_sec, command_env) as output: + apply_trufflehog_diagnostics( + results, output, + output.returncode, 'git', + require_completion=external_trufflehog_lifecycle, + redactions=secrets_to_redact, + ) + append_trufflehog_findings( + results, output.stdout_lines(redactions=secrets_to_redact), + ) + + if boundary.get('since_commit') or target_metadata or target_branch: + results.setdefault("scan_meta", {}).update({**boundary, 'branch': target_branch, 'target_metadata': target_metadata}) + + return apply_finding_filters(results, original_target) + + except ScanSlotFatalError: + raise + except Exception as e: + logger.error(f"Error scanning repository {repo_url}: {str(e)}") + return {"findings": [], "errors": [f"Scan failed: {str(e)}"]} + + +DOCKER_ARCHIVE_MAX_DECODED_BYTES = 1 << 30 + + +def _require_docker_archive_policy(limits): + # This independent decoded-stream ceiling is part of docker-layer-execution-v4. + if limits['archive_max_size_bytes'] > DOCKER_ARCHIVE_MAX_DECODED_BYTES: + raise DockerLayerInfrastructureError( + 'archive_policy_incompatible', 'Docker member policy exceeds the decoded validation ceiling', + category='source_configuration', + ) + + +def validate_docker_content_artifact( + path, descriptor, *, deadline=None, max_member_bytes=256 << 20, + max_decoded_bytes=DOCKER_ARCHIVE_MAX_DECODED_BYTES, max_members=100000, +): + import zlib + + deadline = min(float(deadline) if deadline is not None else float('inf'), time.monotonic() + 30) + if not math.isfinite(deadline) or any( + isinstance(value, bool) or not isinstance(value, int) or not 0 < value <= bound + for value, bound in ((max_member_bytes, DOCKER_ARCHIVE_MAX_DECODED_BYTES), + (max_decoded_bytes, DOCKER_ARCHIVE_MAX_DECODED_BYTES), (max_members, 100000)) + ): + raise DockerLayerInfrastructureError( + 'archive_validation_bounds', 'Docker archive validation bounds are invalid', category='source_configuration', + ) + + def check_deadline(): + _raise_if_scan_slot_fatal() + if time.monotonic() >= deadline: + raise DockerContentScanError('archive_timeout', 'Docker archive validation deadline expired', True) + + check_deadline() + kind = str((descriptor or {}).get('kind') or '') + media_type = str((descriptor or {}).get('media_type') or '').strip().lower() + expected_size = (descriptor or {}).get('size') + try: + actual_size = os.path.getsize(path) + except OSError as exc: + raise DockerLayerInfrastructureError( + 'private_storage', 'Docker content artifact is unavailable', + category='source_resource', + ) from exc + if actual_size != expected_size: + raise DockerContentScanError( + 'size_mismatch', 'Docker content artifact size changed after verification', + ) + + if kind == 'config': + if media_type not in DOCKER_CONFIG_MEDIA_TYPES: + raise DockerContentScanError( + 'unsupported_media_type', 'Docker configuration media type is unsupported', + ) + if actual_size > 64 << 20: + raise DockerContentScanError('archive_limit', 'Docker configuration exceeds the validation bound') + try: + with open(path, 'rb') as config_file: + raw_config = config_file.read(actual_size + 1) + except OSError as exc: + raise DockerLayerInfrastructureError( + 'private_storage', 'Docker configuration artifact cannot be read', + category='source_resource', + ) from exc + try: + config = json.loads(raw_config.decode('utf-8')) + except (RecursionError, UnicodeDecodeError, TypeError, ValueError) as exc: + raise DockerContentScanError( + 'invalid_config_json', 'Docker configuration is not valid UTF-8 JSON', + ) from exc + if not isinstance(config, dict): + raise DockerContentScanError( + 'invalid_config_json', 'Docker configuration JSON must be an object', + ) + check_deadline() + return config + + if kind != 'layer' or media_type not in DOCKER_LAYER_MEDIA_TYPES: + raise DockerContentScanError( + 'unsupported_media_type', 'Docker layer media type is unsupported', + ) + zstd = None + if media_type.endswith('+zstd'): + try: + import zstandard as zstd + except ImportError as exc: + raise DockerLayerInfrastructureError( + 'archive_decoder_unavailable', 'Docker zstd validation capability is unavailable', + category='source_configuration', + ) from exc + decoded_bytes = 0 + members = 0 + terminated = False + + class TimedInput: + def read(self, size=-1): + check_deadline() + data = layer_file.read(min(size if size >= 0 else 65536, 65536)) + check_deadline() + return data + + class BoundedReader: + def read(self, size=-1): + nonlocal decoded_bytes + check_deadline() + data = decoder.read(min(size if size >= 0 else 65536, 65536, max_decoded_bytes - decoded_bytes + 1)) + decoded_bytes += len(data) + if decoded_bytes > max_decoded_bytes: + raise DockerContentScanError('archive_limit', 'Docker archive decoded byte bound exceeded') + check_deadline() + return data + + class BoundedTarInfo(tarfile.TarInfo): + @classmethod + def fromtarfile(cls, archive): + nonlocal terminated + try: + check_deadline() + member = super().fromtarfile(archive) + check_deadline() + return member + except tarfile.EOFHeaderError: + terminated = True + raise + except tarfile.HeaderError as exc: + # TarFile.next otherwise tolerates some corrupt headers after member one. + raise DockerContentScanError('invalid_layer_archive', 'Docker tar header is invalid') from exc + + def _proc_member(self, archive): + nonlocal members + check_deadline() + members += 1 + metadata = self.type in (tarfile.XHDTYPE, tarfile.XGLTYPE, tarfile.SOLARIS_XHDTYPE, + tarfile.GNUTYPE_LONGNAME, tarfile.GNUTYPE_LONGLINK) + if self.size < 0: + raise DockerContentScanError('invalid_layer_archive', 'Docker archive member size is negative') + if members > max_members or self.size > min(max_member_bytes, 1 << 20 if metadata else max_member_bytes): + raise DockerContentScanError('archive_limit', 'Docker archive member bound exceeded') + if self.type == tarfile.GNUTYPE_SPARSE: + raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse tar validation is unsupported') + member = super()._proc_member(archive) + check_deadline() + return member + + def _proc_pax(self, archive): + # Do not enter tarfile's unbounded hdrcharset/length regexes or sparse + # map parsers. Validate complete records before decoding/applying fields. + if self.size > 64 << 10: + raise DockerContentScanError('archive_limit', 'Docker PAX parse byte bound exceeded') + body = archive.fileobj.read(self._block(self.size)) + if len(body) != self._block(self.size): + raise DockerContentScanError('invalid_layer_archive', 'Docker PAX header is truncated') + body = body[:self.size] + records = [] + position = 0 + while position < len(body): + check_deadline() + if len(records) >= min(max_members, 1024): + raise DockerContentScanError('archive_limit', 'Docker PAX record bound exceeded') + space = body.find(b' ', position, min(position + 9, len(body))) + if space < 0: + raise DockerContentScanError('archive_limit', 'Docker PAX record length field exceeds its bound') + digits = body[position:space] + if not digits or not digits.isdigit(): + raise DockerContentScanError('invalid_layer_archive', 'Docker PAX record length is invalid') + end = position + int(digits) + if end > len(body) or end <= space + 3 or body[end - 1:end] != b'\n': + raise DockerContentScanError('invalid_layer_archive', 'Docker PAX record boundary is invalid') + equals = body.find(b'=', space + 1, min(end - 1, space + 258)) + if equals <= space + 1: + raise DockerContentScanError('invalid_layer_archive', 'Docker PAX keyword is invalid or oversized') + key, value = body[space + 1:equals], body[equals + 1:end - 1] + if key.startswith(b'GNU.sparse.'): + raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse PAX validation is unsupported') + converter = tarfile.PAX_NUMBER_FIELDS.get(key.decode('utf-8')) + if converter is not None: + if len(value) > (32 if converter is int else 64): + raise DockerContentScanError('archive_limit', 'Docker PAX numeric field exceeds its bound') + number = converter(value) + if converter is float and not math.isfinite(number): + raise DockerContentScanError('invalid_layer_archive', 'Docker PAX numeric field is not finite') + if key == b'size' and not 0 <= number <= max_member_bytes: + raise DockerContentScanError('archive_limit', 'Docker PAX member size exceeds its bound') + records.append((key, value)) + position = end + check_deadline() + pax_headers = archive.pax_headers.copy() + charset = next((value.decode('utf-8') for key, value in records if key == b'hdrcharset'), + pax_headers.get('hdrcharset')) + encoding = archive.encoding if charset == 'BINARY' else 'utf-8' + for key, value in records: + check_deadline() + key = self._decode_pax_field(key, 'utf-8', 'utf-8', archive.errors) + if key in tarfile.PAX_NAME_FIELDS: + value = self._decode_pax_field(value, encoding, archive.encoding, archive.errors) + else: + value = self._decode_pax_field(value, 'utf-8', 'utf-8', archive.errors) + pax_headers[key] = value + if len(pax_headers) > 1024: + raise DockerContentScanError('archive_limit', 'Docker global PAX field bound exceeded') + if self.type == tarfile.XGLTYPE: + archive.pax_headers = pax_headers + try: + member = self.fromtarfile(archive) + except tarfile.HeaderError as exc: + raise DockerContentScanError('invalid_layer_archive', 'Docker PAX header has no following member') from exc + if self.type in (tarfile.XHDTYPE, tarfile.SOLARIS_XHDTYPE): + check_deadline() + member._apply_pax_info(pax_headers, archive.encoding, archive.errors) + member.offset = self.offset + if 'size' in pax_headers: + archive.offset = member.offset_data + if member.isreg() or member.type not in tarfile.SUPPORTED_TYPES: + archive.offset += member._block(member.size) + check_deadline() + return member + + try: + with open(path, 'rb') as layer_file: + magic = layer_file.read(4) + layer_file.seek(0) + if zstd is not None: + # stream_reader can silently accept a truncated final frame. Check physical + # frame/block boundaries separately, then let the decoder verify checksums. + frames = 0 + while layer_file.tell() < actual_size: + check_deadline() + frames += 1 + start = layer_file.tell() + header = layer_file.read(18) + if frames > max_members: + raise DockerContentScanError('archive_limit', 'Docker zstd frame bound exceeded') + if len(header) >= 8 and 0x184d2a50 <= int.from_bytes(header[:4], 'little') <= 0x184d2a5f: + end = start + 8 + int.from_bytes(header[4:8], 'little') + if end > actual_size: + raise DockerContentScanError('invalid_layer_archive', 'Docker zstd skippable frame is truncated') + layer_file.seek(end) + continue + if header[:4] != b'\x28\xb5\x2f\xfd': + raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame header is invalid') + header_size = zstd.frame_header_size(header) + params = zstd.get_frame_parameters(header) + if params.window_size > max_decoded_bytes or ( + params.content_size != zstd.CONTENTSIZE_UNKNOWN and params.content_size > max_decoded_bytes + ): + raise DockerContentScanError('archive_limit', 'Docker zstd window bound exceeded') + layer_file.seek(start + header_size) + while True: + check_deadline() + block = layer_file.read(3) + if len(block) != 3: + raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame is truncated') + value = int.from_bytes(block, 'little') + block_type, block_size = (value >> 1) & 3, value >> 3 + if block_type == 3 or block_size > 128 << 10: + raise DockerContentScanError('invalid_layer_archive', 'Docker zstd block is invalid') + end = layer_file.tell() + (1 if block_type == 1 else block_size) + if value & 1 and params.has_checksum: + end += 4 + if end > actual_size: + raise DockerContentScanError('invalid_layer_archive', 'Docker zstd frame is truncated') + layer_file.seek(end) + if value & 1: + break + layer_file.seek(0) + decoder = zstd.ZstdDecompressor(max_window_size=max(1024, max_decoded_bytes)).stream_reader( + TimedInput(), read_across_frames=True, closefd=False, + ) + elif media_type in DOCKER_LAYER_GZIP_MEDIA_TYPES: + if not magic.startswith(b'\x1f\x8b'): + raise DockerContentScanError('invalid_layer_archive', 'Docker layer does not match its gzip media type') + decoder = gzip.GzipFile(fileobj=TimedInput()) + else: + decoder = layer_file + try: + with tarfile.open(fileobj=BoundedReader(), mode='r|', tarinfo=BoundedTarInfo) as archive: + for member in archive: + if member.size > max_member_bytes: + raise DockerContentScanError('archive_limit', 'Docker archive member bound exceeded') + if member.sparse is not None: + raise DockerContentScanError('unsupported_layer_archive', 'Docker sparse tar validation is unsupported') + if member.isfile(): + with archive.extractfile(member) as body: + while body.read(65536): + check_deadline() + if not terminated or archive.fileobj.read(512) != b'\0' * 512: + raise DockerContentScanError('invalid_layer_archive', 'Docker tar end marker is missing') + while True: + padding = archive.fileobj.read(65536) + if not padding: + break + if padding.strip(b'\0'): + raise DockerContentScanError('invalid_layer_archive', 'Docker tar has trailing non-padding content') + if decoded_bytes % 512: + raise DockerContentScanError('invalid_layer_archive', 'Docker tar padding is truncated') + finally: + if decoder is not layer_file: + decoder.close() + except DockerContentScanError: + raise + except (tarfile.TarError, EOFError, gzip.BadGzipFile, zlib.error, ValueError, RecursionError) as exc: + raise DockerContentScanError( + 'invalid_layer_archive', 'Docker layer is not a valid bounded tar archive', + ) from exc + except OSError as exc: + raise DockerLayerInfrastructureError( + 'private_storage', 'Docker layer artifact cannot be read', + category='source_resource', + ) from exc + except Exception as exc: + if zstd is not None and isinstance(exc, zstd.ZstdError): + raise DockerContentScanError('invalid_layer_archive', 'Docker zstd stream is invalid') from exc + raise + + +def attach_docker_content_provenance( + findings, plan, descriptor, positions, private_blob_path, +): + location = ( + f'docker://{plan["repository"]}@{plan["manifest_digest"]}/' + f'{descriptor["kind"]}/{descriptor["digest"]}' + ) + private_blob_path = os.path.normcase(os.path.abspath(private_blob_path)) + for finding in findings: + if not isinstance(finding, dict): + continue + source = finding.setdefault('SourceMetadata', {}) + data = source.setdefault('Data', {}) if isinstance(source, dict) else {} + filesystem = data.get('Filesystem') if isinstance(data, dict) else None + if isinstance(filesystem, dict): + original = str(filesystem.get('file') or '') + if private_blob_path and private_blob_path in os.path.normcase(original): + original = original[len(private_blob_path):].lstrip('/\\:') + filesystem['file'] = location + (f'/{original}' if original else '') + if isinstance(data, dict): + docker_content = { + 'image': plan['image'], + 'manifest_digest': plan['manifest_digest'], + 'blob_digest': descriptor['digest'], + 'descriptor_kind': descriptor['kind'], + 'positions': list(positions), + } + if plan['version'] == 2: + docker_content['payload_class'] = descriptor['payload_class'] + data['DockerContent'] = docker_content + return findings + + +def _docker_content_error_code(value, fallback='scan_failed'): + text = re.sub(r'[^a-z0-9_]+', '_', str(value or '').strip().lower()).strip('_') + return text[:64] if re.fullmatch(r'[a-z][a-z0-9_]{0,63}', text) else fallback + + +def _scan_docker_content_file( + destination, descriptor, limits, deadline, detectors=None, + exclude_detectors=None, no_verification=False, trufflehog_config=None, +): + _require_docker_archive_policy(limits) + validate_docker_content_artifact( + destination, descriptor, + deadline=min(deadline, time.monotonic() + limits['archive_timeout_sec']), + max_member_bytes=limits['archive_max_size_bytes'], + ) + remaining = deadline - time.monotonic() + if remaining <= 0: + raise DockerContentScanError('archive_timeout', 'Docker blob deadline expired before scanning', True) + command = [ + get_trufflehog_cmd(), 'filesystem', destination, '--json', '--no-update', + '--log-level', '2', + '--archive-max-size', f'{limits["archive_max_size_bytes"]}B', + '--archive-max-depth', str(limits['archive_max_depth']), + '--archive-timeout', f'{limits["archive_timeout_sec"]}s', + '--concurrency', str(limits['filesystem_concurrency']), + ] + append_trufflehog_scan_args(command, detectors, exclude_detectors, no_verification, trufflehog_config) + result = {'findings': [], 'errors': []} + with run_command_streamed( + command, remaining, os.environ.copy(), deadline=deadline, + staging_roots=(os.path.dirname(os.path.abspath(destination)),), + ) as output: + apply_trufflehog_diagnostics( + result, output, output.returncode, 'filesystem', require_completion=True, + ) + append_trufflehog_findings(result, output.stdout_lines()) + # Only wrapper-owned diagnostics can attribute a watchdog stop to staging. + staging_error = output.synthetic_stderr.removeprefix('Error: ').split(';', 1)[0] + if staging_error in ('TruffleHog staging limit exceeded', 'Unable to monitor TruffleHog staging'): + monitor_failed = staging_error == 'Unable to monitor TruffleHog staging' + result.update( + errors=[staging_error], + error_class='source_resource' if monitor_failed else 'staging_limit', + retryable=monitor_failed, + source_failure=monitor_failed, + source_failure_auth_related=False, + ) + result.pop('source_failure_category', None) + if monitor_failed: + result['source_failure_category'] = 'source_resource' + return result + if time.monotonic() >= deadline: + result['errors'].append('Docker blob scan exceeded its absolute deadline') + result['error_class'] = 'timeout' + result['retryable'] = True + return result + + +def scan_docker_layer_plan( + image_name, docker_layer_work, timeout_sec=600, detectors=None, + exclude_detectors=None, no_verification=False, trufflehog_config=None, + *, log_target=True, +): + try: + plan = validate_docker_layer_plan((docker_layer_work or {}).get('plan')) + plan_bytes = canonical_docker_layer_plan_bytes(plan) + plan_sha256 = hashlib.sha256(plan_bytes).hexdigest() + if plan_sha256 != str((docker_layer_work or {}).get('plan_sha256') or ''): + raise ValueError('Docker layer work plan hash is invalid') + if parse_docker_target(image_name)['image'].lower() != plan['image']: + raise ValueError('Docker layer work target does not match its plan') + except (TypeError, ValueError) as exc: + raise DockerLayerInfrastructureError( + 'invalid_bound_plan', 'Docker layer bound plan is invalid', + category='source_configuration', + ) from exc + + _require_docker_archive_policy(plan['limits']) + leased_by_digest = {} + for descriptor in plan['descriptors']: + if descriptor['coverage_state'] == 'leased': + entry = leased_by_digest.setdefault(descriptor['digest'], { + 'descriptor': descriptor, 'positions': [], + }) + entry['positions'].append(descriptor['position']) + + work_root = None + retain_work = False + created_paths = [] + findings = [] + finding_digests = {} + records = [] + failures = [] + bearer_auth = (docker_layer_work or {}).get('bearer_auth') + min_free_bytes = max(0, int((docker_layer_work or {}).get('min_free_bytes') or 0)) + execution_deadline = time.monotonic() + max(0.001, float(timeout_sec or 0.001)) + supplied_deadline = (docker_layer_work or {}).get('deadline') + if supplied_deadline is not None: + try: + supplied_deadline = float(supplied_deadline) + except (TypeError, ValueError) as exc: + raise DockerLayerInfrastructureError( + 'invalid_deadline', 'Docker layer execution deadline is invalid', + category='source_configuration', + ) from exc + if not math.isfinite(supplied_deadline): + raise DockerLayerInfrastructureError( + 'invalid_deadline', 'Docker layer execution deadline is invalid', + category='source_configuration', + ) + execution_deadline = min(execution_deadline, supplied_deadline) + try: + if leased_by_digest: + if time.monotonic() >= execution_deadline: + raise DockerLayerInfrastructureError( + 'execution_deadline', 'Docker layer execution deadline expired before work began', + category='remote_transient', + ) + try: + work_root = tempfile.mkdtemp(prefix='docker-layer-', dir=get_work_dir()) + harden_private_directory(work_root) + write_temp_owner(work_root, ['docker-layer-content'], os.getpid(), required=True) + except ScanSlotFatalError: + raise + except (OSError, RuntimeError, ValueError) as exc: + raise DockerLayerInfrastructureError( + 'private_storage', 'Docker layer private work storage is unavailable', + category='source_resource', + ) from exc + for digest, entry in leased_by_digest.items(): + descriptor = entry['descriptor'] + blob_started = time.monotonic() + blob_deadline = min( + execution_deadline, + blob_started + plan['limits']['blob_timeout_sec'], + ) + destination = os.path.join(work_root, f'blob-{len(created_paths):04d}') + verified_bytes = 0 + transfer_bytes = 0 + transfer_duration_ms = 0 + scan_duration_ms = 0 + blob_findings = [] + error_code = '' + blob_retryable = True + try: + if not records: + emit_client_scan_phase('downloading', { + 'operation': 'docker_blob_transfer', + }) + else: + emit_client_scan_phase('downloading', { + 'operation': 'additional_docker_blob_transfer', + }) + outcome = stream_docker_registry_blob( + plan['repository'], descriptor, destination, bearer_auth, + deadline=blob_deadline, min_free_bytes=min_free_bytes, + ) + bearer_auth = outcome.bearer_auth + created_paths.append(destination) + verified_bytes = outcome.verified_bytes + transfer_bytes = outcome.transfer_bytes + transfer_duration_ms = outcome.duration_ms + scan_started = time.monotonic() + emit_client_scan_phase('scanning', { + 'operation': 'docker_blob_scan', + }) + result = _scan_docker_content_file( + destination, descriptor, plan['limits'], blob_deadline, + detectors, exclude_detectors, no_verification, trufflehog_config, + ) + scan_duration_ms = max(0, int((time.monotonic() - scan_started) * 1000)) + if time.monotonic() >= blob_deadline and not result.get('errors'): + result['errors'] = ['Docker layer scan exceeded its absolute deadline'] + result['error_class'] = 'timeout' + result['retryable'] = True + blob_findings = list(result.get('findings') or []) + attach_docker_content_provenance( + blob_findings, plan, descriptor, entry['positions'], destination, + ) + finding_digests.update( + (id(finding), digest) + for finding in blob_findings if isinstance(finding, dict) + ) + if result.get('source_failure'): + raise DockerLayerInfrastructureError( + _docker_content_error_code( + result.get('source_failure_category'), 'scanner_infrastructure', + ), + 'Docker layer scanner infrastructure is unavailable', + category=str( + result.get('source_failure_category') or 'source_resource' + ), + auth_related=bool(result.get('source_failure_auth_related')), + ) + if ( + result.get('errors') or result.get('skipped') + or result.get('warnings') or result.get('degraded') + ): + diagnostic_code = result.get('error_class') + if not diagnostic_code and result.get('warning_classes'): + diagnostic_code = result['warning_classes'][0] + error_code = _docker_content_error_code( + diagnostic_code, + 'scan_incomplete' if result.get('warnings') else 'scan_failed', + ) + blob_retryable = bool( + result.get('retryable', not bool(result.get('skipped'))) + ) + except DockerLayerInfrastructureError: + raise + except DockerContentTransferError as exc: + error_code = _docker_content_error_code(exc.error_code, 'transfer_failed') + blob_retryable = exc.retryable + transfer_bytes = max(transfer_bytes, int(exc.transfer_bytes or 0)) + transfer_duration_ms = max( + transfer_duration_ms, int(exc.duration_ms or 0), + ) + except DockerContentScanError as exc: + error_code = _docker_content_error_code(exc.error_code, 'invalid_content') + blob_retryable = exc.retryable + except ScanSlotFatalError: + retain_work = True + raise + except Exception as exc: + raise DockerLayerInfrastructureError( + 'scanner_infrastructure', 'Docker layer scanner infrastructure failed', + category='source_resource', + ) from exc + finally: + if not retain_work and destination in created_paths: + try: + durable_unlink(destination) + except OSError as exc: + raise DockerLayerInfrastructureError( + 'private_cleanup', 'Docker layer artifact cleanup failed', + category='source_resource', + ) from exc + created_paths.remove(destination) + + terminal = ( + not blob_retryable + or int(descriptor['attempt']) >= int(descriptor['max_attempts']) + ) + status = ( + 'covered' if not error_code + else 'terminal_failed' if terminal + else 'retryable_failed' + ) + records.append({ + 'digest': digest, + 'lease_token': descriptor['lease_token'], + 'status': status, + 'verified_bytes': verified_bytes, + 'transfer_bytes': transfer_bytes, + 'transfer_duration_ms': transfer_duration_ms, + 'scan_duration_ms': scan_duration_ms, + 'finding_count': len(blob_findings), + 'error_code': error_code or None, + }) + findings.extend(blob_findings) + if error_code: + failures.append(f'Docker content {digest[:19]} failed: {error_code}') + except ScanSlotFatalError: + retain_work = True + raise + finally: + # Fatal process outcomes leave payloads and ownership evidence for the janitor. + if not retain_work: + for path in created_paths: + if os.path.lexists(path): + try: + durable_unlink(path) + except OSError as exc: + raise DockerLayerInfrastructureError( + 'private_cleanup', 'Docker layer artifact cleanup failed', + category='source_resource', + ) from exc + if work_root: + cleanup_command_work_dir(work_root) + + records_by_digest = {item['digest']: item for item in records} + if set(records_by_digest) != set(leased_by_digest): + raise DockerLayerInfrastructureError( + 'missing_execution', 'Docker layer execution metadata is incomplete', + category='source_resource', + ) + effective_descriptors = [] + for descriptor in plan['descriptors']: + effective_state = descriptor['coverage_state'] + if effective_state == 'leased': + effective_state = records_by_digest[descriptor['digest']]['status'] + effective_descriptors.append((descriptor, effective_state)) + + static_states = {item['coverage_state'] for item in plan['descriptors']} + if 'selected' in static_states: + failures.append('Docker content remains selected for the next durable checkpoint') + if 'shared_pending' in static_states: + failures.append('Docker content is pending under another fenced reservation') + result_states = {state for _, state in effective_descriptors} + has_retryable = bool( + result_states & {'selected', 'shared_pending', 'retryable_failed'} + ) + has_terminal = 'terminal_failed' in result_states + if has_terminal: + failures.append('Docker content exhausted its bounded attempt budget') + result = { + 'findings': findings, + 'errors': failures, + 'retryable': bool(has_retryable), + 'error_class': ('docker_content_retry' if has_retryable else 'docker_content_terminal') + if failures else None, + 'docker_layer_plan': plan, + 'docker_layer_execution': { + 'version': plan['version'], + 'plan_sha256': plan_sha256, + 'blobs': records, + }, + 'scan_meta': { + 'docker_layer_scope': { + 'manifest_digest': plan['manifest_digest'], + 'coverage_complete': all( + state == 'covered' for _, state in effective_descriptors + ), + 'selected_descriptors': sum( + 1 for item in plan['descriptors'] if item['selected'] + ), + 'leased_blobs': len(leased_by_digest), + 'covered_blobs': len({ + item['digest'] for item, state in effective_descriptors + if state == 'covered' + }), + 'newly_covered_blobs': sum( + 1 for item in records if item['status'] == 'covered' + ), + 'globally_reused_blobs': len({ + item['digest'] for item in plan['descriptors'] + if item['coverage_state'] == 'covered' + }), + 'retryable_failed_blobs': sum( + 1 for item in records if item['status'] == 'retryable_failed' + ), + 'terminal_failed_blobs': sum( + 1 for item in records if item['status'] == 'terminal_failed' + ), + 'pending_checkpoint_blobs': sum( + 1 for item in plan['descriptors'] + if item['coverage_state'] == 'selected' + ), + 'shared_pending_blobs': sum( + 1 for item in plan['descriptors'] + if item['coverage_state'] == 'shared_pending' + ), + 'selected_bytes': sum( + item['size'] for item in plan['descriptors'] if item['selected'] + ), + 'covered_bytes': sum( + item['size'] for item, state in effective_descriptors + if state == 'covered' + ), + 'skipped_bytes': sum( + item['size'] for item, state in effective_descriptors + if state == 'skipped' + ), + 'shared_pending_bytes': sum( + item['size'] for item, state in effective_descriptors + if state == 'shared_pending' + ), + 'retryable_failed_bytes': sum( + item['size'] for item, state in effective_descriptors + if state == 'retryable_failed' + ), + 'terminal_failed_bytes': sum( + item['size'] for item, state in effective_descriptors + if state == 'terminal_failed' + ), + 'transfer_bytes': sum(item['transfer_bytes'] for item in records), + 'transfer_duration_ms': sum( + item['transfer_duration_ms'] for item in records + ), + 'scan_duration_ms': sum(item['scan_duration_ms'] for item in records), + 'timeout_blobs': sum( + 1 for item in records + if item['error_code'] in ('timeout', 'transfer_timeout') + ), + 'skipped_descriptors': sum( + 1 for item in plan['descriptors'] if item['coverage_state'] == 'skipped' + ), + }, + }, + } + if not failures: + result.pop('error_class') + result.pop('retryable') + if 'skipped' in static_states: + result['degraded'] = True + result['warnings'] = ['Docker content plan intentionally skipped bounded descriptors'] + result['warning_classes'] = ['docker_content_budget'] + try: + result = apply_finding_filters( + result, plan['image'], log_target=log_target, + ) + except Exception as exc: + raise DockerLayerInfrastructureError( + 'result_filter', 'Docker layer result filtering failed', + category='source_configuration', + ) from exc + filtered_counts = Counter( + finding_digests.get(id(finding)) + for finding in result.get('findings', []) + if finding_digests.get(id(finding)) + ) + for record in result['docker_layer_execution']['blobs']: + record['finding_count'] = filtered_counts[record['digest']] + return result + + +def _docker_implicit_auth_present(): + if os.environ.get('DOCKER_TOKEN') or os.environ.get('REGISTRY_AUTH_FILE') or docker_token_manager.has_accounts(): + return True + homes = {os.environ.get('HOME'), os.environ.get('USERPROFILE'), os.path.expanduser('~')} + if os.environ.get('HOMEDRIVE') and os.environ.get('HOMEPATH'): + homes.add(os.environ['HOMEDRIVE'] + os.environ['HOMEPATH']) + paths = [os.path.join(home, '.docker', 'config.json') for home in homes if home and home != '~'] + xdg = os.environ.get('XDG_RUNTIME_DIR', '') + if xdg and not os.path.isabs(xdg): + return True + paths.append(os.path.join(xdg, 'containers', 'auth.json')) + for path in paths: + try: + os.lstat(path) + return True + except (FileNotFoundError, NotADirectoryError): + continue + except OSError: + return True + return False + + +def _recover_docker_image_contents( + image_ref, deadline, config_dir, limits, min_free_bytes, + detectors, exclude_detectors, no_verification, trufflehog_config, *, + implicit_auth_unsupported=False, anonymous_public_client=False, +): + from scanner_db import validate_docker_layer_limits + + result = {'findings': [], 'errors': []} + scope = {'coverage_complete': False, 'scanned_descriptors': 0, 'blob_transfer_attempted': False} + result['scan_meta'] = {'docker_full_recovery': scope} + phase = 'configuration' + diagnostic = {} + descriptor = None + work_root = None + retain_work = False + try: + limits = validate_docker_layer_limits(limits if limits is not None else { + 'config_max_bytes': 1 << 20, 'layer_max_bytes': 256 << 20, + 'image_max_bytes': 1 << 30, 'max_layers': 8, + 'archive_max_size_bytes': 256 << 20, 'archive_max_depth': 4, + 'archive_timeout_sec': 30, 'blob_timeout_sec': 600, + 'filesystem_concurrency': 2, 'blob_max_attempts': 3, + }) + _require_docker_archive_policy(limits) + phase = 'preflight' + if time.monotonic() >= deadline: + raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True) + try: + _, _, repository, digest = _dockerhub_manifest_target_parts(image_ref) + except (TypeError, ValueError) as exc: + raise DockerContentScanError('unsupported_recovery_target', 'Docker recovery requires an immutable Docker Hub target') from exc + phase = 'authentication' + bearer_auth = None + if anonymous_public_client: + if config_dir: + raise DockerLayerInfrastructureError( + 'recovery_auth_unavailable', + 'Anonymous Docker recovery received credential configuration', + category='source_configuration', + ) + elif config_dir: + # Only the already-managed credential pool is trusted. Never load an arbitrary + # Docker config or run its external credential helpers in the recovery path. + with docker_token_manager.lock: + matched = next((account for account in docker_token_manager.accounts if + os.path.normcase(os.path.abspath(account.config_dir)) == os.path.normcase(os.path.abspath(config_dir))), None) + if matched is None: + raise DockerLayerInfrastructureError( + 'recovery_auth_unavailable', 'Docker recovery cannot use this credential configuration', + category='source_configuration', + ) + excluded = {account.name for account in docker_token_manager.accounts if account.name != matched.name} + bearer_auth = docker_registry_bearer_token( + 'Bearer realm="https://auth.docker.io/token",service="registry.docker.io"', + repository, excluded_accounts=excluded, deadline=deadline, + ) + else: + # The native default keychain may use Docker/Podman configs without + # DOCKER_CONFIG. Inspect existence only; never read or invoke helpers. + if implicit_auth_unsupported or _docker_implicit_auth_present(): + raise DockerLayerInfrastructureError( + 'recovery_auth_unavailable', 'Docker recovery cannot map implicit credential identity', + category='source_configuration', + ) + phase = 'manifest_resolution' + emit_client_scan_phase('resolving', { + 'operation': 'docker_manifest_resolution_recovery', + }) + resolved, bearer_auth = resolve_docker_content_manifest( + image_ref, bearer_auth=bearer_auth, deadline=deadline, + anonymous_only=anonymous_public_client, + ) + phase = 'preflight' + if resolved['image'] != image_ref or resolved['manifest_digest'] != digest or resolved['repository'] != repository: + raise DockerContentScanError('recovery_identity_mismatch', 'Docker recovery manifest identity changed') + descriptors = [dict(resolved['config'], kind='config', position=0)] + [ + dict(item, kind='layer', position=index) for index, item in enumerate(resolved['layers'], 1) + ] + scope['descriptor_count'] = len(descriptors) + if len(resolved['layers']) > limits['max_layers']: + diagnostic.update(reason='count_bound', observed=len(resolved['layers']), limit=limits['max_layers']) + raise DockerContentScanError('recovery_budget', 'All Docker layers do not fit the recovery count bound') + unique = {} + for descriptor in descriptors: + kind, media, size = descriptor['kind'], descriptor.get('media_type'), descriptor['size'] + allowed = DOCKER_CONFIG_MEDIA_TYPES if kind == 'config' else DOCKER_LAYER_MEDIA_TYPES + if not isinstance(media, str) or media not in allowed: + # Reporting only: these official types do not expand recovery support. + known_media = DOCKER_CONFIG_MEDIA_TYPES | DOCKER_LAYER_MEDIA_TYPES | { + 'application/vnd.oci.image.layer.nondistributable.v1.tar', + 'application/vnd.oci.image.layer.nondistributable.v1.tar+gzip', + 'application/vnd.oci.image.layer.nondistributable.v1.tar+zstd', + 'application/vnd.docker.image.rootfs.foreign.diff.tar', + 'application/vnd.docker.image.rootfs.foreign.diff.tar.gzip', + 'application/vnd.oci.image.manifest.v1+json', + 'application/vnd.oci.image.index.v1+json', + 'application/vnd.docker.distribution.manifest.v1+json', + 'application/vnd.docker.distribution.manifest.v1+prettyjws', + 'application/vnd.docker.distribution.manifest.v2+json', + 'application/vnd.docker.distribution.manifest.list.v2+json', + } + diagnostic.update( + reason='media_type', + media_type=media if isinstance(media, str) and media in known_media else 'other', + ) + raise DockerContentScanError('unsupported_media_type', 'Docker recovery descriptor is unsupported') + if not normalize_docker_digest(descriptor['digest']): + diagnostic['reason'] = 'integrity' + raise DockerContentScanError('unsupported_media_type', 'Docker recovery descriptor is unsupported') + if isinstance(size, bool) or not isinstance(size, int) or size < 0: + raise DockerContentScanError('invalid_descriptor', 'Docker recovery descriptor size is invalid') + byte_limit = limits['config_max_bytes' if kind == 'config' else 'layer_max_bytes'] + if size > byte_limit: + diagnostic.update(reason='descriptor_byte_bound', observed=size, limit=byte_limit) + raise DockerContentScanError('recovery_budget', 'All Docker content does not fit the recovery byte bound') + entry = unique.setdefault(descriptor['digest'], {'descriptor': descriptor, 'positions': []}) + previous = entry['descriptor'] + if any(previous[key] != descriptor[key] for key in ('kind', 'size', 'media_type')): + raise DockerContentScanError('conflicting_descriptors', 'Docker recovery digest descriptors conflict') + entry['positions'].append(descriptor['position']) + descriptor = None + image_bytes = sum(entry['descriptor']['size'] for entry in unique.values()) + if image_bytes > limits['image_max_bytes']: + diagnostic.update(reason='image_byte_bound', observed=image_bytes, limit=limits['image_max_bytes']) + raise DockerContentScanError('recovery_budget', 'All Docker content does not fit the cumulative recovery bound') + if time.monotonic() >= deadline: + raise DockerContentScanError('timeout', 'Docker recovery deadline expired after preflight', True) + phase = 'staging' + work_root = tempfile.mkdtemp(prefix='docker-layer-', dir=get_work_dir()) + harden_private_directory(work_root) + write_temp_owner(work_root, ['docker-full-recovery'], os.getpid(), required=True) + for index, entry in enumerate(unique.values()): + descriptor = entry['descriptor'] + phase = 'blob_transfer' + blob_deadline = min(deadline, time.monotonic() + limits['blob_timeout_sec']) + if time.monotonic() >= blob_deadline: + raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True) + destination = os.path.join(work_root, f'blob-{index:04d}') + try: + blob_min_free_bytes = max(0, int(min_free_bytes)) + scope['blob_transfer_attempted'] = True + emit_client_scan_phase('downloading', { + 'operation': 'docker_blob_transfer_recovery', + 'descriptor_index': index, + }) + outcome = stream_docker_registry_blob( + repository, descriptor, destination, bearer_auth, + deadline=blob_deadline, min_free_bytes=blob_min_free_bytes, + anonymous_only=anonymous_public_client, + ) + bearer_auth = outcome.bearer_auth + phase = 'blob_scan' + emit_client_scan_phase('scanning', { + 'operation': 'docker_blob_scan_recovery', + 'descriptor_index': index, + }) + scanned = _scan_docker_content_file( + destination, descriptor, limits, blob_deadline, + detectors, exclude_detectors, no_verification, trufflehog_config, + ) + result['findings'].extend(attach_docker_content_provenance( + scanned['findings'], resolved, descriptor, entry['positions'], destination, + )) + if scanned.get('source_failure'): + raise DockerLayerInfrastructureError( + 'recovery_scanner_unavailable', 'Docker recovery scanner is unavailable', + category=scanned.get('source_failure_category') or 'source_resource', + ) + if any(scanned.get(key) for key in ('errors', 'warnings', 'degraded', 'skipped')): + raise DockerContentScanError( + 'recovery_scan_incomplete', 'Docker recovery scanner did not cover a blob', + bool(scanned.get('retryable', False)), + ) + scope['scanned_descriptors'] += len(entry['positions']) + except ScanSlotFatalError: + retain_work = True + raise + finally: + if not retain_work and os.path.lexists(destination): + previous_phase = phase + phase = 'cleanup' + durable_unlink(destination) + phase = previous_phase + descriptor = None + phase = 'completion' + if time.monotonic() >= deadline: + raise DockerContentScanError('timeout', 'Docker recovery deadline expired', True) + except ScanSlotFatalError: + retain_work = True + raise + except Exception as exc: + code = _docker_content_error_code(getattr(exc, 'error_code', None), 'recovery_infrastructure') + if isinstance(exc, DockerRemoteAccessError): + code = _docker_content_error_code(exc.status, 'recovery_remote') + elif isinstance(exc, DockerRegistryResolutionError): + code = 'recovery_manifest_invalid' + diagnostic.setdefault('reason', { + 'recovery_identity_mismatch': 'integrity', + 'invalid_descriptor': 'integrity', + 'conflicting_descriptors': 'integrity', + 'digest_mismatch': 'integrity', + 'size_mismatch': 'integrity', + 'invalid_layer_archive': 'integrity', + 'invalid_config_json': 'integrity', + 'recovery_manifest_invalid': 'manifest_invalid', + 'timeout': 'timeout', + 'transfer_timeout': 'timeout', + 'archive_timeout': 'timeout', + 'recovery_scan_incomplete': 'scan_incomplete', + }.get(code, 'configuration' if phase == 'configuration' else 'other')) + diagnostic['phase'] = phase + if descriptor is not None: + diagnostic.update(descriptor_kind=descriptor['kind'], descriptor_index=descriptor['position']) + scope['diagnostic'] = diagnostic + result['errors'].append(f'Docker full-image recovery incomplete: {code}') + result['error_class'] = code + result['retryable'] = bool(getattr(exc, 'retryable', True)) + if isinstance(exc, DockerRegistryResolutionError): + result['retryable'] = ( + isinstance(exc, DockerRemoteAccessError) + and exc.status not in ('target_forbidden', 'auth_failed') + ) + if result['retryable']: + result['source_failure'] = True + result['source_failure_category'] = 'remote_auth' if exc.status == 'auth_failed' else 'remote_rate_limit' if exc.status == 'rate_limited' else 'remote_transient' + elif anonymous_public_client and exc.status == 'auth_failed': + result['error_class'] = 'docker_registry_access' + result.pop('source_failure', None) + result.pop('source_failure_category', None) + elif isinstance(exc, DockerLayerInfrastructureError) or not isinstance(exc, (DockerContentScanError, DockerContentTransferError)): + result['source_failure'] = True + result['source_failure_category'] = getattr(exc, 'category', 'source_configuration' if isinstance(exc, ValueError) else 'source_resource') + result['source_failure_auth_related'] = bool(getattr(exc, 'auth_related', False)) + finally: + if work_root and not retain_work: + try: + cleanup_command_work_dir(work_root) + except ScanSlotFatalError: + raise + except (OSError, RuntimeError): + scope.setdefault('diagnostic', {'phase': 'cleanup', 'reason': 'cleanup'}) + result['errors'].append('Docker full-image recovery private cleanup failed') + result.update(error_class='private_cleanup', retryable=True, source_failure=True, + source_failure_category='source_resource') + if not result['errors'] and time.monotonic() >= deadline: + scope['diagnostic'] = {'phase': 'completion', 'reason': 'timeout'} + result.update(errors=['Docker full-image recovery exceeded its absolute deadline'], + error_class='timeout', retryable=True) + scope['coverage_complete'] = not result['errors'] and scope['scanned_descriptors'] == scope.get('descriptor_count', 0) > 0 + return result + + +def scan_docker_image(image_name, timeout_sec=1800, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, config_dir=None, trufflehog_concurrency=0, *, log_target=True, docker_recovery_limits=None, docker_recovery_min_free_bytes=20 << 30): + """Scan a Docker image for secrets""" + started = time.monotonic() + try: + timeout_sec = float(timeout_sec) + if not math.isfinite(timeout_sec) or timeout_sec <= 0: + raise ValueError('invalid timeout') + except (TypeError, ValueError): + return {'findings': [], 'errors': ['Docker scan requires a positive finite time budget'], + 'error_class': 'timeout', 'retryable': False} + deadline = started + timeout_sec + anonymous_public_client = ( + _client_remote_execution_kind.get() == 'docker_direct_v1' + ) + if anonymous_public_client and config_dir: + raise RuntimeError('remote Docker direct execution cannot use credentials') + try: + image_ref = ( + parse_dockerhub_digest_target(image_name)['image'] + if anonymous_public_client else parse_docker_target(image_name)['image'] + ) + except (TypeError, ValueError) as exc: + return { + "findings": [], "errors": [f"Docker image target rejected: {exc}"], + "error_class": "invalid_target", "retryable": False, + } + if log_target: + logger.info(f"Scanning Docker image: {image_ref}") + + cmd = [get_trufflehog_cmd(), 'docker', '--image', image_ref, '--json', '--no-update', '--local-dev', '--log-level', '2'] + concurrency = max(0, min(64, int(trufflehog_concurrency or 0))) + if concurrency: + cmd.extend(['--concurrency', str(concurrency)]) + append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) + + env = os.environ.copy() + if config_dir: + env['DOCKER_CONFIG'] = config_dir + implicit_auth_unsupported = ( + not anonymous_public_client + and not (config_dir or env.get('DOCKER_CONFIG')) + and _docker_implicit_auth_present() + ) + + results = {"findings": [], "errors": []} + emit_client_scan_phase('scanning', { + 'integrated_operation': 'docker_pull_and_scan', + }) + with run_command_streamed(cmd, max(0, deadline - time.monotonic()), env, deadline=deadline) as output: + apply_trufflehog_diagnostics(results, output, output.returncode, 'docker') + append_trufflehog_findings(results, output.stdout_lines()) + + for line in results.get('errors', []): + try: + diagnostic = json.loads(line) + except (TypeError, ValueError): + continue + if ( + isinstance(diagnostic, dict) and diagnostic.get('msg') == 'error processing layer' + and diagnostic.get('error') == 'unexpected EOF' + ): + results.setdefault('scan_meta', {})['docker_native_diagnostic'] = { + 'phase': 'native_layer_processing', 'subcause': 'unexpected_eof_ambiguous', + 'coverage_complete': False, + } + break + + codec_warnings = [] + for line in results.get('warnings', []): + try: + diagnostic = json.loads(line) + except (TypeError, ValueError): + continue + if isinstance(diagnostic, dict) and diagnostic.get('msg') == 'error processing layer' and diagnostic.get('error') == 'gzip: invalid header': + codec_warnings.append(line) + if codec_warnings: + if not results.get('errors') and not results.get('source_failure') and not results.get('skipped') and len(codec_warnings) == len(results.get('warnings', [])): + recovered = _recover_docker_image_contents( + image_ref, deadline, ( + None if anonymous_public_client + else config_dir or env.get('DOCKER_CONFIG') + ), + docker_recovery_limits, docker_recovery_min_free_bytes, + detectors, exclude_detectors, no_verification, trufflehog_config, + implicit_auth_unsupported=implicit_auth_unsupported, + anonymous_public_client=anonymous_public_client, + ) + results['findings'].extend(recovered.pop('findings')) + results.setdefault('scan_meta', {}).update(recovered.pop('scan_meta')) + if not recovered['errors'] and results['scan_meta']['docker_full_recovery']['coverage_complete']: + for key in ('warnings', 'warning_classes', 'degraded', 'retryable'): + results.pop(key, None) + results['scan_meta']['docker_full_recovery']['recovered_codec'] = True + else: + if not recovered['errors']: + recovered.update(errors=['Docker full-image recovery coverage is incomplete'], + error_class='docker_recovery_incomplete', retryable=False) + results['scan_meta']['docker_full_recovery'].setdefault( + 'diagnostic', {'phase': 'completion', 'reason': 'scan_incomplete'}, + ) + results.update(recovered) + else: + results['errors'].append('Docker codec recovery cannot clear unrelated scan diagnostics') + results.setdefault('error_class', 'docker_recovery_incomplete') + results.setdefault('scan_meta', {})['docker_full_recovery'] = { + 'coverage_complete': False, 'blob_transfer_attempted': False, + 'diagnostic': {'phase': 'native_diagnostics', 'reason': 'unrelated_diagnostics'}, + } + results = apply_finding_filters(results, image_ref, log_target=log_target) + if time.monotonic() >= deadline: + had_errors = bool(results.get('errors')) + results.setdefault('errors', []).append('Docker scan exceeded its absolute deadline including cleanup and filtering') + if not had_errors: + results.update(error_class='timeout', retryable=True) + metadata = results.setdefault('scan_meta', {}) + metadata['docker_deadline_exceeded'] = True + if 'docker_full_recovery' in metadata: + metadata['docker_full_recovery']['coverage_complete'] = False + metadata['docker_full_recovery'].setdefault('diagnostic', {'phase': 'completion', 'reason': 'timeout'}) + return results + + +def _append_windows_git_longpaths(command_env, operation): + if os.name != 'nt': + return + config_count = command_env.get('GIT_CONFIG_COUNT') or '0' + if not re.fullmatch(r'[0-9]{1,3}', config_count) or int(config_count) > 255: + raise ValueError(f'{operation} GIT_CONFIG_COUNT must be an integer from 0 to 255') + config_count = int(config_count) + command_env[f'GIT_CONFIG_KEY_{config_count}'] = 'core.longpaths' + command_env[f'GIT_CONFIG_VALUE_{config_count}'] = 'true' + command_env['GIT_CONFIG_COUNT'] = str(config_count + 1) + + +def scan_huggingface_space(space_id, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None): + anonymous_public_client = ( + _client_remote_execution_kind.get() == 'huggingface_space_v1' + ) + if anonymous_public_client and token: + raise RuntimeError('remote HuggingFace direct execution cannot use a token') + logger.info(f"Scanning HuggingFace Space: {space_id}") + + cmd = [get_trufflehog_cmd(), 'huggingface', '--space', space_id, '--json', '--no-update'] + secrets_to_redact = [] + command_env = os.environ.copy() + _append_windows_git_longpaths(command_env, 'HuggingFace') + if token: + command_env['HUGGINGFACE_TOKEN'] = token + command_env['HF_TOKEN'] = token + secrets_to_redact.append(token) + append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) + + results = {"findings": [], "errors": []} + emit_client_scan_phase('scanning', { + 'integrated_operation': 'huggingface_clone_and_scan', + }) + with run_command_streamed(cmd, timeout_sec, command_env) as output: + apply_trufflehog_diagnostics( + results, output, output.returncode, 'huggingface', + redactions=secrets_to_redact, + ) + append_trufflehog_findings( + results, output.stdout_lines(redactions=secrets_to_redact), + ) + + return apply_finding_filters(results, space_id) + +WINDOWS_RESERVED_NAMES = { + 'con', 'prn', 'aux', 'nul', + *(f'com{index}' for index in range(1, 10)), + *(f'lpt{index}' for index in range(1, 10)), +} + + +def validate_archive_member_name(name): + normalized = str(name or '').replace('\\', '/') + if normalized.startswith('/') or '..' in normalized.split('/'): + raise ValueError(f'Unsafe archive member path: {name}') + for component in [part for part in normalized.split('/') if part]: + stem = component.split('.', 1)[0].lower() + if ':' in component or component.endswith((' ', '.')) or stem in WINDOWS_RESERVED_NAMES: + raise ValueError(f'Unsafe Windows archive member name: {name}') + return normalized + + +class LimitedReader: + def __init__(self, source, limit): + self.source = source + self.remaining = max(0, int(limit)) + + def read(self, size=-1): + if self.remaining <= 0: + raise ValueError('Archive decompressed stream exceeds safety budget') + if size is None or size < 0: + size = self.remaining + 1 + data = self.source.read(min(size, self.remaining + 1)) + self.remaining -= len(data) + if self.remaining < 0: + raise ValueError('Archive decompressed stream exceeds safety budget') + return data + + +def safe_extract_tar(tar_path, destination, max_files=100000, max_total_size_mb=1024, max_file_size_mb=512): + _raise_if_scan_slot_fatal() + destination_abs = os.path.abspath(destination) + max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024 + max_file_bytes = int(max_file_size_mb or 0) * 1024 * 1024 + stream_budget = max_total_bytes + (64 * 1024 * 1024) + raw_source = open(tar_path, 'rb') + magic = raw_source.read(6) + raw_source.seek(0) + if magic.startswith(b'\x1f\x8b'): + decompressed = gzip.GzipFile(fileobj=raw_source) + elif magic.startswith(b'BZh'): + decompressed = bz2.BZ2File(raw_source) + elif magic.startswith(b'\xfd7zXZ\x00'): + decompressed = lzma.LZMAFile(raw_source) + else: + decompressed = raw_source + try: + archive = tarfile.open(fileobj=LimitedReader(decompressed, stream_budget), mode='r|') + try: + total_size = 0 + member_count = 0 + for member in archive: + _raise_if_scan_slot_fatal() + if max_files and member_count >= int(max_files): + raise ValueError(f'Tar archive exceeds {max_files} members') + member_count += 1 + validate_archive_member_name(member.name) + member_path = os.path.abspath(os.path.join(destination, member.name)) + if not member_path.startswith(destination_abs + os.sep) and member_path != destination_abs: + raise ValueError(f"Unsafe tar member path: {member.name}") + if member.issym() or member.islnk() or member.isdev() or member.isfifo(): + raise ValueError(f"Unsafe tar member type: {member.name}") + if member.isfile(): + if max_file_bytes and member.size > max_file_bytes: + raise ValueError(f"Tar member exceeds {max_file_size_mb} MB: {member.name}") + total_size += max(0, int(member.size or 0)) + if max_total_bytes and total_size > max_total_bytes: + raise ValueError(f"Tar archive exceeds {max_total_size_mb} MB extracted") + archive.extract(member, destination, filter='data') + _raise_if_scan_slot_fatal() + finally: + archive.close() + finally: + if decompressed is not raw_source: + decompressed.close() + raw_source.close() + + +def validate_zip_central_directory(zip_path, max_files, max_metadata_size_mb=16): + _raise_if_scan_slot_fatal() + with open(zip_path, 'rb') as source: + source.seek(0, os.SEEK_END) + size = source.tell() + source.seek(max(0, size - 65557)) + tail = source.read() + offset = tail.rfind(b'PK\x05\x06') + if offset < 0 or offset + 22 > len(tail): + raise ValueError('Zip end-of-central-directory record is missing') + disk_number = int.from_bytes(tail[offset + 4:offset + 6], 'little') + central_disk = int.from_bytes(tail[offset + 6:offset + 8], 'little') + entry_count = int.from_bytes(tail[offset + 10:offset + 12], 'little') + central_size = int.from_bytes(tail[offset + 12:offset + 16], 'little') + central_offset = int.from_bytes(tail[offset + 16:offset + 20], 'little') + if disk_number or central_disk or entry_count == 0xFFFF or central_size == 0xFFFFFFFF or central_offset == 0xFFFFFFFF: + raise ValueError('Multi-disk and ZIP64 archives are not accepted') + if max_files and entry_count > int(max_files): + raise ValueError(f'Zip archive exceeds {max_files} members') + max_metadata_bytes = int(max_metadata_size_mb or 0) * 1024 * 1024 + if max_metadata_bytes and central_size > max_metadata_bytes: + raise ValueError(f'Zip central directory exceeds {max_metadata_size_mb} MB') + if central_offset < 0 or central_size < 0 or central_offset + central_size > size: + raise ValueError('Zip central directory points outside the archive') + with open(zip_path, 'rb') as source: + source.seek(central_offset) + consumed = 0 + for _ in range(entry_count): + _raise_if_scan_slot_fatal() + header = source.read(46) + if len(header) != 46 or header[:4] != b'PK\x01\x02': + raise ValueError('Invalid zip central-directory entry') + name_len = int.from_bytes(header[28:30], 'little') + extra_len = int.from_bytes(header[30:32], 'little') + comment_len = int.from_bytes(header[32:34], 'little') + variable_size = name_len + extra_len + comment_len + source.seek(variable_size, os.SEEK_CUR) + consumed += 46 + variable_size + if consumed > central_size: + raise ValueError('Zip central-directory size mismatch') + if consumed != central_size: + raise ValueError('Zip central-directory entry count mismatch') + return entry_count + + +def safe_extract_package_zip(zip_path, destination, max_files=100000, max_total_size_mb=1024, max_file_size_mb=512): + _raise_if_scan_slot_fatal() + destination_abs = os.path.abspath(destination) + max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024 + max_file_bytes = int(max_file_size_mb or 0) * 1024 * 1024 + validate_zip_central_directory(zip_path, max_files) + with zipfile.ZipFile(zip_path) as archive: + members = archive.infolist() + if max_files and len(members) > int(max_files): + raise ValueError(f'Zip archive exceeds {max_files} members') + total_size = 0 + for member in members: + _raise_if_scan_slot_fatal() + validate_archive_member_name(member.filename) + member_path = os.path.abspath(os.path.join(destination, member.filename)) + if not member_path.startswith(destination_abs + os.sep) and member_path != destination_abs: + raise ValueError(f"Unsafe zip member path: {member.filename}") + if member.is_dir(): + continue + if max_file_bytes and member.file_size > max_file_bytes: + raise ValueError(f"Zip member exceeds {max_file_size_mb} MB: {member.filename}") + total_size += max(0, int(member.file_size or 0)) + if max_total_bytes and total_size > max_total_bytes: + raise ValueError(f"Zip archive exceeds {max_total_size_mb} MB extracted") + for member in members: + _raise_if_scan_slot_fatal() + archive.extract(member, destination) + _raise_if_scan_slot_fatal() + +def safe_extract_archive(archive_path, destination, max_total_size_mb=1024): + _raise_if_scan_slot_fatal() + if tarfile.is_tarfile(archive_path): + safe_extract_tar(archive_path, destination, max_total_size_mb=max_total_size_mb) + return + if zipfile.is_zipfile(archive_path): + safe_extract_package_zip(archive_path, destination, max_total_size_mb=max_total_size_mb) + return + raise ValueError("Unsupported package archive format") + +def download_file(url, path, max_size_mb=50, timeout=60): + _raise_if_scan_slot_fatal() + max_bytes = max_size_mb * 1024 * 1024 + with requests.Session() as session: + session.trust_env = False + with session.get(url, stream=True, headers={'User-Agent': 'GitSecretsScanner/2.0'}, timeout=timeout) as response: + _raise_if_scan_slot_fatal() + response.raise_for_status() + total = 0 + with open(path, 'wb') as f: + for chunk in response.iter_content(chunk_size=1024 * 256): + _raise_if_scan_slot_fatal() + if not chunk: + continue + total += len(chunk) + if max_bytes and total > max_bytes: + raise ValueError(f"Artifact exceeds {max_size_mb} MB") + f.write(chunk) + _raise_if_scan_slot_fatal() + harden_private_file(path) + return total + +def scan_npm_package(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=50): + """Download, extract, and scan an npm package tarball with TruffleHog filesystem.""" + package = parse_npm_target(target) + package_id = npm_package_id(package) + logger.info(f"Scanning npm package: {package_id}") + + _raise_if_scan_slot_fatal() + work_dir = create_command_work_dir() + _raise_if_scan_slot_fatal() + if not work_dir: + return {"findings": [], "errors": ["Unable to create npm work dir"]} + + try: + tarball_path = os.path.join(work_dir, 'package.tgz') + extract_dir = os.path.join(work_dir, 'extract') + ensure_private_directory(extract_dir, reject_reparse=True) + downloaded = download_file(package['tarball'], tarball_path, max_artifact_size_mb, timeout=min(timeout_sec, 120)) + _raise_if_scan_slot_fatal() + safe_extract_tar(tarball_path, extract_dir, max_total_size_mb=max(256, int(max_artifact_size_mb or 0) * 4)) + _raise_if_scan_slot_fatal() + harden_private_tree(extract_dir) + _raise_if_scan_slot_fatal() + harvest_warnings = [] + postman_targets = find_postman_artifacts( + extract_dir, + 'npm', + package, + max_artifact_size_mb=max_artifact_size_mb, + warnings=harvest_warnings, + ) + + cmd = [get_trufflehog_cmd(), 'filesystem', extract_dir, '--json', '--no-update'] + append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) + results = {"findings": [], "errors": [], "package": package, "bytes": downloaded, "postman_targets": postman_targets} + with run_command_streamed(cmd, timeout_sec) as output: + apply_trufflehog_diagnostics(results, output, output.returncode, 'npm') + append_trufflehog_findings(results, output.stdout_lines()) + attach_nearby_context(results) + _attach_postman_harvest_warnings(results, harvest_warnings) + return apply_finding_filters(results, package_id) + except ScanSlotFatalError: + raise + except Exception as e: + return {"findings": [], "errors": [f"npm scan failed: {str(e)}"], "package": package} + finally: + cleanup_command_work_dir(work_dir) + +def scan_pypi_package(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=50): + """Download, extract, and scan a PyPI sdist/wheel with TruffleHog filesystem.""" + package = parse_pypi_target(target) + package_id = pypi_package_id(package) + logger.info(f"Scanning PyPI package: {package_id}") + + declared_size = int(package.get('size') or 0) + max_bytes = int(max_artifact_size_mb or 0) * 1024 * 1024 + if max_bytes and declared_size > max_bytes: + return { + "findings": [], + "errors": [], + "skipped": f"artifact exceeds {max_artifact_size_mb} MB", + "package": package, + } + + _raise_if_scan_slot_fatal() + work_dir = create_command_work_dir() + _raise_if_scan_slot_fatal() + if not work_dir: + return {"findings": [], "errors": ["Unable to create PyPI work dir"]} + + try: + artifact_path = os.path.join(work_dir, 'package-artifact') + extract_dir = os.path.join(work_dir, 'extract') + ensure_private_directory(extract_dir, reject_reparse=True) + downloaded = download_file(package['artifact'], artifact_path, max_artifact_size_mb, timeout=min(timeout_sec, 120)) + _raise_if_scan_slot_fatal() + safe_extract_archive(artifact_path, extract_dir, max_total_size_mb=max(256, int(max_artifact_size_mb or 0) * 4)) + _raise_if_scan_slot_fatal() + harden_private_tree(extract_dir) + _raise_if_scan_slot_fatal() + harvest_warnings = [] + postman_targets = find_postman_artifacts( + extract_dir, + 'pypi', + package, + max_artifact_size_mb=max_artifact_size_mb, + warnings=harvest_warnings, + ) + + cmd = [get_trufflehog_cmd(), 'filesystem', extract_dir, '--json', '--no-update'] + append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) + results = {"findings": [], "errors": [], "package": package, "bytes": downloaded, "postman_targets": postman_targets} + with run_command_streamed(cmd, timeout_sec) as output: + apply_trufflehog_diagnostics(results, output, output.returncode, 'pypi') + append_trufflehog_findings(results, output.stdout_lines()) + attach_nearby_context(results) + _attach_postman_harvest_warnings(results, harvest_warnings) + return apply_finding_filters(results, package_id) + except ScanSlotFatalError: + raise + except Exception as e: + return {"findings": [], "errors": [f"PyPI scan failed: {str(e)}"], "package": package} + finally: + cleanup_command_work_dir(work_dir) + +def scan_package_git_repo(target, timeout_sec=900, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, max_depth=None, max_commit_age_days=None, commit_lookup_pages=3, skip_if_commit_lookup_fails=True): + data = parse_package_git_target(target) + repo_url = data.get('repo_url') + provider = data.get('provider') + if not repo_url: + return {"findings": [], "errors": ["package_git target missing repo_url"], "package": data} + logger.info( + f"Scanning package git repo: {repo_url} " + f"({data.get('package_source')}:{data.get('name')}@{data.get('version')})" + ) + candidate = normalize_git_repo_candidate(repo_url) or {} + canonical_provider = candidate.get('provider') or provider + provider_token = token if canonical_provider == 'github' else None + preflight = preflight_package_git_repo(data, provider_token, min(int(timeout_sec or 30), 20)) + if preflight and preflight.get('skip'): + reason = preflight.get('reason') or 'package_git repo unavailable' + logger.info(f"Skipping package git repo {repo_url}: {reason}") + return {"findings": [], "errors": [], "skipped": reason, "package": data, "scan_meta": {"preflight": preflight}} + scan_provider = (preflight or {}).get('provider') or canonical_provider + scan_token = None if preflight and preflight.get('retry_unauthenticated') else provider_token + result = scan_git_repo( + repo_url, + timeout_sec=timeout_sec, + detectors=detectors, + exclude_detectors=exclude_detectors, + no_verification=no_verification, + trufflehog_config=trufflehog_config, + token=scan_token, + provider=scan_provider, + max_depth=max_depth, + max_commit_age_days=max_commit_age_days, + commit_lookup_pages=commit_lookup_pages, + skip_if_commit_lookup_fails=skip_if_commit_lookup_fails, + ) + convert_package_git_unavailable_to_skip(result) + result['package'] = data + return result + + +def preflight_package_git_repo(data, token=None, timeout=15): + repo_url = data.get('repo_url') or '' + provider = (data.get('provider') or '').lower() + candidate = normalize_git_repo_candidate(repo_url) + if not candidate: + return {'skip': True, 'reason': 'package_git repo URL is unsupported or invalid'} + provider = candidate.get('provider') or provider + repo_path = candidate.get('repo_path') or '' + try: + if provider == 'github': + response = api_request( + 'GET', + f'https://api.github.com/repos/{repo_path}', + headers=github_headers(token), + timeout=timeout, + retry_statuses={500, 502, 503, 504}, + ) + if response.status_code == 200: + return {'skip': False, 'provider': provider, 'repo_path': repo_path} + message = response_message(response).lower() + if response.status_code == 404: + return {'skip': True, 'provider': provider, 'repo_path': repo_path, 'reason': 'package_git repo not found or private'} + if response.status_code in (401, 403) and 'rate limit' not in message: + return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code, 'retry_unauthenticated': True} + return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code} + if provider == 'gitlab': + response = api_request( + 'GET', + f'https://gitlab.com/api/v4/projects/{quote(repo_path, safe="")}', + headers=gitlab_headers(token), + timeout=timeout, + retry_statuses={500, 502, 503, 504}, + ) + if response.status_code == 200: + return {'skip': False, 'provider': provider, 'repo_path': repo_path} + if response.status_code == 404: + return {'skip': True, 'provider': provider, 'repo_path': repo_path, 'reason': 'package_git project not found or private'} + if response.status_code in (401, 403): + return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code, 'retry_unauthenticated': True} + return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': response.status_code} + except ScanSlotFatalError: + raise + except Exception as e: + logger.warning(f"Package git preflight failed for {repo_url}: {str(e)[:300]}") + return {'skip': False, 'provider': provider, 'repo_path': repo_path, 'preflight_status': 'unknown'} + + +def postman_stage_filename(target_data): + kind = target_data.get('kind') or 'artifact' + digest = target_data.get('sha256') or target_data.get('sha') or 'postman' + suffix = POSTMAN_COLLECTION_SUFFIX if kind == 'collection' else POSTMAN_ENVIRONMENT_SUFFIX if kind == 'environment' else 'postman.json' + return f'{str(digest)[:16]}.{suffix}' + + +def is_postman_placeholder(value): + text = str(value or '').strip() + return bool(text and POSTMAN_PLACEHOLDER_RE.match(text)) + + +def load_postman_context(cache_path, payload=None): + max_input_bytes = min( + POSTMAN_JSON_HARD_MAX_INPUT_BYTES, + max(1, int(getattr(scan_config, 'postman_context_max_input_bytes', POSTMAN_JSON_HARD_MAX_INPUT_BYTES))), + ) + max_nodes = max(1, int(getattr(scan_config, 'postman_context_max_nodes', 100000))) + max_depth = max(1, int(getattr(scan_config, 'postman_context_max_depth', 64))) + max_scalar_bytes = max(1, int(getattr(scan_config, 'postman_context_max_scalar_bytes', 16 * 1024 * 1024))) + max_items = max(1, int(getattr(scan_config, 'postman_context_max_items', 50000))) + try: + if payload is None: + size = os.path.getsize(cache_path) + if size <= 0 or size > max_input_bytes: + raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit') + with open(cache_path, 'rb') as f: + payload = f.read(max_input_bytes + 1) + elif not isinstance(payload, bytes): + raise PostmanCacheValidationError('Postman context input must be bytes') + size = len(payload) + if size <= 0 or size > max_input_bytes: + raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit') + if len(payload) > max_input_bytes: + raise PostmanCacheValidationError('Postman context input exceeds its bounded byte limit') + data = json.loads(payload.decode('utf-8-sig')) + except PostmanCacheValidationError: + raise + except (OSError, UnicodeDecodeError, ValueError, RecursionError) as exc: + raise PostmanCacheValidationError('Postman context is not bounded valid JSON') from exc + contexts = [] + traversed = 0 + scalar_bytes = 0 + + def charge(*values): + nonlocal scalar_bytes + scalar_bytes += sum(len(str(value or '').encode('utf-8', errors='replace')) for value in values) + if scalar_bytes > max_scalar_bytes: + raise PostmanCacheValidationError('Postman context scalar byte limit exceeded') + + def host_from_url(value): + try: + return urlsplit(str(value)).hostname or '' + except Exception: + return '' + + def add_context(path, key, value, endpoint='', auth_type='', location='value'): + if value is None: + return + endpoint = str(endpoint or '') + text = str(value) + charge(path, key, text, endpoint, auth_type, location) + if len(contexts) >= max_items: + raise PostmanCacheValidationError('Postman context item limit exceeded') + contexts.append({ + 'path': path, + 'key': str(key or ''), + 'value': text, + 'endpoint': endpoint, + 'host': host_from_url(endpoint), + 'auth_type': str(auth_type or ''), + 'location': location, + }) + + stack = [(data, '$', '', '', 'value', 0)] + while stack: + value, path, endpoint, auth_type, inherited_location, depth = stack.pop() + traversed += 1 + if traversed > max_nodes: + raise PostmanCacheValidationError('Postman context traversal item limit exceeded') + if depth > max_depth: + raise PostmanCacheValidationError('Postman context depth limit exceeded') + if len(path.encode('utf-8', errors='replace')) > 4096: + raise PostmanCacheValidationError('Postman context path limit exceeded') + if isinstance(value, dict): + local_endpoint = endpoint + url_value = value.get('url') + if isinstance(url_value, str): + local_endpoint = url_value + elif isinstance(url_value, dict) and url_value.get('raw'): + local_endpoint = str(url_value.get('raw')) + local_auth = auth_type + auth = value.get('auth') + if isinstance(auth, dict): + local_auth = str(auth.get('type') or local_auth or '') + if traversed + len(stack) + len(value) > max_nodes: + raise PostmanCacheValidationError('Postman context traversal item limit exceeded') + children = [] + for key, item in value.items(): + lower_key = str(key).lower() + location = 'value' + if lower_key in ('header', 'headers'): + location = 'header' + elif lower_key in ('query', 'queryparam', 'query_params'): + location = 'query_param' + elif lower_key in ('body', 'raw'): + location = 'body' + elif lower_key in ('variable', 'values'): + location = 'environment_variable' + child_path = f'{path}.{key}' + charge(key) + children.append((item, child_path, local_endpoint, local_auth, location, depth + 1)) + stack.extend(reversed(children)) + elif isinstance(value, list): + if traversed + len(stack) + len(value) > max_nodes: + raise PostmanCacheValidationError('Postman context traversal item limit exceeded') + stack.extend( + (value[index], f'{path}[{index}]', endpoint, auth_type, inherited_location, depth + 1) + for index in range(len(value) - 1, -1, -1) + ) + else: + add_context(path, '', value, endpoint, auth_type, inherited_location) + return contexts + + +def provider_from_postman(detector_name='', host='', value=''): + detector = str(detector_name or '').lower() + if detector in ('openai', 'anthropic', 'github', 'gitlab', 'stripe', 'slack'): + return detector + text = ' '.join([str(host or '').lower(), str(value or '').lower()]) + if 'api.openai.com' in text or 'sk-proj-' in text or re.search(r'\bsk-[A-Za-z0-9]{20,}', str(value or '')): + return 'openai' + if 'anthropic.com' in text or 'sk-ant-' in text: + return 'anthropic' + if 'generativelanguage.googleapis.com' in text or 'aiplatform.googleapis.com' in text: + return 'google' + if 'huggingface.co' in text or str(value or '').startswith('hf_'): + return 'huggingface' + if 'github.com' in text or str(value or '').startswith(GITHUB_TOKEN_PREFIXES): + return 'github' + if 'gitlab' in text or str(value or '').startswith(GITLAB_TOKEN_PREFIXES): + return 'gitlab' + if 'stripe.com' in text or str(value or '').startswith(('sk_live_', 'rk_live_')): + return 'stripe' + return '' + + +def credential_kind_from_postman(context, value): + key = str(context.get('key') or '').lower() + auth_type = str(context.get('auth_type') or '').lower() + location = str(context.get('location') or '').lower() + text = str(value or '').strip() + if is_postman_placeholder(text): + return 'placeholder' + if auth_type == 'bearer' or key == 'authorization' or text.lower().startswith('bearer '): + return 'jwt' if re.match(r'^(?:bearer\s+)?eyJ[A-Za-z0-9_-]+\.', text, re.IGNORECASE) else 'bearer_token' + if ('api' in key and 'key' in key) or key in ('x-api-key', 'apikey'): + return 'api_key' + if 'client_secret' in key or 'client-secret' in key: + return 'oauth_client_secret' + if 'password' in key: + return 'basic_auth_password' if auth_type == 'basic' else 'password' + if location == 'query_param' and ('token' in key or 'key' in key): + return 'api_key' + if re.match(r'^eyJ[A-Za-z0-9_-]+\.', text): + return 'jwt' + return 'unknown' + + +POSTMAN_GEMINI_KEY_RE = re.compile(r'(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})') +POSTMAN_AZURE_OPENAI_KEY_RE = re.compile(r'\b[a-f0-9]{32}\b', re.IGNORECASE) +POSTMAN_AZURE_OPENAI_ENDPOINT_RE = re.compile(r'([a-z0-9-]+\.openai\.azure\.com)', re.IGNORECASE) +FOUNDRY_ENDPOINT_HOST_RE = r'[a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models\.ai\.azure\.com|services\.ai\.azure\.com|inference\.ai\.azure\.com)' +POSTMAN_FOUNDRY_ENDPOINT_RE = re.compile(r'((?:https?://)?' + FOUNDRY_ENDPOINT_HOST_RE + r'(?:/[^\s:"\'<>\\]*)?)', re.IGNORECASE) +FOUNDRY_ASSIGNMENT_RE = re.compile(r'''(?ix) + (?:authorization|api[_-]?key|key|token|secret|credential|bearer) + [^\n:=]{0,80} + [:=] + \s*["']?(?:bearer\s+)? + ([A-Za-z0-9_./+=\-]{20,512}) +''') +NON_FOUNDRY_KEY_PREFIXES = ( + 'sk-', 'sk_', 'sk-or-', 'xai-', 'ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_', + 'glpat-', 'glrt-', 'hf_', 'AIza', 'AQ.', 'zai-', 'gsk_', 'r8_', 'nvapi-', +) + + +def keycheck_output_path(service, filename): + root = getattr(scan_config, 'keycheck_dir', None) or os.path.join(get_results_dir() or os.getcwd(), 'keychecks') + directory = os.path.join(root, service) + ensure_private_directory(directory, reject_reparse=True) + return os.path.join(directory, filename) + + +def _candidate_limits(): + return { + 'artifact_items': max(1, int(getattr(scan_config, 'keycheck_candidate_artifact_max_items', 2000))), + 'artifact_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_artifact_max_bytes', 2 * 1024 * 1024))), + 'file_items': max(1, int(getattr(scan_config, 'keycheck_candidate_file_max_items', 100000))), + 'file_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_file_max_bytes', 32 * 1024 * 1024))), + 'line_bytes': max(1, int(getattr(scan_config, 'keycheck_candidate_line_max_bytes', 8192))), + } + + +class CandidateQueueCapacityError(RuntimeError): + pass + + +_CANDIDATE_CHECKED_LEDGERS = { + 'gem.txt': 'geminiChecked.txt', + 'azureOpenAI.txt': 'azureChecked.txt', + 'azureFoundry.txt': 'azureChecked.txt', +} + + +def _candidate_identity(value): + return str(value or '').strip().split('\t', 1)[0].strip() + + +def _read_bounded_candidate_rows(path, limits, description): + if not os.path.exists(path): + return [], set(), 0, 0 + reject_reparse_components(path) + if os.path.getsize(path) > limits['file_bytes']: + raise RuntimeError(f'{description} exceeds its aggregate byte bound: {path}') + rows = [] + identities = set() + total_bytes = 0 + with open(path, 'rb') as handle: + for raw_line in handle: + total_bytes += len(raw_line) + if total_bytes > limits['file_bytes']: + raise RuntimeError(f'{description} exceeds its aggregate byte bound: {path}') + if len(raw_line) > limits['line_bytes']: + raise RuntimeError(f'{description} line exceeds its byte bound: {path}') + text = raw_line.rstrip(b'\r\n').decode('utf-8', errors='strict') + if not text: + continue + rows.append(text) + if len(rows) > limits['file_items']: + raise RuntimeError(f'{description} exceeds its aggregate item bound: {path}') + identity = _candidate_identity(text) + if identity: + identities.add(identity) + return rows, identities, len(rows), total_bytes + + +def _candidate_additions(lines, existing_identities, limits, checked_identities=()): + seen = set(existing_identities) + seen.update(checked_identities) + additions = [] + added_bytes = 0 + for value in lines: + line = str(value or '').strip() + if not line or '\n' in line or '\r' in line: + continue + encoded = (line + '\n').encode('utf-8') + identity = _candidate_identity(line) + if not identity or len(encoded) > limits['line_bytes'] or identity in seen: + continue + additions.append(encoded) + added_bytes += len(encoded) + seen.add(identity) + return additions, added_bytes + + +def _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits): + return ( + existing_items + len(additions) <= limits['file_items'] + and existing_bytes + added_bytes <= limits['file_bytes'] + ) + + +def _rewrite_candidate_rows_atomic(path, rows): + temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp' + descriptor = None + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + try: + descriptor = os.open(temporary, flags, 0o600) + os.close(descriptor) + descriptor = None + harden_private_file(temporary) + with open(temporary, 'wb') as handle: + for row in rows: + handle.write((row + '\n').encode('utf-8')) + handle.flush() + os.fsync(handle.fileno()) + harden_private_file(temporary) + durable_replace(temporary, path) + harden_private_file(path) + finally: + if descriptor is not None: + os.close(descriptor) + try: + if os.path.exists(temporary): + os.remove(temporary) + except OSError: + pass + + +def _compact_checked_candidate_rows_unlocked(path, rows, existing_bytes, limits): + ledger_name = _CANDIDATE_CHECKED_LEDGERS.get(os.path.basename(path)) + if not ledger_name: + return rows, set(), existing_bytes + checked_path = os.path.join(os.path.dirname(path), ledger_name) + checked_lock_path = f'{checked_path}.lock' + # Candidate locks are always outermost; checked-ledger writers never take them. + checked_lock = acquire_file_lock(checked_lock_path, stale_sec=120, timeout_sec=10) + try: + _, checked_identities, _, _ = _read_bounded_candidate_rows( + checked_path, limits, 'keycheck checked ledger', + ) + finally: + release_file_lock(checked_lock, checked_lock_path) + retained = [row for row in rows if _candidate_identity(row) not in checked_identities] + if len(retained) != len(rows): + encoded_sizes = [len((row + '\n').encode('utf-8')) for row in retained] + retained_bytes = sum(encoded_sizes) + if retained_bytes > limits['file_bytes'] or any( + size > limits['line_bytes'] for size in encoded_sizes + ): + raise RuntimeError(f'compacted keycheck candidate file would exceed its bounds: {path}') + _rewrite_candidate_rows_atomic(path, retained) + else: + retained_bytes = existing_bytes + return retained, checked_identities, retained_bytes + + +def _append_unique_lines_unlocked(path, lines, limits): + ensure_private_directory(os.path.dirname(path), reject_reparse=True) + rows, existing, existing_items, existing_bytes = _read_bounded_candidate_rows( + path, limits, 'keycheck candidate file', + ) + additions, added_bytes = _candidate_additions(lines, existing, limits) + if not additions: + return 0 + if not _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits): + rows, checked, existing_bytes = _compact_checked_candidate_rows_unlocked( + path, rows, existing_bytes, limits, + ) + existing = {_candidate_identity(row) for row in rows if _candidate_identity(row)} + existing_items = len(rows) + additions, added_bytes = _candidate_additions(lines, existing, limits, checked) + if not additions: + return 0 + if not _candidate_batch_fits(existing_items, existing_bytes, additions, added_bytes, limits): + raise CandidateQueueCapacityError( + f'keycheck candidate queue has insufficient capacity for the complete offered batch: {path}' + ) + with open(path, 'ab') as f: + f.write(b''.join(additions)) + f.flush() + os.fsync(f.fileno()) + harden_private_file(path) + return len(additions) + + +def append_unique_lines_locked(path, lines): + limits = _candidate_limits() + lock_path = f'{path}.lock' + lock = acquire_file_lock(lock_path, stale_sec=120, timeout_sec=10) + try: + return _append_unique_lines_unlocked(path, lines, limits) + finally: + release_file_lock(lock, lock_path) + + +def append_unique_line(path, line): + return bool(append_unique_lines_locked(path, [line])) + + +def append_unique_line_locked(path, line): + return bool(append_unique_lines_locked(path, [line])) + + +def _collect_candidate_line(batches, seen, budget, path, line): + limits = budget['limits'] + text = str(line or '').strip() + encoded_size = len((text + '\n').encode('utf-8')) if text else 0 + if not text or encoded_size > limits['line_bytes'] or text in seen.setdefault(path, set()): + return False + if budget['items'] >= limits['artifact_items'] or budget['bytes'] + encoded_size > limits['artifact_bytes']: + budget['truncated'] = True + return False + seen[path].add(text) + batches.setdefault(path, []).append(text) + budget['items'] += 1 + budget['bytes'] += encoded_size + return True + + +def normalize_foundry_endpoint(value): + text = str(value or '').strip().strip('"\'`,;') + if not text: + return '' + split_text = text if re.match(r'(?i)^https?://', text) else 'https://' + text + try: + parsed = urlsplit(split_text) + host = parsed.netloc or parsed.path.split('/', 1)[0] + path = parsed.path if parsed.netloc else ('/' + parsed.path.split('/', 1)[1] if '/' in parsed.path else '') + except Exception: + host, path = re.sub(r'(?i)^https?://', '', text).split('/', 1)[0], '' + path = path.rstrip('.,;:)]}/') + terminal_routes = ( + ('/models/chat/completions', ''), + ('/openai/v1/chat/completions', '/openai/v1'), + ('/v1/chat/completions', '/v1'), + ('/chat/completions', ''), + ('/v1/models', '/v1'), + ('/models', ''), + ) + lower_path = path.lower() + for suffix, replacement in terminal_routes: + if lower_path.endswith(suffix): + path = path[:-len(suffix)] + replacement + break + return (host + path).strip('/').lower() + + +def dedupe_foundry_endpoints(endpoints): + normalized = [] + for endpoint in endpoints or []: + endpoint = normalize_foundry_endpoint(endpoint) + if endpoint and endpoint not in normalized: + normalized.append(endpoint) + kept = [] + for endpoint in sorted(normalized, key=len, reverse=True): + if any(other.startswith(endpoint + '/') for other in kept): + continue + kept.append(endpoint) + return list(reversed(kept)) + + +def foundry_keyish(value): + text = re.sub(r'(?i)^bearer\s+', '', str(value or '').strip().strip('"\'`,;')).strip() + lower = text.lower() + if not (20 <= len(text) <= 512): + return False + if any(marker in lower for marker in ('http://', 'https://', '{{', '${', '<', 'azure.com')): + return False + if any(ch.isspace() for ch in text): + return False + if re.match(r'(?i)^(?:authorization|api[_-]?key|key|token|secret|credential|bearer)\s*[:=]', text): + return False + if text.startswith(NON_FOUNDRY_KEY_PREFIXES): + return False + return bool(re.search(r'[A-Za-z]', text) and re.search(r'[0-9]', text)) + + +def is_foundry_detector(finding): + detector = str((finding or {}).get('DetectorName') or '').lower() + extra = (finding or {}).get('ExtraData') if isinstance((finding or {}).get('ExtraData'), dict) else {} + name = str(extra.get('name') or '').lower() + return detector.startswith('azurefoundry') or (detector == 'customregex' and name.startswith('azurefoundry')) + + +def foundry_finding_text(finding): + parts = [] + for value in finding_raw_values(finding): + parts.append(value) + if is_foundry_detector(finding): + return '\n'.join(part for part in parts if part) + context = finding.get('ScannerContext') if isinstance(finding, dict) else None + if isinstance(context, dict): + parts.extend(str(context.get(item) or '') for item in ('nearby', 'file')) + postman_context = finding.get('PostmanContext') if isinstance(finding, dict) else None + if isinstance(postman_context, dict): + parts.extend(str(postman_context.get(item) or '') for item in ('endpoint', 'host', 'variable_name')) + extra = finding.get('ExtraData') if isinstance(finding, dict) else None + if isinstance(extra, dict): + parts.extend(str(value) for value in extra.values() if isinstance(value, str)) + return '\n'.join(part for part in parts if part) + + +def foundry_candidate_keys(finding, text): + keys = [] + if is_foundry_detector(finding): + for value in finding_raw_values(finding): + if foundry_keyish(value) and value not in keys: + keys.append(value.strip().strip('"\'`,;')) + for match in FOUNDRY_ASSIGNMENT_RE.findall(text or ''): + candidate = re.sub(r'(?i)^bearer\s+', '', str(match or '').strip().strip('"\'`,;')).strip() + if foundry_keyish(candidate) and candidate not in keys: + keys.append(candidate) + return keys[:5] + + +def foundry_candidate_origin(result, finding): + file_path, line_number = finding_source_location(finding) + location = f'{file_path}:{line_number}' if file_path and line_number else file_path or '' + target = result.get('target') or '' + scan_type = result.get('scan_type') or '' + return ':'.join(part for part in (scan_type, str(target), location) if part) + + +def write_foundry_keycheck_candidates_from_findings(result): + findings = result.get('findings') or [] + if not findings: + return 0 + azure_foundry_path = keycheck_output_path('azure', 'azureFoundry.txt') + batches = {} + seen = {} + budget = {'items': 0, 'bytes': 0, 'truncated': False, 'limits': _candidate_limits()} + for finding in findings: + if not is_foundry_detector(finding): + continue + text = foundry_finding_text(finding) + endpoints = [] + for match in POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text): + endpoint = normalize_foundry_endpoint(match) + if endpoint and endpoint not in endpoints: + endpoints.append(endpoint) + if not endpoints: + continue + endpoints = dedupe_foundry_endpoints(endpoints) + keys = foundry_candidate_keys(finding, text) + if not keys: + continue + origin = foundry_candidate_origin(result, finding) + for endpoint in endpoints[:5]: + for key in keys[:5]: + metadata = json.dumps({ + 'origin': origin, + 'finding_uid': finding.get('finding_uid') or '', + }, ensure_ascii=False, separators=(',', ':')) + _collect_candidate_line( + batches, seen, budget, azure_foundry_path, + f'{endpoint}:{key}\t{metadata}', + ) + if budget['truncated']: + break + if budget['truncated']: + break + if budget['truncated']: + break + if budget['truncated']: + logger.warning('Azure Foundry candidates reached the per-artifact bound; remaining values were capped') + return append_unique_lines_locked(azure_foundry_path, batches.get(azure_foundry_path, [])) + + +def artifact_origin_label(target_data, cache_path): + origin = target_data.get('origin') if isinstance(target_data.get('origin'), dict) else {} + source = target_data.get('source') or origin.get('provider') or 'artifact' + repo = target_data.get('repo') or origin.get('repo') or '' + path = target_data.get('path') or origin.get('path') or os.path.basename(cache_path or '') + sha = target_data.get('sha') or origin.get('sha') or target_data.get('sha256') or '' + return f'{source}:{repo}:{path}:{sha}' + + +def context_values_for_pairing(contexts): + endpoints = [] + values = [] + for context in contexts or []: + value = str(context.get('value') or '').strip() + if not value or is_postman_placeholder(value): + continue + text = ' '.join([value, str(context.get('endpoint') or ''), str(context.get('host') or ''), str(context.get('key') or '')]) + values.append((context, value, text)) + for pattern in (POSTMAN_AZURE_OPENAI_ENDPOINT_RE, POSTMAN_FOUNDRY_ENDPOINT_RE): + for match in pattern.findall(text): + endpoint = normalize_foundry_endpoint(match) if pattern is POSTMAN_FOUNDRY_ENDPOINT_RE else str(match).lower().strip('/') + if endpoint not in endpoints: + endpoints.append(endpoint) + return endpoints, values + + +def write_structured_keycheck_candidates(cache_path, target_data): + contexts = load_postman_context(cache_path) + if not contexts: + return {} + context_value_by_path = {str(item.get('path') or ''): str(item.get('value') or '') for item in contexts} + endpoints, values = context_values_for_pairing(contexts) + origin = artifact_origin_label(target_data or {}, cache_path) + counts = {'gemini': 0, 'azure_openai': 0, 'azure_foundry': 0} + + gemini_path = keycheck_output_path('gemini', 'gem.txt') + azure_openai_path = keycheck_output_path('azure', 'azureOpenAI.txt') + azure_foundry_path = keycheck_output_path('azure', 'azureFoundry.txt') + batches = {} + seen = {} + budget = {'items': 0, 'bytes': 0, 'truncated': False, 'limits': _candidate_limits()} + labels = { + gemini_path: 'gemini', + azure_openai_path: 'azure_openai', + azure_foundry_path: 'azure_foundry', + } + + for context, value, text in values: + for key in POSTMAN_GEMINI_KEY_RE.findall(value): + _collect_candidate_line( + batches, seen, budget, gemini_path, + f'{key}\t{origin}\t{context.get("path") or ""}', + ) + + azure_key_match = POSTMAN_AZURE_OPENAI_KEY_RE.search(value) + if azure_key_match: + local_azure = [str(match).lower().strip('/') for match in POSTMAN_AZURE_OPENAI_ENDPOINT_RE.findall(text)] + azure_endpoints = local_azure or [endpoint for endpoint in endpoints if POSTMAN_AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint)] + for endpoint in azure_endpoints[:5]: + _collect_candidate_line( + batches, seen, budget, azure_openai_path, + f'{endpoint}:{azure_key_match.group(0)}\t{origin}\t{context.get("path") or ""}', + ) + + sibling_name = '' + path = str(context.get('path') or '') + if path.endswith('.value'): + sibling_name = context_value_by_path.get(path[:-6] + '.key', '') + key_context = ' '.join([str(context.get('key') or ''), sibling_name]).lower() + foundry_value = re.sub(r'(?i)^bearer\s+', '', value.strip().strip('"\'`,;')).strip() + if foundry_keyish(foundry_value) and any(word in key_context for word in ('authorization', 'bearer', 'key', 'token', 'secret', 'api')): + foundry_endpoints = dedupe_foundry_endpoints(POSTMAN_FOUNDRY_ENDPOINT_RE.findall(text)) + for endpoint in foundry_endpoints[:5]: + if endpoint: + _collect_candidate_line( + batches, seen, budget, azure_foundry_path, + f'{endpoint}:{foundry_value}\t{origin}\t{context.get("path") or ""}', + ) + if budget['truncated']: + break + if budget['truncated']: + logger.warning('Structured keycheck candidates reached the per-artifact bound; remaining values were capped') + for path, lines in batches.items(): + counts[labels[path]] = append_unique_lines_locked(path, lines) + return {key: value for key, value in counts.items() if value} + + +def _postman_substring_match(budget, needle, value): + if _context_budget_expired(budget): + return None + if budget['postman_comparisons'] >= budget['max_postman_comparisons']: + return None + budget['postman_comparisons'] += 1 + return needle in value + + +def attach_postman_context(results, cache_path, budget=None): + budget = budget or context_enrichment_budget() + results.pop('structured_keycheck_pending', None) + try: + payload, complete, reason = _read_context_source(cache_path, budget) + if payload is None or not complete: + if reason in ('elapsed', 'source_bytes'): + dimension = 'elapsed-time' if reason == 'elapsed' else 'source-byte' + _add_context_warning(results, 'budget', f'{dimension} budget was exhausted; findings were retained') + else: + _add_context_warning(results, 'failure', 'Postman context could not be read; findings were retained') + return results + contexts = load_postman_context(cache_path, payload=payload) + except Exception: + logger.warning('Optional Postman context parsing failed; parsed findings were retained') + _add_context_warning(results, 'failure', 'Postman JSON context parsing failed; findings were retained') + return results + + results['structured_keycheck_pending'] = True + if not contexts: + return results + + try: + placeholder_count = 0 + context_values = [] + exact_values = {} + for index, context in enumerate(contexts): + if index % 256 == 0 and _context_budget_expired(budget): + _add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; findings were retained') + return results + if not isinstance(context, dict): + continue + value = str(context.get('value') or '') + context_values.append((context, value)) + if value: + exact_values.setdefault(value, context) + placeholder_count += int(is_postman_placeholder(value)) + results['postman_context'] = { + 'context_count': len(contexts), + 'placeholder_count': placeholder_count, + } + + for finding in results.get('findings') or []: + if _context_budget_expired(budget): + _add_context_warning(results, 'budget', 'elapsed-time budget was exhausted; remaining findings were retained') + return results + if not isinstance(finding, dict): + continue + if not _context_budget_claim_finding(budget, finding): + _add_context_warning(results, 'budget', 'finding-count budget was exhausted; remaining findings were retained') + return results + raw_values = finding_raw_values(finding) + matched = next((exact_values.get(raw) for raw in raw_values if raw and exact_values.get(raw)), None) + for raw in raw_values if not matched else (): + if not raw: + continue + for context, value in context_values: + comparison = _postman_substring_match(budget, raw, value) + if comparison is None: + _add_context_warning(results, 'budget', 'Postman comparison budget was exhausted; remaining findings were retained') + return results + if comparison: + matched = context + break + if matched: + break + if not matched and raw_values and raw_values[0][:8]: + prefix = raw_values[0][:8] + for context, value in context_values: + comparison = _postman_substring_match(budget, prefix, value) + if comparison is None: + _add_context_warning(results, 'budget', 'Postman comparison budget was exhausted; remaining findings were retained') + return results + if comparison: + matched = context + break + if not matched: + continue + raw_value = raw_values[0] if raw_values else matched.get('value') + kind = credential_kind_from_postman(matched, raw_value) + confidence = 'verified' if finding.get('Verified') else 'detector_match' if finding.get('DetectorName') else 'structured_complete' if kind != 'unknown' else 'context_only' + if kind == 'placeholder': + confidence = 'placeholder' + endpoint = sanitize_endpoint(matched.get('endpoint')) + host = sanitize_endpoint_host(endpoint) or sanitize_endpoint_host(matched.get('host')) + finding['PostmanContext'] = { + 'provider': provider_from_postman(finding.get('DetectorName'), host, raw_value), + 'credential_kind': kind, + 'credential_confidence': confidence, + 'context_location': matched.get('location'), + 'variable_name': matched.get('key'), + 'endpoint': endpoint, + 'host': host, + 'auth_type': matched.get('auth_type'), + 'json_path': matched.get('path'), + 'placeholder': is_postman_placeholder(matched.get('value')), + } + except Exception: + logger.warning('Optional Postman context matching failed; parsed findings were retained') + _add_context_warning(results, 'failure', 'Postman context matching failed; findings were retained') + return results + + +def scan_postman_target(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, max_artifact_size_mb=20, token=None): + data = parse_postman_target(target) + cache_path = data.get('cache_path') or data.get('local_path') + if not cache_path: + return {"findings": [], "errors": ["Postman target missing cached artifact"], "package": data.get('origin') or data} + try: + cache_path, declared_size = validate_postman_cache_artifact(data, max_artifact_size_mb) + except PostmanCacheTooLarge as exc: + return {"findings": [], "errors": [], "skipped": str(exc), "package": data.get('origin') or data} + except (OSError, PostmanCacheValidationError) as exc: + return {"findings": [], "errors": [f"Postman cache path rejected: {exc}"], "package": data.get('origin') or data} + _raise_if_scan_slot_fatal() + work_dir = create_command_work_dir() + _raise_if_scan_slot_fatal() + if not work_dir: + return {"findings": [], "errors": ["Unable to create Postman work dir"], "package": data.get('origin') or data} + try: + staged_path = os.path.join(work_dir, postman_stage_filename(data)) + _raise_if_scan_slot_fatal() + shutil.copyfile(cache_path, staged_path) + _raise_if_scan_slot_fatal() + harden_private_file(staged_path) + cmd = [get_trufflehog_cmd(), 'filesystem', work_dir, '--json', '--no-update'] + append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) + results = { + "findings": [], "errors": [], "package": data.get('origin') or data, + "postman": data, "bytes": declared_size, + "postman_max_artifact_size_mb": int(max_artifact_size_mb or 0), + } + with run_command_streamed(cmd, timeout_sec) as output: + apply_trufflehog_diagnostics(results, output, output.returncode, 'postman') + append_trufflehog_findings(results, output.stdout_lines()) + enrichment_budget = context_enrichment_budget() + attach_nearby_context(results, enrichment_budget) + artifact_path = str(data.get('path') or data.get('name') or data.get('url') or '').split('?', 1)[0].lower() + if str(data.get('kind') or '').lower() != 'bruno' or not artifact_path.endswith('.bru'): + attach_postman_context(results, staged_path, enrichment_budget) + return apply_finding_filters(results, cache_path) + except ScanSlotFatalError: + raise + except Exception as e: + return {"findings": [], "errors": [f"Postman scan failed: {str(e)}"], "package": data.get('origin') or data} + finally: + cleanup_command_work_dir(work_dir) + + +def safe_path_component(value, max_len=120): + text = str(value or '')[:max_len] + return re.sub(r'[^A-Za-z0-9_.-]+', '_', text).strip('._') or 'item' + + +def run_filesystem_scan(root_dir, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None): + cmd = [get_trufflehog_cmd(), 'filesystem', root_dir, '--json', '--no-update'] + append_trufflehog_scan_args(cmd, detectors, exclude_detectors, no_verification, trufflehog_config) + results = {"findings": [], "errors": []} + with run_command_streamed(cmd, timeout_sec) as output: + apply_trufflehog_diagnostics(results, output, output.returncode, 'filesystem') + append_trufflehog_findings(results, output.stdout_lines()) + attach_nearby_context(results) + return apply_finding_filters(results, root_dir) + + +CI_ARTIFACT_ALLOWED_SUFFIXES = { + '.env', '.log', '.txt', '.json', '.yaml', '.yml', '.xml', '.html', '.lcov', '.sarif', + '.tfstate', '.tfplan', '.py', '.js', '.jsx', '.ts', '.tsx', '.sh', '.toml', '.ini', + '.cfg', '.conf', '.properties', '.ipynb', '.md', '.out', '.err', '.csv', '.tsv', +} +CI_ARTIFACT_ALLOWED_NAMES = {'dockerfile', 'makefile', 'procfile'} +CI_ARTIFACT_SKIP_DIRS = { + '.git', '.hg', '.svn', 'node_modules', '__pycache__', '.venv', 'venv', 'env', + '.mypy_cache', '.pytest_cache', '.tox', '.gradle', '.idea', '.vscode', +} + + +def ci_artifact_file_allowed(name, allowed_suffixes=None): + if allowed_suffixes is None: + return True + normalized = name.replace('\\', '/').strip('/') + parts = [part.lower() for part in normalized.split('/') if part] + if any(part in CI_ARTIFACT_SKIP_DIRS for part in parts[:-1]): + return False + base = parts[-1] if parts else '' + if base in CI_ARTIFACT_ALLOWED_NAMES or base.startswith('.env'): + return True + return any(base.endswith(suffix) for suffix in allowed_suffixes) + + +def safe_extract_zip(zip_path, destination, max_file_size_mb=20, max_files=1000, allowed_suffixes=None, max_total_size_mb=250): + _raise_if_scan_slot_fatal() + extracted = 0 + max_bytes = int(max_file_size_mb or 0) * 1024 * 1024 + max_total_bytes = int(max_total_size_mb or 0) * 1024 * 1024 + total_bytes = 0 + validate_zip_central_directory(zip_path, max_files) + with zipfile.ZipFile(zip_path) as archive: + for info in archive.infolist(): + _raise_if_scan_slot_fatal() + if info.is_dir(): + continue + if max_files and extracted >= max_files: + raise ValueError(f'Zip archive exceeds {max_files} extracted files') + if max_bytes and info.file_size > max_bytes: + raise ValueError(f'Zip member exceeds {max_file_size_mb} MB: {info.filename}') + if max_total_bytes and total_bytes + max(0, int(info.file_size or 0)) > max_total_bytes: + raise ValueError(f'Zip archive exceeds {max_total_size_mb} MB extracted') + name = info.filename.replace('\\', '/') + validate_archive_member_name(name) + if not ci_artifact_file_allowed(name, allowed_suffixes): + continue + target_path = os.path.abspath(os.path.normpath(os.path.join(destination, name))) + if os.path.commonpath([os.path.abspath(destination), target_path]) != os.path.abspath(destination): + continue + os.makedirs(os.path.dirname(target_path), exist_ok=True) + with archive.open(info) as src, open(target_path, 'wb') as dst: + while True: + _raise_if_scan_slot_fatal() + chunk = src.read(256 * 1024) + if not chunk: + break + dst.write(chunk) + _raise_if_scan_slot_fatal() + extracted += 1 + total_bytes += max(0, int(info.file_size or 0)) + return extracted + + +def remove_file_quiet(path): + try: + reject_reparse_components(path) + os.remove(path) + except OSError: + pass + + +@dataclass(frozen=True) +class DownloadOutcome: + path: str + status_code: int + bytes_written: int + error: str = '' + + @property + def ok(self): + return bool(self.path and not self.error) + + +def download_to_file( + url, destination, headers=None, timeout=20, max_size_mb=0, + max_size_bytes=None, byte_budget=None, +): + _raise_if_scan_slot_fatal() + destination = os.path.abspath(destination) + require_private_directory(os.path.dirname(destination), create=False) + reject_reparse_components(os.path.dirname(destination)) + current_url = str(url) + current_headers = dict(headers or {}) + configured_max = int(max_size_mb or 0) * 1024 * 1024 + explicit_max = max(0, int(max_size_bytes or 0)) + max_bytes = min(value for value in (configured_max, explicit_max) if value > 0) if configured_max and explicit_max else configured_max or explicit_max + if byte_budget is not None and int(byte_budget.get('remaining', 0)) <= 0: + return DownloadOutcome('', 0, 0, 'target byte budget exhausted') + for _ in range(6): + _raise_if_scan_slot_fatal() + try: + response = _direct_request( + 'GET', current_url, headers=current_headers, timeout=timeout, + allow_redirects=False, stream=True, + ) + except requests.RequestException as exc: + return DownloadOutcome('', 0, 0, str(exc)) + if _scan_slot_fatal_event.is_set(): + response.close() + _raise_if_scan_slot_fatal() + if response.status_code in (301, 302, 303, 307, 308) and response.headers.get('Location'): + next_url = urljoin(current_url, response.headers['Location']) + old_host = (urlsplit(current_url).hostname or '').lower() + parsed_next = urlsplit(next_url) + response.close() + if parsed_next.scheme != 'https': + return DownloadOutcome('', 0, 0, f'unsafe redirect scheme: {parsed_next.scheme}') + if (parsed_next.hostname or '').lower() != old_host: + current_headers = { + key: value for key, value in current_headers.items() + if key.lower() not in ('authorization', 'private-token', 'cookie', 'proxy-authorization') + } + current_url = next_url + continue + if response.status_code >= 400: + try: + prefix = bytearray() + for chunk in response.iter_content(chunk_size=300): + _raise_if_scan_slot_fatal() + prefix.extend(chunk[:max(0, 300 - len(prefix))]) + if len(prefix) >= 300: + break + message = bytes(prefix).decode(response.encoding or 'utf-8', errors='replace') + finally: + response.close() + return DownloadOutcome('', response.status_code, 0, message) + content_length = response.headers.get('Content-Length') + if content_length: + try: + declared = int(content_length) + remaining = int(byte_budget.get('remaining', 0)) if byte_budget is not None else 0 + if max_bytes and declared > max_bytes: + response.close() + return DownloadOutcome('', response.status_code, 0, f'response exceeds configured {max_bytes}-byte limit') + if byte_budget is not None and declared > remaining: + response.close() + return DownloadOutcome('', response.status_code, 0, 'target byte budget exhausted') + except ValueError: + pass + total = 0 + temporary = f'{destination}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.partial' + descriptor = None + try: + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) + descriptor = os.open(temporary, flags, 0o600) + os.close(descriptor) + descriptor = None + harden_private_file(temporary) + with open(temporary, 'wb', buffering=0) as output: + for chunk in response.iter_content(chunk_size=256 * 1024): + _raise_if_scan_slot_fatal() + if not chunk: + continue + if max_bytes and total + len(chunk) > max_bytes: + raise CommandOutputLimitError(f'object response exceeds configured {max_bytes}-byte limit') + if byte_budget is not None: + remaining = int(byte_budget.get('remaining', 0)) + if len(chunk) > remaining: + byte_budget['remaining'] = 0 + raise CommandOutputLimitError('target byte budget exhausted') + byte_budget['remaining'] = remaining - len(chunk) + output.write(chunk) + total += len(chunk) + output.flush() + os.fsync(output.fileno()) + harden_private_file(temporary) + durable_replace(temporary, destination) + if not private_file_ready(destination): + raise OSError('download destination lost its private file identity') + return DownloadOutcome(destination, response.status_code, total) + except (CommandOutputLimitError, OSError, requests.RequestException) as exc: + return DownloadOutcome('', response.status_code, total, str(exc)) + finally: + response.close() + if descriptor is not None: + os.close(descriptor) + if os.path.exists(temporary): + try: + os.remove(temporary) + except OSError: + pass + return DownloadOutcome('', 0, 0, 'too many redirects') + + +def github_api_get(url, token=None, timeout=20, stream=False): + response = api_request( + 'GET', url, headers=github_headers(token), timeout=timeout, stream=stream, + ) + if token and response.status_code in (401, 403): + response.close() + anonymous = api_request( + 'GET', url, headers=github_headers(None), timeout=timeout, stream=stream, + ) + if anonymous.status_code < 400: + return anonymous + response = anonymous + if response.status_code >= 400: + raise github_api_error(response) + return response + + +def bounded_response_json(response, max_bytes=8 * 1024 * 1024): + max_bytes = max(1, int(max_bytes)) + declared = response.headers.get('Content-Length') + if declared: + try: + declared = int(declared) + except ValueError as exc: + response.close() + raise ApiRequestError('API JSON response has an invalid Content-Length') from exc + if declared > max_bytes: + response.close() + raise ApiRequestError(f'API JSON response exceeds {max_bytes} bytes') + payload = bytearray() + try: + for chunk in response.iter_content(chunk_size=64 * 1024): + if not chunk: + continue + if len(payload) + len(chunk) > max_bytes: + raise ApiRequestError(f'API JSON response exceeds {max_bytes} bytes') + payload.extend(chunk) + return json.loads(bytes(payload).decode('utf-8', errors='strict')) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ApiRequestError('API response is not bounded valid UTF-8 JSON') from exc + finally: + response.close() + + +def select_ci_runs(runs, runs_per_repo=5, lookback_days=30, failed_first=True): + cutoff = datetime.now(timezone.utc) - timedelta(days=int(lookback_days or 0)) if int(lookback_days or 0) > 0 else None + kept = [] + for run in runs or []: + created = parse_postman_time(run.get('created_at')) + if cutoff and created and created < cutoff: + continue + kept.append(run) + def created_ts(item): + parsed = parse_postman_time(item.get('created_at')) + return parsed.timestamp() if parsed else 0 + if failed_first: + kept.sort(key=lambda item: (0 if item.get('conclusion') not in ('success', None) else 1, -created_ts(item))) + else: + kept.sort(key=lambda item: -created_ts(item)) + return kept[:max(1, int(runs_per_repo or 5))] + + +def scan_github_actions_repo(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, ci_runs_per_repo=5, ci_lookback_days=30, ci_max_log_archive_mb=50, ci_max_log_file_mb=20, ci_failed_first=True, ci_scan_artifacts=False, ci_max_artifacts_per_run=3, ci_max_artifact_archive_mb=50, ci_max_artifact_file_mb=10, ci_max_artifact_files=1000, ci_target_max_download_mb=500, fetch_timeout=20): + repo, repo_url = parse_github_repo_target(target) + if not repo: + return {"findings": [], "errors": ["Unable to parse GitHub repo target"]} + logger.info(f"Scanning GitHub Actions logs: {repo}") + _raise_if_scan_slot_fatal() + work_dir = create_command_work_dir() + _raise_if_scan_slot_fatal() + if not work_dir: + return {"findings": [], "errors": ["Unable to create GitHub Actions work dir"]} + try: + runs_url = f'https://api.github.com/repos/{repo}/actions/runs?per_page={max(1, min(100, int(ci_runs_per_repo or 5) * 3))}' + response = github_api_get(runs_url, token, fetch_timeout, stream=True) + payload = bounded_response_json(response) + if not isinstance(payload, dict) or 'workflow_runs' not in payload or not isinstance(payload.get('workflow_runs'), list): + raise ApiRequestError('invalid GitHub Actions runs payload') + runs = select_ci_runs(payload.get('workflow_runs') or [], ci_runs_per_repo, ci_lookback_days, ci_failed_first) + if not runs: + return {"findings": [], "errors": [], "skipped": "no recent workflow runs", "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}} + max_archive_bytes = int(ci_max_log_archive_mb or 0) * 1024 * 1024 + max_artifact_archive_bytes = int(ci_max_artifact_archive_mb or 0) * 1024 * 1024 + extracted_total = 0 + artifact_extracted_total = 0 + download_failures = [] + downloaded_total = 0 + target_download_limit = int(ci_target_max_download_mb or 0) * 1024 * 1024 + download_budget = {'remaining': target_download_limit} if target_download_limit else None + run_meta = [] + for run in runs: + _raise_if_scan_slot_fatal() + if download_budget is not None and download_budget['remaining'] <= 0: + download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap') + break + run_id = run.get('id') + if not run_id: + continue + extracted = 0 + artifacts_meta = [] + logs_url = f'https://api.github.com/repos/{repo}/actions/runs/{run_id}/logs' + zip_path = os.path.join(work_dir, f'github_actions_{safe_path_component(repo)}_{run_id}.zip') + authenticated_status = None + try: + download = download_to_file( + logs_url, zip_path, github_headers(token), fetch_timeout, ci_max_log_archive_mb, + max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, + byte_budget=download_budget, + ) + authenticated_status = download.status_code + except requests.RequestException as exc: + download = DownloadOutcome('', 0, 0, str(exc)) + if not download.ok and token and download.status_code in (401, 403): + anonymous = download_to_file( + logs_url, zip_path, github_headers(None), fetch_timeout, ci_max_log_archive_mb, + max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, + byte_budget=download_budget, + ) + if not anonymous.ok: + download = DownloadOutcome( + '', authenticated_status or anonymous.status_code, 0, + f'authenticated HTTP {authenticated_status}; anonymous HTTP ' + f'{anonymous.status_code}: {anonymous.error}', + ) + else: + download = anonymous + if download.ok: + downloaded_total += download.bytes_written + run_dir = os.path.join(work_dir, 'github_actions', safe_path_component(repo), str(run_id)) + ensure_private_directory(run_dir, reject_reparse=True) + try: + extracted = safe_extract_zip( + zip_path, run_dir, ci_max_log_file_mb, + max_total_size_mb=ci_max_log_archive_mb, + ) + harden_private_tree(run_dir) + except (zipfile.BadZipFile, ValueError) as exc: + logger.warning(f"GitHub Actions logs archive rejected for {repo} run {run_id}: {exc}") + download_failures.append(f'log archive rejected: {exc}') + extracted = 0 + remove_file_quiet(zip_path) + else: + if download.status_code not in (404, 410): + download_failures.append(f'log download HTTP {download.status_code}: {download.error}') + run_dir = os.path.join(work_dir, 'github_actions', safe_path_component(repo), str(run_id)) + ensure_private_directory(run_dir, reject_reparse=True) + if ci_scan_artifacts: + artifacts_url = f'https://api.github.com/repos/{repo}/actions/runs/{run_id}/artifacts?per_page={max(1, min(100, int(ci_max_artifacts_per_run or 3)))}' + try: + artifacts_response = github_api_get( + artifacts_url, token, fetch_timeout, stream=True, + ) + artifact_payload = bounded_response_json(artifacts_response) + if not isinstance(artifact_payload, dict) or 'artifacts' not in artifact_payload or not isinstance(artifact_payload.get('artifacts'), list): + raise ApiRequestError('invalid GitHub Actions artifacts payload') + artifacts = artifact_payload.get('artifacts') or [] + except ScanSlotFatalError: + raise + except Exception as exc: + logger.warning(f"Unable to list GitHub Actions artifacts for {repo} run {run_id}: {exc}") + download_failures.append(f'artifact listing failed: {exc}') + artifacts = [] + for artifact in artifacts[:max(0, int(ci_max_artifacts_per_run or 3))]: + _raise_if_scan_slot_fatal() + if download_budget is not None and download_budget['remaining'] <= 0: + download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap') + break + if artifact.get('expired'): + continue + size = int(artifact.get('size_in_bytes') or 0) + if max_artifact_archive_bytes and size and size > max_artifact_archive_bytes: + download_failures.append(f'artifact declared size exceeds {ci_max_artifact_archive_mb} MB') + continue + download_url = artifact.get('archive_download_url') + if not download_url: + continue + artifact_id = artifact.get('id') or safe_path_component(artifact.get('name')) + artifact_zip = os.path.join(work_dir, f'github_actions_artifact_{safe_path_component(repo)}_{run_id}_{artifact_id}.zip') + download = download_to_file( + download_url, artifact_zip, github_headers(token), fetch_timeout, ci_max_artifact_archive_mb, + max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, + byte_budget=download_budget, + ) + authenticated_status = download.status_code + if not download.ok and token and download.status_code in (401, 403): + anonymous = download_to_file( + download_url, artifact_zip, github_headers(None), fetch_timeout, ci_max_artifact_archive_mb, + max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, + byte_budget=download_budget, + ) + if not anonymous.ok: + download = DownloadOutcome( + '', authenticated_status or anonymous.status_code, 0, + f'authenticated HTTP {authenticated_status}; anonymous HTTP ' + f'{anonymous.status_code}: {anonymous.error}', + ) + else: + download = anonymous + if not download.ok: + if download.status_code not in (404, 410): + logger.warning(f"GitHub Actions artifact download HTTP {download.status_code} for {repo} run {run_id}: {download.error}") + download_failures.append(f'artifact download HTTP {download.status_code}: {download.error}') + continue + downloaded_total += download.bytes_written + artifact_dir = os.path.join(run_dir, 'artifacts', safe_path_component(artifact.get('name') or artifact_id)) + ensure_private_directory(artifact_dir, reject_reparse=True) + try: + artifact_files = safe_extract_zip( + artifact_zip, + artifact_dir, + ci_max_artifact_file_mb, + ci_max_artifact_files, + CI_ARTIFACT_ALLOWED_SUFFIXES, + max_total_size_mb=ci_max_artifact_archive_mb, + ) + harden_private_tree(artifact_dir) + except (zipfile.BadZipFile, ValueError) as exc: + logger.warning(f"GitHub Actions artifact rejected for {repo} run {run_id}: {artifact.get('name')}: {exc}") + download_failures.append(f'artifact archive rejected: {exc}') + artifact_files = 0 + remove_file_quiet(artifact_zip) + artifact_extracted_total += artifact_files + artifacts_meta.append({ + 'artifact_id': artifact.get('id'), + 'name': artifact.get('name'), + 'size_in_bytes': size, + 'expired': artifact.get('expired'), + 'extracted_files': artifact_files, + }) + extracted_total += extracted + run_meta.append({ + 'run_id': run_id, + 'run_number': run.get('run_number'), + 'workflow_name': run.get('name'), + 'status': run.get('status'), + 'conclusion': run.get('conclusion'), + 'created_at': run.get('created_at'), + 'updated_at': run.get('updated_at'), + 'extracted_files': extracted, + 'artifacts': artifacts_meta, + }) + if extracted_total == 0 and artifact_extracted_total == 0: + if download_failures: + text = '; '.join(download_failures[:5]) + all_failures = '; '.join(download_failures) + auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any( + token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit') + ) + return { + "findings": [], "errors": [text], "error_class": 'source_auth' if auth_failure else 'remote_transient', + "retryable": True, "source_failure": auth_failure, + "source_failure_category": 'auth_forbidden' if auth_failure else 'remote_transient', + "source_failure_auth_related": auth_failure, + "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta}, + } + return {"findings": [], "errors": [], "skipped": "no downloadable workflow logs or artifacts", "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta}} + _raise_if_scan_slot_fatal() + results = run_filesystem_scan(work_dir, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config) + results['package'] = {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url, "runs": run_meta, "log_files": extracted_total, "artifact_files": artifact_extracted_total} + if download_failures: + text = '; '.join(download_failures[:5]) + all_failures = '; '.join(download_failures) + auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any( + token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit') + ) + results['errors'] = list(results.get('errors') or []) + [text] + results['error_class'] = 'source_auth' if auth_failure else 'remote_transient' + results['retryable'] = True + results['source_failure'] = auth_failure + results['source_failure_category'] = 'auth_forbidden' if auth_failure else 'remote_transient' + results['source_failure_auth_related'] = auth_failure + return results + except ScanSlotFatalError: + raise + except RateLimitError as exc: + category = getattr(exc, 'category', '') + if category == 'not_found': + return {"findings": [], "errors": [], "skipped": str(exc), "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}} + return { + "findings": [], "errors": [str(exc)], "error_class": 'source_auth' if getattr(exc, 'auth_related', True) else 'remote_transient', + "retryable": True, "source_failure": True, + "source_failure_category": category or 'rate_limit', + "source_failure_auth_related": bool(getattr(exc, 'auth_related', True)), + "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}, + **({'_diagnostic_http': exc.diagnostic_http} if exc.diagnostic_http else {}), + } + except Exception as exc: + return { + "findings": [], "errors": [f"GitHub Actions acquisition failed: {exc}"], + "error_class": "remote_transient", "retryable": True, "source_failure": True, + "source_failure_category": "network", "source_failure_auth_related": False, + "package": {"ci_provider": "github_actions", "repo": repo, "repo_url": repo_url}, + } + finally: + cleanup_command_work_dir(work_dir) + + +def gitlab_api_get(url, token=None, timeout=20, stream=False): + response = api_request( + 'GET', url, headers=gitlab_headers(token), timeout=timeout, stream=stream, + ) + if token and response.status_code in (401, 403): + response.close() + anonymous = api_request( + 'GET', url, headers=gitlab_headers(None), timeout=timeout, stream=stream, + ) + if anonymous.status_code < 400: + return anonymous + response = anonymous + if response.status_code >= 400: + raise gitlab_api_error(response) + return response + + +def scan_gitlab_ci_project(target, timeout_sec=300, detectors=None, exclude_detectors=None, no_verification=False, trufflehog_config=None, token=None, ci_pipelines_per_project=5, ci_jobs_per_pipeline=20, ci_lookback_days=30, ci_max_trace_mb=20, ci_scan_artifacts=False, ci_max_artifacts_per_pipeline=5, ci_max_artifact_archive_mb=50, ci_max_artifact_file_mb=10, ci_max_artifact_files=1000, ci_target_max_download_mb=500, fetch_timeout=20): + project, project_url = parse_gitlab_project_target(target) + if not project: + return {"findings": [], "errors": ["Unable to parse GitLab project target"]} + logger.info(f"Scanning GitLab CI traces: {project}") + _raise_if_scan_slot_fatal() + work_dir = create_command_work_dir() + _raise_if_scan_slot_fatal() + if not work_dir: + return {"findings": [], "errors": ["Unable to create GitLab CI work dir"]} + try: + encoded = quote(project, safe='') + pipeline_url = f'https://gitlab.com/api/v4/projects/{encoded}/pipelines?per_page={max(1, min(100, int(ci_pipelines_per_project or 5)))}&order_by=updated_at&sort=desc' + response = gitlab_api_get(pipeline_url, token, fetch_timeout, stream=True) + pipelines = bounded_response_json(response) or [] + if not isinstance(pipelines, list): + raise ApiRequestError('invalid GitLab pipelines payload') + cutoff = datetime.now(timezone.utc) - timedelta(days=int(ci_lookback_days or 0)) if int(ci_lookback_days or 0) > 0 else None + selected = [] + for pipeline in pipelines: + updated = parse_postman_time(pipeline.get('updated_at') or pipeline.get('created_at')) + if cutoff and updated and updated < cutoff: + continue + selected.append(pipeline) + if len(selected) >= int(ci_pipelines_per_project or 5): + break + if not selected: + return {"findings": [], "errors": [], "skipped": "no recent pipelines", "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}} + trace_dir = os.path.join(work_dir, 'gitlab_ci', safe_path_component(project)) + ensure_private_directory(trace_dir, reject_reparse=True) + max_trace_bytes = int(ci_max_trace_mb or 0) * 1024 * 1024 + max_artifact_archive_bytes = int(ci_max_artifact_archive_mb or 0) * 1024 * 1024 + pipeline_meta = [] + written = 0 + artifact_extracted_total = 0 + download_failures = [] + downloaded_total = 0 + target_download_limit = int(ci_target_max_download_mb or 0) * 1024 * 1024 + download_budget = {'remaining': target_download_limit} if target_download_limit else None + for pipeline in selected: + _raise_if_scan_slot_fatal() + if download_budget is not None and download_budget['remaining'] <= 0: + download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap') + break + pipeline_id = pipeline.get('id') + if not pipeline_id: + continue + jobs_url = f'https://gitlab.com/api/v4/projects/{encoded}/pipelines/{pipeline_id}/jobs?per_page={max(1, min(100, int(ci_jobs_per_pipeline or 20)))}' + try: + jobs_response = gitlab_api_get(jobs_url, token, fetch_timeout, stream=True) + jobs = bounded_response_json(jobs_response) or [] + if not isinstance(jobs, list): + raise ApiRequestError('invalid GitLab jobs payload') + except ScanSlotFatalError: + raise + except Exception as exc: + logger.warning(f"Unable to list GitLab CI jobs for {project} pipeline {pipeline_id}: {exc}") + download_failures.append(f'job listing failed: {exc}') + jobs = [] + job_meta = [] + artifacts_seen = 0 + for job in jobs[:max(1, int(ci_jobs_per_pipeline or 20))]: + _raise_if_scan_slot_fatal() + if download_budget is not None and download_budget['remaining'] <= 0: + download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap') + break + job_id = job.get('id') + if not job_id: + continue + artifact_meta = None + trace_written = False + trace_url = f'https://gitlab.com/api/v4/projects/{encoded}/jobs/{job_id}/trace' + filename = f"pipeline_{pipeline_id}_job_{job_id}_{safe_path_component(job.get('name'))}.log" + trace_path = os.path.join(trace_dir, filename) + try: + trace_download = download_to_file( + trace_url, trace_path, gitlab_headers(token), fetch_timeout, ci_max_trace_mb, + max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, + byte_budget=download_budget, + ) + except requests.RequestException as exc: + logger.warning(f"GitLab CI trace download failed for {project} job {job_id}: {exc}") + download_failures.append(f'trace download failed: {exc}') + trace_download = DownloadOutcome('', 0, 0, str(exc)) + authenticated_status = trace_download.status_code + if not trace_download.ok and token and trace_download.status_code in (401, 403): + anonymous = download_to_file( + trace_url, trace_path, gitlab_headers(None), fetch_timeout, ci_max_trace_mb, + max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, + byte_budget=download_budget, + ) + if not anonymous.ok: + trace_download = DownloadOutcome( + '', authenticated_status or anonymous.status_code, 0, + f'authenticated HTTP {authenticated_status}; anonymous HTTP ' + f'{anonymous.status_code}: {anonymous.error}', + ) + else: + trace_download = anonymous + if trace_download.ok: + downloaded_total += trace_download.bytes_written + written += 1 + trace_written = True + elif trace_download.status_code not in (404, 410): + download_failures.append( + f'trace download HTTP {trace_download.status_code}: {trace_download.error}' + ) + if ci_scan_artifacts and artifacts_seen < int(ci_max_artifacts_per_pipeline or 5): + if download_budget is not None and download_budget['remaining'] <= 0: + download_failures.append(f'target download reached {ci_target_max_download_mb} MB cap') + job_meta.append({'job_id': job_id, 'name': job.get('name'), 'status': job.get('status'), 'stage': job.get('stage'), 'trace': trace_written, 'artifact': None}) + break + artifact_file = job.get('artifacts_file') if isinstance(job.get('artifacts_file'), dict) else {} + artifact_size = int(artifact_file.get('size') or 0) + if artifact_file and max_artifact_archive_bytes and artifact_size and artifact_size > max_artifact_archive_bytes: + download_failures.append(f'artifact declared size exceeds {ci_max_artifact_archive_mb} MB') + artifact_file = {} + if artifact_file and (not max_artifact_archive_bytes or not artifact_size or artifact_size <= max_artifact_archive_bytes): + artifacts_url = f'https://gitlab.com/api/v4/projects/{encoded}/jobs/{job_id}/artifacts' + artifact_zip = os.path.join(work_dir, f'gitlab_ci_artifact_{safe_path_component(project)}_{job_id}.zip') + download = download_to_file( + artifacts_url, artifact_zip, gitlab_headers(token), fetch_timeout, ci_max_artifact_archive_mb, + max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, + byte_budget=download_budget, + ) + authenticated_status = download.status_code + if not download.ok and token and download.status_code in (401, 403): + anonymous = download_to_file( + artifacts_url, artifact_zip, gitlab_headers(None), fetch_timeout, ci_max_artifact_archive_mb, + max_size_bytes=(target_download_limit - downloaded_total) if target_download_limit else None, + byte_budget=download_budget, + ) + if not anonymous.ok: + download = DownloadOutcome( + '', authenticated_status or anonymous.status_code, 0, + f'authenticated HTTP {authenticated_status}; anonymous HTTP ' + f'{anonymous.status_code}: {anonymous.error}', + ) + else: + download = anonymous + if download.ok: + downloaded_total += download.bytes_written + artifact_dir = os.path.join(trace_dir, 'artifacts', f'pipeline_{pipeline_id}', f'job_{job_id}') + ensure_private_directory(artifact_dir, reject_reparse=True) + try: + artifact_files = safe_extract_zip( + artifact_zip, + artifact_dir, + ci_max_artifact_file_mb, + ci_max_artifact_files, + CI_ARTIFACT_ALLOWED_SUFFIXES, + max_total_size_mb=ci_max_artifact_archive_mb, + ) + harden_private_tree(artifact_dir) + except (zipfile.BadZipFile, ValueError) as exc: + logger.warning(f"GitLab CI artifact rejected for {project} job {job_id}: {exc}") + download_failures.append(f'artifact archive rejected: {exc}') + artifact_files = 0 + remove_file_quiet(artifact_zip) + artifact_extracted_total += artifact_files + artifacts_seen += 1 + artifact_meta = { + 'filename': artifact_file.get('filename'), + 'size': artifact_size, + 'extracted_files': artifact_files, + } + else: + if download.status_code not in (404, 410): + logger.warning(f"GitLab CI artifact download HTTP {download.status_code} for {project} job {job_id}: {download.error}") + download_failures.append(f'artifact download HTTP {download.status_code}: {download.error}') + job_meta.append({'job_id': job_id, 'name': job.get('name'), 'status': job.get('status'), 'stage': job.get('stage'), 'trace': trace_written, 'artifact': artifact_meta}) + pipeline_meta.append({'pipeline_id': pipeline_id, 'status': pipeline.get('status'), 'ref': pipeline.get('ref'), 'updated_at': pipeline.get('updated_at'), 'jobs': job_meta}) + if written == 0 and artifact_extracted_total == 0: + if download_failures: + text = '; '.join(download_failures[:5]) + all_failures = '; '.join(download_failures) + auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any( + token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit') + ) + return { + "findings": [], "errors": [text], "error_class": 'source_auth' if auth_failure else 'remote_transient', + "retryable": True, "source_failure": auth_failure, + "source_failure_category": 'auth_forbidden' if auth_failure else 'remote_transient', + "source_failure_auth_related": auth_failure, + "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta}, + } + return {"findings": [], "errors": [], "skipped": "no downloadable job traces or artifacts", "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta}} + _raise_if_scan_slot_fatal() + results = run_filesystem_scan(work_dir, timeout_sec, detectors, exclude_detectors, no_verification, trufflehog_config) + results['package'] = {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url, "pipelines": pipeline_meta, "trace_files": written, "artifact_files": artifact_extracted_total} + if download_failures: + text = '; '.join(download_failures[:5]) + all_failures = '; '.join(download_failures) + auth_failure = any(f'HTTP {status}' in all_failures for status in (401, 403, 429)) or any( + token in all_failures.lower() for token in ('auth forbidden', 'unauthorized', 'rate limit') + ) + results['errors'] = list(results.get('errors') or []) + [text] + results['error_class'] = 'source_auth' if auth_failure else 'remote_transient' + results['retryable'] = True + results['source_failure'] = auth_failure + results['source_failure_category'] = 'auth_forbidden' if auth_failure else 'remote_transient' + results['source_failure_auth_related'] = auth_failure + return results + except ScanSlotFatalError: + raise + except RateLimitError as exc: + category = getattr(exc, 'category', '') + if category == 'not_found': + return {"findings": [], "errors": [], "skipped": str(exc), "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}} + return { + "findings": [], "errors": [str(exc)], "error_class": 'source_auth' if getattr(exc, 'auth_related', True) else 'remote_transient', + "retryable": True, "source_failure": True, + "source_failure_category": category or 'rate_limit', + "source_failure_auth_related": bool(getattr(exc, 'auth_related', True)), + "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}, + **({'_diagnostic_http': exc.diagnostic_http} if exc.diagnostic_http else {}), + } + except Exception as exc: + return { + "findings": [], "errors": [f"GitLab CI acquisition failed: {exc}"], + "error_class": "remote_transient", "retryable": True, "source_failure": True, + "source_failure_category": "network", "source_failure_auth_related": False, + "package": {"ci_provider": "gitlab_ci", "project": project, "project_url": project_url}, + } + finally: + cleanup_command_work_dir(work_dir) + +def send_webhook_notification(webhook_url, finding, target): + """Send webhook notification for a finding""" + if not webhook_url: + return + + try: + payload = { + "text": f"Secret Found in {target}", + "attachments": [{ + "color": "danger", + "fields": [ + {"title": "Detector", "value": finding.get('DetectorName', 'Unknown'), "short": True}, + {"title": "Target", "value": target, "short": True}, + {"title": "Verified", "value": str(finding.get('Verified', False)), "short": True} + ] + }] + } + + response = requests.post(webhook_url, json=payload, timeout=10) + response.raise_for_status() + logger.info(f"Webhook notification sent for {target}") + except Exception as e: + logger.error(f"Failed to send webhook notification: {str(e)}") + + +def _captured_http_failure(error): + current = error + seen = set() + for _index in range(8): + if current is None or id(current) in seen: + break + seen.add(id(current)) + direct_status = getattr(current, 'status_code', None) + if type(direct_status) is int: + direct_body = getattr(current, 'body', b'') + result = { + 'status_code': direct_status, + 'content_type': getattr(current, 'content_type', None), + 'request_id': getattr(current, 'request_id', None), + 'operation': getattr(current, 'operation', 'provider-api'), + 'headers_b64': base64.b64encode(json.dumps( + getattr(current, 'headers', None) or {}, + ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).decode('ascii'), + } + result['body_capture_truncated'] = bool(getattr( + current, 'body_capture_truncated', False + )) + if direct_body is not None: + if isinstance(direct_body, str): + direct_body = direct_body.encode('utf-8') + if not isinstance(direct_body, bytes): + direct_body = bytes(direct_body or b'') + result['body_b64'] = base64.b64encode(direct_body).decode('ascii') + original_size = getattr(current, 'body_original_size', None) + stored_size = getattr(current, 'body_stored_size', None) + body_sha256 = getattr(current, 'body_sha256', None) + if original_size is None or stored_size is None or body_sha256 is None: + material = make_body_material(direct_body) + original_size = material.original_size + stored_size = material.stored_size + body_sha256 = material.sha256 + result['body_capture_truncated'] = material.truncated + result.update({ + 'body_original_size': original_size, + 'body_stored_size': stored_size, + 'body_sha256': body_sha256, + }) + return result + response = getattr(current, 'response', None) + status = getattr(response, 'status_code', None) + if type(status) is int: + body_material = getattr(response, '_truf_diagnostic_body_material', None) + if body_material is None and ( + not getattr(response, 'raw', None) + or getattr(response, '_content_consumed', False) + ): + try: + body = getattr(response, 'content', b'') + except RuntimeError: + body = None + if isinstance(body, str): + body = body.encode('utf-8') + if body is not None and not isinstance(body, bytes): + body = bytes(body) + if body is not None: + body_material = make_body_material(body) + headers = getattr(response, 'headers', {}) or {} + result = { + 'status_code': status, + 'content_type': str(headers.get('Content-Type') or '') or None, + 'request_id': str( + headers.get('X-Request-Id') or headers.get('X-GitHub-Request-Id') or '' + ) or None, + 'operation': 'provider-request', + 'headers_b64': base64.b64encode(json.dumps( + {str(key): str(value) for key, value in headers.items()}, + ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).decode('ascii'), + 'body_capture_truncated': False, + } + if body_material is not None: + body = diagnostic_material_bytes(body_material) + result.update({ + 'body_b64': base64.b64encode(body).decode('ascii'), + 'body_original_size': body_material.original_size, + 'body_stored_size': body_material.stored_size, + 'body_sha256': body_material.sha256, + 'body_capture_truncated': body_material.truncated, + }) + return result + current = getattr(current, '__cause__', None) or getattr( + current, '__context__', None + ) + return None + + +def _captured_http_body_material(payload, capture): + material = make_body_material(payload) + metadata_fields = ( + 'body_original_size', 'body_stored_size', 'body_sha256', + ) + if not any(name in capture for name in metadata_fields): + if capture.get('body_capture_truncated'): + raise ValueError('truncated diagnostic HTTP material lacks capture metadata') + return material + if ( + not all(name in capture for name in metadata_fields) + or type(capture.get('body_capture_truncated')) is not bool + or isinstance(capture['body_original_size'], bool) + or not isinstance(capture['body_original_size'], int) + or isinstance(capture['body_stored_size'], bool) + or not isinstance(capture['body_stored_size'], int) + or capture['body_stored_size'] != len(payload) + or capture['body_original_size'] < capture['body_stored_size'] + or capture['body_capture_truncated'] + != (capture['body_original_size'] > capture['body_stored_size']) + or not isinstance(capture['body_sha256'], str) + or re.fullmatch(r'[a-f0-9]{64}', capture['body_sha256']) is None + or ( + not capture['body_capture_truncated'] + and capture['body_sha256'] != hashlib.sha256(payload).hexdigest() + ) + ): + raise ValueError('captured diagnostic HTTP material metadata is invalid') + return replace( + material, + original_size=capture['body_original_size'], + stored_size=capture['body_stored_size'], + sha256=capture['body_sha256'], + truncated=capture['body_capture_truncated'], + ) + + +def scan_target_result(target, scan_type, scan_event_id, scan_kwargs=None): + scan_kwargs = dict(scan_kwargs or {}) + scan_started_at = datetime.now(timezone.utc).isoformat() + started = time.perf_counter() + try: + direct_kind = _client_remote_execution_kind.get() + expected_direct_kind = { + 'docker': 'docker_direct_v1', + 'huggingface': 'huggingface_space_v1', + }.get(scan_type) + if ( + _client_scan_manifest.get() is not None + and expected_direct_kind is not None + and direct_kind != expected_direct_kind + ): + raise RuntimeError( + 'remote direct scan lacks its execution authority' + ) + if direct_kind is not None and direct_kind != expected_direct_kind: + raise RuntimeError('remote direct execution platform changed') + if scan_type in ('git', 'github', 'github_archive', 'gitlab'): + provider = 'github' if scan_type == 'github_archive' else scan_type if scan_type in ('github', 'gitlab') else None + result = scan_git_repo(target, provider=provider, **scan_kwargs) + elif scan_type == 'docker': + docker_layer_work = scan_kwargs.pop('docker_layer_work', None) + if docker_layer_work is not None: + scan_kwargs.pop('trufflehog_concurrency', None) + scan_kwargs.pop('docker_recovery_limits', None) + scan_kwargs.pop('docker_recovery_min_free_bytes', None) + try: + result = scan_docker_layer_plan( + target, docker_layer_work, **scan_kwargs, + ) + except (ScanSlotFatalError, DockerLayerInfrastructureError): + raise + except Exception as exc: + raise DockerLayerInfrastructureError( + 'scanner_infrastructure', + 'Docker layer scanner infrastructure failed', + category='source_resource', + ) from exc + else: + result = scan_docker_image( + target, + config_dir=( + None if direct_kind == 'docker_direct_v1' + else docker_token_manager.get_next_config() + ), + **scan_kwargs, + ) + elif scan_type == 'huggingface': + result = scan_huggingface_space(target, **scan_kwargs) + elif scan_type == 'npm': + result = scan_npm_package(target, **scan_kwargs) + elif scan_type == 'pypi': + result = scan_pypi_package(target, **scan_kwargs) + elif scan_type == 'package_git': + result = scan_package_git_repo(target, **scan_kwargs) + elif scan_type in ('postman', 'github_gists', 'github_archive_files'): + result = scan_postman_target(target, **scan_kwargs) + elif scan_type == 'github_actions': + result = scan_github_actions_repo(target, **scan_kwargs) + elif scan_type == 'gitlab_ci': + result = scan_gitlab_ci_project(target, **scan_kwargs) + else: + result = {'findings': [], 'errors': [f'Unknown scan type: {scan_type}']} + except (ScanSlotFatalError, DockerLayerInfrastructureError): + raise + except Exception as exc: + result = {'findings': [], 'errors': [str(exc)]} + http_failure = _captured_http_failure(exc) + if http_failure is not None: + result['_diagnostic_http'] = http_failure + apply_result_error_scope(result) + result['target'] = target + result['scan_type'] = scan_type + result['scan_event_id'] = str(scan_event_id) + result['scan_started_at'] = scan_started_at + result['duration_sec'] = time.perf_counter() - started + result['timestamp'] = datetime.now(timezone.utc).isoformat() + assign_finding_uids(result) + return result + + +@dataclass(frozen=True) +class StagedResult: + target: str + scan_event_id: str + bundle_id: str + reservation_id: int + scan_event_hash: str + actual_bytes: int + relative_path: str + frame_count: int + finding_count: int + error_count: int + candidate_count: int + queue_status: str + source_failure: bool + source_failure_category: str + source_failure_auth_related: bool + first_error: str + + def as_dict(self): + return dict(self.__dict__) + + +def stage_result_bundle( + result, reservation, bundle_root, scan_options, queue_disposition, + candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024, + require_s_drive=False, fault=None, diagnostic_slot_id=0, + diagnostic_attempt=1, +): + reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation) + if not isinstance(result, dict): + raise ValueError('scan result must be an object') + canonical_reservation_target = normalize_target( + reservation.target, reservation.platform, + ) + if not reservation.normalized_target: + reservation = replace( + reservation, normalized_target=canonical_reservation_target, + ) + + def optional_integer_identity(name, expected): + if name not in result or result[name] is None: + return + value = result[name] + if isinstance(value, bool): + raise ValueError('scan result identity does not match its reservation') + try: + value = int(value) + except (TypeError, ValueError, OverflowError): + raise ValueError('scan result identity does not match its reservation') from None + if value != int(expected): + raise ValueError('scan result identity does not match its reservation') + + optional_integer_identity('reservation_id', reservation.reservation_id) + optional_integer_identity('result_reservation_id', reservation.reservation_id) + optional_integer_identity('queue_id', reservation.queue_id) + for name, expected in ( + ('bundle_id', reservation.bundle_id), + ('source', reservation.source), + ('platform', reservation.platform), + ('query', reservation.query), + ('normalized_target', reservation.normalized_target), + ): + if name in result and result[name] is not None and str(result[name]) != str(expected): + raise ValueError('scan result identity does not match its reservation') + if ( + str(result.get('scan_event_id') or '') != reservation.scan_event_id + or str(result.get('target') or '') != reservation.target + or str(result.get('scan_type') or '') != reservation.platform + or normalize_target(result.get('target'), reservation.platform) + != canonical_reservation_target + or reservation.normalized_target != canonical_reservation_target + ): + raise ValueError('scan result identity does not match its reservation') + + strip_nearby_context_for_persistence(result) + findings = result.get('findings') or [] + errors = result.get('errors') or [] + candidates = [] + candidate_bytes = 0 + candidate_identities = set() + candidate_truncated = False + + def add_candidate(candidate, attribution): + nonlocal candidate_bytes, candidate_truncated + identity = ( + candidate.service, candidate.credential_hash, + '' if candidate.service == 'provider_resolver' else str( + (attribution or {}).get('finding_uid') or (attribution or {}).get('origin') or '' + ), + ) + if identity in candidate_identities: + return + frame = candidate.as_frame(attribution) + encoded_size = len(json.dumps( + frame, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8')) + if len(candidates) >= max(0, int(candidate_max_items)): + candidate_truncated = True + return + if candidate_bytes + encoded_size > max(0, int(candidate_max_bytes)): + candidate_truncated = True + return + candidate_identities.add(identity) + candidates.append(frame) + candidate_bytes += encoded_size + + for finding in findings: + attribution = { + 'finding_uid': str(finding.get('finding_uid') or ''), + 'detector_name': str(finding.get('DetectorName') or finding.get('DetectorType') or ''), + } + try: + for candidate in extract_candidates(finding, attribution): + add_candidate(candidate, attribution) + except ValueError as exc: + result.setdefault('warnings', []).append( + f'Optional keycheck candidate was omitted: {type(exc).__name__}' + ) + result['degraded'] = True + if result.get('structured_keycheck_pending') and isinstance(result.get('postman'), dict): + try: + postman_data = result['postman'] + cache_path, _ = validate_postman_cache_artifact( + postman_data, + int(result.get('postman_max_artifact_size_mb') or 20), + expected_size=result.get('bytes'), + ) + contexts = load_postman_context(cache_path) + origin = artifact_origin_label(postman_data, cache_path) + attribution = {'origin': origin} + for candidate in extract_structured_candidates({ + 'contexts': contexts, 'origin': origin, + }, attribution): + structured_origin = candidate.metadata.get('structured_origin') or origin + add_candidate(candidate, {'origin': structured_origin}) + except Exception as exc: + result.setdefault('warnings', []).append( + f'Optional structured keycheck extraction failed: {type(exc).__name__}' + ) + result['degraded'] = True + if candidate_truncated: + result.setdefault('warnings', []).append( + 'Keycheck candidates reached their pre-reserved per-event bound; remaining candidates were omitted' + ) + result['degraded'] = True + status = target_status(result) + first_error = '' + for error in errors: + first_error = next((line.strip() for line in str(error).splitlines() if line.strip()), '') + if first_error: + break + diagnostics = list(result.get('diagnostics') or ()) + if errors and not diagnostics: + timestamp = str(result.get('timestamp') or result.get('scan_started_at') or '') + try: + occurred = datetime.fromisoformat(timestamp.replace('Z', '+00:00')) + except ValueError: + occurred = datetime.now(timezone.utc) + if occurred.tzinfo is None: + occurred = datetime.now(timezone.utc) + occurred_at = occurred.astimezone(timezone.utc).isoformat( + timespec='milliseconds' + ).replace('+00:00', 'Z') + scan_meta = result.get('scan_meta') or {} + timed_out = bool(scan_meta.get('command_timed_out')) or result.get('error_class') == 'timeout' + category_name = str( + result.get('source_failure_category') or result.get('error_class') or '' + ).lower() + category = { + 'source_auth': DiagnosticCategory.AUTHORIZATION, + 'auth_forbidden': DiagnosticCategory.AUTHORIZATION, + 'rate_limit': DiagnosticCategory.RATE_LIMIT, + 'not_found': DiagnosticCategory.NOT_FOUND, + 'network': DiagnosticCategory.NETWORK, + 'remote_transient': DiagnosticCategory.NETWORK, + 'timeout': DiagnosticCategory.TIMEOUT, + 'source_resource': DiagnosticCategory.STORAGE, + 'storage': DiagnosticCategory.STORAGE, + }.get(category_name, DiagnosticCategory.SCANNER) + phase_name = str(scan_meta.get('timed_out_phase') or WorkerPhase.SCANNING.value) + try: + diagnostic_phase = WorkerPhase(phase_name) + except ValueError: + diagnostic_phase = WorkerPhase.SCANNING + scan_outcome = { + 'clean': ScanOutcome.CLEAN, + 'found': ScanOutcome.FOUND, + 'degraded': ScanOutcome.DEGRADED, + 'error': ScanOutcome.ERROR, + 'skipped': ScanOutcome.SKIPPED, + }.get(status, ScanOutcome.ERROR) + try: + raw_stdout = base64.b64decode( + str(result['_diagnostic_raw_stdout_b64']).encode('ascii'), + validate=True, + ) if '_diagnostic_raw_stdout_b64' in result else None + raw_stderr = base64.b64decode( + str(result['_diagnostic_raw_stderr_b64']).encode('ascii'), + validate=True, + ) if '_diagnostic_raw_stderr_b64' in result else None + except (UnicodeEncodeError, ValueError, binascii.Error) as exc: + raise ValueError('captured diagnostic process material is invalid') from exc + stdout_limit = stderr_limit = 0 + if raw_stdout is not None or raw_stderr is not None: + desired = [ + min(len(raw_stdout), MAX_DIAGNOSTIC_LOG_BYTES) + if raw_stdout is not None else 0, + min(len(raw_stderr), MAX_DIAGNOSTIC_LOG_BYTES) + if raw_stderr is not None else 0, + ] + total = sum(desired) + if total <= MAX_DIAGNOSTIC_LOG_BYTES: + stdout_limit, stderr_limit = desired + elif total: + stdout_limit = ( + MAX_DIAGNOSTIC_LOG_BYTES * desired[0] + ) // total + stderr_limit = MAX_DIAGNOSTIC_LOG_BYTES - stdout_limit + process = None + if raw_stdout is not None or raw_stderr is not None: + process = DiagnosticProcessContext( + name='trufflehog', + exit_code=( + int(scan_meta['trufflehog_returncode']) + if type(scan_meta.get('trufflehog_returncode')) is int else None + ), + signal=None, + timed_out=timed_out, + stdout=( + make_log_material(raw_stdout, maximum=stdout_limit) + if raw_stdout is not None else None + ), + stderr=( + make_log_material(raw_stderr, maximum=stderr_limit) + if raw_stderr is not None else None + ), + ) + http_value = result.get('_diagnostic_http') + http = None + if isinstance(http_value, dict) and type(http_value.get('status_code')) is int: + raw_body = None + raw_headers = None + if 'body_b64' in http_value: + try: + raw_body = base64.b64decode( + str(http_value['body_b64']).encode('ascii'), + validate=True, + ) + except (UnicodeEncodeError, ValueError, binascii.Error) as exc: + raise ValueError('captured diagnostic HTTP material is invalid') from exc + if 'headers_b64' in http_value: + try: + raw_headers = base64.b64decode( + str(http_value['headers_b64']).encode('ascii'), + validate=True, + ) + except (UnicodeEncodeError, ValueError, binascii.Error) as exc: + raise ValueError('captured diagnostic HTTP headers are invalid') from exc + http = DiagnosticHTTPContext( + operation=str(http_value.get('operation') or 'provider-request'), + status_code=http_value['status_code'], + content_type=http_value.get('content_type'), + request_id=http_value.get('request_id'), + body=( + _captured_http_body_material(raw_body, http_value) + if raw_body is not None else None + ), + headers=( + make_body_material(raw_headers) + if raw_headers is not None else None + ), + ) + transformation = str(result.get('_diagnostic_stderr_transformation') or ( + 'HTTP status and bounded body evidence preserve captured values; parsed ' + 'response headers were serialized as a deterministic mapping because raw ' + 'wire order and casing are unavailable' + + ( + '; response body capture reached its explicit byte bound' + if http_value and http_value.get('body_capture_truncated') else '' + ) + if http is not None else + 'raw provider/process material was unavailable; canonical diagnostic ' + 'was projected from the existing legacy error representation' + )) + fingerprint = hashlib.sha256( + ('\n'.join(str(error) for error in errors)).encode('utf-8') + ).hexdigest() + kind = ( + DiagnosticKind.PROVIDER_HTTP if http is not None + else DiagnosticKind.SCANNER_PROCESS if process is not None + else DiagnosticKind.EXCEPTION + ) + diagnostics.append(build_diagnostic_envelope( + occurrence_id=f'{reservation.scan_event_id}:scan-result', + reservation_id=reservation.reservation_id, + scan_event_id=reservation.scan_event_id, + slot_id=int(diagnostic_slot_id), + source=reservation.source, + phase=diagnostic_phase, + kind=kind, + category=(DiagnosticCategory.TIMEOUT if timed_out else category), + code=('scan.stage_timeout' if timed_out else 'scan.result_error'), + summary=(first_error or 'scan returned one or more errors')[:1000], + retryable=bool(result.get('retryable', False)), + attempt=max(1, int(diagnostic_attempt)), + assignment_outcome=AssignmentOutcome.ACCEPTED, + scan_outcome=scan_outcome, + occurred_at=occurred_at, + captured_at=occurred_at, + http=http, + process=process, + exception=DiagnosticExceptionContext( + type='truf.diagnostic.TechnicalTransformation', + message=transformation, + fingerprint=fingerprint, + ), + )) + metadata = { + key: value for key, value in result.items() + if key not in ('findings', 'errors', 'diagnostics') + and not key.startswith('_diagnostic_') + } + metadata.update({ + 'reservation_id': reservation.reservation_id, + 'queue_id': reservation.queue_id, + 'bundle_id': reservation.bundle_id, + 'source': reservation.source, + 'platform': reservation.platform, + 'query': reservation.query, + 'normalized_target': reservation.normalized_target, + 'status': status, + 'findings_count': len(findings), + 'verified_findings_count': sum(1 for finding in findings if finding.get('Verified')), + 'error_count': len(errors), + 'first_error_summary': first_error[:500], + 'scan_options': dict(scan_options or {}), + 'derived_postman_targets': list(result.get('postman_targets') or ()), + **dict(queue_disposition or {}), + }) + with ResultBundleWriter.open( + bundle_root, reservation, fault=fault, require_s_drive=require_s_drive, + ) as writer: + for finding in findings: + writer.write_finding(finding) + for error in errors: + writer.write_error(error) + for diagnostic in diagnostics: + writer.write_diagnostic(diagnostic) + for candidate in candidates: + writer.write_candidate(candidate) + commit = writer.finish(metadata) + findings.clear() + errors.clear() + diagnostics.clear() + candidates.clear() + return StagedResult( + target=str(result.get('target') or ''), + scan_event_id=commit.scan_event_id, + bundle_id=commit.bundle_id, + reservation_id=commit.reservation_id, + scan_event_hash=commit.scan_event_hash, + actual_bytes=commit.actual_bytes, + relative_path=commit.relative_path, + frame_count=commit.frame_count, + finding_count=commit.finding_count, + error_count=commit.error_count, + candidate_count=commit.candidate_count, + queue_status=str(metadata.get('queue_status') or ''), + source_failure=bool(result.get('source_failure')), + source_failure_category=str(result.get('source_failure_category') or ''), + source_failure_auth_related=bool(result.get('source_failure_auth_related')), + first_error=first_error[:500], + ) + + +class ResultSinkError(RuntimeError): + def __init__(self, failures, results): + self.failures = list(failures) + self.results = list(results) + targets = ', '.join(str(target) for target, _ in self.failures[:5]) + super().__init__(f'result sink failed for {len(self.failures)} completed target(s): {targets}') + + +def scan_targets_batch( + targets, + scan_type, + progress_callback=None, + max_workers=4, + persist_results=True, + result_sink=None, + scan_slot_leases=None, + sink_within_scan_slot=False, + **kwargs, +): + """Scan multiple targets with parallel processing and progress tracking""" + from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait + import threading + + _raise_if_scan_slot_fatal() + + results = [] + total = len(targets) if hasattr(targets, '__len__') else None + completed = 0 + results_lock = threading.Lock() + sink_failures = [] + provided_leases = list(scan_slot_leases or []) + if provided_leases and len(provided_leases) != total: + raise ValueError('one pre-acquired scan slot lease is required for every target') + + def apply_worker_sink(result): + if not sink_within_scan_slot or result_sink is None: + return result + try: + result_sink(result) + result['_result_sink_applied'] = True + except Exception as exc: + result['_result_sink_exception'] = exc + return result + + def scan_single_target(target, scan_event_id, provided_lease=None): + """Scan a single target""" + _raise_if_scan_slot_fatal() + scan_started_at = datetime.now(timezone.utc).isoformat() + started = time.perf_counter() + try: + with scan_slot_scope( + ['scan-target', scan_type], kwargs.get('timeout_sec'), lease=provided_lease, + ): + try: + _raise_if_scan_slot_fatal() + result = scan_target_result(target, scan_type, scan_event_id, kwargs) + return apply_worker_sink(result) + except ScanSlotFatalError: + raise + except Exception as e: + result = { + "target": target, + "scan_type": scan_type, + "scan_event_id": scan_event_id, + "scan_started_at": scan_started_at, + "duration_sec": time.perf_counter() - started, + "timestamp": datetime.now(timezone.utc).isoformat(), + "findings": [], + "errors": [str(e)] + } + return apply_worker_sink(apply_result_error_scope(result)) + except ScanSlotFatalError: + raise + except Exception as e: + result = { + "target": target, + "scan_type": scan_type, + "scan_event_id": scan_event_id, + "scan_started_at": scan_started_at, + "duration_sec": time.perf_counter() - started, + "timestamp": datetime.now(timezone.utc).isoformat(), + "findings": [], + "errors": [str(e)] + } + return apply_worker_sink(apply_result_error_scope(result)) + + max_workers = max(1, int(max_workers or 1)) + executor = ThreadPoolExecutor(max_workers=max_workers) + future_to_target = {} + target_iterator = iter(enumerate(targets)) + + def submit_next(): + try: + index, target = next(target_iterator) + except StopIteration: + return False + provided_lease = provided_leases[index] if provided_leases else None + future_to_target[ + executor.submit(scan_single_target, target, str(uuid.uuid4()), provided_lease) + ] = target + return True + + try: + for _ in range(max_workers): + _raise_if_scan_slot_fatal() + if not submit_next(): + break + while future_to_target: + _raise_if_scan_slot_fatal() + done, _ = wait(tuple(future_to_target), timeout=0.2, return_when=FIRST_COMPLETED) + for future in done: + target = future_to_target.pop(future) + result = future.result() + + with results_lock: + results.append(result) + persistence_error = result.pop('_result_sink_exception', None) + sink_applied = bool(result.pop('_result_sink_applied', False)) + if result_sink is not None and not sink_applied and persistence_error is None: + try: + result_sink(result) + except Exception as exc: + persistence_error = exc + if persist_results: + try: + if not save_scan_result(result): + raise RuntimeError('save_scan_result returned false') + except Exception as exc: + persistence_error = persistence_error or exc + if persistence_error is not None: + result['persistence_failure'] = str(persistence_error)[:500] + sink_failures.append((target, persistence_error)) + logger.error(f'Persistence failed for completed target {target}: {persistence_error}') + submit_next() + continue + with results_lock: + completed += 1 + findings_count = len(result.get('findings', [])) + errors_count = len(result.get('errors', [])) + skipped = bool(result.get('skipped')) + logger.info( + f"Completed {completed}/{total}: {target} " + f"(findings={findings_count}, errors={errors_count}, skipped={skipped})" + ) + + if progress_callback: + progress_callback(completed, total, target) + submit_next() + except ScanSlotFatalError as exc: + _set_scan_slot_fatal(str(exc)) + for future in future_to_target: + future.cancel() + wait(tuple(future_to_target), timeout=1.0) + try: + executor.shutdown(wait=False, cancel_futures=True) + except TypeError: + executor.shutdown(wait=False) + executor = None + raise + finally: + if executor is not None: + executor.shutdown(wait=True) + for lease in provided_leases: + if lease.heartbeat_thread is None: + lease.release() + + if progress_callback: + progress_callback( + completed, + total, + "Scan completed" if not sink_failures else "Scan completed with persistence failures", + ) + + if sink_failures: + raise ResultSinkError(sink_failures, results) + return results + +def extract_trufflehog_error_lines(output): + error_lines = [] + for index, line in enumerate(_iter_output_lines(output), 1): + if index > 2000: + break + line = line.strip() + if not line: + continue + try: + payload = json.loads(line) + level = str(payload.get('level', '')).lower() + message = str(payload.get('msg', '')).lower() + if message == 'error cleaning temporary artifacts': + continue + if 'error' in level or 'error' in message or payload.get('error'): + error_lines.append(line) + except json.JSONDecodeError: + if 'error' in line.lower() or 'failed' in line.lower(): + error_lines.append(line) + return error_lines diff --git a/app/scanner_db.py b/app/scanner_db.py new file mode 100644 index 0000000..b9ea7e5 --- /dev/null +++ b/app/scanner_db.py @@ -0,0 +1,30695 @@ +import hashlib +import hmac +import importlib +import ipaddress +import json +import logging +import math +import os +import random +import re +import secrets +import threading +import time +import uuid +from datetime import datetime, timedelta, timezone +from urllib.parse import urlsplit + +from db_backend import ( + POSTGRES_APPLICATION_SCHEMA, + connect_host_agent_postgres, + connect_postgres, + connect_sqlite, + database_url_from_env, + is_postgres_url, + redact_database_url, +) +from result_spool import SpoolHashConflictError, prepare_scan_event +from keycheck_candidates import candidate_uid, stored_provider_key_hash +from process_identity import exact_process_identity_state +from target_identity import ( + docker_target_identity, + normalize_huggingface_space_id, + parse_dockerhub_digest_target, + parse_docker_target, + postman_target_identity, + validate_docker_image_reference, +) + + +logger = logging.getLogger(__name__) +REMOTE_PROGRESS_RESOLUTION_GRACE_SECONDS = 120 +REMOTE_PROGRESS_CLOCK_TOLERANCE_SECONDS = 60 +ADMIN_DISCARDED_QUEUE_REASON = 'discarded stale unassigned backlog by source administrator' +ADMIN_DISCARDED_QUEUE_SQL = ( + "target_queue.status = 'quarantined' AND target_queue.last_error = " + f"'{ADMIN_DISCARDED_QUEUE_REASON}'" +) + + +class RuntimeSafetySchemaError(RuntimeError): + pass + + +class PipelineCapacityUnavailable(RuntimeError): + pass + + +class RuntimeControlConflictError(RuntimeError): + pass + + +class RuntimeControlRevisionConflictError(RuntimeControlConflictError): + def __init__(self, expected_revision, current_state): + self.expected_revision = expected_revision + self.current_state = current_state + super().__init__( + f'runtime control revision changed from {expected_revision} ' + f'to {current_state["revision"]}' + ) + + +class RuntimeOperationIdentityConflictError(RuntimeControlConflictError): + pass + + +class RuntimeOperationTransitionError(RuntimeControlConflictError): + pass + + +class RuntimeControlTransitionError(RuntimeControlConflictError): + pass + + +class DiscoveryPausedError(RuntimeError): + def __init__(self, control_state): + self.control_state = dict(control_state) + super().__init__('provider discovery is paused') + + +class ScanEventConflictError(RuntimeError): + pass + + +class WorkerObservabilityConflictError(ScanEventConflictError): + pass + + +class WorkerProgressInactiveError(WorkerObservabilityConflictError): + pass + + +def _canonical_worker_contract(kind, payload): + validators = { + 'progress': ('validate_worker_event', 'validate_progress_event', 'validate_event'), + 'diagnostic': ( + 'validate_worker_diagnostic', 'validate_diagnostic_envelope', + 'validate_diagnostic', + ), + } + if kind not in validators: + raise ValueError(f'unknown worker contract kind: {kind}') + try: + contracts = importlib.import_module('worker_contracts') + except ModuleNotFoundError as exc: + if exc.name == 'worker_contracts': + raise RuntimeError( + f'worker_contracts is required to persist worker {kind} records' + ) from exc + raise RuntimeError(f'worker_contracts import failed: missing {exc.name}') from exc + except ImportError as exc: + raise RuntimeError(f'worker_contracts import failed: {exc}') from exc + codecs = { + 'progress': ('decode_worker_event', 'encode_worker_event'), + 'diagnostic': ('decode_diagnostic_envelope', 'encode_diagnostic_envelope'), + } + decoder = getattr(contracts, codecs[kind][0], None) + encoder = getattr(contracts, codecs[kind][1], None) + if callable(decoder) and callable(encoder): + try: + raw = json.dumps( + dict(payload or {}), ensure_ascii=True, sort_keys=True, + separators=(',', ':'), + ).encode('ascii') + canonical_bytes = encoder(decoder(raw)) + normalized = json.loads(canonical_bytes.decode('ascii')) + except (TypeError, ValueError, UnicodeError) as exc: + raise ValueError(f'worker {kind} contract is invalid') from exc + canonical = canonical_bytes.decode('ascii') + return normalized, canonical, hashlib.sha256(canonical_bytes).hexdigest() + validator = next( + (getattr(contracts, name) for name in validators[kind] + if callable(getattr(contracts, name, None))), + None, + ) + if validator is None: + expected = ', '.join((*codecs[kind], *validators[kind])) + raise RuntimeError( + f'worker_contracts has no {kind} validator; expected one of: {expected}' + ) + normalized = validator(dict(payload or {})) + if normalized is None: + normalized = dict(payload or {}) + elif not isinstance(normalized, dict): + to_dict = getattr(normalized, 'to_dict', None) + if not callable(to_dict) or not isinstance((normalized := to_dict()), dict): + raise RuntimeError(f'worker_contracts {kind} validator returned a non-mapping') + try: + canonical = json.dumps( + normalized, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ) + except (TypeError, ValueError) as exc: + raise ValueError(f'worker {kind} contract is not canonical JSON') from exc + return normalized, canonical, hashlib.sha256(canonical.encode('utf-8')).hexdigest() + + +class DockerCoverageDispositionConflictError(ScanEventConflictError): + pass + + +class DockerFindingAttributionLimitError(ValueError): + pass + + +class DiscoveryRetryLeaseError(RuntimeError): + pass + + +def env_int(name, default): + try: + return int(os.getenv(name, default)) + except (TypeError, ValueError): + return default + + +def env_float(name, default): + try: + return float(os.getenv(name, default)) + except (TypeError, ValueError): + return default + + +DB_FILENAME = 'scanner.db' +SQLITE_BUSY_TIMEOUT_MS = max(1000, env_int('SCANNER_SQLITE_BUSY_TIMEOUT_MS', 15000)) +SQLITE_CONNECT_TIMEOUT_SEC = max(1, env_int('SCANNER_SQLITE_CONNECT_TIMEOUT_SEC', max(1, SQLITE_BUSY_TIMEOUT_MS // 1000))) +SQLITE_LOCK_RETRY_ATTEMPTS = max(1, env_int('SCANNER_SQLITE_LOCK_RETRY_ATTEMPTS', 8)) +SQLITE_LOCK_RETRY_BASE_SEC = max(0.05, env_float('SCANNER_SQLITE_LOCK_RETRY_BASE_SEC', 0.25)) +SQLITE_LOCK_RETRY_MAX_SEC = max(0.5, env_float('SCANNER_SQLITE_LOCK_RETRY_MAX_SEC', 5.0)) +STALE_RUN_FINALIZE_BATCH_SIZE = max(1, env_int('SCANNER_STALE_RUN_FINALIZE_BATCH_SIZE', 250)) +STALE_RUN_FINALIZE_MAX_BATCHES = max(1, env_int('SCANNER_STALE_RUN_FINALIZE_MAX_BATCHES', 20)) +KNOWN_TARGET_LOOKUP_BATCH_SIZE = min(64, max(1, env_int('SCANNER_KNOWN_TARGET_LOOKUP_BATCH_SIZE', 64))) +DISCOVERY_RETRY_MAX_QUERY_CHARS = 256 +DISCOVERY_RETRY_MAX_PAGE = 30 +DISCOVERY_RETRY_MAX_REPOSITORIES_PER_PAGE = 100 +DISCOVERY_RETRY_MAX_ALLOWLIST = 256 +DISCOVERY_RETRY_MAX_CLAIM = 10 +DISCOVERY_RETRY_MAX_ATTEMPTS = 1000000 +DISCOVERY_RETRY_BASE_DELAY_SEC = min( + 3600, max(1, env_int('DOCKERHUB_DISCOVERY_RETRY_BASE_SEC', 60)), +) +DISCOVERY_RETRY_MAX_DELAY_SEC = min(86400, max( + DISCOVERY_RETRY_BASE_DELAY_SEC, + env_int('DOCKERHUB_DISCOVERY_RETRY_MAX_SEC', 21600), +)) +DISCOVERY_RETRY_MAX_TRUSTED_DELAY_SEC = 86400 +DISCOVERY_RETRY_PASS_KINDS = frozenset(('ordinary', 'deep')) +DISCOVERY_RETRY_WORK_KINDS = frozenset(('query', 'page', 'range')) +DISCOVERY_RETRY_ERROR_CATEGORIES = frozenset(( + 'account_pool_exhausted', + 'auth_forbidden', + 'auth_invalid', + 'auth_unavailable', + 'invalid_payload', + 'network', + 'page_unavailable', + 'policy_mismatch', + 'provider_cooldown', + 'provider_unavailable', + 'query_removed', + 'rate_limit', + 'remote_transient', + 'request_failed', + 'tail_unavailable', + 'transport', +)) +DISCOVERY_RETRY_DUE_INDEX_PREDICATE = "status = 'pending'" +DISCOVERY_RETRY_LEASE_INDEX_PREDICATE = "status = 'leased'" +CLAIM_CANDIDATE_MIN = min(1000, max(1, env_int('SCANNER_CLAIM_CANDIDATE_MIN', 64))) +CLAIM_CANDIDATE_MAX = max(CLAIM_CANDIDATE_MIN, min(1000, env_int('SCANNER_CLAIM_CANDIDATE_MAX', 1000))) +RESULT_SPOOL_ADVISORY_CLASS = 1414681926 +RESULT_SPOOL_ADVISORY_OBJECT = 1397772111 +RESULT_SPOOL_PROGRESS_MAX_WAIT_SEC = max(3600, env_int('RESULT_SPOOL_PROGRESS_MAX_WAIT_SEC', 86400)) +PIPELINE_ADVISORY_CLASS = 1414681926 +PIPELINE_ADVISORY_OBJECTS = { + 'result_ingester': 1768842867, + 'jsonl_projector': 1785753445, +} +CI_SOFT_SKIP_REASONS = { + 'github_actions': frozenset(('no recent workflow runs', 'no downloadable workflow logs or artifacts')), + 'gitlab_ci': frozenset(('no recent pipelines', 'no downloadable job traces or artifacts')), +} + + +def _ci_cooldown_source_predicate(source): + reasons = ' OR '.join( + f"skipped_reason = '{reason}'" for reason in sorted(CI_SOFT_SKIP_REASONS[source]) + ) + return f"source = '{source}' AND ({reasons})" + + +CI_COOLDOWN_INDEX_PREDICATE = "status = 'skipped' AND (" + ' OR '.join( + _ci_cooldown_source_predicate(source) for source in sorted(CI_SOFT_SKIP_REASONS) +) + ')' +TARGET_QUEUE_OBSERVABILITY_STATUSES = ( + 'pending', 'deferred', 'in_progress', 'done', 'failed', 'quarantined', 'cold', +) +TARGET_QUEUE_OBSERVABILITY_STATUS_LIMIT = min( + 10000, max(100, env_int('TARGET_QUEUE_OBSERVABILITY_STATUS_LIMIT', 100)), +) +TARGET_QUEUE_OBSERVABILITY_TIMEOUT_MS = min( + 10000, max(1000, env_int('TARGET_QUEUE_OBSERVABILITY_TIMEOUT_MS', 5000)), +) +TARGET_QUEUE_OBSERVABILITY_RETRY_BACKOFF_SEC = 300 +RETIREMENT_STATEMENT_TIMEOUT_MS = 300000 +KEYCHECK_RESULT_PROJECTION_RESERVE_BYTES = 3 * 1024 * 1024 +HAS_CLAIMABLE_TARGETS_V2_SQL = '''SELECT ( + EXISTS ( + SELECT 1 FROM target_queue + WHERE source = ? AND platform = ? + AND current_result_reservation_id IS NULL + AND status = 'pending' + AND (available_after IS NULL OR available_after <= ?) + AND (? = 0 OR COALESCE(attempts, 0) < ?) + AND (resolver_state IS NULL OR resolver_state = 'resolved') + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_targets experiment_target + WHERE experiment_target.target_queue_id = target_queue.id + ) + ) OR EXISTS ( + SELECT 1 FROM target_queue + WHERE source = ? AND platform = ? + AND current_result_reservation_id IS NULL + AND status = 'deferred' + AND available_after IS NOT NULL AND available_after <= ? + AND (? = 0 OR COALESCE(attempts, 0) < ?) + AND (resolver_state IS NULL OR resolver_state = 'resolved') + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_targets experiment_target + WHERE experiment_target.target_queue_id = target_queue.id + ) + ) +) AS present''' +PACKAGE_CANDIDATE_LOOKUP_MAX_RESULTS = 10000 +PACKAGE_CANDIDATE_SCAN_MAX_ROWS = 50000 +PACKAGE_CANDIDATE_SCAN_PAGE_SIZE = 250 +PACKAGE_CANDIDATE_WRITE_CHUNK_SIZE = 25 +PACKAGE_REPO_NONEMPTY_PREDICATE = "repo_url IS NOT NULL AND repo_url <> ''" +DOCKER_LEGACY_PROVENANCE_SEED_MAX_ROWS = 1000000 +DOCKER_LEGACY_PROVENANCE_SEED_PAGE_SIZE = 1000 +DOCKER_DEPTH_EXPERIMENT_MIGRATION = '20260909_18_docker_depth_experiment' +DOCKER_DEPTH_ROLLOUT_AUTHORITY_MIGRATION = '20260910_22_docker_depth_rollout_authority' +DOCKER_DEPTH_SCHEMA_SELECTOR_AUTHORITY_MIGRATION = ( + '20260910_23_docker_schema_selector_authority' +) +DOCKER_DEPTH_SCARCITY_COHORT_MIGRATION = ( + '20260910_24_docker_depth_scarcity_cohort' +) +DOCKER_DEPTH_HOLD_QUERY_INDEX_MIGRATION = ( + '20260910_25_docker_depth_hold_query_index' +) +DOCKER_DEPTH_RESOLVER_REFUND_MIGRATION = ( + '20260910_26_docker_depth_resolver_attempt_refund' +) +DOCKER_DEPTH_RESOLVER_DISPOSITION_MIGRATION = ( + '20260910_27_docker_depth_resolver_disposition' +) +REMOTE_WORKER_MIGRATION = '20260917_28_remote_worker_admission_transport' +OPERATIONS_CONTROL_MIGRATION = '20260919_29_operations_control_audit' +WORKER_OBSERVABILITY_MIGRATION = '20260923_30_worker_observability' +DIAGNOSTIC_PROJECTION_AUTHORITY_MIGRATION = ( + '20260924_31_diagnostic_projection_authority' +) +ASSIGNMENT_UPLOAD_TIMEOUT_MIGRATION = ( + '20260924_32_assignment_upload_timeout' +) +REMOTE_ASSIGNMENT_CAPACITY_MIGRATION = ( + '20260930_33_remote_assignment_capacity' +) +RUNTIME_CONTROL_TARGET_KIND = 'runtime-control' +RUNTIME_CONTROL_DRAIN_STATES = frozenset(('normal', 'draining', 'drained')) +RUNTIME_CONTROL_ACTION_TARGETS = { + 'control.discovery.pause': 'discovery', + 'control.discovery.resume': 'discovery', + 'control.dispatch.pause': 'dispatch', + 'control.dispatch.resume': 'dispatch', + 'control.drain.start': 'drain', + 'control.drain.cancel': 'drain', + 'control.drain.complete': 'drain', +} +RUNTIME_CONTROL_ACTIONS = frozenset(RUNTIME_CONTROL_ACTION_TARGETS) +RUNTIME_CONTROL_MAX_REVISION = 9223372036854775807 +RUNTIME_CONTROL_MAX_EXPECTED_REVISION = RUNTIME_CONTROL_MAX_REVISION - 1 +RUNTIME_ASYNC_TARGET_KIND = 'runtime-deployment' +RUNTIME_ASYNC_ACTION_TARGETS = { + 'apply-config': 'config', + 'apply-secrets': 'secrets', + 'apply-both': 'config-secrets', + 'restart': 'runtime', +} +RUNTIME_ASYNC_ACTIONS = frozenset(RUNTIME_ASYNC_ACTION_TARGETS) +RUNTIME_SOURCE_TARGET_KIND = 'managed-source' +RUNTIME_SOURCE_IDS = frozenset(( + 'discovery-producer:gitlab', + 'discovery-producer:dockerhub', + 'discovery-producer:huggingface', + 'result-ingester', 'jsonl-projector', 'janitor', 'worker-api', + 'keychecks', 'docker-shadow', 'dashboard', +)) +RUNTIME_SOURCE_ACTIONS = frozenset(( + 'start', 'stop', 'restart', 'pause', 'resume', 'set-interval', + 'once', 'set-mode', 'set-restart', 'set-restart-delay', +)) +RUNTIME_SOURCE_OPERATION_ACTIONS = { + f'supervisor.source.{action}': action for action in RUNTIME_SOURCE_ACTIONS +} +RUNTIME_WORKER_ADMIN_TARGET_KIND = 'worker-admin' +RUNTIME_WORKER_ADMIN_ACTION_TARGETS = { + 'workers.user.create': 'user', + 'workers.user.set-cap': 'user', + 'workers.user.disable': 'user', + 'workers.user.enable': 'user', + 'workers.device.issue': 'device', + 'workers.device.rotate': 'device', + 'workers.device.revoke': 'device', + 'workers.device.unrevoke': 'device', + 'workers.queue.requeue': 'deferred-queue', +} +RUNTIME_DOCUMENT_TARGET_KIND = 'runtime-document' +RUNTIME_DOCUMENT_ACTION_TARGETS = { + 'runtime.config.save': 'config', + 'runtime.secrets.save': 'secrets', +} +RUNTIME_MANAGED_FILE_TARGET_KIND = 'managed-file' +RUNTIME_MANAGED_FILE_ACTIONS = frozenset(( + 'files.create', 'files.replace', 'files.delete', +)) +RUNTIME_MANAGED_FILE_ADVISORY_CLASS = 1836212590 +_RUNTIME_MANAGED_FILE_SQLITE_LOCKS = tuple( + threading.Lock() for _ in range(64) +) +RUNTIME_ASYNC_TERMINAL_RESULTS = frozenset(( + 'succeeded', 'failed', 'rolled_back', 'failed_hold', +)) +RUNTIME_OPERATION_SAFE_CATEGORIES = frozenset(( + 'agent_failed', 'agent_unavailable', 'apply_failed', + 'health_check_failed', 'operation_conflict', 'restart_failed', + 'result_invalid', 'revision_conflict', 'rollback_failed', + 'supervisor_action_failed', 'validation_failed', 'worker_admin_mutation_failed', + 'runtime_document_save_failed', 'managed_file_mutation_failed', +)) +RUNTIME_OPERATION_SAFE_DETAILS = frozenset(( + 'active_hash_mismatch', 'apply_failed', 'bounded_result', + 'candidate_hash_mismatch', 'health_check_failed', 'lock_busy', + 'restart_failed', 'rollback_failed', 'validation_failed', +)) +RUNTIME_AGENT_RESULT_MAX_BYTES = 16 * 1024 +RUNTIME_AGENT_RESULT_MAX_DEPTH = 64 +PIPELINE_MIGRATION_VERSIONS = ( + '20260727_01_pipeline_capacity_bundles', + '20260727_02_projection_queue', + '20260727_03_keycheck_queue', + '20260727_04_normalized_compat', + '20260727_05_provider_canonical_keycheck_identity', + '20260727_06_second_audit_high_fixes', + '20260727_07_final_high_blockers', + '20260727_08_admission_intent_and_capacity_closure', + '20260727_09_serialized_recovery_and_bounded_retirement', + '20260729_10_keycheck_projection_single_writer', + '20260828_11_exact_git_scan_plans', + '20260831_12_projection_findings_index', + '20260901_13_docker_layer_content_scanning', + '20260902_14_adaptive_docker_payload_scanning', + '20260906_15_cold_policy_stale_backlog', + '20260907_16_docker_quarantine_rescan', + '20260909_17_dockerhub_discovery_retry', + DOCKER_DEPTH_EXPERIMENT_MIGRATION, + '20260910_19_docker_retry_provenance', + '20260910_20_docker_depth_resolver', + '20260910_21_docker_finding_layer_attribution', + DOCKER_DEPTH_ROLLOUT_AUTHORITY_MIGRATION, + DOCKER_DEPTH_SCHEMA_SELECTOR_AUTHORITY_MIGRATION, + DOCKER_DEPTH_SCARCITY_COHORT_MIGRATION, + DOCKER_DEPTH_HOLD_QUERY_INDEX_MIGRATION, + DOCKER_DEPTH_RESOLVER_REFUND_MIGRATION, + DOCKER_DEPTH_RESOLVER_DISPOSITION_MIGRATION, + REMOTE_WORKER_MIGRATION, + OPERATIONS_CONTROL_MIGRATION, + WORKER_OBSERVABILITY_MIGRATION, + DIAGNOSTIC_PROJECTION_AUTHORITY_MIGRATION, + ASSIGNMENT_UPLOAD_TIMEOUT_MIGRATION, + REMOTE_ASSIGNMENT_CAPACITY_MIGRATION, +) +PIPELINE_QUARANTINE_REVIEW_STATUSES = frozenset(( + 'pending', 'approved_retry', 'approved_rescan', 'discarded', 'resolved', +)) +PIPELINE_QUARANTINE_LEGACY_REVIEW_STATUSES = frozenset(( + 'pending', 'approved_retry', 'discarded', 'resolved', +)) +FINAL_CUTOVER_MARKER = 'postgres-normalized-v2-authority' +PIPELINE_REQUIRED_COLUMNS = { + 'runtime_schema_migrations': {'version', 'applied_at', 'code_sha256'}, + 'runtime_final_cutover': {'id', 'marker', 'checked_at', 'evidence_sha256'}, + 'runtime_operations': { + 'operation_id', 'actor', 'action', 'target_kind', 'target_ref', 'status', + 'safe_category', 'safe_detail', 'expected_revision', 'resulting_revision', + 'expected_identity_json', 'resulting_identity_json', 'agent_state', + 'agent_result_sha256', 'requested_at', 'started_at', 'completed_at', + 'agent_reconciled_at', 'updated_at', + }, + 'runtime_operations_control': { + 'id', 'revision', 'discovery_paused', 'dispatch_paused', 'drain_state', + 'actor', 'operation_id', 'created_at', 'updated_at', + }, + 'runtime_audit_events': { + 'id', 'operation_id', 'actor', 'action', 'target_kind', 'target_ref', + 'result', 'safe_category', 'before_identity_json', 'after_identity_json', + 'before_bytes', 'after_bytes', 'previous_event_id', + 'previous_event_sha256', 'event_sha256', 'created_at', + }, + 'pipeline_capacity': { + 'id', 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'keycheck_items', 'keycheck_bytes', 'quarantine_items', 'quarantine_bytes', 'updated_at', + }, + 'result_reservations': { + 'id', 'reservation_token', 'bundle_id', 'scan_event_id', 'queue_id', 'state', + 'declared_bundle_bytes', 'reserved_bundle_bytes', 'reserved_projection_bytes', + 'reserved_candidate_items', + 'reserved_candidate_bytes', 'ready_relative_path', 'producer_pid', + 'producer_creation_time', 'producer_executable', 'bundle_credit_released', + 'cleanup_attempts', 'cleanup_available_after', 'git_scan_plan_json', + 'git_scan_plan_sha256', 'docker_layer_plan_json', 'docker_layer_plan_sha256', + 'assignment_kind', 'remote_user_id', 'remote_device_id', 'remote_issued_at', + 'remote_expires_at', 'remote_effective_config_sha256', + 'remote_result_upload_body_timeout_seconds', + 'remote_client_compat_sha256', 'remote_resolution_kind', + 'remote_execution_snapshot_json', 'remote_execution_snapshot_sha256', + 'remote_payload_sha256', 'remote_receipt_id', 'remote_resolution_json', + 'remote_resolved_at', 'remote_diagnostic_projection_version', + 'remote_diagnostic_count', 'remote_diagnostic_uids_sha256', + }, + 'worker_progress_events': { + 'id', 'reservation_id', 'remote_device_id', 'schema_version', 'sequence', + 'event_type', 'phase', 'event_timestamp', 'phase_started_at', 'instance_id', + 'slot_id', 'source', 'event_json', 'event_sha256', 'received_at', + }, + 'worker_diagnostics': { + 'id', 'diagnostic_uid', 'reservation_id', 'target_scan_id', 'schema_version', + 'scan_event_id', 'slot_id', 'attempt', 'source', 'phase', 'kind', 'category', + 'code', 'summary', 'retryable', 'occurred_at', 'captured_at', 'envelope_json', + 'envelope_sha256', 'body_payload_json', 'log_payload_json', 'received_at', + }, + 'admission_intents': { + 'reservation_token', 'intent_sha256', 'state', 'reservation_id', + 'remote_user_id', 'remote_device_id', 'created_at', 'updated_at', + }, + 'remote_worker_users': { + 'id', 'user_key', 'active_assignment_cap', 'disabled_at', 'created_at', 'updated_at', + }, + 'remote_worker_devices': { + 'id', 'user_id', 'device_key', 'token_sha256', 'revoked_at', + 'last_contact_at', 'created_at', 'updated_at', + }, + 'admission_intent_retirement': { + 'id', 'retired_count', 'chain_sha256', 'cursor_token', 'updated_at', + }, + 'pipeline_artifact_retirement': { + 'id', 'retired_count', 'chain_sha256', 'cursor_id', 'updated_at', + }, + 'result_bundles': { + 'reservation_id', 'bundle_id', 'scan_event_id', 'scan_event_hash', 'relative_path', + 'actual_bytes', 'state', 'ingest_lease_generation', 'ingest_lease_token', + }, + 'pipeline_leases': {'worker_name', 'generation', 'lease_token', 'state', 'lease_expires_at'}, + 'scan_result_compat': {'target_scan_id', 'metadata_json', 'metadata_sha256', 'metadata_bytes'}, + 'finding_compat_payloads': {'finding_id', 'payload_sha256', 'payload_bytes', 'payload_omitted'}, + 'projection_jobs': {'id', 'job_kind', 'event_id', 'event_hash', 'status', 'lease_token'}, + 'projection_streams': {'stream_name', 'base_relative_path', 'current_generation', 'rotation_bytes'}, + 'projection_cursors': {'stream_name', 'generation', 'committed_offset', 'last_job_id'}, + 'projection_appends': {'id', 'job_id', 'stream_name', 'byte_offset', 'byte_length', 'state'}, + 'projection_append_audit': { + 'id', 'append_id', 'job_id', 'stream_name', 'state', 'reason_code', + }, + 'projection_rotations': {'id', 'stream_name', 'from_generation', 'to_generation', 'state'}, + 'keycheck_credentials': { + 'id', 'service', 'credential_hash', 'provider_key_hash', 'candidate_kind', + }, + 'keycheck_candidates': { + 'id', 'candidate_uid', 'credential_id', 'service', 'routed_service', + 'secret_hash', 'state', + 'lease_token', 'result_projection_reserved_bytes', + 'result_projection_credit_transferred', + }, + 'keycheck_current_state': {'credential_id', 'service', 'status', 'last_result_id', 'state_version'}, + 'pipeline_quarantine': { + 'id', 'subsystem', 'object_type', 'reason_code', 'review_status', + 'capacity_credit_applied', 'capacity_items', 'capacity_bytes', + }, + 'pipeline_artifacts': { + 'id', 'subsystem', 'artifact_kind', 'owner_id', 'owner_key', + 'relative_path', 'payload_sha256', 'byte_count', 'state', 'updated_at', + 'cleanup_attempts', 'cleanup_available_after', + }, + 'janitor_cursors': {'layout_name', 'last_name', 'wrap_count', 'updated_at'}, + 'keycheck_recheck_cursors': { + 'cursor_key', 'service', 'status_scope', 'last_credential_id', + 'wrap_count', 'updated_at', + }, + 'docker_content_blobs': { + 'digest', 'coverage_policy_sha256', 'descriptor_kind', 'declared_bytes', 'media_type', 'state', + 'attempts', 'max_attempts', 'available_after', 'lease_reservation_id', + 'lease_token', 'lease_plan_sha256', 'lease_expires_at', + 'covered_reservation_id', 'covered_scan_event_id', 'covered_policy_sha256', + 'verified_bytes', 'covered_at', 'last_error_code', 'last_error_detail', + 'created_at', 'updated_at', + }, + 'docker_image_blob_coverage': { + 'queue_id', 'manifest_digest', 'position', 'blob_digest', 'coverage_policy_sha256', 'descriptor_kind', + 'selection_policy_sha256', 'plan_sha256', 'reservation_id', 'selected', 'selection_reason', + 'coverage_state', 'covered_at', 'last_error_code', 'created_at', 'updated_at', + }, + 'docker_adaptive_shadow_reports': { + 'id', 'report_token', 'evaluator_version', 'state', 'scan_policy_sha256', + 'execution_policy_sha256', 'selection_policy_sha256', 'cohort_size', + 'completed_pairs', 'full_routed_count', 'adaptive_routed_count', + 'routed_intersection_count', 'full_detector_count', 'adaptive_detector_count', + 'detector_intersection_count', 'full_slot_ms', 'adaptive_slot_ms', + 'omitted_descriptor_count', 'failure_count', 'privacy_violation_count', + 'safety_regression_count', 'selection_metrics_json', + 'sink_checkpoint_count', 'routed_recall_ppm', 'slot_ratio_ppm', + 'recall_threshold_ppm', 'slot_threshold_ppm', 'passed', 'lease_owner', + 'lease_token', 'lease_expires_at', 'started_at', 'completed_at', + 'created_at', 'updated_at', + }, + 'discovery_retry_queue': { + 'id', 'work_key', 'source', 'query', 'source_cycle_id', 'policy_sha256', 'pass_kind', + 'work_kind', 'page_start', 'page_end', 'next_page', 'status', 'attempts', + 'available_after', 'lease_owner', 'lease_token', 'leased_at', + 'lease_expires_at', 'last_error_category', 'held_at', 'created_at', + 'updated_at', + }, + 'target_queue_policy_events': { + 'id', 'queue_id', 'action', 'prior_status', 'next_status', 'source', + 'platform', 'query', 'reason_code', 'config_sha256', 'policy_sha256', + 'manifest_sha256', 'review_audit_sha256', 'reverses_event_id', + 'experiment_id', 'prior_updated_at', 'created_at', + }, +} +REDACTED = '***REDACTED***' +SECRET_KEY_PARTS = ( + 'token', + 'secret', + 'password', + 'authorization', + 'credential', + 'apikey', + 'api_key', + 'private_key', +) +DATABASE_SECRET_KEY_PARTS = ( + 'database_url', + 'dashboard_db_url', + 'db_url', + 'dsn', + 'connection_string', + 'postgres_url', +) +ENDPOINT_METADATA_MAX_CHARS = 2048 +POSTMAN_RESOURCE_MAX_CHARS = 4096 +_ENDPOINT_DNS_LABEL_RE = re.compile(r'^[A-Za-z0-9](?:[A-Za-z0-9-]{0,61}[A-Za-z0-9])?$') + + +def _validated_endpoint_host(value, require_domain=False): + host = str(value or '').strip().rstrip('.') + if not host or len(host) > 253 or not host.isascii(): + return '' + try: + return ipaddress.ip_address(host).compressed.lower() + except ValueError: + labels = host.split('.') + if any(not _ENDPOINT_DNS_LABEL_RE.fullmatch(label) for label in labels): + return '' + if require_domain and len(labels) < 2 and host.lower() != 'localhost': + return '' + if len(labels) > 1 and not ( + re.fullmatch(r'[A-Za-z]{2,63}', labels[-1]) + or re.fullmatch(r'xn--[A-Za-z0-9-]{1,59}', labels[-1], re.IGNORECASE) + ): + return '' + return host.lower() + + +def _validated_endpoint_port(parsed): + authority = parsed.netloc.rsplit('@', 1)[-1] + try: + port = parsed.port + except ValueError: + return None, False + if authority.startswith('['): + close = authority.find(']') + suffix = authority[close + 1:] if close >= 0 else authority + if close < 0 or (suffix and not re.fullmatch(r':[0-9]+', suffix)): + return None, False + elif ':' in authority: + if authority.count(':') != 1 or not re.fullmatch(r'[0-9]+', authority.rsplit(':', 1)[1]): + return None, False + if port is not None and not 1 <= port <= 65535: + return None, False + return port, True + + +def _safe_endpoint_path(value): + path = str(value or '') + if not path: + return '' + if not path.startswith('/') or any( + character.isspace() or ord(character) < 32 or ord(character) == 127 or character in '\\?#' + for character in path + ): + return '' + return path + + +def _format_endpoint(scheme, host, port, path): + authority = f'[{host}]' if ':' in host else host + if port is not None: + authority += f':{port}' + prefix = f'{scheme}://{authority}' if scheme else authority + remaining = max(0, ENDPOINT_METADATA_MAX_CHARS - len(prefix)) + return prefix + path[:remaining] + + +def sanitize_endpoint(value): + """Return bounded endpoint metadata without URL credentials or parameters.""" + try: + text = str(value or '').strip() + except Exception: + return '' + if not text or len(text) > 65536 or any(ord(character) < 32 or ord(character) == 127 for character in text): + return '' + + scheme_match = re.match(r'^([A-Za-z][A-Za-z0-9+.-]*):\/\/', text) + if scheme_match: + scheme = scheme_match.group(1).lower() + if scheme not in ('http', 'https'): + return '' + try: + parsed = urlsplit(text) + host = _validated_endpoint_host(parsed.hostname) + port, valid_port = _validated_endpoint_port(parsed) + except (TypeError, ValueError): + return '' + if parsed.scheme.lower() != scheme or not parsed.netloc or not host or not valid_port: + return '' + return _format_endpoint(scheme, host, port, _safe_endpoint_path(parsed.path)) + + if re.match(r'(?i)^https?:', text) or '://' in text: + return '' + base = re.split(r'[?#]', text, maxsplit=1)[0] + if not base or base.startswith('/') and not base.startswith('//'): + return '' + try: + parsed = urlsplit(base if base.startswith('//') else '//' + base) + host = _validated_endpoint_host(parsed.hostname, require_domain=True) + port, valid_port = _validated_endpoint_port(parsed) + except (TypeError, ValueError): + return '' + if not parsed.netloc or not host or not valid_port: + return '' + return _format_endpoint('', host, port, _safe_endpoint_path(parsed.path)) + + +def sanitize_endpoint_host(value): + endpoint = sanitize_endpoint(value) + if not endpoint: + return '' + try: + parsed = urlsplit(endpoint if re.match(r'(?i)^https?://', endpoint) else '//' + endpoint) + return _validated_endpoint_host(parsed.hostname) + except (TypeError, ValueError): + return '' + + +def _bounded_postman_label(value, max_chars): + text = str(value or '') + if any(ord(character) < 32 or ord(character) == 127 for character in text): + return '' + if re.search(r'(?i)https?://', text) or re.search( + r'(?i)(?:^|[?&])(?:api[_-]?key|token|password|secret|credential)=', text, + ): + return '' + return text[:max_chars] + + +def _sanitize_postman_resource(value): + text = str(value or '').strip() + if not text or any(ord(character) < 32 or ord(character) == 127 for character in text): + return '' + if '://' in text or text.startswith('//') or any(marker in text for marker in ('@', '?', '#')): + return sanitize_endpoint(text) + return text[:POSTMAN_RESOURCE_MAX_CHARS] + + +def sanitize_postman_context(context): + if not isinstance(context, dict): + return {} + safe = {} + label_limits = { + 'provider': 128, + 'credential_kind': 128, + 'credential_confidence': 128, + 'context_location': 128, + 'variable_name': 512, + 'auth_type': 128, + } + for key, limit in label_limits.items(): + if key in context: + safe[key] = _bounded_postman_label(context.get(key), limit) + if 'json_path' in context: + safe['json_path'] = _sanitize_postman_resource(context.get('json_path')) + if 'placeholder' in context: + safe['placeholder'] = bool(context.get('placeholder')) + + endpoint = sanitize_endpoint(context.get('endpoint')) + host = sanitize_endpoint_host(endpoint) or sanitize_endpoint_host(context.get('host')) + if 'endpoint' in context or 'host' in context: + safe['endpoint'] = endpoint + safe['host'] = host + return safe + + +def sanitize_postman_finding(finding): + if not isinstance(finding, dict) or not isinstance(finding.get('PostmanContext'), dict): + return finding + safe = dict(finding) + safe['PostmanContext'] = sanitize_postman_context(finding['PostmanContext']) + return safe + +RUNS_COLUMN_SPECS = { + 'id': ('id', True), 'started_at': ('text', True), 'ended_at': ('text', False), + 'duration_sec': ('real', False), 'status': ('text', True), + 'invocation_mode': ('text', False), 'command_line': ('text', False), + 'argv_json': ('text', False), 'selected_source': ('text', False), + 'selected_platform': ('text', False), 'config_path': ('text', False), + 'config_hash': ('text', False), 'enabled_sources_json': ('text', False), + 'db_path': ('text', False), 'total_fetched': ('integer', False), + 'total_queued_new': ('integer', False), 'total_scan_requested': ('integer', False), + 'total_scanned': ('integer', False), 'total_clean': ('integer', False), + 'total_found': ('integer', False), 'total_skipped': ('integer', False), + 'total_errors': ('integer', False), 'total_findings': ('integer', False), + 'total_verified_findings': ('integer', False), 'total_unique_secrets': ('integer', False), + 'total_unique_findings': ('integer', False), 'error': ('text', False), + 'total_staged': ('integer', True), 'total_quarantined': ('integer', True), + 'created_at': ('text', True), 'updated_at': ('text', True), +} + +SOURCE_CYCLES_COLUMN_SPECS = { + 'id': ('id', True), 'run_id': ('id_ref', False), 'source': ('text', False), + 'platform': ('text', False), 'mode': ('text', False), 'query': ('text', False), + 'query_index': ('integer', False), 'query_count': ('integer', False), + 'auth_name': ('text', False), 'started_at': ('text', True), 'ended_at': ('text', False), + 'duration_sec': ('real', False), 'status': ('text', True), 'message': ('text', False), + 'config_json': ('text', False), 'queue_todo_before': ('integer', False), + 'queue_checked_before': ('integer', False), 'queue_todo_after': ('integer', False), + 'queue_checked_after': ('integer', False), 'fetched_count': ('integer', False), + 'queued_new_count': ('integer', False), 'queued_updated_count': ('integer', True), + 'scan_requested_count': ('integer', False), + 'scanned_count': ('integer', False), 'clean_count': ('integer', False), + 'found_count': ('integer', False), 'skipped_count': ('integer', False), + 'error_count': ('integer', False), 'findings_count': ('integer', False), + 'verified_findings_count': ('integer', False), 'unique_secrets_count': ('integer', False), + 'unique_findings_count': ('integer', False), 'targets_per_hour': ('real', False), + 'hit_rate': ('real', False), 'verified_hit_rate': ('real', False), + 'error_rate': ('real', False), 'created_at': ('text', True), 'updated_at': ('text', True), + 'staged_count': ('integer', True), 'ingested_count': ('integer', True), + 'quarantined_count': ('integer', True), +} + +ERRORS_COLUMN_SPECS = { + 'id': ('id', True), 'run_id': ('id_ref', False), 'cycle_id': ('id_ref', False), + 'target_scan_id': ('id_ref', False), 'source': ('text', False), 'query': ('text', False), + 'target': ('text', False), 'normalized_target': ('text', False), + 'category': ('text', False), 'summary': ('text', False), 'raw_error': ('text', False), + 'created_at': ('text', True), +} + +QUEUE_SNAPSHOTS_COLUMN_SPECS = { + 'id': ('id', True), 'run_id': ('id_ref', False), 'cycle_id': ('id_ref', False), + 'source': ('text', False), 'phase': ('text', False), 'todo_count': ('integer', False), + 'checked_count': ('integer', False), 'todo_file': ('text', False), + 'checked_file': ('text', False), 'captured_at': ('text', True), +} + +CONFIG_SNAPSHOTS_COLUMN_SPECS = { + 'id': ('id', True), 'run_id': ('id_ref', False), 'cycle_id': ('id_ref', False), + 'scope': ('text', False), 'source': ('text', False), 'config_json': ('text', False), + 'captured_at': ('text', True), +} + +PACKAGE_REPO_CANDIDATES_COLUMN_SPECS = { + 'id': ('id', True), 'package_source': ('text', True), 'package_name': ('text', True), + 'package_version': ('text', True), 'query': ('text', False), 'repo_url': ('text', True), + 'provider': ('text', False), 'evidence_json': ('text', False), + 'confidence': ('text', False), 'first_seen_at': ('text', True), + 'last_seen_at': ('text', True), 'last_run_id': ('id_ref', False), + 'last_cycle_id': ('id_ref', False), +} + +KEYCHECK_COLUMN_SPECS = { + 'id': ('id', True), + 'service': ('text', True), + 'status': ('text', True), + 'status_group': ('text', True), + 'checked_at': ('text', True), + 'key_hash': ('text', False), + 'secret_hash': ('text', False), + 'key_masked': ('text', False), + 'finding_id': ('id_ref', False), + 'target_scan_id': ('id_ref', False), + 'cycle_id': ('id_ref', False), + 'run_id': ('id_ref', False), + 'source': ('text', False), + 'query': ('text', False), + 'target': ('text', False), + 'detector_name': ('text', False), + 'found_at': ('text', False), + 'message': ('text', False), + 'metadata_json': ('text', False), + 'source_line': ('text', False), + 'detector_secret_hash': ('text', False), + 'event_id': ('text', False), + 'finding_uid': ('text', False), + 'link_status': ('text', False), + 'link_attempts': ('integer', False), + 'linked_at': ('text', False), + 'link_error': ('text', False), + 'created_at': ('text', True), + 'candidate_id': ('id_ref', False), + 'credential_id': ('id_ref', False), + 'result_source': ('text', True), +} + +OUTBOX_COLUMN_SPECS = { + 'id': ('id', True), + 'target_scan_id': ('id_ref', True), + 'payload_json': ('text', True), + 'status': ('text', True), + 'attempts': ('integer', False), + 'last_error': ('text', False), + 'lease_owner': ('text', False), + 'lease_expires_at': ('text', False), + 'available_after': ('text', False), + 'created_at': ('text', True), + 'delivered_at': ('text', False), + 'updated_at': ('text', True), +} + +CURSOR_COLUMN_SPECS = { + 'source_file': ('text', True), + 'file_identity': ('text', True), + 'file_size': ('id_ref', True), + 'file_mtime_ns': ('id_ref', True), + 'source': ('text', True), + 'platform': ('text', True), + 'byte_offset': ('id_ref', True), + 'line_number': ('id_ref', True), + 'discarding_oversized': ('integer', True), + 'oversized_line_start': ('id_ref', False), + 'cumulative_rows': ('id_ref', True), + 'cumulative_bytes': ('id_ref', True), + 'cumulative_inserted': ('id_ref', True), + 'cumulative_rejected': ('id_ref', True), + 'completed_at': ('text', False), + 'last_report_json': ('text', False), + 'updated_at': ('text', True), +} + +RECONCILIATION_ISSUE_COLUMN_SPECS = { + 'id': ('id', True), + 'source_file': ('text', True), + 'file_identity': ('text', True), + 'source': ('text', True), + 'platform': ('text', True), + 'line_number': ('id_ref', True), + 'byte_offset': ('id_ref', True), + 'reason': ('text', True), + 'target_preview': ('text', False), + 'created_at': ('text', True), + 'resolved_at': ('text', False), +} + +TARGET_QUEUE_COLUMN_SPECS = { + 'id': ('id', True), 'source': ('text', True), 'platform': ('text', True), + 'query': ('text', False), 'target': ('text', True), 'normalized_target': ('text', True), + 'status': ('text', True), 'attempts': ('integer', False), + 'lease_owner': ('text', False), 'lease_token': ('text', False), + 'claim_batch': ('text', False), 'leased_at': ('text', False), + 'lease_expires_at': ('text', False), 'available_after': ('text', False), + 'target_scan_id': ('id_ref', False), 'last_error': ('text', False), + 'created_at': ('text', True), 'updated_at': ('text', True), 'completed_at': ('text', False), + 'resolver_state': ('text', False), 'resolver_due_at': ('text', False), + 'resolver_attempts': ('integer', True), 'resolver_token': ('text', False), + 'current_result_reservation_id': ('id_ref', False), 'claim_event_id': ('text', False), + 'remote_modified_at': ('text', False), 'scan_remote_modified_at': ('text', False), + 'covered_ref': ('text', False), 'covered_head': ('text', False), +} + +DISCOVERY_RETRY_QUEUE_COLUMN_SPECS = { + 'id': ('id', True), + 'work_key': ('text', True), + 'source': ('text', True), + 'query': ('text', True), + 'source_cycle_id': ('id_ref', False), + 'policy_sha256': ('text', True), + 'pass_kind': ('text', True), + 'work_kind': ('text', True), + 'page_start': ('integer', True), + 'page_end': ('integer', True), + 'next_page': ('integer', True), + 'status': ('text', True), + 'attempts': ('integer', True), + 'available_after': ('text', False), + 'lease_owner': ('text', False), + 'lease_token': ('text', False), + 'leased_at': ('text', False), + 'lease_expires_at': ('text', False), + 'last_error_category': ('text', False), + 'held_at': ('text', False), + 'created_at': ('text', True), + 'updated_at': ('text', True), +} + +TARGET_QUEUE_POLICY_EVENT_COLUMN_SPECS = { + 'id': ('id', True), 'queue_id': ('id_ref', True), + 'action': ('text', True), 'prior_status': ('text', True), + 'next_status': ('text', True), 'source': ('text', True), + 'platform': ('text', True), 'query': ('text', True), + 'reason_code': ('text', True), 'config_sha256': ('text', True), + 'policy_sha256': ('text', True), 'manifest_sha256': ('text', True), + 'review_audit_sha256': ('text', True), + 'reverses_event_id': ('id_ref', False), + 'experiment_id': ('id_ref', False), + 'prior_updated_at': ('text', True), 'created_at': ('text', True), +} + +DOCKER_DEPTH_EXPERIMENT_COLUMN_SPECS = { + 'id': ('id', True), 'experiment_key': ('text', True), + 'source': ('text', True), 'state': ('text', True), + 'collection_generation': ('text', True), + 'config_sha256': ('text', True), 'ordered_queries_sha256': ('text', True), + 'selector_version': ('text', True), 'selector_sha256': ('text', True), + 'provenance_policy_sha256': ('text', True), + 'query_count': ('integer', True), 'repositories_per_query': ('integer', True), + 'images_per_repository': ('integer', True), 'target_limit': ('integer', True), + 'target_count': ('integer', True), 'selection_count': ('integer', True), + 'fence_generation': ('id_ref', True), 'fence_owner': ('text', False), + 'fence_token': ('text', False), 'fence_expires_at': ('text', False), + 'hold_reason_code': ('text', False), 'plan_sha256': ('text', False), + 'selection_sha256': ('text', False), + 'hold_manifest_sha256': ('text', False), 'created_at': ('text', True), + 'updated_at': ('text', True), 'planned_at': ('text', False), + 'activated_at': ('text', False), 'draining_at': ('text', False), + 'completed_at': ('text', False), 'released_at': ('text', False), + 'held_at': ('text', False), +} + +DOCKER_DEPTH_EXPERIMENT_QUERY_COLUMN_SPECS = { + 'id': ('id', True), 'experiment_id': ('id_ref', True), + 'source': ('text', True), 'query_ordinal': ('integer', True), + 'query': ('text', True), 'query_sha256': ('text', True), + 'required_repository_count': ('integer', True), + 'selected_repository_count': ('integer', True), 'created_at': ('text', True), +} + +DOCKER_DISCOVERY_PASS_COLUMN_SPECS = { + 'id': ('id', True), 'experiment_id': ('id_ref', False), + 'pass_token': ('text', True), 'source': ('text', True), + 'pass_kind': ('text', True), 'policy_sha256': ('text', True), + 'collection_generation': ('text', True), + 'ordered_queries_sha256': ('text', True), 'expected_query_count': ('integer', True), + 'completed_query_count': ('integer', True), 'state': ('text', True), + 'started_at': ('text', True), 'completed_at': ('text', False), + 'created_at': ('text', True), 'updated_at': ('text', True), +} + +DOCKER_DISCOVERY_PAGE_COLUMN_SPECS = { + 'id': ('id', True), 'pass_id': ('id_ref', True), + 'source_cycle_id': ('id_ref', False), 'retry_work_id': ('id_ref', False), + 'query': ('text', True), 'query_ordinal': ('integer', True), + 'page_number': ('integer', True), 'result_count': ('integer', True), + 'total_count': ('id_ref', False), + 'admitted_count': ('integer', True), 'query_complete': ('integer', True), + 'admission_kind': ('text', True), 'page_sha256': ('text', True), + 'observed_at': ('text', True), 'created_at': ('text', True), +} + +DOCKER_REPOSITORY_QUERY_PROVENANCE_COLUMN_SPECS = { + 'source': ('text', True), 'query': ('text', True), + 'repository_queue_id': ('id_ref', True), 'provenance_kind': ('text', True), + 'first_observed_at': ('text', True), 'last_observed_at': ('text', True), + 'first_search_rank': ('integer', False), 'best_search_rank': ('integer', False), + 'last_search_rank': ('integer', False), 'first_cycle_id': ('id_ref', False), + 'last_cycle_id': ('id_ref', False), 'first_page_id': ('id_ref', False), + 'last_page_id': ('id_ref', False), 'first_policy_sha256': ('text', False), + 'last_policy_sha256': ('text', False), 'observation_count': ('integer', True), + 'fresh_observation_count': ('integer', True), + 'fresh_complete_observation_count': ('integer', True), + 'fresh_coverage_eligible': ('integer', True), + 'created_at': ('text', True), 'updated_at': ('text', True), +} + +DOCKER_REPOSITORY_QUERY_OBSERVATION_COLUMN_SPECS = { + 'page_id': ('id_ref', True), 'repository_queue_id': ('id_ref', True), + 'source': ('text', True), 'query': ('text', True), + 'search_rank': ('integer', True), 'observed_at': ('text', True), +} + +DOCKER_IMAGE_MANIFEST_COLUMN_SPECS = { + 'id': ('id', True), 'target_queue_id': ('id_ref', True), + 'source': ('text', True), 'repository': ('text', True), + 'manifest_digest': ('text', True), 'manifest_media_type': ('text', True), + 'config_digest': ('text', False), 'graph_sha256': ('text', True), + 'manifest_size_bytes': ('id_ref', False), 'layer_count': ('integer', True), + 'resolved_at': ('text', True), 'created_at': ('text', True), +} + +DOCKER_MANIFEST_LAYER_COLUMN_SPECS = { + 'id': ('id', True), 'manifest_id': ('id_ref', True), + 'position_from_base': ('integer', True), 'position_from_top': ('integer', True), + 'layer_digest': ('text', True), 'media_type': ('text', True), + 'layer_size_bytes': ('id_ref', True), 'descriptor_sha256': ('text', True), + 'created_at': ('text', True), +} + +DOCKER_DEPTH_EXPERIMENT_REPOSITORY_COLUMN_SPECS = { + 'id': ('id', True), 'experiment_id': ('id_ref', True), + 'query_ordinal': ('integer', True), 'source': ('text', True), + 'query': ('text', True), 'repository_queue_id': ('id_ref', True), + 'eligibility_page_id': ('id_ref', True), 'repository_rank': ('integer', True), + 'planned_is_deep_probe': ('integer', True), + 'is_deep_probe': ('integer', True), 'work_state': ('text', True), + 'resolver_generation': ('id_ref', True), 'resolver_owner': ('text', False), + 'resolver_token': ('text', False), 'resolver_expires_at': ('text', False), + 'resolver_attempts': ('integer', True), 'resolver_due_at': ('text', False), + 'candidate_distinct_graph_count': ('integer', True), + 'selected_image_count': ('integer', True), + 'replacement_repository_queue_id': ('id_ref', False), + 'replacement_eligibility_page_id': ('id_ref', False), + 'replacement_count': ('integer', True), + 'replacement_evidence_sha256': ('text', False), + 'last_error_code': ('text', False), 'created_at': ('text', True), + 'updated_at': ('text', True), 'resolved_at': ('text', False), +} + +DOCKER_DEPTH_EXPERIMENT_TARGET_COLUMN_SPECS = { + 'id': ('id', True), 'experiment_id': ('id_ref', True), + 'target_queue_id': ('id_ref', True), 'manifest_id': ('id_ref', True), + 'counter_ordinal': ('integer', True), 'state': ('text', True), + 'dispatch_wave': ('integer', True), 'dispatch_order': ('id_ref', True), + 'reservation_count': ('integer', True), 'created_at': ('text', True), + 'updated_at': ('text', True), 'terminal_at': ('text', False), +} + +DOCKER_DEPTH_EXPERIMENT_SELECTION_COLUMN_SPECS = { + 'id': ('id', True), 'experiment_id': ('id_ref', True), + 'query_ordinal': ('integer', True), 'experiment_repository_id': ('id_ref', True), + 'experiment_target_id': ('id_ref', True), 'image_rank': ('integer', True), + 'selection_reason': ('text', True), 'selection_evidence_sha256': ('text', True), + 'graph_sha256': ('text', True), 'selected_at': ('text', True), + 'created_at': ('text', True), +} + +DOCKER_DEPTH_EXPERIMENT_SKIP_COLUMN_SPECS = { + 'id': ('id', True), 'experiment_id': ('id_ref', True), + 'experiment_repository_id': ('id_ref', True), + 'repository_queue_id': ('id_ref', True), 'candidate_kind': ('text', True), + 'candidate_ordinal': ('integer', True), 'reason_code': ('text', True), + 'candidate_identity_sha256': ('text', True), + 'evidence_sha256': ('text', True), 'created_at': ('text', True), +} + +DOCKER_DEPTH_RESOLVER_REFUND_COLUMN_SPECS = { + 'id': ('id', True), 'experiment_id': ('id_ref', True), + 'experiment_repository_id': ('id_ref', True), + 'repository_queue_id': ('id_ref', True), + 'query_ordinal': ('integer', True), 'repository_rank': ('integer', True), + 'recovery_kind': ('text', True), 'manifest_sha256': ('text', True), + 'entry_evidence_sha256': ('text', True), 'log_sha256': ('text', True), + 'target_identity_sha256': ('text', True), + 'prior_error_code_sha256': ('text', True), + 'prior_work_state': ('text', True), 'next_work_state': ('text', True), + 'prior_resolver_attempts': ('integer', True), + 'refund_attempts': ('integer', True), + 'next_resolver_attempts': ('integer', True), + 'confirmed_bug_event_count': ('integer', True), + 'applied_at': ('text', True), 'created_at': ('text', True), +} + +DOCKER_DEPTH_RESOLVER_DISPOSITION_COLUMN_SPECS = { + 'id': ('id', True), 'experiment_id': ('id_ref', True), + 'experiment_repository_id': ('id_ref', True), + 'prior_repository_queue_id': ('id_ref', True), + 'replacement_repository_queue_id': ('id_ref', False), + 'replacement_eligibility_page_id': ('id_ref', False), + 'replacement_best_search_rank': ('integer', False), + 'query_ordinal': ('integer', True), 'repository_rank': ('integer', True), + 'disposition_kind': ('text', True), 'outcome': ('text', True), + 'manifest_sha256': ('text', True), 'entry_evidence_sha256': ('text', True), + 'candidate_snapshot_sha256': ('text', True), + 'prior_target_identity_sha256': ('text', True), + 'replacement_target_identity_sha256': ('text', False), + 'prior_error_code_sha256': ('text', True), + 'prior_work_state': ('text', True), 'next_work_state': ('text', True), + 'prior_resolver_attempts': ('integer', True), + 'next_resolver_attempts': ('integer', True), + 'applied_at': ('text', True), 'created_at': ('text', True), +} + +DOCKER_DEPTH_EXPERIMENT_SCAN_BINDING_COLUMN_SPECS = { + 'id': ('id', True), 'experiment_target_id': ('id_ref', True), + 'reservation_id': ('id_ref', True), 'target_scan_id': ('id_ref', False), + 'attempt': ('integer', True), 'state': ('text', True), + 'bound_at': ('text', True), 'scan_bound_at': ('text', False), + 'completed_at': ('text', False), 'created_at': ('text', True), +} + +DOCKER_FINDING_LAYER_ATTRIBUTION_COLUMN_SPECS = { + 'id': ('id', True), 'scan_binding_id': ('id_ref', True), + 'finding_id': ('id_ref', True), 'manifest_layer_id': ('id_ref', False), + 'attribution_state': ('text', True), 'reported_layer_digest': ('text', False), + 'unattributed_reason': ('text', False), + 'position_from_base': ('integer', False), 'position_from_top': ('integer', False), + 'created_at': ('text', True), +} + +TARGET_SCANS_COLUMN_SPECS = { + 'id': ('id', True), 'scan_event_id': ('text', False), 'scan_event_hash': ('text', False), + 'queue_id': ('id_ref', False), 'claim_lease_token': ('text', False), + 'queue_completion_applied': ('integer', True), 'queue_completion_disposition': ('text', False), + 'run_id': ('id_ref', False), 'cycle_id': ('id_ref', False), 'source': ('text', False), + 'query': ('text', False), 'target': ('text', False), 'normalized_target': ('text', False), + 'scan_type': ('text', False), 'status': ('text', False), 'started_at': ('text', False), + 'ended_at': ('text', False), 'duration_sec': ('real', False), 'scan_options_json': ('text', False), + 'package_name': ('text', False), 'package_version': ('text', False), + 'package_artifact': ('text', False), 'package_date': ('text', False), + 'package_filename': ('text', False), 'package_type': ('text', False), + 'package_size': ('integer', False), 'findings_count': ('integer', False), + 'verified_findings_count': ('integer', False), 'error_count': ('integer', False), + 'skipped_reason': ('text', False), 'first_error_summary': ('text', False), + 'raw_result_json': ('text', False), 'created_at': ('text', True), + 'result_reservation_id': ('id_ref', False), + 'compat_schema_version': ('integer', True), + 'raw_result_storage': ('text', True), +} + +DOCKER_CONTENT_BLOB_COLUMN_SPECS = { + 'digest': ('text', True), 'coverage_policy_sha256': ('text', True), + 'descriptor_kind': ('text', True), + 'declared_bytes': ('id_ref', True), 'media_type': ('text', True), + 'state': ('text', True), 'attempts': ('integer', True), + 'max_attempts': ('integer', True), 'available_after': ('text', False), + 'lease_reservation_id': ('id_ref', False), 'lease_token': ('text', False), + 'lease_plan_sha256': ('text', False), 'lease_expires_at': ('text', False), + 'covered_reservation_id': ('id_ref', False), + 'covered_scan_event_id': ('text', False), 'covered_policy_sha256': ('text', False), + 'verified_bytes': ('id_ref', False), 'covered_at': ('text', False), + 'last_error_code': ('text', False), 'last_error_detail': ('text', False), + 'created_at': ('text', True), 'updated_at': ('text', True), +} + +DOCKER_IMAGE_BLOB_COVERAGE_COLUMN_SPECS = { + 'queue_id': ('id_ref', True), 'manifest_digest': ('text', True), + 'reservation_id': ('id_ref', True), 'position': ('integer', True), + 'blob_digest': ('text', True), + 'coverage_policy_sha256': ('text', True), + 'selection_policy_sha256': ('text', True), + 'descriptor_kind': ('text', True), 'plan_sha256': ('text', True), + 'selected': ('integer', True), + 'selection_reason': ('text', True), 'coverage_state': ('text', True), + 'covered_at': ('text', False), 'last_error_code': ('text', False), + 'created_at': ('text', True), 'updated_at': ('text', True), +} + +DOCKER_ADAPTIVE_SHADOW_REPORT_COLUMN_SPECS = { + 'id': ('id', True), 'report_token': ('text', True), + 'evaluator_version': ('text', True), 'state': ('text', True), + 'scan_policy_sha256': ('text', True), 'execution_policy_sha256': ('text', True), + 'selection_policy_sha256': ('text', True), 'cohort_size': ('integer', True), + 'completed_pairs': ('integer', True), 'full_routed_count': ('id_ref', True), + 'adaptive_routed_count': ('id_ref', True), + 'routed_intersection_count': ('id_ref', True), + 'full_detector_count': ('id_ref', True), 'adaptive_detector_count': ('id_ref', True), + 'detector_intersection_count': ('id_ref', True), + 'full_slot_ms': ('id_ref', True), 'adaptive_slot_ms': ('id_ref', True), + 'omitted_descriptor_count': ('id_ref', True), 'failure_count': ('integer', True), + 'privacy_violation_count': ('integer', True), + 'safety_regression_count': ('integer', True), + 'selection_metrics_json': ('text', True), + 'sink_checkpoint_count': ('integer', True), + 'routed_recall_ppm': ('integer', True), 'slot_ratio_ppm': ('integer', True), + 'recall_threshold_ppm': ('integer', True), 'slot_threshold_ppm': ('integer', True), + 'passed': ('integer', True), 'lease_owner': ('text', True), + 'lease_token': ('text', True), 'lease_expires_at': ('text', True), + 'started_at': ('text', True), 'completed_at': ('text', False), + 'created_at': ('text', True), 'updated_at': ('text', True), +} + +FINDINGS_COLUMN_SPECS = { + 'id': ('id', True), 'run_id': ('id_ref', False), 'cycle_id': ('id_ref', False), + 'target_scan_id': ('id_ref', False), 'source': ('text', False), 'query': ('text', False), + 'target': ('text', False), 'normalized_target': ('text', False), + 'detector_name': ('text', False), 'detector_type': ('text', False), + 'verified': ('integer', False), 'raw_secret': ('text', False), + 'redacted_secret': ('text', False), 'secret_hash': ('text', False), + 'detector_secret_hash': ('text', False), 'finding_fingerprint': ('text', False), + 'finding_uid': ('text', False), 'file_path': ('text', False), 'line_number': ('text', False), + 'commit_hash': ('text', False), 'source_timestamp': ('text', False), + 'source_metadata_type': ('text', False), 'source_metadata_json': ('text', False), + 'raw_finding_json': ('text', False), 'provider': ('text', False), + 'credential_kind': ('text', False), 'credential_confidence': ('text', False), + 'required_context_missing': ('integer', False), 'principal': ('text', False), + 'username': ('text', False), 'email': ('text', False), 'project_id': ('text', False), + 'tenant_id': ('text', False), 'organization': ('text', False), 'registry': ('text', False), + 'endpoint': ('text', False), 'scope': ('text', False), 'resource': ('text', False), + 'enrichment_json': ('text', False), 'created_at': ('text', True), + 'raw_payload_sha256': ('text', False), 'raw_payload_bytes': ('id_ref', False), + 'raw_payload_omitted': ('integer', True), +} + +KEYCHECK_EVENT_MAP_COLUMN_SPECS = { + 'event_id': ('text', True), 'keycheck_result_id': ('id_ref', False), 'created_at': ('text', True), +} + +FINDING_UID_MAP_COLUMN_SPECS = { + 'finding_uid': ('text', True), 'finding_id': ('id_ref', True), 'created_at': ('text', True), +} + +RUNTIME_OPERATION_COLUMN_SPECS = { + 'operation_id': ('text', True), 'actor': ('text', True), + 'action': ('text', True), 'target_kind': ('text', True), + 'target_ref': ('text', True), 'status': ('text', True), + 'safe_category': ('text', False), 'safe_detail': ('text', False), + 'expected_revision': ('id_ref', False), 'resulting_revision': ('id_ref', False), + 'expected_identity_json': ('text', True), 'resulting_identity_json': ('text', False), + 'agent_state': ('text', True), 'agent_result_sha256': ('text', False), + 'requested_at': ('text', True), 'started_at': ('text', False), + 'completed_at': ('text', False), 'agent_reconciled_at': ('text', False), + 'updated_at': ('text', True), +} + +RUNTIME_OPERATIONS_CONTROL_COLUMN_SPECS = { + 'id': ('integer', True), 'revision': ('id_ref', True), + 'discovery_paused': ('integer', True), 'dispatch_paused': ('integer', True), + 'drain_state': ('text', True), 'actor': ('text', True), + 'operation_id': ('text', False), 'created_at': ('text', True), + 'updated_at': ('text', True), +} + +RUNTIME_AUDIT_EVENT_COLUMN_SPECS = { + 'id': ('id', True), 'operation_id': ('text', False), + 'actor': ('text', True), 'action': ('text', True), + 'target_kind': ('text', True), 'target_ref': ('text', True), + 'result': ('text', True), 'safe_category': ('text', False), + 'before_identity_json': ('text', False), 'after_identity_json': ('text', False), + 'before_bytes': ('id_ref', False), 'after_bytes': ('id_ref', False), + 'previous_event_id': ('id_ref', False), 'previous_event_sha256': ('text', False), + 'event_sha256': ('text', True), 'created_at': ('text', True), +} + +WORKER_PROGRESS_EVENT_COLUMN_SPECS = { + 'id': ('id', True), 'reservation_id': ('id_ref', True), + 'remote_device_id': ('id_ref', True), 'schema_version': ('integer', True), + 'sequence': ('integer', True), 'event_type': ('text', True), + 'phase': ('text', True), 'event_timestamp': ('text', True), + 'phase_started_at': ('text', False), 'instance_id': ('text', True), + 'slot_id': ('integer', True), 'source': ('text', True), + 'event_json': ('text', True), 'event_sha256': ('text', True), + 'received_at': ('text', True), +} + +WORKER_DIAGNOSTIC_COLUMN_SPECS = { + 'id': ('id', True), 'diagnostic_uid': ('text', True), + 'reservation_id': ('id_ref', True), 'target_scan_id': ('id_ref', False), + 'schema_version': ('integer', True), 'scan_event_id': ('text', False), + 'slot_id': ('integer', True), 'attempt': ('integer', True), + 'source': ('text', True), 'phase': ('text', True), 'kind': ('text', True), + 'category': ('text', True), 'code': ('text', True), 'summary': ('text', True), + 'retryable': ('integer', True), 'occurred_at': ('text', True), + 'captured_at': ('text', True), 'envelope_json': ('text', True), + 'envelope_sha256': ('text', True), 'body_payload_json': ('text', False), + 'log_payload_json': ('text', False), 'received_at': ('text', True), +} + +RUNTIME_TABLE_SPECS = { + 'runtime_operations': RUNTIME_OPERATION_COLUMN_SPECS, + 'runtime_operations_control': RUNTIME_OPERATIONS_CONTROL_COLUMN_SPECS, + 'runtime_audit_events': RUNTIME_AUDIT_EVENT_COLUMN_SPECS, + 'worker_progress_events': WORKER_PROGRESS_EVENT_COLUMN_SPECS, + 'worker_diagnostics': WORKER_DIAGNOSTIC_COLUMN_SPECS, + 'runs': RUNS_COLUMN_SPECS, + 'source_cycles': SOURCE_CYCLES_COLUMN_SPECS, + 'target_queue': TARGET_QUEUE_COLUMN_SPECS, + 'discovery_retry_queue': DISCOVERY_RETRY_QUEUE_COLUMN_SPECS, + 'target_queue_policy_events': TARGET_QUEUE_POLICY_EVENT_COLUMN_SPECS, + 'target_scans': TARGET_SCANS_COLUMN_SPECS, + 'findings': FINDINGS_COLUMN_SPECS, + 'errors': ERRORS_COLUMN_SPECS, + 'queue_snapshots': QUEUE_SNAPSHOTS_COLUMN_SPECS, + 'config_snapshots': CONFIG_SNAPSHOTS_COLUMN_SPECS, + 'package_repo_candidates': PACKAGE_REPO_CANDIDATES_COLUMN_SPECS, + 'scan_publication_outbox': OUTBOX_COLUMN_SPECS, + 'keycheck_results': KEYCHECK_COLUMN_SPECS, + 'keycheck_event_map': KEYCHECK_EVENT_MAP_COLUMN_SPECS, + 'finding_uid_map': FINDING_UID_MAP_COLUMN_SPECS, + 'target_queue_reconciliation_cursors': CURSOR_COLUMN_SPECS, + 'target_queue_reconciliation_issues': RECONCILIATION_ISSUE_COLUMN_SPECS, + 'docker_content_blobs': DOCKER_CONTENT_BLOB_COLUMN_SPECS, + 'docker_image_blob_coverage': DOCKER_IMAGE_BLOB_COVERAGE_COLUMN_SPECS, + 'docker_adaptive_shadow_reports': DOCKER_ADAPTIVE_SHADOW_REPORT_COLUMN_SPECS, + 'docker_depth_experiments': DOCKER_DEPTH_EXPERIMENT_COLUMN_SPECS, + 'docker_depth_experiment_queries': DOCKER_DEPTH_EXPERIMENT_QUERY_COLUMN_SPECS, + 'docker_discovery_passes': DOCKER_DISCOVERY_PASS_COLUMN_SPECS, + 'docker_discovery_pages': DOCKER_DISCOVERY_PAGE_COLUMN_SPECS, + 'docker_repository_query_provenance': DOCKER_REPOSITORY_QUERY_PROVENANCE_COLUMN_SPECS, + 'docker_repository_query_observations': DOCKER_REPOSITORY_QUERY_OBSERVATION_COLUMN_SPECS, + 'docker_image_manifests': DOCKER_IMAGE_MANIFEST_COLUMN_SPECS, + 'docker_manifest_layers': DOCKER_MANIFEST_LAYER_COLUMN_SPECS, + 'docker_depth_experiment_repositories': DOCKER_DEPTH_EXPERIMENT_REPOSITORY_COLUMN_SPECS, + 'docker_depth_experiment_targets': DOCKER_DEPTH_EXPERIMENT_TARGET_COLUMN_SPECS, + 'docker_depth_experiment_selections': DOCKER_DEPTH_EXPERIMENT_SELECTION_COLUMN_SPECS, + 'docker_depth_experiment_candidate_skips': DOCKER_DEPTH_EXPERIMENT_SKIP_COLUMN_SPECS, + 'docker_depth_resolver_attempt_refunds': DOCKER_DEPTH_RESOLVER_REFUND_COLUMN_SPECS, + 'docker_depth_resolver_dispositions': DOCKER_DEPTH_RESOLVER_DISPOSITION_COLUMN_SPECS, + 'docker_depth_experiment_scan_bindings': DOCKER_DEPTH_EXPERIMENT_SCAN_BINDING_COLUMN_SPECS, + 'docker_finding_layer_attributions': DOCKER_FINDING_LAYER_ATTRIBUTION_COLUMN_SPECS, +} + +DOCKER_DEPTH_EXPERIMENT_TABLES = ( + 'docker_depth_experiments', + 'docker_depth_experiment_queries', + 'docker_discovery_passes', + 'docker_discovery_pages', + 'docker_repository_query_provenance', + 'docker_repository_query_observations', + 'docker_image_manifests', + 'docker_manifest_layers', + 'docker_depth_experiment_repositories', + 'docker_depth_experiment_targets', + 'docker_depth_experiment_selections', + 'docker_depth_experiment_candidate_skips', + 'docker_depth_resolver_attempt_refunds', + 'docker_depth_resolver_dispositions', + 'docker_depth_experiment_scan_bindings', + 'docker_finding_layer_attributions', +) +PIPELINE_REQUIRED_COLUMNS.update({ + table: set(RUNTIME_TABLE_SPECS[table]) for table in DOCKER_DEPTH_EXPERIMENT_TABLES +}) + +DOCKER_DEPTH_EXPERIMENT_INDEX_SPECS = ( + ('docker_depth_experiments', 'uq_docker_depth_experiments_key', + ('experiment_key',), True, ''), + ('docker_depth_experiments', 'idx_docker_depth_experiments_state', + ('source', 'state', 'id'), False, ''), + ('docker_depth_experiment_queries', 'uq_docker_depth_experiment_queries_ordinal', + ('experiment_id', 'query_ordinal'), True, ''), + ('docker_depth_experiment_queries', 'uq_docker_depth_experiment_queries_query', + ('experiment_id', 'source', 'query'), True, ''), + ('docker_discovery_passes', 'uq_docker_discovery_passes_token', + ('pass_token',), True, ''), + ('docker_discovery_passes', 'idx_docker_discovery_passes_policy', + ('source', 'policy_sha256', 'state', 'id'), False, ''), + ('docker_discovery_passes', 'idx_docker_discovery_passes_experiment', + ('experiment_id', 'state', 'id'), False, ''), + ('docker_discovery_pages', 'uq_docker_discovery_pages_position', + ('pass_id', 'query_ordinal', 'page_number'), True, ''), + ('docker_discovery_pages', 'idx_docker_discovery_pages_query', + ('pass_id', 'query', 'query_complete', 'page_number'), False, ''), + ('docker_discovery_pages', 'idx_docker_discovery_pages_retry', + ('retry_work_id', 'id'), False, 'retry_work_id IS NOT NULL'), + ('docker_repository_query_provenance', 'idx_docker_repository_provenance_queue', + ('repository_queue_id', 'source', 'query'), False, ''), + ('docker_repository_query_provenance', 'idx_docker_repository_provenance_eligible', + ('source', 'query', 'fresh_coverage_eligible', 'best_search_rank', 'repository_queue_id'), + False, ''), + ('docker_repository_query_observations', 'idx_docker_repository_observations_query', + ('source', 'query', 'repository_queue_id', 'page_id'), False, ''), + ('docker_image_manifests', 'uq_docker_image_manifests_queue', + ('target_queue_id',), True, ''), + ('docker_image_manifests', 'idx_docker_image_manifests_digest', + ('manifest_digest', 'id'), False, ''), + ('docker_manifest_layers', 'uq_docker_manifest_layers_base_position', + ('manifest_id', 'position_from_base'), True, ''), + ('docker_manifest_layers', 'uq_docker_manifest_layers_top_position', + ('manifest_id', 'position_from_top'), True, ''), + ('docker_manifest_layers', 'idx_docker_manifest_layers_digest', + ('layer_digest', 'manifest_id', 'position_from_base'), False, ''), + ('docker_depth_experiment_repositories', 'uq_docker_experiment_repositories_member', + ('experiment_id', 'query_ordinal', 'repository_queue_id'), True, ''), + ('docker_depth_experiment_repositories', 'uq_docker_experiment_repositories_rank', + ('experiment_id', 'query_ordinal', 'repository_rank'), True, ''), + ('docker_depth_experiment_repositories', 'idx_docker_experiment_repository_work', + ('experiment_id', 'work_state', 'repository_rank', 'query_ordinal', 'id'), False, ''), + ('docker_depth_experiment_repositories', 'idx_docker_experiment_repository_due', + ('experiment_id', 'work_state', 'resolver_due_at', 'repository_rank', + 'query_ordinal', 'id'), False, ''), + ('docker_depth_experiment_targets', 'uq_docker_experiment_targets_queue', + ('experiment_id', 'target_queue_id'), True, ''), + ('docker_depth_experiment_targets', 'uq_docker_experiment_targets_counter', + ('experiment_id', 'counter_ordinal'), True, ''), + ('docker_depth_experiment_targets', 'idx_docker_experiment_targets_dispatch', + ('experiment_id', 'state', 'dispatch_wave', 'dispatch_order', 'id'), False, ''), + ('docker_depth_experiment_selections', 'uq_docker_experiment_selections_rank', + ('experiment_repository_id', 'image_rank'), True, ''), + ('docker_depth_experiment_selections', 'idx_docker_experiment_selections_query', + ('experiment_id', 'query_ordinal', 'image_rank', 'experiment_repository_id', 'id'), + False, ''), + ('docker_depth_experiment_selections', 'idx_docker_experiment_selections_target', + ('experiment_target_id', 'id'), False, ''), + ('docker_depth_experiment_candidate_skips', 'uq_docker_experiment_candidate_skip', + ('experiment_repository_id', 'candidate_kind', 'candidate_identity_sha256'), + True, ''), + ('docker_depth_experiment_candidate_skips', 'idx_docker_experiment_candidate_skips', + ('experiment_id', 'experiment_repository_id', 'candidate_kind', 'id'), + False, ''), + ('docker_depth_resolver_attempt_refunds', 'uq_docker_depth_resolver_attempt_refund', + ('experiment_repository_id', 'recovery_kind'), True, ''), + ('docker_depth_resolver_attempt_refunds', 'idx_docker_depth_resolver_attempt_refunds', + ('experiment_id', 'id'), False, ''), + ('docker_depth_resolver_dispositions', 'uq_docker_depth_resolver_disposition', + ('experiment_repository_id', 'disposition_kind'), True, ''), + ('docker_depth_resolver_dispositions', 'idx_docker_depth_resolver_dispositions', + ('experiment_id', 'id'), False, ''), + ('docker_depth_experiment_scan_bindings', 'uq_docker_experiment_bindings_reservation', + ('reservation_id',), True, ''), + ('docker_depth_experiment_scan_bindings', 'uq_docker_experiment_bindings_attempt', + ('experiment_target_id', 'attempt'), True, ''), + ('docker_depth_experiment_scan_bindings', 'uq_docker_experiment_bindings_scan', + ('target_scan_id',), True, 'target_scan_id IS NOT NULL'), + ('docker_depth_experiment_scan_bindings', 'idx_docker_experiment_bindings_target', + ('experiment_target_id', 'state', 'id'), False, ''), + ('docker_finding_layer_attributions', 'uq_docker_finding_layer_exact', + ('finding_id', 'manifest_layer_id'), True, "attribution_state = 'exact'"), + ('docker_finding_layer_attributions', 'uq_docker_finding_layer_unattributed', + ('finding_id',), True, "attribution_state = 'unattributed'"), + ('docker_finding_layer_attributions', 'idx_docker_finding_layer_binding', + ('scan_binding_id', 'finding_id', 'id'), False, ''), + ('target_queue_policy_events', 'idx_target_queue_policy_events_experiment', + ('experiment_id', 'id'), False, 'experiment_id IS NOT NULL'), +) + +RUNTIME_PRIMARY_KEYS = { + 'runtime_operations': ['operation_id'], + 'runtime_operations_control': ['id'], + 'runtime_audit_events': ['id'], + 'worker_progress_events': ['id'], 'worker_diagnostics': ['id'], + 'runs': ['id'], 'source_cycles': ['id'], 'errors': ['id'], + 'queue_snapshots': ['id'], 'config_snapshots': ['id'], + 'package_repo_candidates': ['id'], + 'target_queue': ['id'], 'target_scans': ['id'], 'findings': ['id'], + 'discovery_retry_queue': ['id'], + 'target_queue_policy_events': ['id'], + 'scan_publication_outbox': ['id'], 'keycheck_results': ['id'], + 'keycheck_event_map': ['event_id'], 'finding_uid_map': ['finding_uid'], + 'target_queue_reconciliation_cursors': ['source_file'], + 'target_queue_reconciliation_issues': ['id'], + 'docker_content_blobs': ['digest', 'coverage_policy_sha256'], + 'docker_image_blob_coverage': ['reservation_id', 'position'], + 'docker_adaptive_shadow_reports': ['id'], + 'docker_depth_experiments': ['id'], + 'docker_depth_experiment_queries': ['id'], + 'docker_discovery_passes': ['id'], + 'docker_discovery_pages': ['id'], + 'docker_repository_query_provenance': ['source', 'query', 'repository_queue_id'], + 'docker_repository_query_observations': ['page_id', 'repository_queue_id'], + 'docker_image_manifests': ['id'], + 'docker_manifest_layers': ['id'], + 'docker_depth_experiment_repositories': ['id'], + 'docker_depth_experiment_targets': ['id'], + 'docker_depth_experiment_selections': ['id'], + 'docker_depth_experiment_candidate_skips': ['id'], + 'docker_depth_resolver_attempt_refunds': ['id'], + 'docker_depth_resolver_dispositions': ['id'], + 'docker_depth_experiment_scan_bindings': ['id'], + 'docker_finding_layer_attributions': ['id'], +} + +RUNTIME_COLUMN_DEFAULTS = { + 'runtime_operations': { + 'target_ref': '', 'status': 'requested', + 'expected_identity_json': '{}', 'agent_state': 'not_required', + }, + 'runtime_operations_control': { + 'revision': '0', 'discovery_paused': '0', 'dispatch_paused': '0', + 'drain_state': 'normal', + }, + 'runtime_audit_events': {'target_ref': ''}, + 'runs': { + 'total_fetched': '0', 'total_queued_new': '0', 'total_scan_requested': '0', + 'total_scanned': '0', 'total_clean': '0', 'total_found': '0', + 'total_skipped': '0', 'total_errors': '0', 'total_findings': '0', + 'total_verified_findings': '0', 'total_unique_secrets': '0', + 'total_unique_findings': '0', 'total_staged': '0', 'total_quarantined': '0', + }, + 'source_cycles': { + 'fetched_count': '0', 'queued_new_count': '0', 'queued_updated_count': '0', + 'scan_requested_count': '0', + 'scanned_count': '0', 'clean_count': '0', 'found_count': '0', + 'skipped_count': '0', 'error_count': '0', 'findings_count': '0', + 'verified_findings_count': '0', 'unique_secrets_count': '0', + 'unique_findings_count': '0', 'targets_per_hour': '0', 'hit_rate': '0', + 'verified_hit_rate': '0', 'error_rate': '0', 'staged_count': '0', + 'ingested_count': '0', 'quarantined_count': '0', + }, + 'target_queue': {'status': 'pending', 'attempts': '0', 'resolver_attempts': '0'}, + 'discovery_retry_queue': { + 'page_start': '1', 'page_end': '1', 'next_page': '1', + 'status': 'pending', 'attempts': '0', + }, + 'target_scans': { + 'queue_completion_applied': '0', 'findings_count': '0', + 'verified_findings_count': '0', 'error_count': '0', + 'compat_schema_version': '2', 'raw_result_storage': 'legacy', + }, + 'findings': {'verified': '0', 'required_context_missing': '0', 'raw_payload_omitted': '0'}, + 'scan_publication_outbox': {'status': 'pending', 'attempts': '0'}, + 'keycheck_results': { + 'link_status': 'pending', 'link_attempts': '0', 'result_source': 'api_check', + }, + 'target_queue_reconciliation_cursors': { + 'file_size': '0', 'file_mtime_ns': '0', 'byte_offset': '0', 'line_number': '0', + 'discarding_oversized': '0', 'cumulative_rows': '0', 'cumulative_bytes': '0', + 'cumulative_inserted': '0', 'cumulative_rejected': '0', + }, + 'docker_content_blobs': {'state': 'pending', 'attempts': '0'}, + 'docker_image_blob_coverage': {'selection_policy_sha256': ''}, + 'docker_adaptive_shadow_reports': { + 'state': 'running', 'completed_pairs': '0', 'full_routed_count': '0', + 'adaptive_routed_count': '0', 'routed_intersection_count': '0', + 'full_detector_count': '0', 'adaptive_detector_count': '0', + 'detector_intersection_count': '0', 'full_slot_ms': '0', + 'adaptive_slot_ms': '0', 'omitted_descriptor_count': '0', + 'failure_count': '0', 'privacy_violation_count': '0', + 'safety_regression_count': '0', 'selection_metrics_json': '{}', + 'sink_checkpoint_count': '0', 'routed_recall_ppm': '0', + 'slot_ratio_ppm': '0', + 'recall_threshold_ppm': '850000', + 'slot_threshold_ppm': '400000', + 'passed': '0', + }, + 'docker_depth_experiments': { + 'state': 'collecting', 'target_count': '0', 'selection_count': '0', + 'fence_generation': '0', + 'collection_generation': 'docker-depth-provenance-v1', + }, + 'docker_depth_experiment_queries': { + 'required_repository_count': '10', 'selected_repository_count': '0', + }, + 'docker_discovery_passes': { + 'completed_query_count': '0', 'state': 'collecting', + 'collection_generation': 'docker-depth-provenance-v1', + }, + 'docker_discovery_pages': { + 'result_count': '0', 'admitted_count': '0', 'query_complete': '0', + 'admission_kind': 'main', + }, + 'docker_repository_query_provenance': { + 'observation_count': '1', 'fresh_observation_count': '0', + 'fresh_complete_observation_count': '0', 'fresh_coverage_eligible': '0', + }, + 'docker_depth_experiment_repositories': { + 'planned_is_deep_probe': '0', 'is_deep_probe': '0', + 'work_state': 'pending', 'resolver_generation': '0', + 'resolver_attempts': '0', 'candidate_distinct_graph_count': '0', + 'selected_image_count': '0', 'replacement_count': '0', + }, + 'docker_depth_experiment_targets': {'state': 'pending', 'reservation_count': '0'}, + 'docker_depth_experiment_scan_bindings': {'state': 'reserved'}, +} + +GENERATED_ID_COLUMNS = { + 'runtime_audit_events': 'id', + 'worker_progress_events': 'id', 'worker_diagnostics': 'id', + 'runs': 'id', 'source_cycles': 'id', 'target_queue': 'id', + 'discovery_retry_queue': 'id', + 'target_scans': 'id', 'findings': 'id', 'errors': 'id', + 'queue_snapshots': 'id', 'config_snapshots': 'id', + 'package_repo_candidates': 'id', + 'scan_publication_outbox': 'id', 'keycheck_results': 'id', + 'target_queue_reconciliation_issues': 'id', + 'docker_adaptive_shadow_reports': 'id', + 'target_queue_policy_events': 'id', + 'docker_depth_experiments': 'id', + 'docker_depth_experiment_queries': 'id', + 'docker_discovery_passes': 'id', + 'docker_discovery_pages': 'id', + 'docker_image_manifests': 'id', + 'docker_manifest_layers': 'id', + 'docker_depth_experiment_repositories': 'id', + 'docker_depth_experiment_targets': 'id', + 'docker_depth_experiment_selections': 'id', + 'docker_depth_experiment_candidate_skips': 'id', + 'docker_depth_resolver_attempt_refunds': 'id', + 'docker_depth_resolver_dispositions': 'id', + 'docker_depth_experiment_scan_bindings': 'id', + 'docker_finding_layer_attributions': 'id', +} + +REQUIRED_FOREIGN_KEYS = { + 'runtime_operations_control': [ + (('operation_id',), 'runtime_operations', ('operation_id',)), + ], + 'runtime_audit_events': [ + (('operation_id',), 'runtime_operations', ('operation_id',)), + (('previous_event_id',), 'runtime_audit_events', ('id',)), + ], + 'worker_progress_events': [ + (('reservation_id',), 'result_reservations', ('id',)), + (('remote_device_id',), 'remote_worker_devices', ('id',)), + ], + 'worker_diagnostics': [ + (('reservation_id',), 'result_reservations', ('id',)), + (('target_scan_id',), 'target_scans', ('id',)), + ], + 'source_cycles': [(('run_id',), 'runs', ('id',))], + 'discovery_retry_queue': [ + (('source_cycle_id',), 'source_cycles', ('id',)), + ], + 'target_scans': [ + (('run_id',), 'runs', ('id',)), + (('cycle_id',), 'source_cycles', ('id',)), + (('queue_id',), 'target_queue', ('id',)), + (('result_reservation_id',), 'result_reservations', ('id',)), + ], + 'findings': [ + (('run_id',), 'runs', ('id',)), + (('cycle_id',), 'source_cycles', ('id',)), + (('target_scan_id',), 'target_scans', ('id',)), + ], + 'errors': [ + (('run_id',), 'runs', ('id',)), + (('cycle_id',), 'source_cycles', ('id',)), + (('target_scan_id',), 'target_scans', ('id',)), + ], + 'queue_snapshots': [ + (('run_id',), 'runs', ('id',)), + (('cycle_id',), 'source_cycles', ('id',)), + ], + 'config_snapshots': [ + (('run_id',), 'runs', ('id',)), + (('cycle_id',), 'source_cycles', ('id',)), + ], + 'package_repo_candidates': [ + (('last_run_id',), 'runs', ('id',)), + (('last_cycle_id',), 'source_cycles', ('id',)), + ], + 'target_queue': [ + (('target_scan_id',), 'target_scans', ('id',)), + (('current_result_reservation_id',), 'result_reservations', ('id',)), + ], + 'target_queue_policy_events': [ + (('queue_id',), 'target_queue', ('id',)), + (('reverses_event_id',), 'target_queue_policy_events', ('id',)), + (('experiment_id',), 'docker_depth_experiments', ('id',)), + ], + 'scan_publication_outbox': [(('target_scan_id',), 'target_scans', ('id',))], + 'keycheck_results': [ + (('finding_id',), 'findings', ('id',)), + (('target_scan_id',), 'target_scans', ('id',)), + (('cycle_id',), 'source_cycles', ('id',)), + (('run_id',), 'runs', ('id',)), + (('candidate_id',), 'keycheck_candidates', ('id',)), + (('credential_id',), 'keycheck_credentials', ('id',)), + ], + 'keycheck_event_map': [(('keycheck_result_id',), 'keycheck_results', ('id',))], + 'finding_uid_map': [(('finding_id',), 'findings', ('id',))], + 'result_reservations': [ + (('queue_id',), 'target_queue', ('id',)), + (('run_id',), 'runs', ('id',)), + (('cycle_id',), 'source_cycles', ('id',)), + (('remote_device_id', 'remote_user_id'), 'remote_worker_devices', ('id', 'user_id')), + ], + 'remote_worker_devices': [(('user_id',), 'remote_worker_users', ('id',))], + 'admission_intents': [ + (('reservation_id',), 'result_reservations', ('id',)), + (('remote_device_id', 'remote_user_id'), 'remote_worker_devices', ('id', 'user_id')), + ], + 'docker_content_blobs': [ + (('lease_reservation_id',), 'result_reservations', ('id',)), + (('covered_reservation_id',), 'result_reservations', ('id',)), + ], + 'docker_image_blob_coverage': [ + (('queue_id',), 'target_queue', ('id',)), + ( + ('blob_digest', 'coverage_policy_sha256'), + 'docker_content_blobs', + ('digest', 'coverage_policy_sha256'), + ), + (('reservation_id',), 'result_reservations', ('id',)), + ], + 'result_bundles': [ + (('reservation_id',), 'result_reservations', ('id',)), + (('target_scan_id',), 'target_scans', ('id',)), + ], + 'scan_result_compat': [(('target_scan_id',), 'target_scans', ('id',))], + 'finding_compat_payloads': [(('finding_id',), 'findings', ('id',))], + 'projection_jobs': [ + (('target_scan_id',), 'target_scans', ('id',)), + (('keycheck_result_id',), 'keycheck_results', ('id',)), + ], + 'projection_cursors': [ + (('stream_name',), 'projection_streams', ('stream_name',)), + (('last_append_id',), 'projection_appends', ('id',)), + (('last_job_id',), 'projection_jobs', ('id',)), + ], + 'projection_appends': [ + (('job_id',), 'projection_jobs', ('id',)), + (('stream_name',), 'projection_streams', ('stream_name',)), + ], + 'projection_rotations': [(('stream_name',), 'projection_streams', ('stream_name',))], + 'keycheck_candidates': [ + (('credential_id',), 'keycheck_credentials', ('id',)), + (('finding_id',), 'findings', ('id',)), + (('target_scan_id',), 'target_scans', ('id',)), + (('keycheck_result_id',), 'keycheck_results', ('id',)), + ], + 'keycheck_current_state': [ + (('credential_id',), 'keycheck_credentials', ('id',)), + (('last_result_id',), 'keycheck_results', ('id',)), + ], + 'pipeline_quarantine': [ + (('reservation_id',), 'result_reservations', ('id',)), + (('projection_job_id',), 'projection_jobs', ('id',)), + (('keycheck_candidate_id',), 'keycheck_candidates', ('id',)), + ], + 'docker_depth_experiment_queries': [ + (('experiment_id',), 'docker_depth_experiments', ('id',)), + ], + 'docker_discovery_passes': [ + (('experiment_id',), 'docker_depth_experiments', ('id',)), + ], + 'docker_discovery_pages': [ + (('pass_id',), 'docker_discovery_passes', ('id',)), + (('source_cycle_id',), 'source_cycles', ('id',)), + (('retry_work_id',), 'discovery_retry_queue', ('id',)), + ], + 'docker_repository_query_provenance': [ + (('repository_queue_id',), 'target_queue', ('id',)), + (('first_cycle_id',), 'source_cycles', ('id',)), + (('last_cycle_id',), 'source_cycles', ('id',)), + (('first_page_id',), 'docker_discovery_pages', ('id',)), + (('last_page_id',), 'docker_discovery_pages', ('id',)), + ], + 'docker_repository_query_observations': [ + (('page_id',), 'docker_discovery_pages', ('id',)), + (('repository_queue_id',), 'target_queue', ('id',)), + ( + ('source', 'query', 'repository_queue_id'), + 'docker_repository_query_provenance', + ('source', 'query', 'repository_queue_id'), + ), + ], + 'docker_image_manifests': [ + (('target_queue_id',), 'target_queue', ('id',)), + ], + 'docker_manifest_layers': [ + (('manifest_id',), 'docker_image_manifests', ('id',)), + ], + 'docker_depth_experiment_repositories': [ + ( + ('experiment_id', 'query_ordinal', 'source', 'query'), + 'docker_depth_experiment_queries', + ('experiment_id', 'query_ordinal', 'source', 'query'), + ), + (('repository_queue_id',), 'target_queue', ('id',)), + ( + ('source', 'query', 'repository_queue_id'), + 'docker_repository_query_provenance', + ('source', 'query', 'repository_queue_id'), + ), + ( + ('eligibility_page_id', 'repository_queue_id', 'source', 'query'), + 'docker_repository_query_observations', + ('page_id', 'repository_queue_id', 'source', 'query'), + ), + (('replacement_repository_queue_id',), 'target_queue', ('id',)), + ( + ( + 'replacement_eligibility_page_id', + 'replacement_repository_queue_id', + 'source', + 'query', + ), + 'docker_repository_query_observations', + ('page_id', 'repository_queue_id', 'source', 'query'), + ), + ], + 'docker_depth_experiment_targets': [ + (('experiment_id',), 'docker_depth_experiments', ('id',)), + (('target_queue_id',), 'target_queue', ('id',)), + ( + ('manifest_id', 'target_queue_id'), + 'docker_image_manifests', + ('id', 'target_queue_id'), + ), + ], + 'docker_depth_experiment_selections': [ + ( + ('experiment_id', 'query_ordinal'), + 'docker_depth_experiment_queries', + ('experiment_id', 'query_ordinal'), + ), + ( + ('experiment_id', 'query_ordinal', 'experiment_repository_id'), + 'docker_depth_experiment_repositories', + ('experiment_id', 'query_ordinal', 'id'), + ), + ( + ('experiment_id', 'experiment_target_id'), + 'docker_depth_experiment_targets', + ('experiment_id', 'id'), + ), + ], + 'docker_depth_experiment_candidate_skips': [ + (('experiment_id',), 'docker_depth_experiments', ('id',)), + (('experiment_repository_id',), 'docker_depth_experiment_repositories', ('id',)), + (('repository_queue_id',), 'target_queue', ('id',)), + ], + 'docker_depth_resolver_attempt_refunds': [ + (('experiment_id',), 'docker_depth_experiments', ('id',)), + (('experiment_repository_id',), 'docker_depth_experiment_repositories', ('id',)), + (('repository_queue_id',), 'target_queue', ('id',)), + ], + 'docker_depth_resolver_dispositions': [ + (('experiment_id',), 'docker_depth_experiments', ('id',)), + (('experiment_repository_id',), 'docker_depth_experiment_repositories', ('id',)), + (('prior_repository_queue_id',), 'target_queue', ('id',)), + (('replacement_repository_queue_id',), 'target_queue', ('id',)), + (('replacement_eligibility_page_id',), 'docker_discovery_pages', ('id',)), + ], + 'docker_depth_experiment_scan_bindings': [ + (('experiment_target_id',), 'docker_depth_experiment_targets', ('id',)), + (('reservation_id',), 'result_reservations', ('id',)), + (('target_scan_id',), 'target_scans', ('id',)), + ], + 'docker_finding_layer_attributions': [ + (('scan_binding_id',), 'docker_depth_experiment_scan_bindings', ('id',)), + (('finding_id',), 'findings', ('id',)), + ( + ( + 'manifest_layer_id', 'position_from_base', 'position_from_top', + 'reported_layer_digest', + ), + 'docker_manifest_layers', + ('id', 'position_from_base', 'position_from_top', 'layer_digest'), + ), + ], +} + +REQUIRED_FOREIGN_KEY_ACTIONS = { + (table, columns): ('NO ACTION', 'NO ACTION') + for table, foreign_keys in REQUIRED_FOREIGN_KEYS.items() + for columns, _, _ in foreign_keys +} + + +def _required_foreign_key_actions(table, columns): + return REQUIRED_FOREIGN_KEY_ACTIONS[(table, columns)] + + +def _foreign_key_actions_match(foreign_key, expected_actions): + update_action, delete_action = expected_actions + return ( + str(foreign_key.get('update_action') or '').strip().upper() == update_action + and str(foreign_key.get('delete_action') or '').strip().upper() == delete_action + ) + + +def _schema_type_matches(actual, expected, postgres): + value = re.sub(r'\s+', ' ', str(actual or '').strip().lower()) + if expected in ('id', 'id_ref'): + return value == ('bigint' if postgres else 'integer') + if expected == 'integer': + return value == 'integer' + if expected == 'text': + return value == 'text' + if expected == 'real': + return value == 'real' + return value == expected + + +def _normalized_predicate(value): + text = str(value or '').lower().replace('"', '') + text = re.sub(r'::(?:text|character varying)', '', text) + return re.sub(r'[\s()]', '', text) + + +def _normalized_default(value): + text = str(value or '').strip().lower() + text = re.sub(r'::(?:text|character varying|integer|bigint|smallint)$', '', text) + while len(text) >= 2 and text[0] == '(' and text[-1] == ')': + text = text[1:-1].strip() + if len(text) >= 2 and text[0] == text[-1] == "'": + text = text[1:-1].replace("''", "'") + return text + + +def _sql_enum_check(column, values): + return '(' + ' OR '.join(f"{column} = '{value}'" for value in values) + ')' + + +def _sql_sha256_check(column, nullable=False): + remainder = column + for character in '0123456789abcdef': + remainder = f"replace({remainder}, '{character}', '')" + expression = ( + f'(length({column}) = 64 AND {column} = lower({column}) ' + f'AND length({remainder}) = 0)' + ) + return f'({column} IS NULL OR {expression})' if nullable else expression + + +DOCKER_DEPTH_EXPERIMENT_STATES = ( + 'collecting', 'planned', 'holding', 'resolving', 'active', 'draining', + 'completed', 'released', 'held', +) +DOCKER_DISCOVERY_PASS_STATES = ('collecting', 'complete', 'held', 'failed') +DOCKER_EXPERIMENT_REPOSITORY_STATES = ( + 'pending', 'resolving', 'resolved', 'held', 'failed', 'skipped', +) +DOCKER_EXPERIMENT_TARGET_STATES = ( + 'pending', 'reserved', 'scanning', 'done', 'failed', 'held', + 'quarantined', 'skipped', +) +DOCKER_EXPERIMENT_BINDING_STATES = ( + 'reserved', 'scanning', 'completed', 'failed', 'quarantined', 'released', +) +DOCKER_FINDING_UNATTRIBUTED_REASONS = ( + 'digest_absent', 'digest_invalid', 'digest_not_in_manifest', + 'manifest_mismatch', +) + +DOCKER_DEPTH_EXPERIMENT_CHECK_SPECS = { + 'discovery_retry_queue': { + 'discovery_retry_queue_lifecycle_check': ( + "(status = 'leased' AND lease_owner IS NOT NULL AND lease_token IS NOT NULL " + 'AND leased_at IS NOT NULL AND lease_expires_at IS NOT NULL AND held_at IS NULL) ' + "OR (status = 'pending' AND lease_owner IS NULL AND lease_token IS NULL " + 'AND leased_at IS NULL AND lease_expires_at IS NULL AND held_at IS NULL) ' + "OR (status = 'held' AND lease_owner IS NULL AND lease_token IS NULL " + 'AND leased_at IS NULL AND lease_expires_at IS NULL AND held_at IS NOT NULL)' + ), + 'discovery_retry_queue_sha256_check': ( + f'{_sql_sha256_check("work_key")} ' + f'AND {_sql_sha256_check("policy_sha256")}' + ), + }, + 'target_queue_policy_events': { + 'target_queue_policy_events_transition_check': ( + "(action = 'cold' AND (prior_status = 'pending' OR prior_status = 'deferred') " + "AND next_status = 'cold' AND reverses_event_id IS NULL) OR " + "(action = 'reactivate' AND prior_status = 'cold' " + "AND (next_status = 'pending' OR next_status = 'deferred') " + 'AND reverses_event_id IS NOT NULL)' + ), + 'target_queue_policy_events_hash_check': ( + f'{_sql_sha256_check("config_sha256")} ' + f'AND {_sql_sha256_check("policy_sha256")} ' + f'AND {_sql_sha256_check("manifest_sha256")} ' + f'AND {_sql_sha256_check("review_audit_sha256")}' + ), + 'target_queue_policy_events_experiment_ownership_check': ( + "(experiment_id IS NULL " + "AND reason_code <> 'docker_depth_experiment_hold' " + "AND reason_code <> 'docker_depth_experiment_dynamic_hold' " + "AND reason_code <> 'docker_depth_experiment_reviewed_release') OR " + "(experiment_id IS NOT NULL AND ((action = 'cold' AND " + "(reason_code = 'docker_depth_experiment_hold' " + "OR reason_code = 'docker_depth_experiment_dynamic_hold')) " + "OR (action = 'reactivate' " + "AND reason_code = 'docker_depth_experiment_reviewed_release')))" + ), + }, + 'docker_depth_experiments': { + 'docker_depth_experiments_state_check': _sql_enum_check( + 'state', DOCKER_DEPTH_EXPERIMENT_STATES, + ), + 'docker_depth_experiments_range_check': ( + 'query_count >= 1 AND query_count <= 1000 ' + 'AND repositories_per_query >= 1 AND repositories_per_query <= 39 ' + 'AND images_per_repository >= 1 AND images_per_repository <= 10 ' + 'AND target_limit >= 1 AND target_limit <= 2000 ' + 'AND target_count >= 0 AND target_count <= target_limit ' + 'AND selection_count >= 0 AND fence_generation >= 0' + ), + 'docker_depth_experiments_identity_check': ( + 'length(experiment_key) >= 1 AND length(experiment_key) <= 128 ' + 'AND length(source) >= 1 AND length(source) <= 64 ' + f'AND {_sql_sha256_check("config_sha256")} ' + f'AND {_sql_sha256_check("ordered_queries_sha256")} ' + 'AND length(selector_version) >= 1 AND length(selector_version) <= 128 ' + f'AND {_sql_sha256_check("selector_sha256")} ' + f'AND {_sql_sha256_check("provenance_policy_sha256")} ' + f'AND {_sql_sha256_check("plan_sha256", nullable=True)} ' + f'AND {_sql_sha256_check("hold_manifest_sha256", nullable=True)}' + ), + 'docker_depth_experiments_capacity_check': ( + '(query_count * (repositories_per_query + images_per_repository - 1) ' + "<= target_limit OR (selector_version = 'docker-rank1-breadth-v1' " + 'AND query_count = 61 AND repositories_per_query = 39 ' + 'AND images_per_repository = 1 AND target_limit = 2000))' + ), + 'docker_depth_experiments_selection_check': _sql_sha256_check( + 'selection_sha256', nullable=True, + ), + 'docker_depth_experiments_fence_check': ( + '(fence_owner IS NULL AND fence_token IS NULL AND fence_expires_at IS NULL) ' + 'OR (fence_owner IS NOT NULL AND fence_token IS NOT NULL ' + 'AND fence_expires_at IS NOT NULL)' + ), + }, + 'docker_depth_experiment_queries': { + 'docker_depth_experiment_queries_range_check': ( + 'query_ordinal >= 0 AND required_repository_count >= 1 ' + 'AND required_repository_count <= 39' + ), + 'docker_depth_experiment_queries_identity_check': ( + 'length(source) >= 1 AND length(source) <= 64 ' + 'AND length(query) >= 1 AND length(query) <= 256 ' + f'AND {_sql_sha256_check("query_sha256")}' + ), + 'docker_depth_experiment_queries_selection_check': ( + 'selected_repository_count >= 0 ' + 'AND selected_repository_count <= required_repository_count' + ), + }, + 'docker_discovery_passes': { + 'docker_discovery_passes_kind_check': _sql_enum_check( + 'pass_kind', ('ordinary', 'deep'), + ), + 'docker_discovery_passes_state_check': _sql_enum_check( + 'state', DOCKER_DISCOVERY_PASS_STATES, + ), + 'docker_discovery_passes_count_check': ( + 'expected_query_count >= 1 AND expected_query_count <= 1000 ' + 'AND completed_query_count >= 0 ' + 'AND completed_query_count <= expected_query_count' + ), + 'docker_discovery_passes_identity_check': ( + 'length(pass_token) >= 16 AND length(pass_token) <= 256 ' + 'AND length(source) >= 1 AND length(source) <= 64 ' + f'AND {_sql_sha256_check("policy_sha256")} ' + f'AND {_sql_sha256_check("ordered_queries_sha256")}' + ), + 'docker_discovery_passes_complete_check': ( + "state <> 'complete' OR (completed_query_count = expected_query_count " + 'AND completed_at IS NOT NULL)' + ), + }, + 'docker_discovery_pages': { + 'docker_discovery_pages_range_check': ( + 'query_ordinal >= 0 AND page_number >= 1 AND page_number <= 30 ' + 'AND result_count >= 0 AND admitted_count >= 0 ' + 'AND admitted_count <= result_count' + ), + 'docker_discovery_pages_boolean_check': '(query_complete = 0 OR query_complete = 1)', + 'docker_discovery_pages_kind_check': _sql_enum_check( + 'admission_kind', ('main', 'retry'), + ), + 'docker_discovery_pages_identity_check': ( + 'length(query) >= 1 AND length(query) <= 256 ' + f'AND {_sql_sha256_check("page_sha256")}' + ), + 'docker_discovery_pages_evidence_check': ( + "(admission_kind = 'main' AND retry_work_id IS NULL) " + "OR admission_kind = 'retry'" + ), + 'docker_discovery_pages_total_count_check': ( + 'total_count IS NULL OR (total_count >= 0 AND total_count >= result_count)' + ), + }, + 'docker_repository_query_provenance': { + 'docker_repository_query_provenance_kind_check': _sql_enum_check( + 'provenance_kind', ('legacy_queue', 'fresh_page'), + ), + 'docker_repository_query_provenance_range_check': ( + '(first_search_rank IS NULL OR first_search_rank >= 1) ' + 'AND (best_search_rank IS NULL OR best_search_rank >= 1) ' + 'AND (last_search_rank IS NULL OR last_search_rank >= 1) ' + 'AND observation_count >= 1 AND fresh_observation_count >= 0 ' + 'AND fresh_complete_observation_count >= 0' + ), + 'docker_repository_query_provenance_boolean_check': ( + '(fresh_coverage_eligible = 0 OR fresh_coverage_eligible = 1)' + ), + 'docker_repository_query_provenance_identity_check': ( + 'length(source) >= 1 AND length(source) <= 64 ' + 'AND length(query) >= 1 AND length(query) <= 256 ' + f'AND {_sql_sha256_check("first_policy_sha256", nullable=True)} ' + f'AND {_sql_sha256_check("last_policy_sha256", nullable=True)}' + ), + 'docker_repository_query_provenance_counts_check': ( + 'fresh_complete_observation_count <= fresh_observation_count ' + 'AND fresh_observation_count <= observation_count' + ), + 'docker_repository_query_provenance_legacy_check': ( + "provenance_kind <> 'legacy_queue' OR (fresh_observation_count = 0 " + 'AND fresh_complete_observation_count = 0 AND fresh_coverage_eligible = 0 ' + 'AND first_page_id IS NULL AND last_page_id IS NULL ' + 'AND first_policy_sha256 IS NULL AND last_policy_sha256 IS NULL)' + ), + 'docker_repository_query_provenance_fresh_check': ( + "provenance_kind <> 'fresh_page' OR (fresh_observation_count >= 1 " + 'AND first_search_rank IS NOT NULL AND best_search_rank IS NOT NULL ' + 'AND last_search_rank IS NOT NULL AND first_page_id IS NOT NULL ' + 'AND last_page_id IS NOT NULL AND first_policy_sha256 IS NOT NULL ' + 'AND last_policy_sha256 IS NOT NULL)' + ), + 'docker_repository_query_provenance_eligibility_check': ( + '(fresh_complete_observation_count = 0 AND fresh_coverage_eligible = 0) ' + "OR (fresh_complete_observation_count >= 1 AND fresh_coverage_eligible = 1 " + "AND provenance_kind = 'fresh_page')" + ), + }, + 'docker_repository_query_observations': { + 'docker_repository_query_observations_rank_check': 'search_rank >= 1', + }, + 'docker_image_manifests': { + 'docker_image_manifests_range_check': ( + 'layer_count >= 0 AND (manifest_size_bytes IS NULL OR manifest_size_bytes >= 0)' + ), + 'docker_image_manifests_identity_check': ( + 'length(source) >= 1 AND length(source) <= 64 ' + 'AND length(repository) >= 1 AND length(repository) <= 512 ' + 'AND length(manifest_digest) >= 1 AND length(manifest_digest) <= 256 ' + 'AND length(manifest_media_type) >= 1 AND length(manifest_media_type) <= 256 ' + 'AND (config_digest IS NULL OR (length(config_digest) >= 1 ' + 'AND length(config_digest) <= 256)) ' + f'AND {_sql_sha256_check("graph_sha256")}' + ), + }, + 'docker_manifest_layers': { + 'docker_manifest_layers_range_check': ( + 'position_from_base >= 1 AND position_from_top >= 1 AND layer_size_bytes >= 0' + ), + 'docker_manifest_layers_identity_check': ( + 'length(layer_digest) >= 1 AND length(layer_digest) <= 256 ' + 'AND length(media_type) >= 1 AND length(media_type) <= 256 ' + f'AND {_sql_sha256_check("descriptor_sha256")}' + ), + }, + 'docker_depth_experiment_repositories': { + 'docker_depth_experiment_repositories_range_check': ( + 'query_ordinal >= 0 AND repository_rank >= 1 AND repository_rank <= 39 ' + 'AND resolver_generation >= 0 AND resolver_attempts >= 0 ' + 'AND selected_image_count >= 0 AND selected_image_count <= 10' + ), + 'docker_depth_experiment_repositories_boolean_check': ( + '(is_deep_probe = 0 OR is_deep_probe = 1)' + ), + 'docker_depth_experiment_repositories_state_check': _sql_enum_check( + 'work_state', DOCKER_EXPERIMENT_REPOSITORY_STATES, + ), + 'docker_depth_experiment_repositories_fence_check': ( + "(work_state = 'resolving' AND resolver_owner IS NOT NULL " + 'AND resolver_token IS NOT NULL AND resolver_expires_at IS NOT NULL) ' + "OR (work_state <> 'resolving' AND resolver_owner IS NULL " + 'AND resolver_token IS NULL AND resolver_expires_at IS NULL)' + ), + 'docker_depth_experiment_repositories_resolver_check': ( + '(planned_is_deep_probe = 0 OR planned_is_deep_probe = 1) ' + 'AND resolver_attempts >= 0 AND resolver_attempts <= 3 ' + 'AND candidate_distinct_graph_count >= 0 ' + 'AND candidate_distinct_graph_count <= 100 ' + 'AND selected_image_count <= candidate_distinct_graph_count ' + 'AND replacement_count >= 0 AND replacement_count <= 3000' + ), + 'docker_depth_experiment_repositories_replacement_check': ( + '(replacement_count = 0 AND replacement_repository_queue_id IS NULL ' + 'AND replacement_eligibility_page_id IS NULL ' + 'AND replacement_evidence_sha256 IS NULL) OR ' + '(replacement_count >= 1 AND replacement_repository_queue_id IS NOT NULL ' + 'AND replacement_eligibility_page_id IS NOT NULL ' + f'AND {_sql_sha256_check("replacement_evidence_sha256")})' + ), + }, + 'docker_depth_experiment_targets': { + 'docker_depth_experiment_targets_range_check': ( + 'counter_ordinal >= 1 AND counter_ordinal <= 2000 ' + 'AND dispatch_wave >= 1 AND dispatch_wave <= 3 ' + 'AND dispatch_order >= 1 AND reservation_count >= 0' + ), + 'docker_depth_experiment_targets_state_check': _sql_enum_check( + 'state', DOCKER_EXPERIMENT_TARGET_STATES, + ), + }, + 'docker_depth_experiment_selections': { + 'docker_depth_experiment_selections_range_check': ( + 'query_ordinal >= 0 AND image_rank >= 1 AND image_rank <= 10' + ), + 'docker_depth_experiment_selections_identity_check': ( + 'length(selection_reason) >= 1 AND length(selection_reason) <= 128 ' + f'AND {_sql_sha256_check("selection_evidence_sha256")} ' + f'AND {_sql_sha256_check("graph_sha256")}' + ), + }, + 'docker_depth_experiment_candidate_skips': { + 'docker_depth_experiment_candidate_skips_kind_check': _sql_enum_check( + 'candidate_kind', ('image', 'repository'), + ), + 'docker_depth_experiment_candidate_skips_range_check': ( + 'candidate_ordinal >= 1 AND candidate_ordinal <= 3000' + ), + 'docker_depth_experiment_candidate_skips_identity_check': ( + f'{_sql_sha256_check("candidate_identity_sha256")} ' + f'AND {_sql_sha256_check("evidence_sha256")}' + ), + }, + 'docker_depth_resolver_attempt_refunds': { + 'docker_depth_resolver_attempt_refunds_kind_check': ( + "recovery_kind = 'zero_graph_limit_v1'" + ), + 'docker_depth_resolver_attempt_refunds_state_check': ( + "(prior_work_state = 'pending' OR prior_work_state = 'held') " + "AND next_work_state = 'pending'" + ), + 'docker_depth_resolver_attempt_refunds_range_check': ( + 'query_ordinal >= 0 AND repository_rank >= 1 ' + 'AND prior_resolver_attempts >= 2 AND refund_attempts = 2 ' + 'AND next_resolver_attempts = prior_resolver_attempts - refund_attempts ' + 'AND next_resolver_attempts >= 0 AND confirmed_bug_event_count = 2' + ), + 'docker_depth_resolver_attempt_refunds_hash_check': ( + f'{_sql_sha256_check("manifest_sha256")} ' + f'AND {_sql_sha256_check("entry_evidence_sha256")} ' + f'AND {_sql_sha256_check("log_sha256")} ' + f'AND {_sql_sha256_check("target_identity_sha256")} ' + f'AND {_sql_sha256_check("prior_error_code_sha256")}' + ), + }, + 'docker_depth_resolver_dispositions': { + 'docker_depth_resolver_dispositions_kind_check': ( + "disposition_kind = 'resolver_attempt_limit_replace_or_skip_v1'" + ), + 'docker_depth_resolver_dispositions_outcome_check': ( + "outcome = 'replaced' OR outcome = 'skipped'" + ), + 'docker_depth_resolver_dispositions_state_check': ( + "prior_work_state = 'held' AND (" + "(outcome = 'replaced' AND next_work_state = 'pending' " + "AND replacement_repository_queue_id IS NOT NULL " + "AND replacement_eligibility_page_id IS NOT NULL " + "AND replacement_best_search_rank IS NOT NULL " + "AND replacement_target_identity_sha256 IS NOT NULL " + "AND next_resolver_attempts = 0) OR " + "(outcome = 'skipped' AND next_work_state = 'skipped' " + "AND replacement_repository_queue_id IS NULL " + "AND replacement_eligibility_page_id IS NULL " + "AND replacement_best_search_rank IS NULL " + "AND replacement_target_identity_sha256 IS NULL " + "AND next_resolver_attempts = prior_resolver_attempts))" + ), + 'docker_depth_resolver_dispositions_range_check': ( + 'query_ordinal >= 0 AND repository_rank >= 1 ' + 'AND prior_resolver_attempts >= 3 AND next_resolver_attempts >= 0 ' + 'AND (replacement_best_search_rank IS NULL ' + 'OR replacement_best_search_rank >= 1)' + ), + 'docker_depth_resolver_dispositions_hash_check': ( + f'{_sql_sha256_check("manifest_sha256")} ' + f'AND {_sql_sha256_check("entry_evidence_sha256")} ' + f'AND {_sql_sha256_check("candidate_snapshot_sha256")} ' + f'AND {_sql_sha256_check("prior_target_identity_sha256")} ' + f'AND {_sql_sha256_check("prior_error_code_sha256")} ' + f'AND (replacement_target_identity_sha256 IS NULL ' + f'OR {_sql_sha256_check("replacement_target_identity_sha256")})' + ), + }, + 'docker_depth_experiment_scan_bindings': { + 'docker_depth_experiment_scan_bindings_range_check': 'attempt >= 1', + 'docker_depth_experiment_scan_bindings_state_check': _sql_enum_check( + 'state', DOCKER_EXPERIMENT_BINDING_STATES, + ), + }, + 'docker_finding_layer_attributions': { + 'docker_finding_layer_attributions_state_check': _sql_enum_check( + 'attribution_state', ('exact', 'unattributed'), + ), + 'docker_finding_layer_attributions_evidence_check': ( + "(attribution_state = 'exact' AND manifest_layer_id IS NOT NULL " + 'AND reported_layer_digest IS NOT NULL AND position_from_base IS NOT NULL ' + 'AND position_from_top IS NOT NULL AND position_from_base >= 1 ' + "AND position_from_top >= 1) OR (attribution_state = 'unattributed' " + 'AND manifest_layer_id IS NULL AND reported_layer_digest IS NULL ' + 'AND position_from_base IS NULL AND position_from_top IS NULL)' + ), + 'docker_finding_layer_attributions_reason_check': ( + "(attribution_state = 'exact' AND unattributed_reason IS NULL) OR " + "(attribution_state = 'unattributed' AND unattributed_reason IS NOT NULL AND " + + _sql_enum_check( + 'unattributed_reason', DOCKER_FINDING_UNATTRIBUTED_REASONS, + ) + + ')' + ), + }, +} + +DOCKER_DEPTH_SCHEMA_AUTHORITY_CHECKS = ( + ('discovery_retry_queue', 'discovery_retry_queue_sha256_check'), + ('target_queue_policy_events', 'target_queue_policy_events_transition_check'), + ('target_queue_policy_events', 'target_queue_policy_events_hash_check'), + ( + 'target_queue_policy_events', + 'target_queue_policy_events_experiment_ownership_check', + ), + ('docker_depth_experiments', 'docker_depth_experiments_selection_check'), + ('docker_discovery_pages', 'docker_discovery_pages_total_count_check'), + ( + 'docker_depth_experiment_repositories', + 'docker_depth_experiment_repositories_resolver_check', + ), + ( + 'docker_depth_experiment_repositories', + 'docker_depth_experiment_repositories_replacement_check', + ), + ( + 'docker_depth_experiment_candidate_skips', + 'docker_depth_experiment_candidate_skips_kind_check', + ), + ( + 'docker_depth_experiment_candidate_skips', + 'docker_depth_experiment_candidate_skips_range_check', + ), + ( + 'docker_depth_experiment_candidate_skips', + 'docker_depth_experiment_candidate_skips_identity_check', + ), +) + +DOCKER_DEPTH_SCARCITY_COHORT_CHECKS = ( + ( + 'docker_depth_experiment_queries', + 'docker_depth_experiment_queries_selection_check', + ), +) + + +_CHECK_TOKEN = re.compile( + r"'(?:''|[^'])*'|<>|>=|<=|!=|=|>|<|[A-Za-z_][A-Za-z0-9_$]*|" + r'\d+|[(),+*\-]' +) + + +def _normalized_check_expression(value): + text = str(value or '').strip().lower().replace('"', '') + if text.startswith('check'): + text = text[5:].strip() + text = re.sub( + r'::(?:text|character varying|character|varchar|integer|bigint|smallint|numeric)', + '', + text, + ) + tokens = [] + position = 0 + while position < len(text): + if text[position].isspace(): + position += 1 + continue + match = _CHECK_TOKEN.match(text, position) + if not match: + return ('unparsed', re.sub(r'\s+', '', text.replace('!=', '<>'))) + token = match.group(0) + tokens.append('<>' if token == '!=' else token) + position = match.end() + + cursor = 0 + + def combine(operator, values): + flattened = [] + for item in values: + if item and item[0] == operator: + flattened.extend(item[1]) + else: + flattened.append(item) + return flattened[0] if len(flattened) == 1 else (operator, tuple(flattened)) + + def parse_primary(): + nonlocal cursor + if cursor >= len(tokens): + raise ValueError('missing CHECK expression operand') + token = tokens[cursor] + if token == '(': + cursor += 1 + expression = parse_or() + if cursor >= len(tokens) or tokens[cursor] != ')': + raise ValueError('unclosed CHECK expression group') + cursor += 1 + return expression + cursor += 1 + if re.fullmatch(r'[a-z_][a-z0-9_$]*', token) and ( + cursor < len(tokens) and tokens[cursor] == '(' + ): + cursor += 1 + arguments = [] + if cursor < len(tokens) and tokens[cursor] != ')': + while True: + arguments.append(parse_or()) + if cursor >= len(tokens) or tokens[cursor] != ',': + break + cursor += 1 + if cursor >= len(tokens) or tokens[cursor] != ')': + raise ValueError('unclosed CHECK expression function') + cursor += 1 + return ('call', token, tuple(arguments)) + if token.startswith("'"): + return ('string', token[1:-1].replace("''", "'")) + if token.isdigit(): + return ('number', int(token)) + return ('name', token) + + def parse_multiply(): + nonlocal cursor + expression = parse_primary() + while cursor < len(tokens) and tokens[cursor] == '*': + operator = tokens[cursor] + cursor += 1 + expression = (operator, expression, parse_primary()) + return expression + + def parse_add(): + nonlocal cursor + expression = parse_multiply() + while cursor < len(tokens) and tokens[cursor] in ('+', '-'): + operator = tokens[cursor] + cursor += 1 + expression = (operator, expression, parse_multiply()) + return expression + + def parse_comparison(): + nonlocal cursor + expression = parse_add() + if cursor < len(tokens) and tokens[cursor] == 'is': + cursor += 1 + negated = cursor < len(tokens) and tokens[cursor] == 'not' + if negated: + cursor += 1 + if cursor >= len(tokens) or tokens[cursor] != 'null': + raise ValueError('unsupported CHECK IS expression') + cursor += 1 + return ('is not null' if negated else 'is null', expression) + if cursor < len(tokens) and tokens[cursor] in ('=', '<>', '>=', '<=', '>', '<'): + operator = tokens[cursor] + cursor += 1 + return (operator, expression, parse_add()) + return expression + + def parse_and(): + nonlocal cursor + values = [parse_comparison()] + while cursor < len(tokens) and tokens[cursor] == 'and': + cursor += 1 + values.append(parse_comparison()) + return combine('and', values) + + def parse_or(): + nonlocal cursor + values = [parse_and()] + while cursor < len(tokens) and tokens[cursor] == 'or': + cursor += 1 + values.append(parse_and()) + return combine('or', values) + + try: + normalized = parse_or() + if cursor != len(tokens): + raise ValueError('unsupported trailing CHECK expression') + return normalized + except ValueError: + return ('unparsed', re.sub(r'\s+', '', text.replace('!=', '<>'))) + + +def _docker_depth_check_constraint_problems(conn): + problems = [] + for table, expected_constraints in DOCKER_DEPTH_EXPERIMENT_CHECK_SPECS.items(): + if not conn.table_exists(table): + continue + constraints = conn.table_check_constraints(table) + for name, expected_expression in expected_constraints.items(): + constraint = constraints.get(name) + if ( + not constraint + or not constraint.get('valid', True) + or _normalized_check_expression(constraint.get('expression')) + != _normalized_check_expression(expected_expression) + ): + problems.append(f'check constraint {table}.{name}') + return problems + + +def _docker_depth_check_clause(table, name): + expression = DOCKER_DEPTH_EXPERIMENT_CHECK_SPECS[table][name] + return f'CONSTRAINT {name} CHECK ({expression})' + + +def _index_usable(index): + return bool(index and index.get('valid', True) and index.get('ready', True) and index.get('live', True)) + + +def _normalized_trigger_sql(value): + return re.sub(r'[\s"]', '', str(value or '').lower()) + + +def _runtime_audit_trigger_problems(conn): + if not conn.table_exists('runtime_audit_events'): + return [] + triggers = conn.table_triggers('runtime_audit_events') + if conn.is_postgres: + expected = { + 'runtime_audit_events_reject_mutation': ( + 'before', 'update', 'delete', 'runtime_audit_events', + 'executefunctionreject_runtime_audit_event_mutation()', + ), + 'runtime_audit_events_reject_truncate': ( + 'before', 'truncate', 'runtime_audit_events', + 'executefunctionreject_runtime_audit_event_mutation()', + ), + } + else: + expected = { + 'runtime_audit_events_reject_update': ( + 'beforeupdateonruntime_audit_events', + "raise(abort,'runtime_audit_eventsisappend-only')", + ), + 'runtime_audit_events_reject_delete': ( + 'beforedeleteonruntime_audit_events', + "raise(abort,'runtime_audit_eventsisappend-only')", + ), + } + problems = [] + for name, fragments in expected.items(): + trigger = triggers.get(name) + sql = _normalized_trigger_sql((trigger or {}).get('sql')) + if ( + not trigger + or not trigger.get('enabled', True) + or any(fragment not in sql for fragment in fragments) + or ( + conn.is_postgres + and "raiseexception'runtime_audit_eventsisappend-only'" + not in _normalized_trigger_sql(trigger.get('function_sql')) + ) + ): + problems.append(f'trigger {name}') + return problems + + +def _column_generates_id(details, postgres): + if not details or not details.get('primary_key') or details.get('generated'): + return False + if not postgres: + return str(details.get('type') or '').strip().lower() == 'integer' + identity = str(details.get('identity') or '') + sequence = str(details.get('sequence') or '') + default_sql = str(details.get('default') or '') + if identity in ('a', 'd'): + return bool(sequence) + return bool(sequence and re.search(r'\bnextval\s*\(', default_sql, re.IGNORECASE)) + + +def utc_now_iso(): + return datetime.now(timezone.utc).isoformat(timespec='seconds') + + +def _worker_contract_timestamp(value): + parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.astimezone(timezone.utc).isoformat( + timespec='seconds' + ).replace('+00:00', 'Z') + + +def _canonical_runtime_json(value): + try: + return json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + allow_nan=False, + ) + except (TypeError, ValueError) as exc: + raise ValueError('runtime authority identity must be canonical JSON data') from exc + + +def _runtime_operation_id(value): + if not isinstance(value, str) or len(value) != 36: + raise ValueError('runtime operation ID must be a canonical UUID') + try: + parsed = uuid.UUID(value) + except (AttributeError, TypeError, ValueError) as exc: + raise ValueError('runtime operation ID must be a canonical UUID') from exc + if parsed.int == 0 or str(parsed) != value: + raise ValueError('runtime operation ID must be a canonical UUID') + return value + + +def _runtime_actor(value): + if not isinstance(value, str) or not 1 <= len(value) <= 256: + raise ValueError('runtime operation actor is invalid') + if any(ord(character) < 32 or ord(character) == 127 for character in value): + raise ValueError('runtime operation actor is invalid') + return value + + +def _runtime_revision(value): + if ( + isinstance(value, bool) or not isinstance(value, int) + or not 0 <= value <= RUNTIME_CONTROL_MAX_REVISION + ): + raise ValueError('runtime control revision is invalid') + return value + + +def _runtime_control_identity(value): + if not isinstance(value, dict) or set(value) != { + 'discovery_paused', 'dispatch_paused', 'drain_state', 'revision', + }: + raise RuntimeSafetySchemaError('runtime control identity shape is invalid') + if type(value['discovery_paused']) is not bool or type(value['dispatch_paused']) is not bool: + raise RuntimeSafetySchemaError('runtime control pause identity is invalid') + drain_state = value['drain_state'] + if drain_state not in RUNTIME_CONTROL_DRAIN_STATES: + raise RuntimeSafetySchemaError('runtime control drain identity is invalid') + try: + revision = _runtime_revision(value['revision']) + except ValueError as exc: + raise RuntimeSafetySchemaError('runtime control revision identity is invalid') from exc + return { + 'discovery_paused': value['discovery_paused'], + 'dispatch_paused': value['dispatch_paused'], + 'drain_state': drain_state, + 'revision': revision, + } + + +def _runtime_control_identity_json(value): + return _canonical_runtime_json(_runtime_control_identity(value)) + + +def _stored_runtime_control_identity(value): + if not isinstance(value, str) or not value: + raise RuntimeSafetySchemaError('runtime control identity JSON is missing') + try: + parsed = json.loads(value) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise RuntimeSafetySchemaError('runtime control identity JSON is invalid') from exc + canonical = _runtime_control_identity_json(parsed) + if not hmac.compare_digest(canonical, value): + raise RuntimeSafetySchemaError('runtime control identity JSON is not canonical') + return parsed + + +def _runtime_control_state_from_row(row): + if not row: + raise RuntimeSafetySchemaError('runtime operations control singleton is missing') + try: + revision = row['revision'] + discovery_raw = row['discovery_paused'] + dispatch_raw = row['dispatch_paused'] + except (KeyError, TypeError) as exc: + raise RuntimeSafetySchemaError('runtime operations control state is invalid') from exc + if ( + isinstance(revision, bool) or not isinstance(revision, int) + or revision < 0 or revision > RUNTIME_CONTROL_MAX_REVISION + ): + raise RuntimeSafetySchemaError('runtime operations control revision is invalid') + if ( + isinstance(discovery_raw, bool) or not isinstance(discovery_raw, int) + or isinstance(dispatch_raw, bool) or not isinstance(dispatch_raw, int) + or discovery_raw not in (0, 1) or dispatch_raw not in (0, 1) + ): + raise RuntimeSafetySchemaError('runtime operations control pause state is invalid') + drain_state = str(row['drain_state'] or '') + if drain_state not in RUNTIME_CONTROL_DRAIN_STATES: + raise RuntimeSafetySchemaError('runtime operations control drain state is invalid') + actor = str(row['actor'] or '') + try: + _runtime_actor(actor) + except ValueError as exc: + raise RuntimeSafetySchemaError('runtime operations control actor is invalid') from exc + operation_id = row['operation_id'] + if operation_id is not None: + try: + operation_id = _runtime_operation_id(str(operation_id)) + except ValueError as exc: + raise RuntimeSafetySchemaError('runtime operations control operation ID is invalid') from exc + created_at = str(row['created_at'] or '') + updated_at = str(row['updated_at'] or '') + if not created_at or not updated_at: + raise RuntimeSafetySchemaError('runtime operations control timestamps are invalid') + discovery_paused = bool(discovery_raw) + dispatch_paused = bool(dispatch_raw) + draining = drain_state != 'normal' + return { + 'revision': revision, + 'discovery_paused': discovery_paused, + 'dispatch_paused': dispatch_paused, + 'drain_state': drain_state, + 'effective_discovery_paused': discovery_paused or draining, + 'effective_dispatch_paused': dispatch_paused or draining, + 'actor': actor, + 'operation_id': operation_id, + 'created_at': created_at, + 'updated_at': updated_at, + } + + +def _runtime_control_state_identity(state): + return { + 'discovery_paused': state['discovery_paused'], + 'dispatch_paused': state['dispatch_paused'], + 'drain_state': state['drain_state'], + 'revision': state['revision'], + } + + +def _runtime_control_transition(before, action): + before = _runtime_control_identity(before) + after = dict(before) + if before['revision'] >= RUNTIME_CONTROL_MAX_REVISION: + raise RuntimeControlTransitionError('runtime control revision cannot be advanced') + if action == 'control.discovery.pause': + if before['discovery_paused']: + raise RuntimeControlTransitionError( + 'runtime discovery control is already in the requested state' + ) + after['discovery_paused'] = True + elif action == 'control.discovery.resume': + if not before['discovery_paused']: + raise RuntimeControlTransitionError( + 'runtime discovery control is already in the requested state' + ) + after['discovery_paused'] = False + elif action == 'control.dispatch.pause': + if before['dispatch_paused']: + raise RuntimeControlTransitionError( + 'runtime dispatch control is already in the requested state' + ) + after['dispatch_paused'] = True + elif action == 'control.dispatch.resume': + if not before['dispatch_paused']: + raise RuntimeControlTransitionError( + 'runtime dispatch control is already in the requested state' + ) + after['dispatch_paused'] = False + elif action == 'control.drain.start': + if before['drain_state'] != 'normal': + raise RuntimeControlTransitionError('runtime drain is already active') + after['drain_state'] = 'draining' + elif action == 'control.drain.cancel': + if before['drain_state'] not in ('draining', 'drained'): + raise RuntimeControlTransitionError('runtime drain is not active') + after['drain_state'] = 'normal' + elif action == 'control.drain.complete': + if before['drain_state'] != 'draining': + raise RuntimeControlTransitionError('runtime drain is not awaiting completion') + after['drain_state'] = 'drained' + else: + raise ValueError('runtime control action is invalid') + after['revision'] = before['revision'] + 1 + return after + + +def _runtime_audit_event_sha256(payload): + previous = payload.get('previous_event_sha256') + if previous is None: + previous_digest = b'\x00' * 32 + elif re.fullmatch(r'[a-f0-9]{64}', str(previous or '')): + previous_digest = bytes.fromhex(str(previous)) + else: + raise RuntimeSafetySchemaError('runtime audit parent identity is invalid') + encoded = _canonical_runtime_json(payload).encode('ascii') + return hashlib.sha256( + previous_digest + hashlib.sha256(encoded).digest() + ).hexdigest() + + +def _runtime_sha256(value, field): + if not isinstance(value, str) or not re.fullmatch(r'[a-f0-9]{64}', value): + raise ValueError(f'runtime operation {field} is invalid') + return value + + +def _runtime_safe_code(value, field, *, required=False): + if value is None and not required: + return None + allowed = ( + RUNTIME_OPERATION_SAFE_CATEGORIES + if field == 'safe category' else RUNTIME_OPERATION_SAFE_DETAILS + ) + if not isinstance(value, str) or value not in allowed: + raise ValueError(f'runtime operation {field} is invalid') + return value + + +def _runtime_audit_row_hash(row): + event_id = row['id'] + if type(event_id) is not int or event_id < 1: + raise ValueError('audit event ID') + operation_id = row['operation_id'] + if operation_id is not None: + operation_id = str(operation_id) + actor = str(row['actor'] or '') + action = str(row['action'] or '') + target_kind = str(row['target_kind'] or '') + target_ref = str(row['target_ref'] or '') + result = str(row['result'] or '') + safe_category = row['safe_category'] + if safe_category is not None: + safe_category = str(safe_category) + identities = [] + for field in ('before_identity_json', 'after_identity_json'): + raw = row[field] + if raw is None: + identities.append(None) + continue + raw = str(raw) + identities.append(raw) + byte_counts = [] + for field in ('before_bytes', 'after_bytes'): + value = row[field] + if value is not None and ( + type(value) is not int + or value < 0 + ): + raise ValueError('audit byte count') + byte_counts.append(value) + previous_event_id = row['previous_event_id'] + previous_event_sha256 = row['previous_event_sha256'] + if previous_event_id is None: + if previous_event_sha256 is not None: + raise ValueError('audit parent') + else: + if ( + type(previous_event_id) is not int + or previous_event_id < 1 + or previous_event_id >= event_id + ): + raise ValueError('audit parent') + previous_event_sha256 = _runtime_sha256( + str(previous_event_sha256 or ''), 'audit parent hash', + ) + created_at = str(row['created_at'] or '') + event_sha256 = _runtime_sha256( + str(row['event_sha256'] or ''), 'audit event hash', + ) + payload = { + 'schema': 'runtime-audit-event-v1', + 'operation_id': operation_id, + 'actor': actor, + 'action': action, + 'target_kind': target_kind, + 'target_ref': target_ref, + 'result': result, + 'safe_category': safe_category, + 'before_identity_json': identities[0], + 'after_identity_json': identities[1], + 'before_bytes': byte_counts[0], + 'after_bytes': byte_counts[1], + 'previous_event_id': previous_event_id, + 'previous_event_sha256': previous_event_sha256, + 'created_at': created_at, + } + if not hmac.compare_digest( + _runtime_audit_event_sha256(payload), event_sha256, + ): + raise ValueError('audit event hash') + return event_id, event_sha256 + + +def _runtime_expected_identity(value, action): + fields = { + 'active_config_sha256', 'active_secrets_sha256', + 'candidate_config_sha256', 'candidate_secrets_sha256', + } + if not isinstance(value, dict) or set(value) != fields: + raise ValueError('runtime operation expected identity is invalid') + identity = { + 'active_config_sha256': _runtime_sha256( + value['active_config_sha256'], 'active config hash', + ), + 'active_secrets_sha256': _runtime_sha256( + value['active_secrets_sha256'], 'active secrets hash', + ), + 'candidate_config_sha256': value['candidate_config_sha256'], + 'candidate_secrets_sha256': value['candidate_secrets_sha256'], + } + required_candidates = { + 'apply-config': (True, False), + 'apply-secrets': (False, True), + 'apply-both': (True, True), + 'restart': (False, False), + } + if action not in required_candidates: + raise ValueError('runtime operation action is invalid') + for key, required in zip( + ('candidate_config_sha256', 'candidate_secrets_sha256'), + required_candidates[action], + ): + current = identity[key] + if required: + identity[key] = _runtime_sha256(current, key.replace('_', ' ')) + elif current is not None: + raise ValueError('runtime operation expected identity is invalid') + return identity + + +def _runtime_resulting_identity(value): + if value is None: + return None + if not isinstance(value, dict) or set(value) != { + 'active_config_sha256', 'active_secrets_sha256', + }: + raise ValueError('runtime operation resulting identity is invalid') + return { + 'active_config_sha256': _runtime_sha256( + value['active_config_sha256'], 'resulting config hash', + ), + 'active_secrets_sha256': _runtime_sha256( + value['active_secrets_sha256'], 'resulting secrets hash', + ), + } + + +def _runtime_success_identity(expected, action): + return { + 'active_config_sha256': ( + expected['candidate_config_sha256'] + if action in ('apply-config', 'apply-both') + else expected['active_config_sha256'] + ), + 'active_secrets_sha256': ( + expected['candidate_secrets_sha256'] + if action in ('apply-secrets', 'apply-both') + else expected['active_secrets_sha256'] + ), + } + + +def _runtime_source_id(value): + if ( + type(value) is str + and value != 'all' + and re.fullmatch(r'[A-Za-z0-9][A-Za-z0-9_.:@-]{0,127}', value) + ): + return value + return None + + +def _runtime_source_expected_identity(value, action, target_ref): + source_action = RUNTIME_SOURCE_OPERATION_ACTIONS.get(action) + if source_action is None or _runtime_source_id(target_ref) is None or not isinstance(value, dict): + raise ValueError('runtime source operation identity is invalid') + legacy_fields = {'source_id', 'source_action', 'interval_seconds'} + extended_fields = legacy_fields | { + 'mode', 'restart_enabled', 'restart_delay_seconds', + } + if set(value) not in (legacy_fields, extended_fields): + raise ValueError('runtime source operation identity is invalid') + if value.get('source_id') != target_ref or value.get('source_action') != source_action: + raise ValueError('runtime source operation identity is invalid') + interval = value.get('interval_seconds') + if source_action == 'set-interval': + if type(interval) is not int or not 1 <= interval <= 365 * 24 * 60 * 60: + raise ValueError('runtime source operation interval is invalid') + elif interval is not None: + raise ValueError('runtime source operation identity is invalid') + normalized = { + 'source_id': target_ref, + 'source_action': source_action, + 'interval_seconds': interval, + } + if set(value) == legacy_fields: + if source_action not in ('start', 'stop', 'restart', 'pause', 'resume', 'set-interval'): + raise ValueError('runtime source operation identity is invalid') + return normalized + mode = value.get('mode') + restart_enabled = value.get('restart_enabled') + restart_delay = value.get('restart_delay_seconds') + if source_action == 'set-mode': + if mode not in ('loop', 'once', 'repeat'): + raise ValueError('runtime source operation mode is invalid') + elif mode is not None: + raise ValueError('runtime source operation identity is invalid') + if source_action == 'set-restart': + if type(restart_enabled) is not bool: + raise ValueError('runtime source operation restart setting is invalid') + elif restart_enabled is not None: + raise ValueError('runtime source operation identity is invalid') + if source_action == 'set-restart-delay': + if ( + type(restart_delay) is not int + or not 1 <= restart_delay <= 365 * 24 * 60 * 60 + ): + raise ValueError('runtime source operation restart delay is invalid') + elif restart_delay is not None: + raise ValueError('runtime source operation identity is invalid') + if source_action == 'once' and any( + current is not None for current in (mode, restart_enabled, restart_delay) + ): + raise ValueError('runtime source operation identity is invalid') + normalized.update({ + 'mode': mode, + 'restart_enabled': restart_enabled, + 'restart_delay_seconds': restart_delay, + }) + return normalized + + +def _runtime_source_resulting_identity(value, action, target_ref): + source_action = RUNTIME_SOURCE_OPERATION_ACTIONS.get(action) + if ( + source_action is None + or _runtime_source_id(target_ref) is None + or not isinstance(value, dict) + or set(value) != {'source_id', 'source_action', 'outcome'} + or value.get('source_id') != target_ref + or value.get('source_action') != source_action + or value.get('outcome') not in ('completed', 'dependency-blocked') + ): + raise ValueError('runtime source operation result is invalid') + return { + 'source_id': target_ref, + 'source_action': source_action, + 'outcome': value['outcome'], + } + + +def _stored_runtime_source_identity(value, action, target_ref, *, resulting=False): + if not isinstance(value, str) or not value: + raise RuntimeSafetySchemaError('runtime source operation identity is missing') + try: + parsed = json.loads(value) + normalized = ( + _runtime_source_resulting_identity(parsed, action, target_ref) + if resulting else + _runtime_source_expected_identity(parsed, action, target_ref) + ) + canonical = _canonical_runtime_json(normalized) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise RuntimeSafetySchemaError('runtime source operation identity is invalid') from exc + if not hmac.compare_digest(canonical, value): + raise RuntimeSafetySchemaError('runtime source operation identity is not canonical') + return normalized + + +def _runtime_worker_admin_target(action, target_ref): + target_type = RUNTIME_WORKER_ADMIN_ACTION_TARGETS.get(action) + if target_type is None or not isinstance(target_ref, str): + raise ValueError('runtime worker admin operation target is invalid') + if target_type == 'deferred-queue': + if target_ref != 'deferred-queue': + raise ValueError('runtime worker admin operation target is invalid') + elif not re.fullmatch(r'[A-Za-z0-9][A-Za-z0-9_.:@-]{0,127}', target_ref): + raise ValueError('runtime worker admin operation target is invalid') + return target_ref + + +def _runtime_worker_admin_expected_identity(value, action, target_ref): + target_ref = _runtime_worker_admin_target(action, target_ref) + if ( + not isinstance(value, dict) + or set(value) != {'action', 'target_ref', 'request_sha256'} + or value.get('action') != action + or value.get('target_ref') != target_ref + ): + raise ValueError('runtime worker admin operation identity is invalid') + return { + 'action': action, + 'target_ref': target_ref, + 'request_sha256': _runtime_sha256( + value.get('request_sha256'), 'worker admin request hash', + ), + } + + +def _runtime_worker_admin_resulting_identity(value, action, target_ref): + target_ref = _runtime_worker_admin_target(action, target_ref) + if ( + not isinstance(value, dict) + or set(value) != {'action', 'target_ref', 'outcome', 'affected_count'} + or value.get('action') != action + or value.get('target_ref') != target_ref + or value.get('outcome') != 'completed' + or type(value.get('affected_count')) is not int + or not 0 <= value['affected_count'] <= 10000 + ): + raise ValueError('runtime worker admin operation result is invalid') + return { + 'action': action, + 'target_ref': target_ref, + 'outcome': 'completed', + 'affected_count': value['affected_count'], + } + + +def _stored_runtime_worker_admin_identity( + value, action, target_ref, *, resulting=False, +): + if not isinstance(value, str) or not value: + raise RuntimeSafetySchemaError('runtime worker admin operation identity is missing') + try: + parsed = json.loads(value) + normalized = ( + _runtime_worker_admin_resulting_identity(parsed, action, target_ref) + if resulting else + _runtime_worker_admin_expected_identity(parsed, action, target_ref) + ) + canonical = _canonical_runtime_json(normalized) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise RuntimeSafetySchemaError('runtime worker admin operation identity is invalid') from exc + if not hmac.compare_digest(canonical, value): + raise RuntimeSafetySchemaError('runtime worker admin operation identity is not canonical') + return normalized + + +def _runtime_document_expected_identity(value, action, target_ref): + document = RUNTIME_DOCUMENT_ACTION_TARGETS.get(action) + if ( + document is None or target_ref != document + or not isinstance(value, dict) + or set(value) != { + 'document', 'active_config_sha256', 'active_secrets_sha256', + 'candidate_config_sha256', 'candidate_secrets_sha256', + 'candidate_after_sha256', + 'candidate_before_bytes', 'candidate_after_bytes', + 'candidate_before_present', + } + or value.get('document') != document + ): + raise ValueError('runtime document operation identity is invalid') + before_bytes = value.get('candidate_before_bytes') + after_bytes = value.get('candidate_after_bytes') + before_present = value.get('candidate_before_present') + if ( + type(before_bytes) is not int or type(after_bytes) is not int + or type(before_present) is not bool + or not 0 <= before_bytes <= 4 * 1024 * 1024 + or not 0 <= after_bytes <= 4 * 1024 * 1024 + ): + raise ValueError('runtime document operation byte counts are invalid') + return { + 'document': document, + 'active_config_sha256': _runtime_sha256( + value.get('active_config_sha256'), 'active config hash', + ), + 'active_secrets_sha256': _runtime_sha256( + value.get('active_secrets_sha256'), 'active secrets hash', + ), + 'candidate_config_sha256': _runtime_sha256( + value.get('candidate_config_sha256'), 'candidate config hash', + ), + 'candidate_secrets_sha256': _runtime_sha256( + value.get('candidate_secrets_sha256'), 'candidate secrets hash', + ), + 'candidate_after_sha256': _runtime_sha256( + value.get('candidate_after_sha256'), 'candidate after hash', + ), + 'candidate_before_bytes': before_bytes, + 'candidate_after_bytes': after_bytes, + 'candidate_before_present': before_present, + } + + +def _runtime_document_resulting_identity(value, action, target_ref): + document = RUNTIME_DOCUMENT_ACTION_TARGETS.get(action) + if ( + document is None or target_ref != document + or not isinstance(value, dict) + or set(value) != {'document', 'outcome', 'candidate_sha256', 'written'} + or value.get('document') != document + or value.get('outcome') != 'completed' + or type(value.get('written')) is not bool + ): + raise ValueError('runtime document operation result is invalid') + return { + 'document': document, + 'outcome': 'completed', + 'candidate_sha256': _runtime_sha256( + value.get('candidate_sha256'), 'candidate result hash', + ), + 'written': value['written'], + } + + +def _stored_runtime_document_identity(value, action, target_ref, *, resulting=False): + if not isinstance(value, str) or not value: + raise RuntimeSafetySchemaError('runtime document operation identity is missing') + try: + parsed = json.loads(value) + normalized = ( + _runtime_document_resulting_identity(parsed, action, target_ref) + if resulting else + _runtime_document_expected_identity(parsed, action, target_ref) + ) + canonical = _canonical_runtime_json(normalized) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise RuntimeSafetySchemaError('runtime document operation identity is invalid') from exc + if not hmac.compare_digest(canonical, value): + raise RuntimeSafetySchemaError('runtime document operation identity is not canonical') + return normalized + + +def _runtime_managed_file_target(root_id, relative_path, target_ref): + if ( + not isinstance(root_id, str) + or root_id != target_ref + or not re.fullmatch(r'[a-z][a-z0-9-]{0,63}', root_id) + or not isinstance(relative_path, str) + ): + raise ValueError('runtime managed file operation target is invalid') + try: + encoded_path = relative_path.encode('utf-8', errors='strict') + except UnicodeEncodeError as exc: + raise ValueError('runtime managed file operation target is invalid') from exc + components = relative_path.split('/') + if ( + not encoded_path or len(encoded_path) > 4096 + or relative_path.startswith('/') or '\\' in relative_path or '\x00' in relative_path + or len(components) > 32 + or any( + not component or component in ('.', '..') + or component.startswith('.truf-managed-file-') + or re.fullmatch(r'[A-Za-z]:.*', component) + or len(component.encode('utf-8')) > 255 + for component in components + ) + ): + raise ValueError('runtime managed file operation target is invalid') + return root_id, relative_path + + +def _runtime_optional_sha256(value, label): + return None if value is None else _runtime_sha256(value, label) + + +def _runtime_optional_byte_count(value, label): + if value is None: + return None + if type(value) is not int or not 0 <= value <= 64 * 1024 * 1024: + raise ValueError(f'{label} is invalid') + return value + + +def _runtime_managed_file_expected_identity(value, action, target_ref): + if ( + action not in RUNTIME_MANAGED_FILE_ACTIONS + or not isinstance(value, dict) + or set(value) != { + 'root_id', 'relative_path', 'expected_sha256', + 'proposed_sha256', 'proposed_byte_count', + } + ): + raise ValueError('runtime managed file operation identity is invalid') + root_id, relative_path = _runtime_managed_file_target( + value.get('root_id'), value.get('relative_path'), target_ref, + ) + expected_sha256 = _runtime_optional_sha256( + value.get('expected_sha256'), 'managed file expected hash', + ) + proposed_sha256 = _runtime_optional_sha256( + value.get('proposed_sha256'), 'managed file proposed hash', + ) + proposed_byte_count = _runtime_optional_byte_count( + value.get('proposed_byte_count'), 'managed file proposed byte count', + ) + if ( + action == 'files.create' + and (expected_sha256 is not None or proposed_sha256 is None or proposed_byte_count is None) + or action == 'files.replace' + and (expected_sha256 is None or proposed_sha256 is None or proposed_byte_count is None) + or action == 'files.delete' + and (expected_sha256 is None or proposed_sha256 is not None or proposed_byte_count is not None) + ): + raise ValueError('runtime managed file operation identity is invalid') + return { + 'root_id': root_id, + 'relative_path': relative_path, + 'expected_sha256': expected_sha256, + 'proposed_sha256': proposed_sha256, + 'proposed_byte_count': proposed_byte_count, + } + + +def _runtime_managed_file_resulting_identity(value, action, target_ref): + if ( + action not in RUNTIME_MANAGED_FILE_ACTIONS + or not isinstance(value, dict) + or set(value) != { + 'root_id', 'relative_path', 'outcome', + 'before_sha256', 'before_byte_count', + 'after_sha256', 'after_byte_count', 'written', + } + or value.get('outcome') != 'completed' + or type(value.get('written')) is not bool + ): + raise ValueError('runtime managed file operation result is invalid') + root_id, relative_path = _runtime_managed_file_target( + value.get('root_id'), value.get('relative_path'), target_ref, + ) + before_sha256 = _runtime_optional_sha256( + value.get('before_sha256'), 'managed file before hash', + ) + after_sha256 = _runtime_optional_sha256( + value.get('after_sha256'), 'managed file after hash', + ) + before_byte_count = _runtime_optional_byte_count( + value.get('before_byte_count'), 'managed file before byte count', + ) + after_byte_count = _runtime_optional_byte_count( + value.get('after_byte_count'), 'managed file after byte count', + ) + if ( + action == 'files.create' + and (before_sha256 is not None or before_byte_count is not None + or after_sha256 is None or after_byte_count is None or not value['written']) + or action == 'files.replace' + and (before_sha256 is None or after_sha256 is None or after_byte_count is None) + or action == 'files.delete' + and (before_sha256 is None or after_sha256 is not None or after_byte_count is not None + or not value['written']) + ): + raise ValueError('runtime managed file operation result is invalid') + return { + 'root_id': root_id, + 'relative_path': relative_path, + 'outcome': 'completed', + 'before_sha256': before_sha256, + 'before_byte_count': before_byte_count, + 'after_sha256': after_sha256, + 'after_byte_count': after_byte_count, + 'written': value['written'], + } + + +def _stored_runtime_managed_file_identity( + value, action, target_ref, *, resulting=False, +): + if not isinstance(value, str) or not value: + raise RuntimeSafetySchemaError('runtime managed file operation identity is missing') + try: + parsed = json.loads(value) + normalized = ( + _runtime_managed_file_resulting_identity(parsed, action, target_ref) + if resulting else + _runtime_managed_file_expected_identity(parsed, action, target_ref) + ) + canonical = _canonical_runtime_json(normalized) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise RuntimeSafetySchemaError( + 'runtime managed file operation identity is invalid' + ) from exc + if not hmac.compare_digest(canonical, value): + raise RuntimeSafetySchemaError( + 'runtime managed file operation identity is not canonical' + ) + return normalized + + +def _runtime_result_json_is_too_deep(text): + depth = 0 + in_string = False + escaped = False + for character in text: + if in_string: + if escaped: + escaped = False + elif character == '\\': + escaped = True + elif character == '"': + in_string = False + elif character == '"': + in_string = True + elif character in '[{': + depth += 1 + if depth > RUNTIME_AGENT_RESULT_MAX_DEPTH: + return True + elif character in ']}' and depth: + depth -= 1 + return False + + +def _runtime_result_envelope(payload, max_bytes): + raw = text = parsed = None + try: + if type(payload) is not bytes: + return None, None, 'runtime operation result must be exact bytes' + if ( + isinstance(max_bytes, bool) or not isinstance(max_bytes, int) + or not 1 <= max_bytes <= RUNTIME_AGENT_RESULT_MAX_BYTES + ): + return None, None, 'runtime operation result bound is invalid' + raw = payload + if not raw or len(raw) > max_bytes: + return None, None, 'runtime operation result exceeds its byte bound' + digest = hashlib.sha256(raw).hexdigest() + try: + text = raw.decode('utf-8', errors='strict') + except UnicodeDecodeError: + return None, None, 'runtime operation result is not valid UTF-8' + if _runtime_result_json_is_too_deep(text): + return None, None, 'runtime operation result is invalid JSON' + + def reject_duplicates(pairs): + result = {} + for key, value in pairs: + if key in result: + raise ValueError('duplicate') + result[key] = value + return result + + try: + parsed = json.loads( + text, object_pairs_hook=reject_duplicates, + parse_constant=lambda value: (_ for _ in ()).throw(ValueError('constant')), + ) + except (TypeError, ValueError, json.JSONDecodeError, RecursionError): + return None, None, 'runtime operation result is invalid JSON' + if not isinstance(parsed, dict) or set(parsed) != { + 'schema', 'operation_id', 'action', 'result', 'safe_category', + 'safe_detail', 'resulting_identity', + }: + return None, None, 'runtime operation result shape is invalid' + if parsed['schema'] != 1 or type(parsed['schema']) is not int: + return None, None, 'runtime operation result schema is invalid' + try: + operation_id = _runtime_operation_id(parsed['operation_id']) + action = str(parsed['action']) if isinstance(parsed['action'], str) else '' + if action not in RUNTIME_ASYNC_ACTIONS: + raise ValueError('runtime operation action is invalid') + result = str(parsed['result']) if isinstance(parsed['result'], str) else '' + if result not in RUNTIME_ASYNC_TERMINAL_RESULTS: + raise ValueError('runtime operation result state is invalid') + category = _runtime_safe_code( + parsed['safe_category'], 'safe category', required=result != 'succeeded', + ) + detail = _runtime_safe_code(parsed['safe_detail'], 'safe detail') + if result == 'succeeded' and (category is not None or detail is not None): + raise ValueError('successful runtime operation result must not have an error') + identity = _runtime_resulting_identity(parsed['resulting_identity']) + if result in ('succeeded', 'rolled_back') and identity is None: + raise ValueError('runtime operation result identity is required') + except ValueError as exc: + return None, None, str(exc) + return { + 'operation_id': operation_id, + 'action': action, + 'result': result, + 'safe_category': category, + 'safe_detail': detail, + 'resulting_identity': identity, + }, digest, None + finally: + payload = raw = text = parsed = None + + +def fixed_lease_window(lease_seconds, now=None): + issued_at = now or datetime.now(timezone.utc) + if issued_at.tzinfo is None: + issued_at = issued_at.replace(tzinfo=timezone.utc) + issued_at = issued_at.astimezone(timezone.utc) + expires_at = issued_at + timedelta(seconds=max(60, int(lease_seconds))) + return ( + issued_at.isoformat(timespec='seconds'), + expires_at.isoformat(timespec='seconds'), + ) + + +def compact_delivered_scan_publications(conn, now=None): + """Delivered publication state is represented by absence from the outbox.""" + now_dt = now or datetime.now(timezone.utc) + if isinstance(now_dt, str): + now_dt = parse_time(now_dt) or datetime.now(timezone.utc) + now_text = now_dt.isoformat(timespec='seconds') + conn.execute("DELETE FROM scan_publication_outbox WHERE status = 'delivered'") + return now_text + + +def parse_time(value): + if not value: + return None + try: + parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.astimezone(timezone.utc) + except ValueError: + return None + + +def json_dumps(value): + return json.dumps(value, ensure_ascii=False, default=str, sort_keys=True) + + +def canonical_git_scan_plan_bytes(plan): + try: + encoded = json.dumps( + plan, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('utf-8') + except (TypeError, ValueError) as exc: + raise ValueError('Git scan plan must be canonical JSON data') from exc + if len(encoded) > 16 * 1024: + raise ValueError('Git scan plan exceeds its byte bound') + return encoded + + +def canonical_remote_execution_snapshot_bytes(snapshot): + if not isinstance(snapshot, dict): + raise ValueError('remote execution snapshot must be a JSON object') + # Imported lazily to avoid scanner_db <-> scan_execution import recursion. + from scan_execution import normalize_remote_execution_snapshot + snapshot = normalize_remote_execution_snapshot(snapshot) + try: + encoded = json.dumps( + snapshot, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('utf-8') + except (TypeError, ValueError) as exc: + raise ValueError('remote execution snapshot must be canonical JSON data') from exc + if len(encoded) > 64 * 1024: + raise ValueError('remote execution snapshot exceeds its byte bound') + return encoded + + +def stored_remote_execution_snapshot(reservation): + stored_json = reservation['remote_execution_snapshot_json'] + stored_sha256 = reservation['remote_execution_snapshot_sha256'] + if stored_json is None and stored_sha256 is None: + return None + if not stored_json or not re.fullmatch(r'[a-f0-9]{64}', str(stored_sha256 or '')): + raise ScanEventConflictError('remote execution snapshot identity is incomplete') + try: + value = json.loads(str(stored_json)) + except (TypeError, ValueError) as exc: + raise ScanEventConflictError('remote execution snapshot JSON is invalid') from exc + encoded = canonical_remote_execution_snapshot_bytes(value) + if encoded.decode('ascii') != str(stored_json) or not hmac.compare_digest( + hashlib.sha256(encoded).hexdigest(), str(stored_sha256), + ): + raise ScanEventConflictError('remote execution snapshot identity is invalid') + return value + + +def stored_git_scan_plan(reservation): + stored_json = reservation['git_scan_plan_json'] + stored_sha256 = reservation['git_scan_plan_sha256'] + if stored_json is None and stored_sha256 is None: + return None + if stored_json is None or stored_sha256 is None: + raise RuntimeSafetySchemaError('Git scan reservation plan identity is incomplete') + try: + stored_bytes = str(stored_json).encode('ascii') + plan = json.loads(stored_bytes.decode('ascii')) + except (UnicodeEncodeError, UnicodeDecodeError, json.JSONDecodeError) as exc: + raise RuntimeSafetySchemaError('stored Git scan plan JSON is invalid') from exc + if ( + hashlib.sha256(stored_bytes).hexdigest() != str(stored_sha256) + or canonical_git_scan_plan_bytes(plan) != stored_bytes + ): + raise RuntimeSafetySchemaError('stored Git scan plan identity is invalid') + return plan + + +def _optional_mapping_value(value, key): + try: + return value[key] + except (KeyError, IndexError): + return None + + +def remote_assignment_execution_plan(reservation, *, snapshot=None): + # Imported lazily to avoid scanner_db <-> scan_execution import recursion. + from scan_execution import ScanExecutionError + + try: + if str(reservation['assignment_kind']) != 'remote': + raise ScanEventConflictError('execution plan requires a remote assignment') + snapshot = ( + stored_remote_execution_snapshot(reservation) + if snapshot is None else snapshot + ) + if snapshot is None: + raise ScanEventConflictError('remote assignment has no execution snapshot') + encoded = canonical_remote_execution_snapshot_bytes(snapshot) + snapshot = json.loads(encoded.decode('ascii')) + source = str(reservation['source']) + platform = str(reservation['platform']) + target = str(reservation['target']) + normalized_target = str(reservation['normalized_target']) + effective_sha256 = str( + reservation['remote_effective_config_sha256'] or '' + ) + if ( + snapshot['credential_ref']['source'] != source + or snapshot['execution']['source'] != platform + or snapshot['compatibility']['effective_config_sha256'] + != effective_sha256 + or normalize_target(target, platform) != normalized_target + ): + raise ScanEventConflictError( + 'remote assignment execution context conflicts with its reservation' + ) + kind = snapshot['planning']['kind'] + git_plan = stored_git_scan_plan(reservation) + has_docker_plan = any( + _optional_mapping_value(reservation, name) is not None + for name in ('docker_layer_plan_json', 'docker_layer_plan_sha256') + ) + if kind == 'exact_git_v1': + if source not in ('github', 'gitlab') or platform != source or has_docker_plan: + raise ScanEventConflictError( + 'remote Git execution context conflicts with its reservation' + ) + if git_plan is not None and normalize_target( + git_plan.get('repo_url'), platform, + ) != normalized_target: + raise ScanEventConflictError( + 'remote Git plan target conflicts with its reservation' + ) + return { + 'kind': kind, + 'execution_target': target, + 'bound_plan': git_plan, + } + if git_plan is not None or has_docker_plan: + raise ScanEventConflictError( + 'direct execution context has an unexpected bound plan' + ) + if kind == 'docker_direct_v1': + if source != 'dockerhub' or platform != 'docker': + raise ScanEventConflictError( + 'remote Docker execution context conflicts with its reservation' + ) + parsed = parse_dockerhub_digest_target(target) + if parsed['normalized_target'] != normalized_target: + raise ScanEventConflictError( + 'remote Docker target conflicts with its reservation' + ) + return { + 'kind': kind, + 'execution_target': parsed['image'], + 'bound_plan': None, + } + if kind == 'huggingface_space_v1': + if source != 'huggingface' or platform != 'huggingface': + raise ScanEventConflictError( + 'remote HuggingFace context conflicts with its reservation' + ) + execution_target = normalize_huggingface_space_id(target) + if execution_target.lower() != normalized_target: + raise ScanEventConflictError( + 'remote HuggingFace target conflicts with its reservation' + ) + return { + 'kind': kind, + 'execution_target': execution_target, + 'bound_plan': None, + } + raise ScanEventConflictError('remote assignment planning kind is unsupported') + except ScanEventConflictError: + raise + except ( + KeyError, IndexError, TypeError, ValueError, + RuntimeSafetySchemaError, ScanExecutionError, + ) as exc: + raise ScanEventConflictError( + 'remote assignment execution plan is invalid' + ) from exc + + +def validate_result_git_scan_plan(reservation, metadata): + plan = stored_git_scan_plan(reservation) + metadata_plan = metadata.get('git_scan_plan') + if plan is None: + if metadata_plan is not None: + raise ScanEventConflictError('result has an unbound Git scan plan') + return None + if not isinstance(metadata_plan, dict): + raise ScanEventConflictError('result is missing its bound Git scan plan') + if canonical_git_scan_plan_bytes(metadata_plan) != canonical_git_scan_plan_bytes(plan): + raise ScanEventConflictError('result Git scan plan conflicts with its reservation') + return plan + + +def validate_remote_result_execution_plan(reservation, result_metadata): + if not isinstance(result_metadata, dict): + raise ScanEventConflictError('remote bundle is missing result metadata') + plan = remote_assignment_execution_plan(reservation) + kind = plan['kind'] + if kind == 'exact_git_v1': + if any( + result_metadata.get(name) is not None + for name in ('docker_layer_plan', 'docker_layer_execution') + ): + raise ScanEventConflictError( + 'remote Git result has unexpected Docker execution metadata' + ) + validate_result_git_scan_plan(reservation, result_metadata) + return plan + if any( + result_metadata.get(name) is not None + for name in ( + 'git_scan_plan', 'git_scan_execution', + 'docker_layer_plan', 'docker_layer_execution', + ) + ): + raise ScanEventConflictError( + 'remote direct result has unexpected planning metadata' + ) + return plan + + +def validate_git_resolution(resolved): + resolved = dict(resolved or {}) + provider = str(resolved.get('provider') or '').strip().lower() + if provider not in ('github', 'gitlab'): + raise ValueError('Git scan provider must be github or gitlab') + branch = str(resolved.get('branch') or '') + ref = str(resolved.get('ref') or '') + if not branch or ref != f'refs/heads/{branch}' or len(ref) > 1024: + raise ValueError('Git scan resolution has an invalid branch ref') + if ( + branch.startswith(('/', '.')) or branch.endswith(('/', '.', '.lock')) + or '..' in branch or '@{' in branch or '\\' in branch + or re.search(r'[\x00-\x20\x7f~^:?*\[]', branch) + or any(part in ('', '.', '..') or part.endswith('.lock') for part in branch.split('/')) + ): + raise ValueError('Git scan resolution has an unsafe branch ref') + head_sha = str(resolved.get('head_sha') or '').strip().lower() + if not re.fullmatch(r'[a-f0-9]{40}|[a-f0-9]{64}', head_sha): + raise ValueError('Git scan resolution has an invalid head SHA') + repo_url = str(resolved.get('repo_url') or '').strip() + repo_path = str(resolved.get('repo_path') or '').strip() + expected_host = f'{provider}.com' + parsed = urlsplit(repo_url) + if ( + parsed.scheme != 'https' or (parsed.hostname or '').lower() != expected_host + or parsed.username is not None or parsed.password is not None + or parsed.port is not None or parsed.query or parsed.fragment + or not repo_path or len(repo_path) > 1024 + or parsed.path != f'/{repo_path}.git' + or repo_path.startswith('/') or repo_path.endswith('/') or '\\' in repo_path + or re.search(r'%(?:2f|5c)', repo_path, flags=re.IGNORECASE) + or any(part in ('', '.', '..') for part in repo_path.split('/')) + ): + raise ValueError('Git scan resolution has an invalid canonical repository') + ref_source = str(resolved.get('ref_source') or '') + if ref_source not in ('explicit', 'provider_default'): + raise ValueError('Git scan resolution has an invalid ref source') + return { + 'provider': provider, + 'repo_url': repo_url, + 'repo_path': repo_path, + 'branch': branch, + 'ref': ref, + 'head_sha': head_sha, + 'ref_source': ref_source, + } + + +def matching_git_coverage(reservation, metadata, queue_status, error_count): + plan = validate_result_git_scan_plan(reservation, metadata) + if plan is None: + return False, None, None + stored_sha256 = reservation['git_scan_plan_sha256'] + execution = metadata.get('git_scan_execution') + if queue_status != 'done' or int(error_count or 0) != 0: + return False, None, None + if not isinstance(execution, dict): + raise ScanEventConflictError('successful Git result is missing execution evidence') + if execution.get('success') is not True or execution.get('pinned') is not True: + return False, None, None + if str(execution.get('plan_sha256') or '') != str(stored_sha256): + raise ScanEventConflictError('Git execution evidence has a conflicting plan hash') + if 'coverage_complete' in execution: + if execution['coverage_complete'] is not True: + return False, None, None + elif metadata.get('degraded') or metadata.get('warnings'): + return False, None, None + plan_mode = str(plan.get('mode') or '') + execution_mode = str(execution.get('mode') or '') + valid_execution = ( + (plan_mode == 'baseline' and execution_mode == 'baseline') + or (plan_mode == 'delta' and execution_mode == 'delta') + or ( + plan_mode == 'delta' and execution_mode == 'baseline_reset' + and execution.get('continuity_reset') is True + ) + or (plan_mode == 'noop' and execution_mode == 'noop') + ) + if not valid_execution: + return False, None, None + return True, str(plan['ref']), str(plan['head_sha']) + + +DOCKER_LAYER_PLAN_MAX_BYTES = 1024 * 1024 +DOCKER_LAYER_MAX_DESCRIPTORS = 2048 +# Exact duplicate-position attribution is all-or-nothing per result bundle. +DOCKER_DEPTH_MAX_ATTRIBUTION_ROWS_PER_BUNDLE = 100000 +DOCKER_BLOB_LEASE_MARGIN_SEC = 300 +DOCKER_LAYER_SUPPORTED_MEDIA_TYPES = frozenset(( + 'application/vnd.oci.image.layer.v1.tar', + 'application/vnd.oci.image.layer.v1.tar+gzip', + 'application/vnd.oci.image.layer.v1.tar+zstd', + 'application/vnd.docker.image.rootfs.diff.tar', + 'application/vnd.docker.image.rootfs.diff.tar.gzip', +)) +DOCKER_CONFIG_MEDIA_TYPES = frozenset(( + 'application/vnd.oci.image.config.v1+json', + 'application/vnd.docker.container.image.v1+json', +)) +DOCKER_LAYER_LIMIT_KEYS = frozenset(( + 'config_max_bytes', 'layer_max_bytes', 'image_max_bytes', 'max_layers', + 'archive_max_size_bytes', 'archive_max_depth', 'archive_timeout_sec', + 'blob_timeout_sec', 'filesystem_concurrency', 'blob_max_attempts', +)) +DOCKER_ADAPTIVE_SELECTOR_VERSION = 'docker-adaptive-payload-v1' +DOCKER_ADAPTIVE_EXECUTION_VERSION = 'docker-layer-execution-v4' +DOCKER_ADAPTIVE_SHADOW_EVALUATOR_VERSION = 'docker-adaptive-shadow-v1' +DOCKER_ADAPTIVE_GATE_MIN_CONTROLS = 50 +DOCKER_ADAPTIVE_GATE_MAX_CONTROLS = 100 +DOCKER_ADAPTIVE_GATE_ROUTED_RECALL_PPM = 850000 +DOCKER_ADAPTIVE_GATE_SLOT_RATIO_PPM = 400000 +DOCKER_ADAPTIVE_PAYLOAD_CLASSES = frozenset(( + 'config', 'copy_add', 'app_config_run', 'package_run', 'other_run', + 'bulk_data', 'unknown', +)) +DOCKER_ADAPTIVE_LAYER_CLASS_ORDER = ( + 'copy_add', 'app_config_run', 'package_run', 'unknown', 'other_run', + 'bulk_data', +) +DOCKER_ADAPTIVE_CHECKPOINT_KEYS = frozenset(('max_blobs', 'max_bytes')) +DOCKER_ADAPTIVE_SELECTION_REASONS = frozenset(( + 'config_selected', 'already_covered', 'duplicate_digest', + 'unsupported_media_type', 'config_too_large', 'layer_too_large', + 'image_budget_exhausted', 'layer_limit_exhausted', + *(f'selected_{payload_class}' for payload_class in DOCKER_ADAPTIVE_LAYER_CLASS_ORDER), +)) +DOCKER_ADAPTIVE_SHADOW_SELECTION_METRIC_KEYS = ( + 'selected_config', 'selected_copy_add', 'selected_app_config_run', + 'selected_package_run', 'selected_other_run', 'selected_bulk_data', + 'selected_unknown', 'reuse_already_covered', 'reuse_duplicate_digest', + 'omitted_unsupported_media_type', 'omitted_config_too_large', + 'omitted_layer_too_large', 'omitted_image_budget_exhausted', + 'omitted_layer_limit_exhausted', 'adaptive_checkpoints', +) +DOCKER_ADAPTIVE_SHADOW_OMISSION_METRIC_KEYS = tuple( + name for name in DOCKER_ADAPTIVE_SHADOW_SELECTION_METRIC_KEYS + if name.startswith('omitted_') +) + + +def validate_docker_adaptive_policy_hashes( + scan_policy_sha256, execution_policy_sha256, selection_policy_sha256, +): + values = { + 'scan_policy_sha256': str(scan_policy_sha256 or '').lower(), + 'execution_policy_sha256': str(execution_policy_sha256 or '').lower(), + 'selection_policy_sha256': str(selection_policy_sha256 or '').lower(), + } + for label, value in values.items(): + if not re.fullmatch(r'[a-f0-9]{64}', value): + raise ValueError(f'Docker adaptive {label} must be a lowercase SHA-256') + return values + + +def docker_adaptive_shadow_gate_metrics(values): + names = ( + 'completed_pairs', 'full_routed_count', 'adaptive_routed_count', + 'routed_intersection_count', 'full_detector_count', + 'adaptive_detector_count', 'detector_intersection_count', + 'full_slot_ms', 'adaptive_slot_ms', 'omitted_descriptor_count', + 'failure_count', 'privacy_violation_count', 'safety_regression_count', + ) + metrics = {} + for name in names: + raw = values.get(name, 0) + if isinstance(raw, bool): + raise ValueError(f'Docker adaptive shadow {name} must be an integer') + try: + value = int(raw) + except (TypeError, ValueError, OverflowError) as exc: + raise ValueError(f'Docker adaptive shadow {name} must be an integer') from exc + if value < 0 or value > 9223372036854775807: + raise ValueError(f'Docker adaptive shadow {name} is outside the supported range') + metrics[name] = value + if metrics['routed_intersection_count'] > min( + metrics['full_routed_count'], metrics['adaptive_routed_count'], + ): + raise ValueError('Docker adaptive routed intersection exceeds its evidence sets') + if metrics['detector_intersection_count'] > min( + metrics['full_detector_count'], metrics['adaptive_detector_count'], + ): + raise ValueError('Docker adaptive detector intersection exceeds its evidence sets') + full_routed = metrics['full_routed_count'] + full_slot_ms = metrics['full_slot_ms'] + metrics['routed_recall_ppm'] = ( + metrics['routed_intersection_count'] * 1000000 // full_routed + if full_routed else 0 + ) + metrics['slot_ratio_ppm'] = ( + (metrics['adaptive_slot_ms'] * 1000000 + full_slot_ms - 1) // full_slot_ms + if full_slot_ms else 1000001 + ) + return metrics + + +def validate_docker_adaptive_shadow_selection_metrics(value): + if isinstance(value, str): + try: + value = json.loads(value) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise ValueError('Docker adaptive shadow selection metrics are invalid JSON') from exc + if not isinstance(value, dict) or set(value) != set(DOCKER_ADAPTIVE_SHADOW_SELECTION_METRIC_KEYS): + raise ValueError('Docker adaptive shadow selection metrics have invalid fields') + metrics = {} + for name in DOCKER_ADAPTIVE_SHADOW_SELECTION_METRIC_KEYS: + raw = value[name] + if isinstance(raw, bool): + raise ValueError(f'Docker adaptive shadow selection metric {name} must be an integer') + try: + number = int(raw) + except (TypeError, ValueError, OverflowError) as exc: + raise ValueError( + f'Docker adaptive shadow selection metric {name} must be an integer' + ) from exc + if number < 0 or number > 9223372036854775807: + raise ValueError( + f'Docker adaptive shadow selection metric {name} is outside the supported range' + ) + metrics[name] = number + return metrics + + +def docker_content_media_class(kind, media_type): + kind = str(kind or '').strip().lower() + media_type = str(media_type or '').strip().lower() + if kind == 'config' and media_type in DOCKER_CONFIG_MEDIA_TYPES: + return 'config-json' + if kind == 'layer' and media_type in DOCKER_LAYER_SUPPORTED_MEDIA_TYPES: + if media_type.endswith('+zstd'): + return 'layer-zstd' + if media_type.endswith('+gzip') or media_type.endswith('.gzip'): + return 'layer-gzip' + return 'layer-tar' + return f'{kind}:{media_type}' + + +def _docker_sha256_digest(value, label='Docker content digest'): + digest = str(value or '').strip().lower() + if not re.fullmatch(r'sha256:[a-f0-9]{64}', digest): + raise ValueError(f'{label} is invalid') + return digest + + +def _dockerhub_image_parts(target): + parsed = parse_docker_target(target) + image = str(parsed['image']).lower() + image_name, manifest_digest = image.rsplit('@', 1) + manifest_digest = _docker_sha256_digest(manifest_digest, 'Docker manifest digest') + parts = image_name.split('/') + if len(parts) > 1 and ('.' in parts[0] or ':' in parts[0] or parts[0] == 'localhost'): + registry = parts.pop(0) + if registry not in ('docker.io', 'index.docker.io', 'registry-1.docker.io'): + raise ValueError('Docker layer scanning only supports Docker Hub targets') + if not parts or any(not part for part in parts): + raise ValueError('Docker Hub repository is invalid') + repository = '/'.join(parts) + registry_repository = repository if '/' in repository else f'library/{repository}' + return image, repository, registry_repository, manifest_digest + + +def _normalized_docker_descriptor(value, kind, position=None): + if not isinstance(value, dict) or set(value) != {'digest', 'size', 'media_type'}: + raise ValueError(f'Docker {kind} descriptor has an invalid shape') + digest = _docker_sha256_digest(value.get('digest'), f'Docker {kind} digest') + try: + size = int(value.get('size')) + except (TypeError, ValueError) as exc: + raise ValueError(f'Docker {kind} descriptor size is invalid') from exc + if size < 0 or size > 1024 * 1024 * 1024 * 1024: + raise ValueError(f'Docker {kind} descriptor size is outside its hard bound') + media_type = str(value.get('media_type') or '').strip().lower() + if not media_type or len(media_type) > 256 or any(ord(char) < 32 for char in media_type): + raise ValueError(f'Docker {kind} media type is invalid') + result = { + 'digest': digest, + 'size': size, + 'media_type': media_type, + 'kind': kind, + } + if position is not None: + result['position'] = int(position) + return result + + +def validate_docker_layer_resolution(resolved): + required = { + 'version', 'image', 'repository', 'manifest_digest', 'platform_os', + 'platform_arch', 'manifest_media_type', 'config', 'layers', + } + if not isinstance(resolved, dict) or set(resolved) != required or resolved.get('version') != 1: + raise ValueError('Docker layer resolution has an invalid shape') + image, _, registry_repository, target_digest = _dockerhub_image_parts(resolved.get('image')) + manifest_digest = _docker_sha256_digest( + resolved.get('manifest_digest'), 'Docker resolved manifest digest', + ) + if manifest_digest != target_digest: + raise ValueError('Docker resolved manifest conflicts with the immutable image target') + repository = str(resolved.get('repository') or '').strip().lower() + if repository != registry_repository: + raise ValueError('Docker resolved repository conflicts with the image target') + platform_os = str(resolved.get('platform_os') or '').strip().lower() + platform_arch = str(resolved.get('platform_arch') or '').strip().lower() + if not re.fullmatch(r'[a-z0-9][a-z0-9_.-]{0,63}', platform_os): + raise ValueError('Docker resolved platform OS is invalid') + if not re.fullmatch(r'[a-z0-9][a-z0-9_.-]{0,63}', platform_arch): + raise ValueError('Docker resolved platform architecture is invalid') + manifest_media_type = str(resolved.get('manifest_media_type') or '').strip().lower() + if not manifest_media_type or len(manifest_media_type) > 256: + raise ValueError('Docker resolved manifest media type is invalid') + config = _normalized_docker_descriptor(resolved.get('config'), 'config', 0) + layers = resolved.get('layers') + if not isinstance(layers, list) or len(layers) > DOCKER_LAYER_MAX_DESCRIPTORS: + raise ValueError('Docker layer descriptor count exceeds its bound') + normalized_layers = [ + _normalized_docker_descriptor(descriptor, 'layer', position) + for position, descriptor in enumerate(layers, 1) + ] + return { + 'version': 1, + 'image': image, + 'repository': repository, + 'manifest_digest': manifest_digest, + 'platform_os': platform_os, + 'platform_arch': platform_arch, + 'manifest_media_type': manifest_media_type, + 'config': config, + 'layers': normalized_layers, + } + + +def validate_docker_layer_limits(limits): + if not isinstance(limits, dict) or set(limits) != DOCKER_LAYER_LIMIT_KEYS: + raise ValueError('Docker layer limits have an invalid shape') + bounds = { + 'config_max_bytes': (1024, 64 * 1024 * 1024), + 'layer_max_bytes': (1024, 8 * 1024 * 1024 * 1024), + 'image_max_bytes': (1024, 32 * 1024 * 1024 * 1024), + 'max_layers': (1, 256), + 'archive_max_size_bytes': (1024, 4 * 1024 * 1024 * 1024), + 'archive_max_depth': (1, 16), + 'archive_timeout_sec': (1, 600), + 'blob_timeout_sec': (10, 3600), + 'filesystem_concurrency': (1, 16), + 'blob_max_attempts': (1, 10), + } + normalized = {} + for key, (minimum, maximum) in bounds.items(): + try: + value = int(limits.get(key)) + except (TypeError, ValueError) as exc: + raise ValueError(f'Docker layer limit {key} must be an integer') from exc + if not minimum <= value <= maximum: + raise ValueError(f'Docker layer limit {key} is outside its hard bound') + normalized[key] = value + return normalized + + +def validate_docker_adaptive_checkpoint(checkpoint): + if not isinstance(checkpoint, dict) or set(checkpoint) != DOCKER_ADAPTIVE_CHECKPOINT_KEYS: + raise ValueError('Docker adaptive checkpoint has an invalid shape') + try: + max_blobs = int(checkpoint.get('max_blobs')) + max_bytes = int(checkpoint.get('max_bytes')) + except (TypeError, ValueError) as exc: + raise ValueError('Docker adaptive checkpoint values must be integers') from exc + if not 1 <= max_blobs <= 32: + raise ValueError('Docker adaptive checkpoint blob count is outside its hard bound') + if not 1024 <= max_bytes <= 32 * 1024 * 1024 * 1024: + raise ValueError('Docker adaptive checkpoint bytes are outside its hard bound') + return {'max_blobs': max_blobs, 'max_bytes': max_bytes} + + +def docker_layer_coverage_policy_sha256(scan_policy_sha256, limits): + scan_policy_sha256 = str(scan_policy_sha256 or '').strip().lower() + if not re.fullmatch(r'[a-f0-9]{64}', scan_policy_sha256): + raise ValueError('Docker layer scanner policy hash is invalid') + limits = validate_docker_layer_limits(limits) + payload = { + 'limits': limits, + 'scan_policy_sha256': scan_policy_sha256, + 'validation_version': DOCKER_ADAPTIVE_EXECUTION_VERSION, + } + return hashlib.sha256(json.dumps( + payload, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).hexdigest() + + +def docker_layer_execution_policy_sha256(scan_policy_sha256, limits): + scan_policy_sha256 = str(scan_policy_sha256 or '').strip().lower() + if not re.fullmatch(r'[a-f0-9]{64}', scan_policy_sha256): + raise ValueError('Docker layer scanner policy hash is invalid') + limits = validate_docker_layer_limits(limits) + payload = { + 'archive_semantics': { + key: limits[key] for key in ( + 'archive_max_size_bytes', 'archive_max_depth', 'archive_timeout_sec', + ) + }, + 'content_media_classes': sorted({ + docker_content_media_class('config', media_type) + for media_type in DOCKER_CONFIG_MEDIA_TYPES + } | { + docker_content_media_class('layer', media_type) + for media_type in DOCKER_LAYER_SUPPORTED_MEDIA_TYPES + }), + 'scan_policy_sha256': scan_policy_sha256, + 'version': DOCKER_ADAPTIVE_EXECUTION_VERSION, + } + return hashlib.sha256(json.dumps( + payload, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).hexdigest() + + +def docker_layer_selection_policy_sha256(limits): + limits = validate_docker_layer_limits(limits) + payload = { + 'class_order': list(DOCKER_ADAPTIVE_LAYER_CLASS_ORDER), + 'limits': { + key: limits[key] for key in ( + 'config_max_bytes', 'layer_max_bytes', 'image_max_bytes', 'max_layers', + ) + }, + 'supported_config_media_types': sorted(DOCKER_CONFIG_MEDIA_TYPES), + 'supported_layer_media_types': sorted(DOCKER_LAYER_SUPPORTED_MEDIA_TYPES), + 'tie_breaks': ['position_desc', 'compressed_size_asc', 'digest_asc'], + 'version': DOCKER_ADAPTIVE_SELECTOR_VERSION, + } + return hashlib.sha256(json.dumps( + payload, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).hexdigest() + + +def canonical_docker_layer_plan_bytes(plan): + if not isinstance(plan, dict): + raise ValueError('Docker layer plan must be an object') + payload = json.dumps( + plan, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + if len(payload) > DOCKER_LAYER_PLAN_MAX_BYTES: + raise ValueError('Docker layer plan exceeds its byte bound') + return payload + + +def docker_layer_canary_selected(manifest_digest, basis_points): + manifest_digest = _docker_sha256_digest(manifest_digest, 'Docker canary manifest digest') + basis_points = max(0, min(10000, int(basis_points or 0))) + bucket = int(hashlib.sha256(manifest_digest.encode('ascii')).hexdigest()[:8], 16) % 10000 + return bucket < basis_points + + +def docker_adaptive_canary_selected(manifest_digest, selection_policy_sha256, basis_points): + manifest_digest = _docker_sha256_digest( + manifest_digest, 'Docker adaptive canary manifest digest', + ) + selection_policy_sha256 = str(selection_policy_sha256 or '').strip().lower() + if not re.fullmatch(r'[a-f0-9]{64}', selection_policy_sha256): + raise ValueError('Docker adaptive selector policy hash is invalid') + basis_points = max(0, min(10000, int(basis_points or 0))) + material = ( + f'docker-adaptive-canary-v1:{manifest_digest}:{selection_policy_sha256}' + ).encode('ascii') + bucket = int(hashlib.sha256(material).hexdigest()[:8], 16) % 10000 + return bucket < basis_points + + +def validate_docker_layer_plan(plan): + common_required = { + 'version', 'image', 'repository', 'manifest_digest', 'platform_os', + 'platform_arch', 'manifest_media_type', 'limits', + 'selection_policy_sha256', 'scan_policy_sha256', 'descriptors', + } + if not isinstance(plan, dict): + raise ValueError('Docker layer plan has an invalid shape') + version = plan.get('version') + required = common_required if version == 1 else common_required | { + 'selector_version', 'execution_policy_sha256', 'checkpoint', + } + if version not in (1, 2) or set(plan) != required: + raise ValueError('Docker layer plan has an invalid shape') + image, _, repository, manifest_digest = _dockerhub_image_parts(plan.get('image')) + if str(plan.get('repository') or '') != repository: + raise ValueError('Docker layer plan repository conflicts with its image') + if _docker_sha256_digest(plan.get('manifest_digest')) != manifest_digest: + raise ValueError('Docker layer plan manifest conflicts with its image') + limits = validate_docker_layer_limits(plan.get('limits')) + selection_policy_sha256 = str(plan.get('selection_policy_sha256') or '').lower() + expected_selection_hash = ( + hashlib.sha256(json.dumps( + limits, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).hexdigest() + if version == 1 else docker_layer_selection_policy_sha256(limits) + ) + scan_policy_sha256 = str(plan.get('scan_policy_sha256') or '').lower() + if selection_policy_sha256 != expected_selection_hash: + raise ValueError('Docker layer plan selection policy hash is invalid') + if not re.fullmatch(r'[a-f0-9]{64}', scan_policy_sha256): + raise ValueError('Docker layer plan scanner policy hash is invalid') + selector_version = None + execution_policy_sha256 = None + checkpoint = None + if version == 2: + selector_version = str(plan.get('selector_version') or '') + if selector_version != DOCKER_ADAPTIVE_SELECTOR_VERSION: + raise ValueError('Docker adaptive selector version is invalid') + execution_policy_sha256 = str( + plan.get('execution_policy_sha256') or '' + ).lower() + if execution_policy_sha256 != docker_layer_execution_policy_sha256( + scan_policy_sha256, limits, + ): + raise ValueError('Docker adaptive execution policy hash is invalid') + checkpoint = validate_docker_adaptive_checkpoint(plan.get('checkpoint')) + descriptors = plan.get('descriptors') + if not isinstance(descriptors, list) or not descriptors or len(descriptors) > DOCKER_LAYER_MAX_DESCRIPTORS + 1: + raise ValueError('Docker layer plan descriptor count is invalid') + positions = set() + normalized_descriptors = [] + allowed_coverage = { + 'selected', 'leased', 'shared_pending', 'covered', 'terminal_failed', 'skipped', + } + for descriptor in descriptors: + expected = { + 'digest', 'size', 'media_type', 'kind', 'position', 'selected', + 'selection_reason', 'coverage_state', 'lease_token', 'attempt', + 'max_attempts', + } + if version == 2: + expected.add('payload_class') + if not isinstance(descriptor, dict) or set(descriptor) != expected: + raise ValueError('Docker layer plan descriptor has an invalid shape') + kind = str(descriptor.get('kind') or '') + position = int(descriptor.get('position')) + if kind not in ('config', 'layer') or position < 0 or position in positions: + raise ValueError('Docker layer plan descriptor identity is invalid') + if (position == 0) != (kind == 'config'): + raise ValueError('Docker layer plan configuration position is invalid') + positions.add(position) + normalized = _normalized_docker_descriptor({ + 'digest': descriptor.get('digest'), + 'size': descriptor.get('size'), + 'media_type': descriptor.get('media_type'), + }, kind, position) + selected = descriptor.get('selected') + coverage_state = str(descriptor.get('coverage_state') or '') + reason = str(descriptor.get('selection_reason') or '') + lease_token = descriptor.get('lease_token') + attempt = int(descriptor.get('attempt')) + max_attempts = int(descriptor.get('max_attempts')) + if max_attempts < 1 or max_attempts > 100 or attempt < 0 or attempt > max_attempts: + raise ValueError('Docker layer plan descriptor attempt bounds are invalid') + if not isinstance(selected, bool) or coverage_state not in allowed_coverage: + raise ValueError('Docker layer plan descriptor state is invalid') + if not reason or len(reason) > 64 or not re.fullmatch(r'[a-z0-9_]+', reason): + raise ValueError('Docker layer plan selection reason is invalid') + payload_class = None + if version == 2: + payload_class = str(descriptor.get('payload_class') or '') + if payload_class not in DOCKER_ADAPTIVE_PAYLOAD_CLASSES: + raise ValueError('Docker layer plan payload class is invalid') + if (kind == 'config') != (payload_class == 'config'): + raise ValueError('Docker layer plan payload class conflicts with descriptor kind') + if reason not in DOCKER_ADAPTIVE_SELECTION_REASONS: + raise ValueError('Docker adaptive selection reason is invalid') + if coverage_state == 'leased': + if not isinstance(lease_token, str) or not 32 <= len(lease_token) <= 128: + raise ValueError('Docker layer plan lease token is invalid') + elif lease_token is not None: + raise ValueError('Docker layer plan has an unexpected lease token') + if selected != (coverage_state != 'skipped'): + raise ValueError('Docker layer plan selected state is inconsistent') + normalized.update({ + 'selected': selected, + 'selection_reason': reason, + 'coverage_state': coverage_state, + 'lease_token': lease_token, + 'attempt': attempt, + 'max_attempts': max_attempts, + }) + if version == 2: + normalized['payload_class'] = payload_class + normalized_descriptors.append(normalized) + if positions != set(range(len(descriptors))): + raise ValueError('Docker layer plan positions are not contiguous') + platform_os = str(plan.get('platform_os') or '').lower() + platform_arch = str(plan.get('platform_arch') or '').lower() + manifest_media_type = str(plan.get('manifest_media_type') or '').lower() + if not re.fullmatch(r'[a-z0-9][a-z0-9_.-]{0,63}', platform_os): + raise ValueError('Docker layer plan platform OS is invalid') + if not re.fullmatch(r'[a-z0-9][a-z0-9_.-]{0,63}', platform_arch): + raise ValueError('Docker layer plan platform architecture is invalid') + if not manifest_media_type or len(manifest_media_type) > 256: + raise ValueError('Docker layer plan manifest media type is invalid') + normalized_descriptors.sort(key=lambda item: item['position']) + normalized_plan = { + 'version': version, + 'image': image, + 'repository': repository, + 'manifest_digest': manifest_digest, + 'platform_os': platform_os, + 'platform_arch': platform_arch, + 'manifest_media_type': manifest_media_type, + 'limits': limits, + 'selection_policy_sha256': selection_policy_sha256, + 'scan_policy_sha256': scan_policy_sha256, + 'descriptors': normalized_descriptors, + } + if version == 2: + normalized_plan.update({ + 'selector_version': selector_version, + 'execution_policy_sha256': execution_policy_sha256, + 'checkpoint': checkpoint, + }) + return normalized_plan + + +def docker_layer_plan_coverage_policy_sha256(plan): + if int(plan.get('version') or 0) == 2: + execution_policy_sha256 = str( + plan.get('execution_policy_sha256') or '' + ).lower() + expected = docker_layer_execution_policy_sha256( + plan.get('scan_policy_sha256'), plan.get('limits'), + ) + if execution_policy_sha256 != expected: + raise ValueError('Docker adaptive execution policy hash is invalid') + return execution_policy_sha256 + return docker_layer_coverage_policy_sha256( + plan.get('scan_policy_sha256'), plan.get('limits'), + ) + + +def select_docker_adaptive_payload(descriptors, payload_classes, limits, covered_digests=()): + limits = validate_docker_layer_limits(limits) + descriptors = list(descriptors or []) + payload_classes = list(payload_classes or []) + if ( + not descriptors + or descriptors[0].get('kind') != 'config' + or len(payload_classes) != len(descriptors) - 1 + or any(value not in DOCKER_ADAPTIVE_PAYLOAD_CLASSES - {'config'} for value in payload_classes) + ): + raise ValueError('Docker adaptive payload classes do not match its descriptors') + covered_digests = set(covered_digests or ()) + entries = {} + config = descriptors[0] + if config['digest'] in covered_digests: + entries[config['position']] = (config, True, 'already_covered', 'config') + elif config['media_type'] not in DOCKER_CONFIG_MEDIA_TYPES: + entries[config['position']] = (config, False, 'unsupported_media_type', 'config') + elif config['size'] > limits['config_max_bytes']: + entries[config['position']] = (config, False, 'config_too_large', 'config') + else: + entries[config['position']] = (config, True, 'config_selected', 'config') + + class_rank = { + payload_class: index + for index, payload_class in enumerate(DOCKER_ADAPTIVE_LAYER_CLASS_ORDER) + } + occurrences = {} + for descriptor, payload_class in zip(descriptors[1:], payload_classes): + occurrences.setdefault(descriptor['digest'], []).append((descriptor, payload_class)) + + representatives = {} + for digest, values in occurrences.items(): + representatives[digest] = min( + values, + key=lambda value: ( + class_rank[value[1]], -value[0]['position'], value[0]['size'], digest, + ), + ) + candidates = [] + decisions = {} + for digest, (descriptor, payload_class) in representatives.items(): + if digest in covered_digests: + decisions[digest] = (True, 'already_covered') + elif descriptor['media_type'] not in DOCKER_LAYER_SUPPORTED_MEDIA_TYPES: + decisions[digest] = (False, 'unsupported_media_type') + elif descriptor['size'] > limits['layer_max_bytes']: + decisions[digest] = (False, 'layer_too_large') + else: + candidates.append((descriptor, payload_class)) + candidates.sort(key=lambda value: ( + class_rank[value[1]], -value[0]['position'], value[0]['size'], value[0]['digest'], + )) + all_fit = ( + len(candidates) <= limits['max_layers'] + and sum(descriptor['size'] for descriptor, _ in candidates) <= limits['image_max_bytes'] + ) + selected_count = 0 + selected_bytes = 0 + for descriptor, payload_class in candidates: + if all_fit or ( + selected_count < limits['max_layers'] + and selected_bytes + descriptor['size'] <= limits['image_max_bytes'] + ): + decisions[descriptor['digest']] = (True, f'selected_{payload_class}') + selected_count += 1 + selected_bytes += descriptor['size'] + elif selected_count >= limits['max_layers']: + decisions[descriptor['digest']] = (False, 'layer_limit_exhausted') + else: + decisions[descriptor['digest']] = (False, 'image_budget_exhausted') + + for digest, values in occurrences.items(): + representative, representative_class = representatives[digest] + selected, reason = decisions[digest] + entries[representative['position']] = ( + representative, selected, reason, representative_class, + ) + for descriptor, payload_class in values: + if descriptor['position'] == representative['position']: + continue + entries[descriptor['position']] = ( + descriptor, + selected, + 'duplicate_digest' if selected else reason, + payload_class, + ) + return [entries[position] for position in sorted(entries)] + + +def stored_docker_layer_plan(reservation): + stored_json = reservation['docker_layer_plan_json'] + stored_sha256 = reservation['docker_layer_plan_sha256'] + if stored_json is None and stored_sha256 is None: + return None, None + if stored_json is None or stored_sha256 is None: + raise RuntimeSafetySchemaError('Docker layer reservation plan identity is incomplete') + try: + stored_bytes = str(stored_json).encode('ascii') + plan = json.loads(stored_bytes.decode('ascii')) + plan = validate_docker_layer_plan(plan) + except (UnicodeEncodeError, UnicodeDecodeError, json.JSONDecodeError, TypeError, ValueError) as exc: + raise RuntimeSafetySchemaError('stored Docker layer plan JSON is invalid') from exc + if ( + canonical_docker_layer_plan_bytes(plan) != stored_bytes + or hashlib.sha256(stored_bytes).hexdigest() != str(stored_sha256) + ): + raise RuntimeSafetySchemaError('stored Docker layer plan identity is invalid') + return plan, str(stored_sha256) + + +def validate_docker_layer_execution(execution, plan, plan_sha256): + if not isinstance(execution, dict) or set(execution) != { + 'version', 'plan_sha256', 'blobs', + }: + raise ValueError('Docker layer execution has an invalid shape') + if ( + execution.get('version') != plan['version'] + or str(execution.get('plan_sha256') or '') != plan_sha256 + ): + raise ValueError('Docker layer execution plan identity is invalid') + records = execution.get('blobs') + if not isinstance(records, list) or len(records) > DOCKER_LAYER_MAX_DESCRIPTORS + 1: + raise ValueError('Docker layer execution record count is invalid') + leased = {} + for descriptor in plan['descriptors']: + if descriptor['coverage_state'] == 'leased': + existing = leased.get(descriptor['digest']) + if existing and existing['lease_token'] != descriptor['lease_token']: + raise ValueError('Docker layer plan duplicates a digest with conflicting leases') + leased[descriptor['digest']] = descriptor + normalized = [] + seen = set() + statuses = {'covered', 'retryable_failed', 'terminal_failed'} + for record in records: + expected = { + 'digest', 'lease_token', 'status', 'verified_bytes', 'transfer_bytes', + 'transfer_duration_ms', 'scan_duration_ms', 'finding_count', 'error_code', + } + if not isinstance(record, dict) or set(record) != expected: + raise ValueError('Docker layer execution record has an invalid shape') + digest = _docker_sha256_digest(record.get('digest'), 'execution blob') + descriptor = leased.get(digest) + if not descriptor or digest in seen: + raise ValueError('Docker layer execution has an unclaimed or duplicate blob') + seen.add(digest) + if str(record.get('lease_token') or '') != descriptor['lease_token']: + raise ValueError('Docker layer execution lease token is invalid') + status = str(record.get('status') or '') + if status not in statuses: + raise ValueError('Docker layer execution status is invalid') + numeric = {} + for key, upper in ( + ('verified_bytes', descriptor['size']), + ('transfer_bytes', descriptor['size']), + ('transfer_duration_ms', 24 * 60 * 60 * 1000), + ('scan_duration_ms', 24 * 60 * 60 * 1000), + ('finding_count', 10_000_000), + ): + value = record.get(key) + if isinstance(value, bool) or not isinstance(value, int) or value < 0 or value > upper: + raise ValueError(f'Docker layer execution {key} is invalid') + numeric[key] = value + error_code = record.get('error_code') + if status == 'covered': + if numeric['verified_bytes'] != descriptor['size'] or error_code is not None: + raise ValueError('successful Docker layer execution is incomplete') + elif ( + not isinstance(error_code, str) + or not re.fullmatch(r'[a-z0-9_]{1,64}', error_code) + ): + raise ValueError('failed Docker layer execution error code is invalid') + normalized.append({ + 'digest': digest, + 'lease_token': descriptor['lease_token'], + 'status': status, + **numeric, + 'error_code': error_code, + }) + if seen != set(leased): + raise ValueError('Docker layer execution does not account for every leased blob') + normalized.sort(key=lambda item: item['digest']) + return { + 'version': plan['version'], + 'plan_sha256': plan_sha256, + 'blobs': normalized, + } + + +def docker_layer_canary_metadata_eligible(metadata, target): + if not isinstance(metadata, dict): + return False + raw_plan = metadata.get('docker_layer_plan') + if raw_plan is None: + scan_meta = metadata.get('scan_meta') + scan_meta = scan_meta if isinstance(scan_meta, dict) else {} + return bool( + scan_meta.get('command_timed_out') or metadata.get('error_class') == 'timeout' + ) + try: + plan = validate_docker_layer_plan(raw_plan) + if plan['image'] != _dockerhub_image_parts(target)[0]: + return False + plan_sha256 = hashlib.sha256(canonical_docker_layer_plan_bytes(plan)).hexdigest() + execution = validate_docker_layer_execution( + metadata.get('docker_layer_execution'), plan, plan_sha256, + ) + except (TypeError, ValueError): + return False + if any( + descriptor['coverage_state'] in ('selected', 'shared_pending', 'retryable_failed') + for descriptor in plan['descriptors'] + ): + return True + return any( + record['status'] == 'retryable_failed' for record in execution['blobs'] + ) + + +def sqlite_lock_error(exc): + message = str(exc or '').lower() + return any(item in message for item in ('locked', 'busy', 'deadlock', 'lock timeout', 'could not serialize access')) + + +def sqlite_lock_retry_delay(attempt): + base = SQLITE_LOCK_RETRY_BASE_SEC * (2 ** min(int(attempt or 0), 5)) + jitter = random.uniform(0, SQLITE_LOCK_RETRY_BASE_SEC) + return min(SQLITE_LOCK_RETRY_MAX_SEC, base + jitter) + + +def safe_json_loads(value): + if not value: + return None + if isinstance(value, (dict, list)): + return value + try: + return json.loads(str(value)) + except (TypeError, ValueError, json.JSONDecodeError): + return None + + +def sha256_text(value): + if value is None: + return '' + value = str(value) + if not value: + return '' + return hashlib.sha256(value.encode('utf-8', errors='replace')).hexdigest() + + +def _identity_mapping(value): + if hasattr(value, 'as_dict'): + value = value.as_dict() + value = dict(value or {}) + required = ('pid', 'creation_time', 'executable') + if any(not value.get(key) for key in required): + raise ValueError('exact producer identity is incomplete') + return { + 'pid': int(value['pid']), + 'creation_time': str(value['creation_time']), + 'executable': os.path.normcase(os.path.realpath(os.path.abspath(str(value['executable'])))), + } + + +def hash_file(path): + if not path or not os.path.exists(path): + return None + digest = hashlib.sha256() + with open(path, 'rb') as f: + for chunk in iter(lambda: f.read(1024 * 1024), b''): + digest.update(chunk) + return digest.hexdigest() + + +def get_database_path(results_dir=None, explicit_path=None): + path = explicit_path or os.getenv('SCANNER_DB_PATH') or os.getenv('SCAN_DB_PATH') + if path: + if results_dir and not os.path.isabs(path): + return os.path.join(results_dir, path) + return path + return os.path.join(results_dir or os.getcwd(), DB_FILENAME) + + +def get_database_url(explicit_url=None): + return explicit_url or database_url_from_env() + + +def database_disabled(): + return str(os.getenv('SCANNER_DB_DISABLED', '')).strip().lower() in ('1', 'true', 'yes', 'on') + + +def redact_config(value): + if isinstance(value, dict): + redacted = {} + for key, item in value.items(): + key_text = str(key).lower() + if any(part in key_text for part in DATABASE_SECRET_KEY_PARTS): + redacted[key] = redact_database_url(item) if item else item + elif any(part in key_text for part in SECRET_KEY_PARTS): + redacted[key] = REDACTED if item else item + else: + redacted[key] = redact_config(item) + return redacted + if isinstance(value, list): + return [redact_config(item) for item in value] + return value + + +def redact_argv(argv): + sensitive_flags = { + '--token', '--docker-token', '--password', '--api-key', '--secret', + '--db-url', '--database-url', '--scanner-db-url', + } + output = [] + hide_next = False + for value in argv or []: + text = str(value) + if hide_next: + output.append(REDACTED) + hide_next = False + continue + flag = text.split('=', 1)[0].lower() + if flag in sensitive_flags: + if '=' in text: + output.append(text.split('=', 1)[0] + '=' + REDACTED) + else: + output.append(text) + hide_next = True + continue + output.append(redact_database_url(text)) + return output + + +def normalize_target(target, source): + target_text = str(target or '').strip() + source = 'docker' if source == 'dockerhub' else str(source or '').lower() + lowered = target_text.lower() + if source == 'postman': + return postman_target_identity(target) + if source == 'docker': + return docker_target_identity(target) + if source == 'github_archive': + try: + data = json.loads(target_text) + repo_url = str(data.get('url') or data.get('repo_url') or target_text).strip().lower() + branch = str(data.get('branch') or '').strip().lower() + if repo_url.endswith('.git'): + repo_url = repo_url[:-4] + repo_url = repo_url.rstrip('/') + if repo_url.startswith('http://'): + repo_url = 'https://' + repo_url[7:] + return f'{repo_url}#{branch}' if branch else repo_url + except Exception: + pass + if source in ('github', 'github_archive', 'gitlab', 'git'): + if lowered.endswith('.git'): + lowered = lowered[:-4] + lowered = lowered.rstrip('/') + if lowered.startswith('http://'): + lowered = 'https://' + lowered[7:] + return lowered + if source in ('npm', 'pypi'): + try: + data = json.loads(target_text) + name = data.get('name') or '' + version = data.get('version') or '' + return f'{source}:{name}@{version}'.lower() + except Exception: + return target_text.split('|', 1)[0].lower() + return lowered + + +def _admin_safe_target(target): + text = str(target or '').strip() + try: + parsed = urlsplit(text) + port = parsed.port + except ValueError: + return '[invalid target]' + if parsed.scheme and parsed.hostname: + host = parsed.hostname + if ':' in host and not host.startswith('['): + host = f'[{host}]' + authority = host + (f':{port}' if port is not None else '') + return f'{parsed.scheme.lower()}://{authority}{parsed.path}'[:2048] + if '@' in text: + text = text.rsplit('@', 1)[-1] + return text[:2048] + + +def _admin_assignment_outcome_sql(reservation='r'): + return f'''CASE {reservation}.remote_resolution_kind + WHEN 'bundle_accepted' THEN 'accepted' + WHEN 'prebundle_report' THEN 'prebundle_failed' + WHEN 'expired' THEN 'expired' + ELSE 'unfinished' END''' + + +def _admin_scan_outcome_sql(reservation='r', scan='s'): + return ( + f"CASE WHEN {reservation}.remote_resolution_kind = 'bundle_accepted' " + f"THEN COALESCE({scan}.status, 'unavailable') ELSE 'unavailable' END" + ) + + +def _validated_admin_worker_filters(value): + filters = dict(value or {}) + allowed = { + 'source', 'worker', 'assignment_outcome', 'scan_outcome', 'phase', + 'category', 'code', 'retryable', 'since', + } + if set(filters) - allowed: + raise ValueError('worker administration filters are invalid') + patterns = { + 'source': r'[a-z0-9][a-z0-9_.-]{0,63}', + 'phase': r'[a-z][a-z0-9_]{0,63}', + 'category': r'[a-z][a-z0-9_]{0,127}', + 'code': r'[A-Za-z0-9][A-Za-z0-9._:-]{0,255}', + } + normalized = {} + for name, pattern in patterns.items(): + if name not in filters: + continue + item = str(filters[name] or '') + if re.fullmatch(pattern, item) is None: + raise ValueError('worker administration filters are invalid') + normalized[name] = item + if 'worker' in filters: + item = str(filters['worker'] or '').strip() + if not 1 <= len(item) <= 128 or '\x00' in item: + raise ValueError('worker administration filters are invalid') + normalized['worker'] = item + if 'assignment_outcome' in filters: + item = str(filters['assignment_outcome'] or '') + if item not in {'accepted', 'prebundle_failed', 'expired', 'unfinished'}: + raise ValueError('worker administration filters are invalid') + normalized['assignment_outcome'] = item + if 'scan_outcome' in filters: + item = str(filters['scan_outcome'] or '') + if item not in {'clean', 'found', 'degraded', 'error', 'skipped', 'unavailable'}: + raise ValueError('worker administration filters are invalid') + normalized['scan_outcome'] = item + if 'retryable' in filters: + if type(filters['retryable']) is not bool: + raise ValueError('worker administration filters are invalid') + normalized['retryable'] = filters['retryable'] + if 'since' in filters: + item = str(filters['since'] or '') + parse_time(item) + normalized['since'] = item + return normalized + + +def _admin_worker_filter_sql(filters, *, diagnostic_alias=None): + filters = _validated_admin_worker_filters(filters) + conditions = [] + parameters = [] + if 'source' in filters: + conditions.append('r.source = ?') + parameters.append(filters['source']) + if 'worker' in filters: + conditions.append('d.device_key = ?') + parameters.append(filters['worker']) + if 'assignment_outcome' in filters: + conditions.append(f'({_admin_assignment_outcome_sql()}) = ?') + parameters.append(filters['assignment_outcome']) + if 'scan_outcome' in filters: + conditions.append(f'({_admin_scan_outcome_sql()}) = ?') + parameters.append(filters['scan_outcome']) + if 'since' in filters: + timestamp_column = f'{diagnostic_alias}.occurred_at' if diagnostic_alias else 'r.remote_issued_at' + conditions.append(f'{timestamp_column} >= ?') + parameters.append(filters['since']) + + diagnostic_conditions = [] + if 'phase' in filters: + if diagnostic_alias: + conditions.append(f'{diagnostic_alias}.phase = ?') + parameters.append(filters['phase']) + else: + conditions.append('''( + p.phase = ? OR EXISTS ( + SELECT 1 FROM worker_diagnostics fd + WHERE fd.reservation_id = r.id AND fd.phase = ? + ) + )''') + parameters.extend((filters['phase'], filters['phase'])) + for name in ('category', 'code'): + if name in filters: + diagnostic_conditions.append(f'fd.{name} = ?') + parameters.append(filters[name]) + if 'retryable' in filters: + diagnostic_conditions.append('fd.retryable = ?') + parameters.append(1 if filters['retryable'] else 0) + if diagnostic_conditions: + if diagnostic_alias: + conditions.extend( + condition.replace('fd.', f'{diagnostic_alias}.') + for condition in diagnostic_conditions + ) + else: + conditions.append('''EXISTS ( + SELECT 1 FROM worker_diagnostics fd + WHERE fd.reservation_id = r.id AND %s + )''' % ' AND '.join(diagnostic_conditions)) + return filters, conditions, parameters + + +def _validated_discovery_retry_source(value): + if not isinstance(value, str) or not re.fullmatch(r'[a-z][a-z0-9_-]{0,63}', value): + raise ValueError('discovery retry source is invalid') + return value + + +def _validated_discovery_retry_query(value): + if ( + not isinstance(value, str) + or not value + or value != value.strip() + or len(value) > DISCOVERY_RETRY_MAX_QUERY_CHARS + or any(ord(character) < 32 or ord(character) == 127 for character in value) + ): + raise ValueError('discovery retry query is invalid') + return value + + +def _validated_discovery_retry_policy(value): + policy = str(value or '') + if not re.fullmatch(r'[a-f0-9]{64}', policy): + raise ValueError('discovery retry policy hash is invalid') + return policy + + +def _validated_discovery_retry_pages(work_kind, page_start, page_end): + work_kind = str(work_kind or '') + if work_kind not in DISCOVERY_RETRY_WORK_KINDS: + raise ValueError('discovery retry work kind is invalid') + if isinstance(page_start, bool) or isinstance(page_end, bool): + raise ValueError('discovery retry page range is invalid') + try: + start = int(page_start) + end = int(page_end) + except (TypeError, ValueError, OverflowError): + raise ValueError('discovery retry page range is invalid') from None + if not 1 <= start <= end <= DISCOVERY_RETRY_MAX_PAGE: + raise ValueError('discovery retry page range is invalid') + if work_kind == 'query' and start != 1: + raise ValueError('query-level discovery retry work must start at page one') + if work_kind == 'page' and start != end: + raise ValueError('page-level discovery retry work must identify one page') + return work_kind, start, end + + +def _validated_discovery_retry_error_category(value, required=False): + category = str(value or '') + if not category and not required: + return None + if category not in DISCOVERY_RETRY_ERROR_CATEGORIES: + raise ValueError('discovery retry error category is invalid') + return category + + +def _validated_discovery_retry_fence(retry_id, lease_owner, lease_token): + if isinstance(retry_id, bool): + raise ValueError('discovery retry lease identity is invalid') + try: + retry_id = int(retry_id) + except (TypeError, ValueError, OverflowError): + raise ValueError('discovery retry lease identity is invalid') from None + owner = str(lease_owner or '') + token = str(lease_token or '') + if ( + retry_id < 1 + or not 1 <= len(owner) <= 128 + or not 1 <= len(token) <= 128 + or any(ord(character) < 32 or ord(character) == 127 for character in owner + token) + ): + raise ValueError('discovery retry lease identity is invalid') + return retry_id, owner, token + + +def _validated_discovery_retry_time(value, now=None): + if value is None: + return None + parsed = parse_time(value) + if not parsed: + raise ValueError('discovery retry time is invalid') + now = now or datetime.now(timezone.utc) + if parsed > now + timedelta(seconds=DISCOVERY_RETRY_MAX_TRUSTED_DELAY_SEC): + raise ValueError('discovery retry time exceeds the trusted delay bound') + return max(parsed, now).isoformat(timespec='seconds') + + +def _discovery_retry_allowlist(configured_query_policies): + if isinstance(configured_query_policies, dict): + entries = list(configured_query_policies.items()) + else: + try: + entries = list(configured_query_policies or ()) + except TypeError: + raise ValueError('discovery retry query policy allowlist is invalid') from None + if len(entries) > DISCOVERY_RETRY_MAX_ALLOWLIST: + raise ValueError('discovery retry query policy allowlist exceeds its bound') + allowed = {} + for entry in entries: + if not isinstance(entry, (list, tuple)) or len(entry) != 2: + raise ValueError('discovery retry query policy allowlist is invalid') + query = _validated_discovery_retry_query(entry[0]) + policy = _validated_discovery_retry_policy(entry[1]) + if query in allowed and allowed[query] != policy: + raise ValueError('discovery retry query policy allowlist conflicts') + allowed[query] = policy + return tuple(sorted(allowed.items())) + + +def discovery_retry_work_key( + source, query, policy_sha256, pass_kind, work_kind, page_start, page_end, + pass_id=None, +): + source = _validated_discovery_retry_source(source) + query = _validated_discovery_retry_query(query) + policy_sha256 = _validated_discovery_retry_policy(policy_sha256) + pass_kind = str(pass_kind or '') + if pass_kind not in DISCOVERY_RETRY_PASS_KINDS: + raise ValueError('discovery retry pass kind is invalid') + work_kind, page_start, page_end = _validated_discovery_retry_pages( + work_kind, page_start, page_end, + ) + identity = [ + source, query, policy_sha256, pass_kind, work_kind, page_start, page_end, + ] + if pass_id is not None: + if isinstance(pass_id, bool): + raise ValueError('discovery retry pass identity is invalid') + try: + pass_id = int(pass_id) + except (TypeError, ValueError, OverflowError): + raise ValueError('discovery retry pass identity is invalid') from None + if not 1 <= pass_id <= 0x7fffffffffffffff: + raise ValueError('discovery retry pass identity is invalid') + identity.append(pass_id) + payload = json.dumps( + identity, + ensure_ascii=True, + separators=(',', ':'), + ).encode('utf-8') + digest = hashlib.sha256(payload).hexdigest() + if pass_id is None: + return digest + return f'{pass_id:016x}{digest[:48]}' + + +def _validated_docker_discovery_observation(value, source, query): + if value is None: + return None + if not isinstance(value, dict): + raise ValueError('DockerHub discovery observation metadata is invalid') + required = { + 'cycle_id', 'query_ordinal', 'query_count', 'page_number', 'page_limit', + 'per_page', 'total_count', 'policy_sha256', 'pass_kind', + 'collection_generation', 'ordered_query_hash', 'query_complete', + } + if set(value) != required: + raise ValueError('DockerHub discovery observation metadata shape is invalid') + numeric = {} + for name in ('query_ordinal', 'query_count', 'page_number', 'page_limit', 'per_page'): + raw = value[name] + if isinstance(raw, bool): + raise ValueError('DockerHub discovery observation bounds are invalid') + try: + numeric[name] = int(raw) + except (TypeError, ValueError, OverflowError): + raise ValueError('DockerHub discovery observation bounds are invalid') from None + if not ( + 1 <= numeric['query_count'] <= 1000 + and 0 <= numeric['query_ordinal'] < numeric['query_count'] + and 1 <= numeric['page_number'] <= DISCOVERY_RETRY_MAX_PAGE + and numeric['page_number'] <= numeric['page_limit'] <= DISCOVERY_RETRY_MAX_PAGE + and 1 <= numeric['per_page'] <= DISCOVERY_RETRY_MAX_REPOSITORIES_PER_PAGE + ): + raise ValueError('DockerHub discovery observation bounds are invalid') + cycle_id = value['cycle_id'] + if cycle_id is not None: + if isinstance(cycle_id, bool): + raise ValueError('DockerHub discovery source cycle identity is invalid') + try: + cycle_id = int(cycle_id) + except (TypeError, ValueError, OverflowError): + raise ValueError('DockerHub discovery source cycle identity is invalid') from None + if cycle_id < 1: + raise ValueError('DockerHub discovery source cycle identity is invalid') + pass_kind = str(value['pass_kind'] or '') + if pass_kind not in DISCOVERY_RETRY_PASS_KINDS: + raise ValueError('DockerHub discovery pass kind is invalid') + ordered_query_hash = str(value['ordered_query_hash'] or '') + if not re.fullmatch(r'[a-f0-9]{64}', ordered_query_hash): + raise ValueError('DockerHub discovery ordered-query hash is invalid') + if not isinstance(value['query_complete'], bool): + raise ValueError('DockerHub discovery query completion evidence is invalid') + total_count = value['total_count'] + if total_count is not None: + if isinstance(total_count, bool): + raise ValueError('DockerHub discovery total-count evidence is invalid') + try: + total_count = int(total_count) + except (TypeError, ValueError, OverflowError): + raise ValueError('DockerHub discovery total-count evidence is invalid') from None + if not 0 <= total_count <= 0x7fffffffffffffff: + raise ValueError('DockerHub discovery total-count evidence is invalid') + if value['query_complete'] and total_count is None: + raise ValueError('DockerHub terminal discovery evidence lacks a total count') + from docker_depth_experiment import DOCKER_DEPTH_COLLECTION_GENERATION + collection_generation = str(value['collection_generation'] or '') + if collection_generation != DOCKER_DEPTH_COLLECTION_GENERATION: + raise ValueError('DockerHub discovery collection generation is invalid') + return { + 'cycle_id': cycle_id, + **numeric, + 'policy_sha256': _validated_discovery_retry_policy(value['policy_sha256']), + 'pass_kind': pass_kind, + 'collection_generation': collection_generation, + 'ordered_query_hash': ordered_query_hash, + 'total_count': total_count, + 'query_complete': value['query_complete'], + 'source': source, + 'query': query, + } + + +def _normalized_dockerhub_repositories(repositories): + try: + offered = list(repositories or ()) + except TypeError: + raise ValueError('DockerHub discovery repositories must be iterable') from None + if len(offered) > DISCOVERY_RETRY_MAX_REPOSITORIES_PER_PAGE: + raise ValueError('DockerHub discovery page exceeds its repository bound') + normalized = [] + observed = [] + ordinals = {} + seen = set() + for ordinal, repository in enumerate(offered, 1): + value = repository.get('repo_name') if isinstance(repository, dict) else repository + if not isinstance(value, str) or not value or value != value.strip(): + raise ValueError('DockerHub discovery repository is invalid') + try: + bare = validate_docker_image_reference(value, require_digest=False) + except (TypeError, ValueError): + raise ValueError('DockerHub discovery repository is invalid') from None + if '@' in bare or ':' in bare.rsplit('/', 1)[-1]: + raise ValueError('DockerHub discovery repository must be a bare repository anchor') + identity = docker_target_identity(bare) + if not identity: + raise ValueError('DockerHub discovery repository is invalid') + observed.append(identity) + if identity in seen: + continue + seen.add(identity) + normalized.append(identity) + ordinals[identity] = ordinal + return len(offered), normalized, ordinals, observed + + +def extract_package_metadata(target, result, source): + source = str(source or '').lower() + package = result.get('package') if isinstance(result, dict) else None + if not package and source in ('npm', 'pypi', 'package_git'): + try: + package = json.loads(str(target).strip()) + except Exception: + package = {} + if not isinstance(package, dict): + package = {} + return { + 'package_name': package.get('name'), + 'package_version': package.get('version'), + 'package_artifact': package.get('artifact') or package.get('tarball') or package.get('repo_url'), + 'repo_url': package.get('repo_url'), + 'repo_provider': package.get('provider'), + 'package_date': package.get('date'), + 'package_filename': package.get('filename'), + 'package_type': package.get('packagetype') or package.get('type'), + 'package_size': package.get('size'), + } + + +def extract_raw_secret(finding): + for key in ('RawV2', 'Raw'): + value = finding.get(key) + if value: + return str(value) + structured = finding.get('StructuredData') + if isinstance(structured, dict): + for value in structured.values(): + if isinstance(value, str) and value: + return value + return '' + + +def extract_redacted_secret(finding): + value = finding.get('Redacted') + if value: + return str(value) + return '***REDACTED***' if extract_raw_secret(finding) else '' + + +def extract_finding_location(finding): + metadata = finding.get('SourceMetadata') or {} + data = metadata.get('Data') if isinstance(metadata, dict) else {} + source_type = '' + details = {} + if isinstance(data, dict): + for key, value in data.items(): + if isinstance(value, dict): + source_type = key + details = value + break + if not isinstance(details, dict): + details = {} + return { + 'source_metadata_type': source_type, + 'file_path': details.get('file') or details.get('path') or details.get('File') or '', + 'line_number': str(details.get('line') or details.get('Line') or ''), + 'commit_hash': details.get('commit') or details.get('commitHash') or details.get('commit_hash') or '', + 'source_timestamp': details.get('timestamp') or details.get('Timestamp') or '', + 'source_metadata_json': json_dumps(metadata) if metadata else '', + } + + +def finding_identity(source, normalized_target, finding): + raw_secret = extract_raw_secret(finding) + secret_hash = sha256_text(raw_secret) + detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '') + location = extract_finding_location(finding) + fallback_hash = sha256_text(json_dumps(finding)) if not secret_hash else '' + detector_secret_hash = sha256_text('|'.join([detector, secret_hash or fallback_hash])) + fingerprint = sha256_text('|'.join([ + str(source or ''), + str(normalized_target or ''), + detector, + secret_hash or fallback_hash, + str(location.get('file_path') or ''), + str(location.get('line_number') or ''), + str(location.get('commit_hash') or ''), + str(bool(finding.get('Verified', False))), + ])) + return raw_secret, secret_hash, detector_secret_hash, fingerprint, location + + +def scanner_context(finding): + context = finding.get('ScannerContext') + return context if isinstance(context, dict) else {} + + +def context_text(finding): + context = scanner_context(finding) + return str(context.get('nearby') or '') + + +def find_first(patterns, text): + for pattern in patterns: + match = re.search(pattern, text or '', re.IGNORECASE | re.MULTILINE) + if match: + return match.group(1) + return '' + + +def split_dockerhub_rawv2(rawv2): + if not rawv2 or ':' not in rawv2: + return '', '' + username, token = rawv2.split(':', 1) + return username, token + + +def split_azure_openai_rawv2(rawv2): + rawv2 = rawv2 or '' + match = re.match(r'^([a-f0-9]{32}):(.+\.openai\.azure\.com)$', rawv2, re.IGNORECASE) + if not match: + return '', '' + return match.group(1), match.group(2) + + +def split_azure_devops_rawv2(raw, rawv2): + raw = raw or '' + rawv2 = rawv2 or '' + if raw and rawv2.startswith(raw) and len(rawv2) > len(raw): + return rawv2[len(raw):] + return '' + + +def confidence_for(finding, complete=False, legacy=False, missing=False): + if finding.get('Verified'): + return 'verified' + if legacy: + return 'legacy_unverified' + if missing: + return 'token_only_missing_context' + if complete: + return 'structured_complete' + return 'unverified' + + +def enrich_finding(finding): + finding = sanitize_postman_finding(finding) + detector = str(finding.get('DetectorName') or '') + detector_key = detector.lower() + raw = str(finding.get('Raw') or '') + rawv2 = str(finding.get('RawV2') or '') + extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {} + analysis = finding.get('AnalysisInfo') if isinstance(finding.get('AnalysisInfo'), dict) else {} + text = context_text(finding) + + out = { + 'provider': '', + 'credential_kind': detector, + 'credential_confidence': confidence_for(finding), + 'required_context_missing': 0, + 'principal': '', + 'username': '', + 'email': '', + 'project_id': '', + 'tenant_id': '', + 'organization': '', + 'registry': '', + 'endpoint': '', + 'scope': '', + 'resource': '', + 'enrichment_json': '', + } + + postman_context = finding.get('PostmanContext') if isinstance(finding.get('PostmanContext'), dict) else {} + if postman_context: + postman_kind = postman_context.get('credential_kind') or detector + is_adc = ( + detector_key == 'gcpapplicationdefaultcredentials' + or str(postman_kind or '').lower() == 'application_default_credentials' + ) + out.update({ + 'provider': postman_context.get('provider') or out['provider'], + 'credential_kind': postman_kind, + 'credential_confidence': postman_context.get('credential_confidence') or confidence_for(finding), + 'endpoint': postman_context.get('endpoint') or postman_context.get('host') or '', + 'resource': '' if is_adc else postman_context.get('json_path') or '', + 'scope': postman_context.get('context_location') or '', + 'username': postman_context.get('variable_name') or '', + 'required_context_missing': 0, + 'enrichment_json': json_dumps(postman_context), + }) + return out + + if detector_key == 'gcp': + data = safe_json_loads(rawv2) + out.update({'provider': 'gcp', 'credential_kind': 'service_account'}) + if isinstance(data, dict): + out.update({ + 'project_id': data.get('project_id') or extra.get('project') or '', + 'principal': data.get('client_email') or analysis.get('principal') or '', + 'username': data.get('client_email') or '', + 'resource': data.get('private_key_id') or '', + 'credential_confidence': confidence_for(finding, complete=True), + 'enrichment_json': json_dumps({ + 'type': data.get('type'), + 'client_id': data.get('client_id'), + 'private_key_id': data.get('private_key_id'), + 'client_x509_cert_url': data.get('client_x509_cert_url'), + }), + }) + return out + + if detector_key == 'gcpapplicationdefaultcredentials': + data = safe_json_loads(text) + if not isinstance(data, dict): + data = {} + project_id = data.get('quota_project_id') or data.get('project_id') or extra.get('project') or '' + out.update({ + 'provider': 'gcp', + 'credential_kind': 'application_default_credentials', + 'principal': data.get('client_id') or '', + 'project_id': project_id, + 'resource': '', + 'required_context_missing': 0 if data else 1, + 'credential_confidence': confidence_for(finding, complete=bool(data), missing=not bool(data)), + 'enrichment_json': json_dumps({ + 'client_id': data.get('client_id'), + 'type': data.get('type'), + 'project_id': data.get('project_id'), + 'quota_project_id': data.get('quota_project_id'), + 'has_client_secret': bool(data.get('client_secret')), + 'has_refresh_token': bool(data.get('refresh_token')), + }), + }) + return out + + if detector_key == 'azurecontainerregistry': + data = safe_json_loads(rawv2) + registry = data.get('username') if isinstance(data, dict) else '' + out.update({ + 'provider': 'azure', + 'credential_kind': 'azure_container_registry', + 'registry': registry or find_first([r'([a-z0-9][a-z0-9-]{1,100}[a-z0-9])\.azurecr\.io'], text), + 'endpoint': (registry + '.azurecr.io') if registry else '', + 'username': registry or '', + 'credential_confidence': confidence_for(finding, complete=bool(registry)), + 'required_context_missing': 0 if registry else 1, + }) + return out + + if detector_key == 'azureopenai': + _, endpoint = split_azure_openai_rawv2(rawv2) + endpoint = endpoint or find_first([r'([a-z0-9-]+\.openai\.azure\.com)'], text) + out.update({ + 'provider': 'azure', + 'credential_kind': 'azure_openai_key', + 'endpoint': endpoint, + 'resource': endpoint.split('.')[0] if endpoint else '', + 'credential_confidence': confidence_for(finding, complete=bool(endpoint), missing=not bool(endpoint)), + 'required_context_missing': 0 if endpoint else 1, + }) + return out + + if detector_key == 'azuredevopspersonalaccesstoken': + org = split_azure_devops_rawv2(raw, rawv2) or find_first([ + r'https?://dev\.azure\.com/([A-Za-z0-9][A-Za-z0-9-]{1,80})', + r'https?://([A-Za-z0-9][A-Za-z0-9-]{1,80})\.visualstudio\.com', + ], text) + out.update({ + 'provider': 'azure', + 'credential_kind': 'azure_devops_pat', + 'organization': org, + 'endpoint': f'https://dev.azure.com/{org}' if org else '', + 'credential_confidence': confidence_for(finding, complete=bool(org), missing=not bool(org)), + 'required_context_missing': 0 if org else 1, + }) + return out + + if detector_key in ('googleai', 'googleaistudio') or (detector_key == 'customregex' and str(extra.get('name') or '').lower() == 'googleaistudio'): + is_aistudio = detector_key == 'googleaistudio' or str(extra.get('name') or '').lower() == 'googleaistudio' + out.update({ + 'provider': 'google', + 'credential_kind': 'google_ai_studio_api_key' if is_aistudio else 'google_ai_api_key', + 'credential_confidence': confidence_for(finding, complete=True), + }) + return out + + if detector_key in ('qwendashscope', 'qwen_dashscope', 'qwen', 'dashscope') or (detector_key == 'customregex' and str(extra.get('name') or '').lower() in ('qwendashscope', 'qwen_dashscope')): + endpoint = find_first([ + r'((?:dashscope(?:-intl|-us)?|cn-hongkong\.dashscope)\.aliyuncs\.com)', + r'(cn-hongkong\.aliyuncs\.com)', + ], text) + out.update({ + 'provider': 'alibaba', + 'credential_kind': 'qwen_dashscope_api_key', + 'endpoint': endpoint, + 'resource': 'dashscope', + 'credential_confidence': confidence_for(finding, complete=True), + }) + return out + + if detector_key in ('kimimoonshot', 'moonshotai', 'moonshot', 'kimi') or (detector_key == 'customregex' and str(extra.get('name') or '').lower() in ('kimimoonshot', 'moonshotai')): + endpoint = find_first([ + r'(api\.moonshot\.ai)', + r'(api\.moonshot\.cn)', + r'(platform\.kimi\.(?:ai|com))', + ], text) + out.update({ + 'provider': 'moonshot', + 'credential_kind': 'kimi_moonshot_api_key', + 'endpoint': endpoint, + 'resource': 'kimi', + 'credential_confidence': confidence_for(finding, complete=True), + }) + return out + + if detector_key == 'zaiglm' or (detector_key == 'customregex' and str(extra.get('name') or '').lower() == 'zaiglm'): + endpoint = find_first([ + r'(api\.z\.ai)', + r'(open\.bigmodel\.cn)', + r'(bigmodel\.cn)', + ], text) + out.update({ + 'provider': 'zai', + 'credential_kind': 'glm_api_key', + 'endpoint': endpoint or 'api.z.ai', + 'resource': 'glm', + 'credential_confidence': confidence_for(finding, complete=True), + }) + return out + + if detector_key == 'groq': + out.update({ + 'provider': 'groq', + 'credential_kind': 'groq_api_key', + 'endpoint': 'api.groq.com', + 'credential_confidence': confidence_for(finding, complete=True), + }) + return out + + if detector_key == 'replicate': + out.update({ + 'provider': 'replicate', + 'credential_kind': 'replicate_api_token', + 'endpoint': 'api.replicate.com', + 'credential_confidence': confidence_for(finding, complete=True), + }) + return out + + if detector_key == 'xai': + out.update({ + 'provider': 'xai', + 'credential_kind': 'xai_api_key', + 'endpoint': 'api.x.ai', + 'credential_confidence': confidence_for(finding, complete=True), + }) + return out + + if detector_key == 'huggingface': + out.update({ + 'provider': 'huggingface', + 'credential_kind': 'huggingface_user_access_token', + 'endpoint': 'huggingface.co', + 'credential_confidence': confidence_for(finding, complete=True), + }) + return out + + if detector_key == 'dockerhub': + username, _ = split_dockerhub_rawv2(rawv2) + username = username or extra.get('hub_username') or analysis.get('username') or find_first([ + r'(?i)(?:docker(?:hub)?[_-]?)?(?:user|username|usr|login|id)\s*[:=]\s*["\']?([a-zA-Z0-9][a-zA-Z0-9_.-]{2,60})', + r'([a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,})', + ], text) + out.update({ + 'provider': 'dockerhub', + 'credential_kind': 'dockerhub_pat', + 'username': username, + 'email': extra.get('hub_email') or (username if '@' in username else ''), + 'scope': extra.get('hub_scope') or '', + 'credential_confidence': confidence_for(finding, complete=bool(username), missing=not bool(username)), + 'required_context_missing': 0 if username else 1, + 'enrichment_json': json_dumps({'2fa_required': extra.get('2fa_required')}), + }) + return out + + if detector_key in ('github', 'githuboauth2'): + legacy = detector_key == 'githuboauth2' or (raw and not raw.startswith(('ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_'))) + out.update({ + 'provider': 'github', + 'credential_kind': 'github_oauth_app' if detector_key == 'githuboauth2' else 'github_token', + 'principal': extra.get('username') or analysis.get('key') or '', + 'scope': extra.get('scopes') or '', + 'credential_confidence': confidence_for(finding, complete=not legacy, legacy=legacy), + }) + return out + + if detector_key == 'gitlab': + legacy = raw and not raw.startswith(('glpat-', 'gloas-', 'glcbt-', 'glimt-', 'glrt-', 'glft-', 'glsoat-')) + out.update({ + 'provider': 'gitlab', + 'credential_kind': 'gitlab_token', + 'principal': analysis.get('key') or '', + 'endpoint': analysis.get('host') or '', + 'credential_confidence': confidence_for(finding, complete=not legacy, legacy=legacy), + }) + return out + + return out + + +def first_error_summary(result): + for error in result.get('errors', []) or []: + for line in str(error).splitlines(): + line = line.strip() + if not line: + continue + try: + payload = json.loads(line) + for key in ('error', 'msg', 'message'): + if payload.get(key): + return str(payload[key])[:500] + except Exception: + pass + return line[:500] + return '' + + +def categorize_error(error): + text = str(error or '').lower() + if 'secondary rate limit' in text or 'abuse' in text: + return 'secondary_rate_limit' + if 'rate limit' in text or 'too many requests' in text or ' 429' in text or '429 ' in text: + return 'rate_limit' + if 'auth_invalid' in text or 'authentication failed' in text or 'unauthorized' in text or '401' in text: + return 'auth_invalid' + if 'auth_forbidden' in text or 'forbidden' in text or 'permission denied' in text or 'access denied' in text or '403' in text: + return 'auth_forbidden' + if 'query_invalid' in text or 'validation failed' in text or '422' in text: + return 'query_invalid' + if 'not found' in text or '404' in text: + return 'not_found' + if any(item in text for item in ('server error', '500', '502', '503', '504')): + return 'server_error' + if any(item in text for item in ('no space left', 'not enough free space', 'disk')): + return 'disk_space' + if any(item in text for item in ('timed out', 'timeout', 'deadline exceeded')): + return 'timeout' + if any(item in text for item in ('extract', 'tar', 'zip', 'gzip', 'invalid header')): + return 'extract' + if any(item in text for item in ('download', 'artifact exceeds', 'fetch')): + return 'download' + if any(item in text for item in ('connection', 'dns', 'tls', 'ssl', 'proxy', 'network')): + return 'network' + if 'api' in text or 'http' in text: + return 'api' + if any(item in text for item in ('trufflehog', 'error processing image', 'failed to clone')): + return 'trufflehog' + return 'unknown' + + +def target_status(result): + if result.get('errors'): + return 'error' + if result.get('skipped'): + return 'skipped' + if result.get('findings'): + return 'found' + if result.get('degraded') or result.get('warnings'): + return 'degraded' + return 'clean' + + +def summarize_results(results): + summary = { + 'scanned_count': len(results or []), + 'clean_count': 0, + 'found_count': 0, + 'skipped_count': 0, + 'error_count': 0, + 'degraded_count': 0, + 'findings_count': 0, + 'verified_findings_count': 0, + 'unique_secrets_count': 0, + 'unique_findings_count': 0, + } + secret_hashes = set() + fingerprints = set() + for result in results or []: + status = target_status(result) + summary[f'{status}_count'] += 1 + findings = result.get('findings') or [] + summary['findings_count'] += len(findings) + summary['verified_findings_count'] += sum(1 for finding in findings if finding.get('Verified', False)) + source = result.get('scan_type') or '' + normalized = normalize_target(result.get('target', ''), source) + for finding in findings: + _, secret_hash, _, fingerprint, _ = finding_identity(source, normalized, finding) + if secret_hash: + secret_hashes.add(secret_hash) + if fingerprint: + fingerprints.add(fingerprint) + summary['unique_secrets_count'] = len(secret_hashes) + summary['unique_findings_count'] = len(fingerprints) + return summary + + +def count_file_lines(path): + if not path or not os.path.exists(path): + return 0 + try: + with open(path, 'r', encoding='utf-8') as f: + return sum(1 for line in f if line.strip()) + except OSError: + return 0 + + +def queue_counts(todo_file=None, checked_file=None): + return { + 'todo_count': count_file_lines(todo_file), + 'checked_count': count_file_lines(checked_file), + 'todo_file': todo_file, + 'checked_file': checked_file, + } + + +class ScannerDB: + def __init__(self, results_dir=None, db_path=None, enabled=True, initialize=True, db_url=None): + explicit_url = get_database_url() if db_url is None else db_url + if is_postgres_url(db_path) and not explicit_url: + explicit_url = db_path + db_path = None + self.url = explicit_url if is_postgres_url(explicit_url) else None + self.postgres_required = bool(explicit_url) + self.path = None if self.url else get_database_path(results_dir, db_path) + self.db_display = redact_database_url(self.url) if self.url else self.path + self.conn = None + self._keycheck_result_columns = None + self._last_claim_expectation = None + self._result_spool_publisher_held = False + self._pipeline_advisory_held = set() + self._runtime_managed_file_execution_held = None + self._target_queue_counts_cache = {} + self._target_queue_counts_retry_after = {} + self.last_error = '' + if explicit_url and not self.url: + logger.error(f'Unsupported database URL scheme: {redact_database_url(explicit_url)}') + self.path = None + return + if not enabled or database_disabled(): + self.path = None + self.url = None + self.db_display = None + return + try: + if not self.url: + parent = os.path.dirname(self.path) + if parent: + os.makedirs(parent, exist_ok=True) + self.conn = self._new_connection(timeout_sec=SQLITE_CONNECT_TIMEOUT_SEC) + if initialize and self.conn.is_sqlite: + self.conn.execute('PRAGMA journal_mode=WAL') + if initialize and self.conn.is_sqlite: + self.conn.execute('PRAGMA synchronous=NORMAL') + if initialize and self.conn.is_sqlite: + self.initialize_schema() + except Exception as e: + logger.error(f'Observability DB disabled after init failure: {e}') + self.conn = None + + @property + def enabled(self): + return self.conn is not None + + @classmethod + def host_agent_authority(cls): + database = cls(enabled=False) + database.postgres_required = True + database.db_display = 'fixed host-agent PostgreSQL authority' + connection = connect_host_agent_postgres() + try: + connection.execute("SET application_name = 'truf-host-agent'") + except BaseException: + connection.close() + raise + database.conn = connection + return database + + def _new_connection(self, timeout_sec=None): + if self.url: + conn = connect_postgres(self.url) + conn.execute("SET application_name = 'truf-observability'") + return conn + conn = connect_sqlite(self.path, timeout_sec=max(1, int(timeout_sec or SQLITE_CONNECT_TIMEOUT_SEC))) + conn.execute(f'PRAGMA busy_timeout={SQLITE_BUSY_TIMEOUT_MS}') + conn.execute('PRAGMA foreign_keys=ON') + return conn + + def _reset_connection(self): + held = self._runtime_managed_file_execution_held + self._runtime_managed_file_execution_held = None + connection = self.conn + self.conn = None + cancellation = None + if held is not None and held[2] is not None: + try: + held[2].release() + except RuntimeError: + pass + except BaseException as exc: + cancellation = exc + try: + if connection: + connection.close() + except BaseException as exc: + if not isinstance(exc, Exception) and cancellation is None: + cancellation = exc + self._result_spool_publisher_held = False + self._pipeline_advisory_held.clear() + if cancellation is not None: + raise cancellation + try: + self.conn = self._new_connection() + self._runtime_safety_schema_validated = False + return True + except Exception as e: + logger.error(f'Observability DB reconnect failed: {e}') + self.conn = None + return False + + def close(self): + held = self._runtime_managed_file_execution_held + self._runtime_managed_file_execution_held = None + connection = self.conn + self.conn = None + self._result_spool_publisher_held = False + self._pipeline_advisory_held.clear() + failure = None + if held is not None and held[2] is not None: + try: + held[2].release() + except RuntimeError: + pass + except BaseException as exc: + failure = exc + try: + if connection: + connection.close() + except BaseException as exc: + if ( + failure is None + or isinstance(failure, Exception) and not isinstance(exc, Exception) + ): + failure = exc + if failure is not None: + raise failure + + def set_application_name(self, value): + if not self.conn or not self.conn.is_postgres: + return True + name = str(value or '').strip() + if not name or any(character in name for character in ('\x00', '\r', '\n')): + raise ValueError('PostgreSQL application name is invalid') + self.conn.execute("SELECT set_config('application_name', ?, false)", (name[:63],)) + self.conn.commit() + return True + + def acquire_runtime_managed_file_execution(self, operation_id): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + if self._runtime_managed_file_execution_held is not None: + raise RuntimeOperationTransitionError( + 'runtime managed file execution lock is already held' + ) + lock_key = int.from_bytes( + hashlib.sha256(operation_id.encode('ascii')).digest()[:4], + 'big', signed=True, + ) + if self.conn.is_postgres: + self.conn.execute( + 'SELECT pg_catalog.pg_advisory_lock(?, ?)', + (RUNTIME_MANAGED_FILE_ADVISORY_CLASS, lock_key), + ) + self.conn.commit() + lock = None + else: + lock = _RUNTIME_MANAGED_FILE_SQLITE_LOCKS[ + lock_key % len(_RUNTIME_MANAGED_FILE_SQLITE_LOCKS) + ] + lock.acquire() + self._runtime_managed_file_execution_held = ( + operation_id, lock_key, lock, + ) + return True + + def release_runtime_managed_file_execution(self, operation_id): + operation_id = _runtime_operation_id(operation_id) + held = self._runtime_managed_file_execution_held + if held is None or held[0] != operation_id: + raise RuntimeOperationTransitionError( + 'runtime managed file execution lock is not held' + ) + self._runtime_managed_file_execution_held = None + if self.conn.is_postgres: + row = self.conn.execute( + '''SELECT pg_catalog.pg_advisory_unlock(?, ?) AS released''', + (RUNTIME_MANAGED_FILE_ADVISORY_CLASS, held[1]), + ).fetchone() + self.conn.commit() + if not row or not bool(row['released']): + raise RuntimeSafetySchemaError( + 'runtime managed file execution lock release failed' + ) + else: + held[2].release() + return True + + def acquire_pipeline_lease( + self, worker_name, supervisor_instance_id, owner_identity, + lease_seconds=30, initial_state='starting', + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('pipeline singleton leases require PostgreSQL') + worker_name = str(worker_name or '') + advisory_object = PIPELINE_ADVISORY_OBJECTS.get(worker_name) + if advisory_object is None: + raise ValueError('unknown pipeline worker lease name') + if worker_name in self._pipeline_advisory_held: + raise RuntimeError(f'pipeline advisory lease is already held: {worker_name}') + identity = _identity_mapping(owner_identity) + token = secrets.token_urlsafe(32) + now = utc_now_iso() + expires = datetime.fromtimestamp( + time.time() + max(5, int(lease_seconds)), timezone.utc, + ).isoformat(timespec='seconds') + acquired = False + try: + row = self.conn.execute( + 'SELECT pg_try_advisory_lock(?, ?) AS acquired', + (PIPELINE_ADVISORY_CLASS, advisory_object), + ).fetchone() + acquired = bool(row and row['acquired']) + if not acquired: + self.conn.commit() + return None + current = self.conn.execute( + 'SELECT generation FROM pipeline_leases WHERE worker_name = ? FOR UPDATE', + (worker_name,), + ).fetchone() + generation = int(current['generation'] or 0) + 1 if current else 1 + self.conn.execute( + '''INSERT INTO pipeline_leases( + worker_name, generation, lease_token, supervisor_instance_id, + owner_pid, owner_creation_time, owner_executable, state, + acquired_at, heartbeat_at, lease_expires_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(worker_name) DO UPDATE SET + generation = excluded.generation, + lease_token = excluded.lease_token, + supervisor_instance_id = excluded.supervisor_instance_id, + owner_pid = excluded.owner_pid, + owner_creation_time = excluded.owner_creation_time, + owner_executable = excluded.owner_executable, + state = excluded.state, + acquired_at = excluded.acquired_at, + heartbeat_at = excluded.heartbeat_at, + lease_expires_at = excluded.lease_expires_at, + last_error = NULL, + updated_at = excluded.updated_at''', + ( + worker_name, generation, token, str(supervisor_instance_id or ''), + identity['pid'], identity['creation_time'], identity['executable'], + initial_state, now, now, expires, now, + ), + ) + self.conn.commit() + self._pipeline_advisory_held.add(worker_name) + return {'worker_name': worker_name, 'generation': generation, 'lease_token': token} + except Exception: + self.conn.rollback() + if acquired: + try: + self.conn.execute( + 'SELECT pg_advisory_unlock(?, ?)', + (PIPELINE_ADVISORY_CLASS, advisory_object), + ) + self.conn.commit() + except Exception: + self.conn.rollback() + raise + + def heartbeat_pipeline_lease( + self, worker_name, generation, lease_token, lease_seconds=30, + state='ready', error='', + ): + if not self.conn or not self.conn.is_postgres or worker_name not in self._pipeline_advisory_held: + return False + now = utc_now_iso() + expires = datetime.fromtimestamp( + time.time() + max(5, int(lease_seconds)), timezone.utc, + ).isoformat(timespec='seconds') + cursor = self.conn.execute( + '''UPDATE pipeline_leases SET state = ?, heartbeat_at = ?, lease_expires_at = ?, + last_error = ?, updated_at = ? + WHERE worker_name = ? AND generation = ? AND lease_token = ?''', + ( + state, now, expires, first_line(error, 1000) if error else None, now, + worker_name, int(generation), str(lease_token), + ), + ) + self.conn.commit() + return int(cursor.rowcount or 0) == 1 + + def release_pipeline_lease(self, worker_name, generation, lease_token, state='released', error=''): + if not self.conn or not self.conn.is_postgres: + return False + advisory_object = PIPELINE_ADVISORY_OBJECTS.get(worker_name) + if advisory_object is None or worker_name not in self._pipeline_advisory_held: + return False + now = utc_now_iso() + try: + cursor = self.conn.execute( + '''UPDATE pipeline_leases SET state = ?, heartbeat_at = ?, lease_expires_at = NULL, + lease_token = NULL, last_error = ?, updated_at = ? + WHERE worker_name = ? AND generation = ? AND lease_token = ?''', + ( + state, now, first_line(error, 1000) if error else None, now, + worker_name, int(generation), str(lease_token), + ), + ) + unlocked = self.conn.execute( + 'SELECT pg_advisory_unlock(?, ?) AS released', + (PIPELINE_ADVISORY_CLASS, advisory_object), + ).fetchone() + self.conn.commit() + if unlocked and unlocked['released']: + self._pipeline_advisory_held.discard(worker_name) + return int(cursor.rowcount or 0) == 1 and bool(unlocked and unlocked['released']) + except Exception: + self.conn.rollback() + raise + + def pipeline_worker_health(self, worker_name='result_ingester', supervisor_instance_id=None): + if not self.conn or not self.conn.is_postgres: + return {'healthy': False, 'reason': 'PostgreSQL is unavailable'} + row = self.conn.execute( + '''SELECT worker_name, generation, state, lease_expires_at, + supervisor_instance_id, heartbeat_at, last_error + FROM pipeline_leases WHERE worker_name = ?''', + (worker_name,), + ).fetchone() + self.conn.commit() + healthy = bool( + row and row['state'] == 'ready' and row['lease_expires_at'] + and str(row['lease_expires_at']) > utc_now_iso() + and ( + not supervisor_instance_id + or str(row['supervisor_instance_id'] or '') == str(supervisor_instance_id) + ) + ) + return { + 'healthy': healthy, + 'reason': '' if healthy else str((row or {}).get('last_error') if isinstance(row, dict) else '') or 'worker lease is not ready', + **(dict(row) if row else {}), + } + + def try_acquire_result_spool_publisher(self): + if not self.conn or not self.conn.is_postgres: + return True + if self._result_spool_publisher_held: + raise RuntimeError('result-spool publisher lease is already held by this database session') + try: + row = self.conn.execute( + 'SELECT pg_try_advisory_lock(?, ?) AS acquired', + (RESULT_SPOOL_ADVISORY_CLASS, RESULT_SPOOL_ADVISORY_OBJECT), + ).fetchone() + self.conn.commit() + self.last_error = '' + acquired = bool(row and row['acquired']) + self._result_spool_publisher_held = acquired + return acquired + except Exception: + connection = self.conn + self.conn = None + self._result_spool_publisher_held = False + try: + if connection: + connection.close() + except Exception: + pass + raise + + def result_spool_publisher_state(self): + if not self.conn or not self.conn.is_postgres: + return { + 'status': 'held', 'authenticated': True, + 'application_name': 'local', 'pid': os.getpid(), + 'holder_identity': f'local:{os.getpid()}', 'state': 'active', + } + try: + rows = self.conn.execute( + '''SELECT a.pid, a.application_name, a.state, + a.backend_start::text AS backend_start, + a.backend_type, + COALESCE(a.client_addr::text, '') AS client_addr, + (a.usename = CURRENT_USER) AS same_user, + (a.datname = CURRENT_DATABASE()) AS same_database + FROM pg_catalog.pg_locks l + JOIN pg_catalog.pg_stat_activity a ON a.pid = l.pid + WHERE l.locktype = 'advisory' AND l.classid = ? AND l.objid = ? + AND l.objsubid = 2 AND l.granted + ORDER BY a.pid''', + (RESULT_SPOOL_ADVISORY_CLASS, RESULT_SPOOL_ADVISORY_OBJECT), + ).fetchall() + self.conn.commit() + if not rows: + return {'status': 'free', 'authenticated': True} + if len(rows) != 1: + raise RuntimeError('result-spool advisory lock has ambiguous multiple holders') + row = rows[0] + application_name = str(row['application_name'] or '') + client_addr = str(row['client_addr'] or '') + try: + client_ip = ipaddress.ip_interface(client_addr).ip + except ValueError: + client_ip = None + authenticated = bool( + re.fullmatch(r'truf-source:[a-z0-9_]+', application_name) + and row['same_user'] + and row['same_database'] + and row['backend_type'] == 'client backend' + and client_ip in (ipaddress.ip_address('127.0.0.1'), ipaddress.ip_address('::1')) + and int(row['pid'] or 0) > 0 + and row['backend_start'] + ) + if not authenticated: + raise RuntimeError('result-spool advisory lock holder is not an authenticated managed source') + return { + 'status': 'held', + 'authenticated': True, + 'application_name': application_name, + 'pid': int(row['pid']), + 'state': str(row['state'] or ''), + 'holder_identity': f"{int(row['pid'])}:{row['backend_start']}", + } + except Exception: + try: + self.conn.rollback() + except Exception: + pass + raise + + def release_result_spool_publisher(self): + if not self.conn or not self.conn.is_postgres: + return True + if not self._result_spool_publisher_held: + return False + try: + row = self.conn.execute( + 'SELECT pg_advisory_unlock(?, ?) AS released', + (RESULT_SPOOL_ADVISORY_CLASS, RESULT_SPOOL_ADVISORY_OBJECT), + ).fetchone() + self.conn.commit() + self.last_error = '' + released = bool(row and row['released']) + if released: + self._result_spool_publisher_held = False + return True + connection = self.conn + self.conn = None + self._result_spool_publisher_held = False + try: + connection.close() + except Exception: + pass + return False + except Exception: + connection = self.conn + self.conn = None + self._result_spool_publisher_held = False + try: + if connection: + connection.close() + except Exception: + pass + raise + + def result_spool_reservation_progress(self, reservations): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('result-spool reservation progress requires PostgreSQL') + reservations = list(reservations or []) + if not reservations or len(reservations) > 10000: + raise RuntimeError('result-spool reservation progress input is outside its bound') + now = time.time() + claims = [] + earliest_progress_at = None + for record in reservations: + try: + recover_after = float(record['recover_after']) + expires_at = float(record['expires_at']) + except (KeyError, TypeError, ValueError) as exc: + raise RuntimeError('result-spool reservation deadline metadata is invalid') from exc + if not math.isfinite(recover_after) or not math.isfinite(expires_at) or recover_after < expires_at: + raise RuntimeError('result-spool reservation deadlines are indeterminate') + if recover_after <= now: + raise RuntimeError('result-spool recoverable reservation remained in capacity accounting') + if recover_after - now > RESULT_SPOOL_PROGRESS_MAX_WAIT_SEC: + raise RuntimeError('result-spool reservation recovery deadline exceeds its progress bound') + deadline = min(value for value in (expires_at, recover_after) if value > now) if expires_at > now else recover_after + earliest_progress_at = deadline if earliest_progress_at is None else min(earliest_progress_at, deadline) + for claim in record.get('claims') or []: + if claim.get('queue_id') is None or not claim.get('lease_owner') or not claim.get('lease_token') or not claim.get('claim_batch'): + raise RuntimeError('result-spool reservation claim ownership is incomplete') + claims.append((record, claim)) + rows_by_id = {} + ids = sorted({int(claim['queue_id']) for _, claim in claims}) + for start in range(0, len(ids), 64): + batch = ids[start:start + 64] + placeholders = ','.join('?' for _ in batch) + rows = self.conn.execute( + f'''SELECT id, status, lease_owner, lease_token, claim_batch, lease_expires_at + FROM target_queue WHERE id IN ({placeholders})''', + batch, + ).fetchall() + rows_by_id.update({int(row['id']): row for row in rows}) + self.conn.commit() + counts = {'exact_live': 0, 'exact_expired': 0, 'stale_or_reassigned': 0, 'unbound': 0} + counts['unbound'] = sum(1 for record in reservations if not (record.get('claims') or [])) + for _, claim in claims: + row = rows_by_id.get(int(claim['queue_id'])) + exact = bool( + row + and row['status'] == 'in_progress' + and row['lease_owner'] == claim['lease_owner'] + and row['lease_token'] == claim['lease_token'] + and row['claim_batch'] == claim['claim_batch'] + ) + if not exact: + counts['stale_or_reassigned'] += 1 + continue + lease = parse_time(row['lease_expires_at']) + if lease and lease.timestamp() > now: + counts['exact_live'] += 1 + else: + counts['exact_expired'] += 1 + return { + 'safe_progress': True, + 'counts': counts, + 'earliest_progress_in_sec': max(0, int((earliest_progress_at or now) - now)), + } + + def initialize_schema(self): + if not self.conn: + return + self.conn.executescript(SCHEMA_SQL) + self.conn.execute( + '''CREATE UNIQUE INDEX IF NOT EXISTS uq_keycheck_credentials_provider_key + ON keycheck_credentials(service, provider_key_hash)''' + ) + now = utc_now_iso() + _ensure_runtime_operations_authority(self.conn, now) + self.conn.execute( + 'INSERT INTO pipeline_capacity(id, updated_at) VALUES (1, ?) ON CONFLICT(id) DO NOTHING', + (now,), + ) + for stream_name, relative_path, rotation_bytes in ( + ('scan_results', 'scan_results.jsonl', 256 * 1024 * 1024), + ('found_secrets', 'found_secrets.jsonl', 128 * 1024 * 1024), + ('scan_errors', 'scan_errors.log', 32 * 1024 * 1024), + ): + self.conn.execute( + '''INSERT INTO projection_streams( + stream_name, base_relative_path, current_generation, + rotation_bytes, max_generations, created_at, updated_at + ) VALUES (?, ?, 0, ?, 16, ?, ?) + ON CONFLICT(stream_name) DO NOTHING''', + (stream_name, relative_path, rotation_bytes, now, now), + ) + self.conn.execute( + '''INSERT INTO projection_cursors(stream_name, generation, committed_offset, updated_at) + VALUES (?, 0, 0, ?) ON CONFLICT(stream_name) DO NOTHING''', + (stream_name, now), + ) + migration_code = hashlib.sha256(PIPELINE_SCHEMA_SQL.encode('utf-8')).hexdigest() + for version in PIPELINE_MIGRATION_VERSIONS: + self.conn.execute( + '''INSERT INTO runtime_schema_migrations(version, applied_at, code_sha256) + VALUES (?, ?, ?) ON CONFLICT(version) DO NOTHING''', + (version, now, migration_code), + ) + if self.conn.is_sqlite: + self.conn.execute('PRAGMA user_version=1') + self.conn.commit() + + def _safe(self, label, func, default=None): + if not self.conn: + return default + last_error = None + for attempt in range(SQLITE_LOCK_RETRY_ATTEMPTS): + try: + result = func() + self.last_error = '' + if self.conn and self.conn.is_postgres: + try: + self.conn.commit() + except Exception: + pass + return result + except Exception as e: + last_error = e + self.last_error = str(e) + try: + self.conn.rollback() + except Exception: + pass + if isinstance(e, DiscoveryPausedError): + raise + if not sqlite_lock_error(e) or attempt == SQLITE_LOCK_RETRY_ATTEMPTS - 1: + if sqlite_lock_error(e): + self._reset_connection() + else: + logger.error(f'Observability DB write failed during {label}: {e}') + return default + break + if attempt and attempt % 4 == 0: + self._reset_connection() + if not self.conn: + return default + delay = sqlite_lock_retry_delay(attempt) + logger.warning(f'Observability DB locked during {label}; retrying in {delay:.2f}s ({attempt + 1}/{SQLITE_LOCK_RETRY_ATTEMPTS})') + time.sleep(delay) + logger.error(f'Observability DB write failed during {label}: {last_error}') + return default + + def _finalize_stale_source_runs(self, source, timestamp): + source = str(source or '').strip() + if not source: + return 0, 0 + reason = 'superseded by a new supervised source process' + totals = [] + statements = ( + ( + '''UPDATE source_cycles SET ended_at = ?, status = 'interrupted', + message = COALESCE(NULLIF(message, ''), ?) + WHERE id IN ( + SELECT id FROM source_cycles + WHERE source = ? AND status = 'running' + ORDER BY id LIMIT ? + )''', + (timestamp, reason, source, STALE_RUN_FINALIZE_BATCH_SIZE), + 'source_cycles', + 'source', + ), + ( + '''UPDATE runs SET ended_at = ?, status = 'interrupted', + error = COALESCE(NULLIF(error, ''), ?), updated_at = ? + WHERE id IN ( + SELECT id FROM runs + WHERE selected_source = ? AND status = 'running' + ORDER BY id LIMIT ? + )''', + (timestamp, reason, timestamp, source, STALE_RUN_FINALIZE_BATCH_SIZE), + 'runs', + 'selected_source', + ), + ) + for statement, params, table, source_column in statements: + finalized = 0 + for _ in range(STALE_RUN_FINALIZE_MAX_BATCHES): + cursor = self.conn.execute(statement, params) + changed = max(0, int(getattr(cursor, 'rowcount', 0) or 0)) + finalized += changed + self.conn.commit() + if changed < STALE_RUN_FINALIZE_BATCH_SIZE: + break + remaining = self.conn.execute( + f'''SELECT 1 FROM {table} + WHERE {source_column} = ? AND status = 'running' LIMIT 1''', + (source,), + ).fetchone() + if remaining: + raise RuntimeError( + f'bounded stale-run finalization limit reached for {table}:{source}; retry source start' + ) + totals.append(finalized) + return tuple(totals) + + def start_run(self, invocation_mode, argv=None, selected_source=None, selected_platform=None, config_path=None, config_hash=None, enabled_sources=None, global_config=None): + def op(): + now = utc_now_iso() + if selected_source: + self._finalize_stale_source_runs(selected_source, now) + safe_argv = redact_argv(argv) + command_line = ' '.join(safe_argv) + run_id = self.conn.insert_returning_id( + '''INSERT INTO runs ( + started_at, status, invocation_mode, command_line, argv_json, + selected_source, selected_platform, config_path, config_hash, + enabled_sources_json, db_path, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + now, 'running', invocation_mode, command_line, json_dumps(safe_argv), + selected_source, selected_platform, config_path, config_hash, + json_dumps(enabled_sources or []), self.db_display, now, now, + ), + ) + if global_config is not None: + self.record_config_snapshot(run_id, None, 'global', None, global_config) + self.conn.commit() + return run_id + return self._safe('start_run', op) + + def finish_run(self, run_id, status='completed', error=None): + if not run_id: + return + def op(): + now = utc_now_iso() + row = self.conn.execute('SELECT started_at FROM runs WHERE id = ?', (run_id,)).fetchone() + duration = elapsed_seconds(row['started_at'], now) if row else None + totals = self.conn.execute( + '''SELECT + COALESCE(SUM(fetched_count), 0) AS fetched, + COALESCE(SUM(queued_new_count), 0) AS queued_new, + COALESCE(SUM(scan_requested_count), 0) AS scan_requested, + COALESCE(SUM(scanned_count), 0) AS scanned, + COALESCE(SUM(clean_count), 0) AS clean, + COALESCE(SUM(found_count), 0) AS found, + COALESCE(SUM(skipped_count), 0) AS skipped, + COALESCE(SUM(error_count), 0) AS errors, + COALESCE(SUM(findings_count), 0) AS findings, + COALESCE(SUM(verified_findings_count), 0) AS verified, + COALESCE(SUM(unique_secrets_count), 0) AS unique_secrets, + COALESCE(SUM(unique_findings_count), 0) AS unique_findings + FROM source_cycles WHERE run_id = ?''', + (run_id,), + ).fetchone() + self.conn.execute( + '''UPDATE runs SET ended_at = ?, duration_sec = ?, status = ?, error = ?, + total_fetched = ?, total_queued_new = ?, total_scan_requested = ?, total_scanned = ?, + total_clean = ?, total_found = ?, total_skipped = ?, total_errors = ?, + total_findings = ?, total_verified_findings = ?, total_unique_secrets = ?, + total_unique_findings = ?, updated_at = ? WHERE id = ?''', + ( + now, duration, status, error, + totals['fetched'], totals['queued_new'], totals['scan_requested'], totals['scanned'], + totals['clean'], totals['found'], totals['skipped'], totals['errors'], + totals['findings'], totals['verified'], totals['unique_secrets'], totals['unique_findings'], now, run_id, + ), + ) + self.conn.commit() + self._safe('finish_run', op) + + def start_source_cycle(self, run_id, source, platform, mode, query, query_index=None, query_count=None, auth_name=None, source_config=None, queue_before=None): + def op(): + now = utc_now_iso() + queue_data = queue_before or {} + cycle_id = self.conn.insert_returning_id( + '''INSERT INTO source_cycles ( + run_id, source, platform, mode, query, query_index, query_count, auth_name, + started_at, status, config_json, queue_todo_before, queue_checked_before, + created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + run_id, source, platform, mode, query, query_index, query_count, auth_name, + now, 'running', json_dumps(redact_config(source_config or {})), + queue_data.get('todo_count'), queue_data.get('checked_count'), now, now, + ), + ) + self.record_config_snapshot(run_id, cycle_id, 'source', source, source_config or {}) + self.record_queue_snapshot(run_id, cycle_id, source, 'before', queue_data) + self.conn.commit() + return cycle_id + return self._safe('start_source_cycle', op) + + def finish_source_cycle(self, cycle_id, status='completed', metrics=None, queue_after=None, message=None): + if not cycle_id: + return + def op(): + now = utc_now_iso() + row = self.conn.execute('SELECT started_at, run_id, source FROM source_cycles WHERE id = ?', (cycle_id,)).fetchone() + duration = elapsed_seconds(row['started_at'], now) if row else None + metrics_data = metrics or {} + queue_data = queue_after or {} + if metrics_data.get('authoritative_async'): + staged = int(metrics_data.get('staged_count') or 0) + self.conn.execute( + '''UPDATE source_cycles SET ended_at = ?, duration_sec = ?, status = ?, message = ?, + fetched_count = ?, queued_new_count = ?, queued_updated_count = ?, + scan_requested_count = ?, + staged_count = staged_count + ?, queue_todo_after = ?, + queue_checked_after = ?, updated_at = ? WHERE id = ?''', + ( + now, duration, status, message, + metrics_data.get('fetched_count', 0), + metrics_data.get('queued_new_count', 0), + metrics_data.get('queued_updated_count', 0), + metrics_data.get('scan_requested_count', 0), staged, + queue_data.get('todo_count'), queue_data.get('checked_count'), now, cycle_id, + ), + ) + if row: + self.conn.execute( + 'UPDATE runs SET total_staged = total_staged + ?, updated_at = ? WHERE id = ?', + (staged, now, row['run_id']), + ) + self.record_queue_snapshot(row['run_id'], cycle_id, row['source'], 'after', queue_data) + self.conn.commit() + return + scanned = int(metrics_data.get('scanned_count') or metrics_data.get('scanned') or 0) + found = int(metrics_data.get('found_count') or 0) + errors = int(metrics_data.get('error_count') or 0) + verified = int(metrics_data.get('verified_findings_count') or 0) + targets_per_hour = scanned / (duration / 3600) if duration and duration > 0 else 0 + hit_rate = found / scanned if scanned else 0 + verified_hit_rate = verified / scanned if scanned else 0 + error_rate = errors / scanned if scanned else 0 + self.conn.execute( + '''UPDATE source_cycles SET ended_at = ?, duration_sec = ?, status = ?, message = ?, + fetched_count = ?, queued_new_count = ?, queued_updated_count = ?, + scan_requested_count = ?, scanned_count = ?, + clean_count = ?, found_count = ?, skipped_count = ?, error_count = ?, + findings_count = ?, verified_findings_count = ?, unique_secrets_count = ?, unique_findings_count = ?, + queue_todo_after = ?, queue_checked_after = ?, targets_per_hour = ?, hit_rate = ?, + verified_hit_rate = ?, error_rate = ?, updated_at = ? WHERE id = ?''', + ( + now, duration, status, message, + metrics_data.get('fetched_count', 0), metrics_data.get('queued_new_count', 0), + metrics_data.get('queued_updated_count', 0), metrics_data.get('scan_requested_count', 0), scanned, + metrics_data.get('clean_count', 0), found, metrics_data.get('skipped_count', 0), errors, + metrics_data.get('findings_count', 0), verified, metrics_data.get('unique_secrets_count', 0), metrics_data.get('unique_findings_count', 0), + queue_data.get('todo_count'), queue_data.get('checked_count'), targets_per_hour, hit_rate, + verified_hit_rate, error_rate, now, cycle_id, + ), + ) + if row: + self.record_queue_snapshot(row['run_id'], cycle_id, row['source'], 'after', queue_data) + self.conn.commit() + self._safe('finish_source_cycle', op) + + def record_config_snapshot(self, run_id, cycle_id, scope, source, config): + if not self.conn or run_id is None: + return + self.conn.execute( + 'INSERT INTO config_snapshots (run_id, cycle_id, scope, source, config_json, captured_at) VALUES (?, ?, ?, ?, ?, ?)', + (run_id, cycle_id, scope, source, json_dumps(redact_config(config or {})), utc_now_iso()), + ) + + def record_queue_snapshot(self, run_id, cycle_id, source, phase, counts): + if not self.conn or run_id is None: + return + counts = counts or {} + self.conn.execute( + '''INSERT INTO queue_snapshots ( + run_id, cycle_id, source, phase, todo_count, checked_count, todo_file, checked_file, captured_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + run_id, cycle_id, source, phase, counts.get('todo_count'), counts.get('checked_count'), + counts.get('todo_file'), counts.get('checked_file'), utc_now_iso(), + ), + ) + + def target_queue_available(self): + if not self.conn or not self.conn.table_exists('target_queue'): + return False + required = { + 'id', 'source', 'platform', 'query', 'target', 'normalized_target', 'status', 'attempts', + 'lease_owner', 'lease_token', 'claim_batch', 'leased_at', 'lease_expires_at', 'available_after', 'target_scan_id', + 'last_error', 'created_at', 'updated_at', 'completed_at', + } + return required.issubset(set(self.conn.table_columns('target_queue'))) + + def runtime_safety_schema_available(self, commit=True): + if not self.conn: + self.last_error = 'database connection is unavailable' + return False + required = {table: set(specs) for table, specs in RUNTIME_TABLE_SPECS.items()} + try: + missing = [] + for table, columns in required.items(): + if not self.conn.table_exists(table): + missing.append(f'table {table}') + continue + absent = sorted(columns - set(self.conn.table_columns(table))) + if absent: + missing.append(f'{table} columns {", ".join(absent)}') + table_details = {} + for table, specs in RUNTIME_TABLE_SPECS.items(): + if not self.conn.table_exists(table): + continue + details = self.conn.table_column_details(table) + table_details[table] = details + for name, (expected_type, expected_not_null) in specs.items(): + actual = details.get(name) + if not actual: + continue + if not _schema_type_matches(actual['type'], expected_type, self.conn.is_postgres): + missing.append(f'{table}.{name} type {actual["type"] or ""}') + if bool(actual['not_null']) != bool(expected_not_null): + missing.append(f'{table}.{name} nullability') + primary_key = [name for name in specs if details.get(name, {}).get('primary_key')] + if primary_key != RUNTIME_PRIMARY_KEYS[table]: + missing.append(f'{table} primary key') + generated_id = GENERATED_ID_COLUMNS.get(table) + if generated_id and not _column_generates_id(details.get(generated_id), self.conn.is_postgres): + missing.append(f'{table}.{generated_id} generated ID identity/sequence') + for name, actual in details.items(): + if name != generated_id and (actual.get('identity') or actual.get('generated')): + missing.append(f'{table}.{name} unexpected generated expression') + expected_defaults = RUNTIME_COLUMN_DEFAULTS.get(table, {}) + for name in specs: + if name == generated_id: + continue + actual = details.get(name) + expected_present = name in expected_defaults + expected_default = expected_defaults.get(name, '') + if actual and ( + bool(actual.get('has_default', bool(actual.get('default')))) != expected_present + or ( + expected_present + and _normalized_default(actual['default']) != _normalized_default(expected_default) + ) + ): + missing.append(f'{table}.{name} default') + + if self.conn.is_postgres: + primary_indexes = [ + index for index in self.conn.table_indexes(table).values() + if index.get('primary') + ] + if len(primary_indexes) != 1 or primary_indexes[0]['columns'] != RUNTIME_PRIMARY_KEYS[table] or not _index_usable(primary_indexes[0]): + missing.append(f'{table} primary-key index validity/readiness') + + queue_indexes = self.conn.table_indexes('target_queue') if self.conn.table_exists('target_queue') else {} + if not any( + index['unique'] and index['columns'] == ['source', 'normalized_target'] + and not _normalized_predicate(index['predicate']) + and _index_usable(index) + for index in queue_indexes.values() + ): + missing.append('unique target_queue source/normalized_target index') + + retry_indexes = ( + self.conn.table_indexes('discovery_retry_queue') + if self.conn.table_exists('discovery_retry_queue') else {} + ) + work_key_index = retry_indexes.get('uq_discovery_retry_queue_work_key') + if ( + not work_key_index + or not work_key_index['unique'] + or work_key_index['columns'] != ['work_key'] + or _normalized_predicate(work_key_index['predicate']) + or not _index_usable(work_key_index) + ): + missing.append('unique discovery retry work-key index') + + scan_indexes = self.conn.table_indexes('target_scans') if self.conn.table_exists('target_scans') else {} + event_index = scan_indexes.get('uq_target_scans_scan_event_id') + if ( + not event_index + or not event_index['unique'] + or event_index['columns'] != ['scan_event_id'] + or _normalized_predicate(event_index['predicate']) != 'scan_event_idisnotnull' + or not _index_usable(event_index) + ): + missing.append('unique partial scan-event index') + + for table, name, columns in ( + ( + 'worker_progress_events', 'uq_worker_progress_reservation_sequence', + ['reservation_id', 'sequence'], + ), + ('worker_diagnostics', 'uq_worker_diagnostics_uid', ['diagnostic_uid']), + ): + indexes = self.conn.table_indexes(table) if self.conn.table_exists(table) else {} + index = indexes.get(name) + if ( + not index or not index['unique'] or index['columns'] != columns + or _normalized_predicate(index['predicate']) or not _index_usable(index) + ): + missing.append(f'unique index {name}') + + outbox_indexes = self.conn.table_indexes('scan_publication_outbox') if self.conn.table_exists('scan_publication_outbox') else {} + if not any( + index['unique'] and index['columns'] == ['target_scan_id'] and not _normalized_predicate(index['predicate']) + and _index_usable(index) + for index in outbox_indexes.values() + ): + missing.append('unique outbox target-scan index') + status_index = outbox_indexes.get('idx_scan_publication_outbox_status') + if not status_index or status_index['columns'] != ['status', 'available_after', 'id'] or status_index['unique'] or not _index_usable(status_index): + missing.append('outbox status index') + age_index = outbox_indexes.get('idx_scan_publication_outbox_age') + if not age_index or age_index['columns'] != ['status', 'created_at', 'id'] or age_index['unique'] or not _index_usable(age_index): + missing.append('outbox age index') + + for table, column in (('keycheck_event_map', 'event_id'), ('finding_uid_map', 'finding_uid')): + indexes = self.conn.table_indexes(table) if self.conn.table_exists(table) else {} + if not any( + index['unique'] and index['columns'] == [column] and not _normalized_predicate(index['predicate']) + and _index_usable(index) + for index in indexes.values() + ): + missing.append(f'unique {table}.{column} index') + + issue_indexes = self.conn.table_indexes('target_queue_reconciliation_issues') if self.conn.table_exists('target_queue_reconciliation_issues') else {} + if not any( + index['unique'] + and index['columns'] == ['source_file', 'file_identity', 'line_number', 'byte_offset', 'reason'] + and not _normalized_predicate(index['predicate']) + and _index_usable(index) + for index in issue_indexes.values() + ): + missing.append('unique reconciliation issue identity index') + + package_indexes = self.conn.table_indexes('package_repo_candidates') if self.conn.table_exists('package_repo_candidates') else {} + if not any( + index['unique'] + and index['columns'] == ['package_source', 'package_name', 'package_version', 'repo_url'] + and not _normalized_predicate(index['predicate']) + and _index_usable(index) + for index in package_indexes.values() + ): + missing.append('unique package repository candidate identity index') + + required_indexes = { + 'runs': { + 'idx_runs_started_at': ['started_at'], + 'idx_runs_status': ['status'], + 'idx_runs_selected_source_status': ['selected_source', 'status', 'id'], + }, + 'source_cycles': { + 'idx_source_cycles_run_id': ['run_id'], + 'idx_source_cycles_source_query': ['source', 'query'], + 'idx_source_cycles_source_status': ['source', 'status', 'id'], + }, + 'target_queue': { + 'idx_target_queue_source_status': ['source', 'status', 'updated_at'], + 'idx_target_queue_observe_source_status': ['source', 'status'], + 'idx_target_queue_lease': ['source', 'status', 'lease_expires_at'], + 'idx_target_queue_platform_status': ['platform', 'status'], + 'idx_target_queue_claim_batch': ['claim_batch', 'lease_owner'], + 'idx_target_queue_claim': ['source', 'platform', 'status', 'available_after', 'lease_expires_at', 'id'], + 'idx_target_queue_resolver_claim': ['source', 'platform', 'status', 'resolver_state', 'resolver_due_at', 'id'], + 'idx_target_queue_source_platform_normalized': ['source', 'platform', 'normalized_target'], + 'idx_target_queue_claim_pending': ['source', 'platform', 'id', 'attempts', 'available_after'], + 'idx_target_queue_claim_deferred': ['source', 'platform', 'available_after', 'id', 'attempts'], + 'idx_target_queue_claim_in_progress': ['source', 'platform', 'id', 'attempts', 'available_after', 'lease_expires_at', 'resolver_state'], + 'idx_target_queue_active_lease_owner_token': ['lease_owner', 'lease_token'], + 'idx_target_queue_exhausted_attempts': ['source', 'platform', 'status', 'attempts', 'id', 'lease_expires_at'], + 'idx_target_queue_updated_rescan': ['source', 'platform', 'completed_at', 'remote_modified_at', 'scan_remote_modified_at', 'id'], + 'idx_target_queue_cold': ['source', 'platform', 'query', 'id'], + }, + 'discovery_retry_queue': { + 'idx_discovery_retry_queue_due': ['source', 'available_after', 'id'], + 'idx_discovery_retry_queue_lease': ['lease_expires_at', 'id'], + 'idx_discovery_retry_queue_policy': [ + 'source', 'query', 'policy_sha256', 'status', 'id', + ], + }, + 'target_queue_policy_events': { + 'idx_target_queue_policy_events_queue': ['queue_id', 'id'], + 'idx_target_queue_policy_events_manifest': ['manifest_sha256', 'id'], + }, + 'target_scans': { + 'idx_target_scans_cycle_id': ['cycle_id'], + 'idx_target_scans_queue_id': ['queue_id'], + 'idx_target_scans_source_status': ['source', 'status'], + 'idx_target_scans_source_ended': ['source', 'ended_at'], + 'idx_target_scans_source_ended_id': ['source', 'ended_at', 'id'], + 'idx_target_scans_source_skip_ended': ['source', 'skipped_reason', 'ended_at'], + 'idx_target_scans_cooldown_recent': ['source', 'ended_at', 'id'], + 'idx_target_scans_normalized_target': ['normalized_target'], + 'idx_target_scans_result_reservation': ['result_reservation_id', 'id'], + }, + 'keycheck_results': { + 'idx_keycheck_results_checked': ['checked_at'], + 'idx_keycheck_results_service_status': ['service', 'status_group', 'status'], + 'idx_keycheck_results_finding': ['finding_id'], + 'idx_keycheck_results_cycle': ['cycle_id'], + 'idx_keycheck_results_source_query': ['source', 'query'], + 'idx_keycheck_results_key_hash': ['key_hash'], + 'idx_keycheck_results_secret_hash': ['secret_hash'], + 'idx_keycheck_results_link_repair': ['link_status', 'id'], + }, + 'findings': { + 'idx_findings_target_scan_id_id': ['target_scan_id', 'id'], + 'idx_findings_cycle_id': ['cycle_id'], + 'idx_findings_source': ['source'], + 'idx_findings_secret_hash': ['secret_hash'], + 'idx_findings_finding_uid': ['finding_uid'], + }, + 'errors': { + 'idx_errors_cycle_id': ['cycle_id'], + 'idx_errors_source_category': ['source', 'category'], + 'idx_errors_target_scan_id': ['target_scan_id'], + }, + 'queue_snapshots': { + 'idx_queue_snapshots_source_time': ['source', 'captured_at'], + }, + 'package_repo_candidates': { + 'idx_package_repo_candidates_query': ['query'], + 'idx_package_repo_candidates_repo': ['repo_url'], + 'idx_package_repo_candidates_source_seen': ['package_source', 'last_seen_at'], + 'idx_package_repo_candidates_query_seen': ['query', 'last_seen_at', 'id'], + 'idx_package_repo_candidates_recent_lookup': ['last_seen_at', 'id', 'package_source', 'query', 'package_name'], + }, + 'target_queue_reconciliation_issues': { + 'idx_reconciliation_issues_open': ['source_file', 'resolved_at', 'id'], + }, + 'docker_content_blobs': { + 'idx_docker_content_blobs_reclaim': [ + 'state', 'available_after', 'lease_expires_at', 'digest', + ], + 'idx_docker_content_blobs_reservation': ['lease_reservation_id', 'digest'], + }, + 'docker_image_blob_coverage': { + 'idx_docker_image_blob_coverage_manifest': [ + 'manifest_digest', 'coverage_state', 'position', + ], + 'idx_docker_image_blob_coverage_blob': [ + 'blob_digest', 'coverage_state', 'queue_id', + ], + 'idx_docker_image_blob_coverage_reservation': ['reservation_id', 'position'], + 'idx_docker_image_blob_coverage_selection': [ + 'queue_id', 'manifest_digest', 'selection_policy_sha256', + 'position', 'reservation_id', + ], + }, + 'docker_adaptive_shadow_reports': { + 'idx_docker_adaptive_shadow_reports_gate': [ + 'scan_policy_sha256', 'execution_policy_sha256', + 'selection_policy_sha256', 'state', 'completed_at', 'id', + ], + }, + 'runtime_operations': { + 'idx_runtime_operations_status_updated': [ + 'status', 'updated_at', 'operation_id', + ], + 'idx_runtime_operations_agent_state_updated': [ + 'agent_state', 'updated_at', 'operation_id', + ], + }, + 'runtime_audit_events': { + 'idx_runtime_audit_events_created': ['created_at', 'id'], + 'idx_runtime_audit_events_operation': ['operation_id', 'id'], + }, + 'worker_progress_events': { + 'idx_worker_progress_reservation_received': [ + 'reservation_id', 'received_at', 'id', + ], + 'idx_worker_progress_device_received': [ + 'remote_device_id', 'received_at', 'id', + ], + 'idx_worker_progress_phase_received': ['phase', 'received_at', 'id'], + }, + 'result_reservations': { + 'idx_result_reservations_remote_resolved': [ + 'remote_resolved_at', 'id', + ], + }, + 'worker_diagnostics': { + 'idx_worker_diagnostics_reservation_received': [ + 'reservation_id', 'received_at', 'id', + ], + 'idx_worker_diagnostics_scan_received': [ + 'target_scan_id', 'received_at', 'id', + ], + 'idx_worker_diagnostics_phase_received': ['phase', 'received_at', 'id'], + 'idx_worker_diagnostics_category_code': [ + 'category', 'code', 'received_at', 'id', + ], + 'idx_worker_diagnostics_code_received': ['code', 'received_at', 'id'], + 'idx_worker_diagnostics_kind_received': ['kind', 'received_at', 'id'], + 'idx_worker_diagnostics_retryable_received': [ + 'retryable', 'received_at', 'id', + ], + }, + } + for table, expected_indexes in required_indexes.items(): + indexes = self.conn.table_indexes(table) if self.conn.table_exists(table) else {} + for name, columns in expected_indexes.items(): + index = indexes.get(name) + if not index or index['columns'] != columns or index['unique'] or not _index_usable(index): + missing.append(f'index {name}') + control = self.conn.execute( + 'SELECT id FROM runtime_operations_control WHERE id = 1' + ).fetchone() if self.conn.table_exists('runtime_operations_control') else None + if not control: + missing.append('runtime_operations_control singleton row') + missing.extend(_runtime_audit_trigger_problems(self.conn)) + experiment_indexes = {} + for table, name, columns, unique, predicate in DOCKER_DEPTH_EXPERIMENT_INDEX_SPECS: + indexes = experiment_indexes.setdefault( + table, + self.conn.table_indexes(table) if self.conn.table_exists(table) else {}, + ) + index = indexes.get(name) + if ( + not index + or index['columns'] != list(columns) + or bool(index['unique']) != bool(unique) + or _normalized_predicate(index['predicate']) != _normalized_predicate(predicate) + or not _index_usable(index) + ): + missing.append(f'index {name}') + queue_indexes = self.conn.table_indexes('target_queue') if self.conn.table_exists('target_queue') else {} + for name, predicate in ( + ('idx_target_queue_claim_pending', "status = 'pending' AND current_result_reservation_id IS NULL AND (resolver_state IS NULL OR resolver_state = 'resolved')"), + ('idx_target_queue_claim_deferred', "status = 'deferred' AND current_result_reservation_id IS NULL AND (resolver_state IS NULL OR resolver_state = 'resolved')"), + ('idx_target_queue_claim_in_progress', "status = 'in_progress' AND (resolver_state IS NULL OR resolver_state = 'resolved')"), + ('idx_target_queue_active_lease_owner_token', "status = 'in_progress'"), + ('idx_target_queue_exhausted_attempts', "status = 'pending' OR status = 'deferred' OR status = 'in_progress'"), + ('idx_target_queue_updated_rescan', "status = 'done' AND remote_modified_at IS NOT NULL"), + ('idx_target_queue_cold', "status = 'cold'"), + ): + if _normalized_predicate((queue_indexes.get(name) or {}).get('predicate')) != _normalized_predicate(predicate): + missing.append(f'predicate {name}') + for name, predicate in ( + ('idx_discovery_retry_queue_due', DISCOVERY_RETRY_DUE_INDEX_PREDICATE), + ('idx_discovery_retry_queue_lease', DISCOVERY_RETRY_LEASE_INDEX_PREDICATE), + ): + if _normalized_predicate((retry_indexes.get(name) or {}).get('predicate')) != _normalized_predicate(predicate): + missing.append(f'predicate {name}') + cooldown_index = scan_indexes.get('idx_target_scans_cooldown_recent') + if _normalized_predicate((cooldown_index or {}).get('predicate')) != _normalized_predicate(CI_COOLDOWN_INDEX_PREDICATE): + missing.append('predicate idx_target_scans_cooldown_recent') + for name in ( + 'idx_package_repo_candidates_query_seen', + 'idx_package_repo_candidates_recent_lookup', + ): + if _normalized_predicate((package_indexes.get(name) or {}).get('predicate')) != _normalized_predicate(PACKAGE_REPO_NONEMPTY_PREDICATE): + missing.append(f'predicate {name}') + missing.extend(_docker_depth_check_constraint_problems(self.conn)) + for table, expected_keys in REQUIRED_FOREIGN_KEYS.items(): + if not self.conn.table_exists(table): + continue + foreign_keys = self.conn.table_foreign_keys(table) + for columns, referenced_table, referenced_columns in expected_keys: + authority_matching = [ + foreign_key for foreign_key in foreign_keys.values() + if tuple(foreign_key.get('columns') or ()) == columns + and foreign_key.get('referenced_table') == referenced_table + and tuple(foreign_key.get('referenced_columns') or ()) == referenced_columns + and foreign_key.get('referenced_schema') in ( + self.conn.application_schema if self.conn.is_postgres else 'main', + ) + ] + expected_actions = _required_foreign_key_actions(table, columns) + matching = [ + foreign_key for foreign_key in authority_matching + if _foreign_key_actions_match(foreign_key, expected_actions) + ] + if len(matching) != 1 or not matching[0].get('valid', True): + detail = ( + f'foreign key {table}({", ".join(columns)}) -> ' + f'{referenced_table}({", ".join(referenced_columns)})' + ) + if len(authority_matching) == 1 and not _foreign_key_actions_match( + authority_matching[0], expected_actions, + ): + detail += ( + f' actions ON UPDATE {expected_actions[0]} ' + f'ON DELETE {expected_actions[1]}' + ) + missing.append(detail) + if self.conn.is_postgres and commit: + self.conn.commit() + if missing: + self.last_error = 'runtime safety schema is incomplete: ' + '; '.join(missing) + return False + self.last_error = '' + return True + except Exception as exc: + self.last_error = f'unable to validate runtime safety schema: {exc}' + try: + self.conn.rollback() + except Exception: + pass + return False + + def require_runtime_safety_schema(self, commit=True): + if not self.runtime_safety_schema_available(commit=commit): + detail = self.last_error or 'runtime safety schema is unavailable' + raise RuntimeSafetySchemaError(f'{detail}; run migrate_runtime_safety.py offline with --apply') + if not self.pipeline_schema_available(commit=commit): + detail = self.last_error or 'pipeline schema is unavailable' + raise RuntimeSafetySchemaError(f'{detail}; run migrate_runtime_safety.py offline with --apply') + return True + + def _locked_runtime_control_state(self, shared=False): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + suffix = '' + if self.conn.is_postgres: + suffix = ' FOR SHARE' if shared else ' FOR UPDATE' + row = self.conn.execute( + 'SELECT * FROM runtime_operations_control WHERE id = 1' + suffix + ).fetchone() + return _runtime_control_state_from_row(row) + + def _require_discovery_admission_locked(self): + state = self._locked_runtime_control_state(shared=True) + if state['effective_discovery_paused']: + raise DiscoveryPausedError(state) + return state + + def runtime_control_state(self): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + try: + row = self.conn.execute( + 'SELECT * FROM runtime_operations_control WHERE id = 1' + ).fetchone() + state = _runtime_control_state_from_row(row) + self.conn.commit() + return state + except Exception: + self.conn.rollback() + raise + + def _locked_runtime_operation(self, operation_id): + suffix = ' FOR UPDATE' if self.conn.is_postgres else '' + return self.conn.execute( + 'SELECT * FROM runtime_operations WHERE operation_id = ?' + suffix, + (operation_id,), + ).fetchone() + + @staticmethod + def _runtime_operation_state(row): + if not row: + return None + operation_id = _runtime_operation_id(str(row['operation_id'] or '')) + actor = _runtime_actor(str(row['actor'] or '')) + action = str(row['action'] or '') + target_kind = str(row['target_kind'] or '') + target_ref = str(row['target_ref'] or '') + status = str(row['status'] or '') + agent_state = str(row['agent_state'] or '') + if action in RUNTIME_ASYNC_ACTIONS: + if ( + target_kind != RUNTIME_ASYNC_TARGET_KIND + or target_ref != RUNTIME_ASYNC_ACTION_TARGETS[action] + ): + raise RuntimeSafetySchemaError('runtime operation target is invalid') + try: + expected_raw = json.loads(str(row['expected_identity_json'] or '')) + expected = _runtime_expected_identity(expected_raw, action) + expected_json = _canonical_runtime_json(expected) + resulting_json = row['resulting_identity_json'] + resulting = None + if resulting_json is not None: + resulting = _runtime_resulting_identity(json.loads(str(resulting_json))) + if _canonical_runtime_json(resulting) != str(resulting_json): + raise ValueError('noncanonical') + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise RuntimeSafetySchemaError('runtime operation identity is invalid') from exc + if expected_json != str(row['expected_identity_json'] or ''): + raise RuntimeSafetySchemaError('runtime operation identity is not canonical') + elif action in RUNTIME_CONTROL_ACTIONS: + expected = _stored_runtime_control_identity(str(row['expected_identity_json'] or '')) + resulting = None + if row['resulting_identity_json'] is not None: + resulting = _stored_runtime_control_identity( + str(row['resulting_identity_json']), + ) + elif action in RUNTIME_SOURCE_OPERATION_ACTIONS: + if target_kind != RUNTIME_SOURCE_TARGET_KIND or _runtime_source_id(target_ref) is None: + raise RuntimeSafetySchemaError('runtime source operation target is invalid') + expected = _stored_runtime_source_identity( + str(row['expected_identity_json'] or ''), action, target_ref, + ) + resulting = None + if row['resulting_identity_json'] is not None: + resulting = _stored_runtime_source_identity( + str(row['resulting_identity_json']), action, target_ref, + resulting=True, + ) + elif action in RUNTIME_WORKER_ADMIN_ACTION_TARGETS: + if target_kind != RUNTIME_WORKER_ADMIN_TARGET_KIND: + raise RuntimeSafetySchemaError( + 'runtime worker admin operation target is invalid' + ) + expected = _stored_runtime_worker_admin_identity( + str(row['expected_identity_json'] or ''), action, target_ref, + ) + resulting = None + if row['resulting_identity_json'] is not None: + resulting = _stored_runtime_worker_admin_identity( + str(row['resulting_identity_json']), action, target_ref, + resulting=True, + ) + elif action in RUNTIME_DOCUMENT_ACTION_TARGETS: + if target_kind != RUNTIME_DOCUMENT_TARGET_KIND: + raise RuntimeSafetySchemaError( + 'runtime document operation target is invalid' + ) + expected = _stored_runtime_document_identity( + str(row['expected_identity_json'] or ''), action, target_ref, + ) + resulting = None + if row['resulting_identity_json'] is not None: + resulting = _stored_runtime_document_identity( + str(row['resulting_identity_json']), action, target_ref, + resulting=True, + ) + elif action in RUNTIME_MANAGED_FILE_ACTIONS: + if target_kind != RUNTIME_MANAGED_FILE_TARGET_KIND: + raise RuntimeSafetySchemaError( + 'runtime managed file operation target is invalid' + ) + expected = _stored_runtime_managed_file_identity( + str(row['expected_identity_json'] or ''), action, target_ref, + ) + resulting = None + if row['resulting_identity_json'] is not None: + resulting = _stored_runtime_managed_file_identity( + str(row['resulting_identity_json']), action, target_ref, + resulting=True, + ) + else: + raise RuntimeSafetySchemaError('runtime operation action is invalid') + if status not in ( + 'requested', 'running', 'succeeded', 'failed', 'rolled_back', + 'failed_hold', 'canceled', + ): + raise RuntimeSafetySchemaError('runtime operation status is invalid') + if agent_state not in ( + 'not_required', 'pending', 'running', 'succeeded', 'failed', + 'rolled_back', 'failed_hold', + ): + raise RuntimeSafetySchemaError('runtime operation agent state is invalid') + category = row['safe_category'] + detail = row['safe_detail'] + try: + category = _runtime_safe_code(category, 'safe category') + detail = _runtime_safe_code(detail, 'safe detail') + except ValueError as exc: + raise RuntimeSafetySchemaError('runtime operation safe result is invalid') from exc + result_sha256 = row['agent_result_sha256'] + if result_sha256 is not None: + try: + result_sha256 = _runtime_sha256(str(result_sha256), 'agent result hash') + except ValueError as exc: + raise RuntimeSafetySchemaError('runtime operation result hash is invalid') from exc + return { + 'operation_id': operation_id, + 'actor': actor, + 'action': action, + 'target_kind': target_kind, + 'target_ref': target_ref, + 'status': status, + 'safe_category': category, + 'safe_detail': detail, + 'expected_revision': row['expected_revision'], + 'resulting_revision': row['resulting_revision'], + 'expected_identity': expected, + 'resulting_identity': resulting, + 'agent_state': agent_state, + 'agent_result_sha256': result_sha256, + 'requested_at': str(row['requested_at'] or ''), + 'started_at': str(row['started_at']) if row['started_at'] is not None else None, + 'completed_at': str(row['completed_at']) if row['completed_at'] is not None else None, + 'agent_reconciled_at': ( + str(row['agent_reconciled_at']) + if row['agent_reconciled_at'] is not None else None + ), + 'updated_at': str(row['updated_at'] or ''), + } + + def _append_runtime_audit_event_locked( + self, *, operation_id, actor, action, target_kind, target_ref, + result, safe_category=None, before_identity=None, after_identity=None, + before_bytes=None, after_bytes=None, created_at=None, + ): + if result not in ( + 'accepted', 'succeeded', 'rejected', 'failed', 'rolled_back', + 'canceled', 'failed_hold', + ): + raise ValueError('runtime audit result is invalid') + before_json = ( + _canonical_runtime_json(before_identity) if before_identity is not None else None + ) + after_json = ( + _canonical_runtime_json(after_identity) if after_identity is not None else None + ) + for value in (before_bytes, after_bytes): + if value is not None and ( + isinstance(value, bool) or not isinstance(value, int) or value < 0 + ): + raise ValueError('runtime audit byte count is invalid') + tail = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + ORDER BY id DESC LIMIT 1''' + ).fetchone() + previous_event_id = int(tail['id']) if tail else None + previous_event_sha256 = str(tail['event_sha256']) if tail else None + now = created_at or utc_now_iso() + payload = { + 'schema': 'runtime-audit-event-v1', + 'operation_id': operation_id, + 'actor': actor, + 'action': action, + 'target_kind': target_kind, + 'target_ref': target_ref, + 'result': result, + 'safe_category': safe_category, + 'before_identity_json': before_json, + 'after_identity_json': after_json, + 'before_bytes': before_bytes, + 'after_bytes': after_bytes, + 'previous_event_id': previous_event_id, + 'previous_event_sha256': previous_event_sha256, + 'created_at': now, + } + event_sha256 = _runtime_audit_event_sha256(payload) + event_id = self.conn.insert_returning_id( + '''INSERT INTO runtime_audit_events( + operation_id, actor, action, target_kind, target_ref, + result, safe_category, before_identity_json, + after_identity_json, before_bytes, after_bytes, + previous_event_id, previous_event_sha256, event_sha256, + created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + operation_id, actor, action, target_kind, target_ref, result, + safe_category, before_json, after_json, before_bytes, after_bytes, + previous_event_id, previous_event_sha256, event_sha256, now, + ), + ) + if event_id is None: + raise RuntimeSafetySchemaError('runtime audit insertion failed') + return { + 'audit_event_id': int(event_id), + 'audit_event_sha256': event_sha256, + } + + def runtime_operation(self, operation_id): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + try: + row = self.conn.execute( + 'SELECT * FROM runtime_operations WHERE operation_id = ?', + (operation_id,), + ).fetchone() + result = self._runtime_operation_state(row) + self.conn.commit() + return result + except Exception: + self.conn.rollback() + raise + + def recent_runtime_operations( + self, limit=200, *, before_updated_at=None, before_operation_id=None, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + if type(limit) is not int or not 1 <= limit <= 500: + raise ValueError('runtime operation snapshot limit is out of range') + if (before_updated_at is None) is not (before_operation_id is None): + raise ValueError('runtime operation cursor is incomplete') + if before_updated_at is not None: + if type(before_updated_at) is not str or not before_updated_at: + raise ValueError('runtime operation cursor timestamp is invalid') + before_operation_id = _runtime_operation_id(before_operation_id) + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN') + else: + self.conn.execute( + 'SET TRANSACTION ISOLATION LEVEL REPEATABLE READ READ ONLY' + ) + rows = [] + for status in ( + 'requested', 'running', 'succeeded', 'failed', 'rolled_back', + 'failed_hold', 'canceled', + ): + if before_updated_at is None: + sql = '''SELECT * FROM runtime_operations WHERE status = ? + ORDER BY updated_at DESC, operation_id DESC LIMIT ?''' + parameters = (status, limit) + else: + sql = '''SELECT * FROM runtime_operations WHERE status = ? + AND ( + updated_at < ? OR ( + updated_at = ? AND operation_id < ? + ) + ) + ORDER BY updated_at DESC, operation_id DESC LIMIT ?''' + parameters = ( + status, before_updated_at, before_updated_at, + before_operation_id, limit, + ) + rows.extend(self.conn.execute(sql, parameters).fetchall()) + operations = {} + for row in rows: + operation = self._runtime_operation_state(row) + operations[operation['operation_id']] = operation + result = sorted( + operations.values(), + key=lambda item: (item['updated_at'], item['operation_id']), + reverse=True, + )[:limit] + self.conn.commit() + return result + except Exception: + self.conn.rollback() + raise + + def pending_runtime_agent_operations(self, limit=32): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + if type(limit) is not int or not 1 <= limit <= 128: + raise ValueError('runtime agent operation limit is out of range') + try: + rows = self.conn.execute( + '''SELECT * FROM runtime_operations + WHERE target_kind = ? + AND status IN ('requested', 'running') + AND agent_state IN ('pending', 'running') + ORDER BY CASE status WHEN 'running' THEN 0 ELSE 1 END, + updated_at ASC, operation_id ASC + LIMIT ?''', + (RUNTIME_ASYNC_TARGET_KIND, limit), + ).fetchall() + result = [self._runtime_operation_state(row) for row in rows] + self.conn.commit() + return result + except Exception: + self.conn.rollback() + raise + + def runtime_audit_events(self, before_event_id=None, limit=50): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + if ( + before_event_id is not None + and ( + type(before_event_id) is not int + or not 1 <= before_event_id <= RUNTIME_CONTROL_MAX_REVISION + ) + ): + raise ValueError('runtime audit cursor is out of range') + if type(limit) is not int or not 1 <= limit <= 200: + raise ValueError('runtime audit page limit is out of range') + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN') + else: + self.conn.execute( + 'SET TRANSACTION ISOLATION LEVEL REPEATABLE READ READ ONLY' + ) + parameters = [] + # The console exposes only events whose identities can be checked + # against a durable typed operation. + clauses = ['operation_id IS NOT NULL'] + if before_event_id is not None: + clauses.append('id < ?') + parameters.append(before_event_id) + parameters.append(limit + 1) + rows = self.conn.execute( + 'SELECT * FROM runtime_audit_events WHERE ' + ' AND '.join(clauses) + + ' ORDER BY id DESC LIMIT ?', + tuple(parameters), + ).fetchall() + page_rows = rows[:limit] + operation_ids = [] + for row in page_rows: + try: + operation_id = _runtime_operation_id(str(row['operation_id'] or '')) + except (KeyError, TypeError, ValueError) as exc: + raise RuntimeSafetySchemaError( + 'runtime audit operation identity is invalid' + ) from exc + if operation_id not in operation_ids: + operation_ids.append(operation_id) + operations = {} + if operation_ids: + placeholders = ','.join('?' for _value in operation_ids) + operation_rows = self.conn.execute( + 'SELECT * FROM runtime_operations WHERE operation_id IN (' + + placeholders + ')', + tuple(operation_ids), + ).fetchall() + for row in operation_rows: + operation = self._runtime_operation_state(row) + operations[operation['operation_id']] = operation + + events = [] + allowed_results = { + 'accepted', 'succeeded', 'rejected', 'failed', 'rolled_back', + 'canceled', 'failed_hold', + } + for row in page_rows: + try: + event_id = row['id'] + if type(event_id) is not int or event_id < 1: + raise ValueError('event ID') + operation_id = _runtime_operation_id(str(row['operation_id'] or '')) + operation = operations.get(operation_id) + if operation is None: + raise ValueError('operation') + actor = _runtime_actor(str(row['actor'] or '')) + action = str(row['action'] or '') + target_kind = str(row['target_kind'] or '') + target_ref = str(row['target_ref'] or '') + result = str(row['result'] or '') + if ( + actor != operation['actor'] + or action != operation['action'] + or target_kind != operation['target_kind'] + or target_ref != operation['target_ref'] + or result not in allowed_results + ): + raise ValueError('event identity') + safe_category = _runtime_safe_code( + row['safe_category'], 'safe category', + ) + + identities = [] + identity_json = [] + for field in ('before_identity_json', 'after_identity_json'): + raw = row[field] + if raw is None: + identities.append(None) + identity_json.append(None) + continue + raw = str(raw) + parsed = json.loads(raw) + if not isinstance(parsed, dict) or _canonical_runtime_json(parsed) != raw: + raise ValueError('event identity JSON') + identities.append(parsed) + identity_json.append(raw) + before_identity, after_identity = identities + if before_identity != operation['expected_identity'] or ( + after_identity is not None + and after_identity != operation['resulting_identity'] + ): + raise ValueError('event operation identity') + + byte_counts = [] + for field in ('before_bytes', 'after_bytes'): + value = row[field] + if value is not None and ( + type(value) is not int + or not 0 <= value <= RUNTIME_CONTROL_MAX_REVISION + ): + raise ValueError('event byte count') + byte_counts.append(value) + before_bytes, after_bytes = byte_counts + + previous_event_id = row['previous_event_id'] + previous_event_sha256 = row['previous_event_sha256'] + if previous_event_id is None: + if previous_event_sha256 is not None: + raise ValueError('event parent') + elif ( + type(previous_event_id) is not int + or previous_event_id < 1 + or previous_event_id >= event_id + ): + raise ValueError('event parent') + else: + previous_event_sha256 = _runtime_sha256( + str(previous_event_sha256 or ''), 'audit parent hash', + ) + event_sha256 = _runtime_sha256( + str(row['event_sha256'] or ''), 'audit event hash', + ) + created_at = str(row['created_at'] or '') + if not 1 <= len(created_at) <= 64: + raise ValueError('event time') + payload = { + 'schema': 'runtime-audit-event-v1', + 'operation_id': operation_id, + 'actor': actor, + 'action': action, + 'target_kind': target_kind, + 'target_ref': target_ref, + 'result': result, + 'safe_category': safe_category, + 'before_identity_json': identity_json[0], + 'after_identity_json': identity_json[1], + 'before_bytes': before_bytes, + 'after_bytes': after_bytes, + 'previous_event_id': previous_event_id, + 'previous_event_sha256': previous_event_sha256, + 'created_at': created_at, + } + if not hmac.compare_digest( + _runtime_audit_event_sha256(payload), event_sha256, + ): + raise ValueError('event hash') + except ( + KeyError, TypeError, ValueError, json.JSONDecodeError, + ) as exc: + raise RuntimeSafetySchemaError( + 'runtime audit event is invalid' + ) from exc + events.append({ + 'id': event_id, + 'operation_id': operation_id, + 'actor': actor, + 'action': action, + 'target_kind': target_kind, + 'target_ref': target_ref, + 'result': result, + 'safe_category': safe_category, + 'before_identity': before_identity, + 'after_identity': after_identity, + 'before_bytes': before_bytes, + 'after_bytes': after_bytes, + 'previous_event_id': previous_event_id, + 'previous_event_sha256': previous_event_sha256, + 'event_sha256': event_sha256, + 'created_at': created_at, + }) + next_before_event_id = ( + events[-1]['id'] if len(rows) > limit and events else None + ) + self.conn.commit() + return { + 'events': events, + 'next_before_event_id': next_before_event_id, + } + except Exception: + self.conn.rollback() + raise + + def create_runtime_operation(self, *, operation_id, actor, action, expected_identity): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + actor = _runtime_actor(actor) + if not isinstance(action, str) or action not in RUNTIME_ASYNC_ACTIONS: + raise ValueError('runtime operation action is invalid') + expected_identity = _runtime_expected_identity(expected_identity, action) + expected_json = _canonical_runtime_json(expected_identity) + target_ref = RUNTIME_ASYNC_ACTION_TARGETS[action] + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + existing_row = self._locked_runtime_operation(operation_id) + if existing_row: + existing = self._runtime_operation_state(existing_row) + if not ( + existing['actor'] == actor + and existing['action'] == action + and existing['target_kind'] == RUNTIME_ASYNC_TARGET_KIND + and existing['target_ref'] == target_ref + and existing['expected_identity'] == expected_identity + ): + raise RuntimeOperationIdentityConflictError( + 'runtime operation ID is already bound to another request' + ) + accepted = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + WHERE operation_id = ? AND result = 'accepted' + ORDER BY id LIMIT 2''', + (operation_id,), + ).fetchall() + if len(accepted) != 1: + raise RuntimeSafetySchemaError( + 'runtime operation accepted audit evidence is incomplete' + ) + self.conn.commit() + return { + **existing, 'replayed': True, + 'audit_event_id': int(accepted[0]['id']), + 'audit_event_sha256': str(accepted[0]['event_sha256']), + } + now = utc_now_iso() + self.conn.execute( + '''INSERT INTO runtime_operations( + operation_id, actor, action, target_kind, target_ref, + status, expected_identity_json, agent_state, + requested_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'requested', ?, 'pending', ?, ?)''', + ( + operation_id, actor, action, RUNTIME_ASYNC_TARGET_KIND, + target_ref, expected_json, now, now, + ), + ) + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=actor, action=action, + target_kind=RUNTIME_ASYNC_TARGET_KIND, target_ref=target_ref, + result='accepted', before_identity=expected_identity, + created_at=now, + ) + row = self._locked_runtime_operation(operation_id) + result = self._runtime_operation_state(row) + self.conn.commit() + return {**result, 'replayed': False, **audit} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError('runtime operation creation retry budget exhausted') + + def mark_runtime_operation_running(self, operation_id): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + row = self._locked_runtime_operation(operation_id) + if not row: + self.conn.commit() + return None + current = self._runtime_operation_state(row) + if current['action'] not in RUNTIME_ASYNC_ACTIONS: + raise RuntimeOperationTransitionError( + 'runtime operation does not use agent lifecycle' + ) + if current['status'] == 'running': + self.conn.commit() + return {**current, 'replayed': True} + if current['status'] != 'requested': + raise RuntimeOperationTransitionError( + 'runtime operation cannot enter running state' + ) + now = utc_now_iso() + updated = self.conn.execute( + '''UPDATE runtime_operations + SET status = 'running', agent_state = 'running', + started_at = ?, updated_at = ? + WHERE operation_id = ? AND status = 'requested' + AND agent_state = 'pending' ''', + (now, now, operation_id), + ) + if int(updated.rowcount or 0) != 1: + raise RuntimeOperationTransitionError( + 'runtime operation running transition conflicted' + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return {**result, 'replayed': False} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError('runtime operation transition retry budget exhausted') + + def claim_runtime_operation_execution(self, *, operation_id, action, expected_identity): + """Atomically bind a host-agent claim to persisted request identity.""" + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + if not isinstance(action, str) or action not in RUNTIME_ASYNC_ACTIONS: + raise ValueError('runtime operation action is invalid') + expected_identity = _runtime_expected_identity(expected_identity, action) + expected_json = _canonical_runtime_json(expected_identity) + target_ref = RUNTIME_ASYNC_ACTION_TARGETS[action] + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + row = self._locked_runtime_operation(operation_id) + if not row: + raise RuntimeOperationTransitionError( + 'runtime operation is unavailable' + ) + current = self._runtime_operation_state(row) + if not ( + current['action'] == action + and current['target_kind'] == RUNTIME_ASYNC_TARGET_KIND + and current['target_ref'] == target_ref + and current['expected_identity'] == expected_identity + ): + raise RuntimeOperationIdentityConflictError( + 'runtime operation identity does not match the persisted request' + ) + if any(( + current['expected_revision'] is not None, + current['resulting_identity'] is not None, + current['resulting_revision'] is not None, + current['completed_at'] is not None, + current['agent_reconciled_at'] is not None, + current['safe_category'] is not None, + current['safe_detail'] is not None, + current['agent_result_sha256'] is not None, + )): + raise RuntimeOperationTransitionError( + 'runtime operation cannot be claimed for execution' + ) + requested = ( + current['status'] == 'requested' + and current['agent_state'] == 'pending' + and current['started_at'] is None + ) + running = ( + current['status'] == 'running' + and current['agent_state'] == 'running' + and current['started_at'] is not None + ) + if not requested and not running: + raise RuntimeOperationTransitionError( + 'runtime operation cannot be claimed for execution' + ) + + audits = self.conn.execute( + '''SELECT * FROM runtime_audit_events + WHERE operation_id = ? ORDER BY id LIMIT 2''', + (operation_id,), + ).fetchall() + if len(audits) != 1: + raise RuntimeSafetySchemaError( + 'runtime operation accepted audit evidence is incomplete' + ) + audit = audits[0] + try: + audit_id = int(audit['id']) + before_json = str(audit['before_identity_json'] or '') + previous_event_id = audit['previous_event_id'] + previous_event_sha256 = audit['previous_event_sha256'] + if previous_event_id is None: + if previous_event_sha256 is not None: + raise ValueError('audit parent') + else: + if ( + type(previous_event_id) is not int + or previous_event_id < 1 + or previous_event_id >= audit_id + ): + raise ValueError('audit parent') + previous_event_sha256 = _runtime_sha256( + str(previous_event_sha256 or ''), 'audit parent hash', + ) + predecessor = self.conn.execute( + '''SELECT * FROM runtime_audit_events + WHERE id < ? ORDER BY id DESC LIMIT 1''', + (audit_id,), + ).fetchone() + if predecessor is None: + if previous_event_id is not None: + raise ValueError('audit parent') + else: + predecessor_id, predecessor_sha256 = ( + _runtime_audit_row_hash(predecessor) + ) + if not ( + predecessor_id == previous_event_id + and hmac.compare_digest( + predecessor_sha256, + previous_event_sha256 or '', + ) + ): + raise ValueError('audit parent') + event_sha256 = _runtime_sha256( + str(audit['event_sha256'] or ''), 'audit event hash', + ) + if not ( + audit_id >= 1 + and str(audit['operation_id'] or '') == operation_id + and str(audit['actor'] or '') == current['actor'] + and str(audit['action'] or '') == action + and str(audit['target_kind'] or '') == RUNTIME_ASYNC_TARGET_KIND + and str(audit['target_ref'] or '') == target_ref + and str(audit['result'] or '') == 'accepted' + and audit['safe_category'] is None + and before_json == expected_json + and audit['after_identity_json'] is None + and audit['before_bytes'] is None + and audit['after_bytes'] is None + and str(audit['created_at'] or '') == current['requested_at'] + ): + raise ValueError('accepted audit') + payload = { + 'schema': 'runtime-audit-event-v1', + 'operation_id': operation_id, + 'actor': current['actor'], + 'action': action, + 'target_kind': RUNTIME_ASYNC_TARGET_KIND, + 'target_ref': target_ref, + 'result': 'accepted', + 'safe_category': None, + 'before_identity_json': expected_json, + 'after_identity_json': None, + 'before_bytes': None, + 'after_bytes': None, + 'previous_event_id': previous_event_id, + 'previous_event_sha256': previous_event_sha256, + 'created_at': current['requested_at'], + } + if not hmac.compare_digest( + _runtime_audit_event_sha256(payload), event_sha256, + ): + raise ValueError('audit hash') + except (KeyError, TypeError, ValueError) as exc: + raise RuntimeSafetySchemaError( + 'runtime operation accepted audit evidence is invalid' + ) from exc + + if running: + self.conn.commit() + return { + **current, 'replayed': True, + 'audit_event_id': audit_id, + 'audit_event_sha256': event_sha256, + } + now = utc_now_iso() + updated = self.conn.execute( + '''UPDATE runtime_operations + SET status = 'running', agent_state = 'running', + started_at = ?, updated_at = ? + WHERE operation_id = ? AND action = ? + AND target_kind = ? AND target_ref = ? + AND expected_identity_json = ? + AND status = 'requested' AND agent_state = 'pending' ''', + ( + now, now, operation_id, action, RUNTIME_ASYNC_TARGET_KIND, + target_ref, expected_json, + ), + ) + if int(updated.rowcount or 0) != 1: + raise RuntimeOperationTransitionError( + 'runtime operation claim conflicted' + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return { + **result, 'replayed': False, + 'audit_event_id': audit_id, + 'audit_event_sha256': event_sha256, + } + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError('runtime operation claim retry budget exhausted') + + def create_runtime_source_operation( + self, *, operation_id, actor, source_id, source_action, + interval_seconds=None, mode=None, restart_enabled=None, + restart_delay_seconds=None, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + actor = _runtime_actor(actor) + if _runtime_source_id(source_id) is None or source_action not in RUNTIME_SOURCE_ACTIONS: + raise ValueError('runtime source operation request is invalid') + action = f'supervisor.source.{source_action}' + identity = { + 'source_id': source_id, + 'source_action': source_action, + 'interval_seconds': interval_seconds, + } + if source_action in ('once', 'set-mode', 'set-restart', 'set-restart-delay'): + identity.update({ + 'mode': mode, + 'restart_enabled': restart_enabled, + 'restart_delay_seconds': restart_delay_seconds, + }) + elif any( + value is not None + for value in (mode, restart_enabled, restart_delay_seconds) + ): + raise ValueError('runtime source operation request is invalid') + expected = _runtime_source_expected_identity(identity, action, source_id) + expected_json = _canonical_runtime_json(expected) + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + existing_row = self._locked_runtime_operation(operation_id) + if existing_row: + existing = self._runtime_operation_state(existing_row) + if not ( + existing['actor'] == actor + and existing['action'] == action + and existing['target_kind'] == RUNTIME_SOURCE_TARGET_KIND + and existing['target_ref'] == source_id + and existing['expected_identity'] == expected + ): + raise RuntimeOperationIdentityConflictError( + 'runtime operation ID is already bound to another request' + ) + accepted = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + WHERE operation_id = ? AND result = 'accepted' + ORDER BY id LIMIT 2''', + (operation_id,), + ).fetchall() + if len(accepted) != 1: + raise RuntimeSafetySchemaError( + 'runtime source operation accepted audit evidence is incomplete' + ) + self.conn.commit() + return { + **existing, 'replayed': True, + 'audit_event_id': int(accepted[0]['id']), + 'audit_event_sha256': str(accepted[0]['event_sha256']), + } + now = utc_now_iso() + self.conn.execute( + '''INSERT INTO runtime_operations( + operation_id, actor, action, target_kind, target_ref, + status, expected_identity_json, agent_state, + requested_at, started_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'running', ?, 'not_required', ?, ?, ?)''', + ( + operation_id, actor, action, RUNTIME_SOURCE_TARGET_KIND, + source_id, expected_json, now, now, now, + ), + ) + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=actor, action=action, + target_kind=RUNTIME_SOURCE_TARGET_KIND, target_ref=source_id, + result='accepted', before_identity=expected, created_at=now, + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return {**result, 'replayed': False, **audit} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError('runtime source operation creation retry budget exhausted') + + def complete_runtime_source_operation( + self, operation_id, *, succeeded, outcome=None, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + if type(succeeded) is not bool: + raise ValueError('runtime source operation result is invalid') + if succeeded: + if outcome not in ('completed', 'dependency-blocked'): + raise ValueError('runtime source operation result is invalid') + elif outcome is not None: + raise ValueError('runtime source operation result is invalid') + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + row = self._locked_runtime_operation(operation_id) + if not row: + self.conn.commit() + return None + current = self._runtime_operation_state(row) + if current['action'] not in RUNTIME_SOURCE_OPERATION_ACTIONS: + raise RuntimeOperationTransitionError( + 'runtime operation is not a managed source action' + ) + status = 'succeeded' if succeeded else 'failed' + safe_category = None if succeeded else 'supervisor_action_failed' + resulting = None + if succeeded: + resulting = _runtime_source_resulting_identity({ + 'source_id': current['target_ref'], + 'source_action': RUNTIME_SOURCE_OPERATION_ACTIONS[current['action']], + 'outcome': outcome, + }, current['action'], current['target_ref']) + if current['status'] in ('succeeded', 'failed'): + if ( + current['status'] != status + or current['safe_category'] != safe_category + or current['resulting_identity'] != resulting + ): + raise RuntimeOperationIdentityConflictError( + 'runtime source operation already has another result' + ) + events = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + WHERE operation_id = ? AND result = ? + ORDER BY id LIMIT 2''', + (operation_id, status), + ).fetchall() + if len(events) != 1: + raise RuntimeSafetySchemaError( + 'runtime source operation terminal audit evidence is incomplete' + ) + self.conn.commit() + return { + **current, 'replayed': True, + 'audit_event_id': int(events[0]['id']), + 'audit_event_sha256': str(events[0]['event_sha256']), + } + if current['status'] != 'running' or current['agent_state'] != 'not_required': + raise RuntimeOperationTransitionError( + 'runtime source operation cannot be completed' + ) + now = utc_now_iso() + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=current['actor'], + action=current['action'], target_kind=RUNTIME_SOURCE_TARGET_KIND, + target_ref=current['target_ref'], result=status, + safe_category=safe_category, + before_identity=current['expected_identity'], + after_identity=resulting, created_at=now, + ) + resulting_json = ( + _canonical_runtime_json(resulting) if resulting is not None else None + ) + updated = self.conn.execute( + '''UPDATE runtime_operations + SET status = ?, safe_category = ?, resulting_identity_json = ?, + completed_at = ?, updated_at = ? + WHERE operation_id = ? AND status = 'running' + AND agent_state = 'not_required' ''', + ( + status, safe_category, resulting_json, now, now, + operation_id, + ), + ) + if int(updated.rowcount or 0) != 1: + raise RuntimeOperationTransitionError( + 'runtime source operation completion conflicted' + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return {**result, 'replayed': False, **audit} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError('runtime source operation completion retry budget exhausted') + + def create_runtime_worker_admin_operation( + self, *, operation_id, actor, action, target_ref, request_sha256, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + actor = _runtime_actor(actor) + target_ref = _runtime_worker_admin_target(action, target_ref) + expected = _runtime_worker_admin_expected_identity({ + 'action': action, + 'target_ref': target_ref, + 'request_sha256': request_sha256, + }, action, target_ref) + expected_json = _canonical_runtime_json(expected) + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + existing_row = self._locked_runtime_operation(operation_id) + if existing_row: + existing = self._runtime_operation_state(existing_row) + if not ( + existing['actor'] == actor + and existing['action'] == action + and existing['target_kind'] == RUNTIME_WORKER_ADMIN_TARGET_KIND + and existing['target_ref'] == target_ref + and existing['expected_identity'] == expected + ): + raise RuntimeOperationIdentityConflictError( + 'runtime operation ID is already bound to another request' + ) + accepted = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + WHERE operation_id = ? AND result = 'accepted' + ORDER BY id LIMIT 2''', + (operation_id,), + ).fetchall() + if len(accepted) != 1: + raise RuntimeSafetySchemaError( + 'runtime worker admin accepted audit evidence is incomplete' + ) + self.conn.commit() + return { + **existing, 'replayed': True, + 'audit_event_id': int(accepted[0]['id']), + 'audit_event_sha256': str(accepted[0]['event_sha256']), + } + now = utc_now_iso() + self.conn.execute( + '''INSERT INTO runtime_operations( + operation_id, actor, action, target_kind, target_ref, + status, expected_identity_json, agent_state, + requested_at, started_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'running', ?, 'not_required', ?, ?, ?)''', + ( + operation_id, actor, action, RUNTIME_WORKER_ADMIN_TARGET_KIND, + target_ref, expected_json, now, now, now, + ), + ) + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=actor, action=action, + target_kind=RUNTIME_WORKER_ADMIN_TARGET_KIND, + target_ref=target_ref, result='accepted', + before_identity=expected, created_at=now, + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return {**result, 'replayed': False, **audit} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError( + 'runtime worker admin operation creation retry budget exhausted' + ) + + def complete_runtime_worker_admin_operation( + self, operation_id, *, succeeded, affected_count=None, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + if type(succeeded) is not bool: + raise ValueError('runtime worker admin operation result is invalid') + if succeeded: + if type(affected_count) is not int or not 0 <= affected_count <= 10000: + raise ValueError('runtime worker admin operation result is invalid') + elif affected_count is not None: + raise ValueError('runtime worker admin operation result is invalid') + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + row = self._locked_runtime_operation(operation_id) + if not row: + self.conn.commit() + return None + current = self._runtime_operation_state(row) + if current['action'] not in RUNTIME_WORKER_ADMIN_ACTION_TARGETS: + raise RuntimeOperationTransitionError( + 'runtime operation is not a worker admin action' + ) + status = 'succeeded' if succeeded else 'failed' + safe_category = None if succeeded else 'worker_admin_mutation_failed' + resulting = None + if succeeded: + resulting = _runtime_worker_admin_resulting_identity({ + 'action': current['action'], + 'target_ref': current['target_ref'], + 'outcome': 'completed', + 'affected_count': affected_count, + }, current['action'], current['target_ref']) + if current['status'] in ('succeeded', 'failed'): + if ( + current['status'] != status + or current['safe_category'] != safe_category + or current['resulting_identity'] != resulting + ): + raise RuntimeOperationIdentityConflictError( + 'runtime worker admin operation already has another result' + ) + events = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + WHERE operation_id = ? AND result = ? + ORDER BY id LIMIT 2''', + (operation_id, status), + ).fetchall() + if len(events) != 1: + raise RuntimeSafetySchemaError( + 'runtime worker admin terminal audit evidence is incomplete' + ) + self.conn.commit() + return { + **current, 'replayed': True, + 'audit_event_id': int(events[0]['id']), + 'audit_event_sha256': str(events[0]['event_sha256']), + } + if current['status'] != 'running' or current['agent_state'] != 'not_required': + raise RuntimeOperationTransitionError( + 'runtime worker admin operation cannot be completed' + ) + now = utc_now_iso() + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=current['actor'], + action=current['action'], + target_kind=RUNTIME_WORKER_ADMIN_TARGET_KIND, + target_ref=current['target_ref'], result=status, + safe_category=safe_category, + before_identity=current['expected_identity'], + after_identity=resulting, created_at=now, + ) + resulting_json = ( + _canonical_runtime_json(resulting) if resulting is not None else None + ) + updated = self.conn.execute( + '''UPDATE runtime_operations + SET status = ?, safe_category = ?, resulting_identity_json = ?, + completed_at = ?, updated_at = ? + WHERE operation_id = ? AND status = 'running' + AND agent_state = 'not_required' ''', + ( + status, safe_category, resulting_json, now, now, + operation_id, + ), + ) + if int(updated.rowcount or 0) != 1: + raise RuntimeOperationTransitionError( + 'runtime worker admin operation completion conflicted' + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return {**result, 'replayed': False, **audit} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError( + 'runtime worker admin operation completion retry budget exhausted' + ) + + def create_runtime_managed_file_operation( + self, *, operation_id, actor, action, root_id, relative_path, + expected_sha256, proposed_sha256, proposed_byte_count, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + actor = _runtime_actor(actor) + if action not in RUNTIME_MANAGED_FILE_ACTIONS: + raise ValueError('runtime managed file operation action is invalid') + expected = _runtime_managed_file_expected_identity({ + 'root_id': root_id, + 'relative_path': relative_path, + 'expected_sha256': expected_sha256, + 'proposed_sha256': proposed_sha256, + 'proposed_byte_count': proposed_byte_count, + }, action, root_id) + expected_json = _canonical_runtime_json(expected) + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + existing_row = self._locked_runtime_operation(operation_id) + if existing_row: + existing = self._runtime_operation_state(existing_row) + if not ( + existing['actor'] == actor + and existing['action'] == action + and existing['target_kind'] == RUNTIME_MANAGED_FILE_TARGET_KIND + and existing['target_ref'] == root_id + and existing['expected_identity'] == expected + ): + raise RuntimeOperationIdentityConflictError( + 'runtime operation ID is already bound to another request' + ) + accepted = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + WHERE operation_id = ? AND result = 'accepted' + ORDER BY id LIMIT 2''', + (operation_id,), + ).fetchall() + if len(accepted) != 1: + raise RuntimeSafetySchemaError( + 'runtime managed file accepted audit evidence is incomplete' + ) + self.conn.commit() + return { + **existing, 'replayed': True, + 'audit_event_id': int(accepted[0]['id']), + 'audit_event_sha256': str(accepted[0]['event_sha256']), + } + now = utc_now_iso() + self.conn.execute( + '''INSERT INTO runtime_operations( + operation_id, actor, action, target_kind, target_ref, + status, expected_identity_json, agent_state, + requested_at, started_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'running', ?, 'not_required', ?, ?, ?)''', + ( + operation_id, actor, action, RUNTIME_MANAGED_FILE_TARGET_KIND, + root_id, expected_json, now, now, now, + ), + ) + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=actor, action=action, + target_kind=RUNTIME_MANAGED_FILE_TARGET_KIND, + target_ref=root_id, result='accepted', + before_identity=expected, + after_bytes=expected['proposed_byte_count'], created_at=now, + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return {**result, 'replayed': False, **audit} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError( + 'runtime managed file operation creation retry budget exhausted' + ) + + def complete_runtime_managed_file_operation( + self, operation_id, *, succeeded, before_sha256=None, + before_byte_count=None, after_sha256=None, after_byte_count=None, + written=None, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + if type(succeeded) is not bool: + raise ValueError('runtime managed file operation result is invalid') + supplied = ( + before_sha256, before_byte_count, after_sha256, after_byte_count, written, + ) + if not succeeded and any(value is not None for value in supplied): + raise ValueError('runtime managed file operation result is invalid') + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + row = self._locked_runtime_operation(operation_id) + if not row: + self.conn.commit() + return None + current = self._runtime_operation_state(row) + if current['action'] not in RUNTIME_MANAGED_FILE_ACTIONS: + raise RuntimeOperationTransitionError( + 'runtime operation is not a managed file action' + ) + status = 'succeeded' if succeeded else 'failed' + safe_category = None if succeeded else 'managed_file_mutation_failed' + resulting = None + if succeeded: + resulting = _runtime_managed_file_resulting_identity({ + 'root_id': current['target_ref'], + 'relative_path': current['expected_identity']['relative_path'], + 'outcome': 'completed', + 'before_sha256': before_sha256, + 'before_byte_count': before_byte_count, + 'after_sha256': after_sha256, + 'after_byte_count': after_byte_count, + 'written': written, + }, current['action'], current['target_ref']) + expected = current['expected_identity'] + if ( + current['action'] == 'files.create' + and ( + resulting['after_sha256'] != expected['proposed_sha256'] + or resulting['after_byte_count'] != expected['proposed_byte_count'] + ) + or current['action'] == 'files.replace' + and ( + resulting['before_sha256'] != expected['expected_sha256'] + or resulting['after_sha256'] != expected['proposed_sha256'] + or resulting['after_byte_count'] != expected['proposed_byte_count'] + or not resulting['written'] and ( + resulting['before_sha256'] != resulting['after_sha256'] + or resulting['before_byte_count'] != resulting['after_byte_count'] + ) + ) + or current['action'] == 'files.delete' + and resulting['before_sha256'] != expected['expected_sha256'] + ): + raise RuntimeOperationIdentityConflictError( + 'runtime managed file result does not match its accepted identity' + ) + if current['status'] in ('succeeded', 'failed'): + if ( + current['status'] != status + or current['safe_category'] != safe_category + or current['resulting_identity'] != resulting + ): + raise RuntimeOperationIdentityConflictError( + 'runtime managed file operation already has another result' + ) + events = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + WHERE operation_id = ? AND result = ? + ORDER BY id LIMIT 2''', + (operation_id, status), + ).fetchall() + if len(events) != 1: + raise RuntimeSafetySchemaError( + 'runtime managed file terminal audit evidence is incomplete' + ) + self.conn.commit() + return { + **current, 'replayed': True, + 'audit_event_id': int(events[0]['id']), + 'audit_event_sha256': str(events[0]['event_sha256']), + } + if current['status'] != 'running' or current['agent_state'] != 'not_required': + raise RuntimeOperationTransitionError( + 'runtime managed file operation cannot be completed' + ) + now = utc_now_iso() + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=current['actor'], + action=current['action'], + target_kind=RUNTIME_MANAGED_FILE_TARGET_KIND, + target_ref=current['target_ref'], result=status, + safe_category=safe_category, + before_identity=current['expected_identity'], + after_identity=resulting, + before_bytes=( + resulting['before_byte_count'] if resulting is not None else None + ), + after_bytes=( + resulting['after_byte_count'] if resulting is not None else None + ), + created_at=now, + ) + resulting_json = ( + _canonical_runtime_json(resulting) if resulting is not None else None + ) + updated = self.conn.execute( + '''UPDATE runtime_operations + SET status = ?, safe_category = ?, resulting_identity_json = ?, + completed_at = ?, updated_at = ? + WHERE operation_id = ? AND status = 'running' + AND agent_state = 'not_required' ''', + ( + status, safe_category, resulting_json, now, now, + operation_id, + ), + ) + if int(updated.rowcount or 0) != 1: + raise RuntimeOperationTransitionError( + 'runtime managed file operation completion conflicted' + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return {**result, 'replayed': False, **audit} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError( + 'runtime managed file operation completion retry budget exhausted' + ) + + def create_runtime_document_operation( + self, *, operation_id, actor, action, active_config_sha256, + active_secrets_sha256, candidate_config_sha256, candidate_secrets_sha256, + candidate_after_sha256, candidate_before_bytes, candidate_after_bytes, + candidate_before_present, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + actor = _runtime_actor(actor) + target_ref = RUNTIME_DOCUMENT_ACTION_TARGETS.get(action) + if target_ref is None: + raise ValueError('runtime document operation action is invalid') + expected = _runtime_document_expected_identity({ + 'document': target_ref, + 'active_config_sha256': active_config_sha256, + 'active_secrets_sha256': active_secrets_sha256, + 'candidate_config_sha256': candidate_config_sha256, + 'candidate_secrets_sha256': candidate_secrets_sha256, + 'candidate_after_sha256': candidate_after_sha256, + 'candidate_before_bytes': candidate_before_bytes, + 'candidate_after_bytes': candidate_after_bytes, + 'candidate_before_present': candidate_before_present, + }, action, target_ref) + expected_json = _canonical_runtime_json(expected) + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + existing_row = self._locked_runtime_operation(operation_id) + if existing_row: + existing = self._runtime_operation_state(existing_row) + if not ( + existing['actor'] == actor + and existing['action'] == action + and existing['target_kind'] == RUNTIME_DOCUMENT_TARGET_KIND + and existing['target_ref'] == target_ref + and existing['expected_identity'] == expected + ): + raise RuntimeOperationIdentityConflictError( + 'runtime operation ID is already bound to another request' + ) + accepted = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + WHERE operation_id = ? AND result = 'accepted' + ORDER BY id LIMIT 2''', + (operation_id,), + ).fetchall() + if len(accepted) != 1: + raise RuntimeSafetySchemaError( + 'runtime document accepted audit evidence is incomplete' + ) + self.conn.commit() + return { + **existing, 'replayed': True, + 'audit_event_id': int(accepted[0]['id']), + 'audit_event_sha256': str(accepted[0]['event_sha256']), + } + now = utc_now_iso() + self.conn.execute( + '''INSERT INTO runtime_operations( + operation_id, actor, action, target_kind, target_ref, + status, expected_identity_json, agent_state, + requested_at, started_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'running', ?, 'not_required', ?, ?, ?)''', + ( + operation_id, actor, action, RUNTIME_DOCUMENT_TARGET_KIND, + target_ref, expected_json, now, now, now, + ), + ) + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=actor, action=action, + target_kind=RUNTIME_DOCUMENT_TARGET_KIND, + target_ref=target_ref, result='accepted', + before_identity=expected, + before_bytes=expected['candidate_before_bytes'], + after_bytes=expected['candidate_after_bytes'], created_at=now, + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return {**result, 'replayed': False, **audit} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError( + 'runtime document operation creation retry budget exhausted' + ) + + def complete_runtime_document_operation( + self, operation_id, *, succeeded, candidate_sha256=None, written=None, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + if type(succeeded) is not bool: + raise ValueError('runtime document operation result is invalid') + if succeeded: + if type(written) is not bool: + raise ValueError('runtime document operation result is invalid') + candidate_sha256 = _runtime_sha256( + candidate_sha256, 'candidate result hash', + ) + elif candidate_sha256 is not None or written is not None: + raise ValueError('runtime document operation result is invalid') + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + row = self._locked_runtime_operation(operation_id) + if not row: + self.conn.commit() + return None + current = self._runtime_operation_state(row) + if current['action'] not in RUNTIME_DOCUMENT_ACTION_TARGETS: + raise RuntimeOperationTransitionError( + 'runtime operation is not a document action' + ) + if succeeded and not hmac.compare_digest( + candidate_sha256, + current['expected_identity']['candidate_after_sha256'], + ): + raise RuntimeOperationIdentityConflictError( + 'runtime document result does not match its accepted candidate' + ) + status = 'succeeded' if succeeded else 'failed' + safe_category = None if succeeded else 'runtime_document_save_failed' + resulting = None + if succeeded: + resulting = _runtime_document_resulting_identity({ + 'document': current['target_ref'], + 'outcome': 'completed', + 'candidate_sha256': candidate_sha256, + 'written': written, + }, current['action'], current['target_ref']) + if current['status'] in ('succeeded', 'failed'): + if ( + current['status'] != status + or current['safe_category'] != safe_category + or current['resulting_identity'] != resulting + ): + raise RuntimeOperationIdentityConflictError( + 'runtime document operation already has another result' + ) + events = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + WHERE operation_id = ? AND result = ? + ORDER BY id LIMIT 2''', + (operation_id, status), + ).fetchall() + if len(events) != 1: + raise RuntimeSafetySchemaError( + 'runtime document terminal audit evidence is incomplete' + ) + self.conn.commit() + return { + **current, 'replayed': True, + 'audit_event_id': int(events[0]['id']), + 'audit_event_sha256': str(events[0]['event_sha256']), + } + if current['status'] != 'running' or current['agent_state'] != 'not_required': + raise RuntimeOperationTransitionError( + 'runtime document operation cannot be completed' + ) + now = utc_now_iso() + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=current['actor'], + action=current['action'], target_kind=RUNTIME_DOCUMENT_TARGET_KIND, + target_ref=current['target_ref'], result=status, + safe_category=safe_category, + before_identity=current['expected_identity'], + after_identity=resulting, + before_bytes=current['expected_identity']['candidate_before_bytes'], + after_bytes=current['expected_identity']['candidate_after_bytes'], + created_at=now, + ) + resulting_json = ( + _canonical_runtime_json(resulting) if resulting is not None else None + ) + updated = self.conn.execute( + '''UPDATE runtime_operations + SET status = ?, safe_category = ?, resulting_identity_json = ?, + completed_at = ?, updated_at = ? + WHERE operation_id = ? AND status = 'running' + AND agent_state = 'not_required' ''', + ( + status, safe_category, resulting_json, now, now, + operation_id, + ), + ) + if int(updated.rowcount or 0) != 1: + raise RuntimeOperationTransitionError( + 'runtime document operation completion conflicted' + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return {**result, 'replayed': False, **audit} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError( + 'runtime document operation completion retry budget exhausted' + ) + + def reconcile_runtime_operation_result( + self, result_envelope, *, max_bytes=RUNTIME_AGENT_RESULT_MAX_BYTES, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + parsed, result_sha256, failure = _runtime_result_envelope( + result_envelope, max_bytes, + ) + result_envelope = None + if failure: + raise ValueError(failure) + operation_id = parsed['operation_id'] + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._locked_runtime_control_state() + row = self._locked_runtime_operation(operation_id) + if not row: + self.conn.commit() + return None + current = self._runtime_operation_state(row) + if ( + current['action'] != parsed['action'] + or current['action'] not in RUNTIME_ASYNC_ACTIONS + ): + raise RuntimeOperationIdentityConflictError( + 'runtime operation result does not match its request' + ) + existing_digest = current['agent_result_sha256'] + if existing_digest is not None: + if not hmac.compare_digest(existing_digest, result_sha256): + raise RuntimeOperationIdentityConflictError( + 'runtime operation already has another result' + ) + if ( + current['status'] != parsed['result'] + or current['agent_state'] != parsed['result'] + or current['safe_category'] != parsed['safe_category'] + or current['safe_detail'] != parsed['safe_detail'] + or current['resulting_identity'] != parsed['resulting_identity'] + ): + raise RuntimeSafetySchemaError( + 'runtime operation replay result is inconsistent' + ) + terminal_events = self.conn.execute( + '''SELECT id, event_sha256 FROM runtime_audit_events + WHERE operation_id = ? AND result = ? + ORDER BY id LIMIT 2''', + (operation_id, parsed['result']), + ).fetchall() + if len(terminal_events) != 1: + raise RuntimeSafetySchemaError( + 'runtime operation terminal audit evidence is incomplete' + ) + self.conn.commit() + return { + **current, 'replayed': True, + 'audit_event_id': int(terminal_events[0]['id']), + 'audit_event_sha256': str(terminal_events[0]['event_sha256']), + } + if current['status'] not in ('requested', 'running'): + raise RuntimeOperationTransitionError( + 'runtime operation is already terminal' + ) + if parsed['result'] == 'rolled_back': + expected = current['expected_identity'] + if parsed['resulting_identity'] != { + 'active_config_sha256': expected['active_config_sha256'], + 'active_secrets_sha256': expected['active_secrets_sha256'], + }: + raise RuntimeOperationIdentityConflictError( + 'rolled-back runtime operation identity is invalid' + ) + elif parsed['result'] == 'succeeded': + if parsed['resulting_identity'] != _runtime_success_identity( + current['expected_identity'], current['action'], + ): + raise RuntimeOperationIdentityConflictError( + 'successful runtime operation identity is invalid' + ) + now = utc_now_iso() + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=current['actor'], + action=current['action'], target_kind=current['target_kind'], + target_ref=current['target_ref'], result=parsed['result'], + safe_category=parsed['safe_category'], + before_identity=current['expected_identity'], + after_identity=parsed['resulting_identity'], created_at=now, + ) + resulting_json = ( + _canonical_runtime_json(parsed['resulting_identity']) + if parsed['resulting_identity'] is not None else None + ) + updated = self.conn.execute( + '''UPDATE runtime_operations + SET status = ?, safe_category = ?, safe_detail = ?, + resulting_identity_json = ?, agent_state = ?, + agent_result_sha256 = ?, + started_at = COALESCE(started_at, ?), + completed_at = ?, agent_reconciled_at = ?, updated_at = ? + WHERE operation_id = ? AND status IN ('requested','running') + AND agent_result_sha256 IS NULL''', + ( + parsed['result'], parsed['safe_category'], parsed['safe_detail'], + resulting_json, parsed['result'], result_sha256, now, now, + now, now, operation_id, + ), + ) + if int(updated.rowcount or 0) != 1: + raise RuntimeOperationTransitionError( + 'runtime operation reconciliation conflicted' + ) + result = self._runtime_operation_state( + self._locked_runtime_operation(operation_id), + ) + self.conn.commit() + return {**result, 'replayed': False, **audit} + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError('runtime operation reconciliation retry budget exhausted') + + def _runtime_control_replay_locked( + self, *, operation_id, actor, action, target_ref, expected_revision, + ): + operation = self.conn.execute( + 'SELECT * FROM runtime_operations WHERE operation_id = ?', + (operation_id,), + ).fetchone() + if not operation: + return None + stored_expected_revision = operation['expected_revision'] + stored_resulting_revision = operation['resulting_revision'] + revisions_valid = ( + isinstance(stored_expected_revision, int) + and not isinstance(stored_expected_revision, bool) + and isinstance(stored_resulting_revision, int) + and not isinstance(stored_resulting_revision, bool) + ) + immutable_request = ( + str(operation['actor'] or '') == actor + and str(operation['action'] or '') == action + and str(operation['target_kind'] or '') == RUNTIME_CONTROL_TARGET_KIND + and str(operation['target_ref'] or '') == target_ref + ) + if not immutable_request: + raise RuntimeOperationIdentityConflictError( + 'runtime operation ID is already bound to another request' + ) + if not revisions_valid: + raise RuntimeSafetySchemaError('runtime control replay revisions are invalid') + if stored_expected_revision != expected_revision: + raise RuntimeOperationIdentityConflictError( + 'runtime operation ID is already bound to another request' + ) + if ( + operation['status'] != 'succeeded' + or operation['agent_state'] != 'not_required' + or operation['safe_category'] is not None + or operation['safe_detail'] is not None + or operation['agent_result_sha256'] is not None + or not operation['requested_at'] + or not operation['started_at'] + or not operation['completed_at'] + or not operation['updated_at'] + ): + raise RuntimeSafetySchemaError('runtime control replay operation is incomplete') + before_json = str(operation['expected_identity_json'] or '') + after_json = str(operation['resulting_identity_json'] or '') + before = _stored_runtime_control_identity(before_json) + after = _stored_runtime_control_identity(after_json) + if ( + before['revision'] != expected_revision + or stored_resulting_revision != after['revision'] + or after['revision'] != before['revision'] + 1 + ): + raise RuntimeSafetySchemaError('runtime control replay revisions are invalid') + try: + expected_after = _runtime_control_transition(before, action) + except (RuntimeControlTransitionError, ValueError) as exc: + raise RuntimeSafetySchemaError( + 'runtime control replay transition is invalid' + ) from exc + if after != expected_after: + raise RuntimeSafetySchemaError('runtime control replay transition does not match') + events = self.conn.execute( + '''SELECT * FROM runtime_audit_events WHERE operation_id = ? + ORDER BY id LIMIT 2''', + (operation_id,), + ).fetchall() + if len(events) != 1: + raise RuntimeSafetySchemaError('runtime control replay audit evidence is incomplete') + event = events[0] + if ( + str(event['actor'] or '') != actor + or str(event['action'] or '') != action + or str(event['target_kind'] or '') != RUNTIME_CONTROL_TARGET_KIND + or str(event['target_ref'] or '') != target_ref + or event['result'] != 'succeeded' + or event['safe_category'] is not None + or event['before_identity_json'] != before_json + or event['after_identity_json'] != after_json + or event['before_bytes'] is not None + or event['after_bytes'] is not None + or not event['created_at'] + ): + raise RuntimeSafetySchemaError('runtime control replay audit evidence does not match') + previous_event_id = event['previous_event_id'] + previous_event_sha256 = event['previous_event_sha256'] + if previous_event_id is None: + if previous_event_sha256 is not None: + raise RuntimeSafetySchemaError('runtime control replay audit parent is invalid') + else: + parent = self.conn.execute( + 'SELECT event_sha256 FROM runtime_audit_events WHERE id = ?', + (int(previous_event_id),), + ).fetchone() + if ( + not parent + or not hmac.compare_digest( + str(parent['event_sha256'] or ''), str(previous_event_sha256 or ''), + ) + ): + raise RuntimeSafetySchemaError('runtime control replay audit parent is invalid') + payload = { + 'schema': 'runtime-audit-event-v1', + 'operation_id': operation_id, + 'actor': actor, + 'action': action, + 'target_kind': RUNTIME_CONTROL_TARGET_KIND, + 'target_ref': target_ref, + 'result': 'succeeded', + 'safe_category': None, + 'before_identity_json': before_json, + 'after_identity_json': after_json, + 'before_bytes': None, + 'after_bytes': None, + 'previous_event_id': int(previous_event_id) if previous_event_id is not None else None, + 'previous_event_sha256': previous_event_sha256, + 'created_at': str(event['created_at']), + } + expected_sha256 = _runtime_audit_event_sha256(payload) + if not hmac.compare_digest(expected_sha256, str(event['event_sha256'] or '')): + raise RuntimeSafetySchemaError('runtime control replay audit hash is invalid') + return { + 'operation_id': operation_id, + 'action': action, + 'replayed': True, + 'before': before, + 'after': after, + 'audit_event_id': int(event['id']), + 'audit_event_sha256': expected_sha256, + } + + def _commit_runtime_control_transition_locked( + self, *, state, action, target_ref, actor, operation_id, + ): + before = _runtime_control_state_identity(state) + after = _runtime_control_transition(before, action) + before_json = _runtime_control_identity_json(before) + after_json = _runtime_control_identity_json(after) + now = utc_now_iso() + self.conn.execute( + '''INSERT INTO runtime_operations( + operation_id, actor, action, target_kind, target_ref, + status, expected_revision, expected_identity_json, + agent_state, requested_at, started_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'running', ?, ?, 'not_required', ?, ?, ?)''', + ( + operation_id, actor, action, RUNTIME_CONTROL_TARGET_KIND, + target_ref, before['revision'], before_json, now, now, now, + ), + ) + updated = self.conn.execute( + '''UPDATE runtime_operations_control + SET revision = ?, discovery_paused = ?, dispatch_paused = ?, + drain_state = ?, actor = ?, operation_id = ?, updated_at = ? + WHERE id = 1 AND revision = ?''', + ( + after['revision'], int(after['discovery_paused']), + int(after['dispatch_paused']), after['drain_state'], actor, + operation_id, now, before['revision'], + ), + ) + if int(updated.rowcount or 0) != 1: + raise RuntimeControlRevisionConflictError( + before['revision'], self._locked_runtime_control_state(), + ) + audit = self._append_runtime_audit_event_locked( + operation_id=operation_id, actor=actor, action=action, + target_kind=RUNTIME_CONTROL_TARGET_KIND, target_ref=target_ref, + result='succeeded', before_identity=before, + after_identity=after, created_at=now, + ) + completed = self.conn.execute( + '''UPDATE runtime_operations + SET status = 'succeeded', resulting_revision = ?, + resulting_identity_json = ?, completed_at = ?, updated_at = ? + WHERE operation_id = ? AND status = 'running' ''', + (after['revision'], after_json, now, now, operation_id), + ) + if int(completed.rowcount or 0) != 1: + raise RuntimeSafetySchemaError('runtime control operation completion failed') + return { + 'operation_id': operation_id, + 'action': action, + 'replayed': False, + 'before': before, + 'after': after, + **audit, + } + + def _runtime_control_mutation( + self, *, action, target_ref, expected_revision, actor, operation_id, + ): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + operation_id = _runtime_operation_id(operation_id) + actor = _runtime_actor(actor) + expected_revision = _runtime_revision(expected_revision) + if expected_revision > RUNTIME_CONTROL_MAX_EXPECTED_REVISION: + raise ValueError('runtime control revision cannot be advanced') + if action not in RUNTIME_CONTROL_ACTIONS: + raise ValueError('runtime control action is invalid') + if RUNTIME_CONTROL_ACTION_TARGETS[action] != target_ref: + raise ValueError('runtime control target is invalid') + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + state = self._locked_runtime_control_state() + replay = self._runtime_control_replay_locked( + operation_id=operation_id, actor=actor, action=action, + target_ref=target_ref, expected_revision=expected_revision, + ) + if replay is not None: + self.conn.commit() + return replay + if state['revision'] != expected_revision: + raise RuntimeControlRevisionConflictError(expected_revision, state) + result = self._commit_runtime_control_transition_locked( + state=state, action=action, target_ref=target_ref, + actor=actor, operation_id=operation_id, + ) + self.conn.commit() + return result + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError('runtime control mutation retry budget exhausted') + + def set_runtime_discovery_paused( + self, paused, *, expected_revision, actor, operation_id, + ): + if type(paused) is not bool: + raise ValueError('runtime discovery pause state must be a boolean') + return self._runtime_control_mutation( + action='control.discovery.pause' if paused else 'control.discovery.resume', + target_ref='discovery', expected_revision=expected_revision, + actor=actor, operation_id=operation_id, + ) + + def set_runtime_dispatch_paused( + self, paused, *, expected_revision, actor, operation_id, + ): + if type(paused) is not bool: + raise ValueError('runtime dispatch pause state must be a boolean') + return self._runtime_control_mutation( + action='control.dispatch.pause' if paused else 'control.dispatch.resume', + target_ref='dispatch', expected_revision=expected_revision, + actor=actor, operation_id=operation_id, + ) + + def start_runtime_drain(self, *, expected_revision, actor, operation_id): + return self._runtime_control_mutation( + action='control.drain.start', target_ref='drain', + expected_revision=expected_revision, actor=actor, + operation_id=operation_id, + ) + + def cancel_runtime_drain(self, *, expected_revision, actor, operation_id): + return self._runtime_control_mutation( + action='control.drain.cancel', target_ref='drain', + expected_revision=expected_revision, actor=actor, + operation_id=operation_id, + ) + + def _runtime_drain_blockers_locked(self): + row = self.conn.execute( + '''SELECT + (SELECT COUNT(*) FROM result_reservations + WHERE assignment_kind = 'remote' + AND remote_resolved_at IS NULL) AS live_remote_assignments, + (SELECT COUNT(*) FROM result_bundles + WHERE state IN ('ready', 'ingesting')) AS precommit_result_bundles''' + ).fetchone() + if not row: + raise RuntimeSafetySchemaError('runtime drain progress is unavailable') + values = {} + for name in ('live_remote_assignments', 'precommit_result_bundles'): + value = row[name] + if isinstance(value, bool) or not isinstance(value, int) or value < 0: + raise RuntimeSafetySchemaError('runtime drain progress is invalid') + values[name] = value + values['blocker_count'] = sum(values.values()) + return values + + def runtime_drain_progress(self): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + try: + state = self._locked_runtime_control_state(shared=True) + progress = self._runtime_drain_blockers_locked() + self.conn.commit() + return {**state, **progress} + except Exception: + self.conn.rollback() + raise + + def reconcile_runtime_drain(self): + if not self.conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + actor = 'system:drain-reconciler' + operation_id = str(uuid.uuid4()) + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + for attempt in range(attempts): + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + state = self._locked_runtime_control_state() + if state['drain_state'] != 'draining': + self.conn.commit() + return None + blockers = self._runtime_drain_blockers_locked() + if blockers['blocker_count']: + self.conn.commit() + return None + result = self._commit_runtime_control_transition_locked( + state=state, action='control.drain.complete', + target_ref='drain', actor=actor, operation_id=operation_id, + ) + self.conn.commit() + return result + except Exception as exc: + self.conn.rollback() + if ( + self.conn.is_sqlite and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + raise + raise RuntimeSafetySchemaError('runtime drain reconciliation retry budget exhausted') + + def final_cutover_status(self): + if not self.conn or not self.conn.is_postgres: + return None + row = self.conn.execute( + 'SELECT marker, checked_at, evidence_sha256 FROM runtime_final_cutover WHERE id = 1' + ).fetchone() + self.conn.commit() + if not row: + return None + result = dict(row) + if ( + result['marker'] != FINAL_CUTOVER_MARKER + or not result['checked_at'] + or not re.fullmatch(r'[a-f0-9]{64}', str(result['evidence_sha256'] or '')) + ): + return None + return result + + def require_final_cutover(self): + if not self.final_cutover_status(): + raise RuntimeSafetySchemaError( + 'final PostgreSQL v2 cutover marker is absent or invalid; ' + 'run migrate_runtime_safety.py offline with --apply --sources-stopped' + ) + return True + + def record_final_cutover(self, evidence): + if not self.conn or not self.conn.is_postgres: + raise RuntimeSafetySchemaError('final cutover authority can only be recorded in PostgreSQL') + encoded = json.dumps( + evidence, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8') + if len(encoded) > 1024 * 1024: + raise ValueError('final cutover evidence exceeds its byte bound') + digest = hashlib.sha256(encoded).hexdigest() + now = utc_now_iso() + self.conn.execute( + '''INSERT INTO runtime_final_cutover(id, marker, checked_at, evidence_sha256) + VALUES (1, ?, ?, ?) + ON CONFLICT(id) DO UPDATE SET marker = excluded.marker, + checked_at = excluded.checked_at, + evidence_sha256 = excluded.evidence_sha256''', + (FINAL_CUTOVER_MARKER, now, digest), + ) + self.conn.commit() + return {'marker': FINAL_CUTOVER_MARKER, 'checked_at': now, 'evidence_sha256': digest} + + def revoke_final_cutover(self): + if not self.conn or not self.conn.is_postgres: + raise RuntimeSafetySchemaError('final cutover authority can only be revoked in PostgreSQL') + self.conn.execute('DELETE FROM runtime_final_cutover WHERE id = 1') + self.conn.commit() + return True + + def pipeline_schema_available(self, commit=True): + if not self.conn: + self.last_error = 'database connection is unavailable' + return False + try: + problems = [] + for table, columns in PIPELINE_REQUIRED_COLUMNS.items(): + if not self.conn.table_exists(table): + problems.append(f'table {table}') + continue + missing = columns - set(self.conn.table_columns(table)) + if missing: + problems.append(f'{table} columns {", ".join(sorted(missing))}') + capacity = self.conn.execute( + 'SELECT id FROM pipeline_capacity WHERE id = 1' + ).fetchone() if self.conn.table_exists('pipeline_capacity') else None + if not capacity: + problems.append('pipeline_capacity singleton row') + if self.conn.table_exists('runtime_schema_migrations'): + rows = self.conn.execute( + 'SELECT version, code_sha256 FROM runtime_schema_migrations' + ).fetchall() + applied = {str(row['version']) for row in rows} + missing_versions = set(PIPELINE_MIGRATION_VERSIONS) - applied + if missing_versions: + problems.append('migration versions ' + ', '.join(sorted(missing_versions))) + marker = next( + ( + row for row in rows + if str(row['version']) == PIPELINE_MIGRATION_VERSIONS[-1] + ), + None, + ) + expected_code = hashlib.sha256(PIPELINE_SCHEMA_SQL.encode('utf-8')).hexdigest() + if marker and str(marker['code_sha256'] or '') != expected_code: + problems.append('pipeline migration marker code identity') + if self.conn.table_exists('pipeline_quarantine'): + constraint = _pipeline_quarantine_review_status_constraint(self.conn) + if ( + constraint.get('statuses') != PIPELINE_QUARANTINE_REVIEW_STATUSES + or not constraint.get('valid') + ): + problems.append('pipeline quarantine review-status constraint') + problems.extend(_docker_depth_check_constraint_problems(self.conn)) + required_indexes = { + 'target_queue': ( + 'idx_target_queue_current_reservation', 'idx_target_queue_claimable_v2', + 'idx_target_queue_cold', + ), + 'target_queue_policy_events': ( + 'idx_target_queue_policy_events_queue', + 'idx_target_queue_policy_events_manifest', + ), + 'target_scans': ( + 'idx_target_scans_event_hash', 'idx_target_scans_queue_id', + 'idx_target_scans_result_reservation', + ), + 'findings': ('idx_findings_target_scan_id_id',), + 'result_reservations': ( + 'uq_result_reservation_queue_lease', 'idx_result_reservations_recovery', + 'idx_result_reservations_remote_active_user', + 'idx_result_reservations_remote_expiry', + 'idx_result_reservations_remote_device_history', + 'uq_result_reservations_remote_receipt', + ), + 'remote_worker_users': ('uq_remote_worker_users_key',), + 'remote_worker_devices': ( + 'uq_remote_worker_devices_key', 'uq_remote_worker_devices_token', + ), + 'admission_intents': ('idx_admission_intents_remote_device',), + 'result_bundles': ('idx_result_bundles_ready', 'idx_result_bundles_ingest_lease'), + 'projection_jobs': ('idx_projection_jobs_claim', 'idx_projection_jobs_lease'), + 'keycheck_candidates': ( + 'idx_keycheck_candidates_pending', 'idx_keycheck_candidates_deferred', + 'idx_keycheck_candidates_lease', + ), + 'keycheck_credentials': ('uq_keycheck_credentials_provider_key',), + 'pipeline_artifacts': ('idx_pipeline_artifacts_owner',), + 'docker_content_blobs': ( + 'idx_docker_content_blobs_reclaim', + 'idx_docker_content_blobs_reservation', + ), + 'docker_image_blob_coverage': ( + 'idx_docker_image_blob_coverage_manifest', + 'idx_docker_image_blob_coverage_blob', + 'idx_docker_image_blob_coverage_reservation', + 'idx_docker_image_blob_coverage_selection', + ), + 'docker_adaptive_shadow_reports': ( + 'idx_docker_adaptive_shadow_reports_gate', + ), + 'discovery_retry_queue': ( + 'uq_discovery_retry_queue_work_key', + 'idx_discovery_retry_queue_due', + 'idx_discovery_retry_queue_lease', + 'idx_discovery_retry_queue_policy', + ), + 'runtime_operations': ( + 'idx_runtime_operations_status_updated', + 'idx_runtime_operations_agent_state_updated', + ), + 'runtime_audit_events': ( + 'idx_runtime_audit_events_created', + 'idx_runtime_audit_events_operation', + ), + 'worker_progress_events': ( + 'uq_worker_progress_reservation_sequence', + 'idx_worker_progress_reservation_received', + 'idx_worker_progress_device_received', + 'idx_worker_progress_phase_received', + ), + 'worker_diagnostics': ( + 'uq_worker_diagnostics_uid', + 'idx_worker_diagnostics_reservation_received', + 'idx_worker_diagnostics_scan_received', + 'idx_worker_diagnostics_phase_received', + 'idx_worker_diagnostics_category_code', + 'idx_worker_diagnostics_code_received', + 'idx_worker_diagnostics_kind_received', + 'idx_worker_diagnostics_retryable_received', + ), + } + for table, names in required_indexes.items(): + indexes = self.conn.table_indexes(table) if self.conn.table_exists(table) else {} + for name in names: + if not _index_usable(indexes.get(name)): + problems.append(f'index {name}') + experiment_indexes = {} + for table, name, columns, unique, predicate in DOCKER_DEPTH_EXPERIMENT_INDEX_SPECS: + indexes = experiment_indexes.setdefault( + table, + self.conn.table_indexes(table) if self.conn.table_exists(table) else {}, + ) + index = indexes.get(name) + if ( + not index + or index['columns'] != list(columns) + or bool(index['unique']) != bool(unique) + or _normalized_predicate(index['predicate']) != _normalized_predicate(predicate) + or not _index_usable(index) + ): + problems.append(f'index {name}') + credential_index = ( + self.conn.table_indexes('keycheck_credentials').get('uq_keycheck_credentials_provider_key') + if self.conn.table_exists('keycheck_credentials') else None + ) + if not credential_index or not credential_index['unique'] or credential_index['columns'] != [ + 'service', 'provider_key_hash' + ]: + problems.append('unique provider-canonical keycheck credential index') + if self.conn.is_postgres and commit: + self.conn.commit() + if problems: + self.last_error = 'pipeline schema is incomplete: ' + '; '.join(problems) + return False + self.last_error = '' + return True + except Exception as exc: + self.last_error = f'unable to validate pipeline schema: {exc}' + try: + self.conn.rollback() + except Exception: + pass + return False + + def provider_routing_finding_rows(self, secret_hash, limit, max_json_chars): + if not self.conn: + raise RuntimeError('database connection is unavailable') + secret_hash = str(secret_hash or '').strip().lower() + if not re.fullmatch(r'[a-f0-9]{64}', secret_hash): + raise ValueError('provider routing lookup requires one SHA-256 digest') + limit = min(1025, max(1, int(limit or 0))) + max_json_chars = min(1024 * 1024, max(1024, int(max_json_chars or 0))) + return self.conn.execute( + '''SELECT + detector_name, + SUBSTR(COALESCE(raw_finding_json, ''), 1, ?) AS raw_finding_json, + CASE WHEN LENGTH(COALESCE(raw_finding_json, '')) > ? THEN 1 ELSE 0 END AS truncated + FROM findings + WHERE secret_hash = ? + ORDER BY id DESC + LIMIT ?''', + (max_json_chars, max_json_chars, secret_hash, limit), + ).fetchall() + + def ensure_scan_publication_outbox(self): + if not self.conn or not self.conn.table_exists('scan_publication_outbox'): + self.last_error = 'scan_publication_outbox is unavailable' + return False + required = { + 'id', 'target_scan_id', 'payload_json', 'status', 'attempts', 'last_error', + 'lease_owner', 'lease_expires_at', 'available_after', 'created_at', 'delivered_at', 'updated_at', + } + missing = required - set(self.conn.table_columns('scan_publication_outbox')) + if missing: + self.last_error = f'scan_publication_outbox is missing columns: {", ".join(sorted(missing))}' + return False + self.last_error = '' + return True + + def scan_publication_backlog_health(self, max_items, max_bytes, max_age_sec, additional_items=0): + if not self.conn or not self.conn.table_exists('scan_publication_outbox'): + return {'healthy': False, 'accepting': False, 'reason': 'publication outbox is unavailable'} + + def op(): + now = datetime.now(timezone.utc) + payload_size = ( + "OCTET_LENGTH(COALESCE(s.raw_result_json, ''))" + if self.conn.is_postgres + else "LENGTH(CAST(COALESCE(s.raw_result_json, '') AS BLOB))" + ) + row = self.conn.execute( + f'''SELECT COUNT(*) AS item_count, + COALESCE(SUM({payload_size}), 0) AS payload_bytes, + MIN(o.created_at) AS oldest_at + FROM scan_publication_outbox o + LEFT JOIN target_scans s ON s.id = o.target_scan_id + WHERE o.status IN ('pending', 'delivering', 'dead')''' + ).fetchone() + item_count = int(row['item_count'] or 0) + payload_bytes = int(row['payload_bytes'] or 0) + oldest = parse_time(row['oldest_at']) + oldest_age_sec = max(0, int((now - oldest).total_seconds())) if oldest else 0 + item_limit = max(1, int(max_items)) + byte_limit = max(1, int(max_bytes)) + age_limit = max(1, int(max_age_sec)) + reasons = [] + if item_count > item_limit: + reasons.append(f'items={item_count}>{item_limit}') + if payload_bytes > byte_limit: + reasons.append(f'bytes={payload_bytes}>{byte_limit}') + if oldest_age_sec > age_limit: + reasons.append(f'oldest_age_sec={oldest_age_sec}>{age_limit}') + healthy = not reasons + accepting = healthy and item_count + max(0, int(additional_items or 0)) <= item_limit + if not accepting and not reasons: + reasons.append(f'items={item_count}+{max(0, int(additional_items or 0))}>{item_limit}') + return { + 'healthy': healthy, + 'accepting': accepting, + 'reason': '; '.join(reasons), + 'items': item_count, + 'bytes': payload_bytes, + 'oldest_age_sec': oldest_age_sec, + 'max_items': item_limit, + 'max_bytes': byte_limit, + 'max_age_sec': age_limit, + } + return self._safe('scan_publication_backlog_health', op, { + 'healthy': False, 'accepting': False, 'reason': self.last_error or 'publication backlog query failed', + }) + + def claim_scan_publications(self, lease_owner, limit=100, lease_seconds=300, max_attempts=None): + if not self.conn or not self.conn.table_exists('scan_publication_outbox') or not lease_owner: + return [] + + def op(): + now = utc_now_iso() + compact_delivered_scan_publications(self.conn, now) + lease_until = datetime.fromtimestamp(time.time() + max(60, int(lease_seconds)), timezone.utc).isoformat(timespec='seconds') + self.conn.execute( + '''UPDATE scan_publication_outbox SET status = 'pending', payload_json = '', + lease_owner = NULL, lease_expires_at = NULL, available_after = COALESCE(available_after, ?), updated_at = ? + WHERE status = 'dead' ''', + (now, now), + ) + self.conn.execute("UPDATE scan_publication_outbox SET payload_json = '' WHERE payload_json != ''") + self.conn.execute( + '''UPDATE scan_publication_outbox SET status = 'pending', lease_owner = NULL, lease_expires_at = NULL, updated_at = ? + WHERE status = 'delivering' AND lease_expires_at IS NOT NULL AND lease_expires_at <= ?''', + (now, now), + ) + limit_value = max(1, int(limit or 100)) + if self.conn.is_postgres: + rows = self.conn.execute( + '''WITH picked AS ( + SELECT id FROM scan_publication_outbox + WHERE status = 'pending' AND (available_after IS NULL OR available_after <= ?) + ORDER BY id LIMIT ? FOR UPDATE SKIP LOCKED + ) + UPDATE scan_publication_outbox o SET status = 'delivering', lease_owner = ?, + lease_expires_at = ?, attempts = CASE WHEN COALESCE(attempts, 0) < 1000000 + THEN COALESCE(attempts, 0) + 1 ELSE COALESCE(attempts, 0) END, updated_at = ? + FROM picked WHERE o.id = picked.id''', + (now, limit_value, lease_owner, lease_until, now), + ) + else: + rows = self.conn.execute( + '''SELECT id FROM scan_publication_outbox + WHERE status = 'pending' AND (available_after IS NULL OR available_after <= ?) + ORDER BY id LIMIT ?''', + (now, limit_value), + ).fetchall() + for row in rows: + self.conn.execute( + '''UPDATE scan_publication_outbox SET status = 'delivering', lease_owner = ?, + lease_expires_at = ?, attempts = CASE WHEN COALESCE(attempts, 0) < 1000000 + THEN COALESCE(attempts, 0) + 1 ELSE COALESCE(attempts, 0) END, + updated_at = ? WHERE id = ?''', + (lease_owner, lease_until, now, row['id']), + ) + rows = self.conn.execute( + '''SELECT o.id, o.target_scan_id, COALESCE(s.raw_result_json, '') AS payload_json, o.attempts + FROM scan_publication_outbox o + JOIN target_scans s ON s.id = o.target_scan_id + WHERE o.status = 'delivering' AND o.lease_owner = ? + ORDER BY o.id LIMIT ?''', + (lease_owner, limit_value), + ).fetchall() + self.conn.commit() + return rows + return self._safe('claim_scan_publications', op, []) + + def renew_scan_publication(self, outbox_id, lease_owner, lease_seconds=300): + if not self.conn or outbox_id is None or not lease_owner: + return False + try: + now = utc_now_iso() + lease_until = datetime.fromtimestamp( + time.time() + max(60, int(lease_seconds)), timezone.utc, + ).isoformat(timespec='seconds') + cur = self.conn.execute( + '''UPDATE scan_publication_outbox SET lease_expires_at = ?, updated_at = ? + WHERE id = ? AND status = 'delivering' AND lease_owner = ?''', + (lease_until, now, outbox_id, lease_owner), + ) + renewed = int(getattr(cur, 'rowcount', 0) or 0) == 1 + self.conn.commit() + self.last_error = '' + return renewed + except Exception as exc: + self.last_error = str(exc) + try: + self.conn.rollback() + except Exception: + pass + logger.error(f'Observability DB write failed during renew_scan_publication: {exc}') + return False + + def finish_scan_publication(self, outbox_id, lease_owner, delivered, error=''): + if not self.conn: + return False + + def op(): + now = utc_now_iso() + if delivered: + cur = self.conn.execute( + '''DELETE FROM scan_publication_outbox + WHERE id = ? AND status = 'delivering' AND lease_owner = ?''', + (outbox_id, lease_owner), + ) + else: + claimed = self.conn.execute( + '''SELECT attempts FROM scan_publication_outbox + WHERE id = ? AND status = 'delivering' AND lease_owner = ?''', + (outbox_id, lease_owner), + ).fetchone() + if not claimed: + self.conn.rollback() + return False + attempts = max(1, int(claimed['attempts'] or 1)) + base = max(1, env_int('SCAN_OUTBOX_RETRY_BASE_SEC', 30)) + maximum = max(base, env_int('SCAN_OUTBOX_RETRY_MAX_SEC', 3600)) + delay = min(maximum, base * (2 ** min(attempts - 1, 20))) + available_after = datetime.fromtimestamp( + time.time() + delay, timezone.utc + ).isoformat(timespec='seconds') + cur = self.conn.execute( + '''UPDATE scan_publication_outbox SET status = 'pending', last_error = ?, payload_json = '', + lease_owner = NULL, lease_expires_at = NULL, available_after = ?, updated_at = ? + WHERE id = ? AND status = 'delivering' AND lease_owner = ?''', + (first_line(error, 500), available_after, now, outbox_id, lease_owner), + ) + self.conn.commit() + return int(getattr(cur, 'rowcount', 0) or 0) == 1 + return self._safe('finish_scan_publication', op, False) + + def sync_target_queue_from_files(self, source, platform, todo_targets=None, checked_targets=None, query=None): + if not self.conn: + return False + + def op(): + now = utc_now_iso() + for target in checked_targets or []: + normalized = normalize_target(target, platform) + self.conn.execute( + f'''INSERT INTO target_queue ( + source, platform, query, target, normalized_target, status, created_at, updated_at, completed_at + ) VALUES (?, ?, ?, ?, ?, 'done', ?, ?, ?) + ON CONFLICT(source, normalized_target) DO UPDATE SET + status = CASE WHEN target_queue.status IN ('failed', 'deferred', 'in_progress', 'cold') THEN target_queue.status ELSE 'done' END, + target = excluded.target, + platform = excluded.platform, + completed_at = CASE + WHEN target_queue.status IN ('deferred', 'in_progress', 'cold') THEN target_queue.completed_at + ELSE COALESCE(target_queue.completed_at, excluded.completed_at) + END, + lease_owner = CASE WHEN target_queue.status = 'in_progress' THEN target_queue.lease_owner ELSE NULL END, + lease_token = CASE WHEN target_queue.status = 'in_progress' THEN target_queue.lease_token ELSE NULL END, + claim_batch = CASE WHEN target_queue.status = 'in_progress' THEN target_queue.claim_batch ELSE NULL END, + leased_at = CASE WHEN target_queue.status = 'in_progress' THEN target_queue.leased_at ELSE NULL END, + lease_expires_at = CASE WHEN target_queue.status = 'in_progress' THEN target_queue.lease_expires_at ELSE NULL END, + available_after = CASE WHEN target_queue.status IN ('failed', 'deferred', 'in_progress', 'cold') THEN target_queue.available_after ELSE NULL END, + updated_at = excluded.updated_at + WHERE target_queue.status NOT IN ('done', 'failed', 'deferred', 'in_progress', 'cold') + OR target_queue.target <> excluded.target + OR target_queue.platform <> excluded.platform''', + (source, platform, query, target, normalized, now, now, now), + ) + for target in todo_targets or []: + normalized = normalize_target(target, platform) + self.conn.execute( + f'''INSERT INTO target_queue ( + source, platform, query, target, normalized_target, status, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'pending', ?, ?) + ON CONFLICT(source, normalized_target) DO UPDATE SET + target = excluded.target, + platform = excluded.platform, + query = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN excluded.query ELSE COALESCE(target_queue.query, excluded.query) END, + status = CASE + WHEN {ADMIN_DISCARDED_QUEUE_SQL} THEN 'pending' + WHEN target_queue.status IN ('done', 'failed', 'deferred', 'in_progress', 'cold') THEN target_queue.status + WHEN target_queue.status = 'pending' AND target_queue.available_after IS NOT NULL THEN 'deferred' + ELSE 'pending' + END, + available_after = CASE + WHEN {ADMIN_DISCARDED_QUEUE_SQL} THEN NULL + WHEN target_queue.status IN ('pending', 'deferred', 'cold') THEN target_queue.available_after + ELSE NULL + END, + attempts = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN 0 ELSE target_queue.attempts END, + last_error = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN NULL ELSE target_queue.last_error END, + completed_at = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN NULL ELSE target_queue.completed_at END, + resolver_state = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN NULL ELSE target_queue.resolver_state END, + resolver_due_at = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN NULL ELSE target_queue.resolver_due_at END, + resolver_attempts = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN 0 ELSE target_queue.resolver_attempts END, + resolver_token = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN NULL ELSE target_queue.resolver_token END, + updated_at = excluded.updated_at + WHERE ({ADMIN_DISCARDED_QUEUE_SQL}) + OR target_queue.target <> excluded.target + OR target_queue.platform <> excluded.platform + OR (target_queue.query IS NULL AND excluded.query IS NOT NULL) + OR (target_queue.status = 'pending' AND target_queue.available_after IS NOT NULL)''', + (source, platform, query, target, normalized, now, now), + ) + self.conn.commit() + return True + return self._safe('sync_target_queue_from_files', op, False) + + def _docker_discovery_pass_locked( + self, source, query, observation, now, pass_id=None, + ): + lock_suffix = ' FOR UPDATE' if self.conn.is_postgres else '' + params = ( + source, + observation['pass_kind'], + observation['collection_generation'], + observation['policy_sha256'], + observation['ordered_query_hash'], + observation['query_count'], + ) + if pass_id is not None: + rows = self.conn.execute( + f'''SELECT * FROM docker_discovery_passes + WHERE id = ? AND source = ? AND pass_kind = ? + AND collection_generation = ? AND policy_sha256 = ? + AND ordered_queries_sha256 = ? + AND expected_query_count = ? + AND state IN ('collecting','complete'){lock_suffix}''', + (pass_id, *params), + ).fetchall() + if len(rows) != 1: + raise RuntimeError('DockerHub discovery retry pass identity changed') + candidates = rows + else: + cycle_predicate = ( + 'page.source_cycle_id IS NULL' + if observation['cycle_id'] is None + else 'page.source_cycle_id = ?' + ) + cycle_params = ( + () if observation['cycle_id'] is None + else (observation['cycle_id'],) + ) + candidates = self.conn.execute( + f'''SELECT * FROM docker_discovery_passes + WHERE source = ? AND pass_kind = ? AND collection_generation = ? + AND policy_sha256 = ? + AND ordered_queries_sha256 = ? AND expected_query_count = ? + AND ( + state = 'collecting' OR EXISTS ( + SELECT 1 FROM docker_discovery_pages page + WHERE page.pass_id = docker_discovery_passes.id + AND page.query_ordinal = ? AND page.page_number = ? + AND {cycle_predicate} + ) + ) + ORDER BY id{lock_suffix}''', + ( + *params, observation['query_ordinal'], + observation['page_number'], *cycle_params, + ), + ).fetchall() + + empty_candidate = None + selected = None + for candidate in candidates: + pages = self.conn.execute( + '''SELECT query, page_number, source_cycle_id + FROM docker_discovery_pages + WHERE pass_id = ? AND query_ordinal = ? ORDER BY page_number''', + (candidate['id'], observation['query_ordinal']), + ).fetchall() + if any(str(page['query']) != query for page in pages): + raise RuntimeError('DockerHub discovery pass query identity conflicts') + exact_page = [ + page for page in pages + if int(page['page_number']) == observation['page_number'] + ] + if pass_id is not None: + if str(candidate['state']) == 'complete' and not exact_page: + raise RuntimeError('DockerHub discovery complete retry pass cannot admit another page') + return candidate + same_cycle = any( + ( + page['source_cycle_id'] is None + and observation['cycle_id'] is None + ) or ( + page['source_cycle_id'] is not None + and observation['cycle_id'] is not None + and int(page['source_cycle_id']) == observation['cycle_id'] + ) + for page in pages + ) + if exact_page and (same_cycle or observation['cycle_id'] is None): + selected = candidate + break + if same_cycle: + if str(candidate['state']) == 'complete': + raise RuntimeError('DockerHub discovery complete pass cannot admit another page') + selected = candidate + break + if not pages and str(candidate['state']) == 'collecting' and empty_candidate is None: + empty_candidate = candidate + selected = selected or empty_candidate + if selected is not None: + return selected + if pass_id is not None: + raise RuntimeError('DockerHub discovery retry pass is unavailable') + + token = secrets.token_urlsafe(32) + row = self.conn.execute( + '''INSERT INTO docker_discovery_passes( + experiment_id, pass_token, source, pass_kind, + collection_generation, policy_sha256, + ordered_queries_sha256, expected_query_count, + completed_query_count, state, started_at, completed_at, + created_at, updated_at + ) VALUES (NULL, ?, ?, ?, ?, ?, ?, ?, 0, 'collecting', ?, NULL, ?, ?) + RETURNING *''', + ( + token, source, observation['pass_kind'], + observation['collection_generation'], observation['policy_sha256'], + observation['ordered_query_hash'], observation['query_count'], + now, now, now, + ), + ).fetchone() + if not row: + raise RuntimeError('DockerHub discovery pass could not be opened') + return row + + def persist_dockerhub_discovery_page( + self, source, query, repositories, retry_id=None, lease_owner=None, + lease_token=None, next_page=None, complete=False, observation=None, + experiment_authority=None, final_cutover=False, + ): + """Durably admit one normalized page without changing existing target state.""" + if not self.conn: + raise RuntimeError('database connection is unavailable') + source = _validated_discovery_retry_source(source) + query = _validated_discovery_retry_query(query) + attempted_count, normalized, ordinals, observed = ( + _normalized_dockerhub_repositories(repositories) + ) + observation = _validated_docker_discovery_observation( + observation, source, query, + ) + if observation is not None: + if observation['total_count'] is None: + raise ValueError('DockerHub page admission lacks total-count evidence') + absolute_start = ( + observation['page_number'] - 1 + ) * observation['per_page'] + absolute_bound = absolute_start + attempted_count + terminal_by_count = observation['total_count'] <= ( + observation['page_number'] * observation['per_page'] + ) + if ( + attempted_count > observation['per_page'] + or observation['total_count'] < absolute_bound + or (attempted_count == 0 and observation['total_count'] > absolute_start) + or ( + observation['query_complete'] + and not terminal_by_count + and observation['page_number'] != observation['page_limit'] + ) + or ( + not observation['query_complete'] + and ( + terminal_by_count + or observation['page_number'] == observation['page_limit'] + ) + ) + ): + raise ValueError('DockerHub page cardinality or continuation evidence conflicts') + if not isinstance(complete, bool): + raise ValueError('discovery retry completion flag is invalid') + retry_requested = any( + value is not None for value in (retry_id, lease_owner, lease_token, next_page) + ) or complete + fence = None + if retry_requested: + fence = _validated_discovery_retry_fence(retry_id, lease_owner, lease_token) + if complete and next_page is not None: + raise ValueError('completed discovery retry work cannot retain a next page') + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._require_discovery_admission_locked() + now_dt = datetime.now(timezone.utc) + now = now_dt.isoformat(timespec='seconds') + page_experiment = None + page_authority = None + if self.conn.is_postgres and source == 'dockerhub': + page_experiment, page_authority, authority_reason = ( + self._locked_docker_depth_page_authority( + source, query, observation, experiment_authority, + final_cutover=final_cutover, now=now, + ) + ) + if authority_reason: + self.conn.commit() + raise ScanEventConflictError( + f'Docker depth page authority drifted: {authority_reason}' + ) + retry_row = None + pass_row = None + if fence: + lock_suffix = ' FOR UPDATE' if self.conn.is_postgres else '' + retry_row = self.conn.execute( + f'''SELECT work_key, policy_sha256, pass_kind, work_kind, + page_start, page_end, next_page, source_cycle_id + FROM discovery_retry_queue + WHERE id = ? AND source = ? AND query = ? AND status = 'leased' + AND lease_owner = ? AND lease_token = ? + AND lease_expires_at > ?{lock_suffix}''', + (fence[0], source, query, fence[1], fence[2], now), + ).fetchone() + if not retry_row: + raise DiscoveryRetryLeaseError('discovery retry page admission lost its lease fence') + if next_page is not None: + if isinstance(next_page, bool): + raise ValueError('discovery retry next page is invalid') + try: + next_page = int(next_page) + except (TypeError, ValueError, OverflowError): + raise ValueError('discovery retry next page is invalid') from None + if not ( + int(retry_row['next_page']) <= next_page <= int(retry_row['page_end']) + and 1 <= next_page <= DISCOVERY_RETRY_MAX_PAGE + ): + raise ValueError('discovery retry next page is outside its durable range') + + retry_identity = ( + source, query, retry_row['policy_sha256'], retry_row['pass_kind'], + retry_row['work_kind'], retry_row['page_start'], retry_row['page_end'], + ) + legacy_work_key = discovery_retry_work_key(*retry_identity) + if str(retry_row['work_key']) == legacy_work_key: + observation = None + else: + try: + pass_id = int(str(retry_row['work_key'])[:16], 16) + except (TypeError, ValueError, OverflowError): + raise RuntimeError('DockerHub discovery retry pass identity is invalid') from None + if str(retry_row['work_key']) != discovery_retry_work_key( + *retry_identity, pass_id=pass_id, + ): + raise RuntimeError('DockerHub discovery retry pass identity conflicts') + if observation is None: + raise RuntimeError('DockerHub discovery retry provenance is unavailable') + retry_cycle_id = ( + int(retry_row['source_cycle_id']) + if retry_row['source_cycle_id'] is not None else None + ) + if ( + observation['policy_sha256'] != str(retry_row['policy_sha256']) + or observation['pass_kind'] != str(retry_row['pass_kind']) + or observation['page_number'] != int(retry_row['next_page']) + or observation['cycle_id'] != retry_cycle_id + ): + raise RuntimeError('DockerHub discovery retry provenance changed') + pass_row = self._docker_discovery_pass_locked( + source, query, observation, now, pass_id=pass_id, + ) + elif observation is not None: + pass_row = self._docker_discovery_pass_locked( + source, query, observation, now, + ) + + preexisting = set() + lock_suffix = ' FOR UPDATE' if self.conn.is_postgres else '' + for start in range(0, len(normalized), KNOWN_TARGET_LOOKUP_BATCH_SIZE): + values = normalized[start:start + KNOWN_TARGET_LOOKUP_BATCH_SIZE] + rows = self.conn.execute( + '''SELECT normalized_target FROM target_queue + WHERE source = ? AND platform = 'docker' + AND normalized_target IN ({}){}'''.format( + ','.join('?' for _ in values), lock_suffix, + ), + (source, *values), + ).fetchall() + preexisting.update( + str(row['normalized_target']) for row in rows if row['normalized_target'] + ) + + resolver_due = (now_dt + timedelta( + seconds=max(60, env_int('DOCKER_RESOLVER_RETRY_SEC', 3600)), + )).isoformat(timespec='seconds') + inserted_count = 0 + queue_ids = {} + new_queue_ids = set() + for repository in normalized: + inserted = self.conn.execute( + '''INSERT INTO target_queue ( + source, platform, query, target, normalized_target, status, + available_after, last_error, resolver_state, resolver_due_at, + created_at, updated_at + ) VALUES (?, 'docker', ?, ?, ?, 'deferred', ?, + 'Docker tag resolution unresolved', 'pending', ?, ?, ?) + ON CONFLICT(source, normalized_target) DO NOTHING + RETURNING id''', + ( + source, query, repository, repository, resolver_due, + resolver_due, now, now, + ), + ).fetchone() + inserted_count += int(bool(inserted)) + if inserted: + queue_ids[repository] = int(inserted['id']) + new_queue_ids.add(int(inserted['id'])) + else: + queue_row = self.conn.execute( + f'''SELECT id, platform, status, last_error FROM target_queue + WHERE source = ? AND normalized_target = ?{lock_suffix}''', + (source, repository), + ).fetchone() + if not queue_row or str(queue_row['platform']) != 'docker': + raise RuntimeError('DockerHub repository queue identity conflicts') + queue_id = int(queue_row['id']) + queue_ids[repository] = queue_id + if ( + str(queue_row['status']) == 'quarantined' + and str(queue_row['last_error'] or '') + == ADMIN_DISCARDED_QUEUE_REASON + ): + cursor = self.conn.execute( + '''UPDATE target_queue SET query = ?, target = ?, + status = 'deferred', attempts = 0, + available_after = ?, + last_error = 'Docker tag resolution unresolved', + completed_at = NULL, lease_owner = NULL, + lease_token = NULL, claim_batch = NULL, + leased_at = NULL, lease_expires_at = NULL, + resolver_state = 'pending', resolver_due_at = ?, + resolver_attempts = 0, resolver_token = NULL, + claim_event_id = NULL, updated_at = ? + WHERE id = ? AND status = 'quarantined' + AND last_error = ?''', + ( + query, repository, resolver_due, resolver_due, now, + queue_id, ADMIN_DISCARDED_QUEUE_REASON, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError( + 'discarded DockerHub target reactivation authority changed' + ) + inserted_count += 1 + new_queue_ids.add(queue_id) + preexisting.discard(repository) + + page_id = None + page_inserted_count = 0 + observation_inserted_count = 0 + completed_query_count = 0 + pass_complete = False + query_complete = False + if pass_row is not None: + page_payload = json.dumps( + { + 'per_page': observation['per_page'], + 'page_number': observation['page_number'], + 'page_limit': observation['page_limit'], + 'query': query, + 'query_ordinal': observation['query_ordinal'], + 'repositories': observed, + 'total_count': observation['total_count'], + 'collection_generation': observation['collection_generation'], + 'schema': 'docker-discovery-page-v2', + 'source': source, + }, + ensure_ascii=True, + sort_keys=True, + separators=(',', ':'), + ).encode('utf-8') + page_sha256 = hashlib.sha256(page_payload).hexdigest() + admission_kind = 'retry' if fence else 'main' + inserted_page = self.conn.execute( + '''INSERT INTO docker_discovery_pages( + pass_id, source_cycle_id, retry_work_id, query, + query_ordinal, page_number, result_count, total_count, + admitted_count, + query_complete, admission_kind, page_sha256, + observed_at, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(pass_id, query_ordinal, page_number) DO NOTHING + RETURNING *''', + ( + pass_row['id'], observation['cycle_id'], + fence[0] if fence else None, query, + observation['query_ordinal'], observation['page_number'], + attempted_count, observation['total_count'], len(normalized), + int(observation['query_complete']), + admission_kind, page_sha256, now, now, + ), + ).fetchone() + page_inserted_count = int(bool(inserted_page)) + page_row = inserted_page or self.conn.execute( + f'''SELECT * FROM docker_discovery_pages + WHERE pass_id = ? AND query_ordinal = ? AND page_number = ?{lock_suffix}''', + ( + pass_row['id'], observation['query_ordinal'], + observation['page_number'], + ), + ).fetchone() + if not page_row or ( + str(page_row['query']) != query + or int(page_row['result_count']) != attempted_count + or int(page_row['total_count']) != observation['total_count'] + or int(page_row['admitted_count']) != len(normalized) + or str(page_row['admission_kind']) != admission_kind + or str(page_row['page_sha256']) != page_sha256 + or ( + (page_row['retry_work_id'] is None) != (fence is None) + ) + or ( + fence is not None + and int(page_row['retry_work_id']) != fence[0] + ) + or ( + (page_row['source_cycle_id'] is None) != + (observation['cycle_id'] is None) + ) + or ( + page_row['source_cycle_id'] is not None + and int(page_row['source_cycle_id']) != observation['cycle_id'] + ) + ): + raise RuntimeError('DockerHub discovery page evidence conflicts') + page_id = int(page_row['id']) + page_observed_at = str(page_row['observed_at']) + if observation['query_complete'] and not int(page_row['query_complete']): + self.conn.execute( + '''UPDATE docker_discovery_pages SET query_complete = 1 + WHERE id = ? AND query_complete = 0''', + (page_id,), + ) + + for repository in normalized: + queue_id = queue_ids[repository] + search_rank = ( + (observation['page_number'] - 1) * observation['per_page'] + + ordinals[repository] + ) + existing_observation = self.conn.execute( + '''SELECT search_rank FROM docker_repository_query_observations + WHERE page_id = ? AND repository_queue_id = ?''', + (page_id, queue_id), + ).fetchone() + if existing_observation: + if int(existing_observation['search_rank']) != search_rank: + raise RuntimeError('DockerHub repository observation rank conflicts') + continue + self.conn.execute( + '''INSERT INTO docker_repository_query_provenance( + source, query, repository_queue_id, provenance_kind, + first_observed_at, last_observed_at, first_search_rank, + best_search_rank, last_search_rank, first_cycle_id, + last_cycle_id, first_page_id, last_page_id, + first_policy_sha256, last_policy_sha256, + observation_count, fresh_observation_count, + fresh_complete_observation_count, fresh_coverage_eligible, + created_at, updated_at + ) VALUES (?, ?, ?, 'fresh_page', ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, + 1, 1, 0, 0, ?, ?) + ON CONFLICT(source, query, repository_queue_id) DO UPDATE SET + provenance_kind = 'fresh_page', + last_observed_at = excluded.last_observed_at, + first_search_rank = COALESCE( + docker_repository_query_provenance.first_search_rank, + excluded.first_search_rank + ), + best_search_rank = CASE + WHEN docker_repository_query_provenance.best_search_rank IS NULL + OR excluded.best_search_rank < docker_repository_query_provenance.best_search_rank + THEN excluded.best_search_rank + ELSE docker_repository_query_provenance.best_search_rank + END, + last_search_rank = excluded.last_search_rank, + first_cycle_id = COALESCE( + docker_repository_query_provenance.first_cycle_id, + excluded.first_cycle_id + ), + last_cycle_id = COALESCE( + excluded.last_cycle_id, + docker_repository_query_provenance.last_cycle_id + ), + first_page_id = COALESCE( + docker_repository_query_provenance.first_page_id, + excluded.first_page_id + ), + last_page_id = excluded.last_page_id, + first_policy_sha256 = COALESCE( + docker_repository_query_provenance.first_policy_sha256, + excluded.first_policy_sha256 + ), + last_policy_sha256 = excluded.last_policy_sha256, + observation_count = docker_repository_query_provenance.observation_count + 1, + fresh_observation_count = docker_repository_query_provenance.fresh_observation_count + 1, + updated_at = excluded.updated_at''', + ( + source, query, queue_id, page_observed_at, page_observed_at, + search_rank, search_rank, search_rank, + observation['cycle_id'], observation['cycle_id'], page_id, page_id, + observation['policy_sha256'], observation['policy_sha256'], now, now, + ), + ) + self.conn.execute( + '''INSERT INTO docker_repository_query_observations( + page_id, repository_queue_id, source, query, + search_rank, observed_at + ) VALUES (?, ?, ?, ?, ?, ?)''', + (page_id, queue_id, source, query, search_rank, page_observed_at), + ) + observation_inserted_count += 1 + + pass_pages = self.conn.execute( + '''SELECT query, query_ordinal, page_number, query_complete + FROM docker_discovery_pages WHERE pass_id = ? + ORDER BY query_ordinal, page_number''', + (pass_row['id'],), + ).fetchall() + expected_query_count = int(pass_row['expected_query_count']) + pages_by_query = {} + names_by_query = {} + terminals_by_query = {} + for pass_page in pass_pages: + ordinal = int(pass_page['query_ordinal']) + if not 0 <= ordinal < expected_query_count: + raise RuntimeError('DockerHub discovery pass ordinal is outside its bound') + names_by_query.setdefault(ordinal, set()).add(str(pass_page['query'])) + pages_by_query.setdefault(ordinal, set()).add(int(pass_page['page_number'])) + if int(pass_page['query_complete']): + terminals_by_query.setdefault(ordinal, set()).add( + int(pass_page['page_number']) + ) + if any(len(names) != 1 for names in names_by_query.values()): + raise RuntimeError('DockerHub discovery pass query identity conflicts') + complete_ordinals = set() + for ordinal, terminals in terminals_by_query.items(): + pages = pages_by_query.get(ordinal, set()) + if any(len({page for page in pages if page <= end}) == end for end in terminals): + complete_ordinals.add(ordinal) + completed_query_count = len(complete_ordinals) + next_pass_state = ( + 'complete' + if completed_query_count == expected_query_count + else 'collecting' + ) + previous_pass_state = str(pass_row['state']) + if previous_pass_state == 'complete' and next_pass_state != 'complete': + raise RuntimeError('DockerHub discovery complete pass lost query evidence') + became_complete = ( + previous_pass_state == 'collecting' and next_pass_state == 'complete' + ) + cursor = self.conn.execute( + '''UPDATE docker_discovery_passes + SET completed_query_count = ?, state = ?, + completed_at = CASE WHEN ? = 'complete' + THEN COALESCE(completed_at, ?) ELSE NULL END, + updated_at = ? + WHERE id = ? AND state IN ('collecting','complete')''', + ( + completed_query_count, next_pass_state, next_pass_state, + now, now, pass_row['id'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError('DockerHub discovery pass completion changed') + if became_complete and str(pass_row['pass_kind']) == 'deep': + completed_rows = self.conn.execute( + '''SELECT observation.source, observation.query, + observation.repository_queue_id, + COUNT(*) AS observation_count + FROM docker_repository_query_observations observation + JOIN docker_discovery_pages page ON page.id = observation.page_id + WHERE page.pass_id = ? + GROUP BY observation.source, observation.query, + observation.repository_queue_id''', + (pass_row['id'],), + ).fetchall() + for completed_row in completed_rows: + cursor = self.conn.execute( + '''UPDATE docker_repository_query_provenance + SET fresh_complete_observation_count = + fresh_complete_observation_count + ?, + fresh_coverage_eligible = 1, + updated_at = ? + WHERE source = ? AND query = ? + AND repository_queue_id = ?''', + ( + int(completed_row['observation_count']), now, + completed_row['source'], completed_row['query'], + completed_row['repository_queue_id'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError('DockerHub discovery coverage evidence changed') + query_complete = observation['query_ordinal'] in complete_ordinals + pass_complete = next_pass_state == 'complete' + + dynamic_hold_count = self._hold_new_docker_depth_repositories_locked( + source, query, new_queue_ids, observation, now, + experiment=page_experiment, authority=page_authority, + ) + retry_progress_count = 0 + retry_completed_count = 0 + if fence and complete: + if page_id is None: + cursor = self.conn.execute( + '''DELETE FROM discovery_retry_queue + WHERE id = ? AND status = 'leased' AND lease_owner = ? + AND lease_token = ? AND lease_expires_at > ?''', + (fence[0], fence[1], fence[2], now), + ) + else: + # Keep the inactive work row because admitted pages reference its identity. + cursor = self.conn.execute( + '''UPDATE discovery_retry_queue SET status = 'held', + available_after = NULL, last_error_category = NULL, + lease_owner = NULL, lease_token = NULL, leased_at = NULL, + lease_expires_at = NULL, held_at = ?, updated_at = ? + WHERE id = ? AND status = 'leased' AND lease_owner = ? + AND lease_token = ? AND lease_expires_at > ?''', + (now, now, fence[0], fence[1], fence[2], now), + ) + retry_completed_count = int(cursor.rowcount or 0) + if retry_completed_count != 1: + raise DiscoveryRetryLeaseError('discovery retry completion lost its lease fence') + elif fence and next_page is not None: + cursor = self.conn.execute( + '''UPDATE discovery_retry_queue SET next_page = ?, updated_at = ? + WHERE id = ? AND status = 'leased' AND lease_owner = ? + AND lease_token = ? AND lease_expires_at > ?''', + (next_page, now, fence[0], fence[1], fence[2], now), + ) + retry_progress_count = int(cursor.rowcount or 0) + if retry_progress_count != 1: + raise DiscoveryRetryLeaseError('discovery retry progress lost its lease fence') + self.conn.commit() + return { + 'attempted_count': attempted_count, + 'normalized_count': len(normalized), + 'duplicate_count': attempted_count - len(normalized), + 'preexisting_count': len(preexisting), + 'inserted_count': inserted_count, + 'dynamic_hold_count': dynamic_hold_count, + 'retry_progress_count': retry_progress_count, + 'retry_completed_count': retry_completed_count, + 'normalized_repositories': frozenset(normalized), + 'preexisting_repositories': frozenset(preexisting), + 'pass_id': int(pass_row['id']) if pass_row is not None else None, + 'page_id': page_id, + 'page_inserted_count': page_inserted_count, + 'observation_inserted_count': observation_inserted_count, + 'completed_query_count': completed_query_count, + 'query_complete': query_complete, + 'pass_complete': pass_complete, + } + except Exception: + self.conn.rollback() + raise + + def dockerhub_discovery_generation_complete( + self, source, collection_generation, ordered_query_hash, + expected_query_count, policy_sha256, + ): + """Return durable authority for the generation's first complete deep pass.""" + if not self.conn: + raise RuntimeError('database connection is unavailable') + from docker_depth_experiment import DOCKER_DEPTH_COLLECTION_GENERATION + source = _validated_discovery_retry_source(source) + collection_generation = str(collection_generation or '') + if collection_generation != DOCKER_DEPTH_COLLECTION_GENERATION: + raise ValueError('DockerHub collection generation authority conflicts') + ordered_query_hash = str(ordered_query_hash or '') + if not re.fullmatch(r'[a-f0-9]{64}', ordered_query_hash): + raise ValueError('DockerHub ordered-query generation authority is invalid') + policy_sha256 = _validated_discovery_retry_policy(policy_sha256) + if isinstance(expected_query_count, bool): + raise ValueError('DockerHub generation query count is invalid') + try: + expected_query_count = int(expected_query_count) + except (TypeError, ValueError, OverflowError): + raise ValueError('DockerHub generation query count is invalid') from None + if not 1 <= expected_query_count <= 1000: + raise ValueError('DockerHub generation query count is invalid') + row = self.conn.execute( + '''SELECT 1 FROM docker_discovery_passes + WHERE source = ? AND pass_kind = 'deep' + AND collection_generation = ? AND policy_sha256 = ? + AND ordered_queries_sha256 = ? AND expected_query_count = ? + AND completed_query_count = expected_query_count + AND state = 'complete' AND completed_at IS NOT NULL + AND ( + SELECT COUNT(DISTINCT terminal.query_ordinal) + FROM docker_discovery_pages terminal + WHERE terminal.pass_id = docker_discovery_passes.id + AND terminal.query_complete = 1 + AND terminal.total_count IS NOT NULL + AND terminal.query_ordinal >= 0 + AND terminal.query_ordinal < expected_query_count + AND terminal.total_count >= terminal.result_count + AND ( + SELECT COUNT(DISTINCT page.page_number) + FROM docker_discovery_pages page + WHERE page.pass_id = terminal.pass_id + AND page.query_ordinal = terminal.query_ordinal + AND page.page_number <= terminal.page_number + ) = terminal.page_number + ) = expected_query_count + ORDER BY id DESC LIMIT 1''', + ( + source, collection_generation, policy_sha256, + ordered_query_hash, expected_query_count, + ), + ).fetchone() + self.conn.commit() + return bool(row) + + def dockerhub_discovery_coverage_summary( + self, source, ordered_queries, ordered_query_hash, + configured_query_policies, required_repository_count=10, + collection_generation=None, + ): + """Return a bounded, read-only gate over complete deep-pass evidence.""" + if not self.conn: + raise RuntimeError('database connection is unavailable') + source = _validated_discovery_retry_source(source) + from docker_depth_experiment import DOCKER_DEPTH_COLLECTION_GENERATION + collection_generation = str( + collection_generation or DOCKER_DEPTH_COLLECTION_GENERATION + ) + if collection_generation != DOCKER_DEPTH_COLLECTION_GENERATION: + raise ValueError('DockerHub collection generation authority conflicts') + if isinstance(ordered_queries, (str, bytes)): + raise ValueError('DockerHub discovery ordered queries are invalid') + try: + queries = tuple(ordered_queries) + except TypeError: + raise ValueError('DockerHub discovery ordered queries are invalid') from None + if ( + not 1 <= len(queries) <= DISCOVERY_RETRY_MAX_ALLOWLIST + or len(set(queries)) != len(queries) + ): + raise ValueError('DockerHub discovery ordered queries are invalid') + queries = tuple(_validated_discovery_retry_query(query) for query in queries) + ordered_query_hash = str(ordered_query_hash or '') + expected_hash = hashlib.sha256(json.dumps( + list(queries), ensure_ascii=True, allow_nan=False, sort_keys=True, + separators=(',', ':'), + ).encode('utf-8')).hexdigest() + if ordered_query_hash != expected_hash: + raise ValueError('DockerHub discovery ordered-query hash conflicts') + if isinstance(required_repository_count, bool): + raise ValueError('DockerHub discovery repository requirement is invalid') + try: + required_repository_count = int(required_repository_count) + except (TypeError, ValueError, OverflowError): + raise ValueError('DockerHub discovery repository requirement is invalid') from None + if not 1 <= required_repository_count <= 39: + raise ValueError('DockerHub discovery repository requirement is invalid') + if not isinstance(configured_query_policies, dict): + raise ValueError('DockerHub discovery query policies are invalid') + policy_values = { + query: ( + value.get('policy_sha256') if isinstance(value, dict) else value + ) + for query, value in configured_query_policies.items() + } + policies = dict(_discovery_retry_allowlist(policy_values)) + if set(policies) != set(queries): + raise ValueError('DockerHub discovery query policies do not match ordered queries') + + query_summaries = [] + try: + for query_ordinal, query in enumerate(queries): + rows = self.conn.execute( + '''SELECT provenance.repository_queue_id, + MIN(observation.search_rank) AS best_search_rank + FROM docker_repository_query_provenance provenance + JOIN docker_repository_query_observations observation + ON observation.source = provenance.source + AND observation.query = provenance.query + AND observation.repository_queue_id = provenance.repository_queue_id + JOIN docker_discovery_pages page ON page.id = observation.page_id + JOIN docker_discovery_passes discovery_pass + ON discovery_pass.id = page.pass_id + JOIN target_queue queue + ON queue.id = provenance.repository_queue_id + WHERE provenance.source = ? AND provenance.query = ? + AND provenance.provenance_kind = 'fresh_page' + AND provenance.fresh_coverage_eligible = 1 + AND provenance.fresh_complete_observation_count > 0 + AND page.query_ordinal = ? AND page.query = ? + AND discovery_pass.source = ? + AND discovery_pass.pass_kind = 'deep' + AND discovery_pass.collection_generation = ? + AND discovery_pass.policy_sha256 = ? + AND discovery_pass.ordered_queries_sha256 = ? + AND discovery_pass.expected_query_count = ? + AND discovery_pass.state = 'complete' + AND queue.source = ? AND queue.platform = 'docker' + AND queue.status IN ('pending','deferred') + AND queue.target_scan_id IS NULL + AND queue.target NOT LIKE '%@%' + AND queue.normalized_target NOT LIKE '%@%' + AND queue.lease_owner IS NULL AND queue.lease_token IS NULL + AND queue.claim_batch IS NULL AND queue.leased_at IS NULL + AND queue.lease_expires_at IS NULL + AND queue.current_result_reservation_id IS NULL + AND queue.claim_event_id IS NULL AND queue.resolver_token IS NULL + AND COALESCE(queue.resolver_state, '') <> 'resolving' + AND NOT EXISTS ( + SELECT 1 FROM target_scans scan + WHERE scan.queue_id = queue.id + ) + AND NOT EXISTS ( + SELECT 1 FROM target_queue_policy_events policy_event + WHERE policy_event.queue_id = queue.id + ) + AND NOT EXISTS ( + SELECT 1 FROM result_reservations reservation + WHERE reservation.queue_id = queue.id + AND reservation.state IN ( + 'scanning','ready','ingesting','db_committed' + ) + ) + AND NOT EXISTS ( + SELECT 1 + FROM result_reservations reservation + JOIN pipeline_quarantine quarantine + ON quarantine.reservation_id = reservation.id + WHERE reservation.queue_id = queue.id + ) + AND NOT EXISTS ( + SELECT 1 FROM docker_image_manifests manifest + WHERE manifest.target_queue_id = queue.id + ) + AND NOT EXISTS ( + SELECT 1 + FROM docker_image_blob_coverage coverage + JOIN docker_content_blobs blob + ON blob.digest = coverage.blob_digest + AND blob.coverage_policy_sha256 = coverage.coverage_policy_sha256 + WHERE coverage.queue_id = queue.id + AND blob.state IN ('leased','submitted') + ) + GROUP BY provenance.repository_queue_id + ORDER BY MIN(observation.search_rank), provenance.repository_queue_id + LIMIT ?''', + ( + source, query, query_ordinal, query, source, + collection_generation, policies[query], + ordered_query_hash, len(queries), source, + required_repository_count, + ), + ).fetchall() + eligible_count = len(rows) + query_summaries.append({ + 'query_ordinal': query_ordinal, + 'query': query, + 'policy_sha256': policies[query], + 'eligible_repository_count': eligible_count, + 'required_repository_count': required_repository_count, + 'covered': eligible_count >= required_repository_count, + }) + covered_query_count = sum( + int(item['covered']) for item in query_summaries + ) + ready = covered_query_count == len(queries) + self.conn.commit() + return { + 'source': source, + 'ordered_query_hash': ordered_query_hash, + 'query_count': len(queries), + 'required_repository_count': required_repository_count, + 'covered_query_count': covered_query_count, + 'minimum_eligible_repository_count': min( + item['eligible_repository_count'] for item in query_summaries + ), + 'planning_allowed': ready, + 'state': 'coverage_complete' if ready else 'collecting', + 'counts_capped_at_requirement': True, + 'queries': query_summaries, + } + except Exception: + self.conn.rollback() + raise + + def enqueue_discovery_retry( + self, source, query, policy_sha256, pass_kind, work_kind, + page_start=1, page_end=None, available_after=None, error_category=None, + observation=None, + ): + if not self.conn: + raise RuntimeError('database connection is unavailable') + source = _validated_discovery_retry_source(source) + query = _validated_discovery_retry_query(query) + policy_sha256 = _validated_discovery_retry_policy(policy_sha256) + pass_kind = str(pass_kind or '') + if pass_kind not in DISCOVERY_RETRY_PASS_KINDS: + raise ValueError('discovery retry pass kind is invalid') + if page_end is None: + page_end = DISCOVERY_RETRY_MAX_PAGE if work_kind == 'query' else page_start + work_kind, page_start, page_end = _validated_discovery_retry_pages( + work_kind, page_start, page_end, + ) + observation = _validated_docker_discovery_observation( + observation, source, query, + ) + if observation is not None and ( + observation['policy_sha256'] != policy_sha256 + or observation['pass_kind'] != pass_kind + or observation['page_number'] != page_start + ): + raise ValueError('discovery retry observation identity conflicts') + error_category = _validated_discovery_retry_error_category(error_category) + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + now_dt = datetime.now(timezone.utc) + now = now_dt.isoformat(timespec='seconds') + available_after = _validated_discovery_retry_time(available_after, now_dt) + pass_row = None + source_cycle_id = observation['cycle_id'] if observation is not None else None + if source_cycle_id is not None: + cycle_row = self.conn.execute( + '''SELECT id FROM source_cycles + WHERE id = ? AND source = ? AND query = ?''', + (source_cycle_id, source, query), + ).fetchone() + if not cycle_row: + raise RuntimeError('discovery retry source cycle identity conflicts') + if observation is not None: + pass_row = self._docker_discovery_pass_locked( + source, query, observation, now, + ) + work_key = discovery_retry_work_key( + source, query, policy_sha256, pass_kind, work_kind, + page_start, page_end, + pass_id=pass_row['id'] if pass_row is not None else None, + ) + inserted = self.conn.execute( + '''INSERT INTO discovery_retry_queue ( + work_key, source, query, source_cycle_id, policy_sha256, + pass_kind, work_kind, + page_start, page_end, next_page, status, attempts, + available_after, last_error_category, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 'pending', 0, ?, ?, ?, ?) + ON CONFLICT(work_key) DO NOTHING + RETURNING id, status, attempts, next_page''', + ( + work_key, source, query, source_cycle_id, policy_sha256, + pass_kind, work_kind, + page_start, page_end, page_start, available_after, error_category, + now, now, + ), + ).fetchone() + inserted_count = int(bool(inserted)) + lock_suffix = ' FOR UPDATE' if self.conn.is_postgres else '' + row = inserted or self.conn.execute( + f'''SELECT id, work_key, source, query, policy_sha256, pass_kind, + work_kind, page_start, page_end, next_page, status, attempts, + available_after, source_cycle_id + FROM discovery_retry_queue WHERE work_key = ?{lock_suffix}''', + (work_key,), + ).fetchone() + if not row: + raise RuntimeError('discovery retry enqueue lost its durable row') + if not inserted and ( + str(row['source']) != source + or str(row['query']) != query + or str(row['policy_sha256']) != policy_sha256 + or str(row['pass_kind']) != pass_kind + or str(row['work_kind']) != work_kind + or int(row['page_start']) != page_start + or int(row['page_end']) != page_end + ): + raise RuntimeError('discovery retry work-key identity conflict') + if not inserted and row['status'] == 'held': + self.conn.execute( + '''UPDATE discovery_retry_queue SET status = 'pending', next_page = page_start, + available_after = ?, last_error_category = ?, held_at = NULL, + updated_at = ? WHERE id = ? AND status = 'held' ''', + (available_after, error_category, now, row['id']), + ) + elif not inserted and row['status'] == 'pending': + self.conn.execute( + '''UPDATE discovery_retry_queue SET + available_after = CASE + WHEN available_after IS NULL OR CAST(? AS TEXT) IS NULL THEN NULL + WHEN available_after <= ? THEN available_after ELSE ? END, + last_error_category = COALESCE(?, last_error_category), + updated_at = ? + WHERE id = ? AND status = 'pending' ''', + ( + available_after, available_after, available_after, + error_category, now, row['id'], + ), + ) + row = self.conn.execute( + '''SELECT id, status, attempts, page_start, page_end, next_page, + source_cycle_id + FROM discovery_retry_queue WHERE work_key = ?''', + (work_key,), + ).fetchone() + if not row: + raise RuntimeError('discovery retry enqueue could not verify durability') + self.conn.commit() + return { + 'id': int(row['id']), + 'work_key': work_key, + 'inserted_count': inserted_count, + 'coalesced_count': 1 - inserted_count, + 'status': str(row['status']), + 'attempts': int(row['attempts']), + 'page_start': int(row['page_start']), + 'page_end': int(row['page_end']), + 'next_page': int(row['next_page']), + 'source_cycle_id': ( + int(row['source_cycle_id']) + if row['source_cycle_id'] is not None else None + ), + 'pass_id': int(pass_row['id']) if pass_row is not None else None, + } + except Exception: + self.conn.rollback() + raise + + def claim_discovery_retries( + self, source, configured_query_policies, lease_owner, limit=1, + lease_seconds=300, + ): + if not self.conn: + raise RuntimeError('database connection is unavailable') + source = _validated_discovery_retry_source(source) + allowed = _discovery_retry_allowlist(configured_query_policies) + owner = str(lease_owner or '') + if ( + not 1 <= len(owner) <= 128 + or any(ord(character) < 32 or ord(character) == 127 for character in owner) + ): + raise ValueError('discovery retry lease owner is invalid') + if isinstance(limit, bool) or isinstance(lease_seconds, bool): + raise ValueError('discovery retry claim bounds are invalid') + try: + limit = int(limit) + lease_seconds = int(lease_seconds) + except (TypeError, ValueError, OverflowError): + raise ValueError('discovery retry claim bounds are invalid') from None + if not 1 <= limit <= DISCOVERY_RETRY_MAX_CLAIM or not 1 <= lease_seconds <= 86400: + raise ValueError('discovery retry claim bounds are invalid') + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + state = self._locked_runtime_control_state(shared=True) + if state['effective_discovery_paused']: + self.conn.commit() + return [] + now_dt = datetime.now(timezone.utc) + now = now_dt.isoformat(timespec='seconds') + lease_expires_at = (now_dt + timedelta(seconds=lease_seconds)).isoformat( + timespec='seconds', + ) + pair_sql = ' OR '.join( + '(query = ? AND policy_sha256 = ?)' for _ in allowed + ) or '0 = 1' + pair_params = tuple(value for pair in allowed for value in pair) + query_sql = ','.join('?' for _ in allowed) + query_params = tuple(query for query, _ in allowed) + category_sql = ( + f"CASE WHEN query IN ({query_sql}) THEN 'policy_mismatch' ELSE 'query_removed' END" + if allowed else "'query_removed'" + ) + self.conn.execute( + f'''UPDATE discovery_retry_queue SET status = 'held', + lease_owner = NULL, lease_token = NULL, leased_at = NULL, + lease_expires_at = NULL, available_after = NULL, + held_at = ?, last_error_category = {category_sql}, updated_at = ? + WHERE source = ? + AND (status = 'pending' OR (status = 'leased' AND lease_expires_at <= ?)) + AND NOT ({pair_sql})''', + (now, *query_params, now, source, now, *pair_params), + ) + if not allowed: + self.conn.commit() + return [] + lock_suffix = ' FOR UPDATE SKIP LOCKED' if self.conn.is_postgres else '' + rows = self.conn.execute( + f'''SELECT id, work_key, source, query, policy_sha256, pass_kind, + work_kind, page_start, page_end, next_page, attempts, + last_error_category, source_cycle_id + FROM discovery_retry_queue + WHERE source = ? AND ({pair_sql}) + AND ( + (status = 'pending' AND (available_after IS NULL OR available_after <= ?)) + OR (status = 'leased' AND lease_expires_at IS NOT NULL + AND lease_expires_at <= ?) + ) + ORDER BY COALESCE(available_after, lease_expires_at, created_at), id + LIMIT ?{lock_suffix}''', + (source, *pair_params, now, now, limit), + ).fetchall() + claimed = [] + for row in rows: + token = secrets.token_urlsafe(32) + updated = self.conn.execute( + '''UPDATE discovery_retry_queue SET status = 'leased', + lease_owner = ?, lease_token = ?, leased_at = ?, + lease_expires_at = ?, held_at = NULL, + attempts = CASE WHEN attempts < ? THEN attempts + 1 ELSE attempts END, + updated_at = ? + WHERE id = ? AND source = ? AND ( + (status = 'pending' AND (available_after IS NULL OR available_after <= ?)) + OR (status = 'leased' AND lease_expires_at IS NOT NULL + AND lease_expires_at <= ?) + ) RETURNING attempts''', + ( + owner, token, now, lease_expires_at, DISCOVERY_RETRY_MAX_ATTEMPTS, + now, row['id'], source, now, now, + ), + ).fetchone() + if not updated: + raise DiscoveryRetryLeaseError('discovery retry claim lost its selected row') + claimed.append({ + 'id': int(row['id']), + 'work_key': str(row['work_key']), + 'source': str(row['source']), + 'query': str(row['query']), + 'policy_sha256': str(row['policy_sha256']), + 'pass_kind': str(row['pass_kind']), + 'work_kind': str(row['work_kind']), + 'page_start': int(row['page_start']), + 'page_end': int(row['page_end']), + 'next_page': int(row['next_page']), + 'attempts': int(updated['attempts']), + 'lease_owner': owner, + 'lease_token': token, + 'lease_expires_at': lease_expires_at, + 'last_error_category': row['last_error_category'], + 'source_cycle_id': ( + int(row['source_cycle_id']) + if row['source_cycle_id'] is not None else None + ), + }) + self.conn.commit() + return claimed + except Exception: + self.conn.rollback() + raise + + def finish_discovery_retry(self, retry_id, lease_owner, lease_token): + if not self.conn: + raise RuntimeError('database connection is unavailable') + retry_id, lease_owner, lease_token = _validated_discovery_retry_fence( + retry_id, lease_owner, lease_token, + ) + now = utc_now_iso() + try: + cursor = self.conn.execute( + '''DELETE FROM discovery_retry_queue + WHERE id = ? AND status = 'leased' AND lease_owner = ? + AND lease_token = ? AND lease_expires_at > ?''', + (retry_id, lease_owner, lease_token, now), + ) + if int(cursor.rowcount or 0) != 1: + raise DiscoveryRetryLeaseError('discovery retry completion lost its lease fence') + self.conn.commit() + return {'id': retry_id, 'deleted_count': 1} + except Exception: + self.conn.rollback() + raise + + def renew_discovery_retry_lease( + self, retry_id, lease_owner, lease_token, lease_seconds=300, + ): + if not self.conn: + raise RuntimeError('database connection is unavailable') + retry_id, lease_owner, lease_token = _validated_discovery_retry_fence( + retry_id, lease_owner, lease_token, + ) + if isinstance(lease_seconds, bool): + raise ValueError('discovery retry lease duration is invalid') + try: + lease_seconds = int(lease_seconds) + except (TypeError, ValueError, OverflowError): + raise ValueError('discovery retry lease duration is invalid') from None + if not 1 <= lease_seconds <= 86400: + raise ValueError('discovery retry lease duration is invalid') + now_dt = datetime.now(timezone.utc) + now = now_dt.isoformat(timespec='seconds') + lease_expires_at = (now_dt + timedelta(seconds=lease_seconds)).isoformat( + timespec='seconds', + ) + try: + cursor = self.conn.execute( + '''UPDATE discovery_retry_queue SET lease_expires_at = ?, updated_at = ? + WHERE id = ? AND status = 'leased' AND lease_owner = ? + AND lease_token = ? AND lease_expires_at > ?''', + ( + lease_expires_at, now, retry_id, lease_owner, lease_token, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise DiscoveryRetryLeaseError('discovery retry renewal lost its lease fence') + self.conn.commit() + return { + 'id': retry_id, + 'status': 'leased', + 'lease_expires_at': lease_expires_at, + } + except Exception: + self.conn.rollback() + raise + + def update_discovery_retry( + self, retry_id, lease_owner, lease_token, error_category, + retry_at=None, refund_attempt=False, next_page=None, + ): + if not self.conn: + raise RuntimeError('database connection is unavailable') + retry_id, lease_owner, lease_token = _validated_discovery_retry_fence( + retry_id, lease_owner, lease_token, + ) + error_category = _validated_discovery_retry_error_category( + error_category, required=True, + ) + if not isinstance(refund_attempt, bool): + raise ValueError('discovery retry refund flag is invalid') + if refund_attempt and retry_at is None: + raise ValueError('discovery retry attempt refund requires a trusted retry time') + if next_page is not None: + if isinstance(next_page, bool): + raise ValueError('discovery retry next page is invalid') + try: + next_page = int(next_page) + except (TypeError, ValueError, OverflowError): + raise ValueError('discovery retry next page is invalid') from None + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + now_dt = datetime.now(timezone.utc) + now = now_dt.isoformat(timespec='seconds') + lock_suffix = ' FOR UPDATE' if self.conn.is_postgres else '' + row = self.conn.execute( + f'''SELECT attempts, page_start, page_end, next_page + FROM discovery_retry_queue + WHERE id = ? AND status = 'leased' AND lease_owner = ? + AND lease_token = ? AND lease_expires_at > ?{lock_suffix}''', + (retry_id, lease_owner, lease_token, now), + ).fetchone() + if not row: + raise DiscoveryRetryLeaseError('discovery retry update lost its lease fence') + if next_page is None: + next_page = int(row['next_page']) + elif not ( + int(row['next_page']) <= next_page <= int(row['page_end']) + and 1 <= next_page <= DISCOVERY_RETRY_MAX_PAGE + ): + raise ValueError('discovery retry next page is outside its durable range') + retry_due = _validated_discovery_retry_time(retry_at, now_dt) + attempts = int(row['attempts']) + if retry_due is None: + exponent = min(20, max(0, attempts - 1)) + delay = min( + DISCOVERY_RETRY_MAX_DELAY_SEC, + DISCOVERY_RETRY_BASE_DELAY_SEC * (2 ** exponent), + ) + retry_due = (now_dt + timedelta(seconds=delay)).isoformat(timespec='seconds') + next_attempts = max(0, attempts - 1) if refund_attempt else attempts + cursor = self.conn.execute( + '''UPDATE discovery_retry_queue SET status = 'pending', attempts = ?, + next_page = ?, available_after = ?, last_error_category = ?, + lease_owner = NULL, lease_token = NULL, leased_at = NULL, + lease_expires_at = NULL, held_at = NULL, updated_at = ? + WHERE id = ? AND status = 'leased' AND lease_owner = ? + AND lease_token = ? AND lease_expires_at > ?''', + ( + next_attempts, next_page, retry_due, error_category, now, + retry_id, lease_owner, lease_token, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise DiscoveryRetryLeaseError('discovery retry update lost its lease fence') + self.conn.commit() + return { + 'id': retry_id, + 'status': 'pending', + 'attempts': next_attempts, + 'next_page': next_page, + 'available_after': retry_due, + 'last_error_category': error_category, + } + except Exception: + self.conn.rollback() + raise + + def hold_discovery_retry( + self, retry_id, lease_owner, lease_token, error_category='policy_mismatch', + ): + if not self.conn: + raise RuntimeError('database connection is unavailable') + retry_id, lease_owner, lease_token = _validated_discovery_retry_fence( + retry_id, lease_owner, lease_token, + ) + error_category = _validated_discovery_retry_error_category( + error_category, required=True, + ) + now = utc_now_iso() + try: + cursor = self.conn.execute( + '''UPDATE discovery_retry_queue SET status = 'held', + available_after = NULL, last_error_category = ?, held_at = ?, + lease_owner = NULL, lease_token = NULL, leased_at = NULL, + lease_expires_at = NULL, updated_at = ? + WHERE id = ? AND status = 'leased' AND lease_owner = ? + AND lease_token = ? AND lease_expires_at > ?''', + ( + error_category, now, now, retry_id, lease_owner, lease_token, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise DiscoveryRetryLeaseError('discovery retry hold lost its lease fence') + self.conn.commit() + return { + 'id': retry_id, + 'status': 'held', + 'last_error_category': error_category, + } + except Exception: + self.conn.rollback() + raise + + def enqueue_targets( + self, source, platform, query, targets, requeue_done=False, + unresolved_targets=None, *, discovery_admission=False, + ): + targets = list(targets or []) + unresolved_targets = list(unresolved_targets or []) + if not isinstance(discovery_admission, bool): + raise ValueError('discovery admission flag must be boolean') + if not self.conn or (not targets and not unresolved_targets): + return 0 + + def op(): + if discovery_admission: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + self._require_discovery_admission_locked() + now = utc_now_iso() + count = 0 + for target in targets or []: + normalized = normalize_target(target, platform) + if requeue_done: + reset_status = f"target_queue.status = 'done' OR ({ADMIN_DISCARDED_QUEUE_SQL})" + status_expr = f"CASE WHEN {reset_status} THEN 'pending' ELSE target_queue.status END" + available_after_expr = f"CASE WHEN {reset_status} THEN NULL ELSE target_queue.available_after END" + attempts_expr = f"CASE WHEN {reset_status} THEN 0 ELSE target_queue.attempts END" + completed_at_expr = f"CASE WHEN {reset_status} THEN NULL ELSE target_queue.completed_at END" + conflict_status_where = reset_status + else: + status_expr = f"CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} THEN 'pending' WHEN target_queue.status IN ('done', 'failed', 'deferred', 'in_progress', 'cold') THEN target_queue.status WHEN target_queue.status = 'pending' AND target_queue.available_after IS NOT NULL THEN 'deferred' ELSE 'pending' END" + available_after_expr = f"CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} THEN NULL WHEN target_queue.status IN ('pending', 'deferred', 'cold') THEN target_queue.available_after ELSE NULL END" + attempts_expr = f'CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} THEN 0 ELSE target_queue.attempts END' + completed_at_expr = f'CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} THEN NULL ELSE target_queue.completed_at END' + conflict_status_where = f"target_queue.status = 'pending' AND target_queue.available_after IS NOT NULL OR ({ADMIN_DISCARDED_QUEUE_SQL})" + reset_for_discovery = f"({ADMIN_DISCARDED_QUEUE_SQL}) OR (target_queue.status = 'done' AND {str(bool(requeue_done)).upper()})" + self.conn.execute( + f'''INSERT INTO target_queue ( + source, platform, query, target, normalized_target, status, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'pending', ?, ?) + ON CONFLICT(source, normalized_target) DO UPDATE SET + target = excluded.target, + platform = excluded.platform, + query = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN excluded.query ELSE COALESCE(target_queue.query, excluded.query) END, + status = {status_expr}, + lease_owner = CASE WHEN {reset_for_discovery} THEN NULL ELSE target_queue.lease_owner END, + lease_token = CASE WHEN {reset_for_discovery} THEN NULL ELSE target_queue.lease_token END, + claim_batch = CASE WHEN {reset_for_discovery} THEN NULL ELSE target_queue.claim_batch END, + leased_at = CASE WHEN {reset_for_discovery} THEN NULL ELSE target_queue.leased_at END, + lease_expires_at = CASE WHEN {reset_for_discovery} THEN NULL ELSE target_queue.lease_expires_at END, + available_after = {available_after_expr}, + attempts = {attempts_expr}, + completed_at = {completed_at_expr}, + last_error = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN NULL ELSE target_queue.last_error END, + resolver_state = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN NULL ELSE target_queue.resolver_state END, + resolver_due_at = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN NULL ELSE target_queue.resolver_due_at END, + resolver_attempts = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN 0 ELSE target_queue.resolver_attempts END, + resolver_token = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN NULL ELSE target_queue.resolver_token END, + updated_at = excluded.updated_at + WHERE ({conflict_status_where}) + OR target_queue.target <> excluded.target + OR target_queue.platform <> excluded.platform + OR (target_queue.query IS NULL AND excluded.query IS NOT NULL)''', + (source, platform, query, target, normalized, now, now), + ) + count += 1 + for target in unresolved_targets: + normalized = normalize_target(target, platform) + resolver_due = datetime.fromtimestamp( + time.time() + max(60, env_int('DOCKER_RESOLVER_RETRY_SEC', 3600)), timezone.utc + ).isoformat(timespec='seconds') + self.conn.execute( + f'''INSERT INTO target_queue ( + source, platform, query, target, normalized_target, status, + available_after, last_error, resolver_state, resolver_due_at, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'deferred', ?, 'Docker tag resolution unresolved', 'pending', ?, ?, ?) + ON CONFLICT(source, normalized_target) DO UPDATE SET + status = 'deferred', available_after = excluded.available_after, + query = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN excluded.query ELSE target_queue.query END, + target = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN excluded.target ELSE target_queue.target END, + resolver_state = 'pending', resolver_due_at = excluded.resolver_due_at, + resolver_attempts = CASE WHEN {ADMIN_DISCARDED_QUEUE_SQL} + THEN 0 ELSE target_queue.resolver_attempts END, + resolver_token = NULL, completed_at = NULL, + last_error = excluded.last_error, updated_at = excluded.updated_at + WHERE target_queue.source = excluded.source + AND target_queue.platform = excluded.platform + AND excluded.platform = 'docker' + AND ( + ({ADMIN_DISCARDED_QUEUE_SQL}) + OR (target_queue.resolver_state IS NULL AND target_queue.status IN ('pending', 'deferred')) + OR (target_queue.resolver_state = 'resolving' AND target_queue.status = 'deferred' + AND target_queue.resolver_due_at IS NOT NULL AND target_queue.resolver_due_at <= ?) + )''', + (source, platform, query, target, normalized, resolver_due, resolver_due, now, now, now), + ) + count += 1 + self.conn.commit() + return count + return self._safe('enqueue_targets', op, 0) + + def observe_discovered_targets( + self, source, platform, query, discoveries, rescan_limit=0, cooldown_seconds=0, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('updated-target observation requires PostgreSQL') + + normalized_records = {} + attempted_count = 0 + for discovery in discoveries or []: + attempted_count += 1 + if isinstance(discovery, dict): + target = str(discovery.get('target') or '').strip() + remote_value = discovery.get('remote_modified_at') + else: + target = str(discovery or '').strip() + remote_value = None + if not target: + continue + if source == 'huggingface': + try: + target = normalize_huggingface_space_id(target) + except (TypeError, ValueError): + continue + normalized = normalize_target(target, platform) + remote_time = parse_time(remote_value) + remote_text = remote_time.isoformat(timespec='seconds') if remote_time else None + current = normalized_records.get(normalized) + if current is None: + normalized_records[normalized] = { + 'target': target, + 'remote_modified_at': remote_text, + } + elif remote_text and ( + not current['remote_modified_at'] + or remote_text > current['remote_modified_at'] + ): + current['target'] = target + current['remote_modified_at'] = remote_text + + if not normalized_records: + return { + 'attempted_count': attempted_count, + 'queued_new_count': 0, + 'queued_updated_count': 0, + } + + now_dt = datetime.now(timezone.utc) + now = now_dt.isoformat(timespec='seconds') + cooldown_cutoff = ( + now_dt - timedelta(seconds=max(0, int(cooldown_seconds or 0))) + ).isoformat(timespec='seconds') + observed_existing_ids = [] + queued_new = 0 + queued_updated = 0 + try: + self._require_discovery_admission_locked() + for normalized in sorted(normalized_records): + record = normalized_records[normalized] + inserted = self.conn.execute( + '''INSERT INTO target_queue ( + source, platform, query, target, normalized_target, status, + remote_modified_at, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'pending', ?, ?, ?) + ON CONFLICT(source, normalized_target) DO NOTHING + RETURNING id''', + ( + source, platform, query, record['target'], normalized, + record['remote_modified_at'], now, now, + ), + ).fetchone() + if inserted: + queued_new += 1 + continue + + row = self.conn.execute( + '''SELECT id, target, platform, query, remote_modified_at, + status, last_error + FROM target_queue + WHERE source = ? AND normalized_target = ? + FOR UPDATE''', + (source, normalized), + ).fetchone() + if not row: + raise RuntimeError('discovered target disappeared during observation') + remote_text = record['remote_modified_at'] + if ( + str(row['status']) == 'quarantined' + and str(row['last_error'] or '') == ADMIN_DISCARDED_QUEUE_REASON + ): + cursor = self.conn.execute( + '''UPDATE target_queue SET target = ?, platform = ?, query = ?, + status = 'pending', attempts = 0, available_after = NULL, + last_error = NULL, completed_at = NULL, + lease_owner = NULL, lease_token = NULL, + claim_batch = NULL, leased_at = NULL, + lease_expires_at = NULL, resolver_state = NULL, + resolver_due_at = NULL, resolver_attempts = 0, + resolver_token = NULL, remote_modified_at = ?, + updated_at = ? + WHERE id = ? AND status = 'quarantined' AND last_error = ?''', + ( + record['target'], platform, query, remote_text, now, + row['id'], ADMIN_DISCARDED_QUEUE_REASON, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError('discarded target reactivation authority changed') + queued_updated += 1 + continue + remote_advanced = bool( + remote_text + and ( + not row['remote_modified_at'] + or remote_text > str(row['remote_modified_at']) + ) + ) + if ( + remote_advanced + or str(row['target']) != record['target'] + or str(row['platform']) != str(platform) + or (row['query'] is None and query is not None) + ): + next_remote = remote_text if remote_advanced else row['remote_modified_at'] + self.conn.execute( + '''UPDATE target_queue SET + target = ?, platform = ?, query = COALESCE(query, ?), + remote_modified_at = ?, + updated_at = ? + WHERE id = ?''', + ( + record['target'], platform, query, next_remote, now, row['id'], + ), + ) + if remote_text: + observed_existing_ids.append(int(row['id'])) + + limit = max(0, int(rescan_limit or 0)) + if limit and observed_existing_ids: + placeholders = ','.join('?' for _ in observed_existing_ids) + eligible = self.conn.execute( + f'''SELECT id FROM target_queue + WHERE id IN ({placeholders}) + AND source = ? AND platform = ? AND status = 'done' + AND remote_modified_at IS NOT NULL + AND completed_at IS NOT NULL AND completed_at <= ? + AND ( + (scan_remote_modified_at IS NULL AND remote_modified_at > completed_at) + OR (scan_remote_modified_at IS NOT NULL AND remote_modified_at > scan_remote_modified_at) + ) + AND (resolver_state IS NULL OR resolver_state = 'resolved') + AND resolver_token IS NULL + AND lease_owner IS NULL AND lease_token IS NULL + AND claim_batch IS NULL AND leased_at IS NULL + AND lease_expires_at IS NULL + AND current_result_reservation_id IS NULL + AND claim_event_id IS NULL + ORDER BY remote_modified_at DESC, id + LIMIT ? FOR UPDATE SKIP LOCKED''', + (*observed_existing_ids, source, platform, cooldown_cutoff, limit), + ).fetchall() + eligible_ids = [int(row['id']) for row in eligible] + if eligible_ids: + eligible_placeholders = ','.join('?' for _ in eligible_ids) + promoted = self.conn.execute( + f'''UPDATE target_queue SET + status = 'pending', attempts = 0, available_after = NULL, + last_error = NULL, completed_at = NULL, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, + leased_at = NULL, lease_expires_at = NULL, + current_result_reservation_id = NULL, claim_event_id = NULL, + updated_at = ? + WHERE id IN ({eligible_placeholders}) + AND status = 'done' AND current_result_reservation_id IS NULL''', + (now, *eligible_ids), + ) + promoted_count = int(promoted.rowcount or 0) + if promoted_count != len(eligible_ids): + raise RuntimeError('updated-target promotion authority changed') + queued_updated += promoted_count + self.conn.commit() + return { + 'attempted_count': attempted_count, + 'queued_new_count': queued_new, + 'queued_updated_count': queued_updated, + } + except Exception: + self.conn.rollback() + raise + + @staticmethod + def _admission_intent_sha256(values): + encoded = json.dumps( + values, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8') + return hashlib.sha256(encoded).hexdigest() + + def provision_remote_worker_device( + self, user_key, device_key, token_sha256, active_assignment_cap, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote worker provisioning requires PostgreSQL') + user_key = str(user_key or '').strip() + device_key = str(device_key or '').strip() + token_sha256 = str(token_sha256 or '').strip().lower() + cap = int(active_assignment_cap) + if not 1 <= len(user_key) <= 128 or not 1 <= len(device_key) <= 128: + raise ValueError('remote worker user and device keys must be 1..128 characters') + if not re.fullmatch(r'[a-f0-9]{64}', token_sha256): + raise ValueError('remote worker token digest must be lowercase SHA-256') + if not 0 <= cap <= 10000: + raise ValueError('remote worker active assignment cap is out of range') + now = utc_now_iso() + try: + self.conn.execute( + '''INSERT INTO remote_worker_users( + user_key, active_assignment_cap, created_at, updated_at + ) VALUES (?, ?, ?, ?) + ON CONFLICT(user_key) DO UPDATE SET + active_assignment_cap = excluded.active_assignment_cap, + updated_at = excluded.updated_at''', + (user_key, cap, now, now), + ) + user = self.conn.execute( + 'SELECT * FROM remote_worker_users WHERE user_key = ? FOR UPDATE', + (user_key,), + ).fetchone() + self.conn.execute( + '''INSERT INTO remote_worker_devices( + user_id, device_key, token_sha256, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?) + ON CONFLICT(device_key) DO UPDATE SET + token_sha256 = excluded.token_sha256, + updated_at = excluded.updated_at''', + (user['id'], device_key, token_sha256, now, now), + ) + device = self.conn.execute( + '''SELECT id, user_id, device_key, revoked_at, created_at, updated_at + FROM remote_worker_devices WHERE device_key = ?''', + (device_key,), + ).fetchone() + if int(device['user_id']) != int(user['id']): + raise ScanEventConflictError( + 'remote worker device key already belongs to another user' + ) + self.conn.commit() + return { + 'user_id': int(user['id']), 'user_key': user_key, + 'device_id': int(device['id']), 'device_key': device_key, + 'active_assignment_cap': cap, + 'revoked': device['revoked_at'] is not None, + } + except Exception: + self.conn.rollback() + raise + + def create_remote_worker_user(self, user_key, active_assignment_cap): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote worker user creation requires PostgreSQL') + user_key = str(user_key or '').strip() + cap = int(active_assignment_cap) + if not 1 <= len(user_key) <= 128: + raise ValueError('remote worker user key must be 1..128 characters') + if not 0 <= cap <= 10000: + raise ValueError('remote worker active assignment cap is out of range') + now = utc_now_iso() + try: + cursor = self.conn.execute( + '''INSERT INTO remote_worker_users( + user_key, active_assignment_cap, created_at, updated_at + ) VALUES (?, ?, ?, ?) ON CONFLICT DO NOTHING''', + (user_key, cap, now, now), + ) + if int(cursor.rowcount or 0) != 1: + self.conn.rollback() + return None + row = self.conn.execute( + '''SELECT id, user_key, active_assignment_cap, disabled_at + FROM remote_worker_users WHERE user_key = ?''', + (user_key,), + ).fetchone() + self.conn.commit() + return { + 'user_id': int(row['id']), 'user_key': str(row['user_key']), + 'active_assignment_cap': int(row['active_assignment_cap']), + 'disabled': row['disabled_at'] is not None, + } + except Exception: + self.conn.rollback() + raise + + def set_remote_worker_user_cap(self, user_key, active_assignment_cap): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote worker user update requires PostgreSQL') + user_key = str(user_key or '').strip() + cap = int(active_assignment_cap) + if not 1 <= len(user_key) <= 128: + raise ValueError('remote worker user key must be 1..128 characters') + if not 0 <= cap <= 10000: + raise ValueError('remote worker active assignment cap is out of range') + now = utc_now_iso() + try: + cursor = self.conn.execute( + '''UPDATE remote_worker_users SET active_assignment_cap = ?, updated_at = ? + WHERE user_key = ?''', + (cap, now, user_key), + ) + self.conn.commit() + return int(cursor.rowcount or 0) == 1 + except Exception: + self.conn.rollback() + raise + + def set_remote_worker_user_disabled(self, user_key, disabled=True): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote worker user update requires PostgreSQL') + user_key = str(user_key or '').strip() + if not 1 <= len(user_key) <= 128: + raise ValueError('remote worker user key must be 1..128 characters') + now = utc_now_iso() + try: + cursor = self.conn.execute( + '''UPDATE remote_worker_users SET disabled_at = ?, updated_at = ? + WHERE user_key = ?''', + (now if disabled else None, now, user_key), + ) + self.conn.commit() + return int(cursor.rowcount or 0) == 1 + except Exception: + self.conn.rollback() + raise + + def issue_remote_worker_device( + self, user_key, device_key, token_sha256, *, rotate=False, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote worker device issuance requires PostgreSQL') + user_key = str(user_key or '').strip() + device_key = str(device_key or '').strip() + token_sha256 = str(token_sha256 or '').strip().lower() + if not 1 <= len(user_key) <= 128 or not 1 <= len(device_key) <= 128: + raise ValueError('remote worker user and device keys must be 1..128 characters') + if not re.fullmatch(r'[a-f0-9]{64}', token_sha256): + raise ValueError('remote worker token digest must be lowercase SHA-256') + now = utc_now_iso() + try: + user = self.conn.execute( + '''SELECT id, user_key FROM remote_worker_users + WHERE user_key = ? FOR SHARE''', + (user_key,), + ).fetchone() + if not user: + self.conn.rollback() + return None + if rotate: + cursor = self.conn.execute( + '''UPDATE remote_worker_devices SET token_sha256 = ?, updated_at = ? + WHERE user_id = ? AND device_key = ?''', + (token_sha256, now, user['id'], device_key), + ) + else: + cursor = self.conn.execute( + '''INSERT INTO remote_worker_devices( + user_id, device_key, token_sha256, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?) ON CONFLICT DO NOTHING''', + (user['id'], device_key, token_sha256, now, now), + ) + if int(cursor.rowcount or 0) != 1: + self.conn.rollback() + return None + device = self.conn.execute( + '''SELECT id, user_id, device_key, revoked_at + FROM remote_worker_devices WHERE user_id = ? AND device_key = ?''', + (user['id'], device_key), + ).fetchone() + if not device: + raise ScanEventConflictError('remote worker device issuance was not durable') + self.conn.commit() + return { + 'user_id': int(user['id']), 'user_key': str(user['user_key']), + 'device_id': int(device['id']), 'device_key': str(device['device_key']), + 'revoked': device['revoked_at'] is not None, + } + except Exception: + self.conn.rollback() + raise + + def admin_remote_worker_snapshot(self, limit=200, filters=None): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote worker admin snapshot requires PostgreSQL') + limit = int(limit) + if not 1 <= limit <= 500: + raise ValueError('remote worker admin snapshot limit is out of range') + filters, filter_conditions, filter_parameters = _admin_worker_filter_sql( + filters, + ) + assignment_outcome = _admin_assignment_outcome_sql() + scan_outcome = _admin_scan_outcome_sql() + assignment_where = ' AND '.join(( + "r.assignment_kind = 'remote'", *filter_conditions, + )) + try: + users = self.conn.execute( + '''SELECT user_key, active_assignment_cap, + (disabled_at IS NOT NULL) AS disabled + FROM remote_worker_users ORDER BY id DESC LIMIT ?''', + (limit,), + ).fetchall() + workers = self.conn.execute( + '''SELECT d.device_key, u.user_key, u.active_assignment_cap, + d.last_contact_at, + (d.revoked_at IS NOT NULL) AS revoked, + activity.current_phases, + activity.latest_progress_age_seconds, + activity.known_reasons, + activity.pending_local_recovery, + activity.active_package_identity, + COUNT(r.id) FILTER ( + WHERE r.remote_resolution_kind IS NULL + ) AS unfinished_count, + COUNT(r.id) FILTER ( + WHERE r.remote_resolution_kind IS NULL + ) AS active_slot_count, + COUNT(r.id) FILTER ( + WHERE r.remote_resolution_kind = 'bundle_accepted' + ) AS completed_count, + COUNT(r.id) FILTER ( + WHERE r.remote_resolution_kind = 'prebundle_report' + ) AS failed_count, + COUNT(r.id) FILTER ( + WHERE r.remote_resolution_kind = 'expired' + ) AS expired_count + FROM remote_worker_devices d + JOIN remote_worker_users u ON u.id = d.user_id + LEFT JOIN result_reservations r + ON r.remote_device_id = d.id AND r.assignment_kind = 'remote' + LEFT JOIN LATERAL ( + SELECT STRING_AGG( + DISTINCT COALESCE(active.phase, 'legacy/unavailable'), + ', ' ORDER BY COALESCE(active.phase, 'legacy/unavailable') + ) AS current_phases, + MIN(active.progress_age_seconds) + AS latest_progress_age_seconds, + STRING_AGG( + DISTINCT active.known_reason, ', ' + ORDER BY active.known_reason + ) FILTER (WHERE active.known_reason IS NOT NULL) + AS known_reasons, + COALESCE(BOOL_OR( + active.pending_local_recovery = 'true' + ), FALSE) AS pending_local_recovery, + STRING_AGG( + DISTINCT active.package_identity, ', ' + ORDER BY active.package_identity + ) FILTER (WHERE active.package_identity IS NOT NULL) + AS active_package_identity + FROM ( + SELECT ar.id, lp.phase, + CASE WHEN lp.event_timestamp IS NOT NULL + THEN GREATEST(0, EXTRACT(EPOCH FROM ( + CURRENT_TIMESTAMP + - lp.event_timestamp::timestamptz + ))::BIGINT) END AS progress_age_seconds, + lp.known_reason, lp.pending_local_recovery, + ( + ar.remote_execution_snapshot_json::jsonb #>> + '{compatibility,platform_tag}' + ) || ':' || LEFT(( + ar.remote_execution_snapshot_json::jsonb #>> + '{compatibility,code_manifest_sha256}' + ), 12) AS package_identity + FROM result_reservations ar + LEFT JOIN LATERAL ( + SELECT e.phase, e.event_timestamp, + e.event_json::jsonb #>> '{progress,reason}' + AS known_reason, + e.event_json::jsonb #>> + '{progress,pending_local_recovery}' + AS pending_local_recovery + FROM worker_progress_events e + WHERE e.reservation_id = ar.id + ORDER BY e.sequence DESC, e.id DESC LIMIT 1 + ) lp ON TRUE + WHERE ar.remote_device_id = d.id + AND ar.assignment_kind = 'remote' + AND ar.remote_resolution_kind IS NULL + ) active + ) activity ON TRUE + GROUP BY d.id, d.device_key, u.user_key, + u.active_assignment_cap, d.last_contact_at, d.revoked_at, + activity.current_phases, + activity.latest_progress_age_seconds, + activity.known_reasons, + activity.pending_local_recovery, + activity.active_package_identity + ORDER BY d.id DESC LIMIT ?''', + (limit,), + ).fetchall() + assignments = self.conn.execute( + f'''SELECT r.id AS reservation_id, q.id AS queue_id, + u.user_key, u.active_assignment_cap, d.device_key, r.source, + r.normalized_target AS target, + r.remote_issued_at AS issued_at, + r.remote_expires_at AS assignment_deadline_at, + r.remote_result_upload_body_timeout_seconds, + r.remote_resolved_at AS finished_at, + CASE WHEN r.remote_resolved_at IS NOT NULL + AND r.remote_issued_at IS NOT NULL + THEN GREATEST(0, EXTRACT(EPOCH FROM ( + r.remote_resolved_at::timestamptz + - r.remote_issued_at::timestamptz + ))::BIGINT) + ELSE NULL END AS duration_seconds, + {assignment_outcome} AS assignment_outcome, + {scan_outcome} AS scan_outcome, + COALESCE( + r.remote_resolution_kind = 'bundle_accepted', FALSE + ) AS accepted, + COALESCE( + b.committed_at IS NOT NULL OR b.state IN ( + 'db_committed', 'acknowledged' + ), FALSE + ) AS ingested, + r.last_error_code AS assignment_code, + s.id AS target_scan_id, s.error_count AS scan_error_count, + s.first_error_summary, s.skipped_reason, + src.metadata_json::jsonb #>> '{{warnings,0}}' + AS scan_warning_summary, + src.metadata_json::jsonb #>> '{{warning_classes,0}}' + AS scan_warning_class, + COALESCE(dg.diagnostic_count, 0) AS diagnostic_count, + dg.diagnostic_categories, dg.diagnostic_codes, + dg.primary_diagnostic, + p.phase AS active_phase, p.phase_started_at, + p.event_timestamp AS last_progress_at, + p.received_at AS last_progress_received_at, + p.slot_id, + p.scan_deadline_at, + p.known_reason, p.pending_local_recovery, + CASE WHEN p.phase_started_at IS NOT NULL + THEN GREATEST(0, EXTRACT(EPOCH FROM ( + COALESCE( + r.remote_resolved_at::timestamptz, + CURRENT_TIMESTAMP + ) - p.phase_started_at::timestamptz + ))::BIGINT) END AS phase_age_seconds, + CASE WHEN p.event_timestamp IS NOT NULL + THEN GREATEST(0, EXTRACT(EPOCH FROM ( + COALESCE( + r.remote_resolved_at::timestamptz, + CURRENT_TIMESTAMP + ) - p.event_timestamp::timestamptz + ))::BIGINT) END AS last_progress_age_seconds, + CASE WHEN r.remote_expires_at IS NOT NULL + THEN EXTRACT(EPOCH FROM ( + r.remote_expires_at::timestamptz - COALESCE( + r.remote_resolved_at::timestamptz, + CURRENT_TIMESTAMP + ) + ))::BIGINT END AS assignment_remaining_seconds, + CASE WHEN p.scan_deadline_at IS NOT NULL + THEN EXTRACT(EPOCH FROM ( + p.scan_deadline_at::timestamptz - COALESCE( + r.remote_resolved_at::timestamptz, + CURRENT_TIMESTAMP + ) + ))::BIGINT END AS scan_remaining_seconds, + b.state AS ingestion_state, + pj.status AS projection_state, + r.remote_diagnostic_projection_version AS diagnostic_projection_version, + r.remote_execution_snapshot_json::jsonb #>> + '{{compatibility,protocol_version}}' AS protocol_version, + r.remote_execution_snapshot_json::jsonb #>> + '{{compatibility,bundle_format_version}}' AS bundle_format_version, + r.remote_execution_snapshot_json::jsonb #>> + '{{compatibility,platform_tag}}' AS platform_tag, + r.remote_execution_snapshot_json::jsonb #>> + '{{compatibility,code_manifest_sha256}}' AS code_manifest_sha256, + r.remote_execution_snapshot_json::jsonb #>> + '{{compatibility,detector_policy_sha256}}' AS detector_policy_sha256, + r.remote_execution_snapshot_json::jsonb #>> + '{{compatibility,effective_config_sha256}}' AS effective_config_sha256 + FROM result_reservations r + JOIN remote_worker_users u ON u.id = r.remote_user_id + JOIN remote_worker_devices d ON d.id = r.remote_device_id + JOIN target_queue q ON q.id = r.queue_id + LEFT JOIN result_bundles b ON b.reservation_id = r.id + LEFT JOIN target_scans s ON s.id = b.target_scan_id + LEFT JOIN scan_result_compat src ON src.target_scan_id = s.id + LEFT JOIN projection_jobs pj + ON pj.target_scan_id = s.id AND pj.job_kind = 'scan_event' + LEFT JOIN LATERAL ( + SELECT e.phase, e.phase_started_at, e.event_timestamp, + e.received_at, e.slot_id, + NULLIF(e.event_json::jsonb ->> 'scan_deadline_at', '') + AS scan_deadline_at, + e.event_json::jsonb #>> '{{progress,reason}}' + AS known_reason, + e.event_json::jsonb #>> + '{{progress,pending_local_recovery}}' + AS pending_local_recovery + FROM worker_progress_events e + WHERE e.reservation_id = r.id + ORDER BY e.sequence DESC, e.id DESC LIMIT 1 + ) p ON TRUE + LEFT JOIN LATERAL ( + SELECT COUNT(*) AS diagnostic_count, + STRING_AGG(DISTINCT wd.category, ', ' ORDER BY wd.category) + AS diagnostic_categories, + STRING_AGG(DISTINCT wd.code, ', ' ORDER BY wd.code) + AS diagnostic_codes, + (ARRAY_AGG( + wd.category || '/' || wd.code + ORDER BY CASE wd.category + WHEN 'internal' THEN 0 + WHEN 'storage' THEN 1 + WHEN 'scanner' THEN 2 + WHEN 'timeout' THEN 3 + ELSE 4 END, + wd.occurred_at DESC, wd.id DESC + ))[1] AS primary_diagnostic + FROM worker_diagnostics wd + WHERE wd.reservation_id = r.id + ) dg ON TRUE + WHERE {assignment_where} + ORDER BY r.remote_issued_at DESC, r.id DESC LIMIT ?''', + (*filter_parameters, limit), + ).fetchall() + deferred = self.conn.execute( + '''SELECT id AS queue_id, source, normalized_target AS target, + available_after + FROM target_queue WHERE status = 'deferred' + ORDER BY updated_at DESC, id DESC LIMIT ?''', + (limit,), + ).fetchall() + self.conn.commit() + assignment_rows = [dict(row) for row in assignments] + deferred_rows = [dict(row) for row in deferred] + for row in (*assignment_rows, *deferred_rows): + row['target'] = _admin_safe_target(row.get('target')) + return { + 'filters': filters, + 'users': [dict(row) for row in users], + 'workers': [dict(row) for row in workers], + 'assignments': assignment_rows, + 'deferred_queue': deferred_rows, + } + except Exception: + self.conn.rollback() + raise + + def admin_worker_diagnostic_groups( + self, limit=200, filters=None, *, occurrence_offset=0, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('worker diagnostic administration requires PostgreSQL') + limit = int(limit) + occurrence_offset = int(occurrence_offset) + if not 1 <= limit <= 500 or occurrence_offset < 0: + raise ValueError('worker diagnostic administration limit is out of range') + filters, conditions, parameters = _admin_worker_filter_sql( + filters, diagnostic_alias='wd', + ) + where = ' AND '.join(("r.assignment_kind = 'remote'", *conditions)) + try: + rows = self.conn.execute( + f'''WITH filtered AS ( + SELECT wd.id, wd.diagnostic_uid, wd.reservation_id, + wd.target_scan_id, wd.source, wd.phase, wd.kind, + wd.category, wd.code, wd.summary, wd.retryable, + wd.occurred_at, wd.received_at, wd.envelope_json, + u.user_key, d.device_key, + {_admin_assignment_outcome_sql()} AS assignment_outcome, + {_admin_scan_outcome_sql()} AS scan_outcome, + COALESCE( + NULLIF( + wd.envelope_json::jsonb #>> + '{{exception,fingerprint}}', + '' + ), + 'taxonomy-v1:' || wd.kind || ':' + || wd.category || ':' || wd.code + ) AS fingerprint + FROM worker_diagnostics wd + JOIN result_reservations r ON r.id = wd.reservation_id + JOIN remote_worker_users u ON u.id = r.remote_user_id + JOIN remote_worker_devices d ON d.id = r.remote_device_id + LEFT JOIN result_bundles b ON b.reservation_id = r.id + LEFT JOIN target_scans s ON s.id = b.target_scan_id + WHERE {where} + ), group_stats AS ( + SELECT fingerprint, COUNT(*) AS fingerprint_count, + COUNT(DISTINCT reservation_id) + AS affected_assignment_count + FROM filtered GROUP BY fingerprint + ), counted AS ( + SELECT filtered.*, group_stats.fingerprint_count, + group_stats.affected_assignment_count, + COUNT(*) OVER() AS matched_occurrence_count + FROM filtered + JOIN group_stats USING (fingerprint) + ) + SELECT * FROM counted + ORDER BY occurred_at DESC, id DESC + LIMIT ? OFFSET ?''', + (*parameters, limit + 1, occurrence_offset), + ).fetchall() + self.conn.commit() + except Exception: + self.conn.rollback() + raise + + has_next = len(rows) > limit + page_rows = rows[:limit] + matched_occurrence_count = ( + int(page_rows[0]['matched_occurrence_count']) if page_rows else 0 + ) + grouped = {} + for raw in page_rows: + row = dict(raw) + row.pop('matched_occurrence_count', None) + fingerprint = str(row.pop('fingerprint')) + fingerprint_count = int(row.pop('fingerprint_count')) + affected_assignment_count = int( + row.pop('affected_assignment_count') + ) + try: + envelope = json.loads(str(row.pop('envelope_json'))) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise WorkerObservabilityConflictError( + 'durable worker diagnostic JSON is invalid' + ) from exc + occurrence = { + 'diagnostic_uid': str(row['diagnostic_uid']), + 'reservation_id': int(row['reservation_id']), + 'target_scan_id': ( + int(row['target_scan_id']) + if row['target_scan_id'] is not None else None + ), + 'source': str(row['source']), + 'worker': str(row['device_key']), + 'user': str(row['user_key']), + 'assignment_outcome': str(row['assignment_outcome']), + 'scan_outcome': str(row['scan_outcome']), + 'phase': str(row['phase']), + 'kind': str(row['kind']), + 'category': str(row['category']), + 'code': str(row['code']), + 'summary': str(row['summary']), + 'retryable': bool(row['retryable']), + 'occurred_at': str(row['occurred_at']), + 'received_at': str(row['received_at']), + } + group = grouped.setdefault(fingerprint, { + 'fingerprint': fingerprint, + 'count': fingerprint_count, + 'affected_assignment_count': affected_assignment_count, + 'page_occurrence_count': 0, + 'affected_assignments': [], + 'occurrences': [], + }) + group['page_occurrence_count'] += 1 + group['occurrences'].append(occurrence) + if occurrence['reservation_id'] not in group['affected_assignments']: + group['affected_assignments'].append(occurrence['reservation_id']) + return { + 'filters': filters, + 'matched_occurrence_count': matched_occurrence_count, + 'occurrence_limit': limit, + 'occurrence_offset': occurrence_offset, + 'page_occurrence_count': len(page_rows), + 'has_previous': occurrence_offset > 0, + 'has_next': has_next, + 'previous_occurrence_offset': ( + max(0, occurrence_offset - limit) + if occurrence_offset > 0 else None + ), + 'next_occurrence_offset': ( + occurrence_offset + limit if has_next else None + ), + 'truncated': has_next, + 'groups': list(grouped.values()), + } + + def admin_worker_duration_metrics(self, limit=200, filters=None, *, offset=0): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('worker duration metrics require PostgreSQL') + limit = int(limit) + offset = int(offset) + if not 1 <= limit <= 500 or offset < 0: + raise ValueError('worker duration metrics limit is out of range') + filters = _validated_admin_worker_filters(filters) + conditions = ["r.assignment_kind = 'remote'"] + parameters = [] + if 'source' in filters: + conditions.append('r.source = ?') + parameters.append(filters['source']) + if 'worker' in filters: + conditions.append('d.device_key = ?') + parameters.append(filters['worker']) + if 'assignment_outcome' in filters: + conditions.append(f'({_admin_assignment_outcome_sql()}) = ?') + parameters.append(filters['assignment_outcome']) + if 'scan_outcome' in filters: + conditions.append(f'({_admin_scan_outcome_sql()}) = ?') + parameters.append(filters['scan_outcome']) + diagnostic_conditions = [] + for name in ('category', 'code'): + if name in filters: + diagnostic_conditions.append(f'fd.{name} = ?') + parameters.append(filters[name]) + if 'retryable' in filters: + diagnostic_conditions.append('fd.retryable = ?') + parameters.append(1 if filters['retryable'] else 0) + if diagnostic_conditions: + conditions.append('''EXISTS ( + SELECT 1 FROM worker_diagnostics fd + WHERE fd.reservation_id = r.id AND %s + )''' % ' AND '.join(diagnostic_conditions)) + sample_conditions = ['duration_seconds >= 0'] + sample_parameters = [] + if 'phase' in filters: + sample_conditions.append('phase = ?') + sample_parameters.append(filters['phase']) + if 'since' in filters: + conditions.append( + '(r.remote_resolved_at IS NULL OR r.remote_resolved_at >= ?)' + ) + parameters.append(filters['since']) + sample_conditions.append('completed_at >= ?') + sample_parameters.append(filters['since']) + where = ' AND '.join(conditions) + sample_where = ' AND '.join(sample_conditions) + try: + rows = self.conn.execute( + f'''WITH reservation_scope AS ( + SELECT r.id, r.source, r.remote_issued_at, + r.remote_resolved_at, + CASE WHEN r.remote_resolution_kind = 'bundle_accepted' + THEN COALESCE(s.status, 'unavailable') + ELSE {_admin_assignment_outcome_sql()} END AS outcome + FROM result_reservations r + JOIN remote_worker_devices d ON d.id = r.remote_device_id + LEFT JOIN result_bundles b ON b.reservation_id = r.id + LEFT JOIN target_scans s ON s.id = b.target_scan_id + WHERE {where} + ), ordered_events AS ( + SELECT e.reservation_id, e.phase, e.event_timestamp, + e.phase_started_at, + LAG(e.phase) OVER ( + PARTITION BY e.reservation_id + ORDER BY e.sequence, e.id + ) AS prior_phase, + e.sequence, e.id + FROM worker_progress_events e + JOIN reservation_scope rs ON rs.id = e.reservation_id + ), transitions AS ( + SELECT reservation_id, phase, + COALESCE( + NULLIF(phase_started_at, ''), event_timestamp + ) AS phase_started_at, + sequence, id + FROM ordered_events + WHERE prior_phase IS DISTINCT FROM phase + ), phase_edges AS ( + SELECT t.reservation_id, t.phase, t.phase_started_at, + LEAD(t.phase_started_at) OVER ( + PARTITION BY t.reservation_id + ORDER BY t.sequence, t.id + ) AS next_phase_started_at + FROM transitions t + ), samples AS ( + SELECT rs.source, pe.phase, rs.outcome, + COALESCE( + pe.next_phase_started_at, rs.remote_resolved_at + ) AS completed_at, + EXTRACT(EPOCH FROM ( + COALESCE( + pe.next_phase_started_at, rs.remote_resolved_at + )::timestamptz + - pe.phase_started_at::timestamptz + ))::DOUBLE PRECISION AS duration_seconds + FROM phase_edges pe + JOIN reservation_scope rs ON rs.id = pe.reservation_id + WHERE pe.next_phase_started_at IS NOT NULL + OR rs.remote_resolved_at IS NOT NULL + UNION ALL + SELECT rs.source, 'end_to_end' AS phase, rs.outcome, + rs.remote_resolved_at AS completed_at, + EXTRACT(EPOCH FROM ( + rs.remote_resolved_at::timestamptz + - rs.remote_issued_at::timestamptz + ))::DOUBLE PRECISION AS duration_seconds + FROM reservation_scope rs + WHERE rs.remote_resolved_at IS NOT NULL + AND rs.remote_issued_at IS NOT NULL + ) + , grouped AS ( + SELECT source, phase, outcome, COUNT(*) AS sample_count, + PERCENTILE_CONT(0.50) WITHIN GROUP ( + ORDER BY duration_seconds + ) AS p50_seconds, + PERCENTILE_CONT(0.95) WITHIN GROUP ( + ORDER BY duration_seconds + ) AS p95_seconds, + PERCENTILE_CONT(0.99) WITHIN GROUP ( + ORDER BY duration_seconds + ) AS p99_seconds + FROM samples WHERE {sample_where} + GROUP BY source, phase, outcome + ) + SELECT grouped.*, COUNT(*) OVER() AS total_group_count + FROM grouped + ORDER BY source, phase, outcome LIMIT ? OFFSET ?''', + (*parameters, *sample_parameters, limit + 1, offset), + ).fetchall() + self.conn.commit() + except Exception: + self.conn.rollback() + raise + has_next = len(rows) > limit + page_rows = rows[:limit] + total_group_count = ( + int(page_rows[0]['total_group_count']) if page_rows else 0 + ) + metrics = [] + for raw in page_rows: + row = dict(raw) + count = int(row['sample_count']) + metrics.append({ + 'source': str(row['source']), + 'phase': str(row['phase']), + 'outcome': str(row['outcome']), + 'sample_count': count, + 'sufficient': count >= 5, + 'minimum_sample_count': 5, + 'p50_seconds': float(row['p50_seconds']), + 'p95_seconds': float(row['p95_seconds']), + 'p99_seconds': float(row['p99_seconds']), + }) + return { + 'filters': filters, + 'metrics': metrics, + 'total_group_count': total_group_count, + 'metric_limit': limit, + 'metric_offset': offset, + 'page_group_count': len(metrics), + 'has_previous': offset > 0, + 'has_next': has_next, + 'previous_metric_offset': max(0, offset - limit) if offset > 0 else None, + 'next_metric_offset': offset + limit if has_next else None, + 'truncated': has_next, + } + + def admin_worker_assignment_detail( + self, reservation_id, *, event_limit=500, diagnostic_limit=500, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('worker assignment detail requires PostgreSQL') + reservation_id = int(reservation_id) + event_limit = int(event_limit) + diagnostic_limit = int(diagnostic_limit) + if ( + reservation_id <= 0 + or not 1 <= event_limit <= 2000 + or not 1 <= diagnostic_limit <= 2000 + ): + raise ValueError('worker assignment detail request is out of range') + try: + raw_assignment = self.conn.execute( + f'''SELECT r.*, q.id AS queue_id, q.status AS queue_status, + q.completed_at AS queue_settled_at, + u.user_key, u.active_assignment_cap, d.device_key, + d.last_contact_at, + b.state AS bundle_state, b.ready_at AS bundle_ready_at, + b.committed_at AS bundle_committed_at, + b.acknowledged_at AS bundle_acknowledged_at, + b.actual_bytes AS bundle_bytes, + b.finding_count AS bundle_finding_count, + b.error_count AS bundle_error_count, + s.id AS target_scan_id, s.status AS scan_status, + s.started_at AS scan_started_at, + s.ended_at AS scan_ended_at, + s.duration_sec AS scan_duration_seconds, + s.findings_count, s.verified_findings_count, + s.error_count AS scan_error_count, s.skipped_reason, + s.first_error_summary, s.queue_completion_applied, + s.queue_completion_disposition, + src.metadata_json::jsonb #>> '{{warnings,0}}' + AS scan_warning_summary, + src.metadata_json::jsonb #>> '{{warning_classes,0}}' + AS scan_warning_class, + pj.status AS projection_status, + pj.completed_at AS projection_completed_at, + CURRENT_TIMESTAMP AS authority_at, + { _admin_assignment_outcome_sql() } AS assignment_outcome, + { _admin_scan_outcome_sql() } AS scan_outcome + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + JOIN remote_worker_users u ON u.id = r.remote_user_id + JOIN remote_worker_devices d ON d.id = r.remote_device_id + LEFT JOIN result_bundles b ON b.reservation_id = r.id + LEFT JOIN target_scans s ON s.id = b.target_scan_id + LEFT JOIN scan_result_compat src ON src.target_scan_id = s.id + LEFT JOIN LATERAL ( + SELECT j.status, j.completed_at + FROM projection_jobs j + WHERE j.target_scan_id = s.id + AND j.job_kind = 'scan_event' + ORDER BY j.id DESC LIMIT 1 + ) pj ON TRUE + WHERE r.id = ? AND r.assignment_kind = 'remote' ''', + (reservation_id,), + ).fetchone() + if not raw_assignment: + self.conn.commit() + return None + event_rows = self.conn.execute( + '''SELECT * FROM ( + SELECT e.*, COUNT(*) OVER() AS total_event_count + FROM worker_progress_events e + WHERE e.reservation_id = ? + ORDER BY e.sequence DESC, e.id DESC LIMIT ? + ) recent + ORDER BY sequence, id''', + (reservation_id, event_limit + 1), + ).fetchall() + diagnostic_rows = self.conn.execute( + '''SELECT * FROM worker_diagnostics + WHERE reservation_id = ? + ORDER BY occurred_at, id LIMIT ?''', + (reservation_id, diagnostic_limit + 1), + ).fetchall() + target_scan_id = raw_assignment['target_scan_id'] + error_rows = [] + if target_scan_id is not None: + error_rows = self.conn.execute( + '''SELECT id, category, summary, raw_error, created_at + FROM errors WHERE target_scan_id = ? + ORDER BY id LIMIT ?''', + (int(target_scan_id), diagnostic_limit + 1), + ).fetchall() + self.conn.commit() + except Exception: + self.conn.rollback() + raise + + assignment_row = dict(raw_assignment) + total_event_count = ( + int(event_rows[0]['total_event_count']) if event_rows else 0 + ) + events_truncated = total_event_count > event_limit + diagnostics_truncated = len(diagnostic_rows) > diagnostic_limit + legacy_errors_truncated = len(error_rows) > diagnostic_limit + events = [] + for raw in event_rows[-event_limit:]: + row = dict(raw) + row.pop('total_event_count', None) + events.append(self._worker_observability_row( + row, 'event_json', 'event', + )) + diagnostics = [] + for raw in diagnostic_rows[:diagnostic_limit]: + row = self._worker_observability_row( + raw, 'envelope_json', 'diagnostic', + ) + row['canonical_envelope_json'] = str(row['envelope_json']) + diagnostics.append(row) + legacy_errors = [dict(row) for row in error_rows[:diagnostic_limit]] + + def parsed_json(name): + raw = assignment_row.get(name) + if not raw: + return None + try: + value = json.loads(str(raw)) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise WorkerObservabilityConflictError( + 'durable remote assignment JSON is invalid' + ) from exc + return value + + resolution = parsed_json('remote_resolution_json') + execution_snapshot = parsed_json('remote_execution_snapshot_json') + timeline = [] + + def add_timeline(timestamp, kind, label, **values): + if timestamp: + timeline.append({ + 'timestamp': str(timestamp), 'kind': kind, 'label': label, + **values, + }) + + add_timeline(assignment_row.get('remote_issued_at'), 'assignment', 'issued') + for row in events: + event = row['event'] + add_timeline( + event.get('timestamp') or row.get('event_timestamp'), + 'phase', str(event.get('phase') or row.get('phase') or ''), + sequence=int(row['sequence']), + received_at=str(row['received_at']), + ) + add_timeline(assignment_row.get('bundle_ready_at'), 'transport', 'bundle received') + add_timeline(assignment_row.get('remote_resolved_at'), 'receipt', str( + assignment_row.get('remote_resolution_kind') or 'terminal receipt' + )) + add_timeline(assignment_row.get('bundle_committed_at'), 'ingestion', 'bundle ingested') + add_timeline(assignment_row.get('queue_settled_at'), 'settlement', 'queue settled') + add_timeline( + assignment_row.get('projection_completed_at'), + 'projection', 'projection completed', + ) + timeline.sort(key=lambda item: ( + parse_time(item['timestamp']), item['kind'], item['label'], + )) + + durations = [] + + def duration(label, started, ended, outcome, *, complete=True, authority=None): + if not started or not ended: + return + seconds = max(0.0, (parse_time(ended) - parse_time(started)).total_seconds()) + durations.append({ + 'phase': label, 'duration_seconds': seconds, 'outcome': outcome, + 'complete': bool(complete), 'ended_at': str(ended), + 'authority': authority or ('transition' if complete else 'database_current_time'), + }) + + current_authority = ( + assignment_row.get('remote_resolved_at') + or assignment_row.get('authority_at') + ) + duration( + 'end_to_end', assignment_row.get('remote_issued_at'), + current_authority, + assignment_row['assignment_outcome'], + complete=assignment_row.get('remote_resolved_at') is not None, + authority=( + 'assignment_resolution' + if assignment_row.get('remote_resolved_at') else 'database_current_time' + ), + ) + if assignment_row.get('scan_duration_seconds') is not None: + durations.append({ + 'phase': 'scan_total', + 'duration_seconds': float(assignment_row['scan_duration_seconds']), + 'outcome': assignment_row['scan_outcome'], + 'complete': True, + 'ended_at': assignment_row.get('scan_ended_at'), + 'authority': 'target_scan', + }) + transitions = [] + for row in events: + event = row['event'] + phase = str(event.get('phase') or row.get('phase') or '') + if not transitions or transitions[-1][0] != phase: + transitions.append(( + phase, + str(event.get('phase_started_at') or event.get('timestamp') or ''), + )) + for index, (phase, started) in enumerate(transitions): + final = index == len(transitions) - 1 + ended = current_authority if final else transitions[index + 1][1] + duration( + phase, started, ended, assignment_row['scan_outcome'], + complete=(not final or assignment_row.get('remote_resolved_at') is not None), + authority=( + 'assignment_resolution' if final and assignment_row.get('remote_resolved_at') + else 'database_current_time' if final else 'phase_transition' + ), + ) + + latest_event = events[-1]['event'] if events else None + latest_progress_at = ( + latest_event.get('timestamp') if latest_event else None + ) + phase_started_at = ( + latest_event.get('phase_started_at') if latest_event else None + ) + phase_authority = current_authority if latest_event else None + phase_age_seconds = ( + max(0.0, ( + parse_time(phase_authority) - parse_time(phase_started_at) + ).total_seconds()) + if phase_started_at and phase_authority else None + ) + last_progress_age_seconds = ( + max(0.0, ( + parse_time(phase_authority) - parse_time(latest_progress_at) + ).total_seconds()) + if latest_progress_at and phase_authority else None + ) + + assignment = { + 'reservation_id': int(assignment_row['id']), + 'queue_id': int(assignment_row['queue_id']), + 'source': str(assignment_row['source']), + 'target': _admin_safe_target(assignment_row.get('normalized_target')), + 'user': str(assignment_row['user_key']), + 'worker': str(assignment_row['device_key']), + 'active_assignment_cap': int(assignment_row['active_assignment_cap']), + 'assignment_outcome': str(assignment_row['assignment_outcome']), + 'scan_outcome': str(assignment_row['scan_outcome']), + 'issued_at': assignment_row.get('remote_issued_at'), + 'resolved_at': assignment_row.get('remote_resolved_at'), + 'assignment_code': assignment_row.get('last_error_code'), + 'assignment_detail': assignment_row.get('last_error_detail'), + } + resolution_deadlines = ( + resolution.get('deadlines') if isinstance(resolution, dict) else None + ) + receipt_scan_deadline = ( + resolution_deadlines.get('scan_deadline_at') + if isinstance(resolution_deadlines, dict) else None + ) + deadlines = { + 'assignment_deadline_at': assignment_row.get('remote_expires_at'), + 'scan_deadline_at': ( + receipt_scan_deadline + or (events[-1]['event'].get('scan_deadline_at') if events else None) + ), + 'upload_timeout_seconds': assignment_row.get( + 'remote_result_upload_body_timeout_seconds' + ), + 'upload_timeout_availability': ( + 'persisted at assignment issuance' + if assignment_row.get( + 'remote_result_upload_body_timeout_seconds' + ) is not None else 'legacy/unavailable' + ), + } + compatibility = (execution_snapshot or {}).get('compatibility') or {} + package = { + key: compatibility.get(key) for key in ( + 'protocol_version', 'bundle_format_version', 'platform_tag', + 'code_manifest_sha256', 'detector_policy_sha256', + 'effective_config_sha256', + ) + } + scan = { + 'available': assignment_row.get('target_scan_id') is not None, + 'target_scan_id': assignment_row.get('target_scan_id'), + 'status': assignment_row.get('scan_status'), + 'started_at': assignment_row.get('scan_started_at'), + 'ended_at': assignment_row.get('scan_ended_at'), + 'duration_seconds': assignment_row.get('scan_duration_seconds'), + 'findings_count': assignment_row.get('findings_count'), + 'verified_findings_count': assignment_row.get('verified_findings_count'), + 'error_count': assignment_row.get('scan_error_count'), + 'skipped_reason': assignment_row.get('skipped_reason'), + 'first_error_summary': assignment_row.get('first_error_summary'), + 'warning_summary': assignment_row.get('scan_warning_summary'), + 'warning_class': assignment_row.get('scan_warning_class'), + } + projection_version = assignment_row.get('remote_diagnostic_projection_version') + protocol2 = str(package.get('protocol_version') or '') == '2' + diagnostic_availability = ( + 'current' if diagnostics or projection_version == 1 or protocol2 + else 'legacy/unavailable' + ) + return { + 'schema': 1, + 'assignment': assignment, + 'deadlines': deadlines, + 'package': package, + 'transport': { + 'resolution': resolution, + 'receipt_id': assignment_row.get('remote_receipt_id'), + 'bundle_state': assignment_row.get('bundle_state'), + 'bundle_ready_at': assignment_row.get('bundle_ready_at'), + 'bundle_committed_at': assignment_row.get('bundle_committed_at'), + 'bundle_acknowledged_at': assignment_row.get('bundle_acknowledged_at'), + 'queue_status': assignment_row.get('queue_status'), + 'queue_settled_at': assignment_row.get('queue_settled_at'), + 'projection_status': assignment_row.get('projection_status'), + 'projection_completed_at': assignment_row.get('projection_completed_at'), + }, + 'scan': scan, + 'timeline': timeline, + 'durations': durations, + 'progress': { + 'available': bool(events), + 'availability': ( + 'current' if events + else 'unavailable/no persisted progress' if protocol2 + else 'legacy/unavailable' + ), + 'truncated': events_truncated, + 'total_event_count': total_event_count, + 'omitted_older_event_count': max(0, total_event_count - len(events)), + 'current_phase': ( + str(latest_event.get('phase')) if latest_event else None + ), + 'phase_started_at': phase_started_at, + 'phase_age_seconds': phase_age_seconds, + 'last_progress_at': latest_progress_at, + 'last_progress_age_seconds': last_progress_age_seconds, + 'age_authority': ( + 'assignment_resolution' + if assignment_row.get('remote_resolved_at') else 'database_current_time' + ) if latest_event else None, + 'events': events, + }, + 'diagnostics': { + 'availability': diagnostic_availability, + 'projection_version': projection_version, + 'declared_count': assignment_row.get('remote_diagnostic_count'), + 'truncated': diagnostics_truncated, + 'items': diagnostics, + }, + 'legacy_evidence': { + 'available': bool(legacy_errors) or bool( + assignment_row.get('first_error_summary') + ), + 'explicitly_not_an_envelope': True, + 'truncated': legacy_errors_truncated, + 'errors': legacy_errors, + 'first_error_summary': assignment_row.get('first_error_summary'), + }, + } + + def admin_worker_diagnostic_envelope(self, reservation_id, diagnostic_uid): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('worker diagnostic download requires PostgreSQL') + reservation_id = int(reservation_id) + diagnostic_uid = str(diagnostic_uid or '') + if reservation_id <= 0 or re.fullmatch(r'[a-f0-9]{64}', diagnostic_uid) is None: + raise ValueError('worker diagnostic download identity is invalid') + row = self.conn.execute( + '''SELECT envelope_json, envelope_sha256 FROM worker_diagnostics + WHERE reservation_id = ? AND diagnostic_uid = ?''', + (reservation_id, diagnostic_uid), + ).fetchone() + self.conn.commit() + if not row: + return None + return { + 'canonical_json': str(row['envelope_json']), + 'sha256': str(row['envelope_sha256']), + } + + def admin_requeue_deferred_targets(self, queue_ids, *, max_items=100): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('admin queue requeue requires PostgreSQL') + maximum = min(500, max(1, int(max_items))) + ids = [] + for value in queue_ids or (): + if isinstance(value, bool): + raise ValueError('admin queue IDs must be positive integers') + queue_id = int(value) + if queue_id <= 0 or queue_id > 9223372036854775807: + raise ValueError('admin queue IDs must be positive integers') + ids.append(queue_id) + if not ids or len(ids) > maximum or len(ids) != len(set(ids)): + raise ValueError('admin queue ID selection exceeds its bound') + now = utc_now_iso() + try: + placeholders = ','.join('?' for _ in ids) + cursor = self.conn.execute( + f'''UPDATE target_queue SET status = 'pending', attempts = 0, + available_after = NULL, lease_owner = NULL, lease_token = NULL, + claim_batch = NULL, leased_at = NULL, lease_expires_at = NULL, + last_error = NULL, completed_at = NULL, updated_at = ? + WHERE id IN ({placeholders}) AND status = 'deferred' ''', + (now, *ids), + ) + self.conn.commit() + return int(cursor.rowcount or 0) + except Exception: + self.conn.rollback() + raise + + def admin_discard_queued_source(self, source): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('admin source queue discard requires PostgreSQL') + source = str(source or '').strip().lower() + if re.fullmatch(r'[a-z0-9][a-z0-9_.-]{0,63}', source) is None: + raise ValueError('admin queue source is invalid') + now = utc_now_iso() + try: + cursor = self.conn.execute( + '''UPDATE target_queue SET status = 'quarantined', + completed_at = ?, available_after = NULL, + lease_owner = NULL, lease_token = NULL, + claim_batch = NULL, leased_at = NULL, + lease_expires_at = NULL, resolver_state = 'resolved', + resolver_due_at = NULL, resolver_attempts = 0, + resolver_token = NULL, claim_event_id = NULL, + last_error = ?, + updated_at = ? + WHERE source = ? + AND status IN ('pending', 'deferred', 'cold') + AND current_result_reservation_id IS NULL + AND target_scan_id IS NULL + AND NOT EXISTS ( + SELECT 1 FROM result_reservations reservation + WHERE reservation.queue_id = target_queue.id + ) + AND NOT EXISTS ( + SELECT 1 FROM target_scans scan + WHERE scan.queue_id = target_queue.id + )''', + (now, ADMIN_DISCARDED_QUEUE_REASON, now, source), + ) + self.conn.commit() + return int(cursor.rowcount or 0) + except Exception: + self.conn.rollback() + raise + + def authenticate_remote_worker(self, token_sha256): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote worker authentication requires PostgreSQL') + token_sha256 = str(token_sha256 or '').strip().lower() + if not re.fullmatch(r'[a-f0-9]{64}', token_sha256): + return None + now = utc_now_iso() + try: + row = self.conn.execute( + '''SELECT d.id AS device_id, d.user_id, d.device_key, + u.user_key, u.active_assignment_cap + FROM remote_worker_devices d + JOIN remote_worker_users u ON u.id = d.user_id + WHERE d.token_sha256 = ? AND d.revoked_at IS NULL + AND u.disabled_at IS NULL FOR UPDATE OF d''', + (token_sha256,), + ).fetchone() + if not row: + self.conn.rollback() + return None + self.conn.execute( + 'UPDATE remote_worker_devices SET last_contact_at = ?, updated_at = ? WHERE id = ?', + (now, now, row['device_id']), + ) + self.conn.commit() + return { + 'device_id': int(row['device_id']), 'user_id': int(row['user_id']), + 'device_key': str(row['device_key']), 'user_key': str(row['user_key']), + 'active_assignment_cap': int(row['active_assignment_cap']), + 'token_sha256': token_sha256, + } + except Exception: + self.conn.rollback() + raise + + def set_remote_worker_device_revoked(self, device_key, revoked=True): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote worker revocation requires PostgreSQL') + device_key = str(device_key or '').strip() + if not 1 <= len(device_key) <= 128: + raise ValueError('remote worker device key must be 1..128 characters') + now = utc_now_iso() + cursor = self.conn.execute( + '''UPDATE remote_worker_devices SET revoked_at = ?, updated_at = ? + WHERE device_key = ?''', + (now if revoked else None, now, device_key), + ) + self.conn.commit() + return int(cursor.rowcount or 0) == 1 + + @staticmethod + def _remote_credential_mapping(device_id, token_sha256): + device_id = int(device_id or 0) + token_sha256 = str(token_sha256 or '').strip().lower() + if device_id <= 0 or not re.fullmatch(r'[a-f0-9]{64}', token_sha256): + raise ValueError('remote worker credential identity is invalid') + return device_id, token_sha256 + + def _lock_active_remote_credential(self, device_id, token_sha256, user_id=None): + device_id, token_sha256 = self._remote_credential_mapping( + device_id, token_sha256, + ) + row = self.conn.execute( + '''SELECT d.id AS device_id, d.user_id + FROM remote_worker_devices d + JOIN remote_worker_users u ON u.id = d.user_id + WHERE d.id = ? AND d.token_sha256 = ? AND d.revoked_at IS NULL + AND u.disabled_at IS NULL + FOR SHARE OF d, u''', + (device_id, token_sha256), + ).fetchone() + if row and user_id is not None and int(row['user_id']) != int(user_id): + return None + return row + + @staticmethod + def _worker_observability_row(row, json_column, value_key, replayed=False): + result = dict(row) + try: + result[value_key] = json.loads(str(result[json_column])) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise WorkerObservabilityConflictError( + f'durable worker {value_key} JSON is invalid' + ) from exc + result['replayed'] = bool(replayed) + return result + + def _record_worker_progress_event(self, reservation_id, device_id, event, received_at=None): + reservation_id = int(reservation_id) + device_id = int(device_id) + event, event_json, event_sha256 = _canonical_worker_contract('progress', event) + schema_version = event.get('schema', event.get('schema_version')) + sequence = event.get('sequence') + slot_id = event.get('slot_id') + if ( + isinstance(schema_version, bool) or not isinstance(schema_version, int) + or schema_version <= 0 + or isinstance(sequence, bool) or not isinstance(sequence, int) or sequence <= 0 + or isinstance(slot_id, bool) or not isinstance(slot_id, int) or slot_id < 0 + ): + raise ValueError('worker progress numeric metadata is invalid') + if int(event.get('reservation_id') or 0) != reservation_id: + raise WorkerObservabilityConflictError( + 'worker progress reservation identity does not match the request' + ) + values = { + 'event_type': str(event.get('type') or '').strip(), + 'phase': str(event.get('phase') or '').strip(), + 'event_timestamp': str(event.get('timestamp') or '').strip(), + 'phase_started_at': str(event.get('phase_started_at') or '').strip() or None, + 'instance_id': str(event.get('instance_id') or '').strip(), + 'source': str(event.get('source') or '').strip(), + } + if any(not values[name] for name in ( + 'event_type', 'phase', 'event_timestamp', 'instance_id', 'source', + )): + raise ValueError('worker progress required metadata is incomplete') + if len(event_json.encode('utf-8')) > 8 * 1024: + raise ValueError('worker progress event exceeds its canonical JSON bound') + lock = ' FOR UPDATE OF r, q' if self.conn.is_postgres else '' + reservation = self.conn.execute( + '''SELECT r.id, r.remote_device_id, r.source, r.state, + r.claim_lease_token, r.scan_event_id, r.remote_expires_at, + r.remote_issued_at, r.remote_resolved_at, + r.remote_resolution_kind, q.status AS queue_status, + q.lease_token AS queue_lease_token, + q.current_result_reservation_id, q.claim_event_id + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ? AND r.assignment_kind = 'remote' ''' + lock, + (reservation_id,), + ).fetchone() + now = str(received_at or utc_now_iso()) + if not reservation or ( + int(reservation['remote_device_id'] or 0) != device_id + or str(reservation['source']) != values['source'] + ): + raise WorkerObservabilityConflictError( + 'worker progress does not match the owned reservation' + ) + existing = self.conn.execute( + '''SELECT * FROM worker_progress_events + WHERE reservation_id = ? AND sequence = ?''', + (reservation_id, sequence), + ).fetchone() + if existing: + if ( + int(existing['remote_device_id']) != device_id + or str(existing['event_sha256']) != event_sha256 + or str(existing['event_json']) != event_json + ): + raise WorkerObservabilityConflictError( + 'worker progress sequence already has different canonical content' + ) + return self._worker_observability_row( + existing, 'event_json', 'event', replayed=True, + ) + try: + event_time = datetime.fromisoformat( + str(event['timestamp']).replace('Z', '+00:00') + ).astimezone(timezone.utc) + received_time = datetime.fromisoformat( + now.replace('Z', '+00:00') + ).astimezone(timezone.utc) + issued_time = datetime.fromisoformat( + str(reservation['remote_issued_at']).replace('Z', '+00:00') + ).astimezone(timezone.utc) + except (AttributeError, TypeError, ValueError) as exc: + raise WorkerObservabilityConflictError( + 'worker progress authority timestamps are invalid' + ) from exc + tolerance = timedelta(seconds=REMOTE_PROGRESS_CLOCK_TOLERANCE_SECONDS) + if ( + event_time < issued_time - tolerance + or event_time > received_time + tolerance + ): + raise WorkerObservabilityConflictError( + 'worker progress timestamp is outside its authority window' + ) + resolved = reservation['remote_resolution_kind'] is not None + if resolved: + try: + resolved_time = datetime.fromisoformat( + str(reservation['remote_resolved_at']).replace('Z', '+00:00') + ).astimezone(timezone.utc) + except (AttributeError, TypeError, ValueError) as exc: + raise WorkerObservabilityConflictError( + 'resolved worker progress authority timestamp is invalid' + ) from exc + if received_time > resolved_time + timedelta( + seconds=REMOTE_PROGRESS_RESOLUTION_GRACE_SECONDS + ): + raise WorkerProgressInactiveError( + 'resolved worker progress grace window elapsed' + ) + if event_time > resolved_time + tolerance: + raise WorkerObservabilityConflictError( + 'worker progress timestamp is after terminal resolution' + ) + if event_time < resolved_time - timedelta( + seconds=( + REMOTE_PROGRESS_RESOLUTION_GRACE_SECONDS + + REMOTE_PROGRESS_CLOCK_TOLERANCE_SECONDS + ) + ): + raise WorkerProgressInactiveError( + 'worker progress timestamp is too old for terminal grace' + ) + if not resolved and ( + str(reservation['state']) != 'scanning' + or not reservation['remote_expires_at'] + or str(reservation['remote_expires_at']) <= now + or str(reservation['queue_status']) != 'in_progress' + or str(reservation['queue_lease_token'] or '') + != str(reservation['claim_lease_token'] or '') + or int(reservation['current_result_reservation_id'] or 0) != reservation_id + or str(reservation['claim_event_id'] or '') + != str(reservation['scan_event_id'] or '') + ): + raise WorkerProgressInactiveError( + 'worker progress requires the owned current unresolved reservation' + ) + latest = self.conn.execute( + '''SELECT sequence, event_timestamp FROM worker_progress_events + WHERE reservation_id = ? ORDER BY sequence DESC LIMIT 1''', + (reservation_id,), + ).fetchone() + if latest and sequence <= int(latest['sequence']): + raise WorkerObservabilityConflictError( + 'worker progress sequence must strictly increase' + ) + if latest: + try: + latest_time = datetime.fromisoformat( + str(latest['event_timestamp']).replace('Z', '+00:00') + ).astimezone(timezone.utc) + except (AttributeError, TypeError, ValueError) as exc: + raise WorkerObservabilityConflictError( + 'durable worker progress timestamp is invalid' + ) from exc + if event_time < latest_time - tolerance: + raise WorkerObservabilityConflictError( + 'worker progress timestamp moves backward beyond tolerance' + ) + cursor = self.conn.execute( + '''INSERT INTO worker_progress_events( + reservation_id, remote_device_id, schema_version, sequence, + event_type, phase, event_timestamp, phase_started_at, instance_id, + slot_id, source, event_json, event_sha256, received_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(reservation_id, sequence) DO NOTHING''', + ( + reservation_id, device_id, schema_version, sequence, + values['event_type'], values['phase'], values['event_timestamp'], + values['phase_started_at'], values['instance_id'], slot_id, + values['source'], event_json, event_sha256, now, + ), + ) + stored = self.conn.execute( + '''SELECT * FROM worker_progress_events + WHERE reservation_id = ? AND sequence = ?''', + (reservation_id, sequence), + ).fetchone() + if not stored or ( + int(cursor.rowcount or 0) != 1 + and ( + int(stored['remote_device_id']) != device_id + or str(stored['event_sha256']) != event_sha256 + or str(stored['event_json']) != event_json + ) + ): + raise WorkerObservabilityConflictError( + 'worker progress insert conflicted with different canonical content' + ) + return self._worker_observability_row(stored, 'event_json', 'event') + + def record_worker_progress_event(self, reservation_id, device_id, event): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('worker progress persistence requires PostgreSQL') + try: + result = self._record_worker_progress_event( + reservation_id, device_id, event, + ) + self.conn.commit() + return result + except Exception: + self.conn.rollback() + raise + + def record_remote_worker_progress_event( + self, reservation_id, device_id, token_sha256, event, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote worker progress persistence requires PostgreSQL') + try: + if not self._lock_active_remote_credential(device_id, token_sha256): + raise WorkerObservabilityConflictError( + 'remote worker credential changed before progress acceptance' + ) + result = self._record_worker_progress_event( + reservation_id, device_id, event, + ) + self.conn.commit() + return result + except Exception: + self.conn.rollback() + raise + + def _record_worker_diagnostic( + self, reservation_id, diagnostic, target_scan_id=None, received_at=None, + ): + reservation_id = int(reservation_id) + target_scan_id = int(target_scan_id) if target_scan_id is not None else None + diagnostic, envelope_json, envelope_sha256 = _canonical_worker_contract( + 'diagnostic', diagnostic, + ) + diagnostic_uid = str( + diagnostic.get('diagnostic_uid') + or diagnostic.get('uid') + or diagnostic.get('id') + or '' + ).strip() + schema_version = diagnostic.get('schema', diagnostic.get('schema_version')) + slot_id = diagnostic.get('slot_id') + attempt = diagnostic.get('attempt') + retryable = diagnostic.get('retryable') + scan_event_id = diagnostic.get('scan_event_id') + if not diagnostic_uid or len(diagnostic_uid) > 256 or not diagnostic_uid.isascii(): + raise ValueError('worker diagnostic UID is invalid') + if ( + isinstance(schema_version, bool) or not isinstance(schema_version, int) + or schema_version <= 0 + or isinstance(slot_id, bool) or not isinstance(slot_id, int) or slot_id < 0 + or isinstance(attempt, bool) or not isinstance(attempt, int) or attempt <= 0 + or ( + scan_event_id is not None + and ( + not isinstance(scan_event_id, str) + or re.fullmatch(r'[a-f0-9]{32,64}', scan_event_id) is None + ) + ) + or not isinstance(retryable, bool) + ): + raise ValueError('worker diagnostic identity or numeric metadata is invalid') + if int(diagnostic.get('reservation_id') or 0) != reservation_id: + raise WorkerObservabilityConflictError( + 'worker diagnostic reservation identity does not match the request' + ) + assignment_outcome = diagnostic.get('assignment_outcome') + scan_outcome = diagnostic.get('scan_outcome') + if target_scan_id is not None: + if ( + assignment_outcome != 'accepted' + or scan_outcome not in { + 'clean', 'found', 'degraded', 'error', 'skipped', + } + ): + raise WorkerObservabilityConflictError( + 'accepted bundle diagnostic outcomes are invalid' + ) + elif assignment_outcome == 'accepted': + raise WorkerObservabilityConflictError( + 'accepted diagnostic requires an authoritative target scan' + ) + elif assignment_outcome in {'prebundle_failed', 'expired'} and ( + scan_outcome != 'unavailable' + ): + raise WorkerObservabilityConflictError( + 'terminal diagnostic scan outcome must be unavailable' + ) + values = { + 'source': str(diagnostic.get('source') or '').strip(), + 'phase': str(diagnostic.get('phase') or '').strip(), + 'kind': str(diagnostic.get('kind') or '').strip(), + 'category': str(diagnostic.get('category') or '').strip(), + 'code': str(diagnostic.get('code') or '').strip(), + 'summary': str(diagnostic.get('summary') or '').strip(), + 'occurred_at': str(diagnostic.get('occurred_at') or '').strip(), + 'captured_at': str(diagnostic.get('captured_at') or '').strip(), + } + if any(not value for value in values.values()): + raise ValueError('worker diagnostic required metadata is incomplete') + if len(envelope_json.encode('utf-8')) > 64 * 1024: + raise ValueError('worker diagnostic exceeds its canonical JSON bound') + reservation = self.conn.execute( + '''SELECT id, source, scan_event_id FROM result_reservations + WHERE id = ?''', + (reservation_id,), + ).fetchone() + if not reservation or str(reservation['source']) != values['source']: + raise WorkerObservabilityConflictError( + 'worker diagnostic does not match its reservation' + ) + if ( + scan_event_id is not None + and str(reservation['scan_event_id'] or '') != scan_event_id + ): + raise WorkerObservabilityConflictError( + 'worker diagnostic scan event does not match its remote reservation' + ) + if target_scan_id is not None: + scan = self.conn.execute( + '''SELECT id, result_reservation_id, scan_event_id FROM target_scans + WHERE id = ?''', + (target_scan_id,), + ).fetchone() + if not scan or int(scan['result_reservation_id'] or 0) != reservation_id: + raise WorkerObservabilityConflictError( + 'worker diagnostic target scan does not belong to its reservation' + ) + if scan_event_id is not None and str(scan['scan_event_id'] or '') != scan_event_id: + raise WorkerObservabilityConflictError( + 'worker diagnostic scan event does not match its target scan' + ) + http = diagnostic.get('http') or {} + process = diagnostic.get('process') or {} + body = diagnostic.get('body', diagnostic.get('raw_body')) + if body is None and isinstance(http, dict): + body = http.get('body') + logs = diagnostic.get('logs', diagnostic.get('log')) + if logs is None and isinstance(process, dict): + logs = { + name: process.get(name) for name in ('stdout', 'stderr') + if process.get(name) is not None + } or None + body_json = ( + json.dumps(body, ensure_ascii=True, sort_keys=True, separators=(',', ':')) + if body is not None else None + ) + log_json = ( + json.dumps(logs, ensure_ascii=True, sort_keys=True, separators=(',', ':')) + if logs is not None else None + ) + existing = self.conn.execute( + 'SELECT * FROM worker_diagnostics WHERE diagnostic_uid = ?', + (diagnostic_uid,), + ).fetchone() + if existing: + if ( + int(existing['reservation_id']) != reservation_id + or ( + int(existing['target_scan_id']) + if existing['target_scan_id'] is not None else None + ) != target_scan_id + or str(existing['envelope_sha256']) != envelope_sha256 + or str(existing['envelope_json']) != envelope_json + ): + raise WorkerObservabilityConflictError( + 'worker diagnostic UID already has different canonical content or authority' + ) + return self._worker_observability_row( + existing, 'envelope_json', 'diagnostic', replayed=True, + ) + now = str(received_at or utc_now_iso()) + cursor = self.conn.execute( + '''INSERT INTO worker_diagnostics( + diagnostic_uid, reservation_id, target_scan_id, schema_version, + scan_event_id, slot_id, attempt, source, phase, kind, category, + code, summary, retryable, occurred_at, captured_at, envelope_json, + envelope_sha256, body_payload_json, log_payload_json, received_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(diagnostic_uid) DO NOTHING''', + ( + diagnostic_uid, reservation_id, target_scan_id, schema_version, + scan_event_id, slot_id, attempt, values['source'], + values['phase'], values['kind'], values['category'], values['code'], + values['summary'], int(retryable), values['occurred_at'], + values['captured_at'], envelope_json, envelope_sha256, + body_json, log_json, now, + ), + ) + stored = self.conn.execute( + 'SELECT * FROM worker_diagnostics WHERE diagnostic_uid = ?', + (diagnostic_uid,), + ).fetchone() + if not stored or ( + int(cursor.rowcount or 0) != 1 + and ( + int(stored['reservation_id']) != reservation_id + or ( + int(stored['target_scan_id']) + if stored['target_scan_id'] is not None else None + ) != target_scan_id + or str(stored['envelope_sha256']) != envelope_sha256 + or str(stored['envelope_json']) != envelope_json + ) + ): + raise WorkerObservabilityConflictError( + 'worker diagnostic insert conflicted with different canonical content' + ) + return self._worker_observability_row( + stored, 'envelope_json', 'diagnostic', + ) + + def record_worker_diagnostic( + self, reservation_id, diagnostic, target_scan_id=None, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('worker diagnostic persistence requires PostgreSQL') + try: + result = self._record_worker_diagnostic( + reservation_id, diagnostic, target_scan_id=target_scan_id, + ) + self.conn.commit() + return result + except Exception: + self.conn.rollback() + raise + + def worker_progress_events(self, reservation_id, limit=1000): + if not self.conn: + raise RuntimeError('database connection is unavailable') + limit = int(limit) + if not 1 <= limit <= 10000: + raise ValueError('worker progress query limit is out of range') + rows = self.conn.execute( + '''SELECT e.*, r.remote_user_id, r.assignment_kind, + r.remote_issued_at, r.remote_expires_at, + r.remote_resolution_kind, q.id AS queue_id, + q.status AS queue_status, s.id AS target_scan_id, + s.status AS scan_status, s.started_at AS scan_started_at, + s.ended_at AS scan_ended_at + FROM worker_progress_events e + JOIN result_reservations r ON r.id = e.reservation_id + JOIN target_queue q ON q.id = r.queue_id + LEFT JOIN target_scans s ON s.result_reservation_id = r.id + WHERE e.reservation_id = ? + ORDER BY e.sequence, e.id LIMIT ?''', + (int(reservation_id), limit), + ).fetchall() + self.conn.commit() + return [ + self._worker_observability_row(row, 'event_json', 'event') + for row in rows + ] + + def worker_diagnostics(self, reservation_id, limit=1000): + if not self.conn: + raise RuntimeError('database connection is unavailable') + limit = int(limit) + if not 1 <= limit <= 10000: + raise ValueError('worker diagnostic query limit is out of range') + rows = self.conn.execute( + '''SELECT d.*, r.remote_device_id, r.remote_user_id, r.assignment_kind, + r.remote_issued_at, r.remote_expires_at, + r.remote_resolution_kind, q.id AS queue_id, + q.status AS queue_status, s.status AS scan_status, + s.started_at AS scan_started_at, s.ended_at AS scan_ended_at + FROM worker_diagnostics d + JOIN result_reservations r ON r.id = d.reservation_id + JOIN target_queue q ON q.id = r.queue_id + LEFT JOIN target_scans s ON s.id = d.target_scan_id + WHERE d.reservation_id = ? + ORDER BY d.occurred_at, d.id LIMIT ?''', + (int(reservation_id), limit), + ).fetchall() + self.conn.commit() + return [ + self._worker_observability_row(row, 'envelope_json', 'diagnostic') + for row in rows + ] + + @staticmethod + def _remote_assignment_mapping(value): + if value is None: + return None + value = dict(value) + mapped = { + 'user_id': int(value.get('user_id') or 0), + 'device_id': int(value.get('device_id') or 0), + 'effective_config_sha256': str(value.get('effective_config_sha256') or '').lower(), + 'client_compat_sha256': str(value.get('client_compat_sha256') or '').lower(), + 'token_sha256': str(value.get('token_sha256') or '').lower(), + 'result_upload_body_timeout_seconds': value.get( + 'result_upload_body_timeout_seconds' + ), + } + if mapped['user_id'] <= 0 or mapped['device_id'] <= 0: + raise ValueError('remote assignment requires positive user and device identities') + for key in ('effective_config_sha256', 'client_compat_sha256', 'token_sha256'): + if not re.fullmatch(r'[a-f0-9]{64}', mapped[key]): + raise ValueError(f'remote assignment {key} must be lowercase SHA-256') + timeout = mapped['result_upload_body_timeout_seconds'] + if ( + isinstance(timeout, bool) or not isinstance(timeout, int) + or not 30 <= timeout <= 86400 + ): + raise ValueError('remote assignment upload body timeout is invalid') + snapshot_bytes = canonical_remote_execution_snapshot_bytes( + value.get('execution_snapshot'), + ) + snapshot = json.loads(snapshot_bytes.decode('ascii')) + if ( + snapshot['compatibility']['effective_config_sha256'] + != mapped['effective_config_sha256'] + ): + raise ValueError( + 'remote assignment snapshot effective configuration is invalid' + ) + mapped.update({ + 'execution_snapshot_json': snapshot_bytes.decode('ascii'), + 'execution_snapshot_sha256': hashlib.sha256(snapshot_bytes).hexdigest(), + }) + return mapped + + def ensure_admission_intent( + self, reservation_token, values, remote_user_id=None, remote_device_id=None, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('v2 admission intent requires PostgreSQL') + token = str(reservation_token or '') + if not token: + raise ValueError('admission intent reservation token is required') + digest = self._admission_intent_sha256(values) + now = utc_now_iso() + try: + self.conn.execute( + '''INSERT INTO admission_intents( + reservation_token, intent_sha256, state, remote_user_id, + remote_device_id, created_at, updated_at + ) VALUES (?, ?, 'pending', ?, ?, ?, ?) + ON CONFLICT(reservation_token) DO NOTHING''', + (token, digest, remote_user_id, remote_device_id, now, now), + ) + row = self.conn.execute( + '''SELECT intent_sha256, state, remote_user_id, remote_device_id + FROM admission_intents WHERE reservation_token = ?''', + (token,), + ).fetchone() + if not row or row['intent_sha256'] != digest or ( + str(row['remote_user_id'] or '') != str(remote_user_id or '') + or str(row['remote_device_id'] or '') != str(remote_device_id or '') + ): + raise ScanEventConflictError('admission intent token resolves to conflicting identity') + self.conn.commit() + return dict(row) + except Exception: + self.conn.rollback() + raise + + def admission_intent_resolution(self, reservation_token): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('admission intent resolution requires PostgreSQL') + row = self.conn.execute( + '''SELECT state, resolution_detail FROM admission_intents + WHERE reservation_token = ?''', + (str(reservation_token),), + ).fetchone() + self.conn.commit() + if not row or str(row['state']) != 'aborted': + return None + return str(row['resolution_detail'] or '') or None + + def reserve_and_claim_target( + self, source, platform, producer_identity, supervisor_instance_id, + declared_bundle_bytes, projection_bytes, candidate_items, candidate_bytes, + lease_seconds=3600, max_attempts=3, capacity_limits=None, + reservation_token=None, bundle_id=None, scan_event_id=None, run_id=None, + cycle_id=None, claim_order='oldest', docker_depth_authority=None, + final_cutover=False, remote_assignment=None, reserved_bundle_bytes=None, + remote_max_active=50, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('v2 capacity reservation and target claims require PostgreSQL') + experiment_authority = None + experiment_claim_states = ('active',) + if isinstance(docker_depth_authority, dict) and docker_depth_authority.get( + 'enabled' + ) is False: + self._hold_disabled_docker_depth_experiment(docker_depth_authority) + if isinstance(docker_depth_authority, dict) and docker_depth_authority.get( + 'enabled' + ) is True: + if source != 'dockerhub' or platform != 'docker': + raise ValueError( + 'Docker depth experiment authority is valid only for DockerHub Docker claims' + ) + try: + experiment_authority = self._normalize_docker_depth_resolver_authority( + docker_depth_authority + ) + from docker_depth_experiment import ( + DOCKER_RANK1_BREADTH_SELECTOR_VERSION, + ) + if ( + experiment_authority['selector_version'] + == DOCKER_RANK1_BREADTH_SELECTOR_VERSION + ): + experiment_claim_states = ('resolving', 'active') + except (TypeError, ValueError): + return None + identity = _identity_mapping(producer_identity) + remote = self._remote_assignment_mapping(remote_assignment) + remote_planning_kind = '' + if remote: + remote_snapshot = json.loads(remote['execution_snapshot_json']) + remote_planning_kind = str(remote_snapshot['planning']['kind']) + if remote_planning_kind == 'docker_direct_v1' and ( + source != 'dockerhub' or platform != 'docker' + ): + raise ValueError( + 'Docker direct assignment requires the DockerHub Docker queue' + ) + if remote_planning_kind == 'huggingface_space_v1' and ( + source != 'huggingface' or platform != 'huggingface' + ): + raise ValueError( + 'HuggingFace direct assignment requires the HuggingFace queue' + ) + declared_bundle_bytes = max(1, int(declared_bundle_bytes)) + reserved_bundle_bytes = ( + max(1, int(reserved_bundle_bytes)) + if remote and reserved_bundle_bytes is not None + else declared_bundle_bytes + ) + remote_max_active = int(remote_max_active) + if remote and not 1 <= remote_max_active <= 10000: + raise ValueError('remote active assignment limit is out of range') + projection_bytes = max(1, int(projection_bytes)) + candidate_items = max(0, int(candidate_items)) + candidate_bytes = max(0, int(candidate_bytes)) + limits = { + 'bundle_items': 10000, + 'bundle_bytes': 3 * 1024 * 1024 * 1024, + 'projection_items': 10000, + 'projection_bytes': 2 * 1024 * 1024 * 1024, + 'projection_headroom_bytes': 0, + 'keycheck_items': 100000, + 'keycheck_bytes': 512 * 1024 * 1024, + 'quarantine_items': 10000, + 'quarantine_bytes': 1024 * 1024 * 1024, + } + limits.update({key: int(value) for key, value in (capacity_limits or {}).items() if key in limits}) + if not reservation_token or not bundle_id or not scan_event_id: + raise ValueError('caller must preassign reservation_token, bundle_id, and scan_event_id') + reservation_token = str(reservation_token) + bundle_id = str(bundle_id).lower() + scan_event_id = str(scan_event_id).lower() + if not re.fullmatch(r'[a-f0-9]{32,64}', bundle_id) or not re.fullmatch(r'[a-f0-9]{32,64}', scan_event_id): + raise ValueError('bundle and scan event IDs must be lowercase hexadecimal identities') + claim_order = str(claim_order or 'oldest').strip().lower() + if claim_order not in ('oldest', 'newest', 'balanced'): + raise ValueError('target claim order must be oldest, newest, or balanced') + newest_first = claim_order == 'newest' or ( + claim_order == 'balanced' and int(scan_event_id[-1], 16) % 2 == 0 + ) + claim_direction = 'DESC' if newest_first else 'ASC' + now = utc_now_iso() + owner = ( + f'remote:{remote["device_id"]}' if remote + else f'{source}:{identity["pid"]}:{cycle_id or "cycle"}' + ) + claim_token = secrets.token_urlsafe(32) + claim_batch = reservation_token + ready_relative_path = f'ready/{bundle_id[:2]}/{bundle_id}.trb' + intent_values = { + 'bundle_id': bundle_id, + 'scan_event_id': scan_event_id, + 'source': str(source), + 'platform': str(platform), + 'producer_instance_id': str(supervisor_instance_id or ''), + 'producer_pid': identity['pid'], + 'producer_creation_time': identity['creation_time'], + 'producer_executable': identity['executable'], + 'declared_bundle_bytes': declared_bundle_bytes, + 'reserved_bundle_bytes': reserved_bundle_bytes, + 'reserved_projection_items': 1, + 'reserved_projection_bytes': projection_bytes, + 'reserved_candidate_items': candidate_items, + 'reserved_candidate_bytes': candidate_bytes, + 'run_id': run_id, + 'cycle_id': cycle_id, + 'assignment_kind': 'remote' if remote else 'local', + 'remote_user_id': remote['user_id'] if remote else None, + 'remote_device_id': remote['device_id'] if remote else None, + 'remote_effective_config_sha256': ( + remote['effective_config_sha256'] if remote else None + ), + 'remote_client_compat_sha256': remote['client_compat_sha256'] if remote else None, + 'remote_result_upload_body_timeout_seconds': ( + remote['result_upload_body_timeout_seconds'] if remote else None + ), + 'remote_execution_snapshot_json': ( + remote['execution_snapshot_json'] if remote else None + ), + 'remote_execution_snapshot_sha256': ( + remote['execution_snapshot_sha256'] if remote else None + ), + } + self.ensure_admission_intent( + reservation_token, intent_values, + remote_user_id=remote['user_id'] if remote else None, + remote_device_id=remote['device_id'] if remote else None, + ) + try: + intent = self.conn.execute( + 'SELECT * FROM admission_intents WHERE reservation_token = ? FOR UPDATE', + (reservation_token,), + ).fetchone() + if not intent or intent['intent_sha256'] != self._admission_intent_sha256(intent_values): + raise ScanEventConflictError('admission intent changed before reservation') + if remote and ( + int(intent['remote_user_id'] or 0) != remote['user_id'] + or int(intent['remote_device_id'] or 0) != remote['device_id'] + ): + raise ScanEventConflictError('remote admission intent changed owner identity') + if intent['state'] == 'aborted': + self.conn.commit() + return None + if remote and intent['state'] == 'pending': + control = self._locked_runtime_control_state(shared=True) + if control['effective_dispatch_paused']: + cursor = self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = 'dispatch_gate_closed', + resolved_at = ?, updated_at = ? + WHERE reservation_token = ? AND state = 'pending' ''', + (now, now, reservation_token), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'dispatch gate lost the pending admission intent fence' + ) + self.conn.commit() + return None + experiment = None + if experiment_authority: + experiment, authority_reason = ( + self._locked_docker_depth_experiment_authority( + experiment_authority, final_cutover=final_cutover, now=now, + ) + ) + if not experiment: + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = ?, resolved_at = ?, updated_at = ? + WHERE reservation_token = ?''', + ( + authority_reason or 'experiment_authority_unavailable', + now, now, reservation_token, + ), + ) + self.conn.commit() + return None + if str(experiment['state']) == 'released': + experiment = None + experiment_authority = None + remote_user = None + if remote: + remote_user = self.conn.execute( + '''SELECT * FROM remote_worker_users + WHERE id = ? AND disabled_at IS NULL FOR UPDATE''', + (remote['user_id'],), + ).fetchone() + credential = self._lock_active_remote_credential( + remote['device_id'], remote['token_sha256'], + user_id=remote['user_id'], + ) + if not remote_user or not credential: + raise ScanEventConflictError( + 'remote worker credential is disabled, revoked, or rotated' + ) + existing = self.conn.execute( + '''SELECT r.*, q.attempts, + binding.id AS experiment_binding_id, + binding.experiment_target_id, + binding.attempt AS experiment_attempt, + target.experiment_id, + target.dispatch_wave, target.dispatch_order, + experiment.experiment_key AS bound_experiment_key + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + LEFT JOIN docker_depth_experiment_scan_bindings binding + ON binding.reservation_id = r.id + LEFT JOIN docker_depth_experiment_targets target + ON target.id = binding.experiment_target_id + LEFT JOIN docker_depth_experiments experiment + ON experiment.id = target.experiment_id + WHERE r.reservation_token = ?''', + (reservation_token,), + ).fetchone() + if existing: + expected = { + 'bundle_id': bundle_id, + 'scan_event_id': scan_event_id, + 'source': str(source), + 'platform': str(platform), + 'producer_instance_id': str(supervisor_instance_id or ''), + 'producer_pid': identity['pid'], + 'producer_creation_time': identity['creation_time'], + 'producer_executable': identity['executable'], + 'declared_bundle_bytes': declared_bundle_bytes, + 'reserved_bundle_bytes': reserved_bundle_bytes, + 'reserved_projection_items': 1, + 'reserved_projection_bytes': projection_bytes, + 'reserved_candidate_items': candidate_items, + 'reserved_candidate_bytes': candidate_bytes, + 'run_id': run_id, + 'cycle_id': cycle_id, + 'assignment_kind': 'remote' if remote else 'local', + 'remote_user_id': remote['user_id'] if remote else None, + 'remote_device_id': remote['device_id'] if remote else None, + 'remote_effective_config_sha256': ( + remote['effective_config_sha256'] if remote else None + ), + 'remote_client_compat_sha256': ( + remote['client_compat_sha256'] if remote else None + ), + 'remote_execution_snapshot_json': ( + remote['execution_snapshot_json'] if remote else None + ), + 'remote_execution_snapshot_sha256': ( + remote['execution_snapshot_sha256'] if remote else None + ), + } + if any(str(existing[key]) != str(value) for key, value in expected.items()): + raise ScanEventConflictError('reservation token resolves to conflicting claim identity') + if experiment_authority and ( + existing['experiment_binding_id'] is None + or str(existing['bound_experiment_key']) + != experiment_authority['experiment_key'] + ): + raise ScanEventConflictError( + 'reservation token is not bound to the active Docker depth experiment' + ) + if remote: + remote_assignment_execution_plan(existing) + self.conn.execute( + '''UPDATE admission_intents SET state = 'committed', reservation_id = ?, + resolution_detail = 'existing_reservation', resolved_at = COALESCE(resolved_at, ?), + updated_at = ? WHERE reservation_token = ?''', + (existing['id'], now, now, reservation_token), + ) + self.conn.commit() + return self._result_reservation_claim(existing) + if experiment_authority: + if ( + str(experiment['state']) == 'held' + and str(experiment['hold_reason_code'] or '') + == 'pipeline_capacity_conflict' + ): + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'active', hold_reason_code = NULL, + held_at = NULL, updated_at = ? + WHERE id = ? AND state = 'held' + AND hold_reason_code = 'pipeline_capacity_conflict' ''', + (now, experiment['id']), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth capacity backpressure recovery lost its fence' + ) + experiment = self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ? FOR UPDATE', + (experiment['id'],), + ).fetchone() + if str(experiment['state']) not in experiment_claim_states: + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = ?, resolved_at = ?, updated_at = ? + WHERE reservation_token = ?''', + ( + f"experiment_{experiment['state']}", now, now, + reservation_token, + ), + ) + self.conn.commit() + return None + health = self.conn.execute( + '''SELECT state, lease_expires_at, supervisor_instance_id + FROM pipeline_leases WHERE worker_name = 'result_ingester' ''' + ).fetchone() + if not ( + health and health['state'] == 'ready' + and str(health['lease_expires_at'] or '') > now + and str(health['supervisor_instance_id'] or '') == str(supervisor_instance_id or '') + ): + if experiment: + self._hold_docker_depth_experiment_locked( + experiment, 'ingester_lease_conflict', now, + ) + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = 'ingester_not_ready', resolved_at = ?, updated_at = ? + WHERE reservation_token = ?''', + (now, now, reservation_token), + ) + self.conn.commit() + return None + maximum_attempts = max(0, int(max_attempts or 0)) + if experiment: + row, selection_reason = self._claim_docker_depth_experiment_target_locked( + experiment, experiment_authority, now, maximum_attempts, + ) + if selection_reason: + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = ?, resolved_at = ?, updated_at = ? + WHERE reservation_token = ?''', + (selection_reason, now, now, reservation_token), + ) + self.conn.commit() + return None + if row: + active = self.conn.execute( + '''SELECT * FROM docker_depth_experiments + WHERE id = ? FOR SHARE''', + (experiment['id'],), + ).fetchone() + if active and str(active['state']) not in experiment_claim_states: + active = None + final_reason = self._docker_depth_authority_drift_reason( + active, experiment_authority, + ) if active else None + if not active: + self.conn.rollback() + return None + if final_reason: + self._hold_docker_depth_experiment_locked( + active, final_reason, now, + ) + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = ?, resolved_at = ?, updated_at = ? + WHERE reservation_token = ?''', + (final_reason, now, now, reservation_token), + ) + self.conn.commit() + return None + experiment = active + else: + row = self.conn.execute( + f'''SELECT id, query, target, normalized_target, attempts + FROM target_queue + WHERE source = ? AND platform = ? + AND current_result_reservation_id IS NULL + AND status = 'pending' + AND (available_after IS NULL OR available_after <= ?) + AND (? = 0 OR COALESCE(attempts, 0) < ?) + AND (resolver_state IS NULL OR resolver_state = 'resolved') + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_targets experiment_target + WHERE experiment_target.target_queue_id = target_queue.id + ) + ORDER BY id {claim_direction} + LIMIT 1 FOR UPDATE SKIP LOCKED''', + (source, platform, now, maximum_attempts, maximum_attempts), + ).fetchone() + if not row: + row = self.conn.execute( + f'''SELECT id, query, target, normalized_target, attempts + FROM target_queue + WHERE source = ? AND platform = ? + AND current_result_reservation_id IS NULL + AND status = 'deferred' AND available_after IS NOT NULL + AND available_after <= ? + AND (? = 0 OR COALESCE(attempts, 0) < ?) + AND (resolver_state IS NULL OR resolver_state = 'resolved') + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_targets experiment_target + WHERE experiment_target.target_queue_id = target_queue.id + ) + ORDER BY available_after {claim_direction}, id {claim_direction} + LIMIT 1 FOR UPDATE SKIP LOCKED''', + (source, platform, now, maximum_attempts, maximum_attempts), + ).fetchone() + if not row: + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = 'no_claimable_target', resolved_at = ?, updated_at = ? + WHERE reservation_token = ?''', + (now, now, reservation_token), + ) + self.conn.commit() + if experiment: + self.refresh_docker_depth_experiment_state( + experiment_authority, final_cutover=final_cutover, + ) + return None + if remote: + remote_assignment_execution_plan({ + 'assignment_kind': 'remote', + 'source': source, + 'platform': platform, + 'target': row['target'], + 'normalized_target': row['normalized_target'], + 'remote_effective_config_sha256': remote[ + 'effective_config_sha256' + ], + 'remote_execution_snapshot_json': remote[ + 'execution_snapshot_json' + ], + 'remote_execution_snapshot_sha256': remote[ + 'execution_snapshot_sha256' + ], + 'git_scan_plan_json': None, + 'git_scan_plan_sha256': None, + 'docker_layer_plan_json': None, + 'docker_layer_plan_sha256': None, + }) + # Global accounting is always acquired after the exact workload object. + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + if not capacity: + raise RuntimeSafetySchemaError('pipeline capacity singleton is unavailable') + if remote: + active = self.conn.execute( + '''SELECT COUNT(*) AS user_value, + (SELECT COUNT(*) FROM result_reservations + WHERE assignment_kind = 'remote' + AND remote_resolved_at IS NULL) AS global_value + FROM result_reservations + WHERE assignment_kind = 'remote' AND remote_user_id = ? + AND remote_resolved_at IS NULL''', + (remote['user_id'],), + ).fetchone() + quota_reason = None + if int(active['user_value'] or 0) >= int( + remote_user['active_assignment_cap'] or 0 + ): + quota_reason = 'remote_user_quota_closed' + elif int(active['global_value'] or 0) >= remote_max_active: + quota_reason = 'remote_global_quota_closed' + if quota_reason: + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = ?, resolved_at = ?, updated_at = ? + WHERE reservation_token = ?''', + (quota_reason, now, now, reservation_token), + ) + self.conn.commit() + return None + if ( + int(capacity['quarantine_items'] or 0) >= max(0, limits['quarantine_items']) + or int(capacity['quarantine_bytes'] or 0) >= max(0, limits['quarantine_bytes']) + ): + if experiment: + self._hold_docker_depth_experiment_locked( + experiment, 'quarantine_capacity_conflict', now, + ) + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = 'quarantine_capacity_conflict', + resolved_at = ?, updated_at = ? + WHERE reservation_token = ?''', + (now, now, reservation_token), + ) + self.conn.commit() + return None + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = 'quarantine_admission_closed', + resolved_at = ?, updated_at = ? WHERE reservation_token = ?''', + (now, now, reservation_token), + ) + self.conn.commit() + return None + requested = { + 'bundle_items': 1, + 'bundle_bytes': reserved_bundle_bytes, + 'projection_items': 1, + 'projection_bytes': projection_bytes, + 'keycheck_items': candidate_items, + 'keycheck_bytes': candidate_bytes, + } + projection_ceiling = max( + 0, + limits['projection_bytes'] + - max(0, limits['projection_headroom_bytes']), + ) + capacity_closed = any( + int(capacity[key] or 0) + requested[key] > max(0, limits[key]) + for key in requested if key != 'projection_bytes' + ) or ( + int(capacity['projection_bytes'] or 0) + + requested['projection_bytes'] > projection_ceiling + ) + if capacity_closed: + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = 'pipeline_capacity_closed', + resolved_at = ?, updated_at = ? WHERE reservation_token = ?''', + (now, now, reservation_token), + ) + self.conn.commit() + return None + # Ownership starts only after every contended admission lock is held. + now, lease_until = fixed_lease_window(lease_seconds) + reservation_id = self.conn.insert_returning_id( + '''INSERT INTO result_reservations( + reservation_token, bundle_id, scan_event_id, queue_id, run_id, cycle_id, + source, platform, query, target, normalized_target, + claim_lease_owner, claim_lease_token, claim_batch, + producer_instance_id, producer_pid, producer_creation_time, producer_executable, + assignment_kind, remote_user_id, remote_device_id, remote_issued_at, + remote_expires_at, remote_result_upload_body_timeout_seconds, + remote_effective_config_sha256, + remote_client_compat_sha256, remote_execution_snapshot_json, + remote_execution_snapshot_sha256, + declared_bundle_bytes, reserved_bundle_bytes, + reserved_projection_items, reserved_projection_bytes, + reserved_candidate_items, reserved_candidate_bytes, ready_relative_path, + state, producer_lease_expires_at, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, + ?, ?, 1, ?, ?, ?, ?, + 'scanning', ?, ?, ?)''', + ( + reservation_token, bundle_id, scan_event_id, row['id'], run_id, cycle_id, + source, platform, row['query'], row['target'], row['normalized_target'], + owner, claim_token, claim_batch, str(supervisor_instance_id or ''), + identity['pid'], identity['creation_time'], identity['executable'], + 'remote' if remote else 'local', remote['user_id'] if remote else None, + remote['device_id'] if remote else None, now if remote else None, + lease_until if remote else None, + remote['result_upload_body_timeout_seconds'] if remote else None, + remote['effective_config_sha256'] if remote else None, + remote['client_compat_sha256'] if remote else None, + remote['execution_snapshot_json'] if remote else None, + remote['execution_snapshot_sha256'] if remote else None, + declared_bundle_bytes, reserved_bundle_bytes, projection_bytes, + candidate_items, candidate_bytes, + ready_relative_path, lease_until, now, now, + ), + ) + partial_token = hashlib.sha256(reservation_token.encode('utf-8')).hexdigest()[:24] + for artifact_kind, relative_path, artifact_bytes in ( + ( + 'bundle_partial', + f'tmp/{bundle_id[:2]}/{bundle_id}.{partial_token}.partial', + declared_bundle_bytes, + ), + ('bundle_ready', ready_relative_path, 0), + ): + self.conn.execute( + '''INSERT INTO pipeline_artifacts( + subsystem, artifact_kind, owner_id, owner_key, relative_path, + byte_count, state, created_at, updated_at + ) VALUES ('result_bundle', ?, ?, '', ?, ?, 'expected', ?, ?)''', + ( + artifact_kind, reservation_id, relative_path, + artifact_bytes, now, now, + ), + ) + cursor = self.conn.execute( + '''UPDATE target_queue SET status = 'in_progress', lease_owner = ?, lease_token = ?, + claim_batch = ?, leased_at = ?, lease_expires_at = ?, attempts = COALESCE(attempts, 0) + 1, + current_result_reservation_id = ?, claim_event_id = ?, + scan_remote_modified_at = remote_modified_at, updated_at = ? + WHERE id = ? AND current_result_reservation_id IS NULL + AND status IN ('pending','deferred')''', + ( + owner, claim_token, claim_batch, now, lease_until, reservation_id, + scan_event_id, now, row['id'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError('target ownership changed during capacity reservation') + experiment_binding_id = None + experiment_attempt = None + if experiment: + experiment_attempt = int(row['reservation_count']) + 1 + experiment_binding_id = self.conn.insert_returning_id( + '''INSERT INTO docker_depth_experiment_scan_bindings( + experiment_target_id, reservation_id, attempt, state, + bound_at, created_at + ) VALUES (?, ?, ?, 'reserved', ?, ?)''', + ( + row['experiment_target_id'], reservation_id, + experiment_attempt, now, now, + ), + ) + target_cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_targets + SET state = 'reserved', reservation_count = reservation_count + 1, + terminal_at = NULL, updated_at = ? + WHERE id = ? AND experiment_id = ? AND target_queue_id = ? + AND manifest_id = ? AND state = 'pending' + AND reservation_count = ?''', + ( + now, row['experiment_target_id'], experiment['id'], row['id'], + row['manifest_id'], row['reservation_count'], + ), + ) + if not experiment_binding_id or int(target_cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth binding lost its exact target reservation fence' + ) + self.conn.execute( + '''UPDATE pipeline_capacity SET + bundle_items = bundle_items + 1, + bundle_bytes = bundle_bytes + ?, + projection_items = projection_items + 1, + projection_bytes = projection_bytes + ?, + keycheck_items = keycheck_items + ?, + keycheck_bytes = keycheck_bytes + ?, + updated_at = ? + WHERE id = 1''', + ( + reserved_bundle_bytes, projection_bytes, + candidate_items, candidate_bytes, now, + ), + ) + self.conn.execute( + '''UPDATE admission_intents SET state = 'committed', reservation_id = ?, + resolution_detail = 'reservation_committed', resolved_at = ?, updated_at = ? + WHERE reservation_token = ?''', + (reservation_id, now, now, reservation_token), + ) + self.conn.commit() + created = self.conn.execute( + '''SELECT r.*, q.attempts, + binding.id AS experiment_binding_id, + binding.experiment_target_id, + binding.attempt AS experiment_attempt, + target.experiment_id, + target.dispatch_wave, target.dispatch_order, + experiment.experiment_key AS bound_experiment_key + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + LEFT JOIN docker_depth_experiment_scan_bindings binding + ON binding.reservation_id = r.id + LEFT JOIN docker_depth_experiment_targets target + ON target.id = binding.experiment_target_id + LEFT JOIN docker_depth_experiments experiment + ON experiment.id = target.experiment_id + WHERE r.id = ?''', + (reservation_id,), + ).fetchone() + self.conn.commit() + return self._result_reservation_claim(created) + except Exception: + self.conn.rollback() + raise + + @staticmethod + def _result_reservation_claim(row): + claim = { + 'reservation_id': int(row['id']), + 'reservation_token': str(row['reservation_token']), + 'bundle_id': str(row['bundle_id']), + 'scan_event_id': str(row['scan_event_id']), + 'queue_id': int(row['queue_id']), + 'claim_lease_owner': str(row['claim_lease_owner']), + 'claim_lease_token': str(row['claim_lease_token']), + 'claim_batch': row['claim_batch'], + 'attempts': int(row['attempts'] or 0), + 'declared_bundle_bytes': int(row['declared_bundle_bytes']), + 'ready_relative_path': str(row['ready_relative_path']), + 'source': str(row['source']), + 'platform': str(row['platform']), + 'query': row['query'], + 'target': str(row['target']), + 'normalized_target': str(row['normalized_target']), + 'run_id': row['run_id'], + 'cycle_id': row['cycle_id'], + 'producer_instance_id': str(row['producer_instance_id']), + 'producer_pid': int(row['producer_pid']), + 'producer_creation_time': str(row['producer_creation_time']), + 'producer_executable': str(row['producer_executable']), + 'assignment_kind': str(row['assignment_kind']), + } + if claim['assignment_kind'] == 'remote': + claim.update({ + 'remote_user_id': int(row['remote_user_id']), + 'remote_device_id': int(row['remote_device_id']), + 'remote_issued_at': str(row['remote_issued_at']), + 'remote_expires_at': str(row['remote_expires_at']), + 'remote_result_upload_body_timeout_seconds': ( + int(row['remote_result_upload_body_timeout_seconds']) + if row['remote_result_upload_body_timeout_seconds'] is not None + else None + ), + 'remote_effective_config_sha256': str(row['remote_effective_config_sha256']), + 'remote_client_compat_sha256': str(row['remote_client_compat_sha256']), + }) + if 'experiment_binding_id' in row.keys() and row['experiment_binding_id'] is not None: + claim.update({ + 'experiment_binding_id': int(row['experiment_binding_id']), + 'experiment_target_id': int(row['experiment_target_id']), + 'experiment_attempt': int(row['experiment_attempt']), + 'experiment_id': int(row['experiment_id']), + 'experiment_key': str(row['bound_experiment_key']), + 'dispatch_wave': int(row['dispatch_wave']), + 'dispatch_order': int(row['dispatch_order']), + }) + return claim + + def docker_layer_timeout_canary_eligible(self, reservation_id, claim_lease_token): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('Docker layer canary eligibility requires PostgreSQL') + reservation_id = int(reservation_id) + claim_lease_token = str(claim_lease_token or '') + if not claim_lease_token: + raise ValueError('Docker layer canary eligibility requires a claim lease token') + now = utc_now_iso() + try: + row = self.conn.execute( + '''SELECT q.target, src.metadata_json + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + LEFT JOIN scan_result_compat src ON src.target_scan_id = q.target_scan_id + WHERE r.id = ? AND r.state = 'scanning' + AND r.source = 'dockerhub' AND r.platform = 'docker' + AND r.claim_lease_token = ? + AND r.producer_lease_expires_at > ? + AND q.status = 'in_progress' + AND q.lease_token = ? + AND q.lease_expires_at > ? + AND q.current_result_reservation_id = r.id + AND q.claim_event_id = r.scan_event_id''', + (reservation_id, claim_lease_token, now, claim_lease_token, now), + ).fetchone() + self.conn.commit() + except Exception: + self.conn.rollback() + raise + if not row: + raise ScanEventConflictError('Docker layer canary eligibility fence changed') + text = str(row['metadata_json'] or '') + if not text or len(text.encode('utf-8')) > DOCKER_LAYER_PLAN_MAX_BYTES: + return False + try: + metadata = json.loads(text) + except (TypeError, ValueError, json.JSONDecodeError): + return False + return docker_layer_canary_metadata_eligible(metadata, row['target']) + + def start_docker_adaptive_shadow_report( + self, scan_policy_sha256, execution_policy_sha256, selection_policy_sha256, + cohort_size, lease_owner, lease_seconds=3600, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('Docker adaptive shadow reports require PostgreSQL') + policies = validate_docker_adaptive_policy_hashes( + scan_policy_sha256, execution_policy_sha256, selection_policy_sha256, + ) + if isinstance(cohort_size, bool): + raise ValueError('Docker adaptive shadow cohort size must be an integer') + cohort_size = int(cohort_size) + if not DOCKER_ADAPTIVE_GATE_MIN_CONTROLS <= cohort_size <= DOCKER_ADAPTIVE_GATE_MAX_CONTROLS: + raise ValueError('Docker adaptive shadow cohort size must be between 50 and 100') + lease_owner = str(lease_owner or '').strip() + if not lease_owner or len(lease_owner) > 256 or not lease_owner.isascii(): + raise ValueError('Docker adaptive shadow lease owner is invalid') + if isinstance(lease_seconds, bool): + raise ValueError('Docker adaptive shadow lease duration must be an integer') + lease_seconds = int(lease_seconds) + if not 60 <= lease_seconds <= 86400: + raise ValueError('Docker adaptive shadow lease duration is outside the supported range') + report_token = secrets.token_hex(32) + lease_token = secrets.token_urlsafe(32) + now = utc_now_iso() + lease_expires_at = datetime.fromtimestamp( + time.time() + lease_seconds, timezone.utc, + ).isoformat(timespec='seconds') + lock_identity = ':'.join(policies.values()) + try: + self.conn.execute( + 'SELECT pg_catalog.pg_advisory_xact_lock(' + 'pg_catalog.hashtextextended(?, 1095982177))', + (lock_identity,), + ) + self.conn.execute( + '''UPDATE docker_adaptive_shadow_reports SET + state = 'failed', failure_count = failure_count + 1, + passed = 0, completed_at = ?, updated_at = ? + WHERE scan_policy_sha256 = ? AND execution_policy_sha256 = ? + AND selection_policy_sha256 = ? AND state = 'running' + AND lease_expires_at <= ?''', + ( + now, now, policies['scan_policy_sha256'], + policies['execution_policy_sha256'], + policies['selection_policy_sha256'], now, + ), + ) + active = self.conn.execute( + '''SELECT id FROM docker_adaptive_shadow_reports + WHERE scan_policy_sha256 = ? AND execution_policy_sha256 = ? + AND selection_policy_sha256 = ? AND state = 'running' + LIMIT 1 FOR UPDATE''', + ( + policies['scan_policy_sha256'], policies['execution_policy_sha256'], + policies['selection_policy_sha256'], + ), + ).fetchone() + if active: + raise ScanEventConflictError('Docker adaptive shadow report is already running') + row = self.conn.execute( + '''INSERT INTO docker_adaptive_shadow_reports( + report_token, evaluator_version, state, scan_policy_sha256, + execution_policy_sha256, selection_policy_sha256, cohort_size, + lease_owner, lease_token, lease_expires_at, started_at, + created_at, updated_at + ) VALUES (?, ?, 'running', ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + RETURNING id, report_token, evaluator_version, state, + scan_policy_sha256, execution_policy_sha256, + selection_policy_sha256, cohort_size, lease_owner, + lease_token, lease_expires_at, started_at''', + ( + report_token, DOCKER_ADAPTIVE_SHADOW_EVALUATOR_VERSION, + policies['scan_policy_sha256'], policies['execution_policy_sha256'], + policies['selection_policy_sha256'], cohort_size, lease_owner, + lease_token, lease_expires_at, now, now, now, + ), + ).fetchone() + self.conn.commit() + return dict(row) + except Exception: + self.conn.rollback() + raise + + def docker_adaptive_shadow_controls(self, scan_policy_sha256, cohort_size, selection_salt): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('Docker adaptive shadow controls require PostgreSQL') + scan_policy_sha256 = str(scan_policy_sha256 or '').lower() + selection_salt = str(selection_salt or '').lower() + if not re.fullmatch(r'[a-f0-9]{64}', scan_policy_sha256): + raise ValueError('Docker adaptive shadow scan policy must be a lowercase SHA-256') + if not re.fullmatch(r'[a-f0-9]{64}', selection_salt): + raise ValueError('Docker adaptive shadow selection salt must be a lowercase SHA-256') + if isinstance(cohort_size, bool): + raise ValueError('Docker adaptive shadow cohort size must be an integer') + cohort_size = int(cohort_size) + if not DOCKER_ADAPTIVE_GATE_MIN_CONTROLS <= cohort_size <= DOCKER_ADAPTIVE_GATE_MAX_CONTROLS: + raise ValueError('Docker adaptive shadow cohort size must be between 50 and 100') + try: + rows = self.conn.execute( + '''WITH eligible AS ( + SELECT ts.id AS target_scan_id, ts.normalized_target, + ROW_NUMBER() OVER ( + PARTITION BY ts.normalized_target + ORDER BY ts.ended_at DESC, ts.id DESC + ) AS target_rank + FROM target_scans ts + JOIN target_queue q + ON q.id = ts.queue_id AND q.target_scan_id = ts.id + JOIN result_reservations r + ON r.id = ts.result_reservation_id AND r.queue_id = q.id + JOIN result_bundles b + ON b.reservation_id = r.id AND b.target_scan_id = ts.id + JOIN scan_result_compat src ON src.target_scan_id = ts.id + WHERE ts.source = 'dockerhub' AND ts.scan_type = 'docker' + AND ts.status IN ('clean','found') + AND COALESCE(ts.error_count, 0) = 0 + AND ts.skipped_reason IS NULL AND ts.ended_at IS NOT NULL + AND ts.queue_completion_applied = 1 + AND ts.queue_completion_disposition = 'applied' + AND ts.raw_result_storage = 'normalized_v2' + AND q.source = 'dockerhub' AND q.platform = 'docker' + AND q.status = 'done' + AND r.source = 'dockerhub' AND r.platform = 'docker' + AND r.state = 'acknowledged' AND b.state = 'acknowledged' + AND r.docker_layer_plan_json IS NULL + AND r.docker_layer_plan_sha256 IS NULL + AND ts.normalized_target ~ '@sha256:[a-f0-9]{64}$' + AND (ts.scan_options_json::jsonb ->> 'scan_policy_sha256') = ? + AND (src.metadata_json::jsonb #>> + '{scan_meta,docker_scan_assignment,effective_mode}') = 'full' + ) + SELECT target_scan_id, normalized_target + FROM eligible WHERE target_rank = 1 + ORDER BY md5(? || ':' || normalized_target), target_scan_id + LIMIT ?''', + (scan_policy_sha256, selection_salt, cohort_size), + ).fetchall() + self.conn.commit() + except Exception: + self.conn.rollback() + raise + if len(rows) != cohort_size: + raise RuntimeError('Docker adaptive shadow exact-policy control cohort is incomplete') + controls = [] + for row in rows: + normalized_target = str(row['normalized_target'] or '') + try: + parsed = parse_docker_target(normalized_target) + except (TypeError, ValueError) as exc: + raise RuntimeError('Docker adaptive shadow control target is invalid') from exc + if not re.search(r'@sha256:[a-f0-9]{64}$', str(parsed.get('image') or '')): + raise RuntimeError('Docker adaptive shadow control target is not immutable') + controls.append({ + 'target_scan_id': int(row['target_scan_id']), + 'normalized_target': normalized_target, + }) + return controls + + def docker_adaptive_shadow_control_identities(self, target_scan_id): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('Docker adaptive shadow control identities require PostgreSQL') + target_scan_id = int(target_scan_id) + if target_scan_id <= 0: + raise ValueError('Docker adaptive shadow control scan identity is invalid') + try: + routed_rows = self.conn.execute( + '''SELECT c.routed_service, kc.provider_key_hash + FROM keycheck_candidates c + JOIN keycheck_credentials kc ON kc.id = c.credential_id + WHERE c.target_scan_id = ?''', + (target_scan_id,), + ).fetchall() + detector_rows = self.conn.execute( + '''SELECT detector_secret_hash FROM findings + WHERE target_scan_id = ?''', + (target_scan_id,), + ).fetchall() + self.conn.commit() + except Exception: + self.conn.rollback() + raise + routed = set() + for row in routed_rows: + service = str(row['routed_service'] or '') + provider_key_hash = str(row['provider_key_hash'] or '').lower() + if ( + not re.fullmatch(r'[a-z0-9][a-z0-9_.-]{0,63}', service) + or not re.fullmatch(r'[a-f0-9]{64}', provider_key_hash) + ): + raise RuntimeError('Docker adaptive shadow routed identity snapshot is incomplete') + routed.add((service, provider_key_hash)) + detectors = set() + for row in detector_rows: + detector_secret_hash = str(row['detector_secret_hash'] or '').lower() + if not re.fullmatch(r'[a-f0-9]{64}', detector_secret_hash): + raise RuntimeError('Docker adaptive shadow detector identity snapshot is incomplete') + detectors.add(detector_secret_hash) + return frozenset(routed), frozenset(detectors) + + def finish_docker_adaptive_shadow_report( + self, report_token, lease_owner, lease_token, **values, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('Docker adaptive shadow reports require PostgreSQL') + report_token = str(report_token or '') + lease_owner = str(lease_owner or '') + lease_token = str(lease_token or '') + if not report_token or not lease_owner or not lease_token: + raise ValueError('Docker adaptive shadow report fence is required') + metrics = docker_adaptive_shadow_gate_metrics(values) + selection_metrics = validate_docker_adaptive_shadow_selection_metrics( + values.get('selection_metrics') + ) + if sum( + selection_metrics[name] for name in DOCKER_ADAPTIVE_SHADOW_OMISSION_METRIC_KEYS + ) != metrics['omitted_descriptor_count']: + raise ValueError('Docker adaptive shadow omission metrics do not reconcile') + now = utc_now_iso() + try: + row = self.conn.execute( + '''SELECT * FROM docker_adaptive_shadow_reports + WHERE report_token = ? FOR UPDATE''', + (report_token,), + ).fetchone() + if not row: + raise ScanEventConflictError('Docker adaptive shadow report is absent') + cohort_size = int(row['cohort_size']) + if metrics['completed_pairs'] > cohort_size: + raise ValueError('Docker adaptive completed pairs exceed the cohort size') + if ( + str(row['state']) != 'running' + or str(row['lease_owner']) != lease_owner + or str(row['lease_token']) != lease_token + or str(row['lease_expires_at']) <= now + ): + raise ScanEventConflictError('Docker adaptive shadow report fence changed') + if int(row['sink_checkpoint_count']) != metrics['completed_pairs'] * 2: + raise ValueError('Docker adaptive shadow sink checkpoints do not reconcile') + passed = int( + metrics['completed_pairs'] == cohort_size + and DOCKER_ADAPTIVE_GATE_MIN_CONTROLS <= cohort_size <= DOCKER_ADAPTIVE_GATE_MAX_CONTROLS + and metrics['full_routed_count'] > 0 + and metrics['full_slot_ms'] > 0 + and metrics['adaptive_slot_ms'] > 0 + and metrics['routed_recall_ppm'] >= DOCKER_ADAPTIVE_GATE_ROUTED_RECALL_PPM + and metrics['slot_ratio_ppm'] <= DOCKER_ADAPTIVE_GATE_SLOT_RATIO_PPM + and metrics['failure_count'] == 0 + and metrics['privacy_violation_count'] == 0 + and metrics['safety_regression_count'] == 0 + ) + cursor = self.conn.execute( + '''UPDATE docker_adaptive_shadow_reports SET + state = 'completed', completed_pairs = ?, + full_routed_count = ?, adaptive_routed_count = ?, + routed_intersection_count = ?, full_detector_count = ?, + adaptive_detector_count = ?, detector_intersection_count = ?, + full_slot_ms = ?, adaptive_slot_ms = ?, + omitted_descriptor_count = ?, failure_count = ?, + privacy_violation_count = ?, safety_regression_count = ?, + selection_metrics_json = ?, + routed_recall_ppm = ?, slot_ratio_ppm = ?, passed = ?, + completed_at = ?, updated_at = ? + WHERE id = ? AND state = 'running' AND lease_owner = ? + AND lease_token = ? AND lease_expires_at > ?''', + ( + metrics['completed_pairs'], metrics['full_routed_count'], + metrics['adaptive_routed_count'], metrics['routed_intersection_count'], + metrics['full_detector_count'], metrics['adaptive_detector_count'], + metrics['detector_intersection_count'], metrics['full_slot_ms'], + metrics['adaptive_slot_ms'], metrics['omitted_descriptor_count'], + metrics['failure_count'], metrics['privacy_violation_count'], + metrics['safety_regression_count'], json_dumps(selection_metrics), + metrics['routed_recall_ppm'], + metrics['slot_ratio_ppm'], passed, now, now, int(row['id']), + lease_owner, lease_token, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('Docker adaptive shadow report completion lost its fence') + self.conn.commit() + return { + 'report_id': int(row['id']), + 'passed': bool(passed), + 'cohort_size': cohort_size, + **metrics, + 'selection_metrics': selection_metrics, + 'completed_at': now, + } + except Exception: + self.conn.rollback() + raise + + def checkpoint_docker_adaptive_shadow_report( + self, report_token, lease_owner, lease_token, lease_seconds=3600, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('Docker adaptive shadow reports require PostgreSQL') + if isinstance(lease_seconds, bool): + raise ValueError('Docker adaptive shadow lease duration must be an integer') + lease_seconds = int(lease_seconds) + if not 60 <= lease_seconds <= 86400: + raise ValueError('Docker adaptive shadow lease duration is outside the supported range') + now = utc_now_iso() + lease_expires_at = datetime.fromtimestamp( + time.time() + lease_seconds, timezone.utc, + ).isoformat(timespec='seconds') + try: + row = self.conn.execute( + '''UPDATE docker_adaptive_shadow_reports SET + sink_checkpoint_count = sink_checkpoint_count + 1, + lease_expires_at = ?, updated_at = ? + WHERE report_token = ? AND state = 'running' + AND lease_owner = ? AND lease_token = ? AND lease_expires_at > ? + RETURNING id, sink_checkpoint_count, lease_expires_at''', + ( + lease_expires_at, now, str(report_token or ''), + str(lease_owner or ''), str(lease_token or ''), now, + ), + ).fetchone() + if not row: + raise ScanEventConflictError('Docker adaptive shadow checkpoint lost its fence') + self.conn.commit() + return { + 'report_id': int(row['id']), + 'sink_checkpoint_count': int(row['sink_checkpoint_count']), + 'lease_expires_at': row['lease_expires_at'], + } + except Exception: + self.conn.rollback() + raise + + def fail_docker_adaptive_shadow_report( + self, report_token, lease_owner, lease_token, + privacy_violation_count=0, safety_regression_count=0, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('Docker adaptive shadow reports require PostgreSQL') + privacy_violation_count = int(privacy_violation_count) + safety_regression_count = int(safety_regression_count) + if privacy_violation_count < 0 or safety_regression_count < 0: + raise ValueError('Docker adaptive shadow failure counters must be non-negative') + now = utc_now_iso() + try: + cursor = self.conn.execute( + '''UPDATE docker_adaptive_shadow_reports SET + state = 'failed', failure_count = failure_count + 1, + privacy_violation_count = ?, safety_regression_count = ?, + passed = 0, completed_at = ?, updated_at = ? + WHERE report_token = ? AND state = 'running' + AND lease_owner = ? AND lease_token = ? AND lease_expires_at > ?''', + ( + privacy_violation_count, safety_regression_count, now, now, + str(report_token or ''), str(lease_owner or ''), + str(lease_token or ''), now, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('Docker adaptive shadow report failure lost its fence') + self.conn.commit() + except Exception: + self.conn.rollback() + raise + + def docker_adaptive_gate_report( + self, reservation_id, claim_lease_token, scan_policy_sha256, + execution_policy_sha256, selection_policy_sha256, max_age_sec=604800, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('Docker adaptive gate lookup requires PostgreSQL') + reservation_id = int(reservation_id) + claim_lease_token = str(claim_lease_token or '') + if not claim_lease_token: + raise ValueError('Docker adaptive gate lookup requires a claim lease token') + policies = validate_docker_adaptive_policy_hashes( + scan_policy_sha256, execution_policy_sha256, selection_policy_sha256, + ) + if isinstance(max_age_sec, bool): + raise ValueError('Docker adaptive gate freshness must be an integer') + max_age_sec = int(max_age_sec) + if not 60 <= max_age_sec <= 2592000: + raise ValueError('Docker adaptive gate freshness is outside the supported range') + now = utc_now_iso() + cutoff = (datetime.now(timezone.utc) - timedelta(seconds=max_age_sec)).isoformat( + timespec='seconds', + ) + try: + claim = self.conn.execute( + '''SELECT r.id + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ? AND r.state = 'scanning' + AND r.source = 'dockerhub' AND r.platform = 'docker' + AND r.claim_lease_token = ? AND r.producer_lease_expires_at > ? + AND q.status = 'in_progress' AND q.lease_token = ? + AND q.lease_expires_at > ? + AND q.current_result_reservation_id = r.id + AND q.claim_event_id = r.scan_event_id + FOR UPDATE OF r, q''', + (reservation_id, claim_lease_token, now, claim_lease_token, now), + ).fetchone() + if not claim: + raise ScanEventConflictError('Docker adaptive gate claim fence changed') + row = self.conn.execute( + '''SELECT * FROM docker_adaptive_shadow_reports + WHERE scan_policy_sha256 = ? AND execution_policy_sha256 = ? + AND selection_policy_sha256 = ? + ORDER BY id DESC LIMIT 1''', + ( + policies['scan_policy_sha256'], policies['execution_policy_sha256'], + policies['selection_policy_sha256'], + ), + ).fetchone() + self.conn.commit() + except Exception: + self.conn.rollback() + raise + if not row: + return {'passed': False, 'reason': 'missing'} + result = { + 'report_id': int(row['id']), + 'passed': False, + 'reason': 'failed', + 'completed_at': row['completed_at'], + } + if str(row['state']) != 'completed': + result['reason'] = str(row['state']) + return result + if not row['completed_at'] or str(row['completed_at']) < cutoff: + result['reason'] = 'stale' + return result + metrics = docker_adaptive_shadow_gate_metrics(dict(row)) + try: + selection_metrics = validate_docker_adaptive_shadow_selection_metrics( + row['selection_metrics_json'] + ) + except ValueError: + result['reason'] = 'invalid_evidence' + return result + cohort_size = int(row['cohort_size']) + expected_pass = bool( + str(row['evaluator_version']) == DOCKER_ADAPTIVE_SHADOW_EVALUATOR_VERSION + and DOCKER_ADAPTIVE_GATE_MIN_CONTROLS <= cohort_size <= DOCKER_ADAPTIVE_GATE_MAX_CONTROLS + and metrics['completed_pairs'] == cohort_size + and int(row['sink_checkpoint_count']) == cohort_size * 2 + and sum( + selection_metrics[name] + for name in DOCKER_ADAPTIVE_SHADOW_OMISSION_METRIC_KEYS + ) == metrics['omitted_descriptor_count'] + and metrics['full_routed_count'] > 0 + and metrics['full_slot_ms'] > 0 + and metrics['adaptive_slot_ms'] > 0 + and metrics['routed_recall_ppm'] >= DOCKER_ADAPTIVE_GATE_ROUTED_RECALL_PPM + and metrics['slot_ratio_ppm'] <= DOCKER_ADAPTIVE_GATE_SLOT_RATIO_PPM + and metrics['failure_count'] == 0 + and metrics['privacy_violation_count'] == 0 + and metrics['safety_regression_count'] == 0 + and int(row['recall_threshold_ppm']) == DOCKER_ADAPTIVE_GATE_ROUTED_RECALL_PPM + and int(row['slot_threshold_ppm']) == DOCKER_ADAPTIVE_GATE_SLOT_RATIO_PPM + and int(row['routed_recall_ppm']) == metrics['routed_recall_ppm'] + and int(row['slot_ratio_ppm']) == metrics['slot_ratio_ppm'] + ) + if bool(row['passed']) != expected_pass or not expected_pass: + result['reason'] = 'thresholds' + return result + result.update({ + 'passed': True, + 'reason': 'passed', + 'cohort_size': cohort_size, + 'completed_pairs': metrics['completed_pairs'], + 'routed_recall_ppm': metrics['routed_recall_ppm'], + 'slot_ratio_ppm': metrics['slot_ratio_ppm'], + }) + return result + + def _alias_compatible_legacy_docker_coverage( + self, descriptor, coverage_policy_sha256, scan_policy_sha256, limits, now, + ): + candidates = self.conn.execute( + '''SELECT blob.*, reservation.state AS reservation_state, + reservation.scan_event_id AS reservation_scan_event_id, + reservation.docker_layer_plan_json, + reservation.docker_layer_plan_sha256 + FROM docker_content_blobs blob + JOIN result_reservations reservation + ON reservation.id = blob.covered_reservation_id + WHERE blob.digest = ? AND blob.coverage_policy_sha256 <> ? + AND blob.state = 'covered' + AND blob.covered_policy_sha256 = blob.coverage_policy_sha256 + ORDER BY blob.covered_reservation_id, blob.coverage_policy_sha256 + FOR UPDATE OF blob''', + (descriptor['digest'], coverage_policy_sha256), + ).fetchall() + archive_keys = ( + 'archive_max_size_bytes', 'archive_max_depth', 'archive_timeout_sec', + ) + for candidate in candidates: + try: + plan, plan_sha256 = stored_docker_layer_plan(candidate) + except RuntimeSafetySchemaError: + continue + if ( + not plan + or plan['version'] != 1 + or str(candidate['reservation_state']) not in ('db_committed', 'acknowledged') + or not candidate['covered_reservation_id'] + or not candidate['covered_scan_event_id'] + or str(candidate['covered_scan_event_id']) + != str(candidate['reservation_scan_event_id']) + or not candidate['covered_at'] + or int(candidate['verified_bytes'] or -1) != descriptor['size'] + or str(candidate['descriptor_kind']) != descriptor['kind'] + or int(candidate['declared_bytes']) != descriptor['size'] + or docker_content_media_class( + candidate['descriptor_kind'], candidate['media_type'], + ) != docker_content_media_class(descriptor['kind'], descriptor['media_type']) + or plan['scan_policy_sha256'] != scan_policy_sha256 + or docker_layer_plan_coverage_policy_sha256(plan) + != str(candidate['coverage_policy_sha256']) + or any(plan['limits'][key] != limits[key] for key in archive_keys) + ): + continue + matching_descriptors = [ + item for item in plan['descriptors'] + if item['digest'] == descriptor['digest'] + and item['kind'] == descriptor['kind'] + and item['size'] == descriptor['size'] + and docker_content_media_class(item['kind'], item['media_type']) + == docker_content_media_class(descriptor['kind'], descriptor['media_type']) + and item['selected'] + ] + if not matching_descriptors: + continue + evidence = self.conn.execute( + '''SELECT 1 AS present FROM docker_image_blob_coverage + WHERE reservation_id = ? AND blob_digest = ? + AND coverage_policy_sha256 = ? AND plan_sha256 = ? + AND descriptor_kind = ? AND selected = 1 + AND coverage_state = 'covered' AND covered_at IS NOT NULL + LIMIT 1''', + ( + candidate['covered_reservation_id'], descriptor['digest'], + candidate['coverage_policy_sha256'], plan_sha256, descriptor['kind'], + ), + ).fetchone() + if not evidence: + continue + cursor = self.conn.execute( + '''UPDATE docker_content_blobs SET + state = 'covered', attempts = 0, max_attempts = ?, + available_after = NULL, lease_reservation_id = NULL, + lease_token = NULL, lease_plan_sha256 = NULL, lease_expires_at = NULL, + covered_reservation_id = ?, covered_scan_event_id = ?, + covered_policy_sha256 = ?, verified_bytes = ?, covered_at = ?, + last_error_code = NULL, last_error_detail = NULL, updated_at = ? + WHERE digest = ? AND coverage_policy_sha256 = ? + AND state = 'pending' AND attempts = 0 + AND lease_reservation_id IS NULL AND covered_reservation_id IS NULL''', + ( + limits['blob_max_attempts'], candidate['covered_reservation_id'], + candidate['covered_scan_event_id'], coverage_policy_sha256, + descriptor['size'], candidate['covered_at'], now, + descriptor['digest'], coverage_policy_sha256, + ), + ) + return int(cursor.rowcount or 0) == 1 + return False + + def bind_docker_layer_plan( + self, reservation_id, claim_lease_token, resolved, limits, + scan_policy_sha256, blob_lease_seconds=1800, *, payload_classes=None, + checkpoint=None, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('Docker layer plan binding requires PostgreSQL') + reservation_id = int(reservation_id) + claim_lease_token = str(claim_lease_token or '') + if not claim_lease_token: + raise ValueError('Docker layer plan binding requires a claim lease token') + resolved = validate_docker_layer_resolution(resolved) + limits = validate_docker_layer_limits(limits) + scan_policy_sha256 = str(scan_policy_sha256 or '').strip().lower() + if not re.fullmatch(r'[a-f0-9]{64}', scan_policy_sha256): + raise ValueError('Docker layer scanner policy hash is invalid') + descriptors = [resolved['config']] + list(resolved['layers']) + adaptive = payload_classes is not None or checkpoint is not None + if adaptive: + payload_classes = list(payload_classes or []) + checkpoint = validate_docker_adaptive_checkpoint(checkpoint) + if ( + len(payload_classes) != len(resolved['layers']) + or any( + value not in DOCKER_ADAPTIVE_PAYLOAD_CLASSES - {'config'} + for value in payload_classes + ) + ): + raise ValueError('Docker adaptive payload classes do not match its resolution') + selection_policy_sha256 = docker_layer_selection_policy_sha256(limits) + coverage_policy_sha256 = docker_layer_execution_policy_sha256( + scan_policy_sha256, limits, + ) + else: + policy_bytes = json.dumps( + limits, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + selection_policy_sha256 = hashlib.sha256(policy_bytes).hexdigest() + coverage_policy_sha256 = docker_layer_coverage_policy_sha256( + scan_policy_sha256, limits, + ) + blob_lease_seconds = max( + limits['blob_timeout_sec'] + DOCKER_BLOB_LEASE_MARGIN_SEC, + int(blob_lease_seconds or 1800), + ) + blob_lease_seconds = max(60, min(7200, blob_lease_seconds)) + try: + row = self.conn.execute( + '''SELECT r.*, q.status AS queue_status, q.lease_token AS queue_lease_token, + q.lease_expires_at AS queue_lease_expires_at, + q.current_result_reservation_id AS queue_reservation_id, + q.claim_event_id AS queue_event_id + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ? FOR UPDATE OF r, q''', + (reservation_id,), + ).fetchone() + if not row: + raise ScanEventConflictError('Docker layer reservation is absent') + now, lease_until = fixed_lease_window(blob_lease_seconds) + if ( + str(row['state']) != 'scanning' + or str(row['claim_lease_token']) != claim_lease_token + or str(row['queue_status']) != 'in_progress' + or str(row['queue_lease_token']) != claim_lease_token + or int(row['queue_reservation_id'] or 0) != reservation_id + or str(row['queue_event_id']) != str(row['scan_event_id']) + or str(row['queue_lease_expires_at'] or '') <= now + or str(row['producer_lease_expires_at'] or '') <= now + ): + raise ScanEventConflictError('Docker layer reservation is no longer actively fenced') + if str(row['assignment_kind']) == 'remote': + remote_expires_at = str(row['remote_expires_at'] or '') + if remote_expires_at <= now: + raise ScanEventConflictError('remote Docker layer reservation has expired') + lease_until = remote_expires_at + if str(row['platform']).lower() != 'docker' or str(row['source']).lower() != 'dockerhub': + raise ScanEventConflictError('Docker layer plan conflicts with the reservation source') + if docker_target_identity(row['target']) != resolved['image']: + raise ScanEventConflictError('Docker layer resolution conflicts with the reservation target') + + stored_json = row['docker_layer_plan_json'] + stored_sha256 = row['docker_layer_plan_sha256'] + if stored_json is not None or stored_sha256 is not None: + stored_plan, stored_sha256 = stored_docker_layer_plan(row) + stored_resolution = [ + { + 'digest': descriptor['digest'], + 'size': descriptor['size'], + 'media_type': descriptor['media_type'], + 'kind': descriptor['kind'], + 'position': descriptor['position'], + } + for descriptor in stored_plan['descriptors'] + ] + if ( + stored_plan['image'] != resolved['image'] + or stored_plan['repository'] != resolved['repository'] + or stored_plan['manifest_digest'] != resolved['manifest_digest'] + or stored_plan['platform_os'] != resolved['platform_os'] + or stored_plan['platform_arch'] != resolved['platform_arch'] + or stored_plan['manifest_media_type'] != resolved['manifest_media_type'] + or str(stored_plan.get('scan_policy_sha256') or '') != scan_policy_sha256 + or docker_layer_plan_coverage_policy_sha256(stored_plan) != coverage_policy_sha256 + or stored_plan.get('limits') != limits + or stored_plan.get('selection_policy_sha256') != selection_policy_sha256 + or stored_resolution != descriptors + or (stored_plan['version'] == 2) != adaptive + or ( + adaptive + and ( + stored_plan.get('checkpoint') != checkpoint + or [ + descriptor['payload_class'] + for descriptor in stored_plan['descriptors'][1:] + ] != payload_classes + ) + ) + ): + raise ScanEventConflictError('Docker layer plan replay conflicts with its reservation') + self.conn.commit() + return stored_plan + + for digest in sorted({descriptor['digest'] for descriptor in descriptors}): + self.conn.execute( + 'SELECT pg_catalog.pg_advisory_xact_lock(pg_catalog.hashtextextended(?, 1764436591))', + (digest,), + ) + blob_rows = {} + for descriptor in descriptors: + inserted = self.conn.execute( + '''INSERT INTO docker_content_blobs( + digest, coverage_policy_sha256, descriptor_kind, + declared_bytes, media_type, state, + attempts, max_attempts, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, 'pending', 0, ?, ?, ?) + ON CONFLICT(digest, coverage_policy_sha256) DO NOTHING''', + ( + descriptor['digest'], coverage_policy_sha256, descriptor['kind'], + descriptor['size'], descriptor['media_type'], + limits['blob_max_attempts'], now, now, + ), + ) + blob = self.conn.execute( + '''SELECT * FROM docker_content_blobs + WHERE digest = ? AND coverage_policy_sha256 = ? FOR UPDATE''', + (descriptor['digest'], coverage_policy_sha256), + ).fetchone() + digest_metadata = self.conn.execute( + '''SELECT descriptor_kind, declared_bytes, media_type + FROM docker_content_blobs WHERE digest = ?''', + (descriptor['digest'],), + ).fetchall() + conflicting_metadata = any( + str(metadata['descriptor_kind']) != descriptor['kind'] + or int(metadata['declared_bytes']) != descriptor['size'] + or docker_content_media_class( + metadata['descriptor_kind'], metadata['media_type'], + ) != docker_content_media_class( + descriptor['kind'], descriptor['media_type'], + ) + for metadata in digest_metadata + ) + if not blob or conflicting_metadata or ( + str(blob['descriptor_kind']) != descriptor['kind'] + or int(blob['declared_bytes']) != descriptor['size'] + or docker_content_media_class( + blob['descriptor_kind'], blob['media_type'], + ) != docker_content_media_class( + descriptor['kind'], descriptor['media_type'], + ) + or ( + not adaptive + and int(blob['max_attempts']) != limits['blob_max_attempts'] + ) + ): + raise ScanEventConflictError('Docker content digest resolves to conflicting metadata') + if adaptive and int(inserted.rowcount or 0) == 1: + self._alias_compatible_legacy_docker_coverage( + descriptor, coverage_policy_sha256, scan_policy_sha256, limits, now, + ) + blob = self.conn.execute( + '''SELECT * FROM docker_content_blobs + WHERE digest = ? AND coverage_policy_sha256 = ? FOR UPDATE''', + (descriptor['digest'], coverage_policy_sha256), + ).fetchone() + blob_rows[descriptor['digest']] = blob + + first_selection_rows = self.conn.execute( + '''SELECT coverage.position, coverage.blob_digest, coverage.selected, + coverage.selection_reason + FROM docker_image_blob_coverage coverage + JOIN ( + SELECT position, MIN(reservation_id) AS reservation_id + FROM docker_image_blob_coverage + WHERE queue_id = ? AND manifest_digest = ? + AND ( + (? = 1 AND selection_policy_sha256 = ?) + OR (? = 0 AND coverage_policy_sha256 = ?) + ) + GROUP BY position + ) first_position + ON first_position.position = coverage.position + AND first_position.reservation_id = coverage.reservation_id + WHERE coverage.queue_id = ? AND coverage.manifest_digest = ? + AND ( + (? = 1 AND coverage.selection_policy_sha256 = ?) + OR (? = 0 AND coverage.coverage_policy_sha256 = ?) + ) + ORDER BY coverage.position''', + ( + row['queue_id'], resolved['manifest_digest'], + 1 if adaptive else 0, selection_policy_sha256, + 1 if adaptive else 0, coverage_policy_sha256, + row['queue_id'], resolved['manifest_digest'], + 1 if adaptive else 0, selection_policy_sha256, + 1 if adaptive else 0, coverage_policy_sha256, + ), + ).fetchall() + first_selection = { + int(selection['position']): selection for selection in first_selection_rows + } + if first_selection: + if len(first_selection) != len(descriptors): + raise ScanEventConflictError('Docker image selection baseline is incomplete') + for descriptor in descriptors: + selection = first_selection.get(descriptor['position']) + if not selection or str(selection['blob_digest']) != descriptor['digest']: + raise ScanEventConflictError('Docker image selection baseline changed') + entries = [] + if adaptive: + classes_by_position = { + 0: 'config', + **{ + position: payload_class + for position, payload_class in enumerate(payload_classes, start=1) + }, + } + if first_selection: + entries = [ + ( + descriptor, + bool(first_selection[descriptor['position']]['selected']), + str(first_selection[descriptor['position']]['selection_reason']), + classes_by_position[descriptor['position']], + ) + for descriptor in descriptors + ] + else: + covered_digests = { + digest for digest, blob in blob_rows.items() + if str(blob['state']) == 'covered' + and str(blob['covered_policy_sha256'] or '') == coverage_policy_sha256 + } + entries = select_docker_adaptive_payload( + descriptors, payload_classes, limits, covered_digests, + ) + class_rank = { + payload_class: index + for index, payload_class in enumerate(DOCKER_ADAPTIVE_LAYER_CLASS_ORDER) + } + entries.sort(key=lambda entry: ( + 0 if entry[0]['kind'] == 'config' else 1 + class_rank[entry[3]], + -entry[0]['position'], entry[0]['size'], entry[0]['digest'], + )) + else: + selected_layer_bytes = 0 + selected_layer_count = 0 + selected_layer_digests = set() + config = descriptors[0] + if first_selection: + selection = first_selection[config['position']] + entries.append(( + config, bool(selection['selected']), + str(selection['selection_reason']), None, + )) + elif config['media_type'] not in DOCKER_CONFIG_MEDIA_TYPES: + entries.append((config, False, 'unsupported_media_type', None)) + elif config['size'] > limits['config_max_bytes']: + entries.append((config, False, 'config_too_large', None)) + else: + entries.append((config, True, 'config_selected', None)) + + for descriptor in reversed(descriptors[1:]): + blob = blob_rows[descriptor['digest']] + if first_selection: + selection = first_selection[descriptor['position']] + entries.append(( + descriptor, bool(selection['selected']), + str(selection['selection_reason']), None, + )) + continue + globally_covered = ( + str(blob['state']) == 'covered' + and str(blob['covered_policy_sha256'] or '') == coverage_policy_sha256 + ) + if globally_covered: + entries.append((descriptor, True, 'globally_covered', None)) + elif descriptor['digest'] in selected_layer_digests: + entries.append((descriptor, True, 'duplicate_selected_content', None)) + elif descriptor['media_type'] not in DOCKER_LAYER_SUPPORTED_MEDIA_TYPES: + entries.append((descriptor, False, 'unsupported_media_type', None)) + elif descriptor['size'] > limits['layer_max_bytes']: + entries.append((descriptor, False, 'layer_too_large', None)) + elif selected_layer_count >= limits['max_layers']: + entries.append((descriptor, False, 'layer_limit_exhausted', None)) + elif selected_layer_bytes + descriptor['size'] > limits['image_max_bytes']: + entries.append((descriptor, False, 'image_budget_exhausted', None)) + else: + entries.append((descriptor, True, 'layer_selected', None)) + selected_layer_digests.add(descriptor['digest']) + selected_layer_count += 1 + selected_layer_bytes += descriptor['size'] + planned = [] + pending_leases = [] + lease_tokens_by_digest = {} + leased_bytes = 0 + checkpoint_max_blobs = checkpoint['max_blobs'] if adaptive else 1 + checkpoint_max_bytes = checkpoint['max_bytes'] if adaptive else None + for descriptor, selected, reason, payload_class in entries: + blob = blob_rows[descriptor['digest']] + state = str(blob['state']) + effective_max_attempts = min( + int(blob['max_attempts']), limits['blob_max_attempts'], + ) + active_other = ( + state in ('leased', 'submitted') + and str(blob['lease_expires_at'] or '') > now + and int(blob['lease_reservation_id'] or 0) != reservation_id + ) + same_policy_covered = ( + state == 'covered' + and str(blob['covered_policy_sha256'] or '') == coverage_policy_sha256 + ) + lease_token = None + coverage_state = 'skipped' + if selected and same_policy_covered: + coverage_state = 'covered' + if not adaptive: + reason = 'globally_covered' + elif selected and active_other: + coverage_state = 'shared_pending' + if not adaptive: + reason = 'shared_active' + elif selected and ( + state == 'failed' + or int(blob['attempts'] or 0) >= effective_max_attempts + ): + coverage_state = 'terminal_failed' + if not adaptive: + reason = 'content_attempts_exhausted' + elif selected and str(blob['available_after'] or '') > now: + coverage_state = 'selected' + if not adaptive: + reason = 'content_retry_wait' + elif selected: + lease_token = lease_tokens_by_digest.get(descriptor['digest']) + can_lease = bool(lease_token) or ( + len(lease_tokens_by_digest) < checkpoint_max_blobs + and ( + checkpoint_max_bytes is None + or leased_bytes + descriptor['size'] <= checkpoint_max_bytes + or not lease_tokens_by_digest + ) + ) + if can_lease: + coverage_state = 'leased' + else: + coverage_state = 'selected' + if not adaptive: + reason = 'content_checkpoint_pending' + if can_lease and not lease_token: + lease_token = secrets.token_urlsafe(32) + lease_tokens_by_digest[descriptor['digest']] = lease_token + pending_leases.append((descriptor['digest'], lease_token)) + leased_bytes += descriptor['size'] + planned_descriptor = { + 'digest': descriptor['digest'], + 'size': descriptor['size'], + 'media_type': descriptor['media_type'], + 'kind': descriptor['kind'], + 'position': descriptor['position'], + 'selected': bool(selected), + 'selection_reason': reason, + 'coverage_state': coverage_state, + 'lease_token': lease_token, + 'attempt': ( + int(blob['attempts'] or 0) + 1 + if coverage_state == 'leased' + else min(int(blob['attempts'] or 0), effective_max_attempts) + ), + 'max_attempts': effective_max_attempts, + } + if adaptive: + planned_descriptor['payload_class'] = payload_class + planned.append(planned_descriptor) + + planned.sort(key=lambda item: item['position']) + + plan = { + 'version': 2 if adaptive else 1, + 'image': resolved['image'], + 'repository': resolved['repository'], + 'manifest_digest': resolved['manifest_digest'], + 'platform_os': resolved['platform_os'], + 'platform_arch': resolved['platform_arch'], + 'manifest_media_type': resolved['manifest_media_type'], + 'limits': limits, + 'selection_policy_sha256': selection_policy_sha256, + 'scan_policy_sha256': scan_policy_sha256, + 'descriptors': planned, + } + if adaptive: + plan.update({ + 'selector_version': DOCKER_ADAPTIVE_SELECTOR_VERSION, + 'execution_policy_sha256': coverage_policy_sha256, + 'checkpoint': checkpoint, + }) + plan_bytes = canonical_docker_layer_plan_bytes(plan) + plan_sha256 = hashlib.sha256(plan_bytes).hexdigest() + + for digest, lease_token in pending_leases: + blob = blob_rows[digest] + effective_max = min(int(blob['max_attempts']), limits['blob_max_attempts']) + if ( + str(blob['state']) in ('leased', 'submitted') + and blob['lease_reservation_id'] is not None + and int(blob['lease_reservation_id']) != reservation_id + ): + self.conn.execute( + '''UPDATE docker_image_blob_coverage + SET coverage_state = 'retryable_failed', covered_at = NULL, + last_error_code = 'content_lease_reclaimed', updated_at = ? + WHERE reservation_id = ? AND plan_sha256 = ? + AND blob_digest = ? AND coverage_policy_sha256 = ? + AND coverage_state IN ('leased','shared_pending')''', + ( + now, blob['lease_reservation_id'], blob['lease_plan_sha256'], + digest, coverage_policy_sha256, + ), + ) + cursor = self.conn.execute( + '''UPDATE docker_content_blobs SET + state = 'leased', attempts = attempts + 1, max_attempts = ?, + available_after = NULL, lease_reservation_id = ?, lease_token = ?, + lease_plan_sha256 = ?, lease_expires_at = ?, + covered_reservation_id = NULL, covered_scan_event_id = NULL, + covered_policy_sha256 = NULL, verified_bytes = NULL, covered_at = NULL, + last_error_code = NULL, last_error_detail = NULL, updated_at = ? + WHERE digest = ? AND coverage_policy_sha256 = ? + AND attempts < max_attempts + AND ( + state = 'pending' + OR ( + state = 'leased' + AND lease_expires_at IS NOT NULL AND lease_expires_at <= ? + ) + )''', + ( + effective_max, reservation_id, lease_token, plan_sha256, + lease_until, now, digest, coverage_policy_sha256, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('Docker content lease changed during plan binding') + + for descriptor in planned: + conflicting = self.conn.execute( + '''SELECT 1 AS present FROM docker_image_blob_coverage + WHERE queue_id = ? AND position = ? AND ( + manifest_digest <> ? OR blob_digest <> ? OR descriptor_kind <> ? + ) LIMIT 1 FOR UPDATE''', + ( + row['queue_id'], descriptor['position'], resolved['manifest_digest'], + descriptor['digest'], descriptor['kind'], + ), + ).fetchone() + if conflicting: + raise ScanEventConflictError('Docker image coverage position changed for an immutable target') + self.conn.execute( + '''INSERT INTO docker_image_blob_coverage( + queue_id, manifest_digest, position, blob_digest, + coverage_policy_sha256, selection_policy_sha256, + descriptor_kind, plan_sha256, + reservation_id, selected, selection_reason, + coverage_state, covered_at, last_error_code, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, NULL, ?, ?) + ON CONFLICT(reservation_id, position) DO UPDATE SET + selected = excluded.selected, + selection_reason = excluded.selection_reason, + coverage_state = excluded.coverage_state, + covered_at = excluded.covered_at, + last_error_code = NULL, + updated_at = excluded.updated_at''', + ( + row['queue_id'], resolved['manifest_digest'], descriptor['position'], + descriptor['digest'], coverage_policy_sha256, + selection_policy_sha256, descriptor['kind'], plan_sha256, reservation_id, + 1 if descriptor['selected'] else 0, descriptor['selection_reason'], + descriptor['coverage_state'], + now if descriptor['coverage_state'] == 'covered' else None, + now, now, + ), + ) + + cursor = self.conn.execute( + '''UPDATE result_reservations + SET docker_layer_plan_json = ?, docker_layer_plan_sha256 = ?, + producer_lease_expires_at = CASE + WHEN assignment_kind = 'remote' THEN ? + WHEN producer_lease_expires_at < ? THEN ? + ELSE producer_lease_expires_at + END, + updated_at = ? + WHERE id = ? AND state = 'scanning' + AND claim_lease_token = ? + AND docker_layer_plan_json IS NULL AND docker_layer_plan_sha256 IS NULL''', + ( + plan_bytes.decode('ascii'), plan_sha256, + lease_until, lease_until, lease_until, + now, reservation_id, claim_lease_token, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('Docker layer plan changed during binding') + cursor = self.conn.execute( + '''UPDATE target_queue SET lease_expires_at = CASE + WHEN ? = 'remote' THEN ? + WHEN lease_expires_at < ? THEN ? ELSE lease_expires_at END, + updated_at = ? + WHERE id = ? AND status = 'in_progress' AND lease_token = ? + AND current_result_reservation_id = ? AND claim_event_id = ?''', + ( + str(row['assignment_kind']), lease_until, lease_until, lease_until, + now, row['queue_id'], claim_lease_token, + reservation_id, row['scan_event_id'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('Docker parent lease changed during plan binding') + self.conn.commit() + return plan + except Exception: + self.conn.rollback() + raise + + def _submit_docker_blob_leases_locked(self, reservation, now): + plan, plan_sha256 = stored_docker_layer_plan(reservation) + if plan is None: + return 0 + coverage_policy_sha256 = docker_layer_plan_coverage_policy_sha256(plan) + submitted = 0 + seen = set() + for descriptor in plan['descriptors']: + if descriptor['coverage_state'] != 'leased' or descriptor['digest'] in seen: + continue + seen.add(descriptor['digest']) + cursor = self.conn.execute( + '''UPDATE docker_content_blobs SET state = 'submitted', + updated_at = ? + WHERE digest = ? AND coverage_policy_sha256 = ? + AND state IN ('leased','submitted') + AND lease_reservation_id = ? AND lease_token = ? + AND lease_plan_sha256 = ? + AND lease_expires_at IS NOT NULL AND lease_expires_at > ?''', + ( + now, descriptor['digest'], coverage_policy_sha256, + int(reservation['id']), descriptor['lease_token'], plan_sha256, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker content lease changed before durable bundle submission' + ) + submitted += 1 + return submitted + + def _release_docker_blob_leases_locked( + self, reservation, reason_code, reason_detail, now, *, refund_attempt=False, + ): + plan, plan_sha256 = stored_docker_layer_plan(reservation) + if plan is None: + return 0 + coverage_policy_sha256 = docker_layer_plan_coverage_policy_sha256(plan) + retry_at = ( + datetime.now(timezone.utc) + timedelta(seconds=3600) + ).isoformat(timespec='seconds') + released = 0 + states = {} + for descriptor in plan['descriptors']: + if descriptor['coverage_state'] != 'leased': + continue + digest = descriptor['digest'] + if digest not in states: + blob = self.conn.execute( + '''SELECT * FROM docker_content_blobs + WHERE digest = ? AND coverage_policy_sha256 = ? FOR UPDATE''', + (digest, coverage_policy_sha256), + ).fetchone() + if not blob: + raise ScanEventConflictError( + 'Docker content policy is absent during reservation release' + ) + owns_lease = ( + str(blob['state']) in ('leased', 'submitted') + and int(blob['lease_reservation_id'] or 0) == int(reservation['id']) + and str(blob['lease_token'] or '') == descriptor['lease_token'] + and str(blob['lease_plan_sha256'] or '') == plan_sha256 + ) + if owns_lease: + attempts = max( + 0, + int(blob['attempts'] or 0) - (1 if refund_attempt else 0), + ) + terminal = attempts >= int(blob['max_attempts']) + state = 'failed' if terminal else 'pending' + cursor = self.conn.execute( + '''UPDATE docker_content_blobs SET state = ?, attempts = ?, + available_after = ?, lease_reservation_id = NULL, + lease_token = NULL, lease_plan_sha256 = NULL, + lease_expires_at = NULL, last_error_code = ?, + last_error_detail = ?, updated_at = ? + WHERE digest = ? AND coverage_policy_sha256 = ? + AND state IN ('leased','submitted') + AND lease_reservation_id = ? AND lease_token = ? + AND lease_plan_sha256 = ?''', + ( + state, attempts, None if terminal else retry_at, + str(reason_code)[:64], first_line(reason_detail, 1000), now, + digest, coverage_policy_sha256, int(reservation['id']), + descriptor['lease_token'], plan_sha256, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker content lease changed during reservation release' + ) + states[digest] = { + 'coverage_state': 'terminal_failed' if terminal else 'retryable_failed', + 'covered_at': None, + 'error_code': str(reason_code)[:64], + } + released += 1 + elif ( + blob['lease_reservation_id'] is not None + and int(blob['lease_reservation_id']) == int(reservation['id']) + ): + raise ScanEventConflictError( + 'Docker content lease identity changed within its reservation' + ) + elif ( + str(blob['state']) == 'covered' + and str(blob['covered_policy_sha256'] or '') == coverage_policy_sha256 + ): + states[digest] = { + 'coverage_state': 'covered', + 'covered_at': blob['covered_at'], + 'error_code': None, + } + elif ( + str(blob['state']) == 'failed' + or int(blob['attempts'] or 0) >= int(blob['max_attempts']) + ): + states[digest] = { + 'coverage_state': 'terminal_failed', + 'covered_at': None, + 'error_code': str(blob['last_error_code'] or reason_code)[:64], + } + else: + states[digest] = { + 'coverage_state': 'retryable_failed', + 'covered_at': None, + 'error_code': str(reason_code)[:64], + } + disposition = states[digest] + cursor = self.conn.execute( + '''UPDATE docker_image_blob_coverage SET coverage_state = ?, + last_error_code = ?, covered_at = ?, updated_at = ? + WHERE queue_id = ? AND position = ? AND reservation_id = ? + AND plan_sha256 = ? AND blob_digest = ? + AND coverage_policy_sha256 = ? + AND coverage_state = 'leased' ''', + ( + disposition['coverage_state'], disposition['error_code'], + disposition['covered_at'], now, + int(reservation['queue_id']), descriptor['position'], + int(reservation['id']), plan_sha256, digest, + coverage_policy_sha256, + ), + ) + if int(cursor.rowcount or 0) != 1: + existing = self.conn.execute( + '''SELECT coverage_state FROM docker_image_blob_coverage + WHERE reservation_id = ? AND position = ? AND plan_sha256 = ? + AND blob_digest = ? AND coverage_policy_sha256 = ?''', + ( + int(reservation['id']), descriptor['position'], plan_sha256, + digest, coverage_policy_sha256, + ), + ).fetchone() + if not existing or str(existing['coverage_state']) != disposition['coverage_state']: + raise ScanEventConflictError( + 'Docker image coverage changed during reservation release' + ) + return released + + def _apply_docker_layer_execution_locked( + self, reservation, metadata, queue_status, now, + ): + plan, plan_sha256 = stored_docker_layer_plan(reservation) + metadata_plan = metadata.get('docker_layer_plan') + metadata_execution = metadata.get('docker_layer_execution') + if plan is None: + if metadata_plan is not None or metadata_execution is not None: + raise ScanEventConflictError( + 'unbound reservation contains Docker layer execution metadata' + ) + return None + coverage_policy_sha256 = docker_layer_plan_coverage_policy_sha256(plan) + try: + normalized_metadata_plan = validate_docker_layer_plan(metadata_plan) + metadata_plan_bytes = canonical_docker_layer_plan_bytes(normalized_metadata_plan) + except (TypeError, ValueError) as exc: + raise ScanEventConflictError('Docker layer bundle plan is invalid') from exc + stored_bytes = canonical_docker_layer_plan_bytes(plan) + if ( + metadata_plan_bytes != stored_bytes + or hashlib.sha256(metadata_plan_bytes).hexdigest() != plan_sha256 + ): + raise ScanEventConflictError( + 'Docker layer bundle plan does not match its reservation' + ) + try: + execution = validate_docker_layer_execution( + metadata_execution, plan, plan_sha256, + ) + except (TypeError, ValueError) as exc: + raise ScanEventConflictError('Docker layer bundle execution is invalid') from exc + + retry_at = ( + datetime.now(timezone.utc) + timedelta(seconds=3600) + ).isoformat(timespec='seconds') + plan_by_digest = {} + for descriptor in plan['descriptors']: + if descriptor['coverage_state'] == 'leased': + plan_by_digest.setdefault(descriptor['digest'], descriptor) + for record in execution['blobs']: + descriptor = plan_by_digest[record['digest']] + blob = self.conn.execute( + '''SELECT * FROM docker_content_blobs + WHERE digest = ? AND coverage_policy_sha256 = ? FOR UPDATE''', + (record['digest'], coverage_policy_sha256), + ).fetchone() + if not blob or ( + str(blob['state']) != 'submitted' + or int(blob['lease_reservation_id'] or 0) != int(reservation['id']) + or str(blob['lease_token'] or '') != record['lease_token'] + or str(blob['lease_plan_sha256'] or '') != plan_sha256 + ): + raise ScanEventConflictError( + 'Docker layer execution does not own its submitted content lease' + ) + terminal = ( + record['status'] == 'terminal_failed' + or ( + record['status'] == 'retryable_failed' + and int(blob['attempts']) >= int(blob['max_attempts']) + ) + ) + if record['status'] == 'covered': + blob_state = 'covered' + coverage_state = 'covered' + available_after = None + covered_reservation_id = int(reservation['id']) + covered_scan_event_id = str(reservation['scan_event_id']) + covered_policy_sha256 = coverage_policy_sha256 + covered_at = now + error_code = None + else: + blob_state = 'failed' if terminal else 'pending' + coverage_state = 'terminal_failed' if terminal else 'retryable_failed' + available_after = None if terminal else retry_at + covered_reservation_id = None + covered_scan_event_id = None + covered_policy_sha256 = None + covered_at = None + error_code = record['error_code'] + cursor = self.conn.execute( + '''UPDATE docker_content_blobs SET state = ?, available_after = ?, + lease_reservation_id = NULL, lease_token = NULL, + lease_plan_sha256 = NULL, lease_expires_at = NULL, + covered_reservation_id = ?, covered_scan_event_id = ?, + covered_policy_sha256 = ?, verified_bytes = ?, covered_at = ?, + last_error_code = ?, last_error_detail = NULL, updated_at = ? + WHERE digest = ? AND coverage_policy_sha256 = ? + AND state = 'submitted' + AND lease_reservation_id = ? AND lease_token = ? + AND lease_plan_sha256 = ? + AND lease_expires_at IS NOT NULL AND lease_expires_at > ?''', + ( + blob_state, available_after, covered_reservation_id, + covered_scan_event_id, covered_policy_sha256, + record['verified_bytes'], covered_at, error_code, now, + record['digest'], coverage_policy_sha256, + int(reservation['id']), record['lease_token'], plan_sha256, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker content lease changed during result ingestion' + ) + expected_positions = sum( + 1 for item in plan['descriptors'] + if item['coverage_state'] == 'leased' and item['digest'] == record['digest'] + ) + cursor = self.conn.execute( + '''UPDATE docker_image_blob_coverage SET coverage_state = ?, + covered_at = ?, last_error_code = ?, updated_at = ? + WHERE queue_id = ? AND reservation_id = ? AND plan_sha256 = ? + AND blob_digest = ? AND coverage_policy_sha256 = ? + AND coverage_state = 'leased' ''', + ( + coverage_state, covered_at, error_code, now, + int(reservation['queue_id']), int(reservation['id']), + plan_sha256, record['digest'], coverage_policy_sha256, + ), + ) + if int(cursor.rowcount or 0) != expected_positions: + raise ScanEventConflictError( + 'Docker image coverage changed during result ingestion' + ) + for descriptor in plan['descriptors']: + if descriptor['coverage_state'] != 'shared_pending': + continue + blob = self.conn.execute( + '''SELECT * FROM docker_content_blobs + WHERE digest = ? AND coverage_policy_sha256 = ? FOR UPDATE''', + (descriptor['digest'], coverage_policy_sha256), + ).fetchone() + reconciled = None + if ( + blob and str(blob['state']) == 'covered' + and str(blob['covered_policy_sha256'] or '') == coverage_policy_sha256 + ): + reconciled = 'covered' + elif blob and ( + str(blob['state']) == 'failed' + or int(blob['attempts'] or 0) >= int(blob['max_attempts']) + ): + reconciled = 'terminal_failed' + if reconciled: + cursor = self.conn.execute( + '''UPDATE docker_image_blob_coverage SET coverage_state = ?, + covered_at = ?, last_error_code = ?, updated_at = ? + WHERE queue_id = ? AND position = ? AND reservation_id = ? + AND plan_sha256 = ? AND blob_digest = ? + AND coverage_policy_sha256 = ? + AND coverage_state = 'shared_pending' ''', + ( + reconciled, now if reconciled == 'covered' else None, + None if reconciled == 'covered' else 'content_attempts_exhausted', + now, int(reservation['queue_id']), descriptor['position'], + int(reservation['id']), plan_sha256, descriptor['digest'], + coverage_policy_sha256, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'shared Docker image coverage changed during result ingestion' + ) + + coverage_rows = self.conn.execute( + '''SELECT position, selected, coverage_state, blob_digest + FROM docker_image_blob_coverage + WHERE queue_id = ? AND reservation_id = ? AND plan_sha256 = ? + AND coverage_policy_sha256 = ? + ORDER BY position''', + ( + int(reservation['queue_id']), int(reservation['id']), plan_sha256, + coverage_policy_sha256, + ), + ).fetchall() + if len(coverage_rows) != len(plan['descriptors']): + raise ScanEventConflictError('Docker image coverage row count is incomplete') + states = [str(row['coverage_state']) for row in coverage_rows] + if any(state == 'leased' for state in states): + raise ScanEventConflictError('Docker image coverage retained a consumed lease') + if any(state in ('selected', 'shared_pending', 'retryable_failed') for state in states): + image_state = 'retryable' + expected_queue_status = 'deferred' + elif any(state == 'terminal_failed' for state in states): + image_state = 'terminal_incomplete' + expected_queue_status = 'failed' + elif any(state == 'skipped' for state in states): + image_state = 'partial' + expected_queue_status = 'done' + else: + image_state = 'complete' + expected_queue_status = 'done' + frozen_shared_pending = any( + descriptor['coverage_state'] == 'shared_pending' + for descriptor in plan['descriptors'] + ) + reconciled_shared_completion = ( + queue_status == 'deferred' + and expected_queue_status in ('done', 'failed') + and frozen_shared_pending + ) + if queue_status != expected_queue_status and not reconciled_shared_completion: + raise DockerCoverageDispositionConflictError( + 'Docker image queue disposition conflicts with content coverage' + ) + available_after = metadata.get('available_after') + reset_attempts = metadata.get('reset_attempts') + if queue_status == 'deferred': + deferred_at = parse_time(available_after) + authoritative_now = parse_time(now) + expected_reset = ( + any( + descriptor['coverage_state'] in ('selected', 'shared_pending') + for descriptor in plan['descriptors'] + ) + and not any( + record['status'] in ('retryable_failed', 'terminal_failed') + for record in execution['blobs'] + ) + ) + if ( + reset_attempts is not expected_reset + or deferred_at is None + or authoritative_now is None + or deferred_at <= authoritative_now + ): + raise ScanEventConflictError( + 'Docker retry disposition lacks an authoritative future deadline and reset' + ) + elif available_after is not None or reset_attempts is not False: + raise ScanEventConflictError( + 'terminal Docker image disposition contains retry scheduling metadata' + ) + counts = { + state: sum(1 for item in states if item == state) + for state in ( + 'covered', 'shared_pending', 'retryable_failed', + 'terminal_failed', 'skipped', + ) + } + return {'state': image_state, **counts} + + def bind_git_scan_plan( + self, reservation_id, claim_lease_token, resolved, baseline_depth, + remote_credential=None, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('exact Git scan plan binding requires PostgreSQL') + reservation_id = int(reservation_id) + claim_lease_token = str(claim_lease_token or '') + if not claim_lease_token: + raise ValueError('Git scan plan binding requires a claim lease token') + resolved = validate_git_resolution(resolved) + try: + baseline_depth = int(baseline_depth) + except (TypeError, ValueError) as exc: + raise ValueError('Git baseline depth must be an integer') from exc + if not 1 <= baseline_depth <= 1000000: + raise ValueError('Git baseline depth must be between 1 and 1000000') + + try: + row = self.conn.execute( + '''SELECT r.*, q.status AS queue_status, q.lease_token AS queue_lease_token, + q.lease_expires_at AS queue_lease_expires_at, + q.current_result_reservation_id AS queue_reservation_id, + q.claim_event_id AS queue_event_id, + q.covered_ref AS queue_covered_ref, + q.covered_head AS queue_covered_head + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ? FOR UPDATE OF r, q''', + (reservation_id,), + ).fetchone() + if not row: + raise ScanEventConflictError('Git scan reservation is absent') + if str(row['assignment_kind']) == 'remote': + credential = dict(remote_credential or {}) + if not self._lock_active_remote_credential( + credential.get('device_id'), credential.get('token_sha256'), + user_id=row['remote_user_id'], + ) or int(credential.get('device_id') or 0) != int( + row['remote_device_id'] or 0 + ): + raise ScanEventConflictError( + 'remote worker credential changed before Git plan binding' + ) + now = utc_now_iso() + if ( + str(row['state']) != 'scanning' + or str(row['claim_lease_token']) != claim_lease_token + or str(row['queue_status']) != 'in_progress' + or str(row['queue_lease_token']) != claim_lease_token + or int(row['queue_reservation_id'] or 0) != reservation_id + or str(row['queue_event_id']) != str(row['scan_event_id']) + or str(row['queue_lease_expires_at'] or '') <= now + or str(row['producer_lease_expires_at'] or '') <= now + or ( + str(row['assignment_kind']) == 'remote' + and ( + row['remote_resolution_kind'] is not None + or str(row['remote_expires_at'] or '') <= now + ) + ) + ): + raise ScanEventConflictError('Git scan reservation is no longer actively fenced') + if str(row['platform']).lower() != resolved['provider']: + raise ScanEventConflictError('Git scan resolution provider conflicts with the reservation') + + covered_ref = str(row['queue_covered_ref'] or '') + covered_head = str(row['queue_covered_head'] or '').lower() + if bool(covered_ref) != bool(covered_head): + raise RuntimeSafetySchemaError('Git coverage ref and head must be stored together') + if covered_head and not re.fullmatch(r'[a-f0-9]{40}|[a-f0-9]{64}', covered_head): + raise RuntimeSafetySchemaError('stored Git covered head is invalid') + if covered_ref == resolved['ref'] and covered_head == resolved['head_sha']: + mode = 'noop' + base_sha = covered_head + elif covered_ref == resolved['ref'] and covered_head: + mode = 'delta' + base_sha = covered_head + else: + mode = 'baseline' + base_sha = None + plan = { + 'version': 1, + **resolved, + 'base_sha': base_sha, + 'mode': mode, + 'baseline_depth': baseline_depth, + } + plan_bytes = canonical_git_scan_plan_bytes(plan) + plan_json = plan_bytes.decode('ascii') + plan_sha256 = hashlib.sha256(plan_bytes).hexdigest() + stored_json = row['git_scan_plan_json'] + stored_sha256 = row['git_scan_plan_sha256'] + if stored_json is not None or stored_sha256 is not None: + if str(stored_json or '') != plan_json or str(stored_sha256 or '') != plan_sha256: + raise ScanEventConflictError('Git scan reservation already has a conflicting plan') + self.conn.commit() + return plan + cursor = self.conn.execute( + '''UPDATE result_reservations + SET git_scan_plan_json = ?, git_scan_plan_sha256 = ?, updated_at = ? + WHERE id = ? AND state = 'scanning' AND claim_lease_token = ? + AND git_scan_plan_json IS NULL AND git_scan_plan_sha256 IS NULL''', + (plan_json, plan_sha256, now, reservation_id, claim_lease_token), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('Git scan plan binding lost its reservation fence') + self.conn.commit() + return plan + except Exception: + self.conn.rollback() + raise + + def remote_bound_git_scan_plan( + self, reservation_id, device_id, claim_lease_token, token_sha256, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote Git scan plan recovery requires PostgreSQL') + reservation_id = int(reservation_id) + device_id = int(device_id) + claim_lease_token = str(claim_lease_token or '') + if reservation_id <= 0 or device_id <= 0 or not claim_lease_token: + raise ValueError('remote Git scan plan recovery identity is invalid') + try: + row = self.conn.execute( + '''SELECT r.*, q.status AS queue_status, q.lease_token AS queue_lease_token, + q.lease_expires_at AS queue_lease_expires_at, + q.current_result_reservation_id AS queue_reservation_id, + q.claim_event_id AS queue_event_id + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ? FOR UPDATE OF r, q''', + (reservation_id,), + ).fetchone() + if not row or ( + str(row['assignment_kind']) != 'remote' + or int(row['remote_device_id'] or 0) != device_id + or str(row['state']) != 'scanning' + or row['remote_resolution_kind'] is not None + or str(row['claim_lease_token']) != claim_lease_token + or str(row['queue_status']) != 'in_progress' + or str(row['queue_lease_token']) != claim_lease_token + or int(row['queue_reservation_id'] or 0) != reservation_id + or str(row['queue_event_id']) != str(row['scan_event_id']) + ): + raise ScanEventConflictError('remote Git scan reservation is no longer actively fenced') + if not self._lock_active_remote_credential( + device_id, token_sha256, user_id=row['remote_user_id'], + ): + raise ScanEventConflictError( + 'remote worker credential changed before Git plan recovery' + ) + now = utc_now_iso() + if ( + str(row['remote_expires_at'] or '') <= now + or str(row['queue_lease_expires_at'] or '') <= now + or str(row['producer_lease_expires_at'] or '') <= now + ): + raise ScanEventConflictError( + 'remote Git scan reservation is no longer actively fenced' + ) + plan = stored_git_scan_plan(row) + self.conn.commit() + return plan + except Exception: + self.conn.rollback() + raise + + def result_reservation_by_token(self, reservation_token): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('result reservation recovery requires PostgreSQL') + row = self.conn.execute( + 'SELECT * FROM result_reservations WHERE reservation_token = ?', + (str(reservation_token),), + ).fetchone() + self.conn.commit() + return dict(row) if row else None + + def recover_result_reservation_claim(self, reservation_token, expected): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('result reservation recovery requires PostgreSQL') + token = str(reservation_token) + expected = dict(expected or {}) + digest = self._admission_intent_sha256(expected) + now = utc_now_iso() + try: + intent = self.conn.execute( + 'SELECT * FROM admission_intents WHERE reservation_token = ? FOR UPDATE', + (token,), + ).fetchone() + if not intent: + raise RuntimeError('admission intent is absent during exact recovery') + if intent['intent_sha256'] != digest: + raise ScanEventConflictError( + 'admission intent token resolves to conflicting recovery identity' + ) + if intent['state'] == 'pending': + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = 'serialized_recovery_observed_no_commit', + resolved_at = ?, updated_at = ? WHERE reservation_token = ?''', + (now, now, token), + ) + self.conn.commit() + return None + if intent['state'] == 'aborted': + self.conn.commit() + return None + row = self.conn.execute( + '''SELECT r.*, q.attempts, + q.id AS bound_queue_id, + q.source AS bound_queue_source, + q.platform AS bound_queue_platform, + q.query AS bound_queue_query, + q.target AS bound_queue_target, + q.normalized_target AS bound_queue_normalized_target, + binding.id AS experiment_binding_id, + binding.experiment_target_id, + binding.attempt AS experiment_attempt, + target.experiment_id, + target.dispatch_wave, target.dispatch_order, + experiment.experiment_key AS bound_experiment_key + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + LEFT JOIN docker_depth_experiment_scan_bindings binding + ON binding.reservation_id = r.id + LEFT JOIN docker_depth_experiment_targets target + ON target.id = binding.experiment_target_id + LEFT JOIN docker_depth_experiments experiment + ON experiment.id = target.experiment_id + WHERE r.reservation_token = ? FOR UPDATE OF r, q''', + (token,), + ).fetchone() + if not row or int(intent['reservation_id'] or 0) != int(row['id']): + raise RuntimeError('committed admission intent has no exact reservation') + for key, value in expected.items(): + if str(row[key]) != str(value): + raise ScanEventConflictError( + 'reservation token resolves to conflicting recovered claim identity' + ) + queue_identity = { + 'id': row['bound_queue_id'], + 'source': row['bound_queue_source'], + 'platform': row['bound_queue_platform'], + 'query': row['bound_queue_query'], + 'target': row['bound_queue_target'], + 'normalized_target': row['bound_queue_normalized_target'], + } + self._validate_locked_reservation_queue_identity(row, queue_identity) + if str(row['assignment_kind']) == 'remote': + remote_assignment_execution_plan(row) + self.conn.commit() + return self._result_reservation_claim(row) + except Exception: + self.conn.rollback() + raise + + def reconcile_remote_assignment_request( + self, reservation_token, device_id, token_sha256, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote assignment request reconciliation requires PostgreSQL') + token = str(reservation_token or '') + device_id, token_sha256 = self._remote_credential_mapping( + device_id, token_sha256, + ) + if not re.fullmatch(r'[a-f0-9]{32,64}', token): + raise ValueError('remote assignment request identity is invalid') + now = utc_now_iso() + try: + intent = self.conn.execute( + 'SELECT * FROM admission_intents WHERE reservation_token = ? FOR UPDATE', + (token,), + ).fetchone() + if not intent: + self.conn.commit() + return None + if int(intent['remote_device_id'] or 0) != device_id or not intent['remote_user_id']: + raise ScanEventConflictError( + 'remote assignment request belongs to another device' + ) + if not self._lock_active_remote_credential( + device_id, token_sha256, user_id=int(intent['remote_user_id']), + ): + raise ScanEventConflictError( + 'remote worker credential changed before request reconciliation' + ) + if intent['state'] == 'pending': + self.conn.execute( + '''UPDATE admission_intents SET state = 'aborted', + resolution_detail = 'serialized_remote_recovery_observed_no_commit', + resolved_at = ?, updated_at = ? WHERE reservation_token = ?''', + (now, now, token), + ) + self.conn.commit() + return {'state': 'aborted'} + if intent['state'] == 'aborted': + self.conn.commit() + return {'state': 'aborted'} + if intent['state'] != 'committed': + raise ScanEventConflictError('remote assignment request state is invalid') + row = self.conn.execute( + '''SELECT r.*, q.attempts, + q.id AS bound_queue_id, + q.source AS bound_queue_source, + q.platform AS bound_queue_platform, + q.query AS bound_queue_query, + q.target AS bound_queue_target, + q.normalized_target AS bound_queue_normalized_target + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + WHERE r.reservation_token = ? FOR UPDATE OF r, q''', + (token,), + ).fetchone() + if ( + not row or int(intent['reservation_id'] or 0) != int(row['id']) + or str(row['assignment_kind']) != 'remote' + or int(row['remote_user_id'] or 0) != int(intent['remote_user_id']) + or int(row['remote_device_id'] or 0) != device_id + ): + raise ScanEventConflictError( + 'committed remote assignment request lost its reservation identity' + ) + queue_identity = { + 'id': row['bound_queue_id'], + 'source': row['bound_queue_source'], + 'platform': row['bound_queue_platform'], + 'query': row['bound_queue_query'], + 'target': row['bound_queue_target'], + 'normalized_target': row['bound_queue_normalized_target'], + } + self._validate_locked_reservation_queue_identity(row, queue_identity) + snapshot = stored_remote_execution_snapshot(row) + execution_plan = remote_assignment_execution_plan(row, snapshot=snapshot) + receipt = ( + self._remote_resolution_from_row(row) + if row['remote_resolution_json'] else None + ) + result = { + 'state': 'committed', + 'claim': self._result_reservation_claim(row), + 'execution_snapshot': snapshot, + 'execution_snapshot_sha256': str( + row['remote_execution_snapshot_sha256'] + ), + 'execution_plan': execution_plan, + 'git_plan': ( + execution_plan['bound_plan'] + if execution_plan['kind'] == 'exact_git_v1' else None + ), + 'receipt': receipt, + } + self.conn.commit() + return result + except Exception: + self.conn.rollback() + raise + + def renew_result_claim(self, reservation_id, lease_token, lease_seconds=3600): + if not self.conn or not self.conn.is_postgres: + return False + now = utc_now_iso() + try: + self._lock_docker_depth_experiment_for_reservation(reservation_id) + reservation = self.conn.execute( + '''SELECT r.*, q.status AS queue_status, q.lease_token AS queue_lease_token, + q.lease_expires_at AS queue_lease_expires_at, + q.current_result_reservation_id AS queue_reservation_id, + q.claim_event_id AS queue_event_id + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ? FOR UPDATE OF r, q''', + (int(reservation_id),), + ).fetchone() + if not reservation or ( + str(reservation['state']) != 'scanning' + or str(reservation['assignment_kind']) == 'remote' + or str(reservation['claim_lease_token']) != str(lease_token) + or str(reservation['producer_lease_expires_at'] or '') <= now + or str(reservation['queue_status']) != 'in_progress' + or str(reservation['queue_lease_token']) != str(lease_token) + or int(reservation['queue_reservation_id'] or 0) != int(reservation_id) + or str(reservation['queue_event_id'] or '') != str(reservation['scan_event_id']) + or str(reservation['queue_lease_expires_at'] or '') <= now + ): + self.conn.rollback() + return False + plan, plan_sha256 = stored_docker_layer_plan(reservation) + effective_seconds = max(60, int(lease_seconds)) + if plan is not None: + effective_seconds = max( + effective_seconds, + plan['limits']['blob_timeout_sec'] + DOCKER_BLOB_LEASE_MARGIN_SEC, + ) + lease_until = datetime.fromtimestamp( + time.time() + effective_seconds, timezone.utc, + ).isoformat(timespec='seconds') + if plan is not None: + coverage_policy_sha256 = docker_layer_plan_coverage_policy_sha256(plan) + seen = set() + for descriptor in plan['descriptors']: + if descriptor['coverage_state'] != 'leased' or descriptor['digest'] in seen: + continue + seen.add(descriptor['digest']) + cursor = self.conn.execute( + '''UPDATE docker_content_blobs SET lease_expires_at = ?, updated_at = ? + WHERE digest = ? AND coverage_policy_sha256 = ? AND state = 'leased' + AND lease_reservation_id = ? AND lease_token = ? + AND lease_plan_sha256 = ? + AND lease_expires_at IS NOT NULL AND lease_expires_at > ?''', + ( + lease_until, now, descriptor['digest'], + coverage_policy_sha256, int(reservation_id), + descriptor['lease_token'], plan_sha256, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + self.conn.rollback() + return False + cursor = self.conn.execute( + '''UPDATE result_reservations SET producer_lease_expires_at = ?, updated_at = ? + WHERE id = ? AND claim_lease_token = ? AND state = 'scanning' + AND producer_lease_expires_at > ?''', + (lease_until, now, int(reservation_id), str(lease_token), now), + ) + if int(cursor.rowcount or 0) != 1: + self.conn.rollback() + return False + cursor = self.conn.execute( + '''UPDATE target_queue SET lease_expires_at = ?, updated_at = ? + WHERE id = ? AND status = 'in_progress' AND lease_token = ? + AND current_result_reservation_id = ? AND claim_event_id = ? + AND lease_expires_at > ?''', + ( + lease_until, now, reservation['queue_id'], str(lease_token), + int(reservation_id), reservation['scan_event_id'], now, + ), + ) + if int(cursor.rowcount or 0) != 1: + self.conn.rollback() + return False + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + @staticmethod + def _validate_result_bundle_reservation_identity( + reservation, values, *, require_header=False, + ): + values = dict(values or {}) + ready_relative = str(reservation['ready_relative_path']).replace('\\', '/') + + def integer_matches(value, expected): + if isinstance(value, bool) or value is None: + return False + try: + return int(value) == int(expected) + except (TypeError, ValueError, OverflowError): + return False + + if ( + ('reservation_id' in values and not integer_matches( + values.get('reservation_id'), reservation['id'], + )) + or str(values.get('bundle_id') or '') != str(reservation['bundle_id']) + or str(values.get('scan_event_id') or '') + != str(reservation['scan_event_id']) + or str(values.get('relative_path') or '').replace('\\', '/') + != ready_relative + ): + raise ValueError('result bundle identity does not match its reservation') + + header = values.get('header') + result_metadata = values.get('result_metadata') + if not require_header and header is None and result_metadata is None: + return + if not isinstance(header, dict) or not isinstance(result_metadata, dict): + raise ValueError('result bundle identity metadata is incomplete') + + string_header = { + 'reservation_token': reservation['reservation_token'], + 'bundle_id': reservation['bundle_id'], + 'scan_event_id': reservation['scan_event_id'], + 'claim_lease_token': reservation['claim_lease_token'], + 'source': reservation['source'], + 'platform': reservation['platform'], + 'query': reservation['query'], + 'target': reservation['target'], + 'normalized_target': reservation['normalized_target'], + 'producer_instance_id': reservation['producer_instance_id'], + 'producer_creation_time': reservation['producer_creation_time'], + } + if any( + str(header.get(name) or '') != str(expected or '') + for name, expected in string_header.items() + ) or ( + str(header.get('ready_relative_path') or '').replace('\\', '/') + != ready_relative + ) or ( + str(header.get('producer_executable') or '').replace('\\', '/') + != str(reservation['producer_executable'] or '').replace('\\', '/') + ): + raise ValueError('result bundle header does not match its reservation') + + integer_header = { + 'format_version': 2, + 'reservation_id': reservation['id'], + 'queue_id': reservation['queue_id'], + 'declared_bytes': reservation['declared_bundle_bytes'], + 'producer_pid': reservation['producer_pid'], + } + if any( + not integer_matches(header.get(name), expected) + for name, expected in integer_header.items() + ): + raise ValueError('result bundle header does not match its reservation') + for name in ('run_id', 'cycle_id'): + actual = header.get(name) + expected = reservation[name] + if actual is None or expected is None: + matches = actual is None and expected is None + else: + matches = integer_matches(actual, expected) + if not matches: + raise ValueError('result bundle execution identity does not match its reservation') + + required_metadata = { + 'scan_event_id': reservation['scan_event_id'], + 'target': reservation['target'], + 'scan_type': reservation['platform'], + } + if any( + str(result_metadata.get(name) or '') != str(expected or '') + for name, expected in required_metadata.items() + ): + raise ValueError('result bundle scan identity does not match its reservation') + for name, expected in ( + ('bundle_id', reservation['bundle_id']), + ('source', reservation['source']), + ('platform', reservation['platform']), + ('query', reservation['query']), + ('normalized_target', reservation['normalized_target']), + ): + if ( + name in result_metadata + and result_metadata[name] is not None + and str(result_metadata[name]) != str(expected) + ): + raise ValueError('result bundle scan identity does not match its reservation') + for name, expected in ( + ('reservation_id', reservation['id']), + ('result_reservation_id', reservation['id']), + ('queue_id', reservation['queue_id']), + ): + if name in result_metadata and not integer_matches( + result_metadata.get(name), expected): + raise ValueError('result bundle scan identity does not match its reservation') + if normalize_target( + result_metadata.get('target'), reservation['platform'], + ) != str(reservation['normalized_target']): + raise ValueError('result bundle target identity does not match its reservation') + + scan_event_hash = str(values.get('scan_event_hash') or '').lower() + count_values = { + name: values.get(name) for name in ( + 'actual_bytes', 'frame_count', 'finding_count', 'error_count', + 'candidate_count', + ) + } + diagnostic_count = values.get('diagnostic_count', 0) + if ( + not re.fullmatch(r'[a-f0-9]{64}', scan_event_hash) + or not integer_matches(values.get('reservation_id'), reservation['id']) + ): + raise ValueError('result bundle metadata is outside its reservation bounds') + try: + if any( + value is None or isinstance(value, bool) + for value in count_values.values() + ): + raise ValueError + counts = { + name: int(value) for name, value in count_values.items() + } + if isinstance(diagnostic_count, bool): + raise ValueError + diagnostic_count = int(diagnostic_count or 0) + except (TypeError, ValueError, OverflowError): + raise ValueError( + 'result bundle metadata is outside its reservation bounds' + ) from None + if ( + counts['actual_bytes'] <= 0 + or counts['actual_bytes'] > int(reservation['declared_bundle_bytes']) + or counts['frame_count'] < 3 + or counts['frame_count'] != ( + counts['finding_count'] + counts['error_count'] + + counts['candidate_count'] + diagnostic_count + 3 + ) + or diagnostic_count < 0 + or any(counts[name] < 0 for name in ( + 'finding_count', 'error_count', 'candidate_count', + )) + ): + raise ValueError('result bundle metadata is outside its reservation bounds') + + @staticmethod + def _validate_locked_reservation_queue_identity(reservation, queue): + if not queue or ( + int(queue['id']) != int(reservation['queue_id']) + or str(queue['source']) != str(reservation['source']) + or str(queue['platform']) != str(reservation['platform']) + or str(queue['query'] or '') != str(reservation['query'] or '') + or str(queue['target']) != str(reservation['target']) + or str(queue['normalized_target']) + != str(reservation['normalized_target']) + or normalize_target(queue['target'], queue['platform']) + != str(queue['normalized_target']) + ): + raise ScanEventConflictError( + 'result reservation lost its exact target queue identity' + ) + + @staticmethod + def _validate_result_ingestion_arguments( + current_reservation, current_bundle, supplied_reservation, supplied_bundle, + ): + supplied_reservation = dict(supplied_reservation or {}) + supplied_bundle = dict(supplied_bundle or {}) + reservation_fields = ( + 'bundle_id', 'scan_event_id', 'queue_id', 'source', 'platform', + 'query', 'target', 'normalized_target', 'claim_lease_token', + 'run_id', 'cycle_id', + ) + supplied_reservation_id = supplied_reservation.get( + 'id', supplied_reservation.get('reservation_id'), + ) + try: + reservation_id_matches = ( + not isinstance(supplied_reservation_id, bool) + and int(supplied_reservation_id) == int(current_reservation['id']) + ) + except (TypeError, ValueError, OverflowError): + reservation_id_matches = False + if not reservation_id_matches or any( + str(supplied_reservation.get(name) or '') + != str(current_reservation[name] or '') + for name in reservation_fields + ): + raise ScanEventConflictError( + 'result ingestion reservation argument lost its exact identity' + ) + bundle_fields = ( + 'reservation_id', 'bundle_id', 'scan_event_id', 'scan_event_hash', + 'relative_path', 'actual_bytes', 'frame_count', 'finding_count', + 'error_count', 'candidate_count', + ) + if any( + str(supplied_bundle.get(name) or '').replace('\\', '/') + != str(current_bundle[name] or '').replace('\\', '/') + for name in bundle_fields + ): + raise ScanEventConflictError( + 'result ingestion bundle argument lost its exact identity' + ) + + def mark_result_bundle_ready( + self, reservation_id, metadata, remote_acceptance=None, + bundle_capacity_bytes=None, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('result bundle readiness requires PostgreSQL') + values = metadata.as_dict() if hasattr(metadata, 'as_dict') else dict(metadata or {}) + remote = dict(remote_acceptance or {}) if remote_acceptance is not None else None + diagnostic_projection = None + if remote is not None: + remote = { + 'device_id': int(remote.get('device_id') or 0), + 'payload_sha256': str(remote.get('payload_sha256') or '').lower(), + 'token_sha256': str(remote.get('token_sha256') or '').lower(), + } + if remote['device_id'] <= 0 or not re.fullmatch( + r'[a-f0-9]{64}', remote['payload_sha256'], + ) or not re.fullmatch(r'[a-f0-9]{64}', remote['token_sha256']): + raise ValueError('remote bundle acceptance identity is invalid') + diagnostic_projection = { + 'version': values.get('effective_diagnostic_projection_version'), + 'count': values.get('effective_diagnostic_count'), + 'uids_sha256': str( + values.get('effective_diagnostic_uids_sha256') or '' + ), + } + if ( + diagnostic_projection['version'] != 1 + or isinstance(diagnostic_projection['count'], bool) + or not isinstance(diagnostic_projection['count'], int) + or diagnostic_projection['count'] < 0 + or not re.fullmatch( + r'[a-f0-9]{64}', diagnostic_projection['uids_sha256'], + ) + ): + raise ValueError( + 'remote diagnostic projection authority is invalid' + ) + now = utc_now_iso() + try: + self._lock_docker_depth_experiment_for_reservation(reservation_id) + reservation = self.conn.execute( + '''SELECT r.*, q.id AS bound_queue_id, + q.source AS bound_queue_source, + q.platform AS bound_queue_platform, + q.query AS bound_queue_query, + q.target AS bound_queue_target, + q.normalized_target AS bound_queue_normalized_target, + q.status AS queue_status, q.lease_token AS queue_lease_token, + q.lease_expires_at AS queue_lease_expires_at, + q.current_result_reservation_id AS queue_reservation_id, + q.claim_event_id AS queue_event_id + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ? FOR UPDATE OF r, q''', + (int(reservation_id),), + ).fetchone() + if not reservation: + self.conn.rollback() + return False + if remote is not None: + if ( + str(reservation['assignment_kind']) != 'remote' + or int(reservation['remote_device_id'] or 0) != remote['device_id'] + ): + raise ScanEventConflictError('remote bundle owner identity conflicts') + if not self._lock_active_remote_credential( + remote['device_id'], remote['token_sha256'], + user_id=reservation['remote_user_id'], + ): + raise ScanEventConflictError( + 'remote worker credential changed before bundle acceptance' + ) + if reservation['remote_resolution_kind'] is not None: + if ( + str(reservation['remote_resolution_kind']) != 'bundle_accepted' + or str(reservation['remote_payload_sha256'] or '') + != remote['payload_sha256'] + ): + raise ScanEventConflictError('remote assignment already has a conflicting resolution') + receipt = self._remote_resolution_from_row(reservation) + receipt_diagnostics = receipt.get('diagnostics') or {} + if receipt_diagnostics.get('projection_version') is not None and ( + int(reservation['remote_diagnostic_projection_version'] or -1) + != int(receipt_diagnostics['projection_version']) + or int(reservation['remote_diagnostic_count'] or 0) + != int(receipt_diagnostics.get('count') or 0) + or str(reservation['remote_diagnostic_uids_sha256'] or '') + != str(receipt_diagnostics.get('ordered_uid_set_sha256') or '') + ): + raise ScanEventConflictError( + 'remote diagnostic projection receipt fence is invalid' + ) + self.conn.commit() + return receipt + elif str(reservation['assignment_kind']) == 'remote': + raise ScanEventConflictError('remote bundle requires authenticated acceptance') + now = utc_now_iso() + queue_identity = { + 'id': reservation['bound_queue_id'], + 'source': reservation['bound_queue_source'], + 'platform': reservation['bound_queue_platform'], + 'query': reservation['bound_queue_query'], + 'target': reservation['bound_queue_target'], + 'normalized_target': reservation['bound_queue_normalized_target'], + } + self._validate_locked_reservation_queue_identity( + reservation, queue_identity, + ) + self._validate_result_bundle_reservation_identity( + reservation, values, require_header='header' in values, + ) + if remote is not None: + result_metadata = values.get('result_metadata') + validate_remote_result_execution_plan( + reservation, result_metadata, + ) + actual_bytes = int(values['actual_bytes']) + reserved_bytes = int(reservation['reserved_bundle_bytes']) + if actual_bytes > reserved_bytes: + if bundle_capacity_bytes is None: + raise ValueError( + 'remote bundle expansion requires a bundle capacity limit' + ) + bundle_capacity_bytes = max(0, int(bundle_capacity_bytes)) + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + additional_bytes = actual_bytes - reserved_bytes + if ( + not capacity + or int(capacity['bundle_bytes']) + additional_bytes + > bundle_capacity_bytes + ): + self.conn.rollback() + raise PipelineCapacityUnavailable( + 'remote bundle expansion exceeds available capacity' + ) + cursor = self.conn.execute( + '''UPDATE result_reservations SET reserved_bundle_bytes = ?, + updated_at = ? WHERE id = ? AND state = 'scanning' + AND reserved_bundle_bytes = ?''', + (actual_bytes, now, reservation_id, reserved_bytes), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'remote bundle capacity expansion lost its reservation fence' + ) + self.conn.execute( + '''UPDATE pipeline_capacity SET bundle_bytes = bundle_bytes + ?, + updated_at = ? WHERE id = 1''', + (additional_bytes, now), + ) + active_fence = ( + str(reservation['state']) == 'scanning' + and str(reservation['queue_status']) == 'in_progress' + and str(reservation['queue_lease_token'] or '') == str(reservation['claim_lease_token']) + and int(reservation['queue_reservation_id'] or 0) == int(reservation_id) + and str(reservation['queue_event_id'] or '') == str(reservation['scan_event_id']) + and str(reservation['queue_lease_expires_at'] or '') > now + and str(reservation['producer_lease_expires_at'] or '') > now + and ( + remote is None or str(reservation['remote_expires_at'] or '') > now + ) + ) + existing = self.conn.execute( + 'SELECT scan_event_hash FROM result_bundles WHERE reservation_id = ?', + (int(reservation_id),), + ).fetchone() + receipt = None + if remote is not None: + receipt = { + 'receipt_id': secrets.token_hex(32), + 'resolution': 'bundle_accepted', + 'payload_sha256': remote['payload_sha256'], + 'reservation_id': int(reservation_id), + 'bundle_id': str(reservation['bundle_id']), + 'scan_event_id': str(reservation['scan_event_id']), + 'accepted_at': now, + } + receipt.update(self._remote_assignment_observability_locked( + reservation, + diagnostic_count=int( + values.get( + 'effective_diagnostic_count', + values.get('diagnostic_count') or 0, + ) + ), + )) + receipt['diagnostics'].update({ + 'projection_version': diagnostic_projection['version'], + 'ordered_uid_set_sha256': diagnostic_projection['uids_sha256'], + }) + receipt_json = json.dumps( + receipt, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ) + artifact_sha256 = remote['payload_sha256'] + else: + receipt_json = None + artifact_sha256 = values.get('scan_event_hash') + if existing: + if str(existing['scan_event_hash']) != str(values.get('scan_event_hash')): + raise ScanEventConflictError('ready bundle replay hash conflicts with its reservation') + if reservation['state'] == 'scanning': + if not active_fence: + self.conn.rollback() + return False + self._submit_docker_blob_leases_locked(reservation, now) + elif reservation['state'] not in ('ready', 'ingesting', 'db_committed', 'acknowledged'): + self.conn.rollback() + return False + if reservation['state'] in ('scanning', 'ready', 'ingesting', 'db_committed'): + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'deleted', byte_count = 0, + deleted_at = ?, updated_at = ? + WHERE subsystem = 'result_bundle' AND artifact_kind = 'bundle_partial' + AND owner_id = ?''', + (now, now, reservation_id), + ) + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'present', payload_sha256 = ?, + byte_count = ?, deleted_at = NULL, updated_at = ? + WHERE subsystem = 'result_bundle' AND artifact_kind = 'bundle_ready' + AND owner_id = ?''', + ( + artifact_sha256, int(values.get('actual_bytes') or 0), + now, reservation_id, + ), + ) + if remote is not None: + cursor = self.conn.execute( + '''UPDATE result_reservations SET + remote_resolution_kind = 'bundle_accepted', + remote_payload_sha256 = ?, remote_receipt_id = ?, + remote_resolution_json = ?, remote_resolved_at = ?, + remote_diagnostic_projection_version = ?, + remote_diagnostic_count = ?, + remote_diagnostic_uids_sha256 = ?, updated_at = ? + WHERE id = ? AND assignment_kind = 'remote' + AND remote_device_id = ? AND remote_resolution_kind IS NULL''', + ( + remote['payload_sha256'], receipt['receipt_id'], receipt_json, + now, diagnostic_projection['version'], + diagnostic_projection['count'], + diagnostic_projection['uids_sha256'], + now, reservation_id, remote['device_id'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('remote bundle receipt lost its reservation fence') + if reservation['state'] in ('scanning', 'ready'): + self._transition_docker_depth_binding_locked( + reservation, 'scanning', 'scanning', now, + ) + self.conn.commit() + return receipt if remote is not None else True + if reservation['state'] != 'scanning': + self.conn.rollback() + return False + if not active_fence: + self.conn.rollback() + return False + self._submit_docker_blob_leases_locked(reservation, now) + self.conn.execute( + '''INSERT INTO result_bundles( + reservation_id, bundle_id, scan_event_id, scan_event_hash, format_version, + relative_path, actual_bytes, frame_count, finding_count, error_count, + candidate_count, state, ready_at, updated_at + ) VALUES (?, ?, ?, ?, 2, ?, ?, ?, ?, ?, ?, 'ready', ?, ?)''', + ( + reservation_id, values['bundle_id'], values['scan_event_id'], + values['scan_event_hash'], values['relative_path'], values['actual_bytes'], + values['frame_count'], values['finding_count'], values['error_count'], + values['candidate_count'], now, now, + ), + ) + if remote is None: + self.conn.execute( + "UPDATE result_reservations SET state = 'ready', updated_at = ? WHERE id = ?", + (now, reservation_id), + ) + else: + cursor = self.conn.execute( + '''UPDATE result_reservations SET state = 'ready', + remote_resolution_kind = 'bundle_accepted', + remote_payload_sha256 = ?, remote_receipt_id = ?, + remote_resolution_json = ?, remote_resolved_at = ?, + remote_diagnostic_projection_version = ?, + remote_diagnostic_count = ?, + remote_diagnostic_uids_sha256 = ?, updated_at = ? + WHERE id = ? AND state = 'scanning' AND assignment_kind = 'remote' + AND remote_device_id = ? AND remote_resolution_kind IS NULL''', + ( + remote['payload_sha256'], receipt['receipt_id'], receipt_json, + now, diagnostic_projection['version'], + diagnostic_projection['count'], + diagnostic_projection['uids_sha256'], + now, reservation_id, remote['device_id'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('remote bundle receipt lost its reservation fence') + self._transition_docker_depth_binding_locked( + reservation, 'scanning', 'scanning', now, + ) + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'deleted', byte_count = 0, + deleted_at = ?, updated_at = ? + WHERE subsystem = 'result_bundle' AND artifact_kind = 'bundle_partial' + AND owner_id = ?''', + (now, now, reservation_id), + ) + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'present', payload_sha256 = ?, + byte_count = ?, deleted_at = NULL, updated_at = ? + WHERE subsystem = 'result_bundle' AND artifact_kind = 'bundle_ready' + AND owner_id = ?''', + ( + artifact_sha256, int(values['actual_bytes']), now, reservation_id, + ), + ) + self.conn.commit() + return receipt if remote is not None else True + except Exception: + self.conn.rollback() + raise + + def recover_expired_result_bundle_ready(self, reservation_id, metadata): + """Recover a validated bundle after its exact producer lease expired.""" + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('expired result bundle recovery requires PostgreSQL') + values = metadata.as_dict() if hasattr(metadata, 'as_dict') else dict(metadata or {}) + header = values.get('header') + if not isinstance(header, dict): + raise ValueError('expired ready bundle recovery requires its validated header') + try: + now = utc_now_iso() + self._lock_docker_depth_experiment_for_reservation(reservation_id) + reservation = self.conn.execute( + '''SELECT r.*, q.id AS bound_queue_id, + q.source AS bound_queue_source, + q.platform AS bound_queue_platform, + q.query AS bound_queue_query, + q.target AS bound_queue_target, + q.normalized_target AS bound_queue_normalized_target, + q.status AS queue_status, q.lease_token AS queue_lease_token, + q.lease_expires_at AS queue_lease_expires_at, + q.current_result_reservation_id AS queue_reservation_id, + q.claim_event_id AS queue_event_id + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ? FOR UPDATE OF r, q''', + (int(reservation_id),), + ).fetchone() + if not reservation: + self.conn.rollback() + return False + if str(reservation['assignment_kind']) == 'remote': + self.conn.rollback() + return False + queue_identity = { + 'id': reservation['bound_queue_id'], + 'source': reservation['bound_queue_source'], + 'platform': reservation['bound_queue_platform'], + 'query': reservation['bound_queue_query'], + 'target': reservation['bound_queue_target'], + 'normalized_target': reservation['bound_queue_normalized_target'], + } + self._validate_locked_reservation_queue_identity( + reservation, queue_identity, + ) + self._validate_result_bundle_reservation_identity( + reservation, values, require_header=True, + ) + ready_relative = str(reservation['ready_relative_path']).replace('\\', '/') + scan_event_hash = str(values.get('scan_event_hash') or '').lower() + counts = { + name: int(values.get(name) if values.get(name) is not None else -1) + for name in ( + 'actual_bytes', 'frame_count', 'finding_count', 'error_count', + 'candidate_count', + ) + } + diagnostic_count = int(values.get('diagnostic_count') or 0) + if ( + not re.fullmatch(r'[a-f0-9]{64}', scan_event_hash) + or counts['actual_bytes'] <= 0 + or counts['actual_bytes'] > int(reservation['declared_bundle_bytes']) + or counts['frame_count'] < 3 + or counts['frame_count'] != ( + counts['finding_count'] + counts['error_count'] + + counts['candidate_count'] + diagnostic_count + 3 + ) + or diagnostic_count < 0 + or any(counts[name] < 0 for name in ('finding_count', 'error_count', 'candidate_count')) + ): + raise ValueError('expired ready bundle metadata is outside its reservation bounds') + exact_fence = ( + str(reservation['state']) == 'scanning' + and str(reservation['queue_status']) == 'in_progress' + and str(reservation['queue_lease_token'] or '') == str(reservation['claim_lease_token']) + and int(reservation['queue_reservation_id'] or 0) == int(reservation_id) + and str(reservation['queue_event_id'] or '') == str(reservation['scan_event_id']) + and bool(reservation['queue_lease_expires_at']) + and str(reservation['queue_lease_expires_at']) <= now + and bool(reservation['producer_lease_expires_at']) + and str(reservation['producer_lease_expires_at']) <= now + ) + if not exact_fence: + self.conn.rollback() + return False + if exact_process_identity_state( + reservation['producer_pid'], reservation['producer_creation_time'], + reservation['producer_executable'], + ) not in ('dead', 'reused'): + self.conn.rollback() + return False + if reservation['docker_layer_plan_json'] or reservation['docker_layer_plan_sha256']: + self.conn.rollback() + return False + if self.conn.execute( + '''SELECT 1 AS present FROM docker_content_blobs + WHERE lease_reservation_id = ? AND state IN ('leased','submitted') LIMIT 1''', + (int(reservation_id),), + ).fetchone(): + self.conn.rollback() + return False + if self.conn.execute( + 'SELECT 1 AS present FROM result_bundles WHERE reservation_id = ?', + (int(reservation_id),), + ).fetchone(): + self.conn.rollback() + return False + if self.conn.execute( + '''SELECT 1 AS present FROM pipeline_quarantine + WHERE reservation_id = ? AND review_status = 'pending' LIMIT 1''', + (int(reservation_id),), + ).fetchone(): + self.conn.rollback() + return False + self._docker_depth_binding_for_reservation_locked(reservation) + self.conn.execute( + '''INSERT INTO result_bundles( + reservation_id, bundle_id, scan_event_id, scan_event_hash, format_version, + relative_path, actual_bytes, frame_count, finding_count, error_count, + candidate_count, state, ready_at, updated_at + ) VALUES (?, ?, ?, ?, 2, ?, ?, ?, ?, ?, ?, 'ready', ?, ?)''', + ( + int(reservation_id), values['bundle_id'], values['scan_event_id'], + scan_event_hash, ready_relative, counts['actual_bytes'], counts['frame_count'], + counts['finding_count'], counts['error_count'], counts['candidate_count'], + now, now, + ), + ) + cursor = self.conn.execute( + "UPDATE result_reservations SET state = 'ready', updated_at = ? WHERE id = ? AND state = 'scanning'", + (now, int(reservation_id)), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('expired ready bundle reservation changed during recovery') + self._transition_docker_depth_binding_locked( + reservation, 'scanning', 'scanning', now, + ) + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'deleted', byte_count = 0, + deleted_at = ?, updated_at = ? + WHERE subsystem = 'result_bundle' AND artifact_kind = 'bundle_partial' + AND owner_id = ?''', + (now, now, int(reservation_id)), + ) + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'present', payload_sha256 = ?, + byte_count = ?, deleted_at = NULL, updated_at = ? + WHERE subsystem = 'result_bundle' AND artifact_kind = 'bundle_ready' + AND owner_id = ?''', + (scan_event_hash, counts['actual_bytes'], now, int(reservation_id)), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + def active_result_reservations(self, after_id=0, limit=100): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('result reservation recovery requires PostgreSQL') + rows = self.conn.execute( + '''SELECT * FROM result_reservations + WHERE id > ? AND state IN ('scanning','ready','ingesting','db_committed') + AND (cleanup_available_after IS NULL OR cleanup_available_after <= ?) + ORDER BY id LIMIT ?''', + ( + max(0, int(after_id)), utc_now_iso(), + min(1000, max(1, int(limit))), + ), + ).fetchall() + self.conn.commit() + return [dict(row) for row in rows] + + def defer_result_reservation_cleanup(self, reservation_id, error): + if not self.conn or not self.conn.is_postgres: + return False + now = utc_now_iso() + try: + self._lock_docker_depth_experiment_for_reservation(reservation_id) + row = self.conn.execute( + 'SELECT cleanup_attempts, state FROM result_reservations WHERE id = ? FOR UPDATE', + (int(reservation_id),), + ).fetchone() + if not row: + self.conn.rollback() + return False + attempts = int(row['cleanup_attempts'] or 0) + 1 + delay = min(300, 2 ** min(attempts, 8)) + available = datetime.fromtimestamp( + time.time() + delay, timezone.utc, + ).isoformat(timespec='seconds') + self.conn.execute( + '''UPDATE result_reservations SET cleanup_attempts = ?, + cleanup_available_after = ?, last_error_code = 'storage_cleanup_deferred', + last_error_detail = ?, updated_at = ? WHERE id = ?''', + (attempts, available, first_line(error, 1000), now, int(reservation_id)), + ) + bundle = self.conn.execute( + 'SELECT target_scan_id, state FROM result_bundles WHERE reservation_id = ? FOR UPDATE', + (int(reservation_id),), + ).fetchone() + if bundle: + next_state = 'db_committed' if bundle['target_scan_id'] is not None else 'ready' + self.conn.execute( + '''UPDATE result_bundles SET state = ?, available_after = ?, + ingest_lease_generation = NULL, ingest_lease_token = NULL, + ingest_lease_expires_at = NULL, updated_at = ? + WHERE reservation_id = ?''', + (next_state, available, now, int(reservation_id)), + ) + self.conn.execute( + 'UPDATE result_reservations SET state = ? WHERE id = ?', + (next_state, int(reservation_id)), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + def result_bundle_for_reservation(self, reservation_id): + if not self.conn: + return None + row = self.conn.execute( + 'SELECT * FROM result_bundles WHERE reservation_id = ?', + (int(reservation_id),), + ).fetchone() + if self.conn.is_postgres: + self.conn.commit() + return dict(row) if row else None + + def pending_result_bundle_quarantine(self, reservation_id): + if not self.conn or not self.conn.is_postgres: + return None + row = self.conn.execute( + '''SELECT * FROM pipeline_quarantine + WHERE subsystem = 'result_ingester' AND object_type = 'result_bundle' + AND reservation_id = ? AND review_status = 'pending' ''', + (int(reservation_id),), + ).fetchone() + self.conn.commit() + return dict(row) if row else None + + def claim_ready_result_bundle(self, generation, lease_token, lease_seconds=60): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('result bundle claims require PostgreSQL') + now = utc_now_iso() + expires = datetime.fromtimestamp( + time.time() + max(10, int(lease_seconds)), timezone.utc, + ).isoformat(timespec='seconds') + ingest_token = secrets.token_urlsafe(32) + try: + owner = self.conn.execute( + '''SELECT 1 AS valid FROM pipeline_leases + WHERE worker_name = 'result_ingester' AND generation = ? + AND lease_token = ? AND state IN ('recovering','ready') + AND lease_expires_at > ? FOR SHARE''', + (int(generation), str(lease_token), now), + ).fetchone() + if not owner: + self.conn.rollback() + raise RuntimeError('result ingester singleton fence is not valid') + row = self.conn.execute( + '''SELECT reservation_id FROM result_bundles + WHERE ( + state IN ('ready','db_committed') + OR (state = 'ingesting' AND ( + ingest_lease_generation IS NULL OR ingest_lease_generation <> ? + OR (ingest_lease_expires_at IS NOT NULL AND ingest_lease_expires_at <= ?) + )) + ) AND (available_after IS NULL OR available_after <= ?) + ORDER BY CASE WHEN state = 'db_committed' THEN 0 ELSE 1 END, + ready_at, reservation_id + LIMIT 1 FOR UPDATE SKIP LOCKED''', + (int(generation), now, now), + ).fetchone() + if not row: + self.conn.rollback() + return None + bundle = self.conn.execute( + 'SELECT * FROM result_bundles WHERE reservation_id = ? FOR UPDATE', + (row['reservation_id'],), + ).fetchone() + next_state = 'db_committed' if bundle['state'] == 'db_committed' else 'ingesting' + self.conn.execute( + '''UPDATE result_bundles SET state = ?, ingest_attempts = ingest_attempts + 1, + ingest_lease_generation = ?, ingest_lease_token = ?, + ingest_lease_expires_at = ?, updated_at = ? + WHERE reservation_id = ?''', + ( + next_state, int(generation), ingest_token, expires, now, + row['reservation_id'], + ), + ) + if next_state == 'ingesting': + self.conn.execute( + "UPDATE result_reservations SET state = 'ingesting', updated_at = ? WHERE id = ?", + (now, row['reservation_id']), + ) + reservation = self.conn.execute( + 'SELECT * FROM result_reservations WHERE id = ?', + (row['reservation_id'],), + ).fetchone() + updated = self.conn.execute( + 'SELECT * FROM result_bundles WHERE reservation_id = ?', + (row['reservation_id'],), + ).fetchone() + self.conn.commit() + return { + 'reservation': dict(reservation), + 'bundle': dict(updated), + 'ingest_lease_token': ingest_token, + } + except Exception: + self.conn.rollback() + raise + + def confirm_scan_event(self, event_id, event_hash): + if not self.conn: + raise RuntimeError('database connection is unavailable') + row = self.conn.execute( + '''SELECT id, scan_event_id, scan_event_hash, result_reservation_id, + raw_result_storage, compat_schema_version + FROM target_scans WHERE scan_event_id = ?''', + (str(event_id),), + ).fetchone() + if self.conn.is_postgres: + self.conn.commit() + if not row: + return None + if str(row['scan_event_hash'] or '') != str(event_hash or ''): + raise ScanEventConflictError(f'scan event {event_id} already exists with a different hash') + return dict(row) + + def acknowledge_removed_bundle(self, reservation_id, event_id, event_hash): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('bundle acknowledgement requires PostgreSQL') + now = utc_now_iso() + try: + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + reservation = self.conn.execute( + 'SELECT * FROM result_reservations WHERE id = ? FOR UPDATE', + (int(reservation_id),), + ).fetchone() + bundle = self.conn.execute( + 'SELECT * FROM result_bundles WHERE reservation_id = ? FOR UPDATE', + (int(reservation_id),), + ).fetchone() + event = self.conn.execute( + 'SELECT id, scan_event_hash FROM target_scans WHERE scan_event_id = ?', + (str(event_id),), + ).fetchone() + if not reservation or not bundle or not event or str(event['scan_event_hash']) != str(event_hash): + self.conn.rollback() + return False + if str(bundle['scan_event_hash']) != str(event_hash): + raise ScanEventConflictError('bundle acknowledgement hash conflicts with committed event') + if not reservation['bundle_credit_released']: + if int(capacity['bundle_items']) < 1 or int(capacity['bundle_bytes']) < int(reservation['reserved_bundle_bytes']): + raise RuntimeError('bundle capacity accounting would become negative') + self.conn.execute( + '''UPDATE pipeline_capacity SET bundle_items = bundle_items - 1, + bundle_bytes = bundle_bytes - ?, updated_at = ? WHERE id = 1''', + (reservation['reserved_bundle_bytes'], now), + ) + self.conn.execute( + '''UPDATE result_reservations SET state = 'acknowledged', bundle_credit_released = 1, + released_at = COALESCE(released_at, ?), updated_at = ? WHERE id = ?''', + (now, now, reservation_id), + ) + self.conn.execute( + '''UPDATE result_bundles SET state = 'acknowledged', acknowledged_at = COALESCE(acknowledged_at, ?), + ingest_lease_token = NULL, ingest_lease_expires_at = NULL, updated_at = ? + WHERE reservation_id = ?''', + (now, now, reservation_id), + ) + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'deleted', byte_count = 0, + deleted_at = ?, updated_at = ? + WHERE subsystem = 'result_bundle' AND owner_id = ? + AND artifact_kind IN ('bundle_partial','bundle_ready')''', + (now, now, reservation_id), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + @staticmethod + def _remote_resolution_from_row(row): + raw = str(row['remote_resolution_json'] or '') + try: + value = json.loads(raw) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise ScanEventConflictError('remote assignment durable receipt is invalid') from exc + if not isinstance(value, dict) or str(value.get('receipt_id') or '') != str( + row['remote_receipt_id'] or '' + ): + raise ScanEventConflictError('remote assignment durable receipt identity is invalid') + return value + + def _remote_assignment_observability_locked( + self, row, *, diagnostic_count=None, + ): + def row_value(name): + return row.get(name) if hasattr(row, 'get') else row[name] + + latest = self.conn.execute( + '''SELECT sequence, slot_id, phase, event_timestamp, phase_started_at, + event_json, received_at + FROM worker_progress_events + WHERE reservation_id = ? ORDER BY sequence DESC, id DESC LIMIT 1''', + (int(row['id']),), + ).fetchone() + latest_value = None + known_reason = None + scan_deadline_at = None + if latest: + try: + event = json.loads(str(latest['event_json'])) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise WorkerObservabilityConflictError( + 'durable worker progress JSON is invalid' + ) from exc + latest_value = { + 'sequence': int(latest['sequence']), + 'slot_id': int(latest['slot_id']), + 'phase': str(latest['phase']), + 'event_timestamp': str(latest['event_timestamp']), + 'phase_started_at': str(latest['phase_started_at'] or ''), + 'received_at': str(latest['received_at']), + 'progress': dict(event.get('progress') or {}), + } + scan_deadline_at = event.get('scan_deadline_at') + if latest_value['phase'] in {'idle', 'backoff'}: + reason = latest_value['progress'].get('reason') + known_reason = str(reason) if reason is not None else None + if diagnostic_count is None: + count_row = self.conn.execute( + '''SELECT COUNT(*) AS count FROM worker_diagnostics + WHERE reservation_id = ?''', + (int(row['id']),), + ).fetchone() + diagnostic_count = int(count_row['count'] or 0) if count_row else 0 + accepted_count = row_value('remote_diagnostic_count') + if accepted_count is not None: + diagnostic_count = max( + diagnostic_count, int(accepted_count), + ) + timeout_seconds = None + snapshot_raw = row_value('remote_execution_snapshot_json') + if snapshot_raw: + try: + snapshot = json.loads(str(snapshot_raw)) + timeout = ((snapshot.get('execution') or {}).get('scan_kwargs') or {}).get( + 'timeout_sec' + ) + if type(timeout) in (int, float) and not isinstance(timeout, bool): + timeout_seconds = int(timeout) + except (TypeError, ValueError, json.JSONDecodeError): + timeout_seconds = None + return { + 'deadlines': { + 'assignment_issued_at': ( + str(row_value('remote_issued_at')) + if row_value('remote_issued_at') else None + ), + 'assignment_deadline_at': ( + str(row_value('remote_expires_at')) + if row_value('remote_expires_at') else None + ), + 'scan_deadline_at': scan_deadline_at, + 'target_scan_timeout_seconds': timeout_seconds, + 'result_upload_body_timeout_seconds': ( + int(row_value('remote_result_upload_body_timeout_seconds')) + if row_value('remote_result_upload_body_timeout_seconds') is not None + else None + ), + 'result_upload_body_timeout_availability': ( + 'persisted' + if row_value('remote_result_upload_body_timeout_seconds') is not None + else 'legacy/unavailable' + ), + 'immutable': True, + }, + 'latest_progress': { + 'available': latest_value is not None, + 'event': latest_value, + }, + 'known_reason': known_reason, + 'diagnostics': { + 'available': int(diagnostic_count) > 0, + 'count': int(diagnostic_count), + 'projection_version': row_value( + 'remote_diagnostic_projection_version' + ), + 'ordered_uid_set_sha256': row_value( + 'remote_diagnostic_uids_sha256' + ), + }, + } + + def remote_assignment_status(self, reservation_id, device_id, token_sha256): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote assignment status requires PostgreSQL') + try: + credential = self._lock_active_remote_credential( + device_id, token_sha256, + ) + if not credential: + self.conn.rollback() + return None + row = self.conn.execute( + '''SELECT id, bundle_id, scan_event_id, state, remote_device_id, + remote_issued_at, remote_expires_at, + remote_result_upload_body_timeout_seconds, + remote_execution_snapshot_json, + remote_diagnostic_projection_version, + remote_diagnostic_count, + remote_diagnostic_uids_sha256, + remote_resolution_kind, remote_receipt_id, + remote_resolution_json, remote_payload_sha256, remote_resolved_at + FROM result_reservations + WHERE id = ? AND assignment_kind = 'remote' AND remote_device_id = ?''', + (int(reservation_id), int(device_id)), + ).fetchone() + observability = ( + self._remote_assignment_observability_locked(row) if row else None + ) + self.conn.commit() + except Exception: + self.conn.rollback() + raise + if not row: + return None + if row['remote_resolution_kind'] is not None: + result = dict(self._remote_resolution_from_row(row)) + result.setdefault('deadlines', observability['deadlines']) + for name in ('latest_progress', 'known_reason', 'diagnostics'): + result[name] = observability[name] + return result + return { + 'reservation_id': int(row['id']), + 'bundle_id': str(row['bundle_id']), + 'scan_event_id': str(row['scan_event_id']), + 'state': str(row['state']), + 'expires_at': str(row['remote_expires_at']), + **observability, + } + + def remote_assignment_transport(self, reservation_id, device_id, token_sha256): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote assignment transport requires PostgreSQL') + try: + credential = self._lock_active_remote_credential( + device_id, token_sha256, + ) + if not credential: + self.conn.rollback() + return None + row = self.conn.execute( + '''SELECT id, reservation_token, bundle_id, scan_event_id, queue_id, + claim_lease_token, declared_bundle_bytes, ready_relative_path, + source, platform, query, target, normalized_target, run_id, cycle_id, + producer_instance_id, producer_pid, producer_creation_time, + producer_executable, state, remote_device_id, remote_expires_at, + remote_resolution_kind, remote_payload_sha256, remote_receipt_id, + remote_resolution_json + FROM result_reservations + WHERE id = ? AND assignment_kind = 'remote' AND remote_device_id = ?''', + (int(reservation_id), int(device_id)), + ).fetchone() + self.conn.commit() + except Exception: + self.conn.rollback() + raise + if not row: + return None + result = dict(row) + result['reservation_id'] = int(result['id']) + result['declared_bytes'] = int(result['declared_bundle_bytes']) + result['ready_path'] = str(result['ready_relative_path']) + if result['remote_resolution_kind'] is not None: + result['receipt'] = self._remote_resolution_from_row(row) + return result + + def _resolve_remote_scanning_locked( + self, row, resolution_kind, resolution_payload_sha256, detail, now, + receipt_fields=None, + ): + reservation_id = int(row['id']) + queue = self.conn.execute( + 'SELECT * FROM target_queue WHERE id = ? FOR UPDATE', + (int(row['queue_id']),), + ).fetchone() + if not queue or ( + str(queue['status']) != 'in_progress' + or str(queue['lease_token'] or '') != str(row['claim_lease_token']) + or int(queue['current_result_reservation_id'] or 0) != reservation_id + or str(queue['claim_event_id'] or '') != str(row['scan_event_id']) + ): + raise ScanEventConflictError('remote assignment lost its exact queue fence') + error_code = ( + 'remote_assignment_expired' + if resolution_kind == 'expired' + else 'remote_prebundle_infrastructure_failure' + ) + self._release_docker_blob_leases_locked( + row, error_code, detail, now, refund_attempt=True, + ) + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + deductions = ( + int(row['reserved_bundle_bytes']), int(row['reserved_projection_items']), + int(row['reserved_projection_bytes']), int(row['reserved_candidate_items']), + int(row['reserved_candidate_bytes']), + ) + if not capacity or ( + int(capacity['bundle_items']) < 1 + or int(capacity['bundle_bytes']) < deductions[0] + or int(capacity['projection_items']) < deductions[1] + or int(capacity['projection_bytes']) < deductions[2] + or int(capacity['keycheck_items']) < deductions[3] + or int(capacity['keycheck_bytes']) < deductions[4] + ): + raise RuntimeError('remote assignment resolution would make capacity accounting negative') + cursor = self.conn.execute( + '''UPDATE target_queue SET status = 'pending', + attempts = CASE WHEN attempts > 0 THEN attempts - 1 ELSE 0 END, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, + leased_at = NULL, lease_expires_at = NULL, available_after = NULL, + current_result_reservation_id = NULL, claim_event_id = NULL, + last_error = ?, updated_at = ? + WHERE id = ? AND status = 'in_progress' AND lease_token = ? + AND current_result_reservation_id = ? AND claim_event_id = ?''', + ( + first_line(detail, 500), now, row['queue_id'], row['claim_lease_token'], + reservation_id, row['scan_event_id'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('remote assignment queue changed during resolution') + self.conn.execute( + '''UPDATE pipeline_capacity SET + bundle_items = bundle_items - 1, bundle_bytes = bundle_bytes - ?, + projection_items = projection_items - ?, projection_bytes = projection_bytes - ?, + keycheck_items = keycheck_items - ?, keycheck_bytes = keycheck_bytes - ?, + updated_at = ? WHERE id = 1''', + (*deductions, now), + ) + receipt = { + 'receipt_id': secrets.token_hex(32), + 'resolution': resolution_kind, + 'reservation_id': reservation_id, + 'bundle_id': str(row['bundle_id']), + 'scan_event_id': str(row['scan_event_id']), + 'resolved_at': now, + } + receipt.update(self._remote_assignment_observability_locked(row)) + if resolution_payload_sha256: + receipt['payload_sha256'] = resolution_payload_sha256 + if receipt_fields: + receipt.update(dict(receipt_fields)) + receipt_json = json.dumps( + receipt, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ) + cursor = self.conn.execute( + '''UPDATE result_reservations SET state = 'refunded', bundle_credit_released = 1, + last_error_code = ?, last_error_detail = ?, refunded_at = ?, + released_at = ?, remote_resolution_kind = ?, remote_payload_sha256 = ?, + remote_receipt_id = ?, remote_resolution_json = ?, remote_resolved_at = ?, + updated_at = ? WHERE id = ? AND state = 'scanning' + AND assignment_kind = 'remote' AND remote_resolution_kind IS NULL''', + ( + error_code, first_line(detail, 1000), now, now, resolution_kind, + resolution_payload_sha256, receipt['receipt_id'], receipt_json, + now, now, reservation_id, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('remote assignment changed during resolution') + self._transition_docker_depth_binding_locked(row, 'released', 'pending', now) + self.conn.execute( + '''UPDATE pipeline_artifacts SET cleanup_attempts = 0, + cleanup_available_after = NULL, cleanup_last_error = NULL, + updated_at = ? + WHERE subsystem = 'result_bundle' AND owner_id = ? + AND artifact_kind IN ('bundle_partial','bundle_ready') + AND state IN ('expected','present')''', + (now, reservation_id), + ) + return receipt + + def report_remote_prebundle_failure( + self, reservation_id, device_id, token_sha256, report, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote pre-bundle reporting requires PostgreSQL') + report = dict(report or {}) + failure_code = str(report.get('failure_code') or '').strip().lower() + if failure_code not in { + 'client_process_failed', 'client_storage_failed', 'client_cancelled', + }: + raise ValueError('remote pre-bundle failure code is not permitted') + if set(report) not in ( + {'failure_code', 'detail'}, + {'failure_code', 'detail', 'diagnostics'}, + ) or not isinstance(report.get('detail'), str) or len(report['detail']) > 1000: + raise ValueError('remote pre-bundle report shape is invalid') + detail = first_line(report.get('detail') or failure_code, 1000) + normalized_report = { + 'failure_code': failure_code, + 'detail': report['detail'], + } + diagnostics = report.get('diagnostics') + if diagnostics is not None: + if not isinstance(diagnostics, list) or len(diagnostics) > 32: + raise ValueError('remote pre-bundle diagnostics are invalid') + normalized_report['diagnostics'] = [] + diagnostic_uids = set() + for diagnostic in diagnostics: + normalized = _canonical_worker_contract( + 'diagnostic', diagnostic, + )[0] + if ( + normalized.get('assignment_outcome') != 'prebundle_failed' + or normalized.get('scan_outcome') != 'unavailable' + ): + raise ValueError( + 'remote pre-bundle diagnostic outcomes are invalid' + ) + uid = normalized['diagnostic_uid'] + if uid in diagnostic_uids: + raise ValueError( + 'remote pre-bundle diagnostic UID is duplicated' + ) + diagnostic_uids.add(uid) + normalized_report['diagnostics'].append(normalized) + report_json = json.dumps( + normalized_report, + ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ) + if len(report_json.encode('ascii')) > 16 * 1024: + raise ValueError('remote pre-bundle report exceeds its byte bound') + report_sha256 = hashlib.sha256(report_json.encode('utf-8')).hexdigest() + now = utc_now_iso() + try: + self._lock_docker_depth_experiment_for_reservation(reservation_id) + row = self.conn.execute( + 'SELECT * FROM result_reservations WHERE id = ? FOR UPDATE', + (int(reservation_id),), + ).fetchone() + if not row or str(row['assignment_kind']) != 'remote' or int( + row['remote_device_id'] or 0 + ) != int(device_id): + self.conn.rollback() + return None + if not self._lock_active_remote_credential( + device_id, token_sha256, user_id=row['remote_user_id'], + ): + raise ScanEventConflictError( + 'remote worker credential changed before terminal report' + ) + if row['remote_resolution_kind'] is not None: + if ( + str(row['remote_resolution_kind']) != 'prebundle_report' + or str(row['remote_payload_sha256'] or '') != report_sha256 + ): + raise ScanEventConflictError('remote assignment already has a conflicting resolution') + receipt = self._remote_resolution_from_row(row) + self.conn.commit() + return receipt + now = utc_now_iso() + if ( + str(row['state']) != 'scanning' + or not row['remote_expires_at'] + or str(row['remote_expires_at']) <= now + ): + self.conn.rollback() + return None + for diagnostic in normalized_report.get('diagnostics') or (): + self._record_worker_diagnostic( + int(reservation_id), diagnostic, received_at=now, + ) + receipt = self._resolve_remote_scanning_locked( + row, 'prebundle_report', report_sha256, detail, now, + receipt_fields={'failure_code': failure_code}, + ) + self.conn.commit() + return receipt + except Exception: + self.conn.rollback() + raise + + def expire_remote_assignment(self, reservation_id, now=None): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote assignment expiry requires PostgreSQL') + now = str(now or utc_now_iso()) + try: + self._lock_docker_depth_experiment_for_reservation(reservation_id) + row = self.conn.execute( + 'SELECT * FROM result_reservations WHERE id = ? FOR UPDATE', + (int(reservation_id),), + ).fetchone() + if not row or ( + str(row['assignment_kind']) != 'remote' + or str(row['state']) != 'scanning' + or row['remote_resolution_kind'] is not None + or not row['remote_expires_at'] + or str(row['remote_expires_at']) > now + ): + self.conn.rollback() + return None + contracts = importlib.import_module('worker_contracts') + observability = self._remote_assignment_observability_locked(row) + latest = observability['latest_progress']['event'] + phase_name = ( + latest['phase'] if latest is not None + else contracts.WorkerPhase.ASSIGNED.value + ) + envelope = contracts.build_diagnostic_envelope( + occurrence_id=( + f'reservation:{int(reservation_id)}:' + f'{row["remote_expires_at"]}:expiry' + ), + reservation_id=int(reservation_id), + scan_event_id=None, + slot_id=(latest['slot_id'] if latest is not None else 0), + source=str(row['source']), + phase=contracts.WorkerPhase(phase_name), + kind=contracts.DiagnosticKind.ASSIGNMENT, + category=contracts.DiagnosticCategory.ASSIGNMENT_EXPIRED, + code='assignment.deadline_expired', + summary=( + f'Assignment deadline expired after phase {phase_name}' + if latest is not None else 'Assignment deadline expired without progress' + ), + retryable=True, + attempt=1, + assignment_outcome=contracts.AssignmentOutcome.EXPIRED, + scan_outcome=contracts.ScanOutcome.UNAVAILABLE, + occurred_at=_worker_contract_timestamp(row['remote_expires_at']), + captured_at=_worker_contract_timestamp(now), + ) + diagnostic = json.loads( + contracts.encode_diagnostic_envelope(envelope).decode('ascii') + ) + self._record_worker_diagnostic( + int(reservation_id), diagnostic, received_at=now, + ) + receipt = self._resolve_remote_scanning_locked( + row, 'expired', None, 'remote assignment deadline expired', now, + ) + self.conn.commit() + return receipt + except Exception: + self.conn.rollback() + raise + + def reap_expired_remote_assignments(self, limit=100): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('remote assignment reaper requires PostgreSQL') + now = utc_now_iso() + rows = self.conn.execute( + '''SELECT id FROM result_reservations + WHERE assignment_kind = 'remote' AND state = 'scanning' + AND remote_resolution_kind IS NULL AND remote_expires_at <= ? + ORDER BY remote_expires_at, id LIMIT ?''', + (now, min(1000, max(1, int(limit)))), + ).fetchall() + self.conn.commit() + receipts = [] + for row in rows: + try: + receipt = self.expire_remote_assignment(row['id'], now=now) + except Exception: + logger.error( + 'remote assignment expiry failed for reservation_id=%s', + int(row['id']), + ) + continue + if receipt: + receipts.append(receipt) + return receipts + + def refund_uncommitted_reservation( + self, reservation_id, exact_producer_identity, reason, + partial_absence_confirmed=False, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('reservation refunds require PostgreSQL') + if partial_absence_confirmed is not True: + raise RuntimeError( + 'reservation refund requires tri-state-confirmed deterministic partial absence' + ) + identity = _identity_mapping(exact_producer_identity) + now = utc_now_iso() + try: + self._lock_docker_depth_experiment_for_reservation(reservation_id) + row = self.conn.execute( + 'SELECT * FROM result_reservations WHERE id = ? FOR UPDATE', + (int(reservation_id),), + ).fetchone() + if not row or row['state'] != 'scanning': + self.conn.rollback() + return False + if str(row['assignment_kind']) != 'local': + self.conn.rollback() + return False + if ( + int(row['producer_pid']) != identity['pid'] + or str(row['producer_creation_time']) != identity['creation_time'] + or os.path.normcase(os.path.realpath(str(row['producer_executable']))) != identity['executable'] + ): + self.conn.rollback() + return False + queue = self.conn.execute( + 'SELECT * FROM target_queue WHERE id = ? FOR UPDATE', + (int(row['queue_id']),), + ).fetchone() + if not queue or ( + str(queue['status']) != 'in_progress' + or str(queue['lease_token'] or '') != str(row['claim_lease_token']) + or int(queue['current_result_reservation_id'] or 0) != int(row['id']) + or str(queue['claim_event_id'] or '') != str(row['scan_event_id']) + ): + self.conn.rollback() + return False + self._release_docker_blob_leases_locked( + row, + 'producer_failed_before_handoff', + reason, + now, + refund_attempt=True, + ) + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + cursor = self.conn.execute( + '''UPDATE target_queue SET status = 'pending', + attempts = CASE WHEN attempts > 0 THEN attempts - 1 ELSE 0 END, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, + leased_at = NULL, lease_expires_at = NULL, available_after = NULL, + current_result_reservation_id = NULL, claim_event_id = NULL, + last_error = ?, updated_at = ? + WHERE id = ? AND status = 'in_progress' AND lease_token = ? + AND current_result_reservation_id = ? AND claim_event_id = ?''', + ( + first_line(reason, 500), now, row['queue_id'], row['claim_lease_token'], + row['id'], row['scan_event_id'], + ), + ) + if int(cursor.rowcount or 0) != 1: + self.conn.rollback() + return False + deductions = ( + int(row['reserved_bundle_bytes']), int(row['reserved_projection_items']), + int(row['reserved_projection_bytes']), int(row['reserved_candidate_items']), + int(row['reserved_candidate_bytes']), + ) + if ( + int(capacity['bundle_items']) < 1 + or int(capacity['bundle_bytes']) < deductions[0] + or int(capacity['projection_items']) < deductions[1] + or int(capacity['projection_bytes']) < deductions[2] + or int(capacity['keycheck_items']) < deductions[3] + or int(capacity['keycheck_bytes']) < deductions[4] + ): + raise RuntimeError('reservation refund would make capacity accounting negative') + self.conn.execute( + '''UPDATE pipeline_capacity SET + bundle_items = bundle_items - 1, bundle_bytes = bundle_bytes - ?, + projection_items = projection_items - ?, projection_bytes = projection_bytes - ?, + keycheck_items = keycheck_items - ?, keycheck_bytes = keycheck_bytes - ?, + updated_at = ? WHERE id = 1''', + (*deductions, now), + ) + self.conn.execute( + '''UPDATE result_reservations SET state = 'refunded', bundle_credit_released = 1, + last_error_code = 'producer_failed_before_handoff', last_error_detail = ?, + refunded_at = ?, released_at = ?, updated_at = ? WHERE id = ?''', + (first_line(reason, 1000), now, now, now, row['id']), + ) + self._transition_docker_depth_binding_locked( + row, 'released', 'pending', now, + ) + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'deleted', byte_count = 0, + deleted_at = ?, updated_at = ? + WHERE subsystem = 'result_bundle' AND owner_id = ? + AND artifact_kind IN ('bundle_partial','bundle_ready')''', + (now, now, row['id']), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + def _finalize_result_bundle_quarantine_locked( + self, reservation, reason_code, reason_detail, source_relative_path, + payload_sha256, byte_count, now, + ): + reservation_id = int(reservation['id']) + queue = self.conn.execute( + 'SELECT * FROM target_queue WHERE id = ? FOR UPDATE', + (int(reservation['queue_id']),), + ).fetchone() + if str(reservation['state']) == 'quarantined': + if not queue or ( + str(queue['status']) != 'quarantined' + or int(queue['current_result_reservation_id'] or 0) != reservation_id + or str(queue['claim_event_id'] or '') != str(reservation['scan_event_id']) + or not reservation['bundle_credit_released'] + or str(reservation['last_error_code'] or '') != str(reason_code) + or str(reservation['last_error_detail'] or '') != first_line(reason_detail, 2000) + ): + raise ScanEventConflictError('quarantined bundle lost its exact queue fence') + bundle = self.conn.execute( + 'SELECT state FROM result_bundles WHERE reservation_id = ? FOR UPDATE', + (reservation_id,), + ).fetchone() + if bundle and str(bundle['state']) != 'quarantined': + raise ScanEventConflictError('quarantined bundle row lost its terminal state') + if source_relative_path: + artifact = self.conn.execute( + '''SELECT state, payload_sha256, byte_count FROM pipeline_artifacts + WHERE subsystem = 'result_bundle' + AND artifact_kind = 'bundle_quarantine' AND owner_id = ? + AND relative_path = ? FOR UPDATE''', + (reservation_id, self._artifact_relative_path(source_relative_path)), + ).fetchone() + if not artifact or ( + str(artifact['state']) != 'quarantined' + or str(artifact['payload_sha256'] or '') != str(payload_sha256 or '') + or int(artifact['byte_count'] or 0) != max(0, int(byte_count)) + ): + raise ScanEventConflictError('quarantined bundle artifact evidence changed') + return + else: + if not queue or ( + str(queue['status']) != 'in_progress' + or str(queue['lease_token'] or '') != str(reservation['claim_lease_token']) + or int(queue['current_result_reservation_id'] or 0) != reservation_id + or str(queue['claim_event_id'] or '') != str(reservation['scan_event_id']) + ): + raise ScanEventConflictError('bundle quarantine lost its exact queue fence') + self._release_docker_blob_leases_locked( + reservation, + 'bundle_quarantined', + reason_detail or reason_code, + now, + refund_attempt=False, + ) + cursor = self.conn.execute( + '''UPDATE target_queue SET status = 'quarantined', completed_at = ?, last_error = ?, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, + leased_at = NULL, lease_expires_at = NULL, updated_at = ? + WHERE id = ? AND status = 'in_progress' AND lease_token = ? + AND current_result_reservation_id = ? AND claim_event_id = ?''', + ( + now, first_line(reason_detail or reason_code, 500), now, + reservation['queue_id'], reservation['claim_lease_token'], + reservation_id, reservation['scan_event_id'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('bundle quarantine queue transition lost its fence') + cursor = self.conn.execute( + '''UPDATE result_reservations SET state = 'quarantined', bundle_credit_released = 1, + last_error_code = ?, last_error_detail = ?, released_at = ?, updated_at = ? + WHERE id = ? AND state = ?''', + ( + str(reason_code), first_line(reason_detail, 2000), now, now, + reservation_id, reservation['state'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('bundle quarantine reservation transition lost its fence') + self._transition_docker_depth_binding_locked( + reservation, 'quarantined', 'quarantined', now, + ) + self.conn.execute( + '''UPDATE result_bundles SET state = 'quarantined', ingest_lease_token = NULL, + ingest_lease_expires_at = NULL, updated_at = ? WHERE reservation_id = ?''', + (now, reservation_id), + ) + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'deleted', byte_count = 0, + deleted_at = ?, updated_at = ? + WHERE subsystem = 'result_bundle' AND owner_id = ? + AND artifact_kind IN ('bundle_partial','bundle_ready')''', + (now, now, reservation_id), + ) + if source_relative_path: + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'quarantined', payload_sha256 = ?, + byte_count = ?, deleted_at = NULL, updated_at = ? + WHERE subsystem = 'result_bundle' AND artifact_kind = 'bundle_quarantine' + AND owner_id = ? AND relative_path = ?''', + ( + str(payload_sha256 or ''), max(0, int(byte_count)), now, + reservation_id, self._artifact_relative_path(source_relative_path), + ), + ) + + def quarantine_result_bundle( + self, reservation_id, reason_code, reason_detail='', source_relative_path='', + payload_sha256='', byte_count=0, + quarantine_max_items=10000, quarantine_max_bytes=1024 * 1024 * 1024, + physical_confirmed=False, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('bundle quarantine requires PostgreSQL') + reason_code = str(reason_code or '').strip() + if not reason_code or len(reason_code) > 64 or not re.fullmatch(r'[a-z0-9_]+', reason_code): + raise ValueError('bundle quarantine reason code is invalid') + reason_detail = first_line(reason_detail, 2000) + source_relative_path = ( + self._artifact_relative_path(source_relative_path) if source_relative_path else '' + ) + payload_sha256 = str(payload_sha256 or '').strip().lower() + if payload_sha256 and not re.fullmatch(r'[a-f0-9]{64}', payload_sha256): + raise ValueError('bundle quarantine payload hash is invalid') + byte_count = max(0, int(byte_count)) + now = utc_now_iso() + try: + self._lock_docker_depth_experiment_for_reservation(reservation_id) + row = self.conn.execute( + 'SELECT * FROM result_reservations WHERE id = ? FOR UPDATE', + (int(reservation_id),), + ).fetchone() + if not row: + self.conn.rollback() + raise ValueError('result reservation is absent') + existing = self.conn.execute( + '''SELECT * FROM pipeline_quarantine + WHERE subsystem = 'result_ingester' AND object_type = 'result_bundle' + AND object_id = ? AND review_status = 'pending' ''', + (int(reservation_id),), + ).fetchone() + if existing: + if ( + str(existing['source_relative_path'] or '') != str(source_relative_path or '') + or str(existing['payload_sha256'] or '') != str(payload_sha256 or '') + or int(existing['byte_count']) != byte_count + or str(existing['reason_code']) != reason_code + or str(existing['reason_detail'] or '') != reason_detail + or int(existing['reservation_id'] or 0) != int(reservation_id) + or str(existing['event_id'] or '') != str(row['scan_event_id']) + ): + raise ScanEventConflictError('bundle quarantine preparation conflicts with existing evidence') + if physical_confirmed: + self._finalize_result_bundle_quarantine_locked( + row, reason_code, reason_detail, source_relative_path, + payload_sha256, byte_count, now, + ) + self.conn.commit() + return int(existing['id']) + if str(row['state']) not in ('scanning', 'ready', 'ingesting'): + raise ScanEventConflictError('bundle reservation is not exactly quarantineable') + queue = self.conn.execute( + 'SELECT * FROM target_queue WHERE id = ? FOR UPDATE', + (int(row['queue_id']),), + ).fetchone() + if not queue or ( + str(queue['status']) != 'in_progress' + or str(queue['lease_token'] or '') != str(row['claim_lease_token']) + or int(queue['current_result_reservation_id'] or 0) != int(row['id']) + or str(queue['claim_event_id'] or '') != str(row['scan_event_id']) + ): + raise ScanEventConflictError('bundle quarantine does not own the target queue fence') + bundle = self.conn.execute( + 'SELECT * FROM result_bundles WHERE reservation_id = ? FOR UPDATE', + (int(reservation_id),), + ).fetchone() + if str(row['state']) in ('ready', 'ingesting') and ( + not bundle + or str(bundle['bundle_id']) != str(row['bundle_id']) + or str(bundle['scan_event_id']) != str(row['scan_event_id']) + or str(bundle['relative_path']).replace('\\', '/') + != str(row['ready_relative_path']).replace('\\', '/') + or str(bundle['state']) not in ('ready', 'ingesting') + ): + raise ScanEventConflictError('bundle quarantine row identity is not exact') + if str(row['state']) == 'scanning' and bundle is not None: + raise ScanEventConflictError('scanning quarantine has an unexpected ready bundle row') + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + projection_items = 0 if row['projection_credit_transferred'] else int(row['reserved_projection_items']) + projection_bytes = 0 if row['projection_credit_transferred'] else int(row['reserved_projection_bytes']) + candidate_items = 0 if row['candidate_credit_transferred'] else int(row['reserved_candidate_items']) + candidate_bytes = 0 if row['candidate_credit_transferred'] else int(row['reserved_candidate_bytes']) + bundle_items = 0 if row['bundle_credit_released'] else 1 + bundle_bytes = 0 if row['bundle_credit_released'] else int(row['reserved_bundle_bytes']) + if ( + int(capacity['bundle_items']) < bundle_items + or int(capacity['bundle_bytes']) < bundle_bytes + or int(capacity['projection_items']) < projection_items + or int(capacity['projection_bytes']) < projection_bytes + or int(capacity['keycheck_items']) < candidate_items + or int(capacity['keycheck_bytes']) < candidate_bytes + ): + raise RuntimeError('bundle quarantine would make capacity accounting negative') + quarantine_capacity_items = bundle_items + projection_items + candidate_items + quarantine_capacity_bytes = bundle_bytes + projection_bytes + candidate_bytes + self.conn.execute( + '''UPDATE pipeline_capacity SET + bundle_items = bundle_items - ?, bundle_bytes = bundle_bytes - ?, + projection_items = projection_items - ?, projection_bytes = projection_bytes - ?, + keycheck_items = keycheck_items - ?, keycheck_bytes = keycheck_bytes - ?, + quarantine_items = quarantine_items + ?, + quarantine_bytes = quarantine_bytes + ?, updated_at = ? WHERE id = 1''', + ( + bundle_items, bundle_bytes, projection_items, projection_bytes, + candidate_items, candidate_bytes, quarantine_capacity_items, + quarantine_capacity_bytes, now, + ), + ) + quarantine_id = self.conn.insert_returning_id( + '''INSERT INTO pipeline_quarantine( + subsystem, object_type, object_id, reservation_id, event_id, + payload_sha256, reason_code, reason_detail, source_relative_path, + byte_count, capacity_items, capacity_bytes, detected_at + ) VALUES ('result_ingester', 'result_bundle', ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + reservation_id, reservation_id, row['scan_event_id'], payload_sha256, + reason_code, reason_detail, source_relative_path, + byte_count, quarantine_capacity_items, + quarantine_capacity_bytes, now, + ), + ) + if source_relative_path: + self.conn.execute( + '''INSERT INTO pipeline_artifacts( + subsystem, artifact_kind, owner_id, owner_key, relative_path, + payload_sha256, byte_count, state, created_at, updated_at + ) VALUES ('result_bundle','bundle_quarantine',?,'',?,?,?, + 'expected',?,?) + ON CONFLICT(subsystem, artifact_kind, owner_id, owner_key) + DO UPDATE SET relative_path = excluded.relative_path, + payload_sha256 = excluded.payload_sha256, + byte_count = excluded.byte_count, + deleted_at = NULL, updated_at = excluded.updated_at''', + ( + reservation_id, source_relative_path, + payload_sha256, byte_count, now, now, + ), + ) + if physical_confirmed: + self._finalize_result_bundle_quarantine_locked( + row, reason_code, reason_detail, source_relative_path, + payload_sha256, byte_count, now, + ) + self.conn.commit() + return int(quarantine_id) + except Exception: + self.conn.rollback() + raise + + def pipeline_capacity_snapshot(self): + if not self.conn: + return {} + row = self.conn.execute('SELECT * FROM pipeline_capacity WHERE id = 1').fetchone() + if self.conn.is_postgres: + self.conn.commit() + return dict(row) if row else {} + + @staticmethod + def _artifact_relative_path(relative_path): + value = str(relative_path or '').replace('\\', '/').strip('/') + if not value or value.startswith('/') or any(part in ('', '.', '..') for part in value.split('/')): + raise ValueError('pipeline artifact path is not a canonical relative path') + return value + + def register_pipeline_artifact( + self, subsystem, artifact_kind, owner_id, owner_key, relative_path, + *, state='expected', payload_sha256='', byte_count=0, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('pipeline artifact registration requires PostgreSQL') + if state not in ('expected', 'present', 'quarantined'): + raise ValueError('pipeline artifact registration state is invalid') + relative_path = self._artifact_relative_path(relative_path) + payload_sha256 = str(payload_sha256 or '').lower() + if payload_sha256 and not re.fullmatch(r'[a-f0-9]{64}', payload_sha256): + raise ValueError('pipeline artifact payload identity is invalid') + now = utc_now_iso() + try: + row = self.conn.execute( + '''SELECT * FROM pipeline_artifacts + WHERE subsystem = ? AND artifact_kind = ? + AND owner_id = ? AND owner_key = ? FOR UPDATE''', + (str(subsystem), str(artifact_kind), int(owner_id), str(owner_key or '')), + ).fetchone() + if row and str(row['relative_path']) != relative_path: + raise ScanEventConflictError('pipeline artifact owner resolves to a conflicting path') + if row and row['payload_sha256'] and payload_sha256 and row['payload_sha256'] != payload_sha256: + raise ScanEventConflictError('pipeline artifact owner resolves to conflicting bytes') + if row: + self.conn.execute( + '''UPDATE pipeline_artifacts SET payload_sha256 = ?, byte_count = ?, + state = ?, deleted_at = NULL, updated_at = ? WHERE id = ?''', + ( + payload_sha256 or row['payload_sha256'], max(0, int(byte_count)), + state, now, row['id'], + ), + ) + artifact_id = int(row['id']) + else: + artifact_id = self.conn.insert_returning_id( + '''INSERT INTO pipeline_artifacts( + subsystem, artifact_kind, owner_id, owner_key, relative_path, + payload_sha256, byte_count, state, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + str(subsystem), str(artifact_kind), int(owner_id), str(owner_key or ''), + relative_path, payload_sha256, max(0, int(byte_count)), state, now, now, + ), + ) + self.conn.commit() + return int(artifact_id) + except Exception: + self.conn.rollback() + raise + + def mark_pipeline_artifact_deleted(self, artifact_id): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('pipeline artifact deletion requires PostgreSQL') + now = utc_now_iso() + cursor = self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'deleted', byte_count = 0, + deleted_at = ?, cleanup_attempts = 0, cleanup_available_after = NULL, + cleanup_last_error = NULL, updated_at = ? + WHERE id = ? AND state != 'deleted' ''', + (now, now, int(artifact_id)), + ) + self.conn.commit() + return int(cursor.rowcount or 0) == 1 + + def defer_pipeline_artifact_cleanup(self, artifact_id, error): + if not self.conn or not self.conn.is_postgres: + return False + now = utc_now_iso() + try: + row = self.conn.execute( + 'SELECT cleanup_attempts FROM pipeline_artifacts WHERE id = ? FOR UPDATE', + (int(artifact_id),), + ).fetchone() + if not row: + self.conn.rollback() + return False + attempts = int(row['cleanup_attempts'] or 0) + 1 + delay = min(300, 2 ** min(attempts, 8)) + available = datetime.fromtimestamp( + time.time() + delay, timezone.utc, + ).isoformat(timespec='seconds') + self.conn.execute( + '''UPDATE pipeline_artifacts SET cleanup_attempts = ?, + cleanup_available_after = ?, cleanup_last_error = ?, updated_at = ? + WHERE id = ?''', + (attempts, available, first_line(error, 1000), now, int(artifact_id)), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + def projection_terminal_temp_artifacts(self, limit=100): + if not self.conn or not self.conn.is_postgres: + return [] + rows = self.conn.execute( + '''SELECT a.* FROM pipeline_artifacts a + JOIN projection_jobs j ON j.id = a.owner_id + WHERE a.subsystem = 'jsonl_projector' + AND a.artifact_kind IN ('prepared_stream','projection_tail_temp') + AND a.state IN ('expected','present') + AND j.status IN ('completed','quarantined') + ORDER BY a.id LIMIT ?''', + (min(1000, max(1, int(limit))),), + ).fetchall() + self.conn.commit() + return [dict(row) for row in rows] + + def bundle_terminal_temp_artifacts(self, limit=100): + if not self.conn or not self.conn.is_postgres: + return [] + rows = self.conn.execute( + '''SELECT a.* FROM pipeline_artifacts a + JOIN result_reservations r ON r.id = a.owner_id + WHERE a.subsystem = 'result_bundle' + AND a.artifact_kind IN ('bundle_partial','bundle_ready') + AND a.state IN ('expected','present') + AND r.state IN ('acknowledged','refunded','quarantined') + AND (a.cleanup_available_after IS NULL OR a.cleanup_available_after <= ?) + ORDER BY a.id LIMIT ?''', + (utc_now_iso(), min(1000, max(1, int(limit)))), + ).fetchall() + self.conn.commit() + return [dict(row) for row in rows] + + def register_projection_tail_quarantine( + self, job_id, append_id, stream_name, relative_path, payload_sha256, + byte_count, max_items, max_bytes, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection tail registration requires PostgreSQL') + relative_path = self._artifact_relative_path(relative_path) + payload_sha256 = str(payload_sha256 or '').lower() + if not re.fullmatch(r'[a-f0-9]{64}', payload_sha256): + raise ValueError('projection tail payload identity is invalid') + owner_key = f'{int(append_id)}:{stream_name}:{payload_sha256}' + now = utc_now_iso() + try: + artifact = self.conn.execute( + '''SELECT * FROM pipeline_artifacts + WHERE subsystem = 'jsonl_projector' AND artifact_kind = 'partial_tail' + AND owner_id = ? AND owner_key = ? FOR UPDATE''', + (int(job_id), owner_key), + ).fetchone() + if artifact and ( + artifact['relative_path'] != relative_path + or artifact['payload_sha256'] != payload_sha256 + or int(artifact['byte_count']) != int(byte_count) + ): + raise ScanEventConflictError('projection tail artifact identity conflicts') + if not artifact: + artifact_id = self.conn.insert_returning_id( + '''INSERT INTO pipeline_artifacts( + subsystem, artifact_kind, owner_id, owner_key, relative_path, + payload_sha256, byte_count, state, created_at, updated_at + ) VALUES ('jsonl_projector','partial_tail',?,?,?,?,?,'expected',?,?)''', + ( + int(job_id), owner_key, relative_path, payload_sha256, + max(0, int(byte_count)), now, now, + ), + ) + else: + artifact_id = int(artifact['id']) + quarantine = self.conn.execute( + '''SELECT id FROM pipeline_quarantine + WHERE subsystem = 'jsonl_projector' AND object_type = 'projection_tail' + AND object_id = ? AND review_status = 'pending' FOR UPDATE''', + (artifact_id,), + ).fetchone() + if not quarantine: + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + self.conn.execute( + '''UPDATE pipeline_capacity SET quarantine_items = quarantine_items + 1, + quarantine_bytes = quarantine_bytes + ?, updated_at = ? WHERE id = 1''', + (int(byte_count), now), + ) + quarantine_id = self.conn.insert_returning_id( + '''INSERT INTO pipeline_quarantine( + subsystem, object_type, object_id, projection_job_id, + payload_sha256, reason_code, reason_detail, source_relative_path, + byte_count, capacity_items, capacity_bytes, detected_at + ) VALUES ('jsonl_projector','projection_tail',?,?,?,?,?,?,?,?,?,?)''', + ( + artifact_id, int(job_id), payload_sha256, 'partial_projection_tail', + f'append={int(append_id)} stream={stream_name}', relative_path, + int(byte_count), 1, int(byte_count), now, + ), + ) + else: + quarantine_id = int(quarantine['id']) + self.conn.commit() + return {'artifact_id': int(artifact_id), 'quarantine_id': int(quarantine_id)} + except Exception: + self.conn.rollback() + raise + + def confirm_projection_tail_artifact(self, artifact_id, payload_sha256, byte_count): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection tail confirmation requires PostgreSQL') + now = utc_now_iso() + cursor = self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'quarantined', payload_sha256 = ?, + byte_count = ?, updated_at = ? + WHERE id = ? AND subsystem = 'jsonl_projector' + AND artifact_kind = 'partial_tail' + AND payload_sha256 = ? AND byte_count = ? + AND state IN ('expected','quarantined')''', + ( + str(payload_sha256), int(byte_count), now, int(artifact_id), + str(payload_sha256), int(byte_count), + ), + ) + self.conn.commit() + return int(cursor.rowcount or 0) == 1 + + def review_pipeline_quarantine( + self, quarantine_id, expected_reason_code, expected_payload_sha256, + action, audit_sha256, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('pipeline quarantine review requires PostgreSQL') + action = str(action or '').lower() + if action not in ('discard', 'rescan', 'retry'): + raise ValueError('quarantine review action must be discard, rescan, or retry') + if not re.fullmatch(r'[a-f0-9]{64}', str(audit_sha256 or '')): + raise ValueError('quarantine review audit identity is invalid') + now = utc_now_iso() + try: + preview = self.conn.execute( + '''SELECT object_type, reservation_id FROM pipeline_quarantine + WHERE id = ?''', + (int(quarantine_id),), + ).fetchone() + experiment = None + reservation = None + if ( + preview and str(preview['object_type']) == 'result_bundle' + and preview['reservation_id'] is not None + ): + experiment = self._lock_docker_depth_experiment_for_reservation( + preview['reservation_id'] + ) + reservation = self.conn.execute( + 'SELECT * FROM result_reservations WHERE id = ? FOR UPDATE', + (preview['reservation_id'],), + ).fetchone() + row = self.conn.execute( + 'SELECT * FROM pipeline_quarantine WHERE id = ? FOR UPDATE', + (int(quarantine_id),), + ).fetchone() + if not row: + raise ValueError('pipeline quarantine row is absent') + if row['review_status'] != 'pending': + if row['review_audit_sha256'] == audit_sha256: + self.conn.commit() + return {'reviewed': True, 'duplicate': True, 'status': row['review_status']} + raise ScanEventConflictError('pipeline quarantine row was already reviewed differently') + if ( + str(row['reason_code']) != str(expected_reason_code) + or str(row['payload_sha256'] or '') != str(expected_payload_sha256 or '') + ): + raise ScanEventConflictError('pipeline quarantine review evidence does not match') + if row['object_type'] == 'result_bundle': + if not preview or ( + str(preview['object_type']) != str(row['object_type']) + or int(preview['reservation_id'] or 0) + != int(row['reservation_id'] or 0) + ): + raise ScanEventConflictError( + 'pipeline quarantine review identity changed before locking' + ) + bundle = self.conn.execute( + 'SELECT * FROM result_bundles WHERE reservation_id = ? FOR UPDATE', + (row['reservation_id'],), + ).fetchone() + if not reservation or reservation['state'] != 'quarantined': + raise RuntimeError('bundle quarantine review requires terminal quarantined reservation state') + if bundle and bundle['state'] != 'quarantined': + raise RuntimeError('bundle quarantine review requires terminal quarantined bundle state') + elif row['object_type'] == 'projection_job': + job_state = self.conn.execute( + 'SELECT status FROM projection_jobs WHERE id = ? FOR UPDATE', + (row['projection_job_id'],), + ).fetchone() + if not job_state or job_state['status'] != 'quarantined': + raise RuntimeError('projection review requires terminal quarantined job state') + elif row['object_type'] == 'keycheck_candidate': + candidate_state = self.conn.execute( + 'SELECT * FROM keycheck_candidates WHERE id = ? FOR UPDATE', + (row['keycheck_candidate_id'],), + ).fetchone() + if not candidate_state or candidate_state['state'] != 'quarantined': + raise RuntimeError('keycheck review requires terminal quarantined candidate state') + if action in ('discard', 'rescan') and row['source_relative_path']: + artifact = None + if row['object_type'] == 'projection_tail': + artifact = self.conn.execute( + 'SELECT state FROM pipeline_artifacts WHERE id = ? FOR UPDATE', + (row['object_id'],), + ).fetchone() + elif row['object_type'] == 'result_bundle': + artifact = self.conn.execute( + '''SELECT state FROM pipeline_artifacts + WHERE subsystem = 'result_bundle' + AND artifact_kind = 'bundle_quarantine' AND owner_id = ? FOR UPDATE''', + (row['reservation_id'],), + ).fetchone() + if not artifact or artifact['state'] != 'deleted': + raise RuntimeError( + 'physical quarantine artifact must be absent before capacity credit release' + ) + if action == 'retry': + if not row['capacity_credit_applied']: + raise RuntimeError('quarantine retry requires exact applied capacity credit') + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + if ( + int(capacity['quarantine_items']) < int(row['capacity_items']) + or int(capacity['quarantine_bytes']) < int(row['capacity_bytes']) + ): + raise RuntimeError('quarantine retry would make capacity accounting negative') + if row['object_type'] == 'projection_job': + job = self.conn.execute( + 'SELECT * FROM projection_jobs WHERE id = ? FOR UPDATE', + (row['projection_job_id'],), + ).fetchone() + if not job or job['status'] != 'quarantined' or not job['capacity_released']: + raise RuntimeError('projection quarantine retry state is not exact') + prepared = self.conn.execute( + "SELECT 1 FROM projection_appends WHERE job_id = ? AND state = 'prepared' LIMIT 1", + (job['id'],), + ).fetchone() + if prepared: + raise RuntimeError('projection quarantine retry still has a prepared append') + self.conn.execute( + '''UPDATE pipeline_capacity SET + projection_items = projection_items + ?, + projection_bytes = projection_bytes + ?, + quarantine_items = quarantine_items - ?, + quarantine_bytes = quarantine_bytes - ?, updated_at = ? WHERE id = 1''', + ( + job['capacity_items'], job['capacity_bytes'], + row['capacity_items'], row['capacity_bytes'], now, + ), + ) + self.conn.execute( + '''UPDATE projection_jobs SET status = 'pending', capacity_released = 0, + available_after = NULL, lease_generation = NULL, lease_token = NULL, + lease_expires_at = NULL, last_error_code = NULL, + last_error_detail = NULL, completed_at = NULL, updated_at = ? + WHERE id = ?''', + (now, job['id']), + ) + elif row['object_type'] == 'keycheck_candidate': + candidate = candidate_state + active = self.conn.execute( + '''SELECT 1 FROM keycheck_candidates + WHERE credential_id = ? AND id <> ? + AND state IN ('pending','deferred','leased') LIMIT 1 FOR UPDATE''', + (candidate['credential_id'], candidate['id']), + ).fetchone() + if active: + raise RuntimeError('keycheck quarantine retry conflicts with another active credential candidate') + candidate_items = 1 + candidate_bytes = int(candidate['capacity_bytes']) + if ( + int(row['capacity_items']) < candidate_items + or int(row['capacity_bytes']) < candidate_bytes + ): + raise RuntimeError('keycheck quarantine retry has insufficient capacity credit') + self.conn.execute( + '''UPDATE pipeline_capacity SET + keycheck_items = keycheck_items + ?, + keycheck_bytes = keycheck_bytes + ?, + quarantine_items = quarantine_items - ?, + quarantine_bytes = quarantine_bytes - ?, updated_at = ? WHERE id = 1''', + ( + candidate_items, candidate_bytes, + row['capacity_items'], row['capacity_bytes'], now, + ), + ) + self.conn.execute( + '''UPDATE keycheck_candidates SET state = 'pending', attempts = 0, + available_after = NULL, capacity_released = 0, + result_projection_credit_transferred = 0, + result_projection_reserved_bytes = 0, + lease_owner = NULL, lease_token = NULL, lease_expires_at = NULL, + last_error = NULL, completed_at = NULL, updated_at = ? + WHERE id = ?''', + (now, candidate['id']), + ) + elif row['object_type'] == 'result_bundle': + if reservation['docker_layer_plan_json'] is not None: + raise ValueError( + 'Docker layer bundle quarantine requires a fresh parent claim' + ) + if not reservation['bundle_credit_released']: + raise RuntimeError('bundle quarantine retry requires released bundle capacity') + bundle_items = 1 + bundle_bytes = int(reservation['reserved_bundle_bytes']) + projection_items = ( + 0 if reservation['projection_credit_transferred'] + else int(reservation['reserved_projection_items']) + ) + projection_bytes = ( + 0 if reservation['projection_credit_transferred'] + else int(reservation['reserved_projection_bytes']) + ) + candidate_items = ( + 0 if reservation['candidate_credit_transferred'] + else int(reservation['reserved_candidate_items']) + ) + candidate_bytes = ( + 0 if reservation['candidate_credit_transferred'] + else int(reservation['reserved_candidate_bytes']) + ) + expected_items = bundle_items + projection_items + candidate_items + expected_bytes = bundle_bytes + projection_bytes + candidate_bytes + if ( + int(row['capacity_items']) != expected_items + or int(row['capacity_bytes']) != expected_bytes + ): + raise RuntimeError('bundle quarantine retry capacity evidence is inconsistent') + self.conn.execute( + '''UPDATE pipeline_capacity SET + bundle_items = bundle_items + ?, bundle_bytes = bundle_bytes + ?, + projection_items = projection_items + ?, projection_bytes = projection_bytes + ?, + keycheck_items = keycheck_items + ?, keycheck_bytes = keycheck_bytes + ?, + quarantine_items = quarantine_items - ?, + quarantine_bytes = quarantine_bytes - ?, updated_at = ? WHERE id = 1''', + ( + bundle_items, bundle_bytes, projection_items, projection_bytes, + candidate_items, candidate_bytes, + row['capacity_items'], row['capacity_bytes'], now, + ), + ) + self.conn.execute( + '''UPDATE result_reservations SET state = 'scanning', + bundle_credit_released = 0, released_at = NULL, + cleanup_attempts = 0, cleanup_available_after = NULL, + last_error_code = NULL, last_error_detail = NULL, updated_at = ? + WHERE id = ?''', + (now, reservation['id']), + ) + if bundle: + self.conn.execute( + '''UPDATE result_bundles SET state = 'ready', + relative_path = ?, available_after = NULL, + ingest_lease_generation = NULL, ingest_lease_token = NULL, + ingest_lease_expires_at = NULL, updated_at = ? + WHERE reservation_id = ?''', + (reservation['ready_relative_path'], now, reservation['id']), + ) + self.conn.execute( + '''UPDATE target_queue SET status = 'in_progress', completed_at = NULL, + last_error = NULL, lease_owner = ?, lease_token = ?, + claim_batch = ?, leased_at = ?, lease_expires_at = ?, + claim_event_id = ?, updated_at = ? + WHERE id = ? AND current_result_reservation_id = ?''', + ( + reservation['claim_lease_owner'], reservation['claim_lease_token'], + reservation['claim_batch'], now, reservation['producer_lease_expires_at'], + reservation['scan_event_id'], now, + reservation['queue_id'], reservation['id'], + ), + ) + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'deleted', byte_count = 0, + deleted_at = ?, updated_at = ? + WHERE subsystem = 'result_bundle' + AND artifact_kind = 'bundle_quarantine' AND owner_id = ?''', + (now, now, reservation['id']), + ) + self.conn.execute( + '''UPDATE pipeline_artifacts SET state = 'present', + relative_path = ?, payload_sha256 = ?, byte_count = ?, + deleted_at = NULL, updated_at = ? + WHERE subsystem = 'result_bundle' + AND artifact_kind = 'bundle_ready' AND owner_id = ?''', + ( + reservation['ready_relative_path'], row['payload_sha256'] or '', + int(row['byte_count']), now, reservation['id'], + ), + ) + self._transition_docker_depth_binding_locked( + reservation, 'reserved', 'reserved', now, + ) + else: + raise ValueError('quarantine object type does not support deterministic retry') + review_status = 'approved_retry' + elif action == 'rescan': + if ( + row['object_type'] != 'result_bundle' + or reservation['docker_layer_plan_json'] is None + or str(reservation['source']) != 'dockerhub' + or str(reservation['platform']) != 'docker' + ): + raise ValueError('quarantine rescan requires a Docker layer result bundle') + if not row['capacity_credit_applied']: + raise RuntimeError('quarantine rescan requires exact applied capacity credit') + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + if ( + int(capacity['quarantine_items']) < int(row['capacity_items']) + or int(capacity['quarantine_bytes']) < int(row['capacity_bytes']) + ): + raise RuntimeError('quarantine rescan would make capacity accounting negative') + queue = self.conn.execute( + 'SELECT * FROM target_queue WHERE id = ? FOR UPDATE', + (reservation['queue_id'],), + ).fetchone() + if not queue or ( + str(queue['source']) != 'dockerhub' + or str(queue['platform']) != 'docker' + or str(queue['status']) != 'quarantined' + or int(queue['current_result_reservation_id'] or 0) != int(reservation['id']) + or str(queue['claim_event_id'] or '') != str(reservation['scan_event_id']) + ): + raise ScanEventConflictError( + 'Docker quarantine rescan lost its exact queue fence' + ) + active_blob = self.conn.execute( + '''SELECT 1 FROM docker_content_blobs + WHERE lease_reservation_id = ? LIMIT 1 FOR UPDATE''', + (reservation['id'],), + ).fetchone() + if active_blob: + raise RuntimeError('Docker quarantine rescan retains active content work') + self.conn.execute( + '''UPDATE pipeline_capacity SET quarantine_items = quarantine_items - ?, + quarantine_bytes = quarantine_bytes - ?, updated_at = ? WHERE id = 1''', + (row['capacity_items'], row['capacity_bytes'], now), + ) + cursor = self.conn.execute( + '''UPDATE target_queue SET status = 'pending', attempts = 0, + available_after = NULL, completed_at = NULL, last_error = NULL, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, + leased_at = NULL, lease_expires_at = NULL, + current_result_reservation_id = NULL, claim_event_id = NULL, + updated_at = ? + WHERE id = ? AND status = 'quarantined' + AND current_result_reservation_id = ? AND claim_event_id = ?''', + ( + now, reservation['queue_id'], reservation['id'], + reservation['scan_event_id'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker quarantine rescan queue transition lost its fence' + ) + self._transition_docker_depth_binding_locked( + reservation, 'quarantined', 'pending', now, + ) + review_status = 'approved_rescan' + else: + if row['capacity_credit_applied']: + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + if ( + int(capacity['quarantine_items']) < int(row['capacity_items']) + or int(capacity['quarantine_bytes']) < int(row['capacity_bytes']) + ): + raise RuntimeError('quarantine discard would make capacity accounting negative') + self.conn.execute( + '''UPDATE pipeline_capacity SET quarantine_items = quarantine_items - ?, + quarantine_bytes = quarantine_bytes - ?, updated_at = ? WHERE id = 1''', + (row['capacity_items'], row['capacity_bytes'], now), + ) + review_status = 'discarded' + if experiment and action in ('retry', 'rescan'): + self._resume_docker_depth_after_quarantine_review_locked( + experiment, now, + ) + self.conn.execute( + '''UPDATE pipeline_quarantine SET review_status = ?, resolved_at = ?, + review_audit_sha256 = ? WHERE id = ?''', + (review_status, now, audit_sha256, row['id']), + ) + self.conn.commit() + return {'reviewed': True, 'duplicate': False, 'status': review_status} + except Exception: + self.conn.rollback() + raise + + def _resume_docker_depth_after_quarantine_review_locked(self, experiment, now): + if not experiment: + return + experiment = self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ? FOR UPDATE', + (int(experiment['id']),), + ).fetchone() + if not experiment: + raise ScanEventConflictError( + 'Docker depth quarantine review lost its experiment authority' + ) + state = str(experiment['state']) + if state in ('completed', 'released'): + raise ScanEventConflictError( + 'Docker depth terminal experiment cannot reopen quarantined work' + ) + if state == 'held': + if str(experiment['hold_reason_code'] or '') != 'scan_target_held': + return + remaining = self.conn.execute( + '''SELECT 1 FROM docker_depth_experiment_targets + WHERE experiment_id = ? AND state IN ('held','quarantined') + LIMIT 1 FOR UPDATE''', + (experiment['id'],), + ).fetchone() + if remaining: + return + elif state != 'draining': + return + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'active', hold_reason_code = NULL, held_at = NULL, + draining_at = NULL, updated_at = ? + WHERE id = ? AND state = ?''', + (now, experiment['id'], state), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth quarantine review lost its resume fence' + ) + + def pipeline_quarantine_for_review(self, quarantine_id): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('pipeline quarantine review lookup requires PostgreSQL') + row = self.conn.execute( + 'SELECT * FROM pipeline_quarantine WHERE id = ?', (int(quarantine_id),), + ).fetchone() + self.conn.commit() + return dict(row) if row else None + + def _hold_docker_depth_experiment_locked(self, experiment, reason, now=None): + now = str(now or utc_now_iso()) + reason = first_line(reason, 128) + if not experiment: + return {'status': 'unavailable', 'committed': False, 'reason': reason} + experiment_id = int(experiment['id']) + stable_reason = str(experiment['hold_reason_code'] or reason) + if str(experiment['state']) != 'released': + self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = CASE WHEN work_state = 'resolving' THEN 'pending' + ELSE work_state END, + resolver_owner = NULL, resolver_token = NULL, + resolver_expires_at = NULL, resolver_due_at = NULL, + updated_at = ? + WHERE experiment_id = ? AND work_state = 'resolving' ''', + (now, experiment_id), + ) + self.conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'held', hold_reason_code = COALESCE(hold_reason_code, ?), + held_at = COALESCE(held_at, ?), updated_at = ? + WHERE id = ? AND state <> 'released' ''', + (reason, now, now, experiment_id), + ) + return { + 'status': 'held', 'committed': True, 'reason': stable_reason, + 'experiment_id': experiment_id, + } + + @staticmethod + def _docker_depth_authority_drift_reason(experiment, authority): + if not experiment: + return 'experiment_authority_absent' + comparisons = ( + ('experiment_key', 'experiment_key', 'experiment_identity_drift'), + ('source', 'source', 'experiment_identity_drift'), + ('collection_generation', 'collection_generation', 'collection_generation_drift'), + ('config_sha256', 'config_sha256', 'config_hash_drift'), + ('ordered_queries_sha256', 'ordered_queries_sha256', 'ordered_query_drift'), + ('selector_version', 'selector_version', 'selector_drift'), + ('selector_sha256', 'selector_sha256', 'selector_drift'), + ('provenance_policy_sha256', 'provenance_policy_sha256', 'provenance_policy_drift'), + ) + for column, key, reason in comparisons: + if str(experiment[column]) != str(authority[key]): + return reason + for column, key in ( + ('query_count', 'query_count'), + ('repositories_per_query', 'repositories_per_query'), + ('images_per_repository', 'images_per_repository'), + ('target_limit', 'target_limit'), + ): + if int(experiment[column]) != int(authority[key]): + return 'capacity_limit_drift' + return None + + @staticmethod + def _docker_depth_candidate_skip_evidence( + experiment_id, experiment_repository_id, repository_queue_id, + candidate_kind, candidate_ordinal, candidate_identity, reason, + ): + candidate_identity_sha256 = hashlib.sha256(json.dumps( + candidate_identity, ensure_ascii=True, allow_nan=False, + sort_keys=True, separators=(',', ':'), + ).encode('utf-8')).hexdigest() + evidence = { + 'schema': 1, + 'type': 'docker-depth-candidate-skip-v1', + 'experiment_id': int(experiment_id), + 'experiment_repository_id': int(experiment_repository_id), + 'repository_queue_id': int(repository_queue_id), + 'candidate_kind': str(candidate_kind), + 'candidate_ordinal': int(candidate_ordinal), + 'candidate_identity_sha256': candidate_identity_sha256, + 'reason_code': str(reason), + } + evidence_sha256 = hashlib.sha256(json.dumps( + evidence, ensure_ascii=True, allow_nan=False, sort_keys=True, + separators=(',', ':'), + ).encode('utf-8')).hexdigest() + return candidate_identity_sha256, evidence_sha256 + + def _docker_depth_terminal_repository_evidence_locked( + self, experiment, member, + ): + from docker_depth_experiment import ( + DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON, + DOCKER_DEPTH_REPOSITORY_SKIP_REASON, + ) + + state = str(member['work_state']) + resolver_fields = ( + 'resolver_owner', 'resolver_token', 'resolver_expires_at', + 'resolver_due_at', + ) + if state == 'resolved': + if ( + int(member['selected_image_count']) < 1 + or not member['resolved_at'] + or member['last_error_code'] is not None + or any(member[name] is not None for name in resolver_fields) + ): + return 'repository_state_drift', None + return None, None + if state != 'skipped': + return None, None + current_queue_id = int( + member['replacement_repository_queue_id'] + or member['repository_queue_id'] + ) + skip_reason = str(member['last_error_code'] or '') + if ( + int(member['selected_image_count']) != 0 + or skip_reason not in ( + DOCKER_DEPTH_REPOSITORY_SKIP_REASON, + DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON, + ) + or not member['resolved_at'] + or any(member[name] is not None for name in resolver_fields) + ): + return 'repository_skip_evidence_drift', None + if self.conn.execute( + '''SELECT 1 FROM docker_depth_experiment_selections + WHERE experiment_repository_id = ? LIMIT 1''', + (member['id'],), + ).fetchone(): + return 'repository_skip_evidence_drift', None + rows = self.conn.execute( + '''SELECT candidate_ordinal, candidate_identity_sha256, + evidence_sha256 + FROM docker_depth_experiment_candidate_skips + WHERE experiment_id = ? AND experiment_repository_id = ? + AND repository_queue_id = ? AND candidate_kind = 'repository' + AND reason_code = ? + ORDER BY id LIMIT 2''', + ( + experiment['id'], member['id'], current_queue_id, + skip_reason, + ), + ).fetchall() + expected_ordinal = int(member['replacement_count']) + 1 + identity_sha256, evidence_sha256 = self._docker_depth_candidate_skip_evidence( + experiment['id'], member['id'], current_queue_id, 'repository', + expected_ordinal, {'repository_queue_id': current_queue_id}, + skip_reason, + ) + if ( + len(rows) != 1 + or int(rows[0]['candidate_ordinal']) != expected_ordinal + or str(rows[0]['candidate_identity_sha256']) != identity_sha256 + or str(rows[0]['evidence_sha256']) != evidence_sha256 + ): + return 'repository_skip_evidence_drift', None + return None, evidence_sha256 + + def _docker_depth_runtime_selection_sha256_locked(self, experiment, authority): + from docker_depth_experiment import DOCKER_RANK1_BREADTH_SELECTOR_VERSION + + rank1_breadth = ( + authority.get('selector_version') == DOCKER_RANK1_BREADTH_SELECTOR_VERSION + ) + query_rows = self.conn.execute( + '''SELECT query_ordinal, required_repository_count, + selected_repository_count + FROM docker_depth_experiment_queries + WHERE experiment_id = ? ORDER BY query_ordinal''', + (experiment['id'],), + ).fetchall() + if len(query_rows) != authority['query_count']: + raise ScanEventConflictError('Docker depth runtime query selection is incomplete') + selected_counts = [] + for query_ordinal, row in enumerate(query_rows): + selected_count = int(row['selected_repository_count']) + if ( + int(row['query_ordinal']) != query_ordinal + or int(row['required_repository_count']) + != authority['repositories_per_query'] + or not 0 <= selected_count <= authority['repositories_per_query'] + ): + raise ScanEventConflictError('Docker depth runtime query selection is invalid') + selected_counts.append(selected_count) + rows = self.conn.execute( + '''SELECT id, query_ordinal, repository_rank, repository_queue_id, + replacement_repository_queue_id, + replacement_eligibility_page_id, replacement_count, + replacement_evidence_sha256, is_deep_probe, + work_state, selected_image_count, last_error_code, + resolved_at, resolver_owner, resolver_token, + resolver_expires_at, resolver_due_at + FROM docker_depth_experiment_repositories + WHERE experiment_id = ? + ORDER BY query_ordinal, repository_rank, id''', + (experiment['id'],), + ).fetchall() + expected = sum(selected_counts) + if len(rows) != expected: + raise ScanEventConflictError('Docker depth runtime selection is incomplete') + document = { + 'schema': 2, + 'type': 'docker-depth-runtime-selection-v2', + 'experiment_id': int(experiment['id']), + 'plan_sha256': str(experiment['plan_sha256'] or ''), + 'collection_generation': authority['collection_generation'], + 'queries': [], + } + for query_ordinal in range(authority['query_count']): + members = [ + row for row in rows if int(row['query_ordinal']) == query_ordinal + ] + terminal_evidence = {} + for row in members: + reason, evidence_sha256 = ( + self._docker_depth_terminal_repository_evidence_locked( + experiment, row, + ) + ) + if reason: + raise ScanEventConflictError( + 'Docker depth runtime repository evidence is invalid' + ) + terminal_evidence[int(row['id'])] = evidence_sha256 + image_bearing = sum( + 1 for row in members + if str(row['work_state']) == 'resolved' + and int(row['selected_image_count']) >= 1 + ) + expected_deep_probe_count = 0 if rank1_breadth else (1 if image_bearing else 0) + if ( + len(members) != selected_counts[query_ordinal] + or [int(row['repository_rank']) for row in members] + != list(range(1, selected_counts[query_ordinal] + 1)) + or sum(int(row['is_deep_probe']) for row in members) + != expected_deep_probe_count + or any( + str(row['work_state']) not in ('resolved', 'skipped') + for row in members + ) + ): + raise ScanEventConflictError('Docker depth runtime selection is invalid') + document['queries'].append({ + 'query_ordinal': query_ordinal, + 'selected_repository_count': selected_counts[query_ordinal], + 'repositories': [{ + 'member_id': int(row['id']), + 'repository_rank': int(row['repository_rank']), + 'planned_repository_queue_id': int(row['repository_queue_id']), + 'selected_repository_queue_id': int( + row['replacement_repository_queue_id'] + or row['repository_queue_id'] + ), + 'replacement_eligibility_page_id': ( + int(row['replacement_eligibility_page_id']) + if row['replacement_eligibility_page_id'] is not None else None + ), + 'replacement_count': int(row['replacement_count']), + 'replacement_evidence_sha256': str( + row['replacement_evidence_sha256'] or '' + ), + 'is_deep_probe': bool(row['is_deep_probe']), + 'work_state': str(row['work_state']), + 'selected_image_count': int(row['selected_image_count']), + 'terminal_reason': ( + str(row['last_error_code']) + if str(row['work_state']) == 'skipped' else None + ), + 'skip_evidence_sha256': terminal_evidence[int(row['id'])], + } for row in members], + }) + payload = json.dumps( + document, ensure_ascii=True, allow_nan=False, sort_keys=True, + separators=(',', ':'), + ).encode('utf-8') + return hashlib.sha256(payload).hexdigest() + + def _freeze_docker_depth_runtime_selection_locked( + self, experiment, authority, now, + ): + incomplete = self.conn.execute( + '''SELECT 1 FROM docker_depth_experiment_repositories + WHERE experiment_id = ? + AND work_state NOT IN ('resolved','skipped') LIMIT 1''', + (experiment['id'],), + ).fetchone() + if incomplete: + return experiment, None + rows = self.conn.execute( + '''SELECT * FROM docker_depth_experiment_repositories + WHERE experiment_id = ? ORDER BY id''', + (experiment['id'],), + ).fetchall() + for row in rows: + reason, _evidence_sha256 = ( + self._docker_depth_terminal_repository_evidence_locked( + experiment, row, + ) + ) + if reason: + return experiment, reason + selection_sha256 = self._docker_depth_runtime_selection_sha256_locked( + experiment, authority, + ) + stored = str(experiment['selection_sha256'] or '') + if stored and stored != selection_sha256: + return experiment, 'selection_hash_drift' + if not stored: + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET selection_sha256 = ?, updated_at = ? + WHERE id = ? AND selection_sha256 IS NULL + AND state = 'resolving' ''', + (selection_sha256, now, experiment['id']), + ) + if int(cursor.rowcount or 0) != 1: + return experiment, 'selection_hash_drift' + experiment = self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment['id'],), + ).fetchone() + return experiment, None + + def _docker_depth_hold_event_drift_reason_locked(self, experiment, authority): + from docker_depth_experiment import ( + DOCKER_DEPTH_DYNAMIC_HOLD_REASON, + DOCKER_DEPTH_HOLD_REASON, + DOCKER_DEPTH_RELEASE_REASON, + DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, + ) + + experiment_id = int(experiment['id']) + hold_sha256 = str(experiment['hold_manifest_sha256'] or '') + rows = self.conn.execute( + '''SELECT event.*, queue.status AS queue_status, + queue.source AS queue_source, + queue.platform AS queue_platform, + queue.query AS queue_query, + queue.updated_at AS queue_updated_at + FROM target_queue_policy_events event + JOIN target_queue queue ON queue.id = event.queue_id + WHERE event.experiment_id = ? + ORDER BY event.id LIMIT ?''', + (experiment_id, DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS + 1), + ).fetchall() + if len(rows) > DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS: + return 'target_history_drift' + reversed_ids = { + int(row['reverses_event_id']) for row in rows + if row['reverses_event_id'] is not None + } + for row in rows: + if ( + int(row['queue_id']) < 1 + or str(row['source']) != authority['source'] + or str(row['platform']) != 'docker' + or not str(row['query']) + or str(row['queue_source']) != str(row['source']) + or str(row['queue_platform']) != str(row['platform']) + or str(row['queue_query']) != str(row['query']) + or str(row['config_sha256']) != authority['config_sha256'] + or str(row['policy_sha256']) + != authority['provenance_policy_sha256'] + ): + return 'target_history_drift' + if row['action'] == 'cold': + entry = { + 'queue_id': int(row['queue_id']), + 'source': str(row['source']), + 'platform': str(row['platform']), + 'query': str(row['query']), + 'prior_status': str(row['prior_status']), + 'prior_updated_at': str(row['prior_updated_at']), + } + if ( + str(row['manifest_sha256']) != hold_sha256 + or str(row['reason_code']) not in ( + DOCKER_DEPTH_HOLD_REASON, DOCKER_DEPTH_DYNAMIC_HOLD_REASON, + ) + or str(row['prior_status']) not in ('pending', 'deferred') + or str(row['next_status']) != 'cold' + or int(row['id']) in reversed_ids + or str(row['queue_status']) != 'cold' + or str(row['queue_updated_at']) != str(row['created_at']) + ): + return 'target_history_drift' + elif row['action'] == 'reactivate': + # Reactivation and the completed -> released transition share one + # experiment-row lock, so it is never valid in a pre-release state. + if str(row['reason_code']) != DOCKER_DEPTH_RELEASE_REASON: + return 'target_history_drift' + return 'target_history_drift' + else: + return 'target_history_drift' + if str(row['review_audit_sha256']) != self._target_queue_policy_audit_sha256( + str(row['action']), str(row['manifest_sha256']), entry, + experiment_id, + ): + return 'target_history_drift' + return None + + def _docker_depth_persisted_drift_reason_locked(self, experiment, authority, now): + from docker_depth_experiment import ( + DOCKER_RANK1_BREADTH_SELECTOR_VERSION, + _cohort_plan_document, + canonical_docker_depth_plan_hash, + ) + + experiment_id = int(experiment['id']) + plan_sha256 = str(experiment['plan_sha256'] or '') + hold_sha256 = str(experiment['hold_manifest_sha256'] or '') + if any( + experiment[name] is not None + for name in ('fence_owner', 'fence_token', 'fence_expires_at') + ): + return 'experiment_fence_conflict' + if not re.fullmatch(r'[a-f0-9]{64}', plan_sha256): + return 'plan_hash_drift' + if not re.fullmatch(r'[a-f0-9]{64}', hold_sha256): + return 'hold_hash_drift' + + query_rows = self.conn.execute( + '''SELECT query_ordinal, source, query, query_sha256, + required_repository_count, selected_repository_count + FROM docker_depth_experiment_queries + WHERE experiment_id = ? ORDER BY query_ordinal LIMIT ?''', + (experiment_id, authority['query_count'] + 1), + ).fetchall() + if len(query_rows) != authority['query_count']: + return 'ordered_query_drift' + selected_counts = [] + for ordinal, row in enumerate(query_rows): + query = authority['queries'][ordinal] + selected_count = int(row['selected_repository_count']) + if ( + int(row['query_ordinal']) != ordinal + or str(row['source']) != authority['source'] + or str(row['query']) != query + or str(row['query_sha256']) != hashlib.sha256( + json.dumps( + query, ensure_ascii=True, allow_nan=False, sort_keys=True, + separators=(',', ':'), + ).encode('utf-8') + ).hexdigest() + or int(row['required_repository_count']) + != authority['repositories_per_query'] + or not 0 <= selected_count <= authority['repositories_per_query'] + ): + return 'ordered_query_drift' + selected_counts.append(selected_count) + + expected_repository_count = sum(selected_counts) + repository_rows = self.conn.execute( + '''SELECT id, query_ordinal, source, query, repository_queue_id, + eligibility_page_id, repository_rank, planned_is_deep_probe, + is_deep_probe, work_state, resolver_owner, resolver_token, + resolver_expires_at, resolver_due_at, + candidate_distinct_graph_count, + selected_image_count, replacement_repository_queue_id, + replacement_eligibility_page_id, replacement_count, + replacement_evidence_sha256, last_error_code, resolved_at + FROM docker_depth_experiment_repositories + WHERE experiment_id = ? + ORDER BY query_ordinal, repository_rank, id LIMIT ?''', + (experiment_id, expected_repository_count + 1), + ).fetchall() + if len(repository_rows) != expected_repository_count: + return 'repository_count_drift' + if str(experiment['state']) in ('active', 'draining', 'completed') and any( + str(row['work_state']) not in ('resolved', 'skipped') + for row in repository_rows + ): + return 'repository_state_drift' + repositories_by_query = { + ordinal: [] for ordinal in range(authority['query_count']) + } + for row in repository_rows: + ordinal = int(row['query_ordinal']) + if ordinal not in repositories_by_query: + return 'plan_hash_drift' + repositories_by_query[ordinal].append(row) + replacement_count = int(row['replacement_count']) + replacement_present = ( + row['replacement_repository_queue_id'] is not None + and row['replacement_eligibility_page_id'] is not None + and bool(re.fullmatch( + r'[a-f0-9]{64}', str(row['replacement_evidence_sha256'] or '') + )) + ) + if ( + replacement_count < 0 + or (replacement_count == 0 and ( + row['replacement_repository_queue_id'] is not None + or row['replacement_eligibility_page_id'] is not None + or row['replacement_evidence_sha256'] is not None + )) + or (replacement_count > 0 and not replacement_present) + ): + return 'selection_hash_drift' + if replacement_present and not self.conn.execute( + '''SELECT 1 + FROM docker_repository_query_observations observation + JOIN docker_discovery_pages page ON page.id = observation.page_id + JOIN docker_discovery_passes discovery_pass + ON discovery_pass.id = page.pass_id + JOIN target_queue queue + ON queue.id = observation.repository_queue_id + WHERE observation.page_id = ? + AND observation.repository_queue_id = ? + AND observation.source = ? AND observation.query = ? + AND page.query_ordinal = ? AND page.query = ? + AND discovery_pass.source = ? + AND discovery_pass.pass_kind = 'deep' + AND discovery_pass.collection_generation = ? + AND discovery_pass.policy_sha256 = ? + AND discovery_pass.ordered_queries_sha256 = ? + AND discovery_pass.expected_query_count = ? + AND discovery_pass.state = 'complete' + AND queue.source = ? AND queue.platform = 'docker' + AND queue.target NOT LIKE '%@%' + AND queue.normalized_target NOT LIKE '%@%' + LIMIT 1''', + ( + row['replacement_eligibility_page_id'], + row['replacement_repository_queue_id'], + authority['source'], authority['queries'][ordinal], ordinal, + authority['queries'][ordinal], authority['source'], + authority['collection_generation'], + authority['provenance_policy_sha256'], + authority['ordered_queries_sha256'], authority['query_count'], + authority['source'], + ), + ).fetchone(): + return 'selection_hash_drift' + resolving = str(row['work_state']) == 'resolving' + if resolving and str(experiment['state']) != 'resolving': + return 'stale_resolver_fence' + if resolving and ( + not row['resolver_owner'] + or not row['resolver_token'] + or not row['resolver_expires_at'] + or str(row['resolver_expires_at']) <= now + ): + return 'stale_resolver_fence' + if not resolving and any( + row[name] is not None + for name in ('resolver_owner', 'resolver_token', 'resolver_expires_at') + ): + return 'stale_resolver_fence' + terminal_reason, _evidence_sha256 = ( + self._docker_depth_terminal_repository_evidence_locked( + experiment, row, + ) + ) + if terminal_reason: + return terminal_reason + + planned_queries = [] + for ordinal, query_row in enumerate(query_rows): + members = repositories_by_query[ordinal] + selected_count = selected_counts[ordinal] + if ( + [int(row['repository_rank']) for row in members] + != list(range(1, selected_count + 1)) + or any( + str(row['source']) != authority['source'] + or str(row['query']) != authority['queries'][ordinal] + or int(row['eligibility_page_id']) < 1 + for row in members + ) + or sum(int(row['planned_is_deep_probe']) for row in members) + != (1 if selected_count else 0) + ): + return 'plan_hash_drift' + all_terminal = all( + str(row['work_state']) in ('resolved', 'skipped') for row in members + ) + image_bearing = any( + str(row['work_state']) == 'resolved' + and int(row['selected_image_count']) >= 1 + for row in members + ) + if ( + str(experiment['selector_version']) + == DOCKER_RANK1_BREADTH_SELECTOR_VERSION + and all_terminal + ): + expected_deep_probe_count = 0 + else: + expected_deep_probe_count = ( + (1 if image_bearing else 0) + if all_terminal else (1 if selected_count else 0) + ) + if sum(int(row['is_deep_probe']) for row in members) != ( + expected_deep_probe_count + ): + return 'selection_mutation' + planned_queries.append({ + 'query_ordinal': ordinal, + 'query': authority['queries'][ordinal], + 'query_sha256': str(query_row['query_sha256']), + 'selected_repository_count': selected_count, + 'repositories': [{ + 'repository_queue_id': int(row['repository_queue_id']), + 'eligibility_page_id': int(row['eligibility_page_id']), + 'repository_rank': int(row['repository_rank']), + 'is_deep_probe': bool(row['planned_is_deep_probe']), + } for row in members], + }) + if canonical_docker_depth_plan_hash( + _cohort_plan_document(authority, planned_queries) + ) != plan_sha256: + return 'plan_hash_drift' + selection_sha256 = str(experiment['selection_sha256'] or '') + if selection_sha256: + if ( + not re.fullmatch(r'[a-f0-9]{64}', selection_sha256) + or self._docker_depth_runtime_selection_sha256_locked( + experiment, authority, + ) != selection_sha256 + ): + return 'selection_hash_drift' + elif str(experiment['state']) in ('active', 'draining'): + return 'selection_hash_drift' + + target_rows = self.conn.execute( + '''SELECT target.id AS target_id, target.target_queue_id, + target.manifest_id, target.counter_ordinal, + target.state AS target_state, target.dispatch_wave, + target.dispatch_order, target.reservation_count, + queue.source AS queue_source, queue.platform AS queue_platform, + queue.target AS queue_target, + queue.normalized_target AS queue_normalized_target, + queue.status AS queue_status, queue.lease_owner, + queue.lease_token, queue.claim_batch, queue.leased_at, + queue.lease_expires_at, queue.current_result_reservation_id, + queue.claim_event_id, queue.resolver_token, + manifest.repository, manifest.manifest_digest, + manifest.graph_sha256 AS manifest_graph_sha256 + FROM docker_depth_experiment_targets target + JOIN target_queue queue ON queue.id = target.target_queue_id + JOIN docker_image_manifests manifest + ON manifest.id = target.manifest_id + AND manifest.target_queue_id = target.target_queue_id + WHERE target.experiment_id = ? + ORDER BY target.id LIMIT ?''', + (experiment_id, authority['target_limit'] + 1), + ).fetchall() + if ( + len(target_rows) != int(experiment['target_count']) + or len(target_rows) > authority['target_limit'] + ): + return 'target_counter_drift' + target_by_id = {int(row['target_id']): row for row in target_rows} + if len(target_by_id) != len(target_rows): + return 'target_counter_drift' + + selection_rows = self.conn.execute( + '''SELECT selection.id, selection.query_ordinal, + selection.experiment_repository_id, + selection.experiment_target_id, selection.image_rank, + selection.selection_evidence_sha256, + selection.graph_sha256, + member.query_ordinal AS member_query_ordinal, + member.is_deep_probe + FROM docker_depth_experiment_selections selection + JOIN docker_depth_experiment_repositories member + ON member.id = selection.experiment_repository_id + AND member.experiment_id = selection.experiment_id + WHERE selection.experiment_id = ? + ORDER BY selection.id LIMIT ?''', + (experiment_id, authority['theoretical_max_targets'] + 1), + ).fetchall() + if ( + len(selection_rows) != int(experiment['selection_count']) + or len(selection_rows) > authority['theoretical_max_targets'] + ): + return 'selection_counter_drift' + selected_by_member = {} + selected_ranks_by_member = {} + dispatch_by_target = {} + selected_target_ids = set() + for selection in selection_rows: + target_id = int(selection['experiment_target_id']) + target = target_by_id.get(target_id) + rank = int(selection['image_rank']) + if ( + not target + or int(selection['query_ordinal']) + != int(selection['member_query_ordinal']) + or (rank > 1 and not bool(selection['is_deep_probe'])) + or str(selection['graph_sha256']) + != str(target['manifest_graph_sha256']) + or not re.fullmatch( + r'[a-f0-9]{64}', str(selection['selection_evidence_sha256'] or '') + ) + ): + return 'selection_mutation' + member_id = int(selection['experiment_repository_id']) + selected_by_member[member_id] = selected_by_member.get(member_id, 0) + 1 + selected_ranks_by_member.setdefault(member_id, []).append(rank) + position = self._docker_depth_dispatch_position( + int(selection['query_ordinal']), + next( + int(row['repository_rank']) for row in repository_rows + if int(row['id']) == member_id + ), + rank, authority['query_count'], + ) + dispatch_by_target[target_id] = min( + dispatch_by_target.get(target_id, position), position, + ) + selected_target_ids.add(target_id) + if selected_target_ids != set(target_by_id): + return 'selection_mutation' + for member in repository_rows: + member_id = int(member['id']) + selected_count = selected_by_member.get(member_id, 0) + if ( + selected_count != int(member['selected_image_count']) + or sorted(selected_ranks_by_member.get(member_id, ())) + != list(range(1, selected_count + 1)) + ): + return 'selection_counter_drift' + + for target_id, target in target_by_id.items(): + expected_target = f"{target['repository']}@{target['manifest_digest']}" + if ( + (int(target['dispatch_wave']), int(target['dispatch_order'])) + != dispatch_by_target[target_id] + or str(target['queue_source']) != authority['source'] + or str(target['queue_platform']) != 'docker' + or str(target['queue_target']) != expected_target + or str(target['queue_normalized_target']) != expected_target + ): + return 'selection_mutation' + target_state = str(target['target_state']) + queue_status = str(target['queue_status']) + if target_state in ('held', 'quarantined'): + return 'scan_target_held' + if target_state == 'pending' and queue_status not in ('pending', 'deferred'): + return 'stale_scan_fence' + if target_state in ('reserved', 'scanning') and queue_status != 'in_progress': + return 'stale_scan_fence' + if target_state == 'done' and queue_status != 'done': + return 'stale_scan_fence' + if target_state == 'failed' and queue_status != 'failed': + return 'stale_scan_fence' + if target_state == 'quarantined' and queue_status != 'quarantined': + return 'stale_scan_fence' + if target_state == 'pending' and any( + target[name] is not None for name in ( + 'lease_owner', 'lease_token', 'claim_batch', 'leased_at', + 'lease_expires_at', 'current_result_reservation_id', + 'claim_event_id', 'resolver_token', + ) + ): + return 'stale_scan_fence' + + binding_rows = self.conn.execute( + '''SELECT binding.id, binding.experiment_target_id, + binding.reservation_id, binding.target_scan_id, + binding.attempt, binding.state AS binding_state, + target.target_queue_id, + reservation.queue_id AS reservation_queue_id, + reservation.state AS reservation_state, + reservation.claim_lease_token, + reservation.scan_event_id, + reservation.producer_lease_expires_at, + scan.queue_id AS scan_queue_id, + scan.result_reservation_id AS scan_reservation_id + FROM docker_depth_experiment_scan_bindings binding + JOIN docker_depth_experiment_targets target + ON target.id = binding.experiment_target_id + JOIN result_reservations reservation + ON reservation.id = binding.reservation_id + LEFT JOIN target_scans scan ON scan.id = binding.target_scan_id + WHERE target.experiment_id = ? + ORDER BY binding.experiment_target_id, binding.attempt + LIMIT ?''', + (experiment_id, authority['target_limit'] * 10 + 1), + ).fetchall() + if len(binding_rows) > authority['target_limit'] * 10: + return 'binding_count_drift' + binding_count = {} + active_by_target = {} + latest_by_target = {} + for binding in binding_rows: + target_id = int(binding['experiment_target_id']) + state = str(binding['binding_state']) + reservation_state = str(binding['reservation_state']) + if ( + target_id not in target_by_id + or int(binding['target_queue_id']) + != int(binding['reservation_queue_id']) + or int(binding['attempt']) < 1 + ): + return 'reservation_binding_drift' + has_scan = binding['target_scan_id'] is not None + if has_scan and ( + int(binding['scan_queue_id'] or 0) != int(binding['target_queue_id']) + or int(binding['scan_reservation_id'] or 0) + != int(binding['reservation_id']) + ): + return 'reservation_binding_drift' + if state == 'reserved' and (reservation_state != 'scanning' or has_scan): + return 'reservation_binding_drift' + if state == 'scanning' and ( + reservation_state not in ('ready', 'ingesting') or has_scan + ): + return 'reservation_binding_drift' + if state in ('completed', 'failed') and ( + reservation_state not in ('db_committed', 'acknowledged') or not has_scan + ): + return 'reservation_binding_drift' + if state == 'released' and (reservation_state != 'refunded' or has_scan): + return 'reservation_binding_drift' + if state == 'quarantined' and reservation_state != 'quarantined': + return 'reservation_binding_drift' + if state in ('reserved', 'scanning'): + if ( + state == 'reserved' + and str(binding['producer_lease_expires_at'] or '') <= now + ): + return 'stale_scan_fence' + active_by_target.setdefault(target_id, []).append(binding) + binding_count[target_id] = binding_count.get(target_id, 0) + 1 + latest_by_target[target_id] = binding + for target_id, target in target_by_id.items(): + if int(target['reservation_count']) != binding_count.get(target_id, 0): + return 'binding_count_drift' + active = active_by_target.get(target_id, []) + target_state = str(target['target_state']) + latest = latest_by_target.get(target_id) + if target_state in ('done', 'failed') and ( + latest is None + or str(latest['binding_state']) + != ('completed' if target_state == 'done' else 'failed') + ): + return 'reservation_binding_drift' + if target_state in ('reserved', 'scanning'): + if len(active) != 1: + return 'stale_scan_fence' + binding = active[0] + if ( + int(target['current_result_reservation_id'] or 0) + != int(binding['reservation_id']) + or str(target['lease_token'] or '') + != str(binding['claim_lease_token']) + or str(target['claim_event_id'] or '') + != str(binding['scan_event_id']) + ): + return 'stale_scan_fence' + elif active: + return 'stale_scan_fence' + + unbound = self.conn.execute( + '''SELECT + EXISTS( + SELECT 1 FROM result_reservations reservation + JOIN docker_depth_experiment_targets target + ON target.target_queue_id = reservation.queue_id + LEFT JOIN docker_depth_experiment_scan_bindings binding + ON binding.experiment_target_id = target.id + AND binding.reservation_id = reservation.id + WHERE target.experiment_id = ? AND binding.id IS NULL + ) AS reservation_missing, + EXISTS( + SELECT 1 FROM target_scans scan + JOIN docker_depth_experiment_targets target + ON target.target_queue_id = scan.queue_id + LEFT JOIN docker_depth_experiment_scan_bindings binding + ON binding.experiment_target_id = target.id + AND binding.target_scan_id = scan.id + WHERE target.experiment_id = ? AND binding.id IS NULL + ) AS scan_missing''', + (experiment_id, experiment_id), + ).fetchone() + if bool(unbound['reservation_missing']) or bool(unbound['scan_missing']): + return 'historical_scan_conflict' + + return self._docker_depth_hold_event_drift_reason_locked( + experiment, authority, + ) + + def _locked_docker_depth_experiment_authority( + self, authority, *, final_cutover, now=None, lock=True, + ): + now = str(now or utc_now_iso()) + experiment = self.conn.execute( + '''SELECT * FROM docker_depth_experiments + WHERE experiment_key = ? AND source = ?''' + ( + ' FOR UPDATE' if lock else '' + ), + (authority['experiment_key'], authority['source']), + ).fetchone() + if not experiment: + return None, 'experiment_authority_absent' + reason = self._docker_depth_authority_drift_reason(experiment, authority) + marker = self.conn.execute( + '''SELECT marker, checked_at, evidence_sha256 + FROM runtime_final_cutover WHERE id = 1''' + ).fetchone() + if reason is None and ( + final_cutover is not True + or not marker + or str(marker['marker']) != FINAL_CUTOVER_MARKER + or not marker['checked_at'] + or not re.fullmatch(r'[a-f0-9]{64}', str(marker['evidence_sha256'] or '')) + ): + reason = 'final_cutover_unavailable' + if reason is None and str(experiment['state']) in ( + 'holding', 'resolving', 'active', 'draining', 'completed', 'held', + ): + reason = self._docker_depth_persisted_drift_reason_locked( + experiment, authority, now, + ) + if reason: + self._hold_docker_depth_experiment_locked(experiment, reason, now) + return None, reason + return experiment, None + + def _docker_depth_incomplete_membership_count_locked( + self, experiment_id, *, activation=False, + ): + activation_clause = ( + "OR target.state <> 'pending' OR queue.status NOT IN ('pending','deferred') " + "OR queue.lease_owner IS NOT NULL OR queue.lease_token IS NOT NULL " + "OR queue.claim_batch IS NOT NULL OR queue.leased_at IS NOT NULL " + "OR queue.lease_expires_at IS NOT NULL " + "OR queue.current_result_reservation_id IS NOT NULL " + "OR queue.claim_event_id IS NOT NULL " + "OR EXISTS (SELECT 1 FROM target_scans scan " + " WHERE scan.queue_id = queue.id) " + "OR EXISTS (SELECT 1 FROM result_reservations reservation " + " WHERE reservation.queue_id = queue.id) " + "OR EXISTS (SELECT 1 FROM target_queue_policy_events event " + " WHERE event.queue_id = queue.id) " + "OR EXISTS (SELECT 1 FROM docker_image_blob_coverage coverage " + " JOIN docker_content_blobs blob " + " ON blob.digest = coverage.blob_digest " + " AND blob.coverage_policy_sha256 = coverage.coverage_policy_sha256 " + " WHERE coverage.queue_id = queue.id " + " AND blob.state IN ('leased','submitted'))" + if activation else '' + ) + return int(self.conn.execute( + f'''SELECT COUNT(*) AS count + FROM docker_depth_experiment_repositories member + LEFT JOIN docker_depth_experiment_selections selection + ON selection.experiment_repository_id = member.id + AND selection.image_rank = 1 + LEFT JOIN docker_depth_experiment_targets target + ON target.id = selection.experiment_target_id + AND target.experiment_id = member.experiment_id + LEFT JOIN target_queue queue ON queue.id = target.target_queue_id + WHERE member.experiment_id = ? + AND member.work_state <> 'skipped' + AND (member.selected_image_count < 1 OR selection.id IS NULL + OR target.id IS NULL OR queue.id IS NULL {activation_clause})''', + (experiment_id,), + ).fetchone()['count']) + + def _advance_docker_depth_experiment_state_locked(self, experiment, now=None): + now = str(now or utc_now_iso()) + experiment_id = int(experiment['id']) + state = str(experiment['state']) + unresolved = int(self.conn.execute( + '''SELECT COUNT(*) AS count + FROM docker_depth_experiment_repositories + WHERE experiment_id = ? + AND work_state NOT IN ('resolved','skipped')''', + (experiment_id,), + ).fetchone()['count']) + if state == 'resolving' and unresolved == 0: + from docker_depth_experiment import ( + DOCKER_RANK1_BREADTH_SELECTOR_VERSION, + ) + incremental_scan_profile = ( + str(experiment['selector_version']) + == DOCKER_RANK1_BREADTH_SELECTOR_VERSION + ) + if self._docker_depth_incomplete_membership_count_locked( + experiment_id, activation=not incremental_scan_profile, + ): + self._hold_docker_depth_experiment_locked( + experiment, 'cohort_membership_incomplete', now, + ) + return self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment_id,), + ).fetchone() + experiment, selection_reason = ( + self._freeze_docker_depth_runtime_selection_locked( + experiment, authority={ + 'query_count': int(experiment['query_count']), + 'repositories_per_query': int(experiment['repositories_per_query']), + 'collection_generation': str(experiment['collection_generation']), + 'selector_version': str(experiment['selector_version']), + }, + now=now, + ) + ) + if selection_reason: + self._hold_docker_depth_experiment_locked( + experiment, selection_reason, now, + ) + return self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment_id,), + ).fetchone() + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'active', activated_at = COALESCE(activated_at, ?), + updated_at = ? + WHERE id = ? AND state = 'resolving' ''', + (now, now, experiment_id), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth activation lost its compare-and-swap fence' + ) + state = 'active' + if state == 'active': + nonterminal = int(self.conn.execute( + '''SELECT COUNT(*) AS count + FROM docker_depth_experiment_targets + WHERE experiment_id = ? AND state IN ('pending','reserved','scanning')''', + (experiment_id,), + ).fetchone()['count']) + if nonterminal == 0: + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'draining', draining_at = COALESCE(draining_at, ?), + updated_at = ? + WHERE id = ? AND state = 'active' ''', + (now, now, experiment_id), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth draining transition lost its compare-and-swap fence' + ) + state = 'draining' + if state == 'draining' and unresolved == 0: + if self._docker_depth_incomplete_membership_count_locked(experiment_id): + self._hold_docker_depth_experiment_locked( + experiment, 'cohort_membership_incomplete', now, + ) + return self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment_id,), + ).fetchone() + fences = self.conn.execute( + '''SELECT + (SELECT COUNT(*) FROM docker_depth_experiment_targets target + JOIN target_queue queue ON queue.id = target.target_queue_id + WHERE target.experiment_id = ? + AND (target.state NOT IN ('done','failed','skipped') + OR queue.status NOT IN ('done','failed') + OR queue.lease_owner IS NOT NULL + OR queue.lease_token IS NOT NULL + OR queue.claim_batch IS NOT NULL + OR queue.leased_at IS NOT NULL + OR queue.lease_expires_at IS NOT NULL + OR queue.current_result_reservation_id IS NOT NULL + OR queue.claim_event_id IS NOT NULL)) AS queue_fences, + (SELECT COUNT(*) FROM docker_depth_experiment_scan_bindings binding + JOIN docker_depth_experiment_targets target + ON target.id = binding.experiment_target_id + JOIN result_reservations reservation + ON reservation.id = binding.reservation_id + WHERE target.experiment_id = ? + AND (binding.state IN ('reserved','scanning') + OR reservation.state IN + ('scanning','ready','ingesting','db_committed'))) AS reservation_fences, + (SELECT COUNT(*) FROM docker_image_blob_coverage coverage + JOIN docker_depth_experiment_targets target + ON target.target_queue_id = coverage.queue_id + JOIN docker_content_blobs blob + ON blob.digest = coverage.blob_digest + AND blob.coverage_policy_sha256 = coverage.coverage_policy_sha256 + WHERE target.experiment_id = ? + AND blob.state IN ('leased','submitted')) AS blob_fences''', + (experiment_id, experiment_id, experiment_id), + ).fetchone() + if not any(int(fences[name]) for name in ( + 'queue_fences', 'reservation_fences', 'blob_fences', + )): + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'completed', completed_at = COALESCE(completed_at, ?), + updated_at = ? + WHERE id = ? AND state = 'draining' + AND fence_owner IS NULL AND fence_token IS NULL + AND fence_expires_at IS NULL''', + (now, now, experiment_id), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth completion lost its compare-and-swap fence' + ) + state = 'completed' + if state == str(experiment['state']): + return experiment + return self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment_id,), + ).fetchone() + + def _hold_disabled_docker_depth_experiment(self, authority): + key = str(authority.get('experiment_key') or '') if isinstance( + authority, dict + ) else '' + if not self.conn or not self.conn.is_postgres or not key: + return False + try: + experiment = self.conn.execute( + '''SELECT * FROM docker_depth_experiments + WHERE experiment_key = ? AND source = 'dockerhub' FOR UPDATE''', + (key,), + ).fetchone() + if not experiment or str(experiment['state']) not in ( + 'holding', 'resolving', 'active', 'draining', + ): + self.conn.rollback() + return False + self._hold_docker_depth_experiment_locked( + experiment, 'experiment_disabled', utc_now_iso(), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + return False + + def refresh_docker_depth_experiment_state(self, authority, *, final_cutover=False): + if not self.conn or not self.conn.is_postgres: + return {'status': 'unavailable', 'committed': False} + if isinstance(authority, dict) and authority.get('enabled') is False: + held = self._hold_disabled_docker_depth_experiment(authority) + return { + 'status': 'held' if held else 'disabled', + 'committed': held, + 'reason': 'experiment_disabled' if held else None, + } + if not isinstance(authority, dict) or authority.get('enabled') is not True: + return {'status': 'disabled', 'committed': False} + try: + normalized = self._normalize_docker_depth_resolver_authority(authority) + except (TypeError, ValueError): + return { + 'status': 'invalid', 'committed': False, + 'reason': 'authority_payload_invalid', + } + try: + now = utc_now_iso() + experiment, reason = self._locked_docker_depth_experiment_authority( + normalized, final_cutover=final_cutover, now=now, + ) + if not experiment: + self.conn.commit() + return { + 'status': 'held' if reason != 'experiment_authority_absent' else 'unavailable', + 'committed': reason != 'experiment_authority_absent', + 'reason': reason, + } + experiment = self._advance_docker_depth_experiment_state_locked( + experiment, now, + ) + self.conn.commit() + return { + 'status': str(experiment['state']), 'committed': True, + 'experiment_id': int(experiment['id']), + } + except Exception: + self.conn.rollback() + raise + + def _docker_depth_candidate_selection_drift_reason(self, candidate, authority): + from docker_depth_experiment import ( + canonical_docker_depth_selection_evidence_hash, + canonical_docker_descriptor_hash, + canonical_docker_layer_graph_hash, + ) + + layers = self.conn.execute( + '''SELECT position_from_base, position_from_top, layer_digest, + media_type, layer_size_bytes, descriptor_sha256 + FROM docker_manifest_layers WHERE manifest_id = ? + ORDER BY position_from_base LIMIT 10001''', + (candidate['manifest_id'],), + ).fetchall() + if ( + len(layers) != int(candidate['layer_count']) + or len(layers) > 10000 + ): + return 'selection_mutation' + normalized_layers = [] + graph = [] + for position, layer in enumerate(layers, 1): + digest = str(layer['layer_digest']) + media_type = str(layer['media_type']) + size_bytes = int(layer['layer_size_bytes']) + if ( + int(layer['position_from_base']) != position + or int(layer['position_from_top']) != len(layers) - position + 1 + or str(layer['descriptor_sha256']) + != canonical_docker_descriptor_hash(digest, media_type, size_bytes) + ): + return 'selection_mutation' + graph.append(digest) + normalized_layers.append({ + 'digest': digest, + 'media_type': media_type, + 'size_bytes': size_bytes, + 'descriptor_sha256': str(layer['descriptor_sha256']), + 'position_from_base': position, + 'position_from_top': len(layers) - position + 1, + }) + graph_sha256 = canonical_docker_layer_graph_hash(tuple(graph)) + if graph_sha256 != str(candidate['graph_sha256']): + return 'selection_mutation' + selections = self.conn.execute( + '''SELECT selection.image_rank, selection.selection_reason, + selection.selection_evidence_sha256, + selection.graph_sha256 AS selection_graph_sha256, + member.candidate_distinct_graph_count + FROM docker_depth_experiment_selections selection + JOIN docker_depth_experiment_repositories member + ON member.id = selection.experiment_repository_id + AND member.experiment_id = selection.experiment_id + WHERE selection.experiment_target_id = ? + ORDER BY selection.id LIMIT ?''', + (candidate['experiment_target_id'], authority['query_count'] + 1), + ).fetchall() + if not selections or len(selections) > authority['query_count']: + return 'selection_mutation' + for selection in selections: + evidence = { + 'schema': 1, + 'type': 'docker-depth-selection-evidence-v1', + 'selector_version': authority['selector_version'], + 'selector_sha256': authority['selector_sha256'], + 'candidate_distinct_graph_count': int( + selection['candidate_distinct_graph_count'] + ), + 'image_rank': int(selection['image_rank']), + 'selection_reason': str(selection['selection_reason']), + 'target': str(candidate['target']), + 'repository': str(candidate['repository']), + 'manifest_digest': str(candidate['manifest_digest']), + 'manifest_media_type': str(candidate['manifest_media_type']), + 'manifest_size_bytes': int(candidate['manifest_size_bytes']), + 'config_digest': str(candidate['config_digest']), + 'graph_sha256': graph_sha256, + 'layers': normalized_layers, + } + if ( + str(selection['selection_graph_sha256']) != graph_sha256 + or str(selection['selection_evidence_sha256']) + != canonical_docker_depth_selection_evidence_hash(evidence) + ): + return 'selection_mutation' + return None + + def _terminalize_exhausted_docker_depth_targets_locked( + self, experiment, authority, now, max_attempts, + ): + if max_attempts <= 0: + return 0 + rows = self.conn.execute( + '''SELECT target.id AS experiment_target_id, + target.target_queue_id, target.manifest_id, + target.reservation_count, + queue.attempts, queue.target_scan_id AS queue_target_scan_id, + binding.id AS binding_id, + binding.reservation_id AS binding_reservation_id, + binding.target_scan_id AS binding_target_scan_id, + binding.attempt AS binding_attempt, + binding.state AS binding_state, + reservation.id AS reservation_id, + reservation.queue_id AS reservation_queue_id, + reservation.state AS reservation_state, + scan.id AS scan_id, scan.queue_id AS scan_queue_id, + scan.result_reservation_id AS scan_reservation_id + FROM docker_depth_experiment_targets target + JOIN target_queue queue ON queue.id = target.target_queue_id + LEFT JOIN docker_depth_experiment_scan_bindings binding + ON binding.experiment_target_id = target.id + AND binding.attempt = target.reservation_count + LEFT JOIN result_reservations reservation + ON reservation.id = binding.reservation_id + LEFT JOIN target_scans scan ON scan.id = binding.target_scan_id + WHERE target.experiment_id = ? AND target.state = 'pending' + AND target.reservation_count > 0 + AND queue.source = ? AND queue.platform = 'docker' + AND queue.status = 'deferred' + AND COALESCE(queue.attempts, 0) >= ? + ORDER BY target.id + FOR UPDATE OF target, queue''', + (experiment['id'], authority['source'], max_attempts), + ).fetchall() + for row in rows: + if ( + row['binding_id'] is None + or int(row['binding_attempt'] or 0) + != int(row['reservation_count']) + or str(row['binding_state'] or '') != 'failed' + or row['binding_target_scan_id'] is None + or int(row['reservation_id'] or 0) + != int(row['binding_reservation_id'] or 0) + or int(row['reservation_queue_id'] or 0) + != int(row['target_queue_id']) + or str(row['reservation_state'] or '') + not in ('db_committed', 'acknowledged') + or int(row['scan_id'] or 0) + != int(row['binding_target_scan_id']) + or int(row['scan_queue_id'] or 0) + != int(row['target_queue_id']) + or int(row['scan_reservation_id'] or 0) + != int(row['reservation_id'] or 0) + or int(row['queue_target_scan_id'] or 0) + != int(row['scan_id'] or 0) + ): + raise ScanEventConflictError( + 'Docker depth exhausted retry recovery conflicts with scan history' + ) + queue_cursor = self.conn.execute( + '''UPDATE target_queue + SET status = 'failed', available_after = NULL, + completed_at = COALESCE(completed_at, ?), + last_error = COALESCE( + last_error, 'target retry attempts exhausted' + ), updated_at = ? + WHERE id = ? AND source = ? AND platform = 'docker' + AND status = 'deferred' AND COALESCE(attempts, 0) >= ? + AND target_scan_id = ? + AND current_result_reservation_id IS NULL + AND lease_owner IS NULL AND lease_token IS NULL + AND claim_batch IS NULL AND leased_at IS NULL + AND lease_expires_at IS NULL AND claim_event_id IS NULL + AND resolver_token IS NULL + AND (resolver_state IS NULL OR resolver_state = 'resolved')''', + ( + now, now, row['target_queue_id'], authority['source'], + max_attempts, row['scan_id'], + ), + ) + target_cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_targets + SET state = 'failed', terminal_at = ?, updated_at = ? + WHERE id = ? AND experiment_id = ? AND target_queue_id = ? + AND manifest_id = ? AND state = 'pending' + AND reservation_count = ? AND terminal_at IS NULL''', + ( + now, now, row['experiment_target_id'], experiment['id'], + row['target_queue_id'], row['manifest_id'], + row['reservation_count'], + ), + ) + if ( + int(queue_cursor.rowcount or 0) != 1 + or int(target_cursor.rowcount or 0) != 1 + ): + raise ScanEventConflictError( + 'Docker depth exhausted retry recovery lost its exact fence' + ) + return len(rows) + + def _claim_docker_depth_experiment_target_locked( + self, experiment, authority, now, max_attempts, + ): + self._terminalize_exhausted_docker_depth_targets_locked( + experiment, authority, now, max_attempts, + ) + row = self.conn.execute( + '''SELECT queue.id, queue.query, queue.target, + queue.normalized_target, queue.attempts, + target.id AS experiment_target_id, + target.experiment_id, target.manifest_id, + target.dispatch_wave, target.dispatch_order, + target.reservation_count, + manifest.repository, manifest.manifest_digest, + manifest.manifest_media_type, manifest.manifest_size_bytes, + manifest.config_digest, manifest.graph_sha256, + manifest.layer_count + FROM docker_depth_experiment_targets target + JOIN target_queue queue ON queue.id = target.target_queue_id + JOIN docker_image_manifests manifest + ON manifest.id = target.manifest_id + AND manifest.target_queue_id = queue.id + WHERE target.experiment_id = ? AND target.state = 'pending' + AND queue.source = ? AND queue.platform = 'docker' + AND queue.status IN ('pending','deferred') + AND (queue.status = 'pending' + OR (queue.available_after IS NOT NULL + AND queue.available_after <= ?)) + AND (? = 0 OR COALESCE(queue.attempts, 0) < ?) + AND queue.current_result_reservation_id IS NULL + AND queue.lease_owner IS NULL AND queue.lease_token IS NULL + AND queue.claim_batch IS NULL AND queue.leased_at IS NULL + AND queue.lease_expires_at IS NULL + AND queue.claim_event_id IS NULL AND queue.resolver_token IS NULL + AND (queue.resolver_state IS NULL OR queue.resolver_state = 'resolved') + AND EXISTS ( + SELECT 1 FROM docker_depth_experiment_selections selection + WHERE selection.experiment_id = target.experiment_id + AND selection.experiment_target_id = target.id + ) + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_targets prior + WHERE prior.experiment_id = target.experiment_id + AND prior.dispatch_wave < target.dispatch_wave + AND ( + prior.state NOT IN ('done','failed') + OR NOT EXISTS ( + SELECT 1 + FROM docker_depth_experiment_scan_bindings binding + WHERE binding.experiment_target_id = prior.id + AND binding.attempt = prior.reservation_count + AND ( + (prior.state = 'done' + AND binding.state = 'completed') + OR (prior.state = 'failed' + AND binding.state = 'failed') + ) + ) + ) + ) + ORDER BY target.dispatch_order, target.id + LIMIT 1 FOR UPDATE OF target, queue SKIP LOCKED''', + ( + experiment['id'], authority['source'], now, + max_attempts, max_attempts, + ), + ).fetchone() + if not row: + return None, None + reason = self._docker_depth_candidate_selection_drift_reason(row, authority) + if reason: + self._hold_docker_depth_experiment_locked(experiment, reason, now) + return None, reason + return row, None + + def _docker_depth_experiment_schema_installed(self): + cached = getattr( + self, '_docker_depth_experiment_schema_installed_cache', None, + ) + if cached is None: + cached = bool(self.conn.execute( + '''SELECT 1 AS present FROM runtime_schema_migrations + WHERE version = ?''', + (DOCKER_DEPTH_EXPERIMENT_MIGRATION,), + ).fetchone()) + self._docker_depth_experiment_schema_installed_cache = cached + return cached + + def _docker_depth_binding_for_reservation_locked(self, reservation): + if not self._docker_depth_experiment_schema_installed(): + return None + reservation_id = int(reservation['id']) + binding = self.conn.execute( + '''SELECT binding.*, target.experiment_id, + target.target_queue_id, target.manifest_id, + target.state AS target_state, + target.reservation_count, + queue.source AS bound_source, + queue.platform AS bound_platform, + queue.query AS bound_query, + queue.target AS bound_target, + queue.normalized_target AS bound_normalized_target, + manifest.source AS manifest_source, + manifest.manifest_digest, manifest.layer_count + FROM docker_depth_experiment_scan_bindings binding + JOIN docker_depth_experiment_targets target + ON target.id = binding.experiment_target_id + JOIN target_queue queue ON queue.id = target.target_queue_id + JOIN docker_image_manifests manifest + ON manifest.id = target.manifest_id + AND manifest.target_queue_id = target.target_queue_id + WHERE binding.reservation_id = ? + FOR UPDATE OF binding, target, queue, manifest''', + (reservation_id,), + ).fetchone() + if binding: + if ( + int(binding['target_queue_id']) != int(reservation['queue_id']) + or str(binding['bound_source']) != str(reservation['source']) + or str(binding['bound_platform']) != str(reservation['platform']) + or str(binding['bound_query'] or '') != str(reservation['query'] or '') + or str(binding['bound_target']) != str(reservation['target']) + or str(binding['bound_normalized_target']) + != str(reservation['normalized_target']) + or str(binding['bound_source']) != 'dockerhub' + or str(binding['bound_platform']) != 'docker' + or str(binding['manifest_source']) != str(binding['bound_source']) + or int(binding['attempt']) != int(binding['reservation_count']) + or str(binding['bound_at']) != str(reservation['created_at']) + or str(binding['created_at']) != str(reservation['created_at']) + ): + raise ScanEventConflictError( + 'Docker depth binding lost its exact reservation target identity' + ) + try: + bound_image = parse_docker_target(binding['bound_target'])['image'] + except (TypeError, ValueError): + raise ScanEventConflictError( + 'Docker depth binding has an invalid immutable target identity' + ) from None + if bound_image.rsplit('@', 1)[-1] != str(binding['manifest_digest']): + raise ScanEventConflictError( + 'Docker depth binding manifest identity conflicts with its target' + ) + return binding + experiment_target = self.conn.execute( + '''SELECT id FROM docker_depth_experiment_targets + WHERE target_queue_id = ? LIMIT 1 FOR UPDATE''', + (int(reservation['queue_id']),), + ).fetchone() + if experiment_target: + raise ScanEventConflictError( + 'Docker depth target reservation is missing its atomic binding' + ) + return None + + def _lock_docker_depth_experiment_for_reservation(self, reservation_id): + if not self._docker_depth_experiment_schema_installed(): + return None + identities = self.conn.execute( + '''SELECT DISTINCT identity.experiment_id + FROM ( + SELECT target.experiment_id + FROM docker_depth_experiment_scan_bindings binding + JOIN docker_depth_experiment_targets target + ON target.id = binding.experiment_target_id + WHERE binding.reservation_id = ? + UNION + SELECT target.experiment_id + FROM result_reservations reservation + JOIN docker_depth_experiment_targets target + ON target.target_queue_id = reservation.queue_id + WHERE reservation.id = ? + ) identity + ORDER BY identity.experiment_id LIMIT 2''', + (int(reservation_id), int(reservation_id)), + ).fetchall() + if not identities: + return None + if len(identities) != 1: + raise ScanEventConflictError( + 'Docker depth reservation resolves to conflicting experiment authority' + ) + experiment = self.conn.execute( + '''SELECT * FROM docker_depth_experiments + WHERE id = ? FOR UPDATE''', + (int(identities[0]['experiment_id']),), + ).fetchone() + if not experiment: + raise ScanEventConflictError( + 'Docker depth reservation lost its experiment authority' + ) + return experiment + + def _transition_docker_depth_binding_locked( + self, reservation, binding_state, target_state, now, *, target_scan_id=None, + ): + self._lock_docker_depth_experiment_for_reservation(reservation['id']) + binding = self._docker_depth_binding_for_reservation_locked(reservation) + if not binding: + return None + binding_state = str(binding_state) + target_state = str(target_state) + allowed = { + 'reserved': ('reserved', 'quarantined'), + 'scanning': ('reserved', 'scanning'), + 'completed': ('reserved', 'scanning', 'completed'), + 'failed': ('reserved', 'scanning', 'failed'), + 'quarantined': ('reserved', 'scanning', 'quarantined'), + 'released': ('reserved', 'released'), + } + allowed_targets = { + 'reserved': ('reserved', 'quarantined'), + 'scanning': ('reserved', 'scanning'), + 'done': ('reserved', 'scanning', 'done'), + 'failed': ('reserved', 'scanning', 'failed'), + 'quarantined': ('reserved', 'scanning', 'quarantined'), + 'pending': ('reserved', 'scanning', 'pending', 'quarantined'), + } + if str(binding['state']) not in allowed[binding_state]: + raise ScanEventConflictError( + 'Docker depth binding state conflicts with its reservation transition' + ) + if str(binding['target_state']) not in allowed_targets[target_state]: + raise ScanEventConflictError( + 'Docker depth target state conflicts with its reservation transition' + ) + if binding['target_scan_id'] is not None and int( + binding['target_scan_id'] + ) != int(target_scan_id or 0): + raise ScanEventConflictError( + 'Docker depth binding scan identity changed across replay' + ) + completed_at = now if binding_state in ( + 'completed', 'failed', 'quarantined', 'released', + ) else None + binding_cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_scan_bindings + SET state = ?, target_scan_id = COALESCE(target_scan_id, ?), + scan_bound_at = COALESCE(scan_bound_at, ?), + completed_at = CASE WHEN ? = 0 THEN NULL + ELSE COALESCE(completed_at, ?) END + WHERE id = ? AND experiment_target_id = ? + AND reservation_id = ? AND attempt = ? AND state = ?''', + ( + binding_state, target_scan_id, + now if target_scan_id is not None else None, + 1 if completed_at is not None else 0, completed_at, + binding['id'], binding['experiment_target_id'], + reservation['id'], binding['attempt'], binding['state'], + ), + ) + if int(binding_cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth binding transition lost its exact reservation identity' + ) + terminal_at = now if target_state in ( + 'done', 'failed', 'quarantined', 'held', 'skipped', + ) else None + cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_targets + SET state = ?, terminal_at = ?, updated_at = ? + WHERE id = ? AND experiment_id = ? AND target_queue_id = ? + AND manifest_id = ? AND reservation_count = ? AND state = ?''', + ( + target_state, terminal_at, now, binding['experiment_target_id'], + binding['experiment_id'], reservation['queue_id'], + binding['manifest_id'], binding['reservation_count'], + binding['target_state'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth target state lost its reservation binding identity' + ) + return { + 'binding_id': int(binding['id']), + 'experiment_id': int(binding['experiment_id']), + 'experiment_target_id': int(binding['experiment_target_id']), + 'manifest_id': int(binding['manifest_id']), + 'attempt': int(binding['attempt']), + } + + @staticmethod + def _normalize_docker_depth_resolver_authority(authority): + from docker_depth_experiment import ( + DOCKER_DEPTH_COLLECTION_GENERATION, + DOCKER_DEPTH_QUERY_COUNT, + canonical_ordered_query_hash, + canonical_selector_hash, + reviewed_docker_experiment_profile, + ) + + if not isinstance(authority, dict) or authority.get('enabled') is not True: + raise ValueError('Docker depth resolver authority is disabled or invalid') + queries = authority.get('queries') + if isinstance(queries, (str, bytes)): + raise ValueError('Docker depth resolver query authority is invalid') + try: + queries = tuple(queries) + except TypeError: + raise ValueError('Docker depth resolver query authority is invalid') from None + if ( + len(queries) != DOCKER_DEPTH_QUERY_COUNT + or len(set(queries)) != len(queries) + or any(not isinstance(query, str) or not query or query != query.strip() + for query in queries) + ): + raise ValueError('Docker depth resolver query authority is invalid') + selector_version = str(authority.get('selector_version') or '') + profile = reviewed_docker_experiment_profile(selector_version) + integer_values = { + 'query_count': profile['query_count'], + 'repositories_per_query': profile['repositories_per_query'], + 'images_per_repository': profile['images_per_repository'], + 'target_limit': profile['target_limit'], + 'theoretical_max_targets': profile['theoretical_max_targets'], + } + for key, expected in integer_values.items(): + value = authority.get(key) + if isinstance(value, bool) or value != expected: + raise ValueError(f'Docker depth resolver {key} authority conflicts') + hashes = ( + 'config_sha256', 'ordered_queries_sha256', 'selector_sha256', + 'provenance_policy_sha256', + ) + if any(not re.fullmatch(r'[a-f0-9]{64}', str(authority.get(key) or '')) for key in hashes): + raise ValueError('Docker depth resolver hash authority is invalid') + if authority['ordered_queries_sha256'] != canonical_ordered_query_hash(queries): + raise ValueError('Docker depth resolver ordered-query authority conflicts') + if ( + authority.get('selector_version') != selector_version + or authority['selector_sha256'] != canonical_selector_hash(selector_version) + ): + raise ValueError('Docker depth resolver selector authority conflicts') + experiment_key = str(authority.get('experiment_key') or '') + if not re.fullmatch(r'[a-z0-9](?:[a-z0-9._-]{0,126}[a-z0-9])?', experiment_key): + raise ValueError('Docker depth resolver experiment key is invalid') + if authority.get('source') != 'dockerhub': + raise ValueError('Docker depth resolver source authority conflicts') + if authority.get('collection_generation') != DOCKER_DEPTH_COLLECTION_GENERATION: + raise ValueError('Docker depth resolver collection generation conflicts') + return { + **integer_values, + 'enabled': True, + 'experiment_key': experiment_key, + 'source': 'dockerhub', + 'collection_generation': DOCKER_DEPTH_COLLECTION_GENERATION, + 'queries': queries, + 'config_sha256': str(authority['config_sha256']), + 'ordered_queries_sha256': str(authority['ordered_queries_sha256']), + 'selector_version': selector_version, + 'selector_sha256': str(authority['selector_sha256']), + 'provenance_policy_sha256': str(authority['provenance_policy_sha256']), + } + + @staticmethod + def _docker_depth_experiment_matches_authority(row, authority): + expected = { + 'experiment_key': authority['experiment_key'], + 'source': authority['source'], + 'collection_generation': authority['collection_generation'], + 'config_sha256': authority['config_sha256'], + 'ordered_queries_sha256': authority['ordered_queries_sha256'], + 'selector_version': authority['selector_version'], + 'selector_sha256': authority['selector_sha256'], + 'provenance_policy_sha256': authority['provenance_policy_sha256'], + 'query_count': authority['query_count'], + 'repositories_per_query': authority['repositories_per_query'], + 'images_per_repository': authority['images_per_repository'], + 'target_limit': authority['target_limit'], + } + return bool(row) and all( + (int(row[key]) if isinstance(value, int) else str(row[key])) == value + for key, value in expected.items() + ) + + def _terminalize_docker_depth_attempt_limit_locked( + self, experiment, member, authority, now, *, held_recovery=False, + ): + from docker_depth_experiment import ( + DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON, + DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS, + ) + + attempts = int(member['resolver_attempts']) + selected_count = int(member['selected_image_count']) + if attempts < DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS: + raise ScanEventConflictError( + 'Docker depth attempt-limit terminalization is premature' + ) + stored_selection_count = int(self.conn.execute( + '''SELECT COUNT(*) AS count + FROM docker_depth_experiment_selections + WHERE experiment_repository_id = ?''', + (member['id'],), + ).fetchone()['count']) + if stored_selection_count != selected_count: + raise ScanEventConflictError( + 'Docker depth attempt-limit selection evidence drifted' + ) + current_queue_id = int( + member['replacement_repository_queue_id'] + or member['repository_queue_id'] + ) + self._record_docker_depth_candidate_skip_locked( + experiment, member, current_queue_id, 'repository', + int(member['replacement_count']) + 1, + {'repository_queue_id': current_queue_id}, + DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON, now, + ) + next_state = 'skipped' if selected_count == 0 else 'resolved' + next_error = ( + DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON + if next_state == 'skipped' else None + ) + if held_recovery: + if ( + str(experiment['state']) != 'held' + or str(experiment['hold_reason_code'] or '') + != 'resolver_attempt_limit' + or str(member['work_state']) != 'held' + or str(member['last_error_code'] or '') + != 'resolver_attempt_limit' + or any(member[name] is not None for name in ( + 'resolver_owner', 'resolver_token', 'resolver_expires_at', + 'resolver_due_at', + )) + ): + raise ScanEventConflictError( + 'Docker depth held attempt-limit evidence drifted' + ) + held_count = int(self.conn.execute( + '''SELECT COUNT(*) AS count + FROM docker_depth_experiment_repositories + WHERE experiment_id = ? AND work_state = 'held' ''', + (experiment['id'],), + ).fetchone()['count']) + if held_count != 1: + raise ScanEventConflictError( + 'Docker depth attempt-limit hold is ambiguous' + ) + cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = ?, resolver_owner = NULL, + resolver_token = NULL, resolver_expires_at = NULL, + resolver_due_at = NULL, last_error_code = ?, + resolved_at = ?, updated_at = ? + WHERE id = ? AND experiment_id = ? AND work_state = 'held' + AND last_error_code = 'resolver_attempt_limit' + AND resolver_attempts >= ? + AND resolver_owner IS NULL AND resolver_token IS NULL + AND resolver_expires_at IS NULL AND resolver_due_at IS NULL''', + ( + next_state, next_error, now, now, member['id'], + experiment['id'], DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth attempt-limit recovery lost its member fence' + ) + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'resolving', hold_reason_code = NULL, + held_at = NULL, updated_at = ? + WHERE id = ? AND state = 'held' + AND hold_reason_code = 'resolver_attempt_limit' + AND fence_owner IS NULL AND fence_token IS NULL + AND fence_expires_at IS NULL''', + (now, experiment['id']), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth attempt-limit recovery lost its experiment fence' + ) + else: + if ( + str(experiment['state']) != 'resolving' + or str(member['work_state']) != 'resolving' + or not member['resolver_owner'] + or not member['resolver_token'] + or str(member['resolver_expires_at'] or '') <= now + ): + raise ScanEventConflictError( + 'Docker depth attempt-limit lease evidence drifted' + ) + cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = ?, resolver_owner = NULL, + resolver_token = NULL, resolver_expires_at = NULL, + resolver_due_at = NULL, last_error_code = ?, + resolved_at = ?, updated_at = ? + WHERE id = ? AND experiment_id = ? AND work_state = 'resolving' + AND resolver_generation = ? AND resolver_owner = ? + AND resolver_token = ? AND resolver_expires_at > ? + AND resolver_attempts >= ?''', + ( + next_state, next_error, now, now, member['id'], + experiment['id'], member['resolver_generation'], + member['resolver_owner'], member['resolver_token'], now, + DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth attempt-limit terminalization lost its lease fence' + ) + + if selected_count == 0: + self._finalize_docker_depth_query_breadth_locked( + experiment, member, now, + ) + experiment = self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment['id'],), + ).fetchone() + experiment, selection_reason = ( + self._freeze_docker_depth_runtime_selection_locked( + experiment, authority, now, + ) + ) + if selection_reason: + result = self._hold_docker_depth_experiment_locked( + experiment, selection_reason, now, + ) + result['experiment'] = self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment['id'],), + ).fetchone() + return result + drift_reason = self._docker_depth_persisted_drift_reason_locked( + experiment, authority, now, + ) + if drift_reason: + result = self._hold_docker_depth_experiment_locked( + experiment, drift_reason, now, + ) + result['experiment'] = self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment['id'],), + ).fetchone() + return result + experiment = self._advance_docker_depth_experiment_state_locked( + experiment, now, + ) + return { + 'status': 'skipped' if next_state == 'skipped' else 'resolved', + 'committed': True, + 'reason': DOCKER_DEPTH_REMOTE_UNAVAILABLE_SKIP_REASON, + 'experiment_state': str(experiment['state']), + 'experiment': experiment, + } + + def claim_docker_depth_experiment_resolutions( + self, source, limit, lease_owner, lease_seconds=300, *, authority=None, + final_cutover=False, + ): + if ( + not self.conn + or not self.conn.is_postgres + or source != 'dockerhub' + or not lease_owner + ): + return [] + if isinstance(authority, dict) and authority.get('enabled') is False: + self._hold_disabled_docker_depth_experiment(authority) + return [] + if not isinstance(authority, dict) or authority.get('enabled') is not True: + return [] + try: + normalized = self._normalize_docker_depth_resolver_authority(authority) + except (TypeError, ValueError): + return [] + from docker_depth_experiment import ( + DOCKER_RANK1_BREADTH_SELECTOR_VERSION, + _require_released_policy_history, + ) + if normalized['selector_version'] == DOCKER_RANK1_BREADTH_SELECTOR_VERSION: + original_queue_policy_clause = "queue.status IN ('pending','deferred')" + else: + original_queue_policy_clause = '''queue.status IN ('pending','deferred') + AND NOT EXISTS ( + SELECT 1 FROM target_queue_policy_events event + WHERE event.queue_id = queue.id + )''' + + def op(): + state = self._locked_runtime_control_state(shared=True) + if state['effective_discovery_paused']: + self.conn.commit() + return [] + now = utc_now_iso() + lease_until = datetime.fromtimestamp( + time.time() + max(60, min(3600, int(lease_seconds or 300))), + timezone.utc, + ).isoformat(timespec='seconds') + experiment, _reason = self._locked_docker_depth_experiment_authority( + normalized, final_cutover=final_cutover, now=now, + ) + if not experiment: + self.conn.commit() + return [] + if ( + str(experiment['state']) == 'held' + and str(experiment['hold_reason_code'] or '') + == 'resolver_attempt_limit' + ): + held_members = self.conn.execute( + '''SELECT * FROM docker_depth_experiment_repositories + WHERE experiment_id = ? AND work_state = 'held' + ORDER BY id LIMIT 2 FOR UPDATE''', + (experiment['id'],), + ).fetchall() + if len(held_members) != 1: + self.conn.commit() + return [] + terminal = self._terminalize_docker_depth_attempt_limit_locked( + experiment, held_members[0], normalized, now, + held_recovery=True, + ) + experiment = terminal['experiment'] + if str(experiment['state']) != 'resolving': + self.conn.commit() + return [] + if ( + str(experiment['state']) == 'held' + and str(experiment['hold_reason_code'] or '') + == 'stale_resolver_fence' + ): + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'resolving', hold_reason_code = NULL, + held_at = NULL, updated_at = ? + WHERE id = ? AND state = 'held' + AND hold_reason_code = 'stale_resolver_fence' + AND fence_owner IS NULL AND fence_token IS NULL + AND fence_expires_at IS NULL + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_repositories member + WHERE member.experiment_id = ? + AND (member.work_state = 'resolving' + OR member.resolver_owner IS NOT NULL + OR member.resolver_token IS NOT NULL + OR member.resolver_expires_at IS NOT NULL) + )''', + (now, experiment['id'], experiment['id']), + ) + if int(cursor.rowcount or 0) != 1: + self.conn.commit() + return [] + experiment = self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment['id'],), + ).fetchone() + if str(experiment['state']) == 'holding': + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET state = 'resolving', updated_at = ? + WHERE id = ? AND state = 'holding' + AND plan_sha256 = ? AND hold_manifest_sha256 = ?''', + ( + now, experiment['id'], experiment['plan_sha256'], + experiment['hold_manifest_sha256'], + ), + ) + if int(cursor.rowcount or 0) != 1: + self._hold_docker_depth_experiment_locked( + experiment, 'activation_fence_conflict', now, + ) + self.conn.commit() + return [] + experiment = self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment['id'],), + ).fetchone() + experiment = self._advance_docker_depth_experiment_state_locked( + experiment, now, + ) + if str(experiment['state']) != 'resolving': + self.conn.commit() + return [] + + output = [] + maximum = max(1, min(100, int(limit or 1))) + if normalized['selector_version'] == DOCKER_RANK1_BREADTH_SELECTOR_VERSION: + maximum = 1 + for _ in range(maximum): + row = self.conn.execute( + f'''SELECT member.*, queue.id AS resolution_repository_queue_id, + queue.target + FROM docker_depth_experiment_repositories member + JOIN target_queue queue ON queue.id = COALESCE( + member.replacement_repository_queue_id, + member.repository_queue_id + ) + WHERE member.experiment_id = ? + AND member.selected_image_count = 0 + AND ( + (member.work_state = 'pending' + AND (member.resolver_due_at IS NULL OR member.resolver_due_at <= ?)) + OR (member.work_state = 'resolving' + AND member.resolver_expires_at <= ?) + ) + AND queue.source = ? AND queue.platform = 'docker' + AND ( + ({original_queue_policy_clause}) + OR (member.replacement_repository_queue_id IS NOT NULL + AND queue.status = 'cold' + AND (SELECT COUNT(*) FROM target_queue_policy_events event + WHERE event.queue_id = queue.id) = 1 + AND EXISTS ( + SELECT 1 FROM target_queue_policy_events event + WHERE event.queue_id = queue.id + AND event.action = 'cold' + AND event.experiment_id = member.experiment_id + AND event.reason_code IN ( + 'docker_depth_experiment_hold', + 'docker_depth_experiment_dynamic_hold' + ) + AND event.config_sha256 = ? + AND event.policy_sha256 = ? + AND event.manifest_sha256 = ? + ))) + AND queue.target_scan_id IS NULL + AND queue.target NOT LIKE '%@%' AND queue.normalized_target NOT LIKE '%@%' + AND queue.lease_owner IS NULL AND queue.lease_token IS NULL + AND queue.claim_batch IS NULL AND queue.current_result_reservation_id IS NULL + AND queue.claim_event_id IS NULL AND queue.resolver_token IS NULL + AND NOT EXISTS ( + SELECT 1 FROM target_scans scan + WHERE scan.queue_id = queue.id + ) + AND NOT EXISTS ( + SELECT 1 FROM result_reservations reservation + WHERE reservation.queue_id = queue.id + ) + ORDER BY member.repository_rank, member.query_ordinal, member.id + LIMIT 1 FOR UPDATE OF member SKIP LOCKED''', + ( + experiment['id'], now, now, source, + experiment['config_sha256'], + experiment['provenance_policy_sha256'], + experiment['hold_manifest_sha256'], + ), + ).fetchone() + stage = 'breadth' + if not row: + unfinished_breadth = self.conn.execute( + '''SELECT COUNT(*) AS count + FROM docker_depth_experiment_repositories + WHERE experiment_id = ? AND selected_image_count = 0 + AND work_state IN ('pending','resolving')''', + (experiment['id'],), + ).fetchone()['count'] + if int(unfinished_breadth): + break + experiment, selection_reason = ( + self._freeze_docker_depth_runtime_selection_locked( + experiment, normalized, now, + ) + ) + if selection_reason: + self._hold_docker_depth_experiment_locked( + experiment, selection_reason, now, + ) + break + row = self.conn.execute( + '''SELECT member.*, queue.id AS resolution_repository_queue_id, + queue.target + FROM docker_depth_experiment_repositories member + JOIN target_queue queue ON queue.id = COALESCE( + member.replacement_repository_queue_id, + member.repository_queue_id + ) + WHERE member.experiment_id = ? AND member.is_deep_probe = 1 + AND member.selected_image_count >= 1 + AND member.selected_image_count < member.candidate_distinct_graph_count + AND member.selected_image_count < ? + AND ( + (member.work_state = 'pending' + AND (member.resolver_due_at IS NULL OR member.resolver_due_at <= ?)) + OR (member.work_state = 'resolving' + AND member.resolver_expires_at <= ?) + ) + AND queue.source = ? AND queue.platform = 'docker' + AND ( + (queue.status IN ('pending','deferred') AND NOT EXISTS ( + SELECT 1 FROM target_queue_policy_events event + WHERE event.queue_id = queue.id + )) + OR (member.replacement_repository_queue_id IS NOT NULL + AND queue.status = 'cold' + AND (SELECT COUNT(*) FROM target_queue_policy_events event + WHERE event.queue_id = queue.id) = 1 + AND EXISTS ( + SELECT 1 FROM target_queue_policy_events event + WHERE event.queue_id = queue.id + AND event.action = 'cold' + AND event.experiment_id = member.experiment_id + AND event.reason_code IN ( + 'docker_depth_experiment_hold', + 'docker_depth_experiment_dynamic_hold' + ) + AND event.config_sha256 = ? + AND event.policy_sha256 = ? + AND event.manifest_sha256 = ? + ))) + AND queue.target_scan_id IS NULL + AND queue.lease_owner IS NULL AND queue.lease_token IS NULL + AND queue.claim_batch IS NULL + AND queue.current_result_reservation_id IS NULL + AND queue.claim_event_id IS NULL AND queue.resolver_token IS NULL + AND NOT EXISTS ( + SELECT 1 FROM target_scans scan + WHERE scan.queue_id = queue.id + ) + AND NOT EXISTS ( + SELECT 1 FROM result_reservations reservation + WHERE reservation.queue_id = queue.id + ) + ORDER BY member.query_ordinal, member.repository_rank, member.id + LIMIT 1 FOR UPDATE OF member SKIP LOCKED''', + ( + experiment['id'], normalized['images_per_repository'], + now, now, source, experiment['config_sha256'], + experiment['provenance_policy_sha256'], + experiment['hold_manifest_sha256'], + ), + ).fetchone() + stage = 'deep' + if not row: + break + if ( + normalized['selector_version'] + == DOCKER_RANK1_BREADTH_SELECTOR_VERSION + and row['replacement_repository_queue_id'] is None + ): + try: + _require_released_policy_history( + self.conn, [int(row['resolution_repository_queue_id'])], + ) + except RuntimeError: + self._hold_docker_depth_experiment_locked( + experiment, 'target_history_drift', now, + ) + self.conn.commit() + return [] + token = secrets.token_urlsafe(32) + cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = 'resolving', + resolver_generation = resolver_generation + 1, + resolver_owner = ?, resolver_token = ?, resolver_expires_at = ?, + resolver_due_at = NULL, resolver_attempts = resolver_attempts + 1, + updated_at = ? + WHERE id = ?''', + (lease_owner, token, lease_until, now, row['id']), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('Docker depth resolver claim lost its row fence') + output.append({ + 'id': int(row['id']), + 'experiment_id': int(experiment['id']), + 'repository_queue_id': int(row['resolution_repository_queue_id']), + 'query_ordinal': int(row['query_ordinal']), + 'query': str(row['query']), + 'repository_rank': int(row['repository_rank']), + 'target': str(row['target']), + 'stage': stage, + 'selection_limit': ( + 1 if stage == 'breadth' + else normalized['images_per_repository'] + ), + 'resolver_owner': str(lease_owner), + 'resolver_token': token, + 'resolver_generation': int(row['resolver_generation']) + 1, + 'resolver_attempts': int(row['resolver_attempts']) + 1, + 'resolver_expires_at': lease_until, + 'config_sha256': normalized['config_sha256'], + 'ordered_queries_sha256': normalized['ordered_queries_sha256'], + 'selector_sha256': normalized['selector_sha256'], + }) + self.conn.commit() + return output + + return self._safe('claim_docker_depth_experiment_resolutions', op, []) + + def renew_docker_depth_experiment_resolution( + self, source, repository_id, resolver_generation, resolver_token, *, + resolver_owner, lease_seconds=300, authority=None, final_cutover=False, + ): + if ( + not self.conn + or not self.conn.is_postgres + or source != 'dockerhub' + or not resolver_owner + or not resolver_token + ): + return {'status': 'unavailable', 'renewed': False} + try: + normalized = self._normalize_docker_depth_resolver_authority(authority) + repository_id = int(repository_id) + resolver_generation = int(resolver_generation) + lease_seconds = int(lease_seconds) + if ( + repository_id < 1 + or resolver_generation < 1 + or not 60 <= lease_seconds <= 3600 + ): + raise ValueError('Docker depth resolver renewal identity is invalid') + except (TypeError, ValueError, OverflowError): + return {'status': 'invalid', 'renewed': False} + now_dt = datetime.now(timezone.utc) + now = now_dt.isoformat(timespec='seconds') + lease_until = (now_dt + timedelta(seconds=lease_seconds)).isoformat( + timespec='seconds', + ) + try: + experiment = self.conn.execute( + '''SELECT * FROM docker_depth_experiments + WHERE experiment_key = ? AND source = ? FOR UPDATE''', + (normalized['experiment_key'], source), + ).fetchone() + member = self.conn.execute( + '''SELECT * FROM docker_depth_experiment_repositories + WHERE id = ? AND experiment_id = ? FOR UPDATE''', + (repository_id, experiment['id'] if experiment else 0), + ).fetchone() + if not ( + experiment + and str(experiment['state']) == 'resolving' + and member + and str(member['work_state']) == 'resolving' + and int(member['resolver_generation']) == resolver_generation + and str(member['resolver_owner']) == str(resolver_owner) + and str(member['resolver_token']) == str(resolver_token) + and str(member['resolver_expires_at'] or '') > now + ): + self.conn.rollback() + return {'status': 'stale', 'renewed': False} + checked, reason = self._locked_docker_depth_experiment_authority( + normalized, final_cutover=final_cutover, now=now, + ) + if not checked: + self.conn.commit() + return { + 'status': 'held', 'renewed': False, 'reason': reason, + } + cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET resolver_expires_at = ?, updated_at = ? + WHERE id = ? AND experiment_id = ? + AND work_state = 'resolving' + AND resolver_generation = ? AND resolver_owner = ? + AND resolver_token = ? AND resolver_expires_at > ?''', + ( + lease_until, now, repository_id, experiment['id'], + resolver_generation, resolver_owner, resolver_token, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + self.conn.rollback() + return {'status': 'stale', 'renewed': False} + self.conn.commit() + return { + 'status': 'renewed', 'renewed': True, + 'resolver_expires_at': lease_until, + } + except Exception: + self.conn.rollback() + raise + + def reserve_docker_depth_experiment_target_capacity_locked( + self, experiment_id, target_queue_id, manifest_id, + dispatch_wave, dispatch_order, now=None, + ): + """Reserve one deduplicated target slot inside the caller's transaction.""" + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('Docker depth target capacity requires PostgreSQL') + values = ( + experiment_id, target_queue_id, manifest_id, dispatch_wave, dispatch_order, + ) + if any(isinstance(value, bool) for value in values): + raise ValueError('Docker depth target capacity identity is invalid') + try: + experiment_id, target_queue_id, manifest_id, dispatch_wave, dispatch_order = ( + int(value) for value in values + ) + except (TypeError, ValueError, OverflowError): + raise ValueError('Docker depth target capacity identity is invalid') from None + if ( + min(experiment_id, target_queue_id, manifest_id, dispatch_order) < 1 + or dispatch_wave not in (1, 2, 3) + ): + raise ValueError('Docker depth target capacity identity is invalid') + experiment = self.conn.execute( + '''SELECT id, state, target_count, target_limit + FROM docker_depth_experiments WHERE id = ? FOR UPDATE''', + (experiment_id,), + ).fetchone() + if not experiment or experiment['state'] not in ('resolving', 'active'): + raise ScanEventConflictError('Docker depth target capacity has no active authority') + actual_count = self.conn.execute( + '''SELECT COUNT(*) AS count FROM docker_depth_experiment_targets + WHERE experiment_id = ?''', + (experiment_id,), + ).fetchone()['count'] + target_count = int(experiment['target_count']) + if int(actual_count) != target_count: + raise ScanEventConflictError('Docker depth target capacity counter drifted') + existing = self.conn.execute( + '''SELECT * FROM docker_depth_experiment_targets + WHERE experiment_id = ? AND target_queue_id = ? FOR UPDATE''', + (experiment_id, target_queue_id), + ).fetchone() + if existing: + if int(existing['manifest_id']) != manifest_id: + raise ScanEventConflictError('Docker depth target manifest identity conflicts') + if (dispatch_wave, dispatch_order) < ( + int(existing['dispatch_wave']), int(existing['dispatch_order']), + ): + existing = self.conn.execute( + '''UPDATE docker_depth_experiment_targets + SET dispatch_wave = ?, dispatch_order = ?, updated_at = ? + WHERE id = ? RETURNING *''', + (dispatch_wave, dispatch_order, str(now or utc_now_iso()), existing['id']), + ).fetchone() + return {'target': dict(existing), 'inserted': False} + if target_count >= int(experiment['target_limit']): + raise ScanEventConflictError('Docker depth unique target capacity is exhausted') + now = str(now or utc_now_iso()) + target = self.conn.execute( + '''INSERT INTO docker_depth_experiment_targets( + experiment_id, target_queue_id, manifest_id, counter_ordinal, + state, dispatch_wave, dispatch_order, created_at, updated_at + ) VALUES (?, ?, ?, ?, 'pending', ?, ?, ?, ?) RETURNING *''', + ( + experiment_id, target_queue_id, manifest_id, target_count + 1, + dispatch_wave, dispatch_order, now, now, + ), + ).fetchone() + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET target_count = target_count + 1, updated_at = ? + WHERE id = ? AND target_count = ? AND target_count < target_limit''', + (now, experiment_id, target_count), + ) + if not target or int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('Docker depth target capacity lost its serialized fence') + return {'target': dict(target), 'inserted': True} + + @staticmethod + def _normalize_docker_depth_resolution_outcome(outcome, repository, authority, stage): + from docker_depth_experiment import ( + canonical_docker_depth_selection_evidence_hash, + canonical_docker_descriptor_hash, + canonical_docker_layer_graph_hash, + ) + + def value(name, default=None): + if isinstance(outcome, dict): + return outcome.get(name, default) + return getattr(outcome, name, default) + + if ( + not outcome + or value('selector_version') != authority['selector_version'] + or value('selector_hash') != authority['selector_sha256'] + or value('fresh_graph_evidence') is not True + or value('cache_bypassed') is not True + ): + raise ScanEventConflictError('Docker depth resolver outcome lacks fresh selector evidence') + candidate_count = value('candidate_distinct_graph_count') + if ( + isinstance(candidate_count, bool) + or not isinstance(candidate_count, int) + or not 0 <= candidate_count <= 100 + ): + raise ScanEventConflictError('Docker depth candidate graph count is invalid') + selected_records = value('selection_records', ()) + if isinstance(selected_records, (str, bytes)): + raise ScanEventConflictError('Docker depth selection evidence is invalid') + try: + selected_records = list(selected_records) + except TypeError: + raise ScanEventConflictError('Docker depth selection evidence is invalid') from None + candidate_records = value('candidate_records', selected_records) + if not candidate_records: + candidate_records = selected_records + if isinstance(candidate_records, (str, bytes)): + raise ScanEventConflictError('Docker depth candidate evidence is invalid') + try: + candidate_records = list(candidate_records) + except TypeError: + raise ScanEventConflictError('Docker depth candidate evidence is invalid') from None + selected_maximum = 1 if stage == 'breadth' else authority['images_per_repository'] + if ( + len(selected_records) > selected_maximum + or len(candidate_records) > 100 + or candidate_count < len(candidate_records) + or candidate_records[:len(selected_records)] != selected_records + ): + raise ScanEventConflictError('Docker depth selection count conflicts with graph evidence') + tags = value('tags', ()) + if list(tags or ()) != [ + record.get('target') for record in selected_records + if isinstance(record, dict) + ]: + raise ScanEventConflictError('Docker depth selected targets conflict with selection evidence') + + normalized_records = [] + seen_targets = set() + seen_graphs = set() + for expected_rank, raw in enumerate(candidate_records, 1): + if not isinstance(raw, dict): + raise ScanEventConflictError('Docker depth selection record is invalid') + rank = raw.get('image_rank', raw.get('rank')) + reason = str(raw.get('selection_reason', raw.get('reason')) or '') + if ( + isinstance(rank, bool) + or rank != expected_rank + or not reason + or len(reason) > 128 + ): + raise ScanEventConflictError('Docker depth selection rank or reason is invalid') + target = str(raw.get('target') or '').strip() + try: + parsed = parse_docker_target(target) + except (TypeError, ValueError) as exc: + raise ScanEventConflictError('Docker depth immutable target is invalid') from exc + image = str(parsed['image']).lower() + if '@' not in image or parsed['target'] != target: + raise ScanEventConflictError('Docker depth immutable target is not canonical') + target_repository, target_digest = image.rsplit('@', 1) + manifest_digest = str(raw.get('manifest_digest') or '').lower() + if ( + target_repository != repository + or str(raw.get('repository') or '').lower() != repository + or manifest_digest != target_digest + or not re.fullmatch(r'sha256:[a-f0-9]{64}', manifest_digest) + ): + raise ScanEventConflictError('Docker depth repository or manifest identity conflicts') + manifest_media_type = str(raw.get('manifest_media_type') or '') + config_digest = str(raw.get('config_digest') or '').lower() + manifest_size_bytes = raw.get('manifest_size_bytes') + if ( + not manifest_media_type + or manifest_media_type != manifest_media_type.strip().lower() + or len(manifest_media_type) > 256 + or not re.fullmatch(r'sha256:[a-f0-9]{64}', config_digest) + or isinstance(manifest_size_bytes, bool) + or not isinstance(manifest_size_bytes, int) + or not 0 <= manifest_size_bytes <= 128 * 1024 * 1024 + ): + raise ScanEventConflictError('Docker depth manifest metadata is invalid') + layer_metadata = raw.get('layer_metadata') + if isinstance(layer_metadata, (str, bytes)): + raise ScanEventConflictError('Docker depth layer descriptors are invalid') + try: + layer_metadata = list(layer_metadata) + except TypeError: + raise ScanEventConflictError('Docker depth layer descriptors are invalid') from None + if not layer_metadata or len(layer_metadata) > 10000: + raise ScanEventConflictError('Docker depth layer descriptor count is invalid') + layers = [] + layer_count = len(layer_metadata) + for position, descriptor in enumerate(layer_metadata, 1): + if not isinstance(descriptor, dict): + raise ScanEventConflictError('Docker depth layer descriptor is invalid') + digest = str(descriptor.get('digest') or '').lower() + media_type = str(descriptor.get('media_type') or '') + size_bytes = descriptor.get('size_bytes') + descriptor_sha256 = str(descriptor.get('descriptor_sha256') or '') + if ( + not re.fullmatch(r'sha256:[a-f0-9]{64}', digest) + or not media_type + or media_type != media_type.strip().lower() + or len(media_type) > 256 + or isinstance(size_bytes, bool) + or not isinstance(size_bytes, int) + or not 0 <= size_bytes <= 1024 * 1024 * 1024 * 1024 + or descriptor.get('position_from_base') != position + or descriptor.get('position_from_top') != layer_count - position + 1 + or descriptor_sha256 + != canonical_docker_descriptor_hash(digest, media_type, size_bytes) + ): + raise ScanEventConflictError('Docker depth layer descriptor evidence conflicts') + layers.append({ + 'digest': digest, + 'media_type': media_type, + 'size_bytes': size_bytes, + 'descriptor_sha256': descriptor_sha256, + 'position_from_base': position, + 'position_from_top': layer_count - position + 1, + }) + graph = tuple(layer['digest'] for layer in layers) + raw_graph = tuple(raw.get('graph', raw.get('layers', ())) or ()) + graph_sha256 = canonical_docker_layer_graph_hash(graph) + if ( + raw_graph != graph + or raw.get('layer_count') != layer_count + or raw.get('graph_sha256', raw.get('graph_hash')) != graph_sha256 + ): + raise ScanEventConflictError('Docker depth ordered layer graph evidence conflicts') + normalized_target = normalize_target(target, 'docker') + if normalized_target != image or normalized_target in seen_targets or graph in seen_graphs: + raise ScanEventConflictError('Docker depth duplicate target or graph consumed a rank') + seen_targets.add(normalized_target) + seen_graphs.add(graph) + evidence = { + 'schema': 1, + 'type': 'docker-depth-selection-evidence-v1', + 'selector_version': authority['selector_version'], + 'selector_sha256': authority['selector_sha256'], + 'candidate_distinct_graph_count': candidate_count, + 'image_rank': rank, + 'selection_reason': reason, + 'target': target, + 'repository': repository, + 'manifest_digest': manifest_digest, + 'manifest_media_type': manifest_media_type, + 'manifest_size_bytes': manifest_size_bytes, + 'config_digest': config_digest, + 'graph_sha256': graph_sha256, + 'layers': layers, + } + selection_evidence_sha256 = canonical_docker_depth_selection_evidence_hash(evidence) + if str(raw.get('selection_evidence_sha256') or '') != selection_evidence_sha256: + raise ScanEventConflictError('Docker depth selector evidence hash conflicts') + normalized_records.append({ + **evidence, + 'normalized_target': normalized_target, + 'selection_evidence_sha256': selection_evidence_sha256, + }) + return candidate_count, normalized_records + + def _docker_depth_exact_reactivation_locked(self, experiment_id, queue): + from docker_depth_experiment import ( + DOCKER_DEPTH_DYNAMIC_HOLD_REASON, + DOCKER_DEPTH_HOLD_REASON, + DOCKER_DEPTH_RELEASE_REASON, + ) + + experiment = self.conn.execute( + '''SELECT config_sha256, provenance_policy_sha256, + hold_manifest_sha256 + FROM docker_depth_experiments WHERE id = ?''', + (experiment_id,), + ).fetchone() + events = self.conn.execute( + '''SELECT * FROM target_queue_policy_events + WHERE queue_id = ? ORDER BY id LIMIT 3 FOR UPDATE''', + (queue['id'],), + ).fetchall() + if not experiment or len(events) != 2: + return False + cold, reactivation = events + if ( + cold['action'] != 'cold' + or reactivation['action'] != 'reactivate' + or int(reactivation['reverses_event_id'] or 0) != int(cold['id']) + or any( + event['experiment_id'] is None + or int(event['experiment_id']) != int(experiment_id) + or int(event['queue_id']) != int(queue['id']) + or str(event['source']) != str(queue['source']) + or str(event['platform']) != str(queue['platform']) + or str(event['query']) != str(queue['query']) + or str(event['config_sha256']) != str(experiment['config_sha256']) + or str(event['policy_sha256']) + != str(experiment['provenance_policy_sha256']) + for event in events + ) + or str(cold['manifest_sha256']) + != str(experiment['hold_manifest_sha256']) + or str(cold['reason_code']) not in ( + DOCKER_DEPTH_HOLD_REASON, DOCKER_DEPTH_DYNAMIC_HOLD_REASON, + ) + or str(reactivation['reason_code']) != DOCKER_DEPTH_RELEASE_REASON + or str(cold['next_status']) != 'cold' + or str(reactivation['prior_status']) != 'cold' + or str(reactivation['next_status']) != str(queue['status']) + or str(cold['prior_status']) != str(queue['status']) + ): + return False + cold_entry = { + 'queue_id': int(queue['id']), + 'source': str(queue['source']), + 'platform': str(queue['platform']), + 'query': str(queue['query']), + 'prior_status': str(cold['prior_status']), + 'prior_updated_at': str(cold['prior_updated_at']), + } + reactivation_entry = { + 'queue_id': int(queue['id']), + 'source': str(queue['source']), + 'platform': str(queue['platform']), + 'query': str(queue['query']), + 'cold_event_id': int(cold['id']), + 'restore_status': str(reactivation['next_status']), + 'prior_updated_at': str(reactivation['prior_updated_at']), + } + return ( + str(cold['review_audit_sha256']) + == self._target_queue_policy_audit_sha256( + 'cold', cold['manifest_sha256'], cold_entry, experiment_id, + ) + and str(reactivation['review_audit_sha256']) + == self._target_queue_policy_audit_sha256( + 'reactivate', reactivation['manifest_sha256'], + reactivation_entry, experiment_id, + ) + ) + + def _record_docker_depth_candidate_skip_locked( + self, experiment, member, repository_queue_id, candidate_kind, + candidate_ordinal, candidate_identity, reason, now, + ): + candidate_identity_sha256, evidence_sha256 = ( + self._docker_depth_candidate_skip_evidence( + experiment['id'], member['id'], repository_queue_id, + candidate_kind, candidate_ordinal, candidate_identity, reason, + ) + ) + self.conn.execute( + '''INSERT INTO docker_depth_experiment_candidate_skips( + experiment_id, experiment_repository_id, repository_queue_id, + candidate_kind, candidate_ordinal, reason_code, + candidate_identity_sha256, evidence_sha256, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT( + experiment_repository_id, candidate_kind, + candidate_identity_sha256 + ) DO NOTHING''', + ( + experiment['id'], member['id'], repository_queue_id, + candidate_kind, candidate_ordinal, reason, + candidate_identity_sha256, evidence_sha256, now, + ), + ) + + def _docker_depth_immutable_candidate_conflict_locked( + self, experiment_id, source, record, + ): + row = self.conn.execute( + '''SELECT * FROM target_queue + WHERE source = ? AND normalized_target = ? FOR UPDATE''', + (source, record['normalized_target']), + ).fetchone() + if not row: + return None, None + experiment_target = self.conn.execute( + '''SELECT id FROM docker_depth_experiment_targets + WHERE experiment_id = ? AND target_queue_id = ?''', + (experiment_id, row['id']), + ).fetchone() + historical = self.conn.execute( + '''SELECT + EXISTS(SELECT 1 FROM target_scans WHERE queue_id = ?) AS scanned, + EXISTS(SELECT 1 FROM result_reservations WHERE queue_id = ?) AS reserved, + EXISTS(SELECT 1 FROM target_queue_policy_events WHERE queue_id = ?) AS policy, + EXISTS( + SELECT 1 FROM result_reservations reservation + JOIN pipeline_quarantine quarantine + ON quarantine.reservation_id = reservation.id + WHERE reservation.queue_id = ? + ) AS quarantined, + EXISTS( + SELECT 1 FROM docker_image_blob_coverage coverage + JOIN docker_content_blobs blob + ON blob.digest = coverage.blob_digest + AND blob.coverage_policy_sha256 = coverage.coverage_policy_sha256 + WHERE coverage.queue_id = ? + AND blob.state IN ('leased','submitted') + ) AS blob_fenced''', + (row['id'], row['id'], row['id'], row['id'], row['id']), + ).fetchone() + if row['status'] in ('done', 'failed') or row['completed_at'] is not None or bool( + historical['scanned'] + ): + return 'immutable_target_terminal', int(row['id']) + if row['status'] == 'quarantined' or bool(historical['quarantined']): + return 'immutable_target_quarantined', int(row['id']) + if row['status'] == 'cold' or bool(historical['policy']): + return 'immutable_target_independently_cold', int(row['id']) + fenced = ( + row['status'] == 'in_progress' + or bool(historical['reserved']) + or bool(historical['blob_fenced']) + or any(row[name] is not None for name in ( + 'lease_owner', 'lease_token', 'claim_batch', 'leased_at', + 'lease_expires_at', 'current_result_reservation_id', + 'claim_event_id', 'resolver_token', + )) + ) + if fenced: + return 'immutable_target_fenced', int(row['id']) + if experiment_target and row['status'] in ('pending', 'deferred'): + return None, int(row['id']) + return 'immutable_target_existing', int(row['id']) + + def _docker_depth_repository_candidate_conflict( + self, experiment, queue_id, *, lock, + ): + suffix = ' FOR UPDATE' if lock else '' + queue = self.conn.execute( + f'SELECT * FROM target_queue WHERE id = ?{suffix}', + (queue_id,), + ).fetchone() + if not queue: + return 'repository_candidate_absent' + fences = any(queue[name] is not None for name in ( + 'target_scan_id', 'lease_owner', 'lease_token', 'claim_batch', + 'leased_at', 'lease_expires_at', 'current_result_reservation_id', + 'claim_event_id', 'resolver_token', + )) or str(queue['resolver_state'] or '') == 'resolving' + if fences or self.conn.execute( + '''SELECT 1 FROM target_scans WHERE queue_id = ? + UNION ALL SELECT 1 FROM result_reservations WHERE queue_id = ? + LIMIT 1''', + (queue_id, queue_id), + ).fetchone(): + return 'repository_candidate_fenced' + events = self.conn.execute( + '''SELECT * FROM target_queue_policy_events + WHERE queue_id = ? ORDER BY id LIMIT 3''', + (queue_id,), + ).fetchall() + if queue['status'] in ('pending', 'deferred') and not events: + return None + if queue['status'] == 'cold' and len(events) == 1: + event = events[0] + if ( + event['action'] == 'cold' + and int(event['experiment_id'] or 0) == int(experiment['id']) + and str(event['reason_code']) in ( + 'docker_depth_experiment_hold', + 'docker_depth_experiment_dynamic_hold', + ) + and str(event['config_sha256']) == str(experiment['config_sha256']) + and str(event['policy_sha256']) + == str(experiment['provenance_policy_sha256']) + and str(event['manifest_sha256']) + == str(experiment['hold_manifest_sha256']) + and str(event['next_status']) == 'cold' + and str(event['prior_status']) in ('pending', 'deferred') + and str(event['review_audit_sha256']) + == self._target_queue_policy_audit_sha256( + 'cold', event['manifest_sha256'], { + 'queue_id': int(queue['id']), + 'source': str(queue['source']), + 'platform': str(queue['platform']), + 'query': str(queue['query']), + 'prior_status': str(event['prior_status']), + 'prior_updated_at': str(event['prior_updated_at']), + }, int(experiment['id']), + ) + ): + return None + if queue['status'] in ('done', 'failed'): + return 'repository_candidate_terminal' + if queue['status'] == 'quarantined': + return 'repository_candidate_quarantined' + if queue['status'] == 'cold' or events: + return 'repository_candidate_independently_cold' + return 'repository_candidate_ineligible' + + def _docker_depth_repository_candidate_conflict_locked( + self, experiment, queue_id, + ): + return self._docker_depth_repository_candidate_conflict( + experiment, queue_id, lock=True, + ) + + def _docker_depth_repository_replacement_candidates( + self, experiment, member, authority, *, lock, + ): + candidates = self.conn.execute( + '''SELECT provenance.repository_queue_id, + MIN(observation.search_rank) AS best_search_rank, + MIN(observation.page_id) AS eligibility_page_id, + MIN(queue.normalized_target) AS normalized_target + FROM docker_repository_query_provenance provenance + JOIN docker_repository_query_observations observation + ON observation.source = provenance.source + AND observation.query = provenance.query + AND observation.repository_queue_id = provenance.repository_queue_id + JOIN docker_discovery_pages page ON page.id = observation.page_id + JOIN docker_discovery_passes discovery_pass ON discovery_pass.id = page.pass_id + JOIN target_queue queue ON queue.id = provenance.repository_queue_id + WHERE provenance.source = ? AND provenance.query = ? + AND provenance.provenance_kind = 'fresh_page' + AND provenance.fresh_coverage_eligible = 1 + AND provenance.fresh_complete_observation_count > 0 + AND page.query_ordinal = ? AND page.query = ? + AND discovery_pass.source = ? AND discovery_pass.pass_kind = 'deep' + AND discovery_pass.collection_generation = ? + AND discovery_pass.policy_sha256 = ? + AND discovery_pass.ordered_queries_sha256 = ? + AND discovery_pass.expected_query_count = ? + AND discovery_pass.state = 'complete' + AND queue.source = ? AND queue.platform = 'docker' + AND queue.target NOT LIKE '%@%' AND queue.normalized_target NOT LIKE '%@%' + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_repositories other + WHERE other.experiment_id = ? AND other.query_ordinal = ? + AND (other.repository_queue_id = queue.id + OR other.replacement_repository_queue_id = queue.id) + ) + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_candidate_skips skipped + WHERE skipped.experiment_repository_id = ? + AND skipped.candidate_kind = 'repository' + AND skipped.repository_queue_id = queue.id + ) + GROUP BY provenance.repository_queue_id + ORDER BY MIN(observation.search_rank), provenance.repository_queue_id + LIMIT ?''', + ( + authority['source'], member['query'], member['query_ordinal'], + member['query'], authority['source'], + authority['collection_generation'], authority['provenance_policy_sha256'], + authority['ordered_queries_sha256'], authority['query_count'], + authority['source'], experiment['id'], member['query_ordinal'], + member['id'], 3000, + ), + ).fetchall() + if lock and candidates: + ids = sorted({int(row['repository_queue_id']) for row in candidates}) + placeholders = ','.join('?' for _ in ids) + self.conn.execute( + f'''SELECT id FROM target_queue WHERE id IN ({placeholders}) + ORDER BY id FOR UPDATE''', + tuple(ids), + ).fetchall() + inspected = [] + for candidate in candidates: + conflict = self._docker_depth_repository_candidate_conflict( + experiment, int(candidate['repository_queue_id']), lock=lock, + ) + inspected.append({ + 'repository_queue_id': int(candidate['repository_queue_id']), + 'best_search_rank': int(candidate['best_search_rank']), + 'eligibility_page_id': int(candidate['eligibility_page_id']), + 'target_identity_sha256': hashlib.sha256( + str(candidate['normalized_target']).encode('utf-8') + ).hexdigest(), + 'conflict_reason': str(conflict or ''), + }) + if not conflict: + break + return inspected + + def _finalize_docker_depth_query_breadth_locked( + self, experiment, member, now, + ): + query_row = self.conn.execute( + '''SELECT selected_repository_count + FROM docker_depth_experiment_queries + WHERE experiment_id = ? AND query_ordinal = ? FOR UPDATE''', + (experiment['id'], member['query_ordinal']), + ).fetchone() + if not query_row: + raise ScanEventConflictError('Docker depth query authority is unavailable') + query_members = self.conn.execute( + '''SELECT * FROM docker_depth_experiment_repositories + WHERE experiment_id = ? AND query_ordinal = ? + ORDER BY repository_rank, repository_queue_id, id + FOR UPDATE''', + (experiment['id'], member['query_ordinal']), + ).fetchall() + if len(query_members) != int(query_row['selected_repository_count']): + raise ScanEventConflictError('Docker depth query cohort membership drifted') + if not all( + str(row['work_state']) in ('resolved', 'skipped') + for row in query_members + ): + return False + for row in query_members: + reason, _evidence_sha256 = ( + self._docker_depth_terminal_repository_evidence_locked( + experiment, row, + ) + ) + if reason: + raise ScanEventConflictError( + 'Docker depth terminal repository evidence drifted' + ) + from docker_depth_experiment import DOCKER_RANK1_BREADTH_SELECTOR_VERSION + if str(experiment['selector_version']) == DOCKER_RANK1_BREADTH_SELECTOR_VERSION: + self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET is_deep_probe = 0, updated_at = ? + WHERE experiment_id = ? AND query_ordinal = ? + AND is_deep_probe <> 0''', + (now, experiment['id'], member['query_ordinal']), + ) + return True + if self.conn.execute( + '''SELECT 1 FROM docker_depth_experiment_selections + WHERE experiment_id = ? AND query_ordinal = ? AND image_rank > 1 + LIMIT 1''', + (experiment['id'], member['query_ordinal']), + ).fetchone(): + return True + eligible = [ + row for row in query_members + if str(row['work_state']) == 'resolved' + and int(row['selected_image_count']) >= 1 + ] + deep_id = None + if eligible: + deep = min(eligible, key=lambda row: ( + -int(row['candidate_distinct_graph_count']), + int(row['repository_rank']), int(row['repository_queue_id']), + )) + deep_id = int(deep['id']) + self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET is_deep_probe = CASE WHEN id = ? THEN 1 ELSE 0 END, + work_state = CASE + WHEN id = ? AND candidate_distinct_graph_count > selected_image_count + AND selected_image_count >= 1 + THEN 'pending' ELSE work_state END, + resolved_at = CASE + WHEN id = ? AND candidate_distinct_graph_count > selected_image_count + AND selected_image_count >= 1 + THEN NULL ELSE resolved_at END, + updated_at = ? + WHERE experiment_id = ? AND query_ordinal = ?''', + ( + deep_id, deep_id, deep_id, now, + experiment['id'], member['query_ordinal'], + ), + ) + return True + + def _replace_docker_depth_repository_locked( + self, experiment, member, authority, reason, now, candidate_count, + ): + current_queue_id = int( + member['replacement_repository_queue_id'] + or member['repository_queue_id'] + ) + self._record_docker_depth_candidate_skip_locked( + experiment, member, current_queue_id, 'repository', + int(member['replacement_count']) + 1, + {'repository_queue_id': current_queue_id}, reason, now, + ) + candidates = self._docker_depth_repository_replacement_candidates( + experiment, member, authority, lock=True, + ) + replacement = None + for candidate in candidates: + candidate_queue_id = candidate['repository_queue_id'] + conflict = candidate['conflict_reason'] + if conflict: + self._record_docker_depth_candidate_skip_locked( + experiment, member, candidate_queue_id, 'repository', + int(candidate['best_search_rank']), + {'repository_queue_id': candidate_queue_id}, conflict, now, + ) + continue + replacement = candidate + break + if not replacement: + from docker_depth_experiment import DOCKER_DEPTH_REPOSITORY_SKIP_REASON + + cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = 'skipped', resolver_owner = NULL, + resolver_token = NULL, resolver_expires_at = NULL, + resolver_due_at = NULL, + candidate_distinct_graph_count = CASE + WHEN candidate_distinct_graph_count >= ? + THEN candidate_distinct_graph_count ELSE ? END, + selected_image_count = 0, last_error_code = ?, + resolved_at = ?, updated_at = ? + WHERE id = ? AND experiment_id = ? + AND work_state = 'resolving' + AND resolver_generation = ? AND resolver_owner = ? + AND resolver_token = ? AND resolver_expires_at > ?''', + ( + candidate_count, candidate_count, + DOCKER_DEPTH_REPOSITORY_SKIP_REASON, now, now, + member['id'], experiment['id'], member['resolver_generation'], + member['resolver_owner'], member['resolver_token'], now, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth repository skip lost its lease fence' + ) + self._finalize_docker_depth_query_breadth_locked( + experiment, member, now, + ) + experiment = self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment['id'],), + ).fetchone() + experiment, selection_reason = ( + self._freeze_docker_depth_runtime_selection_locked( + experiment, authority, now, + ) + ) + if selection_reason: + return self._hold_docker_depth_experiment_locked( + experiment, selection_reason, now, + ) + drift_reason = self._docker_depth_persisted_drift_reason_locked( + experiment, authority, now, + ) + if drift_reason: + return self._hold_docker_depth_experiment_locked( + experiment, drift_reason, now, + ) + experiment = self._advance_docker_depth_experiment_state_locked( + experiment, now, + ) + return { + 'status': 'skipped', 'committed': True, + 'reason': DOCKER_DEPTH_REPOSITORY_SKIP_REASON, + 'experiment_state': str(experiment['state']), + } + previous_hash = str(member['replacement_evidence_sha256'] or '') + replacement_document = { + 'schema': 1, + 'type': 'docker-depth-repository-replacement-v1', + 'experiment_id': int(experiment['id']), + 'experiment_repository_id': int(member['id']), + 'replacement_number': int(member['replacement_count']) + 1, + 'from_repository_queue_id': current_queue_id, + 'to_repository_queue_id': int(replacement['repository_queue_id']), + 'eligibility_page_id': int(replacement['eligibility_page_id']), + 'best_search_rank': int(replacement['best_search_rank']), + 'previous_evidence_sha256': previous_hash, + 'reason_code': str(reason), + } + replacement_sha256 = hashlib.sha256(json.dumps( + replacement_document, ensure_ascii=True, allow_nan=False, + sort_keys=True, separators=(',', ':'), + ).encode('utf-8')).hexdigest() + self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET replacement_repository_queue_id = ?, + replacement_eligibility_page_id = ?, + replacement_count = replacement_count + 1, + replacement_evidence_sha256 = ?, work_state = 'pending', + resolver_owner = NULL, resolver_token = NULL, + resolver_expires_at = NULL, resolver_due_at = NULL, + resolver_attempts = 0, selected_image_count = 0, + candidate_distinct_graph_count = 0, + last_error_code = 'repository_candidate_replaced', + resolved_at = NULL, updated_at = ? WHERE id = ?''', + ( + replacement['repository_queue_id'], replacement['eligibility_page_id'], + replacement_sha256, now, member['id'], + ), + ) + return { + 'status': 'replaced', 'committed': True, + 'reason': 'repository_candidate_replaced', + } + + @staticmethod + def _rerank_docker_depth_selection_record(record, image_rank): + from docker_depth_experiment import canonical_docker_depth_selection_evidence_hash + evidence_keys = ( + 'schema', 'type', 'selector_version', 'selector_sha256', + 'candidate_distinct_graph_count', 'selection_reason', 'target', + 'repository', 'manifest_digest', 'manifest_media_type', + 'manifest_size_bytes', 'config_digest', 'graph_sha256', 'layers', + ) + evidence = {key: record[key] for key in evidence_keys} + original_rank = int(record['image_rank']) + evidence['image_rank'] = int(image_rank) + if original_rank != image_rank: + evidence['selection_reason'] = 'replacement_after_exclusion' + return { + **record, + **evidence, + 'selection_evidence_sha256': canonical_docker_depth_selection_evidence_hash( + evidence + ), + } + + def _docker_depth_target_queue_locked( + self, experiment_id, source, query, record, now, + ): + row = self.conn.execute( + '''SELECT * FROM target_queue + WHERE source = ? AND normalized_target = ? FOR UPDATE''', + (source, record['normalized_target']), + ).fetchone() + inserted = False + if not row: + row = self.conn.execute( + '''INSERT INTO target_queue( + source, platform, query, target, normalized_target, + status, created_at, updated_at + ) VALUES (?, 'docker', ?, ?, ?, 'pending', ?, ?) + ON CONFLICT(source, normalized_target) DO NOTHING + RETURNING *''', + ( + source, query, record['target'], record['normalized_target'], + now, now, + ), + ).fetchone() + if not row: + row = self.conn.execute( + '''SELECT * FROM target_queue + WHERE source = ? AND normalized_target = ? FOR UPDATE''', + (source, record['normalized_target']), + ).fetchone() + else: + inserted = True + experiment_target = self.conn.execute( + '''SELECT * FROM docker_depth_experiment_targets + WHERE experiment_id = ? AND target_queue_id = ? FOR UPDATE''', + (experiment_id, row['id']), + ).fetchone() if row else None + audited_reactivation = bool( + row and not inserted and self._docker_depth_exact_reactivation_locked( + experiment_id, row, + ) + ) + if ( + not row + or row['source'] != source + or row['platform'] != 'docker' + or row['status'] not in ('pending', 'deferred') + or normalize_target(row['target'], 'docker') != record['normalized_target'] + or row['target_scan_id'] is not None + or row['lease_owner'] is not None + or row['lease_token'] is not None + or row['claim_batch'] is not None + or row['leased_at'] is not None + or row['lease_expires_at'] is not None + or row['current_result_reservation_id'] is not None + or row['claim_event_id'] is not None + or row['completed_at'] is not None + or (not inserted and not experiment_target and not audited_reactivation) + ): + raise ScanEventConflictError( + 'Docker depth immutable target has prior or conflicting queue state' + ) + historical = self.conn.execute( + '''SELECT + EXISTS(SELECT 1 FROM target_scans WHERE queue_id = ?) AS scanned, + EXISTS(SELECT 1 FROM result_reservations WHERE queue_id = ?) AS reserved, + EXISTS(SELECT 1 FROM target_queue_policy_events WHERE queue_id = ?) AS policy, + EXISTS( + SELECT 1 FROM result_reservations reservation + JOIN pipeline_quarantine quarantine + ON quarantine.reservation_id = reservation.id + WHERE reservation.queue_id = ? + ) AS quarantined''', + (row['id'], row['id'], row['id'], row['id']), + ).fetchone() + if ( + bool(historical['scanned']) + or bool(historical['reserved']) + or bool(historical['quarantined']) + or (bool(historical['policy']) and not audited_reactivation) + ): + raise ScanEventConflictError( + 'Docker depth immutable target has historical or independently cold state' + ) + return dict(row) + + def _persist_docker_depth_manifest_locked(self, queue_id, source, record, now): + expected = { + 'source': source, + 'repository': record['repository'], + 'manifest_digest': record['manifest_digest'], + 'manifest_media_type': record['manifest_media_type'], + 'config_digest': record['config_digest'], + 'graph_sha256': record['graph_sha256'], + 'manifest_size_bytes': record['manifest_size_bytes'], + 'layer_count': len(record['layers']), + } + manifest = self.conn.execute( + '''SELECT * FROM docker_image_manifests + WHERE target_queue_id = ? FOR UPDATE''', + (queue_id,), + ).fetchone() + if manifest: + if any( + (int(manifest[key]) if isinstance(value, int) else str(manifest[key])) != value + for key, value in expected.items() + ): + raise ScanEventConflictError('Docker depth immutable manifest metadata conflicts') + else: + manifest = self.conn.execute( + '''INSERT INTO docker_image_manifests( + target_queue_id, source, repository, manifest_digest, + manifest_media_type, config_digest, graph_sha256, + manifest_size_bytes, layer_count, resolved_at, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) RETURNING *''', + ( + queue_id, source, record['repository'], record['manifest_digest'], + record['manifest_media_type'], record['config_digest'], + record['graph_sha256'], record['manifest_size_bytes'], + len(record['layers']), now, now, + ), + ).fetchone() + for layer in record['layers']: + self.conn.execute( + '''INSERT INTO docker_manifest_layers( + manifest_id, position_from_base, position_from_top, + layer_digest, media_type, layer_size_bytes, + descriptor_sha256, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?)''', + ( + manifest['id'], layer['position_from_base'], + layer['position_from_top'], layer['digest'], + layer['media_type'], layer['size_bytes'], + layer['descriptor_sha256'], now, + ), + ) + stored_layers = self.conn.execute( + '''SELECT position_from_base, position_from_top, layer_digest, + media_type, layer_size_bytes, descriptor_sha256 + FROM docker_manifest_layers WHERE manifest_id = ? + ORDER BY position_from_base''', + (manifest['id'],), + ).fetchall() + expected_layers = [( + layer['position_from_base'], layer['position_from_top'], layer['digest'], + layer['media_type'], layer['size_bytes'], layer['descriptor_sha256'], + ) for layer in record['layers']] + if [( + int(layer['position_from_base']), int(layer['position_from_top']), + str(layer['layer_digest']), str(layer['media_type']), + int(layer['layer_size_bytes']), str(layer['descriptor_sha256']), + ) for layer in stored_layers] != expected_layers: + raise ScanEventConflictError('Docker depth immutable manifest layers conflict') + return int(manifest['id']) + + @staticmethod + def _docker_depth_dispatch_position(query_ordinal, repository_rank, image_rank, query_count): + if image_rank == 1: + return 1, (repository_rank - 1) * query_count + query_ordinal + 1 + if image_rank <= 3: + return 2, (image_rank - 2) * query_count + query_ordinal + 1 + return 3, (image_rank - 4) * query_count + query_ordinal + 1 + + def _defer_or_hold_docker_depth_resolution_conflict( + self, source, repository_id, resolver_generation, resolver_owner, + resolver_token, authority, error, *, final_cutover, + ): + from docker_depth_experiment import ( + DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS, + DOCKER_DEPTH_RESOLVER_RETRY_MAX_SECONDS, + DOCKER_DEPTH_RESOLVER_RETRY_SECONDS, + ) + + now = utc_now_iso() + experiment = self.conn.execute( + '''SELECT * FROM docker_depth_experiments + WHERE experiment_key = ? AND source = ? FOR UPDATE''', + (authority['experiment_key'], source), + ).fetchone() + member = self.conn.execute( + '''SELECT * FROM docker_depth_experiment_repositories + WHERE id = ? AND experiment_id = ? FOR UPDATE''', + (repository_id, experiment['id'] if experiment else 0), + ).fetchone() + if not ( + experiment + and str(experiment['state']) == 'resolving' + and member + and member['work_state'] == 'resolving' + and int(member['resolver_generation']) == int(resolver_generation) + and member['resolver_owner'] == resolver_owner + and member['resolver_token'] == resolver_token + and str(member['resolver_expires_at'] or '') > now + ): + self.conn.rollback() + return {'status': 'stale', 'committed': False} + checked, authority_reason = self._locked_docker_depth_experiment_authority( + authority, final_cutover=final_cutover, now=now, + ) + if not checked: + self.conn.commit() + return { + 'status': 'held', 'committed': True, + 'reason': authority_reason, + } + attempts = int(member['resolver_attempts']) + capacity_conflict = 'capacity' in str(error).lower() + if capacity_conflict or attempts >= DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS: + reason = ( + 'resolver_capacity_conflict' if capacity_conflict + else 'resolver_evidence_attempt_limit' + ) + self._hold_docker_depth_experiment_locked(experiment, reason, now) + self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = 'held', resolver_owner = NULL, + resolver_token = NULL, resolver_expires_at = NULL, + resolver_due_at = NULL, last_error_code = ?, updated_at = ? + WHERE id = ?''', + (reason, now, repository_id), + ) + self.conn.commit() + return {'status': 'held', 'committed': True, 'reason': reason} + delay = min( + DOCKER_DEPTH_RESOLVER_RETRY_MAX_SECONDS, + DOCKER_DEPTH_RESOLVER_RETRY_SECONDS + * (2 ** min(8, max(0, attempts - 1))), + ) + due = datetime.fromtimestamp( + time.time() + delay, timezone.utc, + ).isoformat(timespec='seconds') + self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = 'pending', resolver_owner = NULL, + resolver_token = NULL, resolver_expires_at = NULL, + resolver_due_at = ?, last_error_code = 'resolver_evidence_conflict', + updated_at = ? WHERE id = ?''', + (due, now, repository_id), + ) + self.conn.commit() + return { + 'status': 'conflict_retry', 'committed': True, + 'retry_at': due, 'reason': 'resolver_evidence_conflict', + } + + def finish_docker_depth_experiment_resolution( + self, source, repository_id, resolver_generation, resolver_token, + outcome=None, error='', complete=None, *, resolver_owner=None, + authority=None, retry_at=None, claim_attempt_consumed=True, + final_cutover=False, + ): + if ( + not self.conn + or not self.conn.is_postgres + or source != 'dockerhub' + or repository_id is None + or not resolver_token + or not resolver_owner + ): + return {'status': 'unavailable', 'committed': False} + if isinstance(authority, dict) and authority.get('enabled') is False: + held = self._hold_disabled_docker_depth_experiment(authority) + return { + 'status': 'held' if held else 'unavailable', + 'committed': held, 'reason': 'experiment_disabled', + } + if not isinstance(authority, dict) or authority.get('enabled') is not True: + return {'status': 'unavailable', 'committed': False} + try: + normalized = self._normalize_docker_depth_resolver_authority(authority) + except (TypeError, ValueError): + return { + 'status': 'invalid', 'committed': False, + 'reason': 'authority_payload_invalid', + } + outcome_tags = ( + outcome.get('tags', ()) if isinstance(outcome, dict) + else getattr(outcome, 'tags', ()) + ) + complete = bool(outcome_tags) if complete is None else bool(complete) + if not claim_attempt_consumed and (complete or not retry_at): + return {'status': 'invalid', 'committed': False} + try: + if complete: + self._require_discovery_admission_locked() + now = utc_now_iso() + experiment = self.conn.execute( + '''SELECT * FROM docker_depth_experiments + WHERE experiment_key = ? AND source = ? FOR UPDATE''', + (normalized['experiment_key'], source), + ).fetchone() + member = self.conn.execute( + '''SELECT member.*, queue.normalized_target AS repository + FROM docker_depth_experiment_repositories member + JOIN target_queue queue ON queue.id = COALESCE( + member.replacement_repository_queue_id, + member.repository_queue_id + ) + WHERE member.id = ? AND member.experiment_id = ? + FOR UPDATE OF member, queue''', + (repository_id, experiment['id'] if experiment else 0), + ).fetchone() + if not ( + experiment + and str(experiment['state']) == 'resolving' + and member + and member['source'] == source + and member['work_state'] == 'resolving' + and int(member['resolver_generation']) == int(resolver_generation) + and member['resolver_owner'] == resolver_owner + and member['resolver_token'] == resolver_token + and str(member['resolver_expires_at'] or '') > now + ): + self.conn.rollback() + return {'status': 'stale', 'committed': False} + experiment, authority_reason = self._locked_docker_depth_experiment_authority( + normalized, final_cutover=final_cutover, now=now, + ) + if not experiment: + self.conn.commit() + return { + 'status': 'held', 'committed': True, + 'reason': authority_reason, + } + stage = 'breadth' if int(member['selected_image_count']) == 0 else 'deep' + if stage == 'deep' and not bool(member['is_deep_probe']): + raise ScanEventConflictError('Docker depth deep resolver authority conflicts') + + if not complete: + from docker_depth_experiment import ( + DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS, + DOCKER_DEPTH_RESOLVER_RETRY_MAX_SECONDS, + DOCKER_DEPTH_RESOLVER_RETRY_SECONDS, + ) + + if ( + claim_attempt_consumed + and int(member['resolver_attempts']) + >= DOCKER_DEPTH_RESOLVER_MAX_ATTEMPTS + ): + terminal = self._terminalize_docker_depth_attempt_limit_locked( + experiment, member, normalized, now, + ) + self.conn.commit() + terminal.pop('experiment', None) + return terminal + now_dt = datetime.now(timezone.utc) + if retry_at: + try: + due_dt = datetime.fromisoformat(str(retry_at).replace('Z', '+00:00')) + if due_dt.tzinfo is None: + due_dt = due_dt.replace(tzinfo=timezone.utc) + due_dt = due_dt.astimezone(timezone.utc) + except (TypeError, ValueError): + self.conn.rollback() + return {'status': 'invalid', 'committed': False} + if due_dt > now_dt + timedelta(seconds=3605): + self.conn.rollback() + return {'status': 'invalid', 'committed': False} + due_dt = max(due_dt, now_dt) + else: + exponent = min(8, max(0, int(member['resolver_attempts']) - 1)) + due_dt = datetime.fromtimestamp( + time.time() + min( + DOCKER_DEPTH_RESOLVER_RETRY_MAX_SECONDS, + DOCKER_DEPTH_RESOLVER_RETRY_SECONDS * (2 ** exponent), + ), + timezone.utc, + ) + due = due_dt.isoformat(timespec='seconds') + cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = 'pending', resolver_owner = NULL, + resolver_token = NULL, resolver_expires_at = NULL, + resolver_due_at = ?, resolver_attempts = CASE + WHEN ? = 0 AND resolver_attempts > 0 + THEN resolver_attempts - 1 ELSE resolver_attempts END, + last_error_code = ?, updated_at = ? + WHERE id = ? AND experiment_id = ? + AND resolver_generation = ? AND resolver_owner = ? + AND resolver_token = ? AND resolver_expires_at > ?''', + ( + due, 1 if claim_attempt_consumed else 0, + first_line(error or 'docker_depth_resolution_deferred', 128), + now, repository_id, experiment['id'], resolver_generation, + resolver_owner, resolver_token, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + self.conn.rollback() + return {'status': 'stale', 'committed': False} + self.conn.commit() + return {'status': 'deferred', 'committed': True, 'retry_at': due} + + candidate_count, candidate_records = self._normalize_docker_depth_resolution_outcome( + outcome, str(member['repository']), normalized, stage, + ) + existing_selections = self.conn.execute( + '''SELECT selection.image_rank, selection.graph_sha256, + queue.normalized_target + FROM docker_depth_experiment_selections selection + JOIN docker_depth_experiment_targets target + ON target.id = selection.experiment_target_id + JOIN target_queue queue ON queue.id = target.target_queue_id + WHERE selection.experiment_repository_id = ? + ORDER BY selection.image_rank''', + (repository_id,), + ).fetchall() + existing_graphs = { + str(selection['graph_sha256']) for selection in existing_selections + } + existing_targets = { + str(selection['normalized_target']) for selection in existing_selections + } + eligible_records = [] + resolution_repository_queue_id = int( + member['replacement_repository_queue_id'] + or member['repository_queue_id'] + ) + for candidate in candidate_records: + if ( + candidate['graph_sha256'] in existing_graphs + or candidate['normalized_target'] in existing_targets + ): + continue + conflict, _candidate_queue_id = ( + self._docker_depth_immutable_candidate_conflict_locked( + experiment['id'], source, candidate, + ) + ) + if conflict: + self._record_docker_depth_candidate_skip_locked( + experiment, member, resolution_repository_queue_id, + 'image', candidate['image_rank'], { + 'normalized_target': candidate['normalized_target'], + 'graph_sha256': candidate['graph_sha256'], + }, conflict, now, + ) + continue + eligible_records.append(candidate) + remaining = normalized['images_per_repository'] - len(existing_selections) + if stage == 'breadth': + remaining = 1 + records = [ + self._rerank_docker_depth_selection_record( + record, len(existing_selections) + index, + ) + for index, record in enumerate(eligible_records[:remaining], 1) + ] + if stage == 'breadth' and not records: + replacement = self._replace_docker_depth_repository_locked( + experiment, member, normalized, + 'no_eligible_physical_target', now, candidate_count, + ) + self.conn.commit() + return replacement + actual_selection_count = int(self.conn.execute( + '''SELECT COUNT(*) AS count FROM docker_depth_experiment_selections + WHERE experiment_id = ?''', + (experiment['id'],), + ).fetchone()['count']) + actual_target_count = int(self.conn.execute( + '''SELECT COUNT(*) AS count FROM docker_depth_experiment_targets + WHERE experiment_id = ?''', + (experiment['id'],), + ).fetchone()['count']) + if ( + actual_selection_count != int(experiment['selection_count']) + or actual_target_count != int(experiment['target_count']) + ): + raise ScanEventConflictError('Docker depth counter capacity drifted') + + inserted_selections = 0 + for record in records: + queue = self._docker_depth_target_queue_locked( + experiment['id'], source, member['query'], record, now, + ) + manifest_id = self._persist_docker_depth_manifest_locked( + queue['id'], source, record, now, + ) + dispatch_wave, dispatch_order = self._docker_depth_dispatch_position( + int(member['query_ordinal']), int(member['repository_rank']), + record['image_rank'], normalized['query_count'], + ) + capacity = self.reserve_docker_depth_experiment_target_capacity_locked( + experiment['id'], queue['id'], manifest_id, + dispatch_wave, dispatch_order, now, + ) + selection = self.conn.execute( + '''SELECT * FROM docker_depth_experiment_selections + WHERE experiment_repository_id = ? AND image_rank = ? + FOR UPDATE''', + (repository_id, record['image_rank']), + ).fetchone() + expected_selection = { + 'experiment_id': int(experiment['id']), + 'query_ordinal': int(member['query_ordinal']), + 'experiment_target_id': int(capacity['target']['id']), + 'selection_reason': record['selection_reason'], + 'selection_evidence_sha256': record['selection_evidence_sha256'], + 'graph_sha256': record['graph_sha256'], + } + if selection: + if any( + (int(selection[key]) if isinstance(value, int) else str(selection[key])) + != value for key, value in expected_selection.items() + ): + raise ScanEventConflictError('Docker depth persisted selector evidence conflicts') + else: + if actual_selection_count + inserted_selections + 1 > normalized[ + 'theoretical_max_targets' + ]: + raise ScanEventConflictError( + 'Docker depth theoretical selection capacity is exhausted' + ) + self.conn.execute( + '''INSERT INTO docker_depth_experiment_selections( + experiment_id, query_ordinal, experiment_repository_id, + experiment_target_id, image_rank, selection_reason, + selection_evidence_sha256, graph_sha256, + selected_at, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + experiment['id'], member['query_ordinal'], repository_id, + capacity['target']['id'], record['image_rank'], + record['selection_reason'], record['selection_evidence_sha256'], + record['graph_sha256'], now, now, + ), + ) + inserted_selections += 1 + if inserted_selections: + cursor = self.conn.execute( + '''UPDATE docker_depth_experiments + SET selection_count = selection_count + ?, updated_at = ? + WHERE id = ? AND selection_count = ? + AND selection_count + ? <= ?''', + ( + inserted_selections, now, experiment['id'], + actual_selection_count, inserted_selections, + normalized['theoretical_max_targets'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'Docker depth selection capacity lost its serialized fence' + ) + selected_image_count = int(self.conn.execute( + '''SELECT COUNT(*) AS count FROM docker_depth_experiment_selections + WHERE experiment_repository_id = ?''', + (repository_id,), + ).fetchone()['count']) + cursor = self.conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET work_state = 'resolved', resolver_owner = NULL, + resolver_token = NULL, resolver_expires_at = NULL, + resolver_due_at = NULL, + candidate_distinct_graph_count = CASE + WHEN candidate_distinct_graph_count >= ? + THEN candidate_distinct_graph_count ELSE ? END, + selected_image_count = ?, last_error_code = NULL, + resolved_at = ?, updated_at = ? + WHERE id = ? AND experiment_id = ? + AND resolver_generation = ? AND resolver_owner = ? + AND resolver_token = ? AND resolver_expires_at > ?''', + ( + candidate_count, candidate_count, selected_image_count, now, now, + repository_id, experiment['id'], resolver_generation, + resolver_owner, resolver_token, now, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('Docker depth resolver completion lost its lease fence') + + if stage == 'breadth': + self._finalize_docker_depth_query_breadth_locked( + experiment, member, now, + ) + experiment = self.conn.execute( + 'SELECT * FROM docker_depth_experiments WHERE id = ?', + (experiment['id'],), + ).fetchone() + experiment, selection_reason = ( + self._freeze_docker_depth_runtime_selection_locked( + experiment, normalized, now, + ) + ) + if selection_reason: + held = self._hold_docker_depth_experiment_locked( + experiment, selection_reason, now, + ) + self.conn.commit() + return held + drift_reason = self._docker_depth_persisted_drift_reason_locked( + experiment, normalized, now, + ) + if drift_reason: + held = self._hold_docker_depth_experiment_locked( + experiment, drift_reason, now, + ) + self.conn.commit() + return held + experiment = self._advance_docker_depth_experiment_state_locked( + experiment, now, + ) + self.conn.commit() + return { + 'status': 'resolved', 'committed': True, + 'stage': stage, 'candidate_distinct_graph_count': candidate_count, + 'selected_image_count': selected_image_count, + 'inserted_selection_count': inserted_selections, + 'experiment_state': str(experiment['state']), + } + except DiscoveryPausedError: + self.conn.rollback() + raise + except Exception as exc: + self.last_error = str(exc) + try: + self.conn.rollback() + result = self._defer_or_hold_docker_depth_resolution_conflict( + source, repository_id, resolver_generation, resolver_owner, + resolver_token, normalized, str(exc), + final_cutover=final_cutover, + ) + self.last_error = str(exc) + return result + except Exception as recovery_exc: + self.last_error = f'{exc}; conflict recovery failed: {recovery_exc}' + try: + self.conn.rollback() + except Exception: + pass + return {'status': 'error', 'committed': False} + + def _locked_docker_depth_page_authority( + self, source, query, observation, authority, *, final_cutover, now, + ): + rows = self.conn.execute( + '''SELECT * FROM docker_depth_experiments + WHERE source = ? AND state <> 'released' + ORDER BY id FOR UPDATE''', + (source,), + ).fetchall() + if not rows: + return None, None, None + if len(rows) != 1: + for row in rows: + self._hold_docker_depth_experiment_locked( + row, 'dynamic_hold_authority_ambiguous', now, + ) + return None, None, 'dynamic_hold_authority_ambiguous' + row = rows[0] + if not isinstance(authority, dict) or authority.get('enabled') is not True: + reason = 'authority_payload_invalid' + if isinstance(authority, dict) and authority.get('enabled') is False: + reason = 'experiment_disabled' + self._hold_docker_depth_experiment_locked(row, reason, now) + return None, None, reason + try: + normalized = self._normalize_docker_depth_resolver_authority(authority) + except (TypeError, ValueError): + return None, None, 'authority_payload_invalid' + if ( + str(row['experiment_key']) != normalized['experiment_key'] + or str(row['source']) != normalized['source'] + ): + self._hold_docker_depth_experiment_locked( + row, 'experiment_identity_drift', now, + ) + return None, None, 'experiment_identity_drift' + experiment, reason = self._locked_docker_depth_experiment_authority( + normalized, final_cutover=final_cutover, now=now, + ) + if not experiment: + return None, None, reason + query_row = self.conn.execute( + '''SELECT query_ordinal FROM docker_depth_experiment_queries + WHERE experiment_id = ? AND source = ? AND query = ?''', + (experiment['id'], source, query), + ).fetchone() + if str(experiment['state']) != 'collecting' and ( + observation is None + or not query_row + or observation['query_ordinal'] != int(query_row['query_ordinal']) + or observation['query_count'] != normalized['query_count'] + or observation['collection_generation'] + != normalized['collection_generation'] + or observation['ordered_query_hash'] + != normalized['ordered_queries_sha256'] + or observation['policy_sha256'] + != normalized['provenance_policy_sha256'] + ): + self._hold_docker_depth_experiment_locked( + experiment, 'dynamic_hold_observation_drift', now, + ) + return None, None, 'dynamic_hold_observation_drift' + return experiment, normalized, None + + def _hold_new_docker_depth_repositories_locked( + self, source, query, queue_ids, observation, now, *, experiment=None, + authority=None, + ): + if not queue_ids or not self.conn.is_postgres or not experiment: + return 0 + from docker_depth_experiment import ( + DOCKER_DEPTH_DYNAMIC_HOLD_REASON, + DOCKER_DEPTH_HOLD_ACTIVE_STATES, + DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, + ) + + if ( + str(experiment['state']) not in DOCKER_DEPTH_HOLD_ACTIVE_STATES + or experiment['hold_manifest_sha256'] is None + ): + return 0 + query_row = self.conn.execute( + '''SELECT query_ordinal FROM docker_depth_experiment_queries + WHERE experiment_id = ? AND source = ? AND query = ?''', + (experiment['id'], source, query), + ).fetchone() + if ( + observation is None + or authority is None + or not query_row + or observation['query_ordinal'] != int(query_row['query_ordinal']) + or observation['query_count'] != int(experiment['query_count']) + or observation['collection_generation'] + != experiment['collection_generation'] + or observation['ordered_query_hash'] != experiment['ordered_queries_sha256'] + or observation['policy_sha256'] != experiment['provenance_policy_sha256'] + or str(experiment['config_sha256']) != authority['config_sha256'] + or str(experiment['selector_sha256']) != authority['selector_sha256'] + ): + raise ScanEventConflictError('Docker depth dynamic hold observation drifted') + queue_placeholders = ','.join('?' for _ in queue_ids) + rows = self.conn.execute( + f'''SELECT queue.* FROM target_queue queue + WHERE queue.id IN ({queue_placeholders}) + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_repositories member + WHERE member.experiment_id = ? + AND member.repository_queue_id = queue.id + ) + ORDER BY queue.id FOR UPDATE''', + (*sorted(queue_ids), experiment['id']), + ).fetchall() + entries = [] + for row in rows: + if ( + row['source'] != source + or row['platform'] != 'docker' + or row['query'] != query + or row['status'] != 'deferred' + or row['target_scan_id'] is not None + or '@' in str(row['target']) + or '@' in str(row['normalized_target']) + ): + raise ScanEventConflictError('Docker depth dynamic hold row identity conflicts') + entries.append({ + 'queue_id': int(row['id']), + 'source': str(row['source']), + 'platform': str(row['platform']), + 'query': str(row['query']), + 'prior_status': str(row['status']), + 'prior_updated_at': str(row['updated_at']), + }) + entries = self._normalize_target_queue_policy_entries( + entries, 'cold', max(1, len(entries)), + ) if entries else [] + if entries: + active_holds = int(self.conn.execute( + '''SELECT COUNT(*) AS count + FROM target_queue_policy_events event + LEFT JOIN target_queue_policy_events reverse_event + ON reverse_event.reverses_event_id = event.id + WHERE event.experiment_id = ? AND event.action = 'cold' + AND reverse_event.id IS NULL''', + (experiment['id'],), + ).fetchone()['count']) + if active_holds + len(entries) > DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS: + raise ScanEventConflictError( + 'Docker depth dynamic hold capacity is exhausted' + ) + result = self._cold_target_queue_rows_locked( + entries, + reason_code=DOCKER_DEPTH_DYNAMIC_HOLD_REASON, + config_sha256=str(experiment['config_sha256']), + policy_sha256=str(experiment['provenance_policy_sha256']), + manifest_sha256=str(experiment['hold_manifest_sha256']), + experiment_id=int(experiment['id']), + now=now, + ) + return int(result['transitioned']) + int(result['duplicates']) + + @staticmethod + def _target_queue_policy_audit_sha256( + action, manifest_sha256, entry, experiment_id=None, + ): + payload = { + 'action': str(action), + 'manifest_sha256': str(manifest_sha256), + 'entry': dict(entry), + } + if experiment_id is not None: + payload['experiment_id'] = int(experiment_id) + encoded = json.dumps( + payload, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8') + return hashlib.sha256(encoded).hexdigest() + + @staticmethod + def _normalize_target_queue_policy_entries(entries, action, max_rows): + from docker_depth_experiment import DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS + + maximum = min( + DOCKER_DEPTH_REVIEW_MANIFEST_MAX_ROWS, + max(1, int(max_rows)), + ) + values = [dict(entry) for entry in entries or ()] + if not values or len(values) > maximum: + raise ValueError('target queue policy entry count is outside its bound') + required = { + 'queue_id', 'source', 'platform', 'query', 'prior_updated_at', + } + if action == 'cold': + required.add('prior_status') + else: + required.update(('cold_event_id', 'restore_status')) + normalized = [] + seen = set() + for value in values: + if set(value) != required: + raise ValueError('target queue policy entry shape is invalid') + queue_id = int(value['queue_id']) + if queue_id <= 0 or queue_id in seen: + raise ValueError('target queue policy queue identity is invalid or duplicated') + seen.add(queue_id) + item = { + 'queue_id': queue_id, + 'source': str(value['source'] or '').strip(), + 'platform': str(value['platform'] or '').strip(), + 'query': str(value['query'] or ''), + 'prior_updated_at': str(value['prior_updated_at'] or ''), + } + if not all((item['source'], item['platform'], item['query'], item['prior_updated_at'])): + raise ValueError('target queue policy entry contains an empty identity field') + if action == 'cold': + item['prior_status'] = str(value['prior_status'] or '') + if item['prior_status'] not in ('pending', 'deferred'): + raise ValueError('cold transition requires a pending or deferred prior status') + else: + item['cold_event_id'] = int(value['cold_event_id']) + item['restore_status'] = str(value['restore_status'] or '') + if item['cold_event_id'] <= 0 or item['restore_status'] not in ('pending', 'deferred'): + raise ValueError('reactivation transition identity is invalid') + normalized.append(item) + return sorted(normalized, key=lambda item: item['queue_id']) + + @staticmethod + def _target_queue_policy_chunks(values, size=500): + values = list(values) + for start in range(0, len(values), size): + yield values[start:start + size] + + def _locked_target_queue_policy_rows(self, entries): + rows_by_id = {} + for chunk in self._target_queue_policy_chunks(entries): + placeholders = ','.join('?' for _ in chunk) + rows = self.conn.execute( + f'''SELECT * FROM target_queue + WHERE id IN ({placeholders}) ORDER BY id FOR UPDATE''', + tuple(entry['queue_id'] for entry in chunk), + ).fetchall() + for row in rows: + queue_id = int(row['id']) + if queue_id in rows_by_id: + raise ScanEventConflictError( + 'target queue policy selection contains duplicate rows' + ) + rows_by_id[queue_id] = row + if len(rows_by_id) != len(entries): + raise ScanEventConflictError('target queue policy selection changed') + return [rows_by_id[entry['queue_id']] for entry in entries] + + @staticmethod + def _normalize_target_queue_policy_experiment_id(experiment_id): + if experiment_id is None: + return None + if isinstance(experiment_id, bool): + raise ValueError('target queue policy experiment identity is invalid') + try: + experiment_id = int(experiment_id) + except (TypeError, ValueError, OverflowError): + raise ValueError('target queue policy experiment identity is invalid') from None + if experiment_id <= 0: + raise ValueError('target queue policy experiment identity is invalid') + return experiment_id + + def _require_target_queue_policy_unfenced(self, row, *, lock_rows=True): + fenced_fields = ( + 'lease_owner', 'lease_token', 'claim_batch', 'leased_at', 'lease_expires_at', + 'current_result_reservation_id', 'claim_event_id', 'resolver_token', + ) + if any(row[field] is not None for field in fenced_fields): + raise ScanEventConflictError('target queue policy row has an active claim fence') + if str(row['resolver_state'] or '') == 'resolving': + raise ScanEventConflictError('target queue policy row has an active resolver fence') + lock_suffix = ' FOR UPDATE' if lock_rows else '' + reservation = self.conn.execute( + '''SELECT id FROM result_reservations + WHERE queue_id = ? AND state IN ('scanning','ready','ingesting','db_committed') + LIMIT 1''' + lock_suffix, + (int(row['id']),), + ).fetchone() + if reservation: + raise ScanEventConflictError('target queue policy row has an active result reservation') + blob_lock_suffix = ' FOR UPDATE OF blob' if lock_rows else '' + docker_blob = self.conn.execute( + '''SELECT 1 + FROM docker_image_blob_coverage coverage + JOIN docker_content_blobs blob + ON blob.digest = coverage.blob_digest + AND blob.coverage_policy_sha256 = coverage.coverage_policy_sha256 + WHERE coverage.queue_id = ? AND blob.state IN ('leased','submitted') + LIMIT 1''' + blob_lock_suffix, + (int(row['id']),), + ).fetchone() + if docker_blob: + raise ScanEventConflictError('target queue policy row has active Docker content work') + + def _target_queue_policy_duplicate_count( + self, action, manifest_sha256, entries, *, reason_code, + config_sha256, policy_sha256, experiment_id=None, + ): + entries_by_audit = {} + for entry in entries: + audit_sha256 = self._target_queue_policy_audit_sha256( + action, manifest_sha256, entry, experiment_id, + ) + entries_by_audit[audit_sha256] = entry + events_by_audit = {} + for chunk in self._target_queue_policy_chunks( + sorted(entries_by_audit), + ): + placeholders = ','.join('?' for _ in chunk) + events = self.conn.execute( + f'''SELECT * FROM target_queue_policy_events + WHERE review_audit_sha256 IN ({placeholders}) + ORDER BY id FOR UPDATE''', + tuple(chunk), + ).fetchall() + for event in events: + audit_sha256 = str(event['review_audit_sha256']) + if audit_sha256 in events_by_audit: + raise ScanEventConflictError( + 'target queue policy audit identity is duplicated' + ) + events_by_audit[audit_sha256] = event + if not events_by_audit: + return 0 + if len(events_by_audit) != len(entries): + raise ScanEventConflictError( + 'target queue policy manifest was only partially applied' + ) + reversed_event_ids = set() + if action == 'cold': + event_ids = sorted(int(event['id']) for event in events_by_audit.values()) + for chunk in self._target_queue_policy_chunks(event_ids): + placeholders = ','.join('?' for _ in chunk) + reversed_rows = self.conn.execute( + f'''SELECT reverses_event_id FROM target_queue_policy_events + WHERE reverses_event_id IN ({placeholders}) + ORDER BY reverses_event_id FOR UPDATE''', + tuple(chunk), + ).fetchall() + reversed_event_ids.update( + int(row['reverses_event_id']) for row in reversed_rows + ) + queues = { + int(row['id']): row + for row in self._locked_target_queue_policy_rows(entries) + } + for audit_sha256, entry in entries_by_audit.items(): + event = events_by_audit[audit_sha256] + expected_next = 'cold' if action == 'cold' else entry['restore_status'] + expected_prior = entry['prior_status'] if action == 'cold' else 'cold' + expected_reverse = None if action == 'cold' else entry['cold_event_id'] + if ( + event['action'] != action + or int(event['queue_id']) != entry['queue_id'] + or event['prior_status'] != expected_prior + or event['next_status'] != expected_next + or event['source'] != entry['source'] + or event['platform'] != entry['platform'] + or event['query'] != entry['query'] + or event['reason_code'] != reason_code + or event['config_sha256'] != config_sha256 + or event['policy_sha256'] != policy_sha256 + or event['manifest_sha256'] != manifest_sha256 + or event['prior_updated_at'] != entry['prior_updated_at'] + or ( + (event['experiment_id'] is None and experiment_id is not None) + or ( + event['experiment_id'] is not None + and int(event['experiment_id']) != experiment_id + ) + ) + or ( + (event['reverses_event_id'] is None and expected_reverse is not None) + or ( + event['reverses_event_id'] is not None + and int(event['reverses_event_id']) != expected_reverse + ) + ) + ): + raise ScanEventConflictError('target queue policy audit identity conflicts') + queue = queues.get(entry['queue_id']) + if not queue or queue['status'] != expected_next: + raise ScanEventConflictError('target queue policy duplicate state conflicts') + if action == 'cold' and int(event['id']) in reversed_event_ids: + raise ScanEventConflictError('target queue cold event was already reversed') + return len(entries) + + def _cold_target_queue_rows_locked( + self, entries, *, reason_code, config_sha256, policy_sha256, + manifest_sha256, experiment_id=None, now=None, + ): + experiment_id = self._normalize_target_queue_policy_experiment_id(experiment_id) + now = str(now or utc_now_iso()) + if not entries: + return { + 'transitioned': 0, 'duplicates': 0, 'examined': 0, + 'manifest_sha256': manifest_sha256, + } + experiment = None + if experiment_id is not None: + experiment = self.conn.execute( + '''SELECT id FROM docker_depth_experiments + WHERE id = ? FOR UPDATE''', + (experiment_id,), + ).fetchone() + if not experiment: + raise ScanEventConflictError( + 'target queue policy experiment authority is absent' + ) + rows = self._locked_target_queue_policy_rows(entries) + duplicates = self._target_queue_policy_duplicate_count( + 'cold', manifest_sha256, entries, reason_code=reason_code, + config_sha256=config_sha256, policy_sha256=policy_sha256, + experiment_id=experiment_id, + ) + if duplicates: + return { + 'transitioned': 0, 'duplicates': duplicates, + 'examined': len(entries), 'manifest_sha256': manifest_sha256, + } + for row, entry in zip(rows, entries): + if ( + int(row['id']) != entry['queue_id'] + or row['source'] != entry['source'] + or row['platform'] != entry['platform'] + or row['query'] != entry['query'] + or row['status'] != entry['prior_status'] + or row['updated_at'] != entry['prior_updated_at'] + ): + raise ScanEventConflictError('target queue policy evidence does not match') + self._require_target_queue_policy_unfenced(row) + for row, entry in zip(rows, entries): + audit_sha256 = self._target_queue_policy_audit_sha256( + 'cold', manifest_sha256, entry, experiment_id, + ) + self.conn.execute( + '''INSERT INTO target_queue_policy_events( + queue_id, action, prior_status, next_status, source, platform, + query, reason_code, config_sha256, policy_sha256, + manifest_sha256, review_audit_sha256, reverses_event_id, + experiment_id, prior_updated_at, created_at + ) VALUES (?, 'cold', ?, 'cold', ?, ?, ?, ?, ?, ?, ?, ?, NULL, ?, ?, ?)''', + ( + entry['queue_id'], entry['prior_status'], entry['source'], + entry['platform'], entry['query'], reason_code, config_sha256, + policy_sha256, manifest_sha256, audit_sha256, experiment_id, + entry['prior_updated_at'], now, + ), + ) + cursor = self.conn.execute( + '''UPDATE target_queue SET status = 'cold', updated_at = ? + WHERE id = ? AND status = ? AND updated_at = ?''', + (now, entry['queue_id'], entry['prior_status'], entry['prior_updated_at']), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('target queue cold transition lost its fence') + return { + 'transitioned': len(entries), 'duplicates': 0, + 'examined': len(entries), 'manifest_sha256': manifest_sha256, + } + + def cold_target_queue_rows( + self, entries, *, reason_code, config_sha256, policy_sha256, + manifest_sha256, max_rows=10000, experiment_id=None, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('target queue cold transition requires PostgreSQL') + entries = self._normalize_target_queue_policy_entries(entries, 'cold', max_rows) + hashes = (config_sha256, policy_sha256, manifest_sha256) + if not all(re.fullmatch(r'[a-f0-9]{64}', str(value or '')) for value in hashes): + raise ValueError('target queue policy hash identity is invalid') + reason_code = str(reason_code or '').strip() + if not reason_code or len(reason_code) > 128: + raise ValueError('target queue policy reason code is invalid') + experiment_id = self._normalize_target_queue_policy_experiment_id(experiment_id) + try: + result = self._cold_target_queue_rows_locked( + entries, reason_code=reason_code, config_sha256=config_sha256, + policy_sha256=policy_sha256, manifest_sha256=manifest_sha256, + experiment_id=experiment_id, now=utc_now_iso(), + ) + self.conn.commit() + return result + except Exception: + self.conn.rollback() + raise + + def _reactivate_target_queue_rows_locked( + self, entries, *, reason_code, config_sha256, policy_sha256, + manifest_sha256, experiment_id=None, now=None, + ): + from docker_depth_experiment import ( + DOCKER_DEPTH_DYNAMIC_HOLD_REASON, + DOCKER_DEPTH_HOLD_REASON, + ) + + experiment_id = self._normalize_target_queue_policy_experiment_id(experiment_id) + now = str(now or utc_now_iso()) + if not entries: + return { + 'transitioned': 0, 'duplicates': 0, 'examined': 0, + 'manifest_sha256': manifest_sha256, + } + experiment = None + if experiment_id is not None: + experiment = self.conn.execute( + '''SELECT * FROM docker_depth_experiments + WHERE id = ? FOR UPDATE''', + (experiment_id,), + ).fetchone() + if not experiment: + raise ScanEventConflictError( + 'target queue policy experiment authority is absent' + ) + rows = self._locked_target_queue_policy_rows(entries) + duplicates = self._target_queue_policy_duplicate_count( + 'reactivate', manifest_sha256, entries, reason_code=reason_code, + config_sha256=config_sha256, policy_sha256=policy_sha256, + experiment_id=experiment_id, + ) + if duplicates: + return { + 'transitioned': 0, 'duplicates': duplicates, + 'examined': len(entries), 'manifest_sha256': manifest_sha256, + } + cold_events = {} + cold_event_ids = sorted(entry['cold_event_id'] for entry in entries) + for chunk in self._target_queue_policy_chunks(cold_event_ids): + placeholders = ','.join('?' for _ in chunk) + event_rows = self.conn.execute( + f'''SELECT * FROM target_queue_policy_events + WHERE id IN ({placeholders}) ORDER BY id FOR UPDATE''', + tuple(chunk), + ).fetchall() + cold_events.update((int(event['id']), event) for event in event_rows) + reversed_event_ids = set() + for chunk in self._target_queue_policy_chunks(cold_event_ids): + placeholders = ','.join('?' for _ in chunk) + reversed_rows = self.conn.execute( + f'''SELECT reverses_event_id FROM target_queue_policy_events + WHERE reverses_event_id IN ({placeholders}) + ORDER BY reverses_event_id FOR UPDATE''', + tuple(chunk), + ).fetchall() + reversed_event_ids.update( + int(event['reverses_event_id']) for event in reversed_rows + ) + for row, entry in zip(rows, entries): + cold_event = cold_events.get(entry['cold_event_id']) + cold_experiment_id = ( + None if not cold_event or cold_event['experiment_id'] is None + else int(cold_event['experiment_id']) + ) + cold_entry = { + 'queue_id': entry['queue_id'], + 'source': entry['source'], + 'platform': entry['platform'], + 'query': entry['query'], + 'prior_status': entry['restore_status'], + 'prior_updated_at': ( + str(cold_event['prior_updated_at']) if cold_event else '' + ), + } + owned_event_invalid = bool(experiment_id is not None and cold_event) and ( + str(cold_event['reason_code']) not in ( + DOCKER_DEPTH_HOLD_REASON, DOCKER_DEPTH_DYNAMIC_HOLD_REASON, + ) + or str(cold_event['config_sha256']) != str(config_sha256) + or str(cold_event['policy_sha256']) != str(policy_sha256) + or str(cold_event['manifest_sha256']) + != str(experiment['hold_manifest_sha256']) + or str(cold_event['review_audit_sha256']) + != self._target_queue_policy_audit_sha256( + 'cold', cold_event['manifest_sha256'], cold_entry, + experiment_id, + ) + ) + if ( + int(row['id']) != entry['queue_id'] + or row['source'] != entry['source'] + or row['platform'] != entry['platform'] + or row['query'] != entry['query'] + or row['status'] != 'cold' + or row['updated_at'] != entry['prior_updated_at'] + or not cold_event + or cold_event['action'] != 'cold' + or int(cold_event['queue_id']) != entry['queue_id'] + or cold_event['prior_status'] != entry['restore_status'] + or cold_event['next_status'] != 'cold' + or cold_experiment_id != experiment_id + or owned_event_invalid + or entry['cold_event_id'] in reversed_event_ids + ): + raise ScanEventConflictError('target queue reactivation evidence does not match') + self._require_target_queue_policy_unfenced(row) + for entry in entries: + audit_sha256 = self._target_queue_policy_audit_sha256( + 'reactivate', manifest_sha256, entry, experiment_id, + ) + self.conn.execute( + '''INSERT INTO target_queue_policy_events( + queue_id, action, prior_status, next_status, source, platform, + query, reason_code, config_sha256, policy_sha256, + manifest_sha256, review_audit_sha256, reverses_event_id, + experiment_id, prior_updated_at, created_at + ) VALUES (?, 'reactivate', 'cold', ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + entry['queue_id'], entry['restore_status'], entry['source'], + entry['platform'], entry['query'], reason_code, config_sha256, + policy_sha256, manifest_sha256, audit_sha256, + entry['cold_event_id'], experiment_id, entry['prior_updated_at'], now, + ), + ) + cursor = self.conn.execute( + '''UPDATE target_queue SET status = ?, updated_at = ? + WHERE id = ? AND status = 'cold' AND updated_at = ?''', + ( + entry['restore_status'], now, entry['queue_id'], + entry['prior_updated_at'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError('target queue reactivation lost its fence') + return { + 'transitioned': len(entries), 'duplicates': 0, + 'examined': len(entries), 'manifest_sha256': manifest_sha256, + } + + def reactivate_cold_target_queue_rows( + self, entries, *, reason_code, config_sha256, policy_sha256, + manifest_sha256, max_rows=10000, experiment_id=None, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('target queue reactivation requires PostgreSQL') + entries = self._normalize_target_queue_policy_entries(entries, 'reactivate', max_rows) + hashes = (config_sha256, policy_sha256, manifest_sha256) + if not all(re.fullmatch(r'[a-f0-9]{64}', str(value or '')) for value in hashes): + raise ValueError('target queue policy hash identity is invalid') + reason_code = str(reason_code or '').strip() + if not reason_code or len(reason_code) > 128: + raise ValueError('target queue policy reason code is invalid') + experiment_id = self._normalize_target_queue_policy_experiment_id(experiment_id) + try: + result = self._reactivate_target_queue_rows_locked( + entries, reason_code=reason_code, config_sha256=config_sha256, + policy_sha256=policy_sha256, manifest_sha256=manifest_sha256, + experiment_id=experiment_id, now=utc_now_iso(), + ) + self.conn.commit() + return result + except Exception: + self.conn.rollback() + raise + + @staticmethod + def _retirement_chain(previous, rows): + previous = str(previous or '0' * 64) + payload = json.dumps( + [dict(row) for row in rows], + ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8') + return hashlib.sha256(bytes.fromhex(previous) + hashlib.sha256(payload).digest()).hexdigest() + + def retire_admission_intents(self, retention_seconds=30 * 86400, limit=100): + if not self.conn or not self.conn.is_postgres: + return 0 + now = utc_now_iso() + cutoff = datetime.fromtimestamp( + time.time() - max(3600, int(retention_seconds)), timezone.utc, + ).isoformat(timespec='seconds') + limit = min(500, max(1, int(limit))) + try: + self.conn.execute(f'SET LOCAL statement_timeout = {RETIREMENT_STATEMENT_TIMEOUT_MS}') + self.conn.execute( + '''INSERT INTO admission_intent_retirement( + id, retired_count, chain_sha256, cursor_token, updated_at + ) VALUES (1, 0, ?, '', ?) ON CONFLICT(id) DO NOTHING''', + ('0' * 64, now), + ) + summary = self.conn.execute( + 'SELECT * FROM admission_intent_retirement WHERE id = 1 FOR UPDATE' + ).fetchone() + rows = self.conn.execute( + '''SELECT i.reservation_token, i.intent_sha256, i.state, + i.resolution_detail, i.created_at, i.updated_at, i.resolved_at + FROM admission_intents i + WHERE i.state = 'aborted' AND i.reservation_id IS NULL + AND i.resolved_at IS NOT NULL AND i.resolved_at <= ? + AND i.reservation_token > ? + AND NOT EXISTS ( + SELECT 1 FROM result_reservations r + WHERE r.reservation_token = i.reservation_token + ) + ORDER BY i.reservation_token LIMIT ? FOR UPDATE SKIP LOCKED''', + (cutoff, summary['cursor_token'], limit), + ).fetchall() + if not rows: + if summary['cursor_token']: + self.conn.execute( + "UPDATE admission_intent_retirement SET cursor_token = '', updated_at = ? WHERE id = 1", + (now,), + ) + self.conn.commit() + return 0 + tokens = [row['reservation_token'] for row in rows] + placeholders = ','.join('?' for _ in tokens) + deleted = self.conn.execute( + f'''DELETE FROM admission_intents i + WHERE i.reservation_token IN ({placeholders}) + AND i.state = 'aborted' AND i.reservation_id IS NULL + AND NOT EXISTS ( + SELECT 1 FROM result_reservations r + WHERE r.reservation_token = i.reservation_token + )''', + tokens, + ) + count = int(deleted.rowcount or 0) + if count != len(rows): + raise RuntimeError('admission intent retirement lost its exact row fence') + chain = self._retirement_chain(summary['chain_sha256'], rows) + self.conn.execute( + '''UPDATE admission_intent_retirement SET retired_count = retired_count + ?, + chain_sha256 = ?, cursor_token = ?, updated_at = ? WHERE id = 1''', + (count, chain, tokens[-1], now), + ) + self.conn.commit() + return count + except Exception: + self.conn.rollback() + raise + + def retire_deleted_pipeline_artifacts(self, retention_seconds=30 * 86400, limit=100): + if not self.conn or not self.conn.is_postgres: + return 0 + now = utc_now_iso() + cutoff = datetime.fromtimestamp( + time.time() - max(3600, int(retention_seconds)), timezone.utc, + ).isoformat(timespec='seconds') + limit = min(500, max(1, int(limit))) + try: + self.conn.execute(f'SET LOCAL statement_timeout = {RETIREMENT_STATEMENT_TIMEOUT_MS}') + self.conn.execute( + '''INSERT INTO pipeline_artifact_retirement( + id, retired_count, chain_sha256, cursor_id, updated_at + ) VALUES (1, 0, ?, 0, ?) ON CONFLICT(id) DO NOTHING''', + ('0' * 64, now), + ) + summary = self.conn.execute( + 'SELECT * FROM pipeline_artifact_retirement WHERE id = 1 FOR UPDATE' + ).fetchone() + rows = self.conn.execute( + '''SELECT a.id, a.subsystem, a.artifact_kind, a.owner_id, a.owner_key, + a.relative_path, a.payload_sha256, a.created_at, a.updated_at, a.deleted_at + FROM pipeline_artifacts a + WHERE a.state = 'deleted' AND a.deleted_at IS NOT NULL + AND a.deleted_at <= ? AND a.id > ? + AND NOT EXISTS ( + SELECT 1 FROM pipeline_quarantine q + WHERE q.review_status = 'pending' AND ( + (q.object_type = 'projection_tail' AND q.object_id = a.id) + OR ( + q.object_type = 'result_bundle' + AND a.subsystem = 'result_bundle' + AND a.artifact_kind = 'bundle_quarantine' + AND q.reservation_id = a.owner_id + ) + ) + ) + ORDER BY a.id LIMIT ? FOR UPDATE SKIP LOCKED''', + (cutoff, int(summary['cursor_id'] or 0), limit), + ).fetchall() + if not rows: + if int(summary['cursor_id'] or 0): + self.conn.execute( + 'UPDATE pipeline_artifact_retirement SET cursor_id = 0, updated_at = ? WHERE id = 1', + (now,), + ) + self.conn.commit() + return 0 + ids = [int(row['id']) for row in rows] + placeholders = ','.join('?' for _ in ids) + deleted = self.conn.execute( + f'''DELETE FROM pipeline_artifacts a + WHERE a.id IN ({placeholders}) AND a.state = 'deleted' + AND NOT EXISTS ( + SELECT 1 FROM pipeline_quarantine q + WHERE q.review_status = 'pending' AND ( + (q.object_type = 'projection_tail' AND q.object_id = a.id) + OR ( + q.object_type = 'result_bundle' + AND a.subsystem = 'result_bundle' + AND a.artifact_kind = 'bundle_quarantine' + AND q.reservation_id = a.owner_id + ) + ) + )''', + ids, + ) + count = int(deleted.rowcount or 0) + if count != len(rows): + raise RuntimeError('pipeline artifact retirement lost its exact row fence') + chain = self._retirement_chain(summary['chain_sha256'], rows) + self.conn.execute( + '''UPDATE pipeline_artifact_retirement SET retired_count = retired_count + ?, + chain_sha256 = ?, cursor_id = ?, updated_at = ? WHERE id = 1''', + (count, chain, ids[-1], now), + ) + self.conn.commit() + return count + except Exception: + self.conn.rollback() + raise + + def janitor_cursor(self, layout_name): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('janitor cursor authority requires PostgreSQL') + row = self.conn.execute( + 'SELECT layout_name, last_name, wrap_count FROM janitor_cursors WHERE layout_name = ?', + (str(layout_name),), + ).fetchone() + self.conn.commit() + return dict(row) if row else { + 'layout_name': str(layout_name), 'last_name': '', 'wrap_count': 0, + } + + def advance_janitor_cursor(self, layout_name, last_name, wrapped=False): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('janitor cursor authority requires PostgreSQL') + now = utc_now_iso() + self.conn.execute( + '''INSERT INTO janitor_cursors(layout_name, last_name, wrap_count, updated_at) + VALUES (?, ?, ?, ?) + ON CONFLICT(layout_name) DO UPDATE SET + last_name = excluded.last_name, + wrap_count = janitor_cursors.wrap_count + excluded.wrap_count, + updated_at = excluded.updated_at''', + (str(layout_name), str(last_name), 1 if wrapped else 0, now), + ) + self.conn.commit() + return True + + def claim_projection_job(self, generation, lease_token, lease_seconds=120): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection job claims require PostgreSQL') + now = utc_now_iso() + expires = datetime.fromtimestamp( + time.time() + max(10, int(lease_seconds)), timezone.utc, + ).isoformat(timespec='seconds') + token = secrets.token_urlsafe(32) + try: + owner = self.conn.execute( + '''SELECT 1 AS valid FROM pipeline_leases + WHERE worker_name = 'jsonl_projector' AND generation = ? + AND lease_token = ? AND state = 'ready' AND lease_expires_at > ? FOR SHARE''', + (int(generation), str(lease_token), now), + ).fetchone() + if not owner: + raise RuntimeError('JSONL projector singleton fence is not valid') + row = self.conn.execute( + '''SELECT id FROM projection_jobs + WHERE (status = 'pending' AND (available_after IS NULL OR available_after <= ?)) + OR (status = 'leased' AND ( + lease_generation IS NULL OR lease_generation <> ? + OR (lease_expires_at IS NOT NULL AND lease_expires_at <= ?) + )) + ORDER BY id LIMIT 1 FOR UPDATE SKIP LOCKED''', + (now, int(generation), now), + ).fetchone() + if not row: + self.conn.rollback() + return None + self.conn.execute( + '''UPDATE projection_jobs SET status = 'leased', attempts = attempts + 1, + lease_generation = ?, lease_token = ?, lease_expires_at = ?, updated_at = ? + WHERE id = ?''', + (int(generation), token, expires, now, row['id']), + ) + claimed = self.conn.execute( + 'SELECT * FROM projection_jobs WHERE id = ?', (row['id'],), + ).fetchone() + self.conn.commit() + return dict(claimed) + except Exception: + self.conn.rollback() + raise + + def expand_projection_job_capacity( + self, job_id, job_lease_token, actual_bytes, projection_max_bytes, + defer_seconds=1, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection capacity expansion requires PostgreSQL') + actual_bytes = max(0, int(actual_bytes)) + projection_max_bytes = max(0, int(projection_max_bytes)) + now = utc_now_iso() + available_after = datetime.fromtimestamp( + time.time() + max(1, int(defer_seconds)), timezone.utc, + ).isoformat(timespec='seconds') + try: + job = self.conn.execute( + '''SELECT * FROM projection_jobs WHERE id = ? AND status = 'leased' + AND lease_token = ? FOR UPDATE''', + (int(job_id), str(job_lease_token)), + ).fetchone() + if not job: + self.conn.rollback() + return None + if actual_bytes <= int(job['capacity_bytes']): + self.conn.commit() + return dict(job) + if self.conn.execute( + 'SELECT 1 AS present FROM projection_appends WHERE job_id = ? LIMIT 1', + (int(job_id),), + ).fetchone(): + raise RuntimeError( + 'projection capacity cannot expand after append preparation' + ) + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + additional_bytes = actual_bytes - int(job['capacity_bytes']) + if ( + not capacity + or actual_bytes > projection_max_bytes + or int(capacity['projection_bytes']) + additional_bytes + > projection_max_bytes + ): + cursor = self.conn.execute( + '''UPDATE projection_jobs SET status = 'pending', + available_after = ?, lease_generation = NULL, + lease_token = NULL, lease_expires_at = NULL, + last_error_code = 'projection_capacity_backpressure', + last_error_detail = NULL, updated_at = ? + WHERE id = ? AND status = 'leased' AND lease_token = ?''', + (available_after, now, job_id, job_lease_token), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError( + 'projection capacity deferral lost its lease fence' + ) + self.conn.commit() + return False + cursor = self.conn.execute( + '''UPDATE projection_jobs SET capacity_bytes = ?, + last_error_code = NULL, last_error_detail = NULL, + updated_at = ? + WHERE id = ? AND status = 'leased' AND lease_token = ? + AND capacity_bytes = ?''', + ( + actual_bytes, now, job_id, job_lease_token, + job['capacity_bytes'], + ), + ) + if int(cursor.rowcount or 0) != 1: + raise RuntimeError( + 'projection capacity expansion lost its lease fence' + ) + self.conn.execute( + '''UPDATE pipeline_capacity + SET projection_bytes = projection_bytes + ?, updated_at = ? + WHERE id = 1''', + (additional_bytes, now), + ) + job = dict(job) + job['capacity_bytes'] = actual_bytes + job['updated_at'] = now + self.conn.commit() + return job + except Exception: + self.conn.rollback() + raise + + def projection_stream_state(self, stream_name): + if not self.conn: + raise RuntimeError('database connection is unavailable') + row = self.conn.execute( + '''SELECT s.*, c.generation, c.committed_offset, c.last_append_id, + c.last_job_id, c.last_event_id, c.last_event_hash + FROM projection_streams s + JOIN projection_cursors c ON c.stream_name = s.stream_name + WHERE s.stream_name = ?''', + (str(stream_name),), + ).fetchone() + if self.conn.is_postgres: + self.conn.commit() + return dict(row) if row else None + + def projection_append_for_job(self, job_id, stream_name): + if not self.conn: + return None + row = self.conn.execute( + 'SELECT * FROM projection_appends WHERE job_id = ? AND stream_name = ?', + (int(job_id), str(stream_name)), + ).fetchone() + if self.conn.is_postgres: + self.conn.commit() + return dict(row) if row else None + + def initialize_projection_stream_offset(self, stream_name, exact_file_size): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection stream offset initialization requires PostgreSQL') + exact_file_size = max(0, int(exact_file_size)) + now = utc_now_iso() + try: + cursor = self.conn.execute( + 'SELECT * FROM projection_cursors WHERE stream_name = ? FOR UPDATE', + (str(stream_name),), + ).fetchone() + if not cursor: + raise ValueError('projection stream cursor is absent') + if int(cursor['committed_offset']) == exact_file_size: + self.conn.commit() + return self.projection_stream_state(stream_name) + append_count = self.conn.execute( + 'SELECT COUNT(*) AS count FROM projection_appends WHERE stream_name = ?', + (str(stream_name),), + ).fetchone() + if ( + int(cursor['committed_offset']) != 0 + or cursor['last_append_id'] is not None + or cursor['last_job_id'] is not None + or int(append_count['count'] or 0) != 0 + ): + raise RuntimeError('projection stream/file mismatch has append history') + updated = self.conn.execute( + '''UPDATE projection_cursors SET committed_offset = ?, updated_at = ? + WHERE stream_name = ? AND committed_offset = 0 + AND last_append_id IS NULL AND last_job_id IS NULL''', + (exact_file_size, now, str(stream_name)), + ) + if int(updated.rowcount or 0) != 1: + raise RuntimeError('projection stream offset initialization lost its fence') + self.conn.commit() + return self.projection_stream_state(stream_name) + except Exception: + self.conn.rollback() + raise + + def prepare_projection_append( + self, job_id, job_lease_token, stream_name, generation, + byte_length, payload_sha256, record_count, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection append preparation requires PostgreSQL') + now = utc_now_iso() + try: + job = self.conn.execute( + '''SELECT * FROM projection_jobs WHERE id = ? AND status = 'leased' + AND lease_token = ? FOR UPDATE''', + (int(job_id), str(job_lease_token)), + ).fetchone() + if not job: + self.conn.rollback() + return None + existing = self.conn.execute( + '''SELECT * FROM projection_appends WHERE job_id = ? AND stream_name = ? FOR UPDATE''', + (int(job_id), str(stream_name)), + ).fetchone() + if existing: + self.conn.commit() + return dict(existing) + cursor = self.conn.execute( + 'SELECT * FROM projection_cursors WHERE stream_name = ? FOR UPDATE', + (str(stream_name),), + ).fetchone() + if not cursor or int(cursor['generation']) != int(generation): + raise RuntimeError('projection stream generation changed before append preparation') + append_id = self.conn.insert_returning_id( + '''INSERT INTO projection_appends( + job_id, stream_name, event_id, event_hash, generation, + byte_offset, byte_length, payload_sha256, record_count, + state, prepared_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'prepared', ?)''', + ( + job_id, stream_name, job['event_id'], job['event_hash'], generation, + cursor['committed_offset'], max(0, int(byte_length)), + str(payload_sha256), max(0, int(record_count)), now, + ), + ) + row = self.conn.execute( + 'SELECT * FROM projection_appends WHERE id = ?', (append_id,), + ).fetchone() + self.conn.commit() + return dict(row) + except Exception: + self.conn.rollback() + raise + + def complete_projection_append(self, append_id, job_id, job_lease_token): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection append completion requires PostgreSQL') + now = utc_now_iso() + try: + job = self.conn.execute( + '''SELECT id FROM projection_jobs WHERE id = ? AND status = 'leased' + AND lease_token = ? FOR UPDATE''', + (int(job_id), str(job_lease_token)), + ).fetchone() + append = self.conn.execute( + 'SELECT * FROM projection_appends WHERE id = ? AND job_id = ? FOR UPDATE', + (int(append_id), int(job_id)), + ).fetchone() + if not job or not append: + self.conn.rollback() + return False + expected_offset = int(append['byte_offset']) + int(append['byte_length']) + cursor = self.conn.execute( + 'SELECT * FROM projection_cursors WHERE stream_name = ? FOR UPDATE', + (append['stream_name'],), + ).fetchone() + if append['state'] == 'appended': + valid = bool( + cursor and int(cursor['generation']) == int(append['generation']) + and int(cursor['committed_offset']) >= expected_offset + ) + self.conn.commit() + return valid + if not cursor or int(cursor['generation']) != int(append['generation']) or int(cursor['committed_offset']) != int(append['byte_offset']): + raise RuntimeError('projection cursor no longer matches the prepared append') + self.conn.execute( + "UPDATE projection_appends SET state = 'appended', appended_at = ? WHERE id = ?", + (now, append_id), + ) + self.conn.execute( + '''UPDATE projection_cursors SET committed_offset = ?, last_append_id = ?, + last_job_id = ?, last_event_id = ?, last_event_hash = ?, updated_at = ? + WHERE stream_name = ?''', + ( + expected_offset, append_id, job_id, append['event_id'], + append['event_hash'], now, append['stream_name'], + ), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + def complete_projection_job(self, job_id, job_lease_token): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection job completion requires PostgreSQL') + now = utc_now_iso() + try: + job = self.conn.execute( + '''SELECT * FROM projection_jobs WHERE id = ? AND status = 'leased' + AND lease_token = ? FOR UPDATE''', + (int(job_id), str(job_lease_token)), + ).fetchone() + if not job: + self.conn.rollback() + return False + stream_count = sum( + int(bool(int(job['required_stream_mask']) & mask)) + for mask in (1, 2, 4, 8, 16) + ) + appended = self.conn.execute( + "SELECT stream_name FROM projection_appends WHERE job_id = ? AND state = 'appended'", + (int(job_id),), + ).fetchall() + appended_count = sum( + int(bool( + (row['stream_name'] == 'scan_results' and int(job['required_stream_mask']) & 1) + or (row['stream_name'] == 'found_secrets' and int(job['required_stream_mask']) & 2) + or (row['stream_name'] == 'scan_errors' and int(job['required_stream_mask']) & 4) + or (str(row['stream_name']).startswith('keycheck:') + and str(row['stream_name']).endswith(':results') + and int(job['required_stream_mask']) & 8) + or (str(row['stream_name']).startswith('keycheck:') + and str(row['stream_name']).endswith(':status') + and int(job['required_stream_mask']) & 16) + )) + for row in appended + ) + if appended_count != stream_count: + raise RuntimeError('projection job does not have every required appended stream') + if not job['capacity_released']: + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + if ( + int(capacity['projection_items']) < int(job['capacity_items']) + or int(capacity['projection_bytes']) < int(job['capacity_bytes']) + ): + raise RuntimeError('projection completion would make capacity accounting negative') + self.conn.execute( + '''UPDATE pipeline_capacity SET projection_items = projection_items - ?, + projection_bytes = projection_bytes - ?, updated_at = ? WHERE id = 1''', + (job['capacity_items'], job['capacity_bytes'], now), + ) + self.conn.execute( + '''UPDATE projection_jobs SET status = 'completed', capacity_released = 1, + lease_token = NULL, lease_expires_at = NULL, completed_at = ?, updated_at = ? + WHERE id = ?''', + (now, now, job_id), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + def quarantine_projection_job( + self, job_id, job_lease_token, reason_code, detail='', + quarantine_max_items=10000, quarantine_max_bytes=1024 * 1024 * 1024, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection quarantine requires PostgreSQL') + now = utc_now_iso() + try: + job = self.conn.execute( + '''SELECT * FROM projection_jobs WHERE id = ? AND status = 'leased' + AND lease_token = ? FOR UPDATE''', + (int(job_id), str(job_lease_token)), + ).fetchone() + if not job: + self.conn.rollback() + return False + prepared = self.conn.execute( + '''SELECT * FROM projection_appends + WHERE job_id = ? AND state = 'prepared' ORDER BY id FOR UPDATE''', + (int(job_id),), + ).fetchall() + for append in prepared: + cursor = self.conn.execute( + 'SELECT generation, committed_offset FROM projection_cursors WHERE stream_name = ? FOR UPDATE', + (append['stream_name'],), + ).fetchone() + if not cursor or ( + int(cursor['generation']) != int(append['generation']) + or int(cursor['committed_offset']) != int(append['byte_offset']) + ): + raise RuntimeError('prepared projection append cannot be safely canceled') + self.conn.execute( + '''INSERT INTO projection_append_audit( + append_id, job_id, stream_name, event_id, event_hash, generation, + byte_offset, byte_length, payload_sha256, state, + reason_code, reason_detail, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'quarantined', ?, ?, ?)''', + ( + append['id'], append['job_id'], append['stream_name'], + append['event_id'], append['event_hash'], append['generation'], + append['byte_offset'], append['byte_length'], append['payload_sha256'], + str(reason_code), first_line(detail, 2000), now, + ), + ) + self.conn.execute('DELETE FROM projection_appends WHERE id = ?', (append['id'],)) + existing = self.conn.execute( + '''SELECT id FROM pipeline_quarantine + WHERE subsystem = 'jsonl_projector' AND object_type = 'projection_job' + AND object_id = ? AND review_status = 'pending' ''', + (job_id,), + ).fetchone() + if not existing: + self.conn.insert_returning_id( + '''INSERT INTO pipeline_quarantine( + subsystem, object_type, object_id, projection_job_id, + event_id, payload_sha256, reason_code, reason_detail, + byte_count, capacity_items, capacity_bytes, detected_at + ) VALUES ('jsonl_projector','projection_job',?,?,?,?,?,?,?,?,?,?)''', + ( + job_id, job_id, job['event_id'], job['event_hash'], str(reason_code), + first_line(detail, 2000), job['capacity_bytes'], + job['capacity_items'], job['capacity_bytes'], now, + ), + ) + if not job['capacity_released']: + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + if ( + int(capacity['projection_items']) < int(job['capacity_items']) + or int(capacity['projection_bytes']) < int(job['capacity_bytes']) + ): + raise RuntimeError('projection quarantine would make capacity accounting negative') + self.conn.execute( + '''UPDATE pipeline_capacity SET projection_items = projection_items - ?, + projection_bytes = projection_bytes - ?, + quarantine_items = quarantine_items + ?, + quarantine_bytes = quarantine_bytes + ?, updated_at = ? WHERE id = 1''', + ( + job['capacity_items'], job['capacity_bytes'], job['capacity_items'], + job['capacity_bytes'], now, + ), + ) + self.conn.execute( + '''UPDATE projection_jobs SET status = 'quarantined', capacity_released = 1, + last_error_code = ?, last_error_detail = ?, lease_token = NULL, + lease_expires_at = NULL, completed_at = ?, updated_at = ? WHERE id = ?''', + (str(reason_code), first_line(detail, 2000), now, now, job_id), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + def prepare_projection_rotation(self, stream_name, source_bytes, segment_relative_path): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection rotation requires PostgreSQL') + now = utc_now_iso() + try: + stream = self.conn.execute( + 'SELECT * FROM projection_streams WHERE stream_name = ? FOR UPDATE', + (str(stream_name),), + ).fetchone() + cursor = self.conn.execute( + 'SELECT * FROM projection_cursors WHERE stream_name = ? FOR UPDATE', + (str(stream_name),), + ).fetchone() + if not stream or not cursor or int(cursor['committed_offset']) != int(source_bytes): + raise RuntimeError('projection rotation source size does not match its cursor') + to_generation = int(stream['current_generation']) + 1 + rotation_id = self.conn.insert_returning_id( + '''INSERT INTO projection_rotations( + stream_name, from_generation, to_generation, source_bytes, + segment_relative_path, state, created_at + ) VALUES (?, ?, ?, ?, ?, 'prepared', ?) + ON CONFLICT(stream_name, to_generation) DO NOTHING''', + ( + stream_name, stream['current_generation'], to_generation, + int(source_bytes), str(segment_relative_path), now, + ), + ) + if rotation_id is None: + existing = self.conn.execute( + 'SELECT * FROM projection_rotations WHERE stream_name = ? AND to_generation = ?', + (stream_name, to_generation), + ).fetchone() + self.conn.commit() + return dict(existing) + row = self.conn.execute( + 'SELECT * FROM projection_rotations WHERE id = ?', (rotation_id,), + ).fetchone() + self.conn.commit() + return dict(row) + except Exception: + self.conn.rollback() + raise + + def complete_projection_rotation(self, rotation_id): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('projection rotation completion requires PostgreSQL') + now = utc_now_iso() + try: + rotation = self.conn.execute( + 'SELECT * FROM projection_rotations WHERE id = ? FOR UPDATE', + (int(rotation_id),), + ).fetchone() + if not rotation: + self.conn.rollback() + return False + self.conn.execute( + '''UPDATE projection_streams SET current_generation = ?, updated_at = ? + WHERE stream_name = ? AND current_generation = ?''', + ( + rotation['to_generation'], now, rotation['stream_name'], + rotation['from_generation'], + ), + ) + self.conn.execute( + '''UPDATE projection_cursors SET generation = ?, committed_offset = 0, + last_append_id = NULL, updated_at = ? WHERE stream_name = ?''', + (rotation['to_generation'], now, rotation['stream_name']), + ) + self.conn.execute( + "UPDATE projection_rotations SET state = 'completed', completed_at = ? WHERE id = ?", + (now, rotation_id), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + def pending_projection_rotations(self, limit=100): + if not self.conn: + return [] + rows = self.conn.execute( + '''SELECT r.*, s.base_relative_path + FROM projection_rotations r + JOIN projection_streams s ON s.stream_name = r.stream_name + WHERE r.state IN ('prepared','renamed') ORDER BY r.id LIMIT ?''', + (min(1000, max(1, int(limit))),), + ).fetchall() + if self.conn.is_postgres: + self.conn.commit() + return [dict(row) for row in rows] + + def claim_keycheck_candidate( + self, service, lease_owner, lease_seconds=300, + result_projection_reserve_bytes=KEYCHECK_RESULT_PROJECTION_RESERVE_BYTES, + projection_max_items=10000, + projection_max_bytes=2 * 1024 * 1024 * 1024, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('normal keycheck candidate claims require PostgreSQL') + service = str(service or '').lower() + if not service or not lease_owner: + raise ValueError('keycheck service and lease owner are required') + now = utc_now_iso() + expires = datetime.fromtimestamp( + time.time() + max(30, int(lease_seconds)), timezone.utc, + ).isoformat(timespec='seconds') + token = secrets.token_urlsafe(32) + try: + row = self.conn.execute( + '''SELECT id, result_projection_reserved_bytes, + result_projection_credit_transferred + FROM keycheck_candidates + WHERE service = ? AND ( + state = 'pending' + OR (state = 'deferred' AND available_after IS NOT NULL AND available_after <= ?) + OR (state = 'leased' AND lease_expires_at IS NOT NULL AND lease_expires_at <= ?) + ) + ORDER BY priority DESC, id LIMIT 1 FOR UPDATE SKIP LOCKED''', + (service, now, now), + ).fetchone() + if not row: + self.conn.rollback() + return None + reserve_bytes = max( + KEYCHECK_RESULT_PROJECTION_RESERVE_BYTES, + int(result_projection_reserve_bytes), + ) + if ( + int(row['result_projection_reserved_bytes'] or 0) == 0 + and not row['result_projection_credit_transferred'] + ): + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + if ( + int(capacity['projection_items']) + 1 > int(projection_max_items) + or int(capacity['projection_bytes']) + reserve_bytes > int(projection_max_bytes) + ): + self.conn.rollback() + return { + '_capacity_blocked': True, + 'service': service, + 'reason': 'projection_capacity_saturated', + } + self.conn.execute( + '''UPDATE pipeline_capacity SET projection_items = projection_items + 1, + projection_bytes = projection_bytes + ?, updated_at = ? WHERE id = 1''', + (reserve_bytes, now), + ) + self.conn.execute( + '''UPDATE keycheck_candidates SET result_projection_reserved_bytes = ? + WHERE id = ?''', + (reserve_bytes, row['id']), + ) + self.conn.execute( + '''UPDATE keycheck_candidates SET state = 'leased', attempts = attempts + 1, + lease_owner = ?, lease_token = ?, lease_expires_at = ?, updated_at = ? + WHERE id = ?''', + (str(lease_owner), token, expires, now, row['id']), + ) + candidate = self.conn.execute( + '''SELECT c.*, kc.candidate_kind, kc.credential_hash, + kc.provider_key_hash, kc.secret_text, + kc.secret_json, kc.key_masked, kc.endpoint, kc.principal, + kc.metadata_json AS credential_metadata_json + FROM keycheck_candidates c + JOIN keycheck_credentials kc ON kc.id = c.credential_id + WHERE c.id = ?''', + (row['id'],), + ).fetchone() + finding = None + if candidate['finding_id'] is not None: + finding_row = self.conn.execute( + '''SELECT f.*, cp.raw_value, cp.raw_v2_value, cp.structured_data_json, + cp.extra_data_json, cp.analysis_info_json, cp.extension_json, + cp.payload_sha256, cp.payload_bytes, cp.payload_omitted + FROM findings f + LEFT JOIN finding_compat_payloads cp ON cp.finding_id = f.id + WHERE f.id = ?''', + (candidate['finding_id'],), + ).fetchone() + if finding_row: + if finding_row['payload_omitted']: + finding = { + 'finding_uid': finding_row['finding_uid'], + 'DetectorName': finding_row['detector_name'], + 'finding_omitted': True, + 'payload_sha256': finding_row['payload_sha256'], + } + else: + finding = safe_json_loads(finding_row['extension_json']) or {} + finding.update({ + 'finding_uid': finding_row['finding_uid'], + 'DetectorName': finding_row['detector_name'], + 'DetectorType': finding_row['detector_type'], + 'Verified': bool(finding_row['verified']), + 'Raw': finding_row['raw_value'], + 'RawV2': finding_row['raw_v2_value'], + }) + for column, key in ( + ('structured_data_json', 'StructuredData'), + ('extra_data_json', 'ExtraData'), + ('analysis_info_json', 'AnalysisInfo'), + ): + value = safe_json_loads(finding_row[column]) + if value is not None: + finding[key] = value + output = dict(candidate) + output['finding'] = finding or {} + self.conn.commit() + return output + except Exception: + self.conn.rollback() + raise + + def keycheck_candidate_cached_status(self, candidate_id, lease_token): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('cached keycheck status lookup requires PostgreSQL') + try: + row = self.conn.execute( + '''SELECT c.service, c.state, c.lease_token, c.metadata_json, + s.status, s.status_group, s.last_result_id, s.result_source, + s.checked_at, s.state_version, + EXISTS( + SELECT 1 FROM pipeline_quarantine q + WHERE q.keycheck_candidate_id = c.id + AND q.review_status = 'approved_retry' + ) AS approved_retry + FROM keycheck_candidates c + LEFT JOIN keycheck_current_state s ON s.credential_id = c.credential_id + WHERE c.id = ?''', + (int(candidate_id),), + ).fetchone() + if ( + not row or row['state'] != 'leased' + or str(row['lease_token'] or '') != str(lease_token) + ): + self.conn.rollback() + return None + metadata = safe_json_loads(row['metadata_json']) or {} + reason = '' + if row['service'] == 'provider_resolver': + reason = 'provider_resolution_required' + elif 'recheck_of_status' in metadata or 'recheck_generation' in metadata: + reason = 'explicit_recheck' + elif row['approved_retry']: + reason = 'approved_quarantine_retry' + elif row['state_version'] is None: + reason = 'no_current_state' + self.conn.commit() + if reason: + return {'probe_required': True, 'reason': reason} + return { + 'probe_required': False, + 'status': row['status'], + 'status_group': row['status_group'], + 'last_result_id': int(row['last_result_id']), + 'result_source': row['result_source'], + 'checked_at': row['checked_at'], + 'state_version': int(row['state_version']), + } + except Exception: + self.conn.rollback() + raise + + def defer_keycheck_candidate(self, candidate_id, lease_token, error='', delay_seconds=60): + if not self.conn or not self.conn.is_postgres: + return False + now = utc_now_iso() + available = datetime.fromtimestamp( + time.time() + max(1, int(delay_seconds)), timezone.utc, + ).isoformat(timespec='seconds') + try: + candidate = self.conn.execute( + '''SELECT * FROM keycheck_candidates + WHERE id = ? AND state = 'leased' AND lease_token = ? FOR UPDATE''', + (int(candidate_id), str(lease_token)), + ).fetchone() + if not candidate: + self.conn.rollback() + return False + reserved = ( + int(candidate['result_projection_reserved_bytes'] or 0) + if not candidate['result_projection_credit_transferred'] else 0 + ) + if reserved: + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + if int(capacity['projection_items']) < 1 or int(capacity['projection_bytes']) < reserved: + raise RuntimeError('keycheck defer would make projection capacity negative') + self.conn.execute( + '''UPDATE pipeline_capacity SET projection_items = projection_items - 1, + projection_bytes = projection_bytes - ?, updated_at = ? WHERE id = 1''', + (reserved, now), + ) + self.conn.execute( + '''UPDATE keycheck_candidates SET state = 'deferred', available_after = ?, + lease_owner = NULL, lease_token = NULL, lease_expires_at = NULL, + result_projection_reserved_bytes = 0, + last_error = ?, updated_at = ? WHERE id = ?''', + (available, first_line(error, 1000), now, int(candidate_id)), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + def quarantine_keycheck_candidate(self, candidate_id, lease_token, reason_code, detail=''): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('keycheck candidate quarantine requires PostgreSQL') + now = utc_now_iso() + try: + candidate = self.conn.execute( + '''SELECT * FROM keycheck_candidates + WHERE id = ? AND state = 'leased' AND lease_token = ? FOR UPDATE''', + (int(candidate_id), str(lease_token)), + ).fetchone() + if not candidate: + self.conn.rollback() + return False + existing = self.conn.execute( + '''SELECT id FROM pipeline_quarantine + WHERE subsystem = 'keycheck' AND object_type = 'keycheck_candidate' + AND object_id = ? AND review_status = 'pending' FOR UPDATE''', + (int(candidate_id),), + ).fetchone() + if existing: + self.conn.commit() + return True + projection_bytes = ( + int(candidate['result_projection_reserved_bytes'] or 0) + if not candidate['result_projection_credit_transferred'] else 0 + ) + quarantine_items = (0 if candidate['capacity_released'] else 1) + int(projection_bytes > 0) + quarantine_bytes = ( + (0 if candidate['capacity_released'] else int(candidate['capacity_bytes'])) + + projection_bytes + ) + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + if ( + int(capacity['keycheck_items']) < (0 if candidate['capacity_released'] else 1) + or int(capacity['keycheck_bytes']) < ( + 0 if candidate['capacity_released'] else int(candidate['capacity_bytes']) + ) + or int(capacity['projection_items']) < int(projection_bytes > 0) + or int(capacity['projection_bytes']) < projection_bytes + ): + raise RuntimeError('keycheck quarantine would make capacity accounting negative') + self.conn.execute( + '''UPDATE pipeline_capacity SET + keycheck_items = keycheck_items - ?, keycheck_bytes = keycheck_bytes - ?, + projection_items = projection_items - ?, projection_bytes = projection_bytes - ?, + quarantine_items = quarantine_items + ?, + quarantine_bytes = quarantine_bytes + ?, updated_at = ? WHERE id = 1''', + ( + 0 if candidate['capacity_released'] else 1, + 0 if candidate['capacity_released'] else candidate['capacity_bytes'], + int(projection_bytes > 0), projection_bytes, + quarantine_items, quarantine_bytes, now, + ), + ) + self.conn.insert_returning_id( + '''INSERT INTO pipeline_quarantine( + subsystem, object_type, object_id, keycheck_candidate_id, + reason_code, reason_detail, byte_count, capacity_items, + capacity_bytes, detected_at + ) VALUES ('keycheck','keycheck_candidate',?,?,?,?,?,?,?,?)''', + ( + candidate_id, candidate_id, str(reason_code), first_line(detail, 2000), + 0, quarantine_items, quarantine_bytes, now, + ), + ) + self.conn.execute( + '''UPDATE keycheck_candidates SET state = 'quarantined', capacity_released = 1, + result_projection_credit_transferred = 1, + lease_owner = NULL, lease_token = NULL, lease_expires_at = NULL, + last_error = ?, completed_at = ?, updated_at = ? WHERE id = ?''', + (first_line(detail or reason_code, 1000), now, now, candidate_id), + ) + self.conn.commit() + return True + except Exception: + self.conn.rollback() + raise + + def complete_keycheck_candidate( + self, candidate_id, lease_token, event_id, status, status_group, + checked_at=None, message='', metadata=None, result_source='api_check', + resolved_service='', cached_state_version=None, cached_last_result_id=None, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('normal keycheck completion requires PostgreSQL') + metadata = dict(metadata or {}) + resolved_service = str(resolved_service or '').strip().lower() + cached_completion = ( + cached_state_version is not None or cached_last_result_id is not None + ) + if cached_completion: + if cached_state_version is None or cached_last_result_id is None: + raise ValueError('cached keycheck completion requires an exact current-state fence') + if str(result_source or '') != 'cached_status': + raise ValueError('cached keycheck completion requires cached_status result source') + if resolved_service: + raise ValueError('provider resolution cannot use cached keycheck completion') + if ( + int(metadata.get('cached_state_version') or -1) != int(cached_state_version) + or int(metadata.get('cached_result_id') or -1) != int(cached_last_result_id) + ): + raise ValueError('cached keycheck metadata conflicts with its current-state fence') + if resolved_service: + if not re.fullmatch(r'[a-z][a-z0-9_]{1,63}', resolved_service): + raise ValueError('resolved keycheck service is invalid') + if str(metadata.get('resolved_provider') or '').strip().lower() != resolved_service: + raise ValueError('resolved keycheck service conflicts with result metadata') + if str(metadata.get('provider_resolution') or '') != 'matched': + raise ValueError('resolved keycheck service requires a matched provider resolution') + if not metadata.get('authenticated') and str(status).upper() not in ('VALID', 'ALIVE'): + raise ValueError('resolved keycheck service has no authenticated provider evidence') + event_payload = json.dumps({ + 'candidate_id': int(candidate_id), + 'event_id': str(event_id or ''), + 'status': str(status).upper(), + 'status_group': str(status_group).lower(), + 'checked_at': str(checked_at or ''), + 'message': first_line(message, 1000), + 'metadata': metadata, + 'result_source': str(result_source or 'api_check'), + }, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str).encode('utf-8') + metadata['_event_hash'] = hashlib.sha256(event_payload).hexdigest() + encoded_metadata = json.dumps( + metadata, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8') + if len(encoded_metadata) > 1024 * 1024: + raise ValueError('keycheck result metadata exceeds its byte bound') + event_id = str(event_id or '') + if not re.fullmatch(r'[a-f0-9]{64}', event_id): + raise ValueError('keycheck event ID must be a stable SHA-256 identity') + now = utc_now_iso() + checked = str(checked_at or now) + try: + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + candidate = self.conn.execute( + '''SELECT c.*, kc.credential_hash, kc.provider_key_hash, + kc.key_masked, kc.secret_text, kc.secret_json, + kc.candidate_kind, kc.endpoint, kc.principal, + kc.metadata_json AS credential_metadata_json + FROM keycheck_candidates c + JOIN keycheck_credentials kc ON kc.id = c.credential_id + WHERE c.id = ? FOR UPDATE OF c''', + (int(candidate_id),), + ).fetchone() + if not candidate: + raise ValueError('keycheck candidate is absent') + existing_map = self.conn.execute( + 'SELECT keycheck_result_id FROM keycheck_event_map WHERE event_id = ? FOR UPDATE', + (event_id,), + ).fetchone() + if existing_map: + if int(candidate['keycheck_result_id'] or 0) != int(existing_map['keycheck_result_id'] or 0): + raise ScanEventConflictError('keycheck event replay conflicts with candidate completion') + existing_result = self.conn.execute( + 'SELECT metadata_json FROM keycheck_results WHERE id = ?', + (existing_map['keycheck_result_id'],), + ).fetchone() + existing_metadata = safe_json_loads(existing_result['metadata_json'] if existing_result else '') or {} + if existing_metadata.get('_event_hash') != metadata['_event_hash']: + raise ScanEventConflictError('keycheck event replay has a different payload hash') + self.conn.commit() + return { + 'completed': True, 'duplicate': True, + 'keycheck_result_id': existing_map['keycheck_result_id'], 'event_id': event_id, + } + if candidate['state'] != 'leased' or str(candidate['lease_token'] or '') != str(lease_token): + self.conn.rollback() + return None + candidate = dict(candidate) + if cached_completion: + candidate_metadata = safe_json_loads(candidate['metadata_json']) or {} + approved_retry = self.conn.execute( + '''SELECT 1 FROM pipeline_quarantine + WHERE keycheck_candidate_id = ? AND review_status = 'approved_retry' + LIMIT 1''', + (int(candidate_id),), + ).fetchone() + if ( + candidate['service'] == 'provider_resolver' + or 'recheck_of_status' in candidate_metadata + or 'recheck_generation' in candidate_metadata + or approved_retry + ): + self.conn.rollback() + return {'completed': False, 'probe_required': True} + current_state = self.conn.execute( + '''SELECT service, status, status_group, last_result_id, state_version + FROM keycheck_current_state WHERE credential_id = ? FOR UPDATE''', + (candidate['credential_id'],), + ).fetchone() + if not current_state: + self.conn.rollback() + return {'completed': False, 'probe_required': True} + if ( + current_state['service'] != candidate['service'] + or int(current_state['state_version']) != int(cached_state_version) + or int(current_state['last_result_id']) != int(cached_last_result_id) + or str(current_state['status']).upper() != str(status).upper() + or str(current_state['status_group']).lower() != str(status_group).lower() + ): + self.conn.rollback() + return {'completed': False, 'cached_state_changed': True} + if resolved_service and resolved_service != candidate['service']: + allowed_services = {'provider_resolver', 'qwen', 'deepseek', 'kimi', 'zai'} + if candidate['service'] not in allowed_services or resolved_service not in allowed_services - {'provider_resolver'}: + raise ValueError('provider resolution cannot reassign this keycheck service') + if candidate['candidate_kind'] != 'provider_key' or not candidate['secret_text']: + raise ValueError('provider resolution requires a text provider-key credential') + probe_material = str(candidate['secret_text']) + expected_provider_hash = hashlib.sha256(probe_material.encode('utf-8')).hexdigest() + if expected_provider_hash != candidate['provider_key_hash']: + raise ScanEventConflictError('provider resolution credential material conflicts with its key hash') + resolved_credential_hash = hashlib.sha256('|'.join(( + 'truf-credential-v2', resolved_service, probe_material, + )).encode('utf-8')).hexdigest() + resolved_credential_id = self.conn.insert_returning_id( + '''INSERT INTO keycheck_credentials( + service, credential_hash, provider_key_hash, candidate_kind, + secret_text, secret_json, key_masked, endpoint, principal, + metadata_json, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(service, provider_key_hash) DO NOTHING''', + ( + resolved_service, resolved_credential_hash, candidate['provider_key_hash'], + candidate['candidate_kind'], candidate['secret_text'], candidate['secret_json'], + candidate['key_masked'], candidate['endpoint'], candidate['principal'], + candidate['credential_metadata_json'], now, now, + ), + ) + if resolved_credential_id is None: + resolved_credential = self.conn.execute( + '''SELECT id, credential_hash, secret_text, secret_json + FROM keycheck_credentials + WHERE service = ? AND provider_key_hash = ? FOR UPDATE''', + (resolved_service, candidate['provider_key_hash']), + ).fetchone() + if not resolved_credential or ( + resolved_credential['credential_hash'] != resolved_credential_hash + or resolved_credential['secret_text'] != candidate['secret_text'] + or resolved_credential['secret_json'] != candidate['secret_json'] + ): + raise ScanEventConflictError('resolved provider credential conflicts with canonical material') + resolved_credential_id = resolved_credential['id'] + self.conn.execute( + '''UPDATE keycheck_candidates SET credential_id = ?, service = ?, updated_at = ? + WHERE id = ? AND state = 'leased' AND lease_token = ?''', + ( + resolved_credential_id, resolved_service, now, + int(candidate_id), str(lease_token), + ), + ) + candidate['credential_id'] = resolved_credential_id + candidate['credential_hash'] = resolved_credential_hash + candidate['service'] = resolved_service + result_message = first_line(message, 1000) + projected_result = { + 'event_id': event_id, + 'service': candidate['service'], + 'status': str(status).upper(), + 'status_group': str(status_group).lower(), + 'checked_at': checked, + 'key_hash': candidate['provider_key_hash'], + 'secret_hash': candidate['secret_hash'], + 'key_masked': candidate['key_masked'], + 'finding_uid': candidate['finding_uid'], + 'detector': candidate['detector_name'], + 'source': candidate['source'], + 'message': result_message, + 'metadata': metadata, + 'result_source': str(result_source or 'api_check'), + } + result_projection_bytes = len(json.dumps( + projected_result, ensure_ascii=False, sort_keys=True, + separators=(',', ':'), default=str, + ).encode('utf-8')) + 1 + projection_capacity_bytes = result_projection_bytes + reserved_projection_bytes = int(candidate['result_projection_reserved_bytes'] or 0) + if ( + reserved_projection_bytes < KEYCHECK_RESULT_PROJECTION_RESERVE_BYTES + or candidate['result_projection_credit_transferred'] + ): + raise RuntimeError('keycheck result has no exact pre-probe projection reservation') + if projection_capacity_bytes > reserved_projection_bytes: + raise ValueError( + 'keycheck result compatibility output exceeds its pre-reserved byte capacity' + ) + result_id = self.conn.insert_returning_id( + '''INSERT INTO keycheck_results( + service, status, status_group, checked_at, key_hash, secret_hash, + key_masked, finding_id, target_scan_id, source, query, target, + detector_name, found_at, message, metadata_json, event_id, finding_uid, + link_status, link_attempts, linked_at, link_error, + candidate_id, credential_id, result_source, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, + 'linked', 0, ?, '', ?, ?, ?, ?)''', + ( + candidate['service'], str(status).upper(), str(status_group).lower(), checked, + candidate['provider_key_hash'], candidate['secret_hash'], + candidate['key_masked'], candidate['finding_id'], candidate['target_scan_id'], + candidate['source'], candidate['query'], candidate['target'], + candidate['detector_name'], candidate['found_at'], result_message, + encoded_metadata.decode('utf-8'), event_id, candidate['finding_uid'], now, + candidate_id, candidate['credential_id'], str(result_source or 'api_check'), now, + ), + ) + self.conn.execute( + '''INSERT INTO keycheck_event_map(event_id, keycheck_result_id, created_at) + VALUES (?, ?, ?)''', + (event_id, result_id, now), + ) + if not cached_completion: + self.conn.execute( + '''INSERT INTO keycheck_current_state( + credential_id, service, status, status_group, last_result_id, + result_source, checked_at, recheck_after, state_version, + metadata_json, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 1, ?, ?) + ON CONFLICT(credential_id) DO UPDATE SET + service = excluded.service, status = excluded.status, + status_group = excluded.status_group, last_result_id = excluded.last_result_id, + result_source = excluded.result_source, checked_at = excluded.checked_at, + recheck_after = excluded.recheck_after, + state_version = keycheck_current_state.state_version + 1, + metadata_json = excluded.metadata_json, updated_at = excluded.updated_at''', + ( + candidate['credential_id'], candidate['service'], str(status).upper(), + str(status_group).lower(), result_id, str(result_source or 'api_check'), + checked, metadata.get('recheck_after'), encoded_metadata.decode('utf-8'), now, + ), + ) + if not candidate['capacity_released']: + if ( + int(capacity['keycheck_items']) < 1 + or int(capacity['keycheck_bytes']) < int(candidate['capacity_bytes']) + or int(capacity['projection_items']) < 1 + or int(capacity['projection_bytes']) < reserved_projection_bytes + ): + raise RuntimeError('keycheck completion would make candidate capacity negative') + self.conn.execute( + '''UPDATE pipeline_capacity SET keycheck_items = keycheck_items - 1, + keycheck_bytes = keycheck_bytes - ?, + projection_bytes = projection_bytes - ?, updated_at = ? WHERE id = 1''', + ( + candidate['capacity_bytes'], + reserved_projection_bytes - projection_capacity_bytes, now, + ), + ) + for stream_name, relative_path in (( + f'keycheck:{candidate["service"]}:results', + f'{candidate["service"]}/{candidate["service"]}Results.jsonl', + ),): + self.conn.execute( + '''INSERT INTO projection_streams( + stream_name, base_relative_path, current_generation, + rotation_bytes, max_generations, created_at, updated_at + ) VALUES (?, ?, 0, ?, 16, ?, ?) + ON CONFLICT(stream_name) DO NOTHING''', + (stream_name, relative_path, 32 * 1024 * 1024, now, now), + ) + self.conn.execute( + '''INSERT INTO projection_cursors( + stream_name, generation, committed_offset, updated_at + ) VALUES (?, 0, 0, ?) ON CONFLICT(stream_name) DO NOTHING''', + (stream_name, now), + ) + projection_id = self.conn.insert_returning_id( + '''INSERT INTO projection_jobs( + job_kind, event_id, event_hash, keycheck_result_id, status, + required_stream_mask, capacity_items, capacity_bytes, + created_at, updated_at + ) VALUES ('keycheck_event', ?, ?, ?, 'pending', 8, 1, ?, ?, ?)''', + ( + event_id, hashlib.sha256(encoded_metadata + event_id.encode('ascii')).hexdigest(), + result_id, projection_capacity_bytes, now, now, + ), + ) + self.conn.execute( + '''UPDATE keycheck_candidates SET state = 'completed', keycheck_result_id = ?, + capacity_released = 1, lease_owner = NULL, lease_token = NULL, + lease_expires_at = NULL, result_projection_credit_transferred = 1, + completed_at = ?, updated_at = ? WHERE id = ?''', + (result_id, now, now, candidate_id), + ) + self.conn.commit() + return { + 'completed': True, 'duplicate': False, 'keycheck_result_id': result_id, + 'projection_job_id': projection_id, 'event_id': event_id, + } + except Exception: + self.conn.rollback() + raise + + def keycheck_result_for_projection(self, keycheck_result_id): + if not self.conn: + return None + row = self.conn.execute( + '''SELECT kr.*, kc.candidate_kind, kc.key_masked AS credential_masked, + kc.secret_text AS credential_secret_text, + kc.secret_json AS credential_secret_json, + c.candidate_uid + FROM keycheck_results kr + LEFT JOIN keycheck_credentials kc ON kc.id = kr.credential_id + LEFT JOIN keycheck_candidates c ON c.id = kr.candidate_id + WHERE kr.id = ?''', + (int(keycheck_result_id),), + ).fetchone() + if self.conn.is_postgres: + self.conn.commit() + return dict(row) if row else None + + def enqueue_keycheck_rechecks( + self, service, status_groups=None, max_items=1000, + queue_max_items=100000, queue_max_bytes=512 * 1024 * 1024, + ): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('keycheck recheck generation requires PostgreSQL') + service = str(service or '').lower() + groups = sorted({str(value).lower() for value in (status_groups or []) if value}) + max_items = min(10000, max(1, int(max_items))) + now = utc_now_iso() + status_scope = ','.join(groups) if groups else '*' + cursor_key = hashlib.sha256( + f'truf-keycheck-recheck-cursor-v1|{service}|{status_scope}'.encode('utf-8') + ).hexdigest() + try: + self.conn.execute( + '''INSERT INTO keycheck_recheck_cursors( + cursor_key, service, status_scope, last_credential_id, + wrap_count, updated_at + ) VALUES (?, ?, ?, 0, 0, ?) ON CONFLICT(cursor_key) DO NOTHING''', + (cursor_key, service, status_scope, now), + ) + cursor = self.conn.execute( + 'SELECT * FROM keycheck_recheck_cursors WHERE cursor_key = ? FOR UPDATE', + (cursor_key,), + ).fetchone() + clauses = [ + 's.service = ?', + '''NOT EXISTS ( + SELECT 1 FROM pipeline_quarantine q + JOIN keycheck_candidates blocked ON blocked.id = q.keycheck_candidate_id + WHERE blocked.credential_id = s.credential_id + AND q.review_status = 'pending' + AND q.reason_code IN ( + 'provider_candidate_unconsumed', + 'candidate_provider_route_mismatch' + ) + )''', + ] + params = [service] + if groups: + clauses.append('s.status_group IN ({})'.format(','.join('?' for _ in groups))) + params.extend(groups) + + def select_rows(after_id): + return self.conn.execute( + f'''SELECT s.credential_id, s.state_version, s.status, s.status_group, + c.credential_hash, c.candidate_kind, c.secret_text, c.secret_json, + r.secret_hash, r.source, r.query, r.target, r.detector_name, + r.found_at, r.finding_uid + FROM keycheck_current_state s + JOIN keycheck_credentials c ON c.id = s.credential_id + JOIN keycheck_results r ON r.id = s.last_result_id + WHERE {' AND '.join(clauses)} AND s.credential_id > ? + ORDER BY s.credential_id LIMIT ?''', + (*params, int(after_id), max_items), + ).fetchall() + + rows = select_rows(cursor['last_credential_id']) + wrapped = False + if not rows and int(cursor['last_credential_id'] or 0) > 0: + rows = select_rows(0) + wrapped = True + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + inserted = 0 + inserted_bytes = 0 + last_scanned = 0 if wrapped else int(cursor['last_credential_id'] or 0) + generation = int(cursor['wrap_count'] or 0) + (1 if wrapped else 0) + for row in rows: + uid = hashlib.sha256('|'.join(( + 'truf-keycheck-recheck-v2', service, status_scope, + str(generation), str(row['credential_id']), str(row['state_version']), + )).encode('utf-8')).hexdigest() + capacity_bytes = len(str(row['secret_text'] or row['secret_json'] or '').encode('utf-8')) + 512 + if ( + int(capacity['keycheck_items']) + inserted + 1 > int(queue_max_items) + or int(capacity['keycheck_bytes']) + inserted_bytes + capacity_bytes > int(queue_max_bytes) + ): + break + candidate_id = self.conn.insert_returning_id( + '''INSERT INTO keycheck_candidates( + candidate_uid, credential_id, service, routed_service, secret_hash, + source, query, target, detector_name, found_at, finding_uid, + metadata_json, state, + capacity_bytes, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 'pending', ?, ?, ?) + ON CONFLICT(candidate_uid) DO NOTHING''', + ( + uid, row['credential_id'], service, service, row['secret_hash'], + row['source'], row['query'], row['target'], row['detector_name'], + row['found_at'], row['finding_uid'], + json_dumps({ + 'provider_hint': service, + 'recheck_of_status': row['status'], + 'state_version': row['state_version'], + 'recheck_generation': generation, + }), + capacity_bytes, now, now, + ), + ) + if candidate_id is not None: + inserted += 1 + inserted_bytes += capacity_bytes + last_scanned = int(row['credential_id']) + if inserted: + self.conn.execute( + '''UPDATE pipeline_capacity SET keycheck_items = keycheck_items + ?, + keycheck_bytes = keycheck_bytes + ?, updated_at = ? WHERE id = 1''', + (inserted, inserted_bytes, now), + ) + self.conn.execute( + '''UPDATE keycheck_recheck_cursors SET last_credential_id = ?, + wrap_count = wrap_count + ?, updated_at = ? WHERE cursor_key = ?''', + (last_scanned, 1 if wrapped else 0, now, cursor_key), + ) + self.conn.commit() + return inserted + except Exception: + self.conn.rollback() + raise + + def keycheck_service_has_work(self, service): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('keycheck work inspection requires PostgreSQL') + now = utc_now_iso() + row = self.conn.execute( + '''SELECT ( + EXISTS ( + SELECT 1 FROM keycheck_candidates + WHERE service = ? AND state = 'pending' + ) OR EXISTS ( + SELECT 1 FROM keycheck_candidates + WHERE service = ? AND state = 'deferred' + AND available_after IS NOT NULL AND available_after <= ? + ) OR EXISTS ( + SELECT 1 FROM keycheck_candidates + WHERE service = ? AND state = 'leased' + AND lease_expires_at IS NOT NULL AND lease_expires_at <= ? + ) + ) AS present''', + (str(service).lower(), str(service).lower(), now, str(service).lower(), now), + ).fetchone() + self.conn.commit() + return bool(row and row['present']) + + def has_claimable_targets(self, source, platform, max_attempts=0): + if not self.conn: + return False + + def op(): + now = utc_now_iso() + max_attempts_value = max(0, int(max_attempts or 0)) + row = self.conn.execute( + '''SELECT 1 AS present FROM target_queue + WHERE source = ? AND platform = ? + AND current_result_reservation_id IS NULL + AND (status = 'pending' + OR (status = 'deferred' AND available_after IS NOT NULL AND available_after <= ?) + OR (status = 'in_progress' AND lease_expires_at IS NOT NULL AND lease_expires_at <= ?)) + AND (? = 0 OR COALESCE(attempts, 0) < ?) + AND (available_after IS NULL OR available_after <= ?) + AND (resolver_state IS NULL OR resolver_state = 'resolved') + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_targets experiment_target + WHERE experiment_target.target_queue_id = target_queue.id + ) + ORDER BY id LIMIT 1''', + (source, platform, now, now, max_attempts_value, max_attempts_value, now), + ).fetchone() + if self.conn.is_postgres: + self.conn.commit() + return bool(row) + + return bool(self._safe('has_claimable_targets', op, False)) + + def has_claimable_targets_v2(self, source, platform, max_attempts=0): + if not self.conn: + return False + try: + now = utc_now_iso() + maximum = max(0, int(max_attempts or 0)) + row = self.conn.execute( + HAS_CLAIMABLE_TARGETS_V2_SQL, + ( + source, platform, now, maximum, maximum, + source, platform, now, maximum, maximum, + ), + ).fetchone() + if self.conn.is_postgres: + self.conn.commit() + self.last_error = '' + return bool(row and row['present']) + except Exception as exc: + self.last_error = str(exc) + self.conn.rollback() + return False + + def claim_targets( + self, source, platform, limit, lease_owner=None, lease_seconds=86400, + max_attempts=0, return_rows=False, claim_batch=None, + ): + if not self.conn: + return None + batch = str(claim_batch or secrets.token_urlsafe(24)) + self._last_claim_expectation = None + commit_state = {'started': False} + + def commit_claim(): + commit_state['started'] = True + self.conn.commit() + commit_state['started'] = False + + def remember_expected(owner, rows): + self._last_claim_expectation = { + 'claim_batch': batch, + 'lease_owner': str(owner), + 'claims': [ + { + 'id': int(row['id']), + 'lease_token': str(row['lease_token']), + } + for row in rows + ], + } + + def op(): + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + now = utc_now_iso() + lease_until = datetime.fromtimestamp(time.time() + max(60, int(lease_seconds or 86400)), timezone.utc).isoformat(timespec='seconds') + limit_value = max(1, int(limit or 1000000)) + max_attempts_value = max(0, int(max_attempts or 0)) + owner = lease_owner or f'{source}:{os.getpid()}' + existing = self.conn.execute( + '''SELECT id, target, normalized_target, attempts, lease_owner, lease_token, claim_batch + FROM target_queue WHERE source = ? AND platform = ? AND status = 'in_progress' + AND current_result_reservation_id IS NULL + AND lease_owner = ? AND claim_batch = ? ORDER BY id''', + (source, platform, owner, batch), + ).fetchall() + if existing: + remember_expected(owner, existing) + commit_claim() + return existing if return_rows else [row['target'] for row in existing] + candidate_limit = min( + CLAIM_CANDIDATE_MAX, + max(CLAIM_CANDIDATE_MIN, limit_value * 8), + ) + if max_attempts_value: + if self.conn.is_postgres: + exhausted = self.conn.execute( + '''WITH candidates AS MATERIALIZED ( + SELECT id FROM ( + (SELECT id FROM target_queue + WHERE source = ? AND platform = ? AND status = 'pending' + AND current_result_reservation_id IS NULL + AND attempts >= ? + ORDER BY id LIMIT ?) + UNION ALL + (SELECT id FROM target_queue + WHERE source = ? AND platform = ? AND status = 'deferred' + AND current_result_reservation_id IS NULL + AND attempts >= ? + ORDER BY id LIMIT ?) + UNION ALL + (SELECT id FROM target_queue + WHERE source = ? AND platform = ? AND status = 'in_progress' + AND current_result_reservation_id IS NULL + AND lease_expires_at IS NOT NULL AND lease_expires_at <= ? + AND attempts >= ? + ORDER BY id LIMIT ?) + ) AS status_candidates + ORDER BY id LIMIT ? + ) + SELECT q.id FROM candidates c + JOIN target_queue q ON q.id = c.id + ORDER BY q.id FOR UPDATE OF q SKIP LOCKED''', + ( + source, platform, max_attempts_value, candidate_limit, + source, platform, max_attempts_value, candidate_limit, + source, platform, now, max_attempts_value, candidate_limit, + candidate_limit, + ), + ).fetchall() + exhausted_ids = [row['id'] for row in exhausted] + if exhausted_ids: + placeholders = ','.join('?' for _ in exhausted_ids) + self.conn.execute( + f'''UPDATE target_queue SET status = 'failed', completed_at = COALESCE(completed_at, ?), + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, leased_at = NULL, lease_expires_at = NULL, + available_after = NULL, last_error = COALESCE(last_error, 'target retry attempts exhausted'), updated_at = ? + WHERE id IN ({placeholders})''', + (now, now, *exhausted_ids), + ) + else: + self.conn.execute( + '''UPDATE target_queue SET status = 'failed', completed_at = COALESCE(completed_at, ?), + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, leased_at = NULL, lease_expires_at = NULL, + available_after = NULL, last_error = COALESCE(last_error, 'target retry attempts exhausted'), updated_at = ? + WHERE source = ? AND platform = ? AND COALESCE(attempts, 0) >= ? + AND current_result_reservation_id IS NULL + AND (status IN ('pending', 'deferred') + OR (status = 'in_progress' AND lease_expires_at IS NOT NULL AND lease_expires_at <= ?))''', + (now, now, source, platform, max_attempts_value, now), + ) + if self.conn.is_postgres: + rows = self.conn.execute( + '''WITH candidates AS MATERIALIZED ( + SELECT id FROM ( + (SELECT id FROM target_queue + WHERE source = ? AND platform = ? AND status = 'pending' + AND current_result_reservation_id IS NULL + AND (? = 0 OR COALESCE(attempts, 0) < ?) + AND (available_after IS NULL OR available_after <= ?) + AND (resolver_state IS NULL OR resolver_state = 'resolved') + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_targets experiment_target + WHERE experiment_target.target_queue_id = target_queue.id + ) + ORDER BY id LIMIT ?) + UNION ALL + (SELECT id FROM target_queue + WHERE source = ? AND platform = ? AND status = 'deferred' + AND current_result_reservation_id IS NULL + AND available_after IS NOT NULL AND available_after <= ? + AND (? = 0 OR COALESCE(attempts, 0) < ?) + AND (resolver_state IS NULL OR resolver_state = 'resolved') + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_targets experiment_target + WHERE experiment_target.target_queue_id = target_queue.id + ) + ORDER BY id LIMIT ?) + UNION ALL + (SELECT id FROM target_queue + WHERE source = ? AND platform = ? AND status = 'in_progress' + AND current_result_reservation_id IS NULL + AND lease_expires_at IS NOT NULL AND lease_expires_at <= ? + AND (? = 0 OR COALESCE(attempts, 0) < ?) + AND (available_after IS NULL OR available_after <= ?) + AND (resolver_state IS NULL OR resolver_state = 'resolved') + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_targets experiment_target + WHERE experiment_target.target_queue_id = target_queue.id + ) + ORDER BY id LIMIT ?) + ) AS status_candidates + ORDER BY id LIMIT ? + ) + SELECT q.id, q.target FROM candidates c + JOIN target_queue q ON q.id = c.id + ORDER BY q.id LIMIT ? FOR UPDATE OF q SKIP LOCKED''', + ( + source, platform, max_attempts_value, max_attempts_value, now, candidate_limit, + source, platform, now, max_attempts_value, max_attempts_value, candidate_limit, + source, platform, now, max_attempts_value, max_attempts_value, now, candidate_limit, + candidate_limit, limit_value, + ), + ).fetchall() + else: + rows = self.conn.execute( + '''SELECT id, target + FROM target_queue + WHERE source = ? AND platform = ? + AND current_result_reservation_id IS NULL + AND (status = 'pending' OR (status = 'deferred' AND available_after IS NOT NULL AND available_after <= ?) OR (status = 'in_progress' AND lease_expires_at IS NOT NULL AND lease_expires_at <= ?)) + AND (? = 0 OR COALESCE(attempts, 0) < ?) + AND (available_after IS NULL OR available_after <= ?) + AND (resolver_state IS NULL OR resolver_state = 'resolved') + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_targets experiment_target + WHERE experiment_target.target_queue_id = target_queue.id + ) + ORDER BY id LIMIT ?''', + (source, platform, now, now, max_attempts_value, max_attempts_value, now, limit_value), + ).fetchall() + ids = [row['id'] for row in rows] + for row_id in ids: + lease_token = secrets.token_urlsafe(32) + self.conn.execute( + '''UPDATE target_queue SET status = 'in_progress', lease_owner = ?, lease_token = ?, claim_batch = ?, leased_at = ?, lease_expires_at = ?, + attempts = COALESCE(attempts, 0) + 1, + scan_remote_modified_at = remote_modified_at, updated_at = ? WHERE id = ?''', + (owner, lease_token, batch, now, lease_until, now, row_id), + ) + claimed_rows = None + if return_rows and ids: + placeholders = ','.join('?' for _ in ids) + claimed_rows = self.conn.execute( + f'''SELECT id, target, normalized_target, attempts, lease_owner, lease_token, claim_batch + FROM target_queue WHERE id IN ({placeholders}) ORDER BY id''', + ids, + ).fetchall() + expected_rows = claimed_rows if return_rows and ids else self.conn.execute( + f'''SELECT id, lease_token FROM target_queue + WHERE claim_batch = ? AND lease_owner = ? ORDER BY id''', + (batch, owner), + ).fetchall() + remember_expected(owner, expected_rows) + commit_claim() + if return_rows and ids: + return claimed_rows + return [row['target'] for row in rows] + + last_error = None + for attempt in range(SQLITE_LOCK_RETRY_ATTEMPTS): + commit_state['started'] = False + try: + result = op() + self.last_error = '' + return result + except Exception as exc: + last_error = exc + self.last_error = str(exc) + try: + self.conn.rollback() + except Exception: + pass + if commit_state['started']: + raise + if not sqlite_lock_error(exc) or attempt == SQLITE_LOCK_RETRY_ATTEMPTS - 1: + raise + time.sleep(sqlite_lock_retry_delay(attempt)) + raise last_error + + def claim_recovery_expectation(self, claim_batch, lease_owner): + expectation = self._last_claim_expectation + if not expectation: + return None + if ( + str(expectation.get('claim_batch') or '') != str(claim_batch or '') + or str(expectation.get('lease_owner') or '') != str(lease_owner or '') + ): + return None + return [dict(claim) for claim in expectation.get('claims') or []] + + def recover_claim_batch(self, claim_batch, lease_owner): + if not claim_batch or not lease_owner: + return [] + if not self.conn and not self._reset_connection(): + return [] + + def op(): + rows = self.conn.execute( + '''SELECT id, target, normalized_target, attempts, lease_owner, lease_token, claim_batch + FROM target_queue WHERE status = 'in_progress' AND claim_batch = ? + AND lease_owner = ? ORDER BY id''', + (str(claim_batch), str(lease_owner)), + ).fetchall() + if self.conn.is_postgres: + self.conn.commit() + return rows + rows = self._safe('recover_claim_batch', op, []) + if self.last_error: + first_error = self.last_error + if not self._reset_connection(): + raise RuntimeError(f'unable to reconnect for committed claim batch recovery: {first_error}') + rows = self._safe('recover_claim_batch_after_reconnect', op, []) + if self.last_error: + raise RuntimeError(f'unable to recover committed claim batch: {self.last_error}') + return rows + + def reclaim_target_leases(self, source, platform, lease_owner=None): + if not self.conn: + return 0 + + def op(): + now = utc_now_iso() + cur = self.conn.execute( + '''UPDATE target_queue SET status = 'pending', lease_owner = NULL, lease_token = NULL, claim_batch = NULL, leased_at = NULL, + lease_expires_at = NULL, updated_at = ? + WHERE source = ? AND platform = ? AND status = 'in_progress' + AND current_result_reservation_id IS NULL + AND lease_expires_at IS NOT NULL AND lease_expires_at <= ?''', + (now, source, platform, now), + ) + self.conn.commit() + return int(getattr(cur, 'rowcount', 0) or 0) + return self._safe('reclaim_target_leases', op, 0) + + def complete_target_queue_item(self, source, platform, target, target_scan_id=None, status='done', error=None, available_after=None, lease_owner=None, reset_attempts=False, queue_id=None, lease_token=None): + if not self.conn or queue_id is None or not lease_token: + return False + + def op(): + now = utc_now_iso() + completed = now if status in ('done', 'failed') else None + params = [status, target_scan_id, first_line(error, 500) if error else None, available_after, 1 if reset_attempts else 0, completed, now] + params.extend((queue_id, lease_token)) + cur = self.conn.execute( + '''UPDATE target_queue SET + status = ?, target_scan_id = COALESCE(?, target_scan_id), last_error = ?, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, leased_at = NULL, lease_expires_at = NULL, + available_after = ?, attempts = CASE WHEN ? != 0 THEN 0 ELSE attempts END, + completed_at = COALESCE(?, completed_at), updated_at = ? + WHERE id = ? AND status = 'in_progress' AND lease_token = ?''', + params, + ) + self.conn.commit() + return getattr(cur, 'rowcount', 0) != 0 + return self._safe('complete_target_queue_item', op, False) + + def renew_target_leases(self, lease_owner, lease_seconds=1800, lease_tokens=None): + tokens = [str(token) for token in (lease_tokens or []) if token] + if not self.conn or not lease_owner or not tokens: + return 0 + + def op(): + now = utc_now_iso() + lease_until = datetime.fromtimestamp( + time.time() + max(60, int(lease_seconds or 1800)), timezone.utc + ).isoformat(timespec='seconds') + placeholders = ','.join('?' for _ in tokens) + cur = self.conn.execute( + f'''UPDATE target_queue SET lease_expires_at = ?, updated_at = ? + WHERE status = 'in_progress' AND lease_owner = ? + AND lease_token IN ({placeholders})''', + (lease_until, now, lease_owner, *tokens), + ) + self.conn.commit() + return int(getattr(cur, 'rowcount', 0) or 0) + return self._safe('renew_target_leases', op, 0) + + def active_target_lease_tokens(self, lease_owner, lease_tokens=None): + tokens = [str(token) for token in (lease_tokens or []) if token] + if not self.conn or not lease_owner or not tokens: + return set() + + def op(): + placeholders = ','.join('?' for _ in tokens) + rows = self.conn.execute( + f'''SELECT lease_token FROM target_queue + WHERE status = 'in_progress' AND lease_owner = ? + AND lease_token IN ({placeholders})''', + (lease_owner, *tokens), + ).fetchall() + if self.conn.is_postgres: + self.conn.commit() + return {str(row['lease_token']) for row in rows if row['lease_token']} + return self._safe('active_target_lease_tokens', op, set()) + + def refund_target_claims(self, claims, error='infrastructure persistence failure'): + normalized = [] + for claim in claims or []: + queue_id = claim.get('id') if isinstance(claim, dict) else claim['id'] + lease_token = claim.get('lease_token') if isinstance(claim, dict) else claim['lease_token'] + if queue_id is None or not lease_token: + return False + item = (int(queue_id), str(lease_token)) + if item not in normalized: + normalized.append(item) + if not self.conn or not normalized: + return False + + def op(): + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + now = utc_now_iso() + for queue_id, lease_token in normalized: + cur = self.conn.execute( + '''UPDATE target_queue SET status = 'pending', + attempts = CASE WHEN COALESCE(attempts, 0) > 0 THEN attempts - 1 ELSE 0 END, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, + leased_at = NULL, lease_expires_at = NULL, + available_after = NULL, completed_at = NULL, last_error = ?, updated_at = ? + WHERE id = ? AND status = 'in_progress' AND lease_token = ?''', + (first_line(error, 500), now, queue_id, lease_token), + ) + if int(getattr(cur, 'rowcount', 0) or 0) != 1: + self.conn.rollback() + return False + self.conn.commit() + return True + return self._safe('refund_target_claims', op, False) + + def refund_target_claim(self, queue_id, lease_token, error='infrastructure persistence failure'): + return self.refund_target_claims( + [{'id': queue_id, 'lease_token': lease_token}], error, + ) + + def refund_stopped_result_spool_claims(self, reservation, error='supervisor controlled source stop'): + claims = [dict(claim) for claim in (reservation or {}).get('claims') or []] + if not self.conn or len(claims) > 10000: + return False + if not claims: + return True + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + lock_suffix = ' FOR UPDATE' if self.conn.is_postgres else '' + rows = {} + for claim in claims: + queue_id = int(claim.get('queue_id')) + row = self.conn.execute( + f'''SELECT id, status, lease_owner, lease_token, claim_batch + FROM target_queue WHERE id = ?{lock_suffix}''', + (queue_id,), + ).fetchone() + if not ( + row + and row['status'] == 'in_progress' + and row['lease_owner'] == claim.get('lease_owner') + and row['lease_token'] == claim.get('lease_token') + and row['claim_batch'] == claim.get('claim_batch') + and str(claim.get('claim_batch') or '') == str((reservation or {}).get('reservation_id') or '') + and str(claim.get('lease_owner') or '') == str((reservation or {}).get('owner') or '') + ): + self.conn.rollback() + return False + rows[queue_id] = row + now = utc_now_iso() + for queue_id, row in rows.items(): + cur = self.conn.execute( + '''UPDATE target_queue SET status = 'pending', + attempts = CASE WHEN COALESCE(attempts, 0) > 0 THEN attempts - 1 ELSE 0 END, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, + leased_at = NULL, lease_expires_at = NULL, + available_after = NULL, completed_at = NULL, last_error = ?, updated_at = ? + WHERE id = ? AND status = 'in_progress' + AND lease_owner = ? AND lease_token = ? AND claim_batch = ?''', + ( + first_line(error, 500), now, queue_id, + row['lease_owner'], row['lease_token'], row['claim_batch'], + ), + ) + if int(getattr(cur, 'rowcount', 0) or 0) != 1: + self.conn.rollback() + return False + self.conn.commit() + return True + except Exception: + try: + self.conn.rollback() + except Exception: + pass + raise + + def claim_docker_resolutions( + self, source, limit, lease_owner, lease_seconds=300, periodic_limit=0, + periodic_only=False, allowed_queries=None, + ): + if not self.conn or not source or not lease_owner: + return [] + query_values = None + if allowed_queries is not None: + query_values = [] + for value in allowed_queries: + query = str(value or '').strip() + if not query: + raise ValueError('Docker resolver allowed query is empty') + if query not in query_values: + query_values.append(query) + + def op(): + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + state = self._locked_runtime_control_state(shared=True) + if state['effective_discovery_paused']: + self.conn.commit() + return [] + now = utc_now_iso() + lease_until = datetime.fromtimestamp( + time.time() + max(60, int(lease_seconds or 300)), timezone.utc + ).isoformat(timespec='seconds') + lock_suffix = ' FOR UPDATE SKIP LOCKED' if self.conn.is_postgres else '' + total_limit = max(1, int(limit or 1)) + query_clause = '' + query_params = () + if query_values is not None: + if not query_values: + self.conn.commit() + return [] + query_clause = ' AND query IN (' + ','.join('?' for _ in query_values) + ')' + query_params = tuple(query_values) + retry_rows = [] if periodic_only else self.conn.execute( + f'''SELECT id, target, COALESCE(resolver_attempts, 0) AS resolver_attempts_before + FROM target_queue + WHERE source = ? AND platform = 'docker' AND status = 'deferred' + AND resolver_state IN ('pending', 'retry', 'resolving') + AND resolver_due_at IS NOT NULL AND resolver_due_at <= ?{query_clause} + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_repositories member + JOIN docker_depth_experiments experiment + ON experiment.id = member.experiment_id + WHERE (member.repository_queue_id = target_queue.id + OR member.replacement_repository_queue_id = target_queue.id) + AND experiment.state IN ( + 'planned','holding','resolving','active','draining','held' + ) + ) + ORDER BY resolver_due_at, id LIMIT ?{lock_suffix}''', + (source, now, *query_params, total_limit), + ).fetchall() + rows = [(row, False) for row in retry_rows] + periodic_count = min( + max(0, int(periodic_limit or 0)), + max(0, total_limit - len(rows)), + ) + if periodic_count: + periodic_rows = self.conn.execute( + f'''SELECT id, target, COALESCE(resolver_attempts, 0) AS resolver_attempts_before + FROM target_queue + WHERE source = ? AND platform = 'docker' AND status = 'done' + AND resolver_state = 'resolved' AND resolver_token IS NULL + AND (resolver_due_at IS NULL OR resolver_due_at <= ?){query_clause} + AND NOT EXISTS ( + SELECT 1 FROM docker_depth_experiment_repositories member + JOIN docker_depth_experiments experiment + ON experiment.id = member.experiment_id + WHERE (member.repository_queue_id = target_queue.id + OR member.replacement_repository_queue_id = target_queue.id) + AND experiment.state IN ( + 'planned','holding','resolving','active','draining','held' + ) + ) + ORDER BY COALESCE(resolver_due_at, ''), id + LIMIT ?{lock_suffix}''', + (source, now, *query_params, periodic_count), + ).fetchall() + rows.extend((row, True) for row in periodic_rows) + output = [] + for row, periodic in rows: + token = secrets.token_urlsafe(32) + cur = self.conn.execute( + '''UPDATE target_queue SET status = 'deferred', resolver_state = 'resolving', + resolver_due_at = ?, available_after = ?, + resolver_token = ?, resolver_attempts = COALESCE(resolver_attempts, 0) + 1, + updated_at = ? WHERE id = ?''', + (lease_until, lease_until, token, now, row['id']), + ) + if int(getattr(cur, 'rowcount', 0) or 0) != 1: + self.conn.rollback() + return [] + attempts_before = int(row['resolver_attempts_before'] or 0) + output.append({ + 'id': row['id'], + 'target': row['target'], + 'resolver_token': token, + 'resolver_attempts_before': attempts_before, + 'resolver_attempts': attempts_before + 1, + 'periodic': periodic, + }) + self.conn.commit() + return output + return self._safe('claim_docker_resolutions', op, []) + + def finish_docker_resolution( + self, source, queue_id, resolver_token, tagged_targets=None, error='', complete=None, + *, retry_at=None, claim_attempt_consumed=True, refresh_interval_sec=0, + ): + if not self.conn or queue_id is None or not resolver_token: + return False + tagged_targets = list(tagged_targets or []) + complete = bool(tagged_targets) if complete is None else bool(complete) + if not claim_attempt_consumed and (complete or not retry_at): + return False + + def op(): + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + if tagged_targets: + self._require_discovery_admission_locked() + now = utc_now_iso() + now_dt = datetime.now(timezone.utc) + retry_due = None + if retry_at: + try: + retry_dt = datetime.fromisoformat(str(retry_at).replace('Z', '+00:00')) + if retry_dt.tzinfo is None: + retry_dt = retry_dt.replace(tzinfo=timezone.utc) + retry_dt = retry_dt.astimezone(timezone.utc) + except (TypeError, ValueError): + self.conn.rollback() + return False + if retry_dt > now_dt + timedelta(seconds=3605): + self.conn.rollback() + return False + if retry_dt < now_dt: + retry_dt = now_dt + retry_due = retry_dt.isoformat(timespec='seconds') + lock_suffix = ' FOR UPDATE' if self.conn.is_postgres else '' + row = self.conn.execute( + f'''SELECT query, resolver_attempts FROM target_queue + WHERE id = ? AND source = ? AND platform = 'docker' + AND resolver_state = 'resolving' AND resolver_token = ?{lock_suffix}''', + (queue_id, source, resolver_token), + ).fetchone() + if not row: + self.conn.rollback() + return False + if tagged_targets: + for target in tagged_targets: + normalized = normalize_target(target, 'docker') + if not normalized: + continue + self.conn.execute( + '''INSERT INTO target_queue ( + source, platform, query, target, normalized_target, status, + created_at, updated_at + ) VALUES (?, 'docker', ?, ?, ?, 'pending', ?, ?) + ON CONFLICT(source, normalized_target) DO NOTHING''', + (source, row['query'], target, normalized, now, now), + ) + if complete: + refresh_interval = max(0, int(refresh_interval_sec or 0)) + next_due = ( + datetime.fromtimestamp( + time.time() + refresh_interval, timezone.utc, + ).isoformat(timespec='seconds') + if refresh_interval else None + ) + cur = self.conn.execute( + '''UPDATE target_queue SET status = 'done', resolver_state = 'resolved', + resolver_due_at = ?, resolver_token = NULL, resolver_attempts = 0, + available_after = NULL, last_error = NULL, + completed_at = COALESCE(completed_at, ?), updated_at = ? + WHERE id = ? AND resolver_token = ?''', + (next_due, now, now, queue_id, resolver_token), + ) + elif retry_due is not None: + cur = self.conn.execute( + '''UPDATE target_queue SET status = 'deferred', resolver_state = 'retry', + resolver_due_at = ?, resolver_token = NULL, available_after = ?, + resolver_attempts = CASE + WHEN ? = 0 AND COALESCE(resolver_attempts, 0) > 0 + THEN resolver_attempts - 1 ELSE COALESCE(resolver_attempts, 0) END, + last_error = ?, updated_at = ? + WHERE id = ? AND resolver_token = ?''', + ( + retry_due, retry_due, 1 if claim_attempt_consumed else 0, + first_line(error or 'Docker tag resolution deferred', 500), + now, queue_id, resolver_token, + ), + ) + else: + base_delay = max(60, env_int('DOCKER_RESOLVER_RETRY_SEC', 3600)) + max_delay = max(base_delay, env_int('DOCKER_RESOLVER_RETRY_MAX_SEC', 86400)) + exponent = min(16, max(0, int(row['resolver_attempts'] or 1) - 1)) + delay = min(max_delay, base_delay * (2 ** exponent)) + due = datetime.fromtimestamp(time.time() + delay, timezone.utc).isoformat(timespec='seconds') + cur = self.conn.execute( + '''UPDATE target_queue SET status = 'deferred', resolver_state = 'retry', + resolver_due_at = ?, resolver_token = NULL, available_after = ?, + last_error = ?, updated_at = ? WHERE id = ? AND resolver_token = ?''', + ( + due, due, + first_line(error or 'Docker tag resolution unresolved', 500), + now, queue_id, resolver_token, + ), + ) + if int(getattr(cur, 'rowcount', 0) or 0) != 1: + self.conn.rollback() + return False + self.conn.commit() + return True + return self._safe('finish_docker_resolution', op, False) + + def requeue_target_ids(self, queue_ids): + ids = [int(value) for value in (queue_ids or []) if value is not None] + if not self.conn or not ids: + return 0 + + def op(): + now = utc_now_iso() + placeholders = ','.join('?' for _ in ids) + cur = self.conn.execute( + f'''UPDATE target_queue SET status = 'pending', attempts = 0, available_after = NULL, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, leased_at = NULL, lease_expires_at = NULL, + last_error = NULL, completed_at = NULL, updated_at = ? + WHERE id IN ({placeholders}) AND status = 'deferred' ''', + [now, *ids], + ) + self.conn.commit() + return int(getattr(cur, 'rowcount', 0) or 0) + return self._safe('requeue_target_ids', op, 0) + + def target_queue_item(self, source, platform, target): + if not self.conn: + return None + + def op(): + normalized = normalize_target(target, platform) + return self.conn.execute( + '''SELECT id, status, attempts, available_after, lease_owner, lease_token, lease_expires_at + FROM target_queue WHERE source = ? AND normalized_target = ?''', + (source, normalized), + ).fetchone() + return self._safe('target_queue_item', op) + + def target_queue_counts(self, source=None): + if not self.conn: + return { + 'counts': {}, 'degraded': True, 'stale': True, + 'reason': 'database_unavailable', 'truncated_statuses': [], + } + cache_key = str(source or '*') + cache = getattr(self, '_target_queue_counts_cache', None) + if cache is None: + cache = {} + self._target_queue_counts_cache = cache + retry_after_by_key = getattr(self, '_target_queue_counts_retry_after', None) + if retry_after_by_key is None: + retry_after_by_key = {} + self._target_queue_counts_retry_after = retry_after_by_key + now = time.monotonic() + retry_after = float(retry_after_by_key.get(cache_key) or 0) + if retry_after > now: + cached = cache.get(cache_key) + snapshot = cached or { + 'counts': {}, + 'sample_limit_per_status': TARGET_QUEUE_OBSERVABILITY_STATUS_LIMIT, + 'timeout_ms': TARGET_QUEUE_OBSERVABILITY_TIMEOUT_MS, + 'sampled_rows': 0, + 'truncated_statuses': [], + } + return dict( + snapshot, + counts=dict(snapshot.get('counts') or {}), + degraded=True, + stale=True, + reason='bounded_count_retry_backoff', + retry_after_sec=max(1, min( + TARGET_QUEUE_OBSERVABILITY_RETRY_BACKOFF_SEC, + int(math.ceil(retry_after - now)), + )), + truncated_statuses=list(snapshot.get('truncated_statuses') or []), + ) + try: + counts = {} + truncated = [] + if self.conn.is_postgres: + self.conn.execute( + f'SET LOCAL statement_timeout = {TARGET_QUEUE_OBSERVABILITY_TIMEOUT_MS}' + ) + for status in TARGET_QUEUE_OBSERVABILITY_STATUSES: + if source: + row = self.conn.execute( + '''SELECT COUNT(*) AS count FROM ( + SELECT 1 FROM target_queue + WHERE source = ? AND status = ? LIMIT ? + ) AS sampled''', + (source, status, TARGET_QUEUE_OBSERVABILITY_STATUS_LIMIT), + ).fetchone() + else: + row = self.conn.execute( + '''SELECT COUNT(*) AS count FROM ( + SELECT 1 FROM target_queue + WHERE status = ? LIMIT ? + ) AS sampled''', + (status, TARGET_QUEUE_OBSERVABILITY_STATUS_LIMIT), + ).fetchone() + count = int(row['count'] or 0) + if count: + counts[status] = count + if count >= TARGET_QUEUE_OBSERVABILITY_STATUS_LIMIT: + truncated.append(status) + if self.conn.is_postgres: + self.conn.commit() + snapshot = { + 'counts': counts, + 'degraded': bool(truncated), + 'stale': False, + 'reason': 'bounded_status_sample' if truncated else '', + 'sample_limit_per_status': TARGET_QUEUE_OBSERVABILITY_STATUS_LIMIT, + 'timeout_ms': TARGET_QUEUE_OBSERVABILITY_TIMEOUT_MS, + 'sampled_rows': sum(counts.values()), + 'retry_after_sec': 0, + 'truncated_statuses': truncated, + } + cache[cache_key] = snapshot + retry_after_by_key.pop(cache_key, None) + self.last_error = '' + return dict(snapshot, counts=dict(counts), truncated_statuses=list(truncated)) + except Exception as exc: + self.last_error = str(exc) + rollback_succeeded = False + try: + self.conn.rollback() + rollback_succeeded = True + except Exception: + pass + if rollback_succeeded: + retry_after_by_key[cache_key] = ( + time.monotonic() + TARGET_QUEUE_OBSERVABILITY_RETRY_BACKOFF_SEC + ) + cached = cache.get(cache_key) + logger.warning('Target queue observability degraded after bounded count query failure: %s', type(exc).__name__) + if cached: + return dict( + cached, + counts=dict(cached.get('counts') or {}), + degraded=True, + stale=True, + reason='bounded_count_query_failed', + retry_after_sec=( + TARGET_QUEUE_OBSERVABILITY_RETRY_BACKOFF_SEC if rollback_succeeded else 0 + ), + truncated_statuses=list(cached.get('truncated_statuses') or []), + ) + return { + 'counts': {}, + 'degraded': True, + 'stale': True, + 'reason': 'bounded_count_query_failed', + 'sample_limit_per_status': TARGET_QUEUE_OBSERVABILITY_STATUS_LIMIT, + 'timeout_ms': TARGET_QUEUE_OBSERVABILITY_TIMEOUT_MS, + 'sampled_rows': 0, + 'retry_after_sec': ( + TARGET_QUEUE_OBSERVABILITY_RETRY_BACKOFF_SEC if rollback_succeeded else 0 + ), + 'truncated_statuses': [], + } + + def admin_target_queue_health(self, sources): + if ( + not isinstance(sources, (tuple, list)) or not 1 <= len(sources) <= 16 + or any(type(source) is not str or not source for source in sources) + or len(set(sources)) != len(sources) + ): + raise ValueError('admin queue sources are invalid') + snapshots = [self.target_queue_counts(source) for source in sources] + counts = {} + truncated = set() + for snapshot in snapshots: + for status, count in (snapshot.get('counts') or {}).items(): + counts[status] = counts.get(status, 0) + int(count) + truncated.update(snapshot.get('truncated_statuses') or ()) + degraded = any(snapshot.get('degraded') is True for snapshot in snapshots) + stale = any(snapshot.get('stale') is True for snapshot in snapshots) + return { + 'counts': counts, + 'degraded': degraded, + 'stale': stale, + 'reason': 'bounded_core_source_sample' if degraded else '', + 'sample_limit_per_status': ( + TARGET_QUEUE_OBSERVABILITY_STATUS_LIMIT * len(sources) + ), + 'timeout_ms': TARGET_QUEUE_OBSERVABILITY_TIMEOUT_MS, + 'sampled_rows': sum(counts.values()), + 'retry_after_sec': max( + (int(snapshot.get('retry_after_sec') or 0) for snapshot in snapshots), + default=0, + ), + 'truncated_statuses': sorted(truncated), + } + + def known_target_normalizations(self, source, platform=None): + if not self.conn: + return set() + + def op(): + if platform: + rows = self.conn.execute( + '''SELECT normalized_target FROM target_queue + WHERE source = ? AND platform = ?''', + (source, platform), + ).fetchall() + else: + rows = self.conn.execute( + 'SELECT normalized_target FROM target_queue WHERE source = ?', + (source,), + ).fetchall() + return {str(row['normalized_target']) for row in rows if row['normalized_target']} + return self._safe('known_target_normalizations', op, set()) + + def known_target_normalizations_for(self, source, platform, targets, batch_size=KNOWN_TARGET_LOOKUP_BATCH_SIZE): + """Return only known identities from the offered bounded discovery batch.""" + if not self.conn: + return set() + normalized = [] + seen = set() + for target in targets or []: + value = normalize_target(target, platform) + if value and value not in seen: + seen.add(value) + normalized.append(value) + if not normalized: + return set() + + def op(): + known = set() + size = max(1, min(KNOWN_TARGET_LOOKUP_BATCH_SIZE, int(batch_size or KNOWN_TARGET_LOOKUP_BATCH_SIZE))) + for start in range(0, len(normalized), size): + values = normalized[start:start + size] + rows = self.conn.execute( + '''SELECT normalized_target FROM target_queue + WHERE source = ? AND platform = ? AND normalized_target IN ({})'''.format( + ','.join('?' for _ in values) + ), + (source, platform, *values), + ).fetchall() + known.update(str(row['normalized_target']) for row in rows if row['normalized_target']) + return known + return self._safe('known_target_normalizations_for', op, set()) + + def record_package_repo_candidates(self, run_id, cycle_id, query, candidates): + offered = list(candidates or []) + report = { + 'offered_rows': len(offered), + 'committed_rows': 0, + 'skipped_rows': 0, + 'failed': False, + 'failed_chunk_rows': 0, + 'connection_usable': bool(self.conn), + } + if not offered or not self.conn: + report['skipped_rows'] = len(offered) + report['failed'] = bool(offered) + return report + + for start in range(0, len(offered), PACKAGE_CANDIDATE_WRITE_CHUNK_SIZE): + chunk = offered[start:start + PACKAGE_CANDIDATE_WRITE_CHUNK_SIZE] + attempts = SQLITE_LOCK_RETRY_ATTEMPTS if self.conn.is_sqlite else 1 + failure = None + for attempt in range(attempts): + try: + now = utc_now_iso() + for candidate in chunk: + self.conn.execute( + '''INSERT INTO package_repo_candidates ( + package_source, package_name, package_version, query, repo_url, provider, + evidence_json, confidence, first_seen_at, last_seen_at, last_run_id, last_cycle_id + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(package_source, package_name, package_version, repo_url) + DO UPDATE SET + query = excluded.query, + provider = excluded.provider, + evidence_json = excluded.evidence_json, + confidence = excluded.confidence, + last_seen_at = excluded.last_seen_at, + last_run_id = excluded.last_run_id, + last_cycle_id = excluded.last_cycle_id''', + ( + candidate.get('package_source'), + candidate.get('name') or candidate.get('package_name'), + candidate.get('version') or candidate.get('package_version') or '', + query, + candidate.get('repo_url'), + candidate.get('provider'), + json_dumps(candidate.get('evidence') or []), + candidate.get('confidence') or 'medium', + now, + now, + run_id, + cycle_id, + ), + ) + self.conn.commit() + report['committed_rows'] += len(chunk) + failure = None + break + except Exception as exc: + failure = exc + try: + self.conn.rollback() + except Exception: + report['connection_usable'] = self._reset_connection() + if ( + self.conn + and self.conn.is_sqlite + and sqlite_lock_error(exc) + and attempt < attempts - 1 + ): + time.sleep(sqlite_lock_retry_delay(attempt)) + continue + break + if failure is not None: + self.last_error = str(failure) + report.update({ + 'skipped_rows': len(offered) - report['committed_rows'], + 'failed': True, + 'failed_chunk_rows': len(chunk), + 'connection_usable': bool(self.conn) and report['connection_usable'], + }) + logger.warning( + 'Optional package candidate cache write stopped after a failed bounded chunk: ' + 'committed=%s skipped=%s chunk_rows=%s error=%s', + report['committed_rows'], report['skipped_rows'], len(chunk), type(failure).__name__, + ) + return report + + self.last_error = '' + return report + + def _bounded_package_candidate_rows(self, query, sources, limit): + scan_limit = min( + PACKAGE_CANDIDATE_SCAN_MAX_ROWS, + max(5000, int(limit) * 4), + ) + source_set = set(sources) + exact = [] + if query: + scope = PACKAGE_REPO_NONEMPTY_PREDICATE + source_params = [] + if sources: + scope += ' AND package_source IN ({})'.format(','.join('?' for _ in sources)) + source_params.extend(sources) + try: + if self.conn.is_postgres: + self.conn.execute('SET LOCAL enable_seqscan = off') + exact = self.conn.execute( + f'''SELECT id, package_source, package_name AS name, + package_version AS version, repo_url, provider, + evidence_json, confidence, last_seen_at + FROM package_repo_candidates + WHERE {scope} AND query = ? + ORDER BY last_seen_at DESC, id DESC LIMIT ?''', + (*source_params, query, limit), + ).fetchall() + except Exception as exc: + if self.conn.is_postgres: + self.conn.rollback() + logger.warning( + 'Exact package candidate branch failed; bounded substring results remain available: %s', + type(exc).__name__, + ) + matched_ids = [] + examined = 0 + cursor_seen = None + cursor_id = None + more_rows = False + while examined < scan_limit and len(matched_ids) < limit: + page_limit = min( + PACKAGE_CANDIDATE_SCAN_PAGE_SIZE, + scan_limit - examined, + ) + cursor_clause = '' + params = [] + if cursor_seen is not None and cursor_id is not None: + cursor_clause = 'AND (last_seen_at, id) < (?, ?)' + params.extend((cursor_seen, cursor_id)) + params.append(page_limit) + page = self.conn.execute( + f'''SELECT id, package_source, package_name, query, last_seen_at + FROM package_repo_candidates + WHERE {PACKAGE_REPO_NONEMPTY_PREDICATE} {cursor_clause} + ORDER BY last_seen_at DESC, id DESC + LIMIT ?''', + params, + ).fetchall() + if not page: + more_rows = False + break + examined += len(page) + for row in page: + if source_set and str(row['package_source'] or '').lower() not in source_set: + continue + if not query or ( + str(row['query'] or '') != query + and query in str(row['package_name'] or '') + ): + matched_ids.append(int(row['id'])) + if len(matched_ids) >= limit: + break + cursor_seen = page[-1]['last_seen_at'] + cursor_id = page[-1]['id'] + more_rows = len(page) == page_limit + if len(page) < page_limit: + break + + if len(matched_ids) < limit and examined >= scan_limit and more_rows: + probe = self.conn.execute( + f'''SELECT 1 FROM package_repo_candidates + WHERE {PACKAGE_REPO_NONEMPTY_PREDICATE} + AND (last_seen_at, id) < (?, ?) + ORDER BY last_seen_at DESC, id DESC + LIMIT 1''', + (cursor_seen, cursor_id), + ).fetchone() + if probe: + logger.warning( + 'Package candidate substring lookup reached its bounded %s-row metadata scan; ' + 'older candidates were not examined', + scan_limit, + ) + + rows_by_id = {} + for start in range(0, len(matched_ids), 64): + batch = matched_ids[start:start + 64] + placeholders = ','.join('?' for _ in batch) + rows = self.conn.execute( + f'''SELECT id, package_source, package_name AS name, package_version AS version, + repo_url, provider, evidence_json, confidence, last_seen_at + FROM package_repo_candidates WHERE id IN ({placeholders})''', + batch, + ).fetchall() + rows_by_id.update({int(row['id']): row for row in rows}) + combined = { + int(row['id']): row + for row in (*exact, *(rows_by_id[row_id] for row_id in matched_ids if row_id in rows_by_id)) + } + return sorted( + combined.values(), + key=lambda row: (str(row['last_seen_at'] or ''), int(row['id'])), + reverse=True, + )[:limit] + + def get_package_repo_candidates(self, query=None, package_sources=None, limit=1000): + if not self.conn: + return [] + def op(): + requested_limit = max(1, int(limit or 1000)) + limit_value = min(PACKAGE_CANDIDATE_LOOKUP_MAX_RESULTS, requested_limit) + if requested_limit > limit_value: + logger.warning( + 'Package candidate lookup capped requested result limit from %s to %s', + requested_limit, limit_value, + ) + sources = [item.strip().lower() for item in (package_sources or []) if item.strip()] + search = str(query or '') + rows = self._bounded_package_candidate_rows( + search, sources, limit_value, + ) + candidates = [] + for row in rows: + try: + evidence = json.loads(row['evidence_json'] or '[]') + except json.JSONDecodeError: + evidence = [] + candidates.append({ + 'source': 'package_git', + 'package_source': row['package_source'], + 'name': row['name'], + 'version': row['version'], + 'repo_url': row['repo_url'], + 'provider': row['provider'], + 'evidence': evidence, + 'confidence': row['confidence'], + }) + return candidates + return self._safe('get_package_repo_candidates', op, []) + + @staticmethod + def _compat_payload(finding, max_bytes=16 * 1024 * 1024): + mapped = { + 'DetectorName', 'DetectorType', 'Verified', 'Raw', 'RawV2', 'Redacted', + 'StructuredData', 'ExtraData', 'AnalysisInfo', 'finding_uid', + } + extension = {key: value for key, value in finding.items() if key not in mapped} + payload = { + 'raw_value': finding.get('Raw'), + 'raw_v2_value': finding.get('RawV2'), + 'structured_data_json': json_dumps(finding.get('StructuredData')) if finding.get('StructuredData') is not None else None, + 'extra_data_json': json_dumps(finding.get('ExtraData')) if finding.get('ExtraData') is not None else None, + 'analysis_info_json': json_dumps(finding.get('AnalysisInfo')) if finding.get('AnalysisInfo') is not None else None, + 'extension_json': json_dumps(extension) if extension else None, + } + encoded = json.dumps( + finding, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8') + omitted = len(encoded) > max(1, int(max_bytes)) + if omitted: + payload = {key: None for key in payload} + payload.update({ + 'payload_sha256': hashlib.sha256(encoded).hexdigest(), + 'payload_bytes': len(encoded), + 'payload_omitted': 1 if omitted else 0, + }) + return payload + + def _insert_normalized_finding( + self, run_id, cycle_id, target_scan_id, source, query, target, + normalized, finding, + ): + finding = sanitize_postman_finding(finding) + raw_secret, secret_hash, detector_secret_hash, fingerprint, location = finding_identity( + source, normalized, finding, + ) + finding_uid = str(finding.get('finding_uid') or fingerprint) + enrichment = enrich_finding(finding) + compat = self._compat_payload(finding) + finding_id = self.conn.insert_returning_id( + '''INSERT INTO findings( + run_id, cycle_id, target_scan_id, source, query, target, normalized_target, + detector_name, detector_type, verified, raw_secret, redacted_secret, secret_hash, + detector_secret_hash, finding_fingerprint, finding_uid, file_path, line_number, + commit_hash, source_timestamp, source_metadata_type, source_metadata_json, + raw_finding_json, provider, credential_kind, credential_confidence, + required_context_missing, principal, username, email, project_id, tenant_id, + organization, registry, endpoint, scope, resource, enrichment_json, + raw_payload_sha256, raw_payload_bytes, raw_payload_omitted, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, + NULL, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + run_id, cycle_id, target_scan_id, source, query, target, normalized, + finding.get('DetectorName'), finding.get('DetectorType'), + 1 if finding.get('Verified', False) else 0, raw_secret, + extract_redacted_secret(finding), secret_hash, detector_secret_hash, + fingerprint, finding_uid, location.get('file_path'), location.get('line_number'), + location.get('commit_hash'), location.get('source_timestamp'), + location.get('source_metadata_type'), location.get('source_metadata_json'), + enrichment.get('provider'), enrichment.get('credential_kind'), + enrichment.get('credential_confidence'), enrichment.get('required_context_missing'), + enrichment.get('principal'), enrichment.get('username'), enrichment.get('email'), + enrichment.get('project_id'), enrichment.get('tenant_id'), + enrichment.get('organization'), enrichment.get('registry'), + enrichment.get('endpoint'), enrichment.get('scope'), enrichment.get('resource'), + enrichment.get('enrichment_json'), compat['payload_sha256'], + compat['payload_bytes'], compat['payload_omitted'], utc_now_iso(), + ), + ) + if not finding_id: + raise RuntimeError('normalized finding insert returned no identity') + self.conn.execute( + '''INSERT INTO finding_compat_payloads( + finding_id, raw_value, raw_v2_value, structured_data_json, + extra_data_json, analysis_info_json, extension_json, + payload_sha256, payload_bytes, payload_omitted, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + finding_id, compat['raw_value'], compat['raw_v2_value'], + compat['structured_data_json'], compat['extra_data_json'], + compat['analysis_info_json'], compat['extension_json'], + compat['payload_sha256'], compat['payload_bytes'], compat['payload_omitted'], + utc_now_iso(), + ), + ) + self.conn.execute( + '''INSERT INTO finding_uid_map(finding_uid, finding_id, created_at) + VALUES (?, ?, ?) ON CONFLICT(finding_uid) DO NOTHING''', + (finding_uid, finding_id, utc_now_iso()), + ) + return finding_id, finding_uid + + @staticmethod + def _docker_finding_layer_evidence(finding, manifest_digest): + metadata = finding.get('SourceMetadata') + data = metadata.get('Data') if isinstance(metadata, dict) else None + if not isinstance(data, dict): + return None, 'digest_absent' + + docker = data.get('Docker') + if 'Docker' in data and not isinstance(docker, dict): + return None, 'digest_invalid' + if isinstance(docker, dict) and 'layer' in docker: + digest = docker.get('layer') + if not isinstance(digest, str) or not re.fullmatch( + r'sha256:[0-9a-f]{64}', digest): + return None, 'digest_invalid' + return digest, None + + if 'DockerContent' not in data: + return None, 'digest_absent' + docker_content = data.get('DockerContent') + if not isinstance(docker_content, dict): + return None, 'digest_invalid' + if docker_content.get('descriptor_kind') != 'layer': + return None, 'digest_absent' + digest = docker_content.get('blob_digest') + reported_manifest = docker_content.get('manifest_digest') + if ( + not isinstance(digest, str) + or not re.fullmatch(r'sha256:[0-9a-f]{64}', digest) + or not isinstance(reported_manifest, str) + or not re.fullmatch(r'sha256:[0-9a-f]{64}', reported_manifest) + ): + return None, 'digest_invalid' + if reported_manifest != manifest_digest: + return None, 'manifest_mismatch' + return digest, None + + def _insert_docker_finding_layer_attributions_locked( + self, binding, target_scan_id, finding_id, finding, + ): + if not binding: + return 0 + if binding['target_scan_id'] is not None and int( + binding['target_scan_id']) != int(target_scan_id): + raise ScanEventConflictError( + 'Docker finding attribution binding has a conflicting scan identity' + ) + finding_row = self.conn.execute( + '''SELECT finding.id + FROM findings finding + JOIN target_scans scan ON scan.id = finding.target_scan_id + WHERE finding.id = ? AND finding.target_scan_id = ? + AND scan.queue_id = ? AND scan.result_reservation_id = ? + FOR UPDATE OF finding, scan''', + ( + int(finding_id), int(target_scan_id), + int(binding['target_queue_id']), int(binding['reservation_id']), + ), + ).fetchone() + if not finding_row: + raise ScanEventConflictError( + 'Docker finding attribution lost its exact reservation scan identity' + ) + + digest, reason = self._docker_finding_layer_evidence( + finding, str(binding['manifest_digest']), + ) + layers = [] + if digest is not None: + layers = self.conn.execute( + '''SELECT id, position_from_base, position_from_top, layer_digest + FROM docker_manifest_layers + WHERE manifest_id = ? AND layer_digest = ? + ORDER BY position_from_base, id FOR UPDATE''', + (int(binding['manifest_id']), digest), + ).fetchall() + if not layers: + digest = None + reason = 'digest_not_in_manifest' + + if digest is None: + expected_rows = (( + int(binding['id']), None, 'unattributed', None, reason, None, None, + ),) + else: + expected_rows = tuple( + ( + int(binding['id']), int(layer['id']), 'exact', + str(layer['layer_digest']), None, + int(layer['position_from_base']), int(layer['position_from_top']), + ) + for layer in layers + ) + + now = utc_now_iso() + for row in expected_rows: + self.conn.execute( + '''INSERT INTO docker_finding_layer_attributions( + scan_binding_id, finding_id, manifest_layer_id, + attribution_state, reported_layer_digest, + unattributed_reason, position_from_base, + position_from_top, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT DO NOTHING''', + (row[0], int(finding_id), *row[1:], now), + ) + actual_rows = self.conn.execute( + '''SELECT scan_binding_id, manifest_layer_id, attribution_state, + reported_layer_digest, unattributed_reason, + position_from_base, position_from_top + FROM docker_finding_layer_attributions + WHERE finding_id = ? ORDER BY id FOR UPDATE''', + (int(finding_id),), + ).fetchall() + actual = { + ( + int(row['scan_binding_id']), + int(row['manifest_layer_id']) if row['manifest_layer_id'] is not None else None, + str(row['attribution_state']), row['reported_layer_digest'], + row['unattributed_reason'], + int(row['position_from_base']) if row['position_from_base'] is not None else None, + int(row['position_from_top']) if row['position_from_top'] is not None else None, + ) + for row in actual_rows + } + expected = set(expected_rows) + if len(actual_rows) != len(expected_rows) or actual != expected: + raise ScanEventConflictError( + 'Docker finding layer attribution conflicts with persisted evidence' + ) + return len(expected_rows) + + def _precompute_docker_finding_attribution_rows_locked(self, reader, binding): + if not binding: + return 0 + layers = self.conn.execute( + '''SELECT id, layer_digest, position_from_base, position_from_top + FROM docker_manifest_layers + WHERE manifest_id = ? + ORDER BY position_from_base, id FOR UPDATE''', + (int(binding['manifest_id']),), + ).fetchall() + if len(layers) != int(binding['layer_count']): + raise ScanEventConflictError( + 'Docker attribution manifest layer authority is incomplete' + ) + positions_by_digest = {} + for layer in layers: + positions_by_digest[str(layer['layer_digest'])] = ( + positions_by_digest.get(str(layer['layer_digest']), 0) + 1 + ) + + projected_rows = 0 + finding_count = 0 + for finding in reader.iter_findings(): + finding_count += 1 + digest, _reason = self._docker_finding_layer_evidence( + finding, str(binding['manifest_digest']), + ) + projected_rows += max(1, positions_by_digest.get(digest, 0)) + if projected_rows > DOCKER_DEPTH_MAX_ATTRIBUTION_ROWS_PER_BUNDLE: + raise DockerFindingAttributionLimitError( + 'Docker finding attribution exceeds its strict per-bundle row limit' + ) + if finding_count != int(reader.validate().finding_count): + raise ScanEventConflictError( + 'Docker attribution finding count conflicts with its bundle' + ) + return projected_rows + + def _existing_normalized_finding_for_replay_locked( + self, target_scan_id, source, normalized, finding, + ): + finding = sanitize_postman_finding(finding) + _, _, _, fingerprint, _ = finding_identity(source, normalized, finding) + finding_uid = str(finding.get('finding_uid') or fingerprint) + payload = self._compat_payload(finding) + rows = self.conn.execute( + '''SELECT id, finding_fingerprint, raw_payload_sha256 + FROM findings + WHERE target_scan_id = ? AND finding_uid = ? + ORDER BY id LIMIT 2 FOR UPDATE''', + (int(target_scan_id), finding_uid), + ).fetchall() + if len(rows) != 1 or ( + str(rows[0]['finding_fingerprint']) != fingerprint + or str(rows[0]['raw_payload_sha256']) != payload['payload_sha256'] + ): + raise ScanEventConflictError( + 'replayed Docker finding identity conflicts with persisted evidence' + ) + return int(rows[0]['id']), finding_uid + + def _insert_bundle_candidate( + self, frame, reservation, target_scan_id, finding_ids, + capacity_allocation=None, routed_service_supported=True, + ): + if not isinstance(frame, dict): + raise ValueError('keycheck candidate frame must be an object') + service = str(frame.get('service') or '').lower() + credential_hash = str(frame.get('credential_hash') or '').lower() + provider_key_hash = str(frame.get('provider_key_hash') or '').lower() + secret_hash = str(frame.get('secret_hash') or '').lower() + if ( + not service + or not re.fullmatch(r'[a-f0-9]{64}', credential_hash) + or not re.fullmatch(r'[a-f0-9]{64}', provider_key_hash) + or not re.fullmatch(r'[a-f0-9]{64}', secret_hash) + ): + raise ValueError('keycheck candidate identity is invalid') + secret_text = frame.get('secret_text') + secret_json = frame.get('secret_json') + if (secret_text is None) == (secret_json is None): + raise ValueError('keycheck candidate must contain exactly one secret representation') + metadata = frame.get('metadata') if isinstance(frame.get('metadata'), dict) else {} + attribution = frame.get('attribution') if isinstance(frame.get('attribution'), dict) else {} + now = utc_now_iso() + credential_id = self.conn.insert_returning_id( + '''INSERT INTO keycheck_credentials( + service, credential_hash, provider_key_hash, + candidate_kind, secret_text, secret_json, + key_masked, endpoint, principal, metadata_json, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(service, provider_key_hash) DO NOTHING''', + ( + service, credential_hash, provider_key_hash, + str(frame.get('candidate_kind') or 'raw'), + secret_text, secret_json, str(frame.get('key_masked') or ''), + str(frame.get('endpoint') or ''), str(frame.get('principal') or ''), + json_dumps(metadata), now, now, + ), + ) + if credential_id is None: + credential = self.conn.execute( + '''SELECT id, provider_key_hash, secret_text, secret_json + FROM keycheck_credentials + WHERE service = ? AND provider_key_hash = ?''', + (service, provider_key_hash), + ).fetchone() + if not credential or ( + credential['provider_key_hash'] != provider_key_hash + or credential['secret_text'] != secret_text + or credential['secret_json'] != secret_json + ): + raise ScanEventConflictError('provider credential identity resolves to conflicting secret material') + credential_id = credential['id'] + finding_uid_value = str(attribution.get('finding_uid') or metadata.get('finding_uid') or '') + finding_id = finding_ids.get(finding_uid_value) + origin = finding_uid_value or str(attribution.get('origin') or frame.get('candidate_kind') or 'structured') + uid = candidate_uid(reservation['scan_event_id'], origin, service, credential_hash) + serialized_bytes = len(json.dumps( + frame, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8')) + capacity_bytes = max(serialized_bytes, int(capacity_allocation or serialized_bytes)) + if routed_service_supported: + candidate_sql = '''INSERT INTO keycheck_candidates( + candidate_uid, credential_id, service, routed_service, secret_hash, + finding_id, target_scan_id, + scan_event_id, finding_uid, source, query, target, detector_name, + found_at, metadata_json, state, capacity_bytes, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 'pending', ?, ?, ?) + ON CONFLICT(candidate_uid) DO NOTHING''' + candidate_values = ( + uid, credential_id, service, service, secret_hash, finding_id, target_scan_id, + reservation['scan_event_id'], finding_uid_value, reservation['source'], + reservation['query'], reservation['target'], metadata.get('detector_name'), + now, json_dumps(metadata), capacity_bytes, now, now, + ) + else: + candidate_sql = '''INSERT INTO keycheck_candidates( + candidate_uid, credential_id, service, secret_hash, + finding_id, target_scan_id, + scan_event_id, finding_uid, source, query, target, detector_name, + found_at, metadata_json, state, capacity_bytes, created_at, updated_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 'pending', ?, ?, ?) + ON CONFLICT(candidate_uid) DO NOTHING''' + candidate_values = ( + uid, credential_id, service, secret_hash, finding_id, target_scan_id, + reservation['scan_event_id'], finding_uid_value, reservation['source'], + reservation['query'], reservation['target'], metadata.get('detector_name'), + now, json_dumps(metadata), capacity_bytes, now, now, + ) + candidate_id = self.conn.insert_returning_id(candidate_sql, candidate_values) + return (1, capacity_bytes) if candidate_id is not None else (0, 0) + + def ingest_result_bundle(self, reader, reservation, bundle_row): + if not self.conn or not self.conn.is_postgres: + raise RuntimeError('normalized result bundle ingestion requires PostgreSQL') + validated = reader.validate() + reservation_id = int(reservation['id'] if 'id' in reservation else reservation['reservation_id']) + event_id = str(validated.scan_event_id) + event_hash = str(validated.scan_event_hash) + metadata = reader.metadata() + effective_diagnostics = reader.effective_diagnostics() + contracts = importlib.import_module('worker_contracts') + diagnostic_uid_set_sha256 = contracts.ordered_diagnostic_uid_set_sha256( + effective_diagnostics + ) + if any(key in metadata for key in ('findings', 'errors')): + raise ValueError('bundle metadata must not duplicate findings or errors') + metadata_bytes = json.dumps( + metadata, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str, + ).encode('utf-8') + if len(metadata_bytes) > 16 * 1024 * 1024: + raise ValueError('scan compatibility metadata exceeds its byte bound') + routed_service_supported = ( + 'routed_service' in self.conn.table_columns('keycheck_candidates') + ) + try: + self._lock_docker_depth_experiment_for_reservation(reservation_id) + current_reservation = self.conn.execute( + '''SELECT r.*, q.id AS bound_queue_id, + q.source AS bound_queue_source, + q.platform AS bound_queue_platform, + q.query AS bound_queue_query, + q.target AS bound_queue_target, + q.normalized_target AS bound_queue_normalized_target, + q.status AS bound_queue_status, + q.lease_token AS bound_queue_lease_token, + q.current_result_reservation_id AS bound_queue_reservation_id, + q.claim_event_id AS bound_queue_event_id + FROM result_reservations r + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ? FOR UPDATE OF r, q''', + (reservation_id,), + ).fetchone() + current_bundle = self.conn.execute( + 'SELECT * FROM result_bundles WHERE reservation_id = ? FOR UPDATE', + (reservation_id,), + ).fetchone() + if not current_reservation or not current_bundle: + raise ValueError('bundle reservation state is absent') + self._validate_result_ingestion_arguments( + current_reservation, current_bundle, reservation, bundle_row, + ) + queue_identity = { + 'id': current_reservation['bound_queue_id'], + 'source': current_reservation['bound_queue_source'], + 'platform': current_reservation['bound_queue_platform'], + 'query': current_reservation['bound_queue_query'], + 'target': current_reservation['bound_queue_target'], + 'normalized_target': current_reservation[ + 'bound_queue_normalized_target' + ], + } + self._validate_locked_reservation_queue_identity( + current_reservation, queue_identity, + ) + validated_identity = validated.as_dict() + validated_identity['relative_path'] = current_bundle['relative_path'] + self._validate_result_ingestion_arguments( + current_reservation, current_bundle, + current_reservation, validated_identity, + ) + self._validate_result_bundle_reservation_identity( + current_reservation, validated_identity, require_header=True, + ) + if str(current_reservation['assignment_kind']) == 'remote': + validate_remote_result_execution_plan( + current_reservation, metadata, + ) + projection_version = current_reservation[ + 'remote_diagnostic_projection_version' + ] + if int(projection_version or 0) == 0: + cursor = self.conn.execute( + '''UPDATE result_reservations + SET remote_diagnostic_projection_version = 1, + remote_diagnostic_count = ?, + remote_diagnostic_uids_sha256 = ?, updated_at = ? + WHERE id = ? AND remote_resolution_kind = 'bundle_accepted' + AND COALESCE(remote_diagnostic_projection_version, 0) = 0''', + ( + len(effective_diagnostics), + diagnostic_uid_set_sha256, utc_now_iso(), reservation_id, + ), + ) + if int(cursor.rowcount or 0) != 1: + raise ScanEventConflictError( + 'legacy accepted diagnostic projection fence changed' + ) + elif ( + int(projection_version) != contracts.DIAGNOSTIC_PROJECTION_VERSION + or int(current_reservation['remote_diagnostic_count'] or 0) + != len(effective_diagnostics) + or str(current_reservation['remote_diagnostic_uids_sha256'] or '') + != diagnostic_uid_set_sha256 + ): + raise ScanEventConflictError( + 'accepted diagnostic projection differs from receipt authority' + ) + if str(current_bundle['scan_event_hash']) != event_hash: + raise ScanEventConflictError( + 'validated bundle identity conflicts with its database reservation' + ) + docker_depth_binding = self._docker_depth_binding_for_reservation_locked( + current_reservation, + ) + self._precompute_docker_finding_attribution_rows_locked( + reader, docker_depth_binding, + ) + existing = self.conn.execute( + '''SELECT id, scan_event_hash, result_reservation_id, queue_id, + claim_lease_token, source, query, target, + normalized_target, scan_type, raw_result_storage, + compat_schema_version + FROM target_scans WHERE scan_event_id = ? FOR UPDATE''', + (event_id,), + ).fetchone() + if existing: + if str(existing['scan_event_hash']) != event_hash: + raise ScanEventConflictError(f'scan event {event_id} already exists with a different hash') + if int(existing['result_reservation_id'] or 0) != reservation_id: + raise ScanEventConflictError('scan event replay belongs to a different reservation') + if ( + int(existing['queue_id'] or 0) + != int(current_reservation['queue_id']) + or str(existing['claim_lease_token'] or '') + != str(current_reservation['claim_lease_token']) + or str(existing['source'] or '') + != str(current_reservation['source']) + or str(existing['query'] or '') + != str(current_reservation['query'] or '') + or str(existing['target'] or '') + != str(current_reservation['target']) + or str(existing['normalized_target'] or '') + != str(current_reservation['normalized_target']) + or str(existing['scan_type'] or '') + != str(current_reservation['platform']) + or str(existing['raw_result_storage'] or '') != 'normalized_v2' + or int(existing['compat_schema_version'] or 0) != 2 + ): + raise ScanEventConflictError( + 'authoritative scan replay lost its exact reservation identity' + ) + if docker_depth_binding: + if ( + docker_depth_binding['target_scan_id'] is None + or int(docker_depth_binding['target_scan_id']) != int(existing['id']) + ): + raise ScanEventConflictError( + 'Docker depth binding scan identity conflicts with replay' + ) + replay_finding_ids = {} + for finding in reader.iter_findings(): + finding_id, finding_uid_value = ( + self._existing_normalized_finding_for_replay_locked( + existing['id'], str(current_reservation['source']), + str(current_reservation['normalized_target']), finding, + ) + ) + replay_finding_ids[finding_uid_value] = finding_id + self._insert_docker_finding_layer_attributions_locked( + docker_depth_binding, existing['id'], finding_id, finding, + ) + if len(replay_finding_ids) != validated.finding_count: + raise ScanEventConflictError( + 'replayed Docker finding identities are missing or duplicated' + ) + diagnostic_received_at = utc_now_iso() + for diagnostic in effective_diagnostics: + self._record_worker_diagnostic( + reservation_id, diagnostic, + target_scan_id=int(existing['id']), + received_at=diagnostic_received_at, + ) + self.conn.execute( + "UPDATE result_bundles SET state = 'db_committed', target_scan_id = ?, updated_at = ? WHERE reservation_id = ?", + (existing['id'], utc_now_iso(), reservation_id), + ) + self.conn.execute( + "UPDATE result_reservations SET state = 'db_committed', updated_at = ? WHERE id = ?", + (utc_now_iso(), reservation_id), + ) + self.conn.commit() + return { + 'ingested': True, 'duplicate': True, 'target_scan_id': existing['id'], + 'scan_event_id': event_id, 'scan_event_hash': event_hash, + } + if current_reservation['state'] not in ('ready', 'ingesting'): + raise RuntimeError('bundle reservation is not ingestible') + if current_bundle['state'] not in ('ready', 'ingesting'): + raise RuntimeError('bundle row is not ingestible') + if ( + str(current_reservation['bound_queue_status']) != 'in_progress' + or str(current_reservation['bound_queue_lease_token'] or '') + != str(current_reservation['claim_lease_token']) + or int(current_reservation['bound_queue_reservation_id'] or 0) + != reservation_id + or str(current_reservation['bound_queue_event_id'] or '') != event_id + ): + raise ScanEventConflictError( + 'result ingestion lost its exact target queue fence' + ) + + source = str(current_reservation['source']) + target = str(current_reservation['target']) + normalized = str(current_reservation['normalized_target']) + run_id = current_reservation['run_id'] + cycle_id = current_reservation['cycle_id'] + query = current_reservation['query'] + package = extract_package_metadata(target, metadata, source) + status = str(metadata.get('status') or 'clean') + timestamp = str(metadata.get('timestamp') or utc_now_iso()) + target_scan_id = self.conn.insert_returning_id( + '''INSERT INTO target_scans( + scan_event_id, scan_event_hash, queue_id, claim_lease_token, + queue_completion_applied, queue_completion_disposition, + run_id, cycle_id, source, query, target, normalized_target, scan_type, + status, started_at, ended_at, duration_sec, scan_options_json, + package_name, package_version, package_artifact, package_date, + package_filename, package_type, package_size, findings_count, + verified_findings_count, error_count, skipped_reason, first_error_summary, + raw_result_json, result_reservation_id, compat_schema_version, + raw_result_storage, created_at + ) VALUES (?, ?, ?, ?, 0, 'pending', ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, + ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, NULL, ?, 2, 'normalized_v2', ?)''', + ( + event_id, event_hash, current_reservation['queue_id'], + current_reservation['claim_lease_token'], run_id, cycle_id, source, query, + target, normalized, metadata.get('scan_type'), status, + metadata.get('scan_started_at') or timestamp, timestamp, + metadata.get('duration_sec'), json_dumps(redact_config(metadata.get('scan_options') or {})), + package.get('package_name'), package.get('package_version'), + package.get('package_artifact'), package.get('package_date'), + package.get('package_filename'), package.get('package_type'), + package.get('package_size'), validated.finding_count, + int(metadata.get('verified_findings_count') or 0), validated.error_count, + metadata.get('skipped'), metadata.get('first_error_summary'), + reservation_id, utc_now_iso(), + ), + ) + if not target_scan_id: + raise RuntimeError('normalized target scan insert returned no identity') + self.conn.execute( + '''INSERT INTO scan_result_compat( + target_scan_id, schema_version, metadata_json, metadata_sha256, + metadata_bytes, reconstruction_status, created_at + ) VALUES (?, 2, ?, ?, ?, 'bounded', ?)''', + ( + target_scan_id, metadata_bytes.decode('utf-8'), + hashlib.sha256(metadata_bytes).hexdigest(), len(metadata_bytes), utc_now_iso(), + ), + ) + + diagnostic_received_at = utc_now_iso() + for diagnostic in effective_diagnostics: + self._record_worker_diagnostic( + reservation_id, diagnostic, + target_scan_id=target_scan_id, + received_at=diagnostic_received_at, + ) + + finding_ids = {} + verified_count = 0 + for finding in reader.iter_findings(): + finding_id, finding_uid_value = self._insert_normalized_finding( + run_id, cycle_id, target_scan_id, source, query, target, normalized, finding, + ) + finding_ids[finding_uid_value] = finding_id + self._insert_docker_finding_layer_attributions_locked( + docker_depth_binding, target_scan_id, finding_id, finding, + ) + verified_count += int(bool(finding.get('Verified'))) + if len(finding_ids) != validated.finding_count: + raise ValueError('bundle finding identities are missing or duplicated') + for error in reader.iter_errors(): + self._insert_error( + run_id, cycle_id, target_scan_id, source, query, target, normalized, error, + ) + actual_candidate_items = 0 + actual_candidate_bytes = 0 + per_candidate_capacity = ( + int(current_reservation['reserved_candidate_bytes']) // validated.candidate_count + if validated.candidate_count else 0 + ) + for frame in reader.iter_candidates(): + inserted_items, inserted_bytes = self._insert_bundle_candidate( + frame, current_reservation, target_scan_id, finding_ids, + capacity_allocation=per_candidate_capacity, + routed_service_supported=routed_service_supported, + ) + actual_candidate_items += inserted_items + actual_candidate_bytes += inserted_bytes + reserved_candidate_items = int(current_reservation['reserved_candidate_items']) + reserved_candidate_bytes = int(current_reservation['reserved_candidate_bytes']) + if actual_candidate_items > reserved_candidate_items or actual_candidate_bytes > reserved_candidate_bytes: + raise ValueError('bundle candidates exceed their pre-reserved capacity') + + now = utc_now_iso() + queue_status = str(metadata.get('queue_status') or ('failed' if status == 'error' else 'done')) + if queue_status not in ('done', 'failed', 'deferred'): + raise ValueError('bundle queue disposition is invalid') + self._apply_docker_layer_execution_locked( + current_reservation, metadata, queue_status, now, + ) + advance_git_coverage, covered_ref, covered_head = matching_git_coverage( + current_reservation, metadata, queue_status, validated.error_count, + ) + completed_at = now if queue_status in ('done', 'failed') else None + queue_cursor = self.conn.execute( + '''UPDATE target_queue SET status = ?, target_scan_id = ?, last_error = ?, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, + leased_at = NULL, lease_expires_at = NULL, available_after = ?, + attempts = CASE WHEN ? != 0 THEN 0 ELSE attempts END, + completed_at = COALESCE(?, completed_at), + covered_ref = CASE WHEN ? != 0 THEN ? ELSE covered_ref END, + covered_head = CASE WHEN ? != 0 THEN ? ELSE covered_head END, + current_result_reservation_id = NULL, claim_event_id = NULL, updated_at = ? + WHERE id = ? AND status = 'in_progress' AND lease_token = ? + AND current_result_reservation_id = ? AND claim_event_id = ?''', + ( + queue_status, target_scan_id, first_line(metadata.get('queue_error'), 500), + metadata.get('available_after'), 1 if metadata.get('reset_attempts') else 0, + completed_at, + 1 if advance_git_coverage else 0, covered_ref, + 1 if advance_git_coverage else 0, covered_head, + now, current_reservation['queue_id'], + current_reservation['claim_lease_token'], reservation_id, event_id, + ), + ) + if int(queue_cursor.rowcount or 0) != 1: + raise ScanEventConflictError('fenced target queue completion did not match the reservation') + self.conn.execute( + '''UPDATE target_scans SET queue_completion_applied = 1, + queue_completion_disposition = 'applied', + verified_findings_count = ? WHERE id = ?''', + (verified_count, target_scan_id), + ) + self._transition_docker_depth_binding_locked( + current_reservation, + 'completed' if queue_status == 'done' else 'failed', + queue_status if queue_status in ('done', 'failed') else 'pending', + now, + target_scan_id=target_scan_id, + ) + self._insert_derived_postman_targets(metadata.get('derived_postman_targets'), now) + projection_job_id = self.conn.insert_returning_id( + '''INSERT INTO projection_jobs( + job_kind, event_id, event_hash, target_scan_id, status, + required_stream_mask, capacity_items, capacity_bytes, + created_at, updated_at + ) VALUES ('scan_event', ?, ?, ?, 'pending', ?, ?, ?, ?, ?)''', + ( + event_id, event_hash, target_scan_id, + 1 | (2 if validated.finding_count else 0) | (4 if validated.error_count else 0), + current_reservation['reserved_projection_items'], + current_reservation['reserved_projection_bytes'], now, now, + ), + ) + if not projection_job_id: + raise RuntimeError('projection job insert returned no identity') + unused_items = reserved_candidate_items - actual_candidate_items + unused_bytes = reserved_candidate_bytes - actual_candidate_bytes + # Workload rows are complete before taking the global accounting lock. + capacity = self.conn.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1 FOR UPDATE' + ).fetchone() + if int(capacity['keycheck_items']) < unused_items or int(capacity['keycheck_bytes']) < unused_bytes: + raise RuntimeError('candidate capacity transfer would make accounting negative') + self.conn.execute( + '''UPDATE pipeline_capacity SET keycheck_items = keycheck_items - ?, + keycheck_bytes = keycheck_bytes - ?, updated_at = ? WHERE id = 1''', + (unused_items, unused_bytes, now), + ) + self.conn.execute( + '''UPDATE result_reservations SET state = 'db_committed', + projection_credit_transferred = 1, candidate_credit_transferred = 1, + updated_at = ? WHERE id = ?''', + (now, reservation_id), + ) + self.conn.execute( + '''UPDATE result_bundles SET state = 'db_committed', target_scan_id = ?, + committed_at = ?, updated_at = ? WHERE reservation_id = ?''', + (target_scan_id, now, now, reservation_id), + ) + found_count = 1 if validated.finding_count else 0 + clean_count = 1 if not validated.finding_count and not validated.error_count and not metadata.get('skipped') else 0 + skipped_count = 1 if metadata.get('skipped') else 0 + if cycle_id is not None: + self.conn.execute( + '''UPDATE source_cycles SET ingested_count = ingested_count + 1, + scanned_count = scanned_count + 1, clean_count = clean_count + ?, + found_count = found_count + ?, skipped_count = skipped_count + ?, + error_count = error_count + ?, findings_count = findings_count + ?, + verified_findings_count = verified_findings_count + ?, updated_at = ? + WHERE id = ?''', + ( + clean_count, found_count, skipped_count, validated.error_count, + validated.finding_count, verified_count, now, cycle_id, + ), + ) + if run_id is not None: + self.conn.execute( + '''UPDATE runs SET total_scanned = total_scanned + 1, + total_clean = total_clean + ?, total_found = total_found + ?, + total_skipped = total_skipped + ?, total_errors = total_errors + ?, + total_findings = total_findings + ?, + total_verified_findings = total_verified_findings + ?, updated_at = ? + WHERE id = ?''', + ( + clean_count, found_count, skipped_count, validated.error_count, + validated.finding_count, verified_count, now, run_id, + ), + ) + self.conn.commit() + return { + 'ingested': True, 'duplicate': False, 'target_scan_id': target_scan_id, + 'projection_job_id': projection_job_id, + 'candidate_count': actual_candidate_items, + 'scan_event_id': event_id, 'scan_event_hash': event_hash, + } + except Exception: + self.conn.rollback() + raise + + @staticmethod + def _compat_finding_from_row(row): + if row['payload_omitted']: + return { + 'finding_uid': row['finding_uid'], + 'DetectorName': row['detector_name'], + 'Verified': bool(row['verified']), + 'finding_omitted': True, + 'payload_sha256': row['payload_sha256'], + 'payload_bytes': row['payload_bytes'], + } + finding = safe_json_loads(row['extension_json']) or {} + finding.update({ + 'finding_uid': row['finding_uid'], + 'DetectorName': row['detector_name'], + 'DetectorType': row['detector_type'], + 'Verified': bool(row['verified']), + 'Raw': row['raw_value'], + 'RawV2': row['raw_v2_value'], + 'Redacted': row['redacted_secret'], + }) + for column, key in ( + ('structured_data_json', 'StructuredData'), + ('extra_data_json', 'ExtraData'), + ('analysis_info_json', 'AnalysisInfo'), + ): + value = safe_json_loads(row[column]) + if value is not None: + finding[key] = value + return finding + + def projection_scan_header(self, target_scan_id, max_bytes=16 * 1024 * 1024): + if not self.conn: + raise RuntimeError('database connection is unavailable') + raw_size_sql = ( + 'OCTET_LENGTH(ts.raw_result_json)' if self.conn.is_postgres + else "LENGTH(CAST(ts.raw_result_json AS BLOB))" + ) + scan = self.conn.execute( + f'''SELECT ts.id, ts.scan_event_id, ts.target, ts.scan_type, ts.ended_at, + ts.raw_result_storage, + CASE WHEN ts.raw_result_json IS NULL THEN 0 + ELSE {raw_size_sql} END AS raw_result_bytes, + sc.metadata_sha256, sc.metadata_bytes, sc.reconstruction_status + FROM target_scans ts + LEFT JOIN scan_result_compat sc ON sc.target_scan_id = ts.id + WHERE ts.id = ?''', + (int(target_scan_id),), + ).fetchone() + if not scan: + return None + if scan['raw_result_storage'] != 'normalized_v2': + raw_bytes = int(scan['raw_result_bytes'] or 0) + if raw_bytes <= 0 or raw_bytes > max(1, int(max_bytes)): + raise ValueError('legacy compatibility result exceeds its reconstruction byte bound') + payload_size_sql = ( + 'OCTET_LENGTH(raw_result_json)' if self.conn.is_postgres + else 'LENGTH(CAST(raw_result_json AS BLOB))' + ) + payload = self.conn.execute( + f'''SELECT raw_result_json FROM target_scans + WHERE id = ? AND raw_result_storage != 'normalized_v2' + AND raw_result_json IS NOT NULL + AND {payload_size_sql} = ?''', + (int(target_scan_id), raw_bytes), + ).fetchone() + if not payload: + raise RuntimeError('legacy compatibility result changed during bounded reconstruction') + legacy = safe_json_loads(payload['raw_result_json']) + if isinstance(legacy, dict): + legacy.setdefault( + 'scan_event_id', scan['scan_event_id'] or f'legacy-target-scan-{scan["id"]}', + ) + if self.conn.is_postgres: + self.conn.commit() + return {'storage': 'legacy', 'result': legacy} + if self.conn.is_postgres: + self.conn.commit() + return None + if int(scan['metadata_bytes'] or 0) > max(1, int(max_bytes)): + raise ValueError('normalized compatibility metadata exceeds its reconstruction byte bound') + metadata_size_sql = ( + 'OCTET_LENGTH(metadata_json)' if self.conn.is_postgres + else 'LENGTH(CAST(metadata_json AS BLOB))' + ) + metadata = self.conn.execute( + f'''SELECT metadata_json FROM scan_result_compat + WHERE target_scan_id = ? AND metadata_bytes = ? + AND {metadata_size_sql} <= ?''', + ( + int(target_scan_id), int(scan['metadata_bytes'] or 0), + max(1, int(max_bytes)), + ), + ).fetchone() + if not metadata: + raise RuntimeError('normalized compatibility metadata changed during bounded reconstruction') + result = safe_json_loads(metadata['metadata_json']) or {} + if not isinstance(result, dict): + raise ValueError('normalized compatibility metadata is invalid') + result['scan_event_id'] = scan['scan_event_id'] + result['target'] = scan['target'] + result['scan_type'] = scan['scan_type'] + result['timestamp'] = scan['ended_at'] + result.pop('findings', None) + result.pop('errors', None) + if self.conn.is_postgres: + self.conn.commit() + return {'storage': 'normalized_v2', 'result': result} + + def iter_projection_findings(self, target_scan_id, max_findings=20000, page_size=4): + last_id = 0 + count = 0 + page_size = min(64, max(1, int(page_size))) + while True: + rows = self.conn.execute( + '''SELECT f.*, cp.raw_value, cp.raw_v2_value, cp.structured_data_json, + cp.extra_data_json, cp.analysis_info_json, cp.extension_json, + cp.payload_sha256, cp.payload_bytes, cp.payload_omitted + FROM findings f + LEFT JOIN finding_compat_payloads cp ON cp.finding_id = f.id + WHERE f.target_scan_id = ? AND f.id > ? ORDER BY f.id LIMIT ?''', + (int(target_scan_id), last_id, page_size), + ).fetchall() + if self.conn.is_postgres: + self.conn.commit() + if not rows: + return + for row in rows: + count += 1 + if count > max(0, int(max_findings)): + raise ValueError('normalized compatibility reconstruction exceeds its finding bound') + last_id = int(row['id']) + yield self._compat_finding_from_row(row) + + def iter_projection_errors(self, target_scan_id, max_errors=20000, page_size=64): + last_id = 0 + count = 0 + page_size = min(256, max(1, int(page_size))) + while True: + rows = self.conn.execute( + '''SELECT id, raw_error FROM errors + WHERE target_scan_id = ? AND id > ? ORDER BY id LIMIT ?''', + (int(target_scan_id), last_id, page_size), + ).fetchall() + if self.conn.is_postgres: + self.conn.commit() + if not rows: + return + for row in rows: + count += 1 + if count > max(0, int(max_errors)): + raise ValueError('normalized compatibility reconstruction exceeds its error bound') + last_id = int(row['id']) + yield row['raw_error'] + + def reconstruct_scan_result( + self, target_scan_id, max_bytes=192 * 1024 * 1024, + max_findings=20000, max_errors=20000, + ): + header = self.projection_scan_header(target_scan_id, max_bytes=max_bytes) + if not header: + return None + result = header['result'] + if header['storage'] == 'normalized_v2': + byte_limit = max(1, int(max_bytes)) + used = len(json.dumps( + result, ensure_ascii=True, separators=(',', ':'), default=str, + ).encode('utf-8')) + 64 + findings = [] + for finding in self.iter_projection_findings(target_scan_id, max_findings): + used += len(json.dumps( + finding, ensure_ascii=True, separators=(',', ':'), default=str, + ).encode('utf-8')) + 1 + if used > byte_limit: + raise ValueError('normalized compatibility reconstruction exceeds its aggregate bound') + findings.append(finding) + errors = [] + for error in self.iter_projection_errors(target_scan_id, max_errors): + used += len(json.dumps( + error, ensure_ascii=True, separators=(',', ':'), default=str, + ).encode('utf-8')) + 1 + if used > byte_limit: + raise ValueError('normalized compatibility reconstruction exceeds its aggregate bound') + errors.append(error) + result['findings'] = findings + result['errors'] = errors + encoded = json.dumps(result, ensure_ascii=True, separators=(',', ':'), default=str).encode('utf-8') + if len(encoded) > max(1, int(max_bytes)): + raise ValueError('normalized compatibility reconstruction exceeds its aggregate bound') + return result + + def record_target_result(self, run_id, cycle_id, source, query, target, result, scan_options=None): + if not run_id or not cycle_id: + return None + def op(): + target_scan_id = self._insert_target_result(run_id, cycle_id, source, query, target, result, scan_options) + self.conn.commit() + return target_scan_id + return self._safe('record_target_result', op) + + def _insert_target_result( + self, run_id, cycle_id, source, query, target, result, scan_options=None, + scan_event_id=None, scan_event_hash_value=None, queue_id=None, claim_lease_token=None, + queue_completion_applied=0, queue_completion_disposition=None, + ): + status = target_status(result) + normalized = normalize_target(target, source) + findings = [sanitize_postman_finding(finding) for finding in (result.get('findings') or [])] + persisted_result = dict(result) + if 'findings' in result or findings: + persisted_result['findings'] = findings + errors = result.get('errors') or [] + package = extract_package_metadata(target, result, source) + timestamp = result.get('timestamp') or utc_now_iso() + conflict_clause = ' ON CONFLICT DO NOTHING' if scan_event_id else '' + target_scan_id = self.conn.insert_returning_id( + '''INSERT INTO target_scans ( + scan_event_id, scan_event_hash, queue_id, claim_lease_token, + queue_completion_applied, queue_completion_disposition, + run_id, cycle_id, source, query, target, normalized_target, scan_type, status, + started_at, ended_at, duration_sec, scan_options_json, package_name, package_version, + package_artifact, package_date, package_filename, package_type, package_size, + findings_count, verified_findings_count, error_count, skipped_reason, first_error_summary, + raw_result_json, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''' + conflict_clause, + ( + scan_event_id, scan_event_hash_value, queue_id, claim_lease_token, + 1 if queue_completion_applied else 0, queue_completion_disposition, + run_id, cycle_id, source, query, target, normalized, result.get('scan_type'), status, + result.get('scan_started_at') or timestamp, timestamp, result.get('duration_sec'), json_dumps(redact_config(scan_options or {})), + package.get('package_name'), package.get('package_version'), package.get('package_artifact'), package.get('package_date'), + package.get('package_filename'), package.get('package_type'), package.get('package_size'), + len(findings), sum(1 for finding in findings if finding.get('Verified', False)), len(errors), + result.get('skipped'), first_error_summary(result), json_dumps(persisted_result), utc_now_iso(), + ), + ) + if target_scan_id is None and scan_event_id: + return None + for finding in findings: + self._insert_finding(run_id, cycle_id, target_scan_id, source, query, target, normalized, finding) + for error in errors: + self._insert_error(run_id, cycle_id, target_scan_id, source, query, target, normalized, error) + return target_scan_id + + @staticmethod + def _unique_violation(exc): + code = getattr(exc, 'sqlstate', None) or getattr(exc, 'pgcode', None) + text = str(exc or '').lower() + return code == '23505' or 'unique constraint' in text or 'duplicate key' in text + + def _confirmed_scan_event(self, event_id, event_hash): + row = self.conn.execute( + '''SELECT id, scan_event_hash, queue_completion_applied, queue_completion_disposition + FROM target_scans WHERE scan_event_id = ?''', + (event_id,), + ).fetchone() + if not row: + return None + if str(row['scan_event_hash'] or '') != event_hash: + raise ScanEventConflictError(f'scan event {event_id} already exists with a different hash') + outbox = self.conn.execute( + 'SELECT id FROM scan_publication_outbox WHERE target_scan_id = ?', + (row['id'],), + ).fetchone() + # Delivery deletes the reference row. Its absence on replay therefore + # means the authoritative scan event was already projected. + self.conn.commit() + applied = bool(row['queue_completion_applied']) + return { + 'ingested': True, + 'duplicate': True, + 'stale': not applied, + 'queue_completion_applied': applied, + 'queue_completion_disposition': row['queue_completion_disposition'], + 'target_scan_id': row['id'], + 'outbox_id': outbox['id'] if outbox else None, + 'scan_event_id': event_id, + 'scan_event_hash': event_hash, + } + + def confirmed_scan_event(self, event_id, event_hash): + if not self.conn: + raise RuntimeError('database connection is unavailable') + return self._confirmed_scan_event(str(event_id or ''), str(event_hash or '')) + + def _insert_derived_postman_targets(self, targets, now): + seen = set() + for target in targets or []: + normalized = normalize_target(target, 'postman') + if not normalized or normalized in seen: + continue + seen.add(normalized) + self.conn.execute( + '''INSERT INTO target_queue ( + source, platform, query, target, normalized_target, status, created_at, updated_at + ) VALUES ('postman', 'postman', 'harvested', ?, ?, 'pending', ?, ?) + ON CONFLICT(source, normalized_target) DO NOTHING''', + (target, normalized, now, now), + ) + + def ingest_scan_event(self, envelope): + if not self.conn: + raise RuntimeError('database connection is unavailable') + self.require_runtime_safety_schema() + try: + event = prepare_scan_event(envelope) + except SpoolHashConflictError as exc: + raise ScanEventConflictError(str(exc)) from exc + event_id = event['scan_event_id'] + event_hash = event['scan_event_hash'] + result = event.get('result') + if not isinstance(result, dict): + raise ValueError('scan event result must be an object') + if str(result.get('scan_event_id') or '') != event_id: + raise ValueError('scan event result ID does not match its envelope') + queue_id = event.get('queue_id') + lease_token = str(event.get('claim_lease_token') or '') + if queue_id is None or not lease_token: + raise ValueError('scan event requires queue_id and claim_lease_token') + queue_status = str(event.get('queue_status') or '') + if queue_status not in ('done', 'failed', 'deferred'): + raise ValueError(f'invalid scan event queue status: {queue_status}') + run_id = event.get('run_id') + cycle_id = event.get('cycle_id') + if run_id is None or cycle_id is None: + raise ValueError('scan event requires run_id and cycle_id') + + try: + if self.conn.is_sqlite: + self.conn.execute('BEGIN IMMEDIATE') + lock_suffix = ' FOR UPDATE' if self.conn.is_postgres else '' + existing = self.conn.execute( + f'''SELECT id, scan_event_hash, queue_completion_applied, queue_completion_disposition + FROM target_scans WHERE scan_event_id = ?{lock_suffix}''', + (event_id,), + ).fetchone() + if existing: + if str(existing['scan_event_hash'] or '') != event_hash: + raise ScanEventConflictError(f'scan event {event_id} already exists with a different hash') + self.conn.commit() + return self._confirmed_scan_event(event_id, event_hash) + + target = str(event.get('target') or result.get('target') or '') + source = str(event.get('source') or '') + query = event.get('query') + target_scan_id = self._insert_target_result( + run_id, cycle_id, source, query, target, result, event.get('scan_options') or {}, + scan_event_id=event_id, + scan_event_hash_value=event_hash, + queue_id=queue_id, + claim_lease_token=lease_token, + queue_completion_applied=0, + queue_completion_disposition='pending', + ) + if target_scan_id is None: + confirmed = self._confirmed_scan_event(event_id, event_hash) + if confirmed: + return confirmed + raise RuntimeError(f'scan event {event_id} conflict returned no authoritative row') + now = utc_now_iso() + completed = now if queue_status in ('done', 'failed') else None + cur = self.conn.execute( + '''UPDATE target_queue SET + status = ?, target_scan_id = ?, last_error = ?, + lease_owner = NULL, lease_token = NULL, claim_batch = NULL, leased_at = NULL, lease_expires_at = NULL, + available_after = ?, attempts = CASE WHEN ? != 0 THEN 0 ELSE attempts END, + completed_at = COALESCE(?, completed_at), updated_at = ? + WHERE id = ? AND status = 'in_progress' AND lease_token = ?''', + ( + queue_status, target_scan_id, + first_line(event.get('queue_error'), 500) if event.get('queue_error') else None, + event.get('available_after'), 1 if event.get('reset_attempts') else 0, + completed, now, queue_id, lease_token, + ), + ) + applied = int(getattr(cur, 'rowcount', 0) or 0) == 1 + disposition = 'applied' if applied else 'stale_lease' + self.conn.execute( + '''UPDATE target_scans SET queue_completion_applied = ?, queue_completion_disposition = ? + WHERE id = ?''', + (1 if applied else 0, disposition, target_scan_id), + ) + self._insert_derived_postman_targets(event.get('derived_postman_targets'), now) + outbox_id = self.conn.insert_returning_id( + '''INSERT INTO scan_publication_outbox ( + target_scan_id, payload_json, status, attempts, created_at, updated_at + ) VALUES (?, ?, 'pending', 0, ?, ?)''', + (target_scan_id, '', now, now), + ) + self.conn.commit() + self.last_error = '' + return { + 'ingested': True, + 'duplicate': False, + 'stale': not applied, + 'queue_completion_applied': applied, + 'queue_completion_disposition': disposition, + 'target_scan_id': target_scan_id, + 'outbox_id': outbox_id, + 'scan_event_id': event_id, + 'scan_event_hash': event_hash, + } + except ScanEventConflictError: + try: + self.conn.rollback() + except Exception: + pass + raise + except Exception as exc: + self.last_error = str(exc) + try: + self.conn.rollback() + except Exception: + pass + if self._unique_violation(exc): + confirmed = self._confirmed_scan_event(event_id, event_hash) + if confirmed: + return confirmed + raise + + def record_and_complete_target_result( + self, run_id, cycle_id, source, query, target, result, scan_options, + queue_id, lease_owner, queue_status, queue_error=None, available_after=None, + reset_attempts=False, derived_postman_targets=None, lease_token=None, + ): + event_id = str(result.get('scan_event_id') or '') + if not self.conn or not event_id or queue_id is None or not lease_token: + return None + event = prepare_scan_event({ + 'version': 1, + 'scan_event_id': event_id, + 'run_id': run_id, + 'cycle_id': cycle_id, + 'source': source, + 'query': query, + 'target': target, + 'result': result, + 'scan_options': scan_options or {}, + 'queue_id': queue_id, + 'claim_lease_token': lease_token, + 'claim_lease_owner': lease_owner, + 'queue_status': queue_status, + 'queue_error': queue_error, + 'available_after': available_after, + 'reset_attempts': bool(reset_attempts), + 'derived_postman_targets': list(derived_postman_targets or []), + }) + return self.ingest_scan_event(event) + + def record_target_results(self, run_id, cycle_id, source, query, results, scan_options=None): + output = {} + for result in results or []: + target = result.get('target', '') + target_scan_id = self.record_target_result(run_id, cycle_id, source, query, target, result, scan_options) + if target_scan_id is None: + output[normalize_target(target, source)] = None + output[str(target)] = None + else: + output[normalize_target(target, source)] = target_scan_id + output[str(target)] = target_scan_id + return output + + def _insert_finding(self, run_id, cycle_id, target_scan_id, source, query, target, normalized, finding): + finding = sanitize_postman_finding(finding) + raw_secret, secret_hash, detector_secret_hash, fingerprint, location = finding_identity(source, normalized, finding) + finding_uid = finding.get('finding_uid') or fingerprint + enrichment = enrich_finding(finding) + finding_id = self.conn.insert_returning_id( + '''INSERT INTO findings ( + run_id, cycle_id, target_scan_id, source, query, target, normalized_target, + detector_name, detector_type, verified, raw_secret, redacted_secret, secret_hash, + detector_secret_hash, finding_fingerprint, finding_uid, file_path, line_number, commit_hash, + source_timestamp, source_metadata_type, source_metadata_json, raw_finding_json, + provider, credential_kind, credential_confidence, required_context_missing, + principal, username, email, project_id, tenant_id, organization, registry, + endpoint, scope, resource, enrichment_json, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + run_id, cycle_id, target_scan_id, source, query, target, normalized, + finding.get('DetectorName'), finding.get('DetectorType'), 1 if finding.get('Verified', False) else 0, + raw_secret, extract_redacted_secret(finding), secret_hash, detector_secret_hash, fingerprint, finding_uid, + location.get('file_path'), location.get('line_number'), location.get('commit_hash'), location.get('source_timestamp'), + location.get('source_metadata_type'), location.get('source_metadata_json'), json_dumps(finding), + enrichment.get('provider'), enrichment.get('credential_kind'), enrichment.get('credential_confidence'), + enrichment.get('required_context_missing'), enrichment.get('principal'), enrichment.get('username'), + enrichment.get('email'), enrichment.get('project_id'), enrichment.get('tenant_id'), enrichment.get('organization'), + enrichment.get('registry'), enrichment.get('endpoint'), enrichment.get('scope'), enrichment.get('resource'), + enrichment.get('enrichment_json'), utc_now_iso(), + ), + ) + if finding_uid and finding_id: + self.conn.execute( + '''INSERT INTO finding_uid_map (finding_uid, finding_id, created_at) + VALUES (?, ?, ?) + ON CONFLICT(finding_uid) DO NOTHING''', + (finding_uid, finding_id, utc_now_iso()), + ) + + def _insert_error(self, run_id, cycle_id, target_scan_id, source, query, target, normalized, error): + raw_error = str(error) + self.conn.execute( + '''INSERT INTO errors ( + run_id, cycle_id, target_scan_id, source, query, target, normalized_target, + category, summary, raw_error, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + ( + run_id, cycle_id, target_scan_id, source, query, target, normalized, + categorize_error(raw_error), first_line(raw_error), raw_error, utc_now_iso(), + ), + ) + + def record_keycheck_result(self, service, status, status_group, checked_at=None, key_hash='', secret_hash='', key_masked='', detector_name='', message='', metadata=None, link_findings=True, source_line='', detector_secret_hash=''): + if not self.conn: + return + self.require_runtime_safety_schema() + + def keycheck_result_columns(): + if self._keycheck_result_columns is None: + try: + self._keycheck_result_columns = self.conn.table_columns('keycheck_results') + except Exception: + self._keycheck_result_columns = set() + return self._keycheck_result_columns + + def matching_findings(hash_value): + if not hash_value: + return [] + clauses = ['secret_hash = ?'] + params = [hash_value] + if detector_name: + if str(detector_name).lower() == 'googleaistudio': + extra_name = self.conn.json_extract('raw_finding_json', '$.ExtraData.name') + clauses.append("""( + LOWER(detector_name) = LOWER(?) + OR ( + LOWER(detector_name) = 'customregex' + AND LOWER(COALESCE({extra_name}, '')) = LOWER(?) + ) + )""".format(extra_name=extra_name)) + params.extend([detector_name, detector_name]) + else: + clauses.append('LOWER(detector_name) = LOWER(?)') + params.append(detector_name) + row = self.conn.execute( + f'''SELECT id, run_id, cycle_id, target_scan_id, source, query, target, detector_name, created_at + FROM findings WHERE {' AND '.join(clauses)} ORDER BY id DESC LIMIT 1''', + params, + ).fetchone() + return [row] if row else [] + + def op(): + now = utc_now_iso() + checked = checked_at or now + # Inline linking is intentionally disabled for live keychecks. It can + # misattribute duplicate secrets and should be handled by the bounded + # repair/linker pipeline instead. + rows = [] + if not rows: + rows = [None] + has_link_columns = {'source_line', 'detector_secret_hash', 'link_status', 'link_attempts', 'linked_at', 'link_error'}.issubset(keycheck_result_columns()) + has_event_columns = {'event_id', 'finding_uid'}.issubset(keycheck_result_columns()) + for row in rows: + base_values = ( + service, status, status_group, checked, key_hash or '', secret_hash or '', key_masked or '', + row['id'] if row else None, + row['target_scan_id'] if row else None, + row['cycle_id'] if row else None, + row['run_id'] if row else None, + row['source'] if row else None, + row['query'] if row else None, + row['target'] if row else None, + row['detector_name'] if row else detector_name, + row['created_at'] if row else None, + first_line(message, 1000), + json_dumps(metadata or {}), + ) + if has_link_columns and has_event_columns: + self.conn.execute( + '''INSERT INTO keycheck_results ( + service, status, status_group, checked_at, key_hash, secret_hash, key_masked, + finding_id, target_scan_id, cycle_id, run_id, source, query, target, detector_name, + found_at, message, metadata_json, source_line, detector_secret_hash, + event_id, finding_uid, link_status, link_attempts, linked_at, link_error, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + base_values + ( + source_line or (metadata or {}).get('source_line') or '', + detector_secret_hash or '', + (metadata or {}).get('event_id') or '', + (metadata or {}).get('finding_uid') or '', + 'linked' if row else 'pending', + 0, + now if row else None, + '', + now, + ), + ) + elif has_link_columns: + self.conn.execute( + '''INSERT INTO keycheck_results ( + service, status, status_group, checked_at, key_hash, secret_hash, key_masked, + finding_id, target_scan_id, cycle_id, run_id, source, query, target, detector_name, + found_at, message, metadata_json, source_line, detector_secret_hash, + link_status, link_attempts, linked_at, link_error, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + base_values + ( + source_line or (metadata or {}).get('source_line') or '', + detector_secret_hash or '', + 'linked' if row else 'pending', + 0, + now if row else None, + '', + now, + ), + ) + else: + self.conn.execute( + '''INSERT INTO keycheck_results ( + service, status, status_group, checked_at, key_hash, secret_hash, key_masked, + finding_id, target_scan_id, cycle_id, run_id, source, query, target, detector_name, + found_at, message, metadata_json, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', + base_values + (now,), + ) + self.conn.commit() + return True + return bool(self._safe('record_keycheck_result', op, False)) + + +def _migration_add_columns(conn, table, additions): + existing = conn.table_columns(table) + for name, ddl in additions.items(): + if name not in existing: + conn.execute(f'ALTER TABLE {table} ADD COLUMN {name} {ddl}') + + +def _migration_seed_legacy_docker_provenance( + conn, + max_rows=DOCKER_LEGACY_PROVENANCE_SEED_MAX_ROWS, + page_size=DOCKER_LEGACY_PROVENANCE_SEED_PAGE_SIZE, +): + max_rows = max(1, int(max_rows)) + page_size = max(1, min(int(page_size), max_rows, 10000)) + after_id = 0 + scanned = 0 + inserted = 0 + predicate = """platform = 'docker' AND query IS NOT NULL AND query <> '' + AND (resolver_state IS NOT NULL OR target NOT LIKE '%@%')""" + while scanned < max_rows: + rows = conn.execute( + f'''SELECT id, source, query, target, created_at, updated_at + FROM target_queue + WHERE id > ? AND {predicate} + ORDER BY id LIMIT ?''', + (after_id, min(page_size, max_rows - scanned)), + ).fetchall() + if not rows: + break + for row in rows: + after_id = int(row['id']) + target = str(row['target'] or '').strip() + if not target or '@' in target or ':' in target.rsplit('/', 1)[-1]: + continue + cursor = conn.execute( + '''INSERT INTO docker_repository_query_provenance( + source, query, repository_queue_id, provenance_kind, + first_observed_at, last_observed_at, created_at, updated_at + ) VALUES (?, ?, ?, 'legacy_queue', ?, ?, ?, ?) + ON CONFLICT(source, query, repository_queue_id) DO NOTHING''', + ( + row['source'], row['query'], row['id'], row['created_at'], + row['created_at'], row['created_at'], row['updated_at'], + ), + ) + inserted += max(0, int(getattr(cursor, 'rowcount', 0) or 0)) + scanned += len(rows) + if scanned >= max_rows and conn.execute( + f'''SELECT 1 AS present FROM target_queue + WHERE id > ? AND {predicate} ORDER BY id LIMIT 1''', + (after_id,), + ).fetchone(): + raise RuntimeSafetySchemaError( + f'legacy Docker provenance seed exceeds the reviewed {max_rows}-row bound' + ) + return {'scanned_count': scanned, 'inserted_count': inserted} + + +def _migration_backfill_keycheck_hashes(conn, max_rows=1000000): + processed = 0 + while True: + rows = conn.execute( + '''SELECT id, service, candidate_kind, secret_text, secret_json, endpoint, principal + FROM keycheck_credentials WHERE provider_key_hash = '' ORDER BY id LIMIT 250''' + ).fetchall() + if not rows: + break + for row in rows: + try: + digest = stored_provider_key_hash( + row['service'], row['candidate_kind'], row['secret_text'], row['secret_json'], + row['endpoint'], row['principal'], + ) + except ValueError as exc: + raise RuntimeSafetySchemaError( + f'keycheck credential {row["id"]} cannot be migrated to provider-canonical identity' + ) from exc + conn.execute( + "UPDATE keycheck_credentials SET provider_key_hash = ? WHERE id = ? AND provider_key_hash = ''", + (digest, row['id']), + ) + processed += 1 + if processed > max_rows: + raise RuntimeSafetySchemaError('keycheck credential identity backfill exceeded its row bound') + + processed = 0 + while True: + rows = conn.execute( + '''SELECT c.id, + (SELECT NULLIF(f.secret_hash, '') FROM findings f WHERE f.id = c.finding_id) AS finding_hash, + (SELECT NULLIF(r.secret_hash, '') FROM keycheck_results r + WHERE r.candidate_id = c.id ORDER BY r.id DESC LIMIT 1) AS result_hash, + kc.provider_key_hash + FROM keycheck_candidates c + JOIN keycheck_credentials kc ON kc.id = c.credential_id + WHERE c.secret_hash = '' ORDER BY c.id LIMIT 250''' + ).fetchall() + if not rows: + break + for row in rows: + digest = str(row['finding_hash'] or row['result_hash'] or row['provider_key_hash'] or '').lower() + if not re.fullmatch(r'[a-f0-9]{64}', digest): + raise RuntimeSafetySchemaError( + f'keycheck candidate {row["id"]} cannot be migrated to detector secret identity' + ) + conn.execute( + "UPDATE keycheck_candidates SET secret_hash = ? WHERE id = ? AND secret_hash = ''", + (digest, row['id']), + ) + processed += 1 + if processed > max_rows: + raise RuntimeSafetySchemaError('keycheck candidate identity backfill exceeded its row bound') + + processed = 0 + while True: + rows = conn.execute( + '''SELECT r.id, kc.provider_key_hash, c.secret_hash + FROM keycheck_results r + JOIN keycheck_candidates c ON c.id = r.candidate_id + JOIN keycheck_credentials kc ON kc.id = c.credential_id + WHERE COALESCE(r.key_hash, '') != kc.provider_key_hash + OR COALESCE(r.secret_hash, '') != c.secret_hash + ORDER BY r.id LIMIT 250''' + ).fetchall() + if not rows: + break + for row in rows: + conn.execute( + 'UPDATE keycheck_results SET key_hash = ?, secret_hash = ? WHERE id = ?', + (row['provider_key_hash'], row['secret_hash'], row['id']), + ) + processed += 1 + if processed > max_rows: + raise RuntimeSafetySchemaError('keycheck result identity backfill exceeded its row bound') + + duplicate = conn.execute( + '''SELECT service, provider_key_hash, COUNT(*) AS count + FROM keycheck_credentials WHERE provider_key_hash != '' + GROUP BY service, provider_key_hash HAVING COUNT(*) > 1 LIMIT 1''' + ).fetchone() + if duplicate: + raise RuntimeSafetySchemaError( + 'provider-canonical keycheck credentials contain duplicate persisted identities; ' + 'reviewed credential merge is required before cutover' + ) + + +def _migration_backfill_docker_selection_policies(conn, max_reservations=1000000): + processed = 0 + while True: + reservations = conn.execute( + '''SELECT DISTINCT reservation_id + FROM docker_image_blob_coverage + WHERE COALESCE(selection_policy_sha256, '') = ? + ORDER BY reservation_id LIMIT 250''', + ('',), + ).fetchall() + if not reservations: + break + for identity in reservations: + reservation_id = int(identity['reservation_id']) + reservation = conn.execute( + '''SELECT id, docker_layer_plan_json, docker_layer_plan_sha256 + FROM result_reservations WHERE id = ?''', + (reservation_id,), + ).fetchone() + if not reservation: + raise RuntimeSafetySchemaError( + f'Docker coverage reservation {reservation_id} is absent during selector backfill' + ) + plan, plan_sha256 = stored_docker_layer_plan(reservation) + if plan is None: + raise RuntimeSafetySchemaError( + f'Docker coverage reservation {reservation_id} has no bound plan' + ) + coverage_policy_sha256 = docker_layer_plan_coverage_policy_sha256(plan) + expected = { + descriptor['position']: descriptor for descriptor in plan['descriptors'] + } + rows = conn.execute( + '''SELECT position, blob_digest, coverage_policy_sha256, + selection_policy_sha256, descriptor_kind, plan_sha256, + selected, selection_reason + FROM docker_image_blob_coverage + WHERE reservation_id = ? ORDER BY position''', + (reservation_id,), + ).fetchall() + if len(rows) != len(expected): + raise RuntimeSafetySchemaError( + f'Docker coverage reservation {reservation_id} is incomplete during selector backfill' + ) + empty_count = 0 + for row in rows: + descriptor = expected.get(int(row['position'])) + existing_selector = str(row['selection_policy_sha256'] or '') + if ( + not descriptor + or str(row['blob_digest']) != descriptor['digest'] + or str(row['coverage_policy_sha256']) != coverage_policy_sha256 + or existing_selector not in ('', plan['selection_policy_sha256']) + or str(row['descriptor_kind']) != descriptor['kind'] + or str(row['plan_sha256']) != plan_sha256 + or bool(row['selected']) != descriptor['selected'] + or str(row['selection_reason']) != descriptor['selection_reason'] + ): + raise RuntimeSafetySchemaError( + f'Docker coverage reservation {reservation_id} conflicts with its bound plan' + ) + empty_count += 1 if not existing_selector else 0 + cursor = conn.execute( + '''UPDATE docker_image_blob_coverage SET selection_policy_sha256 = ? + WHERE reservation_id = ? AND COALESCE(selection_policy_sha256, '') = ?''', + (plan['selection_policy_sha256'], reservation_id, ''), + ) + if int(cursor.rowcount or 0) != empty_count: + raise RuntimeSafetySchemaError( + f'Docker coverage reservation {reservation_id} changed during selector backfill' + ) + processed += 1 + if processed > max_reservations: + raise RuntimeSafetySchemaError( + 'Docker selection-policy backfill exceeded its reservation bound' + ) + remaining = conn.execute( + '''SELECT reservation_id FROM docker_image_blob_coverage + WHERE COALESCE(selection_policy_sha256, '') = ? LIMIT 1''', + ('',), + ).fetchone() + if remaining: + raise RuntimeSafetySchemaError('Docker selection-policy backfill is incomplete') + + +def _pipeline_quarantine_review_status_constraint(conn): + if conn.is_postgres: + rows = conn.execute( + '''SELECT con.conname AS name, + pg_catalog.pg_get_constraintdef(con.oid, true) AS definition, + con.convalidated AS is_valid + FROM pg_catalog.pg_constraint con + JOIN pg_catalog.pg_class tbl ON tbl.oid = con.conrelid + JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace + WHERE con.contype = 'c' AND n.nspname = ? + AND tbl.relname = 'pipeline_quarantine' ''', + (conn.application_schema,), + ).fetchall() + matching = [ + row for row in rows + if 'review_status' in str(row['definition'] or '').lower() + ] + if len(matching) != 1: + return {'name': '', 'statuses': None, 'valid': False} + row = matching[0] + statuses = frozenset(re.findall(r"'([^']+)'", str(row['definition'] or ''))) + return { + 'name': str(row['name'] or ''), + 'statuses': statuses, + 'valid': bool(row['is_valid']), + } + row = conn.execute( + "SELECT sql FROM sqlite_master WHERE type = 'table' AND name = 'pipeline_quarantine'" + ).fetchone() + sql = str(row['sql'] if row else '') + match = re.search( + r'review_status\s+TEXT.*?CHECK\s*\(\s*review_status\s+IN\s*\((.*?)\)\s*\)', + sql, re.IGNORECASE | re.DOTALL, + ) + statuses = frozenset(re.findall(r"'([^']+)'", match.group(1))) if match else None + return {'name': '', 'statuses': statuses, 'valid': statuses is not None} + + +def _migration_ensure_pipeline_quarantine_review_status_constraint(conn): + constraint = _pipeline_quarantine_review_status_constraint(conn) + statuses = constraint.get('statuses') + if statuses == PIPELINE_QUARANTINE_REVIEW_STATUSES: + if conn.is_postgres and not constraint.get('valid'): + conn.execute( + 'ALTER TABLE pipeline_quarantine VALIDATE CONSTRAINT ' + + _quoted_pg_name(constraint['name']) + ) + return + if not conn.is_postgres: + raise RuntimeSafetySchemaError( + 'SQLite pipeline_quarantine review-status constraint requires a reviewed table rebuild' + ) + if statuses not in (None, PIPELINE_QUARANTINE_LEGACY_REVIEW_STATUSES): + raise RuntimeSafetySchemaError( + 'PostgreSQL pipeline_quarantine has an unexpected review-status constraint; ' + 'manual reviewed repair is required' + ) + if statuses is not None: + conn.execute( + 'ALTER TABLE pipeline_quarantine DROP CONSTRAINT ' + + _quoted_pg_name(constraint['name']) + ) + conn.execute( + '''ALTER TABLE pipeline_quarantine + ADD CONSTRAINT pipeline_quarantine_review_status_check + CHECK (review_status IN ( + 'pending','approved_retry','approved_rescan','discarded','resolved' + ))''' + ) + + +def _migration_require_pipeline_quiescence(conn): + if not conn.table_exists('runtime_schema_migrations'): + return + applied = { + str(row['version']) for row in conn.execute( + 'SELECT version FROM runtime_schema_migrations' + ).fetchall() + } + if set(PIPELINE_MIGRATION_VERSIONS).issubset(applied) or not all( + conn.table_exists(table) for table in ( + 'pipeline_leases', 'result_reservations', 'target_queue', 'docker_content_blobs', + ) + ): + return + capacity_backfill_only = ( + set(PIPELINE_MIGRATION_VERSIONS[:-1]).issubset(applied) + and REMOTE_ASSIGNMENT_CAPACITY_MIGRATION not in applied + ) + active = { + 'worker_leases': conn.execute( + "SELECT COUNT(*) AS count FROM pipeline_leases WHERE state NOT IN ('released','failed')" + ).fetchone()['count'], + 'result_reservations': 0 if capacity_backfill_only else conn.execute( + """SELECT COUNT(*) AS count FROM result_reservations + WHERE state IN ('scanning','ready','ingesting','db_committed')""" + ).fetchone()['count'], + 'queue_leases': 0 if capacity_backfill_only else conn.execute( + """SELECT COUNT(*) AS count FROM target_queue q + LEFT JOIN result_reservations r ON r.id = q.current_result_reservation_id + WHERE q.status = 'in_progress' + OR r.state IN ('scanning','ready','ingesting','db_committed')""" + ).fetchone()['count'], + 'blob_leases': conn.execute( + """SELECT COUNT(*) AS count FROM docker_content_blobs + WHERE state IN ('leased','submitted') OR lease_reservation_id IS NOT NULL""" + ).fetchone()['count'], + } + if conn.table_exists('discovery_retry_queue'): + active['discovery_retry_leases'] = conn.execute( + "SELECT COUNT(*) AS count FROM discovery_retry_queue WHERE status = 'leased'" + ).fetchone()['count'] + queue_columns = set(conn.table_columns('target_queue')) + if {'resolver_state', 'resolver_token'} <= queue_columns: + active['docker_resolvers'] = conn.execute( + """SELECT COUNT(*) AS count FROM target_queue + WHERE resolver_state = 'resolving' OR resolver_token IS NOT NULL""" + ).fetchone()['count'] + if conn.table_exists('docker_depth_experiments'): + active['experiment_authority_fences'] = conn.execute( + """SELECT COUNT(*) AS count FROM docker_depth_experiments + WHERE fence_token IS NOT NULL + OR state IN ('holding','resolving','active','draining')""" + ).fetchone()['count'] + if conn.table_exists('docker_depth_experiment_repositories'): + active['experiment_resolvers'] = conn.execute( + """SELECT COUNT(*) AS count FROM docker_depth_experiment_repositories + WHERE work_state = 'resolving' OR resolver_token IS NOT NULL""" + ).fetchone()['count'] + if ( + conn.table_exists('target_queue_policy_events') + and 'experiment_id' in conn.table_columns('target_queue_policy_events') + ): + active['experiment_holds'] = conn.execute( + """SELECT COUNT(*) AS count + FROM target_queue_policy_events event + LEFT JOIN docker_depth_experiments experiment + ON experiment.id = event.experiment_id + WHERE event.experiment_id IS NOT NULL AND event.action = 'cold' + AND NOT EXISTS ( + SELECT 1 FROM target_queue_policy_events reversal + WHERE reversal.reverses_event_id = event.id + ) + AND ( + experiment.id IS NULL + OR experiment.state NOT IN ('held','completed','released') + OR experiment.fence_token IS NOT NULL + )""" + ).fetchone()['count'] + if any(int(value or 0) for value in active.values()): + summary = ', '.join( + f'{name}={int(value or 0)}' for name, value in active.items() + if int(value or 0) + ) + raise RuntimeSafetySchemaError( + f'pipeline schema migration requires stopped, reconciled runtime ({summary})' + ) + + +def _migration_postgres_column_shape(conn, table, specs): + if not conn.is_postgres: + return + type_sql = {'id': 'BIGINT', 'id_ref': 'BIGINT', 'integer': 'INTEGER', 'text': 'TEXT', 'real': 'REAL'} + details = conn.table_column_details(table) + for name, (expected_type, expected_not_null) in specs.items(): + actual = details.get(name) + if not actual: + raise RuntimeSafetySchemaError(f'migration did not create {table}.{name}') + if not _schema_type_matches(actual['type'], expected_type, True): + target_type = type_sql[expected_type] + actual_type = re.sub(r'\s+', ' ', str(actual['type'] or '').strip().lower()) + safe_sources = { + 'BIGINT': {'integer', 'smallint'}, + 'INTEGER': {'smallint'}, + 'TEXT': {'character varying', 'character', 'varchar'}, + 'REAL': set(), + } + if actual_type not in safe_sources[target_type]: + raise RuntimeSafetySchemaError( + f'PostgreSQL {table}.{name} type {actual_type or ""} cannot be safely ' + f'changed to {target_type}; manual reviewed conversion is required' + ) + conn.execute( + f'ALTER TABLE {table} ALTER COLUMN {name} TYPE {target_type} USING {name}::{target_type}' + ) + if expected_not_null and not actual['not_null']: + conn.execute(f'ALTER TABLE {table} ALTER COLUMN {name} SET NOT NULL') + elif not expected_not_null and actual['not_null'] and not actual['primary_key']: + conn.execute(f'ALTER TABLE {table} ALTER COLUMN {name} DROP NOT NULL') + + +def _migration_require_sqlite_column_shape(conn, table, specs): + if not conn.is_sqlite: + return + details = conn.table_column_details(table) + problems = [] + for name, (expected_type, expected_not_null) in specs.items(): + actual = details.get(name) + if not actual: + problems.append(f'missing {name}') + continue + if not _schema_type_matches(actual['type'], expected_type, False): + problems.append(f'{name} type {actual["type"] or ""}') + if bool(actual['not_null']) != bool(expected_not_null): + problems.append(f'{name} nullability') + if problems: + raise RuntimeSafetySchemaError( + f'SQLite table {table} has a non-additively-repairable schema ({"; ".join(problems)}); ' + 'manually rebuild this table from a reviewed backup with the exact runtime schema' + ) + + +def _migration_ensure_defaults(conn, table, specs, defaults, generated_id=None): + details = conn.table_column_details(table) + for name in specs: + if name == generated_id: + continue + expected_present = name in defaults + expected = defaults.get(name, '') + actual = details.get(name) + actual_present = bool(actual and actual.get('has_default', bool(actual.get('default')))) + if actual and actual_present == expected_present and ( + not expected_present or _normalized_default(actual['default']) == _normalized_default(expected) + ): + continue + if conn.is_sqlite: + raise RuntimeSafetySchemaError( + f'SQLite table {table}.{name} has default {actual["default"] or ""}; ' + 'manually rebuild the table with the exact runtime default' + ) + if not expected_present: + conn.execute(f'ALTER TABLE {table} ALTER COLUMN {name} DROP DEFAULT') + else: + default_sql = str(int(expected)) if str(expected).isdigit() else "'" + str(expected).replace("'", "''") + "'" + conn.execute(f'ALTER TABLE {table} ALTER COLUMN {name} SET DEFAULT {default_sql}') + + +def _migration_require_primary_key(conn, table, specs, expected): + details = conn.table_column_details(table) + actual = [name for name in specs if details.get(name, {}).get('primary_key')] + if actual != list(expected): + raise RuntimeSafetySchemaError( + f'{table} primary key is {actual or "absent"}, expected {list(expected)}; ' + 'manual reviewed table rebuild is required' + ) + if conn.is_postgres: + primary_indexes = [ + index for index in conn.table_indexes(table).values() + if index.get('primary') + ] + if ( + len(primary_indexes) != 1 + or primary_indexes[0].get('columns') != list(expected) + or not _index_usable(primary_indexes[0]) + ): + raise RuntimeSafetySchemaError( + f'PostgreSQL {table} primary-key index is absent, invalid, or unready; ' + 'manual reviewed primary-key constraint rebuild is required' + ) + + +def _quoted_pg_name(value): + parts = str(value or '').split('.') + if not parts or any(not re.fullmatch(r'[A-Za-z_][A-Za-z0-9_$]*', part) for part in parts): + raise RuntimeSafetySchemaError(f'unsafe PostgreSQL catalog identifier: {value!r}') + return '.'.join('"' + part.replace('"', '""') + '"' for part in parts) + + +def _migration_ensure_generated_id(conn, table, column): + details = conn.table_column_details(table).get(column) + if _column_generates_id(details, conn.is_postgres): + return + if not conn.is_postgres: + raise RuntimeSafetySchemaError( + f'SQLite {table}.{column} does not auto-generate an INTEGER PRIMARY KEY; ' + 'manually rebuild this table from a reviewed backup' + ) + if not details or details.get('type') != 'bigint' or not details.get('primary_key'): + raise RuntimeSafetySchemaError( + f'PostgreSQL {table}.{column} is not a BIGINT PRIMARY KEY; manual reviewed table repair is required' + ) + if details.get('identity') or details.get('generated') or details.get('default') or details.get('sequence'): + raise RuntimeSafetySchemaError( + f'PostgreSQL {table}.{column} has malformed identity/sequence metadata; ' + 'manually repair its generated-ID definition before retrying' + ) + try: + conn.execute(f'ALTER TABLE {table} ALTER COLUMN {column} ADD GENERATED BY DEFAULT AS IDENTITY') + except Exception as exc: + raise RuntimeSafetySchemaError( + f'PostgreSQL {table}.{column} cannot be safely converted to an identity column; ' + 'manually add a generated sequence/identity and reseed it above MAX(id)' + ) from exc + repaired = conn.table_column_details(table).get(column) or {} + sequence = repaired.get('sequence') + if not _column_generates_id(repaired, True) or not sequence: + raise RuntimeSafetySchemaError( + f'PostgreSQL did not create a usable identity sequence for {table}.{column}; ' + 'manual identity repair is required' + ) + row = conn.execute(f'SELECT COALESCE(MAX({column}), 0) + 1 AS next_id FROM {table}').fetchone() + next_id = max(1, int(row['next_id'] if row else 1)) + try: + conn.execute(f'ALTER SEQUENCE {_quoted_pg_name(sequence)} RESTART WITH {next_id}') + except Exception as exc: + raise RuntimeSafetySchemaError( + f'PostgreSQL identity sequence {sequence} could not be reseeded to {next_id}; ' + f'manually restart it above MAX({table}.{column})' + ) from exc + + +def _migration_add_docker_depth_schema_authority_checks(conn, marker_applied): + if not conn.is_postgres or marker_applied: + return + for table, name in DOCKER_DEPTH_SCHEMA_AUTHORITY_CHECKS: + if name in conn.table_check_constraints(table): + continue + expression = DOCKER_DEPTH_EXPERIMENT_CHECK_SPECS[table][name] + try: + conn.execute( + f'ALTER TABLE {_quoted_pg_name(table)} ADD CONSTRAINT ' + f'{_quoted_pg_name(name)} CHECK ({expression}) NOT VALID' + ) + conn.execute( + f'ALTER TABLE {_quoted_pg_name(table)} VALIDATE CONSTRAINT ' + f'{_quoted_pg_name(name)}' + ) + except Exception as exc: + raise RuntimeSafetySchemaError( + f'PostgreSQL {table}.{name} cannot be validated against existing rows; ' + 'manually repair malformed safety data before retrying migration' + ) from exc + + +def _migration_add_docker_depth_scarcity_checks(conn, marker_applied): + if not conn.is_postgres or marker_applied: + return + for table, name in DOCKER_DEPTH_SCARCITY_COHORT_CHECKS: + if name in conn.table_check_constraints(table): + continue + expression = DOCKER_DEPTH_EXPERIMENT_CHECK_SPECS[table][name] + try: + conn.execute( + f'ALTER TABLE {_quoted_pg_name(table)} ADD CONSTRAINT ' + f'{_quoted_pg_name(name)} CHECK ({expression}) NOT VALID' + ) + conn.execute( + f'ALTER TABLE {_quoted_pg_name(table)} VALIDATE CONSTRAINT ' + f'{_quoted_pg_name(name)}' + ) + except Exception as exc: + raise RuntimeSafetySchemaError( + f'PostgreSQL {table}.{name} cannot be validated against existing rows; ' + 'manually repair malformed scarcity cohort data before retrying migration' + ) from exc + + +def _migration_require_docker_depth_check_constraints(conn): + problems = _docker_depth_check_constraint_problems(conn) + prior_reviewed = { + ('docker_depth_experiments', 'docker_depth_experiments_range_check'): ( + 'query_count >= 1 AND query_count <= 1000 ' + 'AND repositories_per_query >= 1 AND repositories_per_query <= 10 ' + 'AND images_per_repository >= 1 AND images_per_repository <= 10 ' + 'AND target_limit >= 1 AND target_limit <= 1200 ' + 'AND target_count >= 0 AND target_count <= target_limit ' + 'AND selection_count >= 0 AND fence_generation >= 0' + ), + ('docker_depth_experiments', 'docker_depth_experiments_capacity_check'): ( + 'query_count * (repositories_per_query + images_per_repository - 1) ' + '<= target_limit' + ), + ( + 'docker_depth_experiment_queries', + 'docker_depth_experiment_queries_range_check', + ): ( + 'query_ordinal >= 0 AND required_repository_count >= 1 ' + 'AND required_repository_count <= 10' + ), + ( + 'docker_depth_experiment_repositories', + 'docker_depth_experiment_repositories_range_check', + ): ( + 'query_ordinal >= 0 AND repository_rank >= 1 AND repository_rank <= 10 ' + 'AND resolver_generation >= 0 AND resolver_attempts >= 0 ' + 'AND selected_image_count >= 0 AND selected_image_count <= 10' + ), + ( + 'docker_depth_experiment_targets', + 'docker_depth_experiment_targets_range_check', + ): ( + 'counter_ordinal >= 1 AND counter_ordinal <= 1200 ' + 'AND dispatch_wave >= 1 AND dispatch_wave <= 3 ' + 'AND dispatch_order >= 1 AND reservation_count >= 0' + ), + } + if conn.is_postgres and problems: + for table, name in prior_reviewed: + label = f'check constraint {table}.{name}' + if label not in problems: + continue + constraint = conn.table_check_constraints(table).get(name) + if ( + not constraint + or not constraint.get('valid', True) + or _normalized_check_expression(constraint.get('expression')) + != _normalized_check_expression(prior_reviewed[(table, name)]) + ): + continue + expression = DOCKER_DEPTH_EXPERIMENT_CHECK_SPECS[table][name] + try: + conn.execute( + f'ALTER TABLE {_quoted_pg_name(table)} ' + f'DROP CONSTRAINT {_quoted_pg_name(name)}' + ) + conn.execute( + f'ALTER TABLE {_quoted_pg_name(table)} ADD CONSTRAINT ' + f'{_quoted_pg_name(name)} CHECK ({expression}) NOT VALID' + ) + conn.execute( + f'ALTER TABLE {_quoted_pg_name(table)} VALIDATE CONSTRAINT ' + f'{_quoted_pg_name(name)}' + ) + except Exception as exc: + raise RuntimeSafetySchemaError( + f'PostgreSQL {table}.{name} reviewed widening failed validation' + ) from exc + problems = _docker_depth_check_constraint_problems(conn) + if problems: + raise RuntimeSafetySchemaError( + 'Docker depth experiment schema has missing, invalid, or noncanonical ' + + '; '.join(problems) + + '; only the exact prior reviewed CHECK constraints can be widened' + ) + + +def _migration_ensure_foreign_keys(conn): + for table, expected_keys in REQUIRED_FOREIGN_KEYS.items(): + foreign_keys = conn.table_foreign_keys(table) + for columns, referenced_table, referenced_columns in expected_keys: + authority_matching = [ + (name, foreign_key) + for name, foreign_key in foreign_keys.items() + if tuple(foreign_key.get('columns') or ()) == columns + and foreign_key.get('referenced_table') == referenced_table + and tuple(foreign_key.get('referenced_columns') or ()) == referenced_columns + and foreign_key.get('referenced_schema') in ( + conn.application_schema if conn.is_postgres else 'main', + ) + ] + if len(authority_matching) > 1: + raise RuntimeSafetySchemaError( + f'{table} has duplicate foreign keys for ({", ".join(columns)}); ' + 'manual reviewed constraint cleanup is required' + ) + expected_actions = _required_foreign_key_actions(table, columns) + if authority_matching and not _foreign_key_actions_match( + authority_matching[0][1], expected_actions, + ): + raise RuntimeSafetySchemaError( + f'{table}({", ".join(columns)}) foreign key actions are not ' + f'ON UPDATE {expected_actions[0]} ON DELETE {expected_actions[1]}; ' + 'manual reviewed constraint repair is required' + ) + if authority_matching: + name, foreign_key = authority_matching[0] + if conn.is_postgres and not foreign_key.get('valid', True): + try: + conn.execute(f'ALTER TABLE {table} VALIDATE CONSTRAINT {_quoted_pg_name(name)}') + except Exception as exc: + raise RuntimeSafetySchemaError( + f'PostgreSQL {table} foreign key {name} cannot be validated against existing rows; ' + 'manually repair orphaned references before retrying migration' + ) from exc + continue + conflicting = [ + foreign_key for foreign_key in foreign_keys.values() + if tuple(foreign_key.get('columns') or ()) == columns + ] + if conflicting: + raise RuntimeSafetySchemaError( + f'{table}({", ".join(columns)}) has a foreign key to the wrong authority; ' + 'manual reviewed constraint repair is required' + ) + if not conn.is_postgres: + raise RuntimeSafetySchemaError( + f'SQLite {table} is missing foreign key ({", ".join(columns)}) -> ' + f'{referenced_table}({", ".join(referenced_columns)}); ' + 'manually rebuild this table from a reviewed backup' + ) + name = f'fk_{table}_{"_".join(columns)}' + try: + conn.execute( + f'ALTER TABLE {table} ADD CONSTRAINT {name} ' + f'FOREIGN KEY ({", ".join(columns)}) REFERENCES ' + f'{POSTGRES_APPLICATION_SCHEMA}.{referenced_table} ({", ".join(referenced_columns)}) ' + f'ON UPDATE {expected_actions[0]} ON DELETE {expected_actions[1]}' + ) + except Exception as exc: + raise RuntimeSafetySchemaError( + f'PostgreSQL {table} cannot add foreign key ({", ".join(columns)}) because ' + 'existing rows do not match the referenced authority; manually repair orphaned rows' + ) from exc + foreign_keys = conn.table_foreign_keys(table) + + +def _migration_ensure_index( + conn, table, name, sql, columns, unique=False, predicate='', required_sql_fragments=(), +): + current = conn.table_indexes(table).get(name) + current_sql = str((current or {}).get('sql') or '').lower() + valid = bool( + current + and bool(current['unique']) == bool(unique) + and current['columns'] == list(columns) + and _normalized_predicate(current['predicate']) == _normalized_predicate(predicate) + and _index_usable(current) + and all(str(fragment).lower() in current_sql for fragment in required_sql_fragments) + ) + if valid: + return + if current: + conn.execute(f'DROP INDEX {name}') + conn.execute(sql) + + +def _ensure_runtime_operations_authority(conn, now=None): + timestamp = now or utc_now_iso() + conn.execute( + '''INSERT INTO runtime_operations_control( + id, revision, discovery_paused, dispatch_paused, drain_state, + actor, operation_id, created_at, updated_at + ) VALUES (1, 0, 0, 0, 'normal', 'system:migration', NULL, ?, ?) + ON CONFLICT(id) DO NOTHING''', + (timestamp, timestamp), + ) + if conn.is_postgres: + conn.execute(''' + CREATE OR REPLACE FUNCTION reject_runtime_audit_event_mutation() + RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + RAISE EXCEPTION 'runtime_audit_events is append-only'; + END; + $$ + ''') + conn.execute( + 'DROP TRIGGER IF EXISTS runtime_audit_events_reject_mutation ' + 'ON runtime_audit_events' + ) + conn.execute(''' + CREATE TRIGGER runtime_audit_events_reject_mutation + BEFORE UPDATE OR DELETE ON runtime_audit_events + FOR EACH ROW EXECUTE FUNCTION reject_runtime_audit_event_mutation() + ''') + conn.execute( + 'DROP TRIGGER IF EXISTS runtime_audit_events_reject_truncate ' + 'ON runtime_audit_events' + ) + conn.execute(''' + CREATE TRIGGER runtime_audit_events_reject_truncate + BEFORE TRUNCATE ON runtime_audit_events + FOR EACH STATEMENT EXECUTE FUNCTION reject_runtime_audit_event_mutation() + ''') + return + conn.execute('DROP TRIGGER IF EXISTS runtime_audit_events_reject_update') + conn.execute(''' + CREATE TRIGGER runtime_audit_events_reject_update + BEFORE UPDATE ON runtime_audit_events + BEGIN + SELECT RAISE(ABORT, 'runtime_audit_events is append-only'); + END + ''') + conn.execute('DROP TRIGGER IF EXISTS runtime_audit_events_reject_delete') + conn.execute(''' + CREATE TRIGGER runtime_audit_events_reject_delete + BEFORE DELETE ON runtime_audit_events + BEGIN + SELECT RAISE(ABORT, 'runtime_audit_events is append-only'); + END + ''') + + +def _migration_retain_pg_trgm_without_name_index(conn): + if not conn.is_postgres: + return + extension = conn.execute( + '''SELECT EXISTS ( + SELECT 1 FROM pg_catalog.pg_available_extensions WHERE name = ? + ) AS available''', + ('pg_trgm',), + ).fetchone() + if extension and extension['available']: + conn.execute('SAVEPOINT package_candidate_trgm') + try: + conn.execute('CREATE EXTENSION IF NOT EXISTS pg_trgm') + conn.execute('RELEASE SAVEPOINT package_candidate_trgm') + except Exception as exc: + conn.execute('ROLLBACK TO SAVEPOINT package_candidate_trgm') + conn.execute('RELEASE SAVEPOINT package_candidate_trgm') + logger.warning( + 'pg_trgm extension retention failed; bounded ordered lookup remains active: %s', + type(exc).__name__, + ) + conn.execute('DROP INDEX IF EXISTS idx_package_repo_candidates_name_trgm') + + +def _migration_reconcile_pipeline_capacity(conn, page_size=1000): + totals = { + 'bundle_items': 0, 'bundle_bytes': 0, + 'projection_items': 0, 'projection_bytes': 0, + 'keycheck_items': 0, 'keycheck_bytes': 0, + 'quarantine_items': 0, 'quarantine_bytes': 0, + } + + def pages(table, columns, where): + after_id = 0 + while True: + rows = conn.execute( + f'''SELECT id, {columns} FROM {table} + WHERE id > ? AND ({where}) ORDER BY id LIMIT ?''', + (after_id, max(1, int(page_size))), + ).fetchall() + if not rows: + return + for row in rows: + after_id = int(row['id']) + yield row + + for row in pages( + 'result_reservations', + '''reserved_bundle_bytes, reserved_projection_items, reserved_projection_bytes, + reserved_candidate_items, reserved_candidate_bytes, bundle_credit_released, + projection_credit_transferred, candidate_credit_transferred''', + "state IN ('scanning','ready','ingesting','db_committed')", + ): + if not row['bundle_credit_released']: + totals['bundle_items'] += 1 + totals['bundle_bytes'] += int(row['reserved_bundle_bytes']) + if not row['projection_credit_transferred']: + totals['projection_items'] += int(row['reserved_projection_items']) + totals['projection_bytes'] += int(row['reserved_projection_bytes']) + if not row['candidate_credit_transferred']: + totals['keycheck_items'] += int(row['reserved_candidate_items']) + totals['keycheck_bytes'] += int(row['reserved_candidate_bytes']) + for row in pages( + 'projection_jobs', 'capacity_items, capacity_bytes', + "status IN ('pending','leased') AND capacity_released = 0", + ): + totals['projection_items'] += int(row['capacity_items']) + totals['projection_bytes'] += int(row['capacity_bytes']) + for row in pages( + 'keycheck_candidates', + 'capacity_bytes, result_projection_reserved_bytes, result_projection_credit_transferred', + "state IN ('pending','leased','deferred') AND capacity_released = 0", + ): + totals['keycheck_items'] += 1 + totals['keycheck_bytes'] += int(row['capacity_bytes']) + if ( + int(row['result_projection_reserved_bytes'] or 0) > 0 + and not row['result_projection_credit_transferred'] + ): + totals['projection_items'] += 1 + totals['projection_bytes'] += int(row['result_projection_reserved_bytes']) + for row in pages( + 'pipeline_quarantine', 'capacity_items, capacity_bytes', + "review_status = 'pending' AND capacity_credit_applied = 1", + ): + totals['quarantine_items'] += int(row['capacity_items']) + totals['quarantine_bytes'] += int(row['capacity_bytes']) + conn.execute( + '''UPDATE pipeline_capacity SET bundle_items = ?, bundle_bytes = ?, + projection_items = ?, projection_bytes = ?, keycheck_items = ?, + keycheck_bytes = ?, quarantine_items = ?, quarantine_bytes = ?, + updated_at = ? WHERE id = 1''', + ( + totals['bundle_items'], totals['bundle_bytes'], totals['projection_items'], + totals['projection_bytes'], totals['keycheck_items'], totals['keycheck_bytes'], + totals['quarantine_items'], totals['quarantine_bytes'], utc_now_iso(), + ), + ) + return totals + + +def migrate_runtime_safety_schema(db, initialize_base=False): + conn = getattr(db, 'conn', None) + if not conn: + raise RuntimeSafetySchemaError('database connection is unavailable') + try: + if initialize_base: + conn.executescript(SCHEMA_SQL) + for table in ('target_queue', 'target_scans', 'findings'): + if not conn.table_exists(table): + raise RuntimeSafetySchemaError(f'base schema table is missing: {table}') + _migration_require_pipeline_quiescence(conn) + + _migration_add_columns(conn, 'target_queue', { + 'lease_token': 'TEXT', + 'claim_batch': 'TEXT', + 'resolver_state': 'TEXT', + 'resolver_due_at': 'TEXT', + 'resolver_attempts': 'INTEGER NOT NULL DEFAULT 0', + 'resolver_token': 'TEXT', + 'current_result_reservation_id': 'BIGINT' if conn.is_postgres else 'INTEGER', + 'claim_event_id': 'TEXT', + 'remote_modified_at': 'TEXT', + 'scan_remote_modified_at': 'TEXT', + 'covered_ref': 'TEXT', + 'covered_head': 'TEXT', + }) + + id_type = 'BIGINT' if conn.is_postgres else 'INTEGER' + _migration_add_columns(conn, 'target_scans', { + 'scan_event_id': 'TEXT', + 'scan_event_hash': 'TEXT', + 'queue_id': id_type, + 'claim_lease_token': 'TEXT', + 'queue_completion_applied': 'INTEGER NOT NULL DEFAULT 0', + 'queue_completion_disposition': 'TEXT', + 'result_reservation_id': id_type, + 'compat_schema_version': 'INTEGER NOT NULL DEFAULT 2', + 'raw_result_storage': "TEXT NOT NULL DEFAULT 'legacy'", + }) + + _migration_add_columns(conn, 'findings', { + 'detector_type': 'TEXT', + 'verified': 'INTEGER DEFAULT 0', + 'raw_secret': 'TEXT', + 'redacted_secret': 'TEXT', + 'secret_hash': 'TEXT', + 'finding_uid': 'TEXT', + 'detector_secret_hash': 'TEXT', + 'finding_fingerprint': 'TEXT', + 'file_path': 'TEXT', + 'line_number': 'TEXT', + 'commit_hash': 'TEXT', + 'source_timestamp': 'TEXT', + 'source_metadata_type': 'TEXT', + 'raw_finding_json': 'TEXT', + 'source_metadata_json': 'TEXT', + 'provider': 'TEXT', + 'credential_kind': 'TEXT', + 'credential_confidence': 'TEXT', + 'required_context_missing': 'INTEGER DEFAULT 0', + 'principal': 'TEXT', + 'username': 'TEXT', + 'email': 'TEXT', + 'project_id': 'TEXT', + 'tenant_id': 'TEXT', + 'organization': 'TEXT', + 'registry': 'TEXT', + 'endpoint': 'TEXT', + 'scope': 'TEXT', + 'resource': 'TEXT', + 'enrichment_json': 'TEXT', + 'raw_payload_sha256': 'TEXT', + 'raw_payload_bytes': id_type, + 'raw_payload_omitted': 'INTEGER NOT NULL DEFAULT 0', + }) + _migration_add_columns(conn, 'runs', { + 'total_staged': 'INTEGER NOT NULL DEFAULT 0', + 'total_quarantined': 'INTEGER NOT NULL DEFAULT 0', + }) + _migration_add_columns(conn, 'source_cycles', { + 'queued_updated_count': 'INTEGER NOT NULL DEFAULT 0', + 'staged_count': 'INTEGER NOT NULL DEFAULT 0', + 'ingested_count': 'INTEGER NOT NULL DEFAULT 0', + 'quarantined_count': 'INTEGER NOT NULL DEFAULT 0', + }) + conn.execute(f'''CREATE TABLE IF NOT EXISTS finding_uid_map ( + finding_uid TEXT PRIMARY KEY, + finding_id {id_type} NOT NULL, + created_at TEXT NOT NULL, + FOREIGN KEY(finding_id) REFERENCES findings(id) + )''') + _migration_postgres_column_shape(conn, 'finding_uid_map', { + 'finding_uid': ('text', True), 'finding_id': ('id_ref', True), 'created_at': ('text', True), + }) + + outbox_id = 'BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY' if conn.is_postgres else 'INTEGER PRIMARY KEY AUTOINCREMENT' + conn.execute(f'''CREATE TABLE IF NOT EXISTS scan_publication_outbox ( + id {outbox_id}, + target_scan_id {id_type} NOT NULL UNIQUE, + payload_json TEXT NOT NULL, + status TEXT NOT NULL DEFAULT 'pending', + attempts INTEGER DEFAULT 0, + last_error TEXT, + lease_owner TEXT, + lease_expires_at TEXT, + available_after TEXT, + created_at TEXT NOT NULL, + delivered_at TEXT, + updated_at TEXT NOT NULL, + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id) + )''') + _migration_add_columns(conn, 'scan_publication_outbox', { + 'payload_json': "TEXT NOT NULL DEFAULT ''", + 'status': "TEXT NOT NULL DEFAULT 'pending'", + 'attempts': 'INTEGER DEFAULT 0', + 'last_error': 'TEXT', + 'lease_owner': 'TEXT', + 'lease_expires_at': 'TEXT', + 'available_after': 'TEXT', + 'created_at': "TEXT NOT NULL DEFAULT ''", + 'delivered_at': 'TEXT', + 'updated_at': "TEXT NOT NULL DEFAULT ''", + }) + _migration_postgres_column_shape(conn, 'scan_publication_outbox', OUTBOX_COLUMN_SPECS) + compact_delivered_scan_publications(conn) + conn.execute( + '''UPDATE scan_publication_outbox SET payload_json = '', + status = CASE WHEN status = 'delivering' THEN 'delivering' ELSE 'pending' END, + lease_owner = CASE WHEN status = 'delivering' THEN lease_owner ELSE NULL END, + lease_expires_at = CASE WHEN status = 'delivering' THEN lease_expires_at ELSE NULL END, + available_after = CASE WHEN status = 'dead' THEN NULL ELSE available_after END + WHERE payload_json != '' OR status NOT IN ('pending', 'delivering')''' + ) + + cursor_integer = 'BIGINT' if conn.is_postgres else 'INTEGER' + conn.execute(f'''CREATE TABLE IF NOT EXISTS target_queue_reconciliation_cursors ( + source_file TEXT PRIMARY KEY, + file_identity TEXT NOT NULL, + file_size {cursor_integer} NOT NULL DEFAULT 0, + file_mtime_ns {cursor_integer} NOT NULL DEFAULT 0, + source TEXT NOT NULL, + platform TEXT NOT NULL, + byte_offset {cursor_integer} NOT NULL DEFAULT 0, + line_number {cursor_integer} NOT NULL DEFAULT 0, + discarding_oversized INTEGER NOT NULL DEFAULT 0, + oversized_line_start {cursor_integer}, + cumulative_rows {cursor_integer} NOT NULL DEFAULT 0, + cumulative_bytes {cursor_integer} NOT NULL DEFAULT 0, + cumulative_inserted {cursor_integer} NOT NULL DEFAULT 0, + cumulative_rejected {cursor_integer} NOT NULL DEFAULT 0, + completed_at TEXT, + last_report_json TEXT, + updated_at TEXT NOT NULL + )''') + _migration_add_columns(conn, 'target_queue_reconciliation_cursors', { + 'file_size': f'{cursor_integer} NOT NULL DEFAULT 0', + 'file_mtime_ns': f'{cursor_integer} NOT NULL DEFAULT 0', + 'discarding_oversized': 'INTEGER NOT NULL DEFAULT 0', + 'oversized_line_start': cursor_integer, + 'cumulative_rows': f'{cursor_integer} NOT NULL DEFAULT 0', + 'cumulative_bytes': f'{cursor_integer} NOT NULL DEFAULT 0', + 'cumulative_inserted': f'{cursor_integer} NOT NULL DEFAULT 0', + 'cumulative_rejected': f'{cursor_integer} NOT NULL DEFAULT 0', + 'completed_at': 'TEXT', + }) + _migration_postgres_column_shape(conn, 'target_queue_reconciliation_cursors', CURSOR_COLUMN_SPECS) + + issue_id = 'BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY' if conn.is_postgres else 'INTEGER PRIMARY KEY AUTOINCREMENT' + conn.execute(f'''CREATE TABLE IF NOT EXISTS target_queue_reconciliation_issues ( + id {issue_id}, + source_file TEXT NOT NULL, + file_identity TEXT NOT NULL, + source TEXT NOT NULL, + platform TEXT NOT NULL, + line_number {cursor_integer} NOT NULL, + byte_offset {cursor_integer} NOT NULL, + reason TEXT NOT NULL, + target_preview TEXT, + created_at TEXT NOT NULL, + resolved_at TEXT + )''') + _migration_postgres_column_shape(conn, 'target_queue_reconciliation_issues', RECONCILIATION_ISSUE_COLUMN_SPECS) + + conn.executescript(KEYCHECK_RESULTS_SQL) + if 'id' not in conn.table_columns('keycheck_results'): + raise RuntimeSafetySchemaError('keycheck_results.id is missing and cannot be repaired additively') + _migration_add_columns(conn, 'keycheck_results', { + 'source_line': 'TEXT', + 'detector_secret_hash': 'TEXT', + 'event_id': 'TEXT', + 'finding_uid': 'TEXT', + 'link_status': "TEXT DEFAULT 'pending'", + 'link_attempts': 'INTEGER DEFAULT 0', + 'linked_at': 'TEXT', + 'link_error': 'TEXT', + 'candidate_id': id_type, + 'credential_id': id_type, + 'result_source': "TEXT NOT NULL DEFAULT 'api_check'", + }) + conn.execute(f'''CREATE TABLE IF NOT EXISTS keycheck_event_map ( + event_id TEXT PRIMARY KEY, + keycheck_result_id {id_type}, + created_at TEXT NOT NULL, + FOREIGN KEY(keycheck_result_id) REFERENCES keycheck_results(id) + )''') + _migration_postgres_column_shape(conn, 'keycheck_results', KEYCHECK_COLUMN_SPECS) + _migration_postgres_column_shape(conn, 'keycheck_event_map', { + 'event_id': ('text', True), 'keycheck_result_id': ('id_ref', False), 'created_at': ('text', True), + }) + conn.executescript(KEYCHECK_RESULTS_SQL) + # The canonical schema creates this index, so an existing marker-13 table + # needs the additive selector column before the schema script reaches it. + if conn.table_exists('docker_image_blob_coverage'): + _migration_add_columns(conn, 'docker_image_blob_coverage', { + 'selection_policy_sha256': "TEXT NOT NULL DEFAULT ''", + }) + if conn.table_exists('target_queue_policy_events'): + experiment_reference = ( + 'BIGINT' if conn.is_postgres + else 'INTEGER REFERENCES docker_depth_experiments(id)' + ) + _migration_add_columns(conn, 'target_queue_policy_events', { + 'experiment_id': experiment_reference, + }) + if conn.table_exists('discovery_retry_queue'): + source_cycle_reference = ( + 'BIGINT' if conn.is_postgres + else 'INTEGER REFERENCES source_cycles(id)' + ) + _migration_add_columns(conn, 'discovery_retry_queue', { + 'source_cycle_id': source_cycle_reference, + }) + rollout_authority_migration = conn.execute( + 'SELECT version FROM runtime_schema_migrations WHERE version = ?', + (DOCKER_DEPTH_ROLLOUT_AUTHORITY_MIGRATION,), + ).fetchone() + schema_selector_authority_migration = conn.execute( + 'SELECT version FROM runtime_schema_migrations WHERE version = ?', + (DOCKER_DEPTH_SCHEMA_SELECTOR_AUTHORITY_MIGRATION,), + ).fetchone() + scarcity_cohort_migration = conn.execute( + 'SELECT version FROM runtime_schema_migrations WHERE version = ?', + (DOCKER_DEPTH_SCARCITY_COHORT_MIGRATION,), + ).fetchone() + if conn.table_exists('docker_depth_experiments'): + _migration_add_columns(conn, 'docker_depth_experiments', { + 'collection_generation': ( + "TEXT NOT NULL DEFAULT 'docker-depth-provenance-v1'" + ), + 'selection_sha256': 'TEXT', + }) + if not rollout_authority_migration: + conn.execute( + "UPDATE docker_depth_experiments " + "SET collection_generation = 'legacy'" + ) + if conn.table_exists('docker_discovery_passes'): + _migration_add_columns(conn, 'docker_discovery_passes', { + 'collection_generation': ( + "TEXT NOT NULL DEFAULT 'docker-depth-provenance-v1'" + ), + }) + if not rollout_authority_migration: + conn.execute( + "UPDATE docker_discovery_passes " + "SET collection_generation = 'legacy'" + ) + if conn.table_exists('docker_discovery_pages'): + count_type = 'BIGINT' if conn.is_postgres else 'INTEGER' + _migration_add_columns(conn, 'docker_discovery_pages', { + 'total_count': count_type, + }) + if ( + conn.table_exists('docker_finding_layer_attributions') + and 'unattributed_reason' not in conn.table_columns( + 'docker_finding_layer_attributions' + )): + if conn.execute( + "SELECT id FROM docker_finding_layer_attributions " + "WHERE attribution_state = 'unattributed' LIMIT 1" + ).fetchone(): + raise RuntimeSafetySchemaError( + 'existing unattributed Docker findings require a reviewed reason backfill' + ) + reason_check = DOCKER_DEPTH_EXPERIMENT_CHECK_SPECS[ + 'docker_finding_layer_attributions' + ]['docker_finding_layer_attributions_reason_check'] + _migration_add_columns(conn, 'docker_finding_layer_attributions', { + 'unattributed_reason': ( + 'TEXT CONSTRAINT docker_finding_layer_attributions_reason_check ' + f'CHECK ({reason_check})' + ), + }) + if conn.table_exists('docker_depth_experiment_repositories'): + resolver_migration = conn.execute( + 'SELECT version FROM runtime_schema_migrations WHERE version = ?', + ('20260910_20_docker_depth_resolver',), + ).fetchone() + _migration_add_columns(conn, 'docker_depth_experiment_repositories', { + 'planned_is_deep_probe': 'INTEGER NOT NULL DEFAULT 0', + 'resolver_due_at': 'TEXT', + 'candidate_distinct_graph_count': 'INTEGER NOT NULL DEFAULT 0', + 'replacement_repository_queue_id': ( + 'BIGINT' if conn.is_postgres else 'INTEGER' + ), + 'replacement_eligibility_page_id': ( + 'BIGINT' if conn.is_postgres else 'INTEGER' + ), + 'replacement_count': 'INTEGER NOT NULL DEFAULT 0', + 'replacement_evidence_sha256': 'TEXT', + }) + if not resolver_migration: + conn.execute( + '''UPDATE docker_depth_experiment_repositories + SET planned_is_deep_probe = is_deep_probe''' + ) + if conn.table_exists('docker_depth_experiment_queries'): + selection_check = DOCKER_DEPTH_EXPERIMENT_CHECK_SPECS[ + 'docker_depth_experiment_queries' + ]['docker_depth_experiment_queries_selection_check'] + _migration_add_columns(conn, 'docker_depth_experiment_queries', { + 'selected_repository_count': ( + 'INTEGER NOT NULL DEFAULT 0 ' + 'CONSTRAINT docker_depth_experiment_queries_selection_check ' + f'CHECK ({selection_check})' + ), + }) + if not scarcity_cohort_migration: + conn.execute( + '''UPDATE docker_depth_experiment_queries + SET selected_repository_count = required_repository_count''' + ) + # Existing installations must gain columns before the schema script + # creates partial indexes that reference them. Fresh installs create + # the complete tables directly from PIPELINE_SCHEMA_SQL below. + remote_capacity_applied = False + if conn.table_exists('result_reservations'): + remote_capacity_applied = bool(conn.execute( + '''SELECT version FROM runtime_schema_migrations + WHERE version = ?''', + (REMOTE_ASSIGNMENT_CAPACITY_MIGRATION,), + ).fetchone()) if conn.table_exists('runtime_schema_migrations') else False + projection_authority_applied = bool(conn.execute( + '''SELECT version FROM runtime_schema_migrations + WHERE version = ?''', + (DIAGNOSTIC_PROJECTION_AUTHORITY_MIGRATION,), + ).fetchone()) if conn.table_exists('runtime_schema_migrations') else False + _migration_add_columns(conn, 'result_reservations', { + 'reserved_bundle_bytes': ( + 'BIGINT CHECK (reserved_bundle_bytes > 0)' + ), + 'assignment_kind': "TEXT NOT NULL DEFAULT 'local' CHECK (assignment_kind IN ('local','remote'))", + 'remote_user_id': 'BIGINT' if conn.is_postgres else 'INTEGER', + 'remote_device_id': 'BIGINT' if conn.is_postgres else 'INTEGER', + 'remote_issued_at': 'TEXT', 'remote_expires_at': 'TEXT', + 'remote_result_upload_body_timeout_seconds': ( + 'INTEGER CHECK (' + 'remote_result_upload_body_timeout_seconds IS NULL OR ' + 'remote_result_upload_body_timeout_seconds BETWEEN 30 AND 86400)' + ), + 'remote_effective_config_sha256': 'TEXT', + 'remote_client_compat_sha256': 'TEXT', + 'remote_execution_snapshot_json': 'TEXT', + 'remote_execution_snapshot_sha256': 'TEXT', + 'remote_resolution_kind': 'TEXT', 'remote_payload_sha256': 'TEXT', + 'remote_receipt_id': 'TEXT', 'remote_resolution_json': 'TEXT', + 'remote_resolved_at': 'TEXT', + 'remote_diagnostic_projection_version': 'INTEGER', + 'remote_diagnostic_count': 'INTEGER', + 'remote_diagnostic_uids_sha256': 'TEXT', + }) + if not remote_capacity_applied: + conn.execute( + '''UPDATE result_reservations + SET reserved_bundle_bytes = declared_bundle_bytes + WHERE reserved_bundle_bytes IS NULL''' + ) + if conn.is_postgres: + conn.execute( + '''ALTER TABLE result_reservations + ALTER COLUMN reserved_bundle_bytes SET NOT NULL''' + ) + if not projection_authority_applied: + conn.execute( + '''UPDATE result_reservations + SET remote_diagnostic_projection_version = 0 + WHERE assignment_kind = 'remote' + AND remote_resolution_kind = 'bundle_accepted' + AND remote_diagnostic_projection_version IS NULL''' + ) + if conn.table_exists('admission_intents'): + _migration_add_columns(conn, 'admission_intents', { + 'remote_user_id': 'BIGINT' if conn.is_postgres else 'INTEGER', + 'remote_device_id': 'BIGINT' if conn.is_postgres else 'INTEGER', + }) + conn.executescript(PIPELINE_SCHEMA_SQL) + _ensure_runtime_operations_authority(conn) + _migration_ensure_pipeline_quarantine_review_status_constraint(conn) + _migration_add_docker_depth_schema_authority_checks( + conn, bool(schema_selector_authority_migration), + ) + _migration_add_docker_depth_scarcity_checks( + conn, bool(scarcity_cohort_migration), + ) + _migration_require_docker_depth_check_constraints(conn) + _migration_add_columns(conn, 'pipeline_quarantine', { + 'capacity_credit_applied': 'INTEGER NOT NULL DEFAULT 1', + 'capacity_items': 'BIGINT NOT NULL DEFAULT 1', + 'capacity_bytes': 'BIGINT NOT NULL DEFAULT 0', + }) + _migration_add_columns(conn, 'result_reservations', { + 'cleanup_attempts': 'INTEGER NOT NULL DEFAULT 0', + 'cleanup_available_after': 'TEXT', + 'git_scan_plan_json': 'TEXT', + 'git_scan_plan_sha256': 'TEXT', + 'docker_layer_plan_json': 'TEXT', + 'docker_layer_plan_sha256': 'TEXT', + }) + _migration_add_columns(conn, 'docker_content_blobs', { + 'coverage_policy_sha256': 'TEXT', + }) + _migration_add_columns(conn, 'docker_image_blob_coverage', { + 'coverage_policy_sha256': 'TEXT', + 'selection_policy_sha256': "TEXT NOT NULL DEFAULT ''", + }) + _migration_backfill_docker_selection_policies(conn) + _migration_add_columns(conn, 'pipeline_artifacts', { + 'cleanup_attempts': 'INTEGER NOT NULL DEFAULT 0', + 'cleanup_available_after': 'TEXT', + 'cleanup_last_error': 'TEXT', + }) + conn.execute( + '''UPDATE pipeline_quarantine SET capacity_bytes = byte_count + WHERE capacity_credit_applied = 1 AND capacity_bytes = 0 AND byte_count > 0''' + ) + _migration_add_columns(conn, 'keycheck_credentials', { + 'provider_key_hash': "TEXT NOT NULL DEFAULT ''", + }) + _migration_add_columns(conn, 'keycheck_candidates', { + 'secret_hash': "TEXT NOT NULL DEFAULT ''", + 'routed_service': "TEXT NOT NULL DEFAULT ''", + 'result_projection_reserved_bytes': 'BIGINT NOT NULL DEFAULT 0', + 'result_projection_credit_transferred': 'INTEGER NOT NULL DEFAULT 0', + }) + _migration_add_columns(conn, 'docker_adaptive_shadow_reports', { + 'selection_metrics_json': "TEXT NOT NULL DEFAULT '{}'", + 'sink_checkpoint_count': 'INTEGER NOT NULL DEFAULT 0', + }) + _migration_backfill_keycheck_hashes(conn) + conn.execute( + '''CREATE UNIQUE INDEX IF NOT EXISTS uq_keycheck_credentials_provider_key + ON keycheck_credentials(service, provider_key_hash)''' + ) + + now = utc_now_iso() + conn.execute( + '''INSERT INTO pipeline_capacity(id, updated_at) VALUES (1, ?) + ON CONFLICT(id) DO NOTHING''', + (now,), + ) + for stream_name, relative_path, rotation_bytes in ( + ('scan_results', 'scan_results.jsonl', 256 * 1024 * 1024), + ('found_secrets', 'found_secrets.jsonl', 128 * 1024 * 1024), + ('scan_errors', 'scan_errors.log', 32 * 1024 * 1024), + ): + conn.execute( + '''INSERT INTO projection_streams( + stream_name, base_relative_path, current_generation, + rotation_bytes, max_generations, created_at, updated_at + ) VALUES (?, ?, 0, ?, 16, ?, ?) + ON CONFLICT(stream_name) DO NOTHING''', + (stream_name, relative_path, rotation_bytes, now, now), + ) + conn.execute( + '''INSERT INTO projection_cursors( + stream_name, generation, committed_offset, updated_at + ) VALUES (?, 0, 0, ?) + ON CONFLICT(stream_name) DO NOTHING''', + (stream_name, now), + ) + migration_code = hashlib.sha256(PIPELINE_SCHEMA_SQL.encode('utf-8')).hexdigest() + capacity_migration = conn.execute( + 'SELECT version FROM runtime_schema_migrations WHERE version = ?', + (PIPELINE_MIGRATION_VERSIONS[0],), + ).fetchone() + if not capacity_migration or not remote_capacity_applied: + _migration_reconcile_pipeline_capacity(conn) + single_writer_migration = conn.execute( + 'SELECT version FROM runtime_schema_migrations WHERE version = ?', + ('20260729_10_keycheck_projection_single_writer',), + ).fetchone() + if not single_writer_migration: + prepared_status = conn.execute( + '''SELECT a.id FROM projection_appends a + JOIN projection_jobs j ON j.id = a.job_id + WHERE j.job_kind = 'keycheck_event' + AND j.status IN ('pending','leased') + AND (j.required_stream_mask & 16) <> 0 + AND a.state = 'prepared' AND a.stream_name LIKE 'keycheck:%:status' + LIMIT 1''' + ).fetchone() + if prepared_status: + raise RuntimeSafetySchemaError( + 'keycheck status projection cutover requires reviewed prepared-append recovery' + ) + conn.execute( + '''UPDATE projection_jobs + SET required_stream_mask = required_stream_mask - 16, updated_at = ? + WHERE job_kind = 'keycheck_event' AND status IN ('pending','leased') + AND (required_stream_mask & 16) <> 0''', + (now,), + ) + for version in PIPELINE_MIGRATION_VERSIONS: + conn.execute( + '''INSERT INTO runtime_schema_migrations(version, applied_at, code_sha256) + VALUES (?, ?, ?) ON CONFLICT(version) DO NOTHING''', + (version, now, migration_code), + ) + conn.execute( + '''UPDATE runtime_schema_migrations SET applied_at = ?, code_sha256 = ? + WHERE version = ?''', + (now, migration_code, PIPELINE_MIGRATION_VERSIONS[-1]), + ) + + for table, specs in RUNTIME_TABLE_SPECS.items(): + if conn.is_postgres: + _migration_postgres_column_shape(conn, table, specs) + else: + _migration_require_sqlite_column_shape(conn, table, specs) + _migration_require_primary_key(conn, table, specs, RUNTIME_PRIMARY_KEYS[table]) + generated_id = GENERATED_ID_COLUMNS.get(table) + if generated_id: + _migration_ensure_generated_id(conn, table, generated_id) + _migration_ensure_defaults( + conn, table, specs, RUNTIME_COLUMN_DEFAULTS.get(table, {}), generated_id, + ) + + _migration_ensure_foreign_keys(conn) + _migration_seed_legacy_docker_provenance(conn) + + for table, name, columns, unique, predicate in DOCKER_DEPTH_EXPERIMENT_INDEX_SPECS: + statement = ( + f'CREATE {"UNIQUE " if unique else ""}INDEX {name} ' + f'ON {table}({", ".join(columns)})' + ) + if predicate: + statement += f' WHERE {predicate}' + _migration_ensure_index( + conn, table, name, statement, list(columns), + unique=unique, predicate=predicate, + ) + + now = utc_now_iso() + unresolved = conn.execute( + '''SELECT id, target FROM target_queue + WHERE platform = 'docker' AND status IN ('pending', 'deferred') + AND resolver_state IS NULL''' + ).fetchall() + resolver_due = datetime.fromtimestamp( + time.time() + max(60, env_int('DOCKER_RESOLVER_RETRY_SEC', 3600)), timezone.utc + ).isoformat(timespec='seconds') + for row in unresolved: + target = str(row['target'] or '').strip() + if target and '@' not in target and ':' not in target.rsplit('/', 1)[-1]: + conn.execute( + '''UPDATE target_queue SET status = 'deferred', resolver_state = 'pending', resolver_due_at = ?, + available_after = COALESCE(available_after, ?), updated_at = ? WHERE id = ?''', + (resolver_due, resolver_due, now, row['id']), + ) + conn.execute( + '''UPDATE target_queue SET + resolver_state = CASE WHEN resolver_state = 'resolving' THEN 'retry' ELSE resolver_state END, + resolver_due_at = ?, resolver_token = CASE WHEN resolver_state = 'resolving' THEN NULL ELSE resolver_token END, + available_after = COALESCE(available_after, ?), updated_at = ? + WHERE platform = 'docker' AND status = 'deferred' + AND resolver_state IN ('pending', 'retry', 'resolving') + AND resolver_due_at IS NULL''', + (resolver_due, resolver_due, now), + ) + + _migration_ensure_index( + conn, 'target_scans', 'uq_target_scans_scan_event_id', + '''CREATE UNIQUE INDEX uq_target_scans_scan_event_id + ON target_scans(scan_event_id) WHERE scan_event_id IS NOT NULL''', + ['scan_event_id'], unique=True, predicate='scan_event_id IS NOT NULL', + ) + _migration_ensure_index( + conn, 'target_scans', 'idx_target_scans_event_hash', + 'CREATE INDEX idx_target_scans_event_hash ON target_scans(scan_event_id, scan_event_hash)', + ['scan_event_id', 'scan_event_hash'], + ) + _migration_ensure_index( + conn, 'target_scans', 'idx_target_scans_queue_id', + 'CREATE INDEX idx_target_scans_queue_id ON target_scans(queue_id)', + ['queue_id'], + ) + _migration_ensure_index( + conn, 'target_scans', 'idx_target_scans_result_reservation', + '''CREATE INDEX idx_target_scans_result_reservation + ON target_scans(result_reservation_id, id)''', + ['result_reservation_id', 'id'], + ) + _migration_ensure_index( + conn, 'target_queue', 'idx_target_queue_current_reservation', + 'CREATE INDEX idx_target_queue_current_reservation ON target_queue(current_result_reservation_id)', + ['current_result_reservation_id'], + ) + _migration_ensure_index( + conn, 'target_queue', 'idx_target_queue_claimable_v2', + '''CREATE INDEX idx_target_queue_claimable_v2 + ON target_queue(source, platform, status, available_after, id) + WHERE current_result_reservation_id IS NULL + AND (status = 'pending' OR status = 'deferred')''', + ['source', 'platform', 'status', 'available_after', 'id'], + predicate="current_result_reservation_id IS NULL AND (status = 'pending' OR status = 'deferred')", + ) + _migration_ensure_index( + conn, 'discovery_retry_queue', 'uq_discovery_retry_queue_work_key', + '''CREATE UNIQUE INDEX uq_discovery_retry_queue_work_key + ON discovery_retry_queue(work_key)''', + ['work_key'], unique=True, + ) + _migration_ensure_index( + conn, 'discovery_retry_queue', 'idx_discovery_retry_queue_due', + '''CREATE INDEX idx_discovery_retry_queue_due + ON discovery_retry_queue(source, available_after, id) + WHERE status = 'pending' ''', + ['source', 'available_after', 'id'], + predicate=DISCOVERY_RETRY_DUE_INDEX_PREDICATE, + ) + _migration_ensure_index( + conn, 'discovery_retry_queue', 'idx_discovery_retry_queue_lease', + '''CREATE INDEX idx_discovery_retry_queue_lease + ON discovery_retry_queue(lease_expires_at, id) + WHERE status = 'leased' ''', + ['lease_expires_at', 'id'], + predicate=DISCOVERY_RETRY_LEASE_INDEX_PREDICATE, + ) + _migration_ensure_index( + conn, 'discovery_retry_queue', 'idx_discovery_retry_queue_policy', + '''CREATE INDEX idx_discovery_retry_queue_policy + ON discovery_retry_queue(source, query, policy_sha256, status, id)''', + ['source', 'query', 'policy_sha256', 'status', 'id'], + ) + target_queue_indexes = conn.table_indexes('target_queue') + if not any( + index['unique'] and index['columns'] == ['source', 'normalized_target'] + and not _normalized_predicate(index['predicate']) + for index in target_queue_indexes.values() + ): + try: + _migration_ensure_index( + conn, 'target_queue', 'uq_target_queue_source_normalized', + 'CREATE UNIQUE INDEX uq_target_queue_source_normalized ON target_queue(source, normalized_target)', + ['source', 'normalized_target'], unique=True, + ) + except Exception as exc: + raise RuntimeSafetySchemaError( + 'target_queue cannot be made unique on (source, normalized_target); ' + 'manually resolve duplicate rows before retrying migration' + ) from exc + _migration_ensure_index( + conn, 'scan_publication_outbox', 'uq_scan_publication_outbox_target_scan_id', + 'CREATE UNIQUE INDEX uq_scan_publication_outbox_target_scan_id ON scan_publication_outbox(target_scan_id)', + ['target_scan_id'], unique=True, + ) + _migration_ensure_index( + conn, 'keycheck_event_map', 'uq_keycheck_event_map_event_id', + 'CREATE UNIQUE INDEX uq_keycheck_event_map_event_id ON keycheck_event_map(event_id)', + ['event_id'], unique=True, + ) + _migration_ensure_index( + conn, 'finding_uid_map', 'uq_finding_uid_map_finding_uid', + 'CREATE UNIQUE INDEX uq_finding_uid_map_finding_uid ON finding_uid_map(finding_uid)', + ['finding_uid'], unique=True, + ) + package_indexes = conn.table_indexes('package_repo_candidates') + if not any( + index['unique'] + and index['columns'] == ['package_source', 'package_name', 'package_version', 'repo_url'] + and not _normalized_predicate(index['predicate']) + and _index_usable(index) + for index in package_indexes.values() + ): + try: + _migration_ensure_index( + conn, 'package_repo_candidates', 'uq_package_repo_candidates_identity', + '''CREATE UNIQUE INDEX uq_package_repo_candidates_identity + ON package_repo_candidates(package_source, package_name, package_version, repo_url)''', + ['package_source', 'package_name', 'package_version', 'repo_url'], unique=True, + ) + except Exception as exc: + raise RuntimeSafetySchemaError( + 'package_repo_candidates cannot be made unique on its package/repository identity; ' + 'manually resolve duplicate rows before retrying migration' + ) from exc + index_specs = ( + ('runs', 'idx_runs_started_at', 'started_at'), + ('runs', 'idx_runs_status', 'status'), + ('runs', 'idx_runs_selected_source_status', 'selected_source, status, id'), + ('source_cycles', 'idx_source_cycles_run_id', 'run_id'), + ('source_cycles', 'idx_source_cycles_source_query', 'source, query'), + ('source_cycles', 'idx_source_cycles_source_status', 'source, status, id'), + ('target_queue', 'idx_target_queue_source_status', 'source, status, updated_at'), + ('target_queue', 'idx_target_queue_observe_source_status', 'source, status'), + ('target_queue', 'idx_target_queue_lease', 'source, status, lease_expires_at'), + ('target_queue', 'idx_target_queue_platform_status', 'platform, status'), + ('target_queue', 'idx_target_queue_claim_batch', 'claim_batch, lease_owner'), + ('target_queue', 'idx_target_queue_claim', 'source, platform, status, available_after, lease_expires_at, id'), + ('target_queue', 'idx_target_queue_resolver_claim', 'source, platform, status, resolver_state, resolver_due_at, id'), + ('target_queue', 'idx_target_queue_source_platform_normalized', 'source, platform, normalized_target'), + ('scan_publication_outbox', 'idx_scan_publication_outbox_status', 'status, available_after, id'), + ('scan_publication_outbox', 'idx_scan_publication_outbox_age', 'status, created_at, id'), + ('target_scans', 'idx_target_scans_cycle_id', 'cycle_id'), + ('target_scans', 'idx_target_scans_source_status', 'source, status'), + ('target_scans', 'idx_target_scans_source_ended', 'source, ended_at'), + ('target_scans', 'idx_target_scans_source_ended_id', 'source, ended_at, id'), + ('target_scans', 'idx_target_scans_source_skip_ended', 'source, skipped_reason, ended_at'), + ('target_scans', 'idx_target_scans_normalized_target', 'normalized_target'), + ('keycheck_results', 'idx_keycheck_results_checked', 'checked_at'), + ('keycheck_results', 'idx_keycheck_results_service_status', 'service, status_group, status'), + ('keycheck_results', 'idx_keycheck_results_finding', 'finding_id'), + ('keycheck_results', 'idx_keycheck_results_cycle', 'cycle_id'), + ('keycheck_results', 'idx_keycheck_results_source_query', 'source, query'), + ('keycheck_results', 'idx_keycheck_results_key_hash', 'key_hash'), + ('keycheck_results', 'idx_keycheck_results_secret_hash', 'secret_hash'), + ('keycheck_results', 'idx_keycheck_results_link_repair', 'link_status, id'), + ('findings', 'idx_findings_finding_uid', 'finding_uid'), + ('findings', 'idx_findings_target_scan_id_id', 'target_scan_id, id'), + ('findings', 'idx_findings_cycle_id', 'cycle_id'), + ('findings', 'idx_findings_source', 'source'), + ('findings', 'idx_findings_secret_hash', 'secret_hash'), + ('findings', 'idx_findings_provider', 'provider'), + ('findings', 'idx_findings_confidence', 'credential_confidence'), + ('errors', 'idx_errors_cycle_id', 'cycle_id'), + ('errors', 'idx_errors_source_category', 'source, category'), + ('errors', 'idx_errors_target_scan_id', 'target_scan_id'), + ('queue_snapshots', 'idx_queue_snapshots_source_time', 'source, captured_at'), + ('package_repo_candidates', 'idx_package_repo_candidates_query', 'query'), + ('package_repo_candidates', 'idx_package_repo_candidates_repo', 'repo_url'), + ('package_repo_candidates', 'idx_package_repo_candidates_source_seen', 'package_source, last_seen_at'), + ('target_queue_reconciliation_issues', 'idx_reconciliation_issues_open', 'source_file, resolved_at, id'), + ('target_queue_policy_events', 'idx_target_queue_policy_events_queue', 'queue_id, id'), + ('target_queue_policy_events', 'idx_target_queue_policy_events_manifest', 'manifest_sha256, id'), + ('docker_content_blobs', 'idx_docker_content_blobs_reclaim', 'state, available_after, lease_expires_at, digest'), + ('docker_content_blobs', 'idx_docker_content_blobs_reservation', 'lease_reservation_id, digest'), + ('docker_image_blob_coverage', 'idx_docker_image_blob_coverage_manifest', 'manifest_digest, coverage_state, position'), + ('docker_image_blob_coverage', 'idx_docker_image_blob_coverage_blob', 'blob_digest, coverage_state, queue_id'), + ('docker_image_blob_coverage', 'idx_docker_image_blob_coverage_reservation', 'reservation_id, position'), + ( + 'docker_image_blob_coverage', + 'idx_docker_image_blob_coverage_selection', + 'queue_id, manifest_digest, selection_policy_sha256, position, reservation_id', + ), + ( + 'docker_adaptive_shadow_reports', + 'idx_docker_adaptive_shadow_reports_gate', + 'scan_policy_sha256, execution_policy_sha256, selection_policy_sha256, ' + 'state, completed_at, id', + ), + ) + for table, name, column_sql in index_specs: + columns = [value.strip() for value in column_sql.split(',')] + _migration_ensure_index( + conn, table, name, f'CREATE INDEX {name} ON {table}({column_sql})', columns, + ) + _migration_ensure_index( + conn, + 'target_scans', + 'idx_target_scans_cooldown_recent', + f'''CREATE INDEX idx_target_scans_cooldown_recent + ON target_scans(source, ended_at DESC, id DESC) + WHERE {CI_COOLDOWN_INDEX_PREDICATE}''', + ['source', 'ended_at', 'id'], + predicate=CI_COOLDOWN_INDEX_PREDICATE, + ) + for name, columns in ( + ( + 'idx_package_repo_candidates_query_seen', + 'query, last_seen_at DESC, id DESC', + ), + ( + 'idx_package_repo_candidates_recent_lookup', + 'last_seen_at DESC, id DESC, package_source, query, package_name', + ), + ): + _migration_ensure_index( + conn, + 'package_repo_candidates', + name, + f'''CREATE INDEX {name} ON package_repo_candidates({columns}) + WHERE {PACKAGE_REPO_NONEMPTY_PREDICATE}''', + [value.strip().split()[0] for value in columns.split(',')], + predicate=PACKAGE_REPO_NONEMPTY_PREDICATE, + ) + _migration_retain_pg_trgm_without_name_index(conn) + for name, columns, predicate in ( + ( + 'idx_target_queue_claim_pending', + 'source, platform, id, attempts, available_after', + "status = 'pending' AND current_result_reservation_id IS NULL AND (resolver_state IS NULL OR resolver_state = 'resolved')", + ), + ( + 'idx_target_queue_claim_deferred', + 'source, platform, available_after, id, attempts', + "status = 'deferred' AND current_result_reservation_id IS NULL AND (resolver_state IS NULL OR resolver_state = 'resolved')", + ), + ( + 'idx_target_queue_claim_in_progress', + 'source, platform, id, attempts, available_after, lease_expires_at, resolver_state', + "status = 'in_progress' AND (resolver_state IS NULL OR resolver_state = 'resolved')", + ), + ( + 'idx_target_queue_active_lease_owner_token', + 'lease_owner, lease_token', + "status = 'in_progress'", + ), + ( + 'idx_target_queue_exhausted_attempts', + 'source, platform, status, attempts, id, lease_expires_at', + "status = 'pending' OR status = 'deferred' OR status = 'in_progress'", + ), + ( + 'idx_target_queue_updated_rescan', + 'source, platform, completed_at, remote_modified_at, scan_remote_modified_at, id', + "status = 'done' AND remote_modified_at IS NOT NULL", + ), + ( + 'idx_target_queue_cold', + 'source, platform, query, id', + "status = 'cold'", + ), + ): + _migration_ensure_index( + conn, + 'target_queue', + name, + f'CREATE INDEX {name} ON target_queue({columns}) WHERE {predicate}', + [value.strip() for value in columns.split(',')], + predicate=predicate, + ) + if conn.is_postgres: + conn.execute( + '''CREATE STATISTICS IF NOT EXISTS st_target_queue_claim_selectivity + (dependencies, mcv) + ON source, platform, status, resolver_state, attempts + FROM target_queue''' + ) + conn.execute('ANALYZE target_queue') + conn.execute('ANALYZE target_scans') + conn.execute('ANALYZE findings') + conn.execute('ANALYZE package_repo_candidates') + _migration_ensure_index( + conn, 'target_queue_reconciliation_issues', 'uq_reconciliation_issue_identity', + '''CREATE UNIQUE INDEX uq_reconciliation_issue_identity + ON target_queue_reconciliation_issues( + source_file, file_identity, line_number, byte_offset, reason + )''', + ['source_file', 'file_identity', 'line_number', 'byte_offset', 'reason'], unique=True, + ) + db._runtime_safety_schema_validated = False + db.require_runtime_safety_schema(commit=False) + conn.commit() + db._docker_depth_experiment_schema_installed_cache = True + return True + except Exception: + try: + conn.rollback() + except Exception: + pass + raise + + +def first_line(value, limit=500): + for line in str(value or '').splitlines(): + line = line.strip() + if line: + return line[:limit] + return '' + + +def elapsed_seconds(start, end): + start_dt = parse_time(start) + end_dt = parse_time(end) + if not start_dt or not end_dt: + return None + return max(0.0, (end_dt - start_dt).total_seconds()) + + +KEYCHECK_RESULTS_SQL = r''' +CREATE TABLE IF NOT EXISTS keycheck_results ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + service TEXT NOT NULL, + status TEXT NOT NULL, + status_group TEXT NOT NULL, + checked_at TEXT NOT NULL, + key_hash TEXT, + secret_hash TEXT, + key_masked TEXT, + finding_id INTEGER, + target_scan_id INTEGER, + cycle_id INTEGER, + run_id INTEGER, + source TEXT, + query TEXT, + target TEXT, + detector_name TEXT, + found_at TEXT, + message TEXT, + metadata_json TEXT, + source_line TEXT, + detector_secret_hash TEXT, + event_id TEXT, + finding_uid TEXT, + link_status TEXT DEFAULT 'pending', + link_attempts INTEGER DEFAULT 0, + linked_at TEXT, + link_error TEXT, + candidate_id INTEGER, + credential_id INTEGER, + result_source TEXT NOT NULL DEFAULT 'api_check', + created_at TEXT NOT NULL, + FOREIGN KEY(finding_id) REFERENCES findings(id), + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id), + FOREIGN KEY(cycle_id) REFERENCES source_cycles(id), + FOREIGN KEY(run_id) REFERENCES runs(id), + FOREIGN KEY(candidate_id) REFERENCES keycheck_candidates(id), + FOREIGN KEY(credential_id) REFERENCES keycheck_credentials(id) +); + +CREATE TABLE IF NOT EXISTS keycheck_event_map ( + event_id TEXT PRIMARY KEY, + keycheck_result_id INTEGER, + created_at TEXT NOT NULL, + FOREIGN KEY(keycheck_result_id) REFERENCES keycheck_results(id) +); + +CREATE INDEX IF NOT EXISTS idx_keycheck_results_checked ON keycheck_results(checked_at); +CREATE INDEX IF NOT EXISTS idx_keycheck_results_service_status ON keycheck_results(service, status_group, status); +CREATE INDEX IF NOT EXISTS idx_keycheck_results_finding ON keycheck_results(finding_id); +CREATE INDEX IF NOT EXISTS idx_keycheck_results_cycle ON keycheck_results(cycle_id); +CREATE INDEX IF NOT EXISTS idx_keycheck_results_source_query ON keycheck_results(source, query); +CREATE INDEX IF NOT EXISTS idx_keycheck_results_key_hash ON keycheck_results(key_hash); +CREATE INDEX IF NOT EXISTS idx_keycheck_results_secret_hash ON keycheck_results(secret_hash); +CREATE INDEX IF NOT EXISTS idx_keycheck_results_link_repair ON keycheck_results(link_status, id); + +DROP VIEW IF EXISTS keycheck_latest_state; +CREATE VIEW IF NOT EXISTS keycheck_latest_state AS +WITH normalized AS ( + SELECT + kr.*, + COALESCE(NULLIF(kr.key_hash, ''), NULLIF(kr.secret_hash, ''), NULLIF(kr.key_masked, '')) AS key_identity, + CASE + WHEN NULLIF(kr.key_hash, '') IS NOT NULL THEN 'key_hash' + WHEN NULLIF(kr.secret_hash, '') IS NOT NULL THEN 'secret_hash' + WHEN NULLIF(kr.key_masked, '') IS NOT NULL THEN 'key_masked' + ELSE 'none' + END AS key_identity_kind + FROM keycheck_results kr +), ranked AS ( + SELECT + normalized.*, + ROW_NUMBER() OVER ( + PARTITION BY service, key_identity + ORDER BY checked_at DESC, id DESC + ) AS latest_rank + FROM normalized + WHERE key_identity IS NOT NULL +) +SELECT * +FROM ranked +WHERE latest_rank = 1; +''' + + +DOCKER_DEPTH_EXPERIMENT_SCHEMA_SQL = f''' +CREATE TABLE IF NOT EXISTS docker_depth_experiments ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_key TEXT NOT NULL, + source TEXT NOT NULL, + state TEXT NOT NULL DEFAULT 'collecting', + collection_generation TEXT NOT NULL DEFAULT 'docker-depth-provenance-v1', + config_sha256 TEXT NOT NULL, + ordered_queries_sha256 TEXT NOT NULL, + selector_version TEXT NOT NULL, + selector_sha256 TEXT NOT NULL, + provenance_policy_sha256 TEXT NOT NULL, + query_count INTEGER NOT NULL, + repositories_per_query INTEGER NOT NULL, + images_per_repository INTEGER NOT NULL, + target_limit INTEGER NOT NULL, + target_count INTEGER NOT NULL DEFAULT 0, + selection_count INTEGER NOT NULL DEFAULT 0, + fence_generation INTEGER NOT NULL DEFAULT 0, + fence_owner TEXT, + fence_token TEXT, + fence_expires_at TEXT, + hold_reason_code TEXT, + plan_sha256 TEXT, + selection_sha256 TEXT, + hold_manifest_sha256 TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + planned_at TEXT, + activated_at TEXT, + draining_at TEXT, + completed_at TEXT, + released_at TEXT, + held_at TEXT, + {_docker_depth_check_clause('docker_depth_experiments', 'docker_depth_experiments_state_check')}, + {_docker_depth_check_clause('docker_depth_experiments', 'docker_depth_experiments_range_check')}, + {_docker_depth_check_clause('docker_depth_experiments', 'docker_depth_experiments_identity_check')}, + {_docker_depth_check_clause('docker_depth_experiments', 'docker_depth_experiments_capacity_check')}, + {_docker_depth_check_clause('docker_depth_experiments', 'docker_depth_experiments_selection_check')}, + {_docker_depth_check_clause('docker_depth_experiments', 'docker_depth_experiments_fence_check')} +); + +CREATE TABLE IF NOT EXISTS docker_depth_experiment_queries ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_id INTEGER NOT NULL, + source TEXT NOT NULL, + query_ordinal INTEGER NOT NULL, + query TEXT NOT NULL, + query_sha256 TEXT NOT NULL, + required_repository_count INTEGER NOT NULL DEFAULT 10, + selected_repository_count INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + UNIQUE(experiment_id, query_ordinal), + UNIQUE(experiment_id, source, query), + UNIQUE(experiment_id, query_ordinal, source, query), + {_docker_depth_check_clause('docker_depth_experiment_queries', 'docker_depth_experiment_queries_range_check')}, + {_docker_depth_check_clause('docker_depth_experiment_queries', 'docker_depth_experiment_queries_identity_check')}, + {_docker_depth_check_clause('docker_depth_experiment_queries', 'docker_depth_experiment_queries_selection_check')}, + FOREIGN KEY(experiment_id) REFERENCES docker_depth_experiments(id) +); + +CREATE TABLE IF NOT EXISTS docker_discovery_passes ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_id INTEGER, + pass_token TEXT NOT NULL, + source TEXT NOT NULL, + pass_kind TEXT NOT NULL, + collection_generation TEXT NOT NULL DEFAULT 'docker-depth-provenance-v1', + policy_sha256 TEXT NOT NULL, + ordered_queries_sha256 TEXT NOT NULL, + expected_query_count INTEGER NOT NULL, + completed_query_count INTEGER NOT NULL DEFAULT 0, + state TEXT NOT NULL DEFAULT 'collecting', + started_at TEXT NOT NULL, + completed_at TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + {_docker_depth_check_clause('docker_discovery_passes', 'docker_discovery_passes_kind_check')}, + {_docker_depth_check_clause('docker_discovery_passes', 'docker_discovery_passes_state_check')}, + {_docker_depth_check_clause('docker_discovery_passes', 'docker_discovery_passes_count_check')}, + {_docker_depth_check_clause('docker_discovery_passes', 'docker_discovery_passes_identity_check')}, + {_docker_depth_check_clause('docker_discovery_passes', 'docker_discovery_passes_complete_check')}, + FOREIGN KEY(experiment_id) REFERENCES docker_depth_experiments(id) +); + +CREATE TABLE IF NOT EXISTS docker_discovery_pages ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + pass_id INTEGER NOT NULL, + source_cycle_id INTEGER, + retry_work_id INTEGER, + query TEXT NOT NULL, + query_ordinal INTEGER NOT NULL, + page_number INTEGER NOT NULL, + result_count INTEGER NOT NULL DEFAULT 0, + total_count INTEGER, + admitted_count INTEGER NOT NULL DEFAULT 0, + query_complete INTEGER NOT NULL DEFAULT 0, + admission_kind TEXT NOT NULL DEFAULT 'main', + page_sha256 TEXT NOT NULL, + observed_at TEXT NOT NULL, + created_at TEXT NOT NULL, + UNIQUE(pass_id, query_ordinal, page_number), + {_docker_depth_check_clause('docker_discovery_pages', 'docker_discovery_pages_range_check')}, + {_docker_depth_check_clause('docker_discovery_pages', 'docker_discovery_pages_boolean_check')}, + {_docker_depth_check_clause('docker_discovery_pages', 'docker_discovery_pages_kind_check')}, + {_docker_depth_check_clause('docker_discovery_pages', 'docker_discovery_pages_identity_check')}, + {_docker_depth_check_clause('docker_discovery_pages', 'docker_discovery_pages_evidence_check')}, + {_docker_depth_check_clause('docker_discovery_pages', 'docker_discovery_pages_total_count_check')}, + FOREIGN KEY(pass_id) REFERENCES docker_discovery_passes(id), + FOREIGN KEY(source_cycle_id) REFERENCES source_cycles(id), + FOREIGN KEY(retry_work_id) REFERENCES discovery_retry_queue(id) +); + +CREATE TABLE IF NOT EXISTS docker_repository_query_provenance ( + source TEXT NOT NULL, + query TEXT NOT NULL, + repository_queue_id INTEGER NOT NULL, + provenance_kind TEXT NOT NULL, + first_observed_at TEXT NOT NULL, + last_observed_at TEXT NOT NULL, + first_search_rank INTEGER, + best_search_rank INTEGER, + last_search_rank INTEGER, + first_cycle_id INTEGER, + last_cycle_id INTEGER, + first_page_id INTEGER, + last_page_id INTEGER, + first_policy_sha256 TEXT, + last_policy_sha256 TEXT, + observation_count INTEGER NOT NULL DEFAULT 1, + fresh_observation_count INTEGER NOT NULL DEFAULT 0, + fresh_complete_observation_count INTEGER NOT NULL DEFAULT 0, + fresh_coverage_eligible INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + PRIMARY KEY(source, query, repository_queue_id), + {_docker_depth_check_clause('docker_repository_query_provenance', 'docker_repository_query_provenance_kind_check')}, + {_docker_depth_check_clause('docker_repository_query_provenance', 'docker_repository_query_provenance_range_check')}, + {_docker_depth_check_clause('docker_repository_query_provenance', 'docker_repository_query_provenance_boolean_check')}, + {_docker_depth_check_clause('docker_repository_query_provenance', 'docker_repository_query_provenance_identity_check')}, + {_docker_depth_check_clause('docker_repository_query_provenance', 'docker_repository_query_provenance_counts_check')}, + {_docker_depth_check_clause('docker_repository_query_provenance', 'docker_repository_query_provenance_legacy_check')}, + {_docker_depth_check_clause('docker_repository_query_provenance', 'docker_repository_query_provenance_fresh_check')}, + {_docker_depth_check_clause('docker_repository_query_provenance', 'docker_repository_query_provenance_eligibility_check')}, + FOREIGN KEY(repository_queue_id) REFERENCES target_queue(id), + FOREIGN KEY(first_cycle_id) REFERENCES source_cycles(id), + FOREIGN KEY(last_cycle_id) REFERENCES source_cycles(id), + FOREIGN KEY(first_page_id) REFERENCES docker_discovery_pages(id), + FOREIGN KEY(last_page_id) REFERENCES docker_discovery_pages(id) +); + +CREATE TABLE IF NOT EXISTS docker_repository_query_observations ( + page_id INTEGER NOT NULL, + repository_queue_id INTEGER NOT NULL, + source TEXT NOT NULL, + query TEXT NOT NULL, + search_rank INTEGER NOT NULL, + observed_at TEXT NOT NULL, + PRIMARY KEY(page_id, repository_queue_id), + UNIQUE(page_id, repository_queue_id, source, query), + {_docker_depth_check_clause('docker_repository_query_observations', 'docker_repository_query_observations_rank_check')}, + FOREIGN KEY(page_id) REFERENCES docker_discovery_pages(id), + FOREIGN KEY(repository_queue_id) REFERENCES target_queue(id), + FOREIGN KEY(source, query, repository_queue_id) + REFERENCES docker_repository_query_provenance(source, query, repository_queue_id) +); + +CREATE TABLE IF NOT EXISTS docker_image_manifests ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + target_queue_id INTEGER NOT NULL, + source TEXT NOT NULL, + repository TEXT NOT NULL, + manifest_digest TEXT NOT NULL, + manifest_media_type TEXT NOT NULL, + config_digest TEXT, + graph_sha256 TEXT NOT NULL, + manifest_size_bytes INTEGER, + layer_count INTEGER NOT NULL, + resolved_at TEXT NOT NULL, + created_at TEXT NOT NULL, + UNIQUE(id, target_queue_id), + {_docker_depth_check_clause('docker_image_manifests', 'docker_image_manifests_range_check')}, + {_docker_depth_check_clause('docker_image_manifests', 'docker_image_manifests_identity_check')}, + FOREIGN KEY(target_queue_id) REFERENCES target_queue(id) +); + +CREATE TABLE IF NOT EXISTS docker_manifest_layers ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + manifest_id INTEGER NOT NULL, + position_from_base INTEGER NOT NULL, + position_from_top INTEGER NOT NULL, + layer_digest TEXT NOT NULL, + media_type TEXT NOT NULL, + layer_size_bytes INTEGER NOT NULL, + descriptor_sha256 TEXT NOT NULL, + created_at TEXT NOT NULL, + UNIQUE(id, position_from_base, position_from_top, layer_digest), + {_docker_depth_check_clause('docker_manifest_layers', 'docker_manifest_layers_range_check')}, + {_docker_depth_check_clause('docker_manifest_layers', 'docker_manifest_layers_identity_check')}, + FOREIGN KEY(manifest_id) REFERENCES docker_image_manifests(id) +); + +CREATE TABLE IF NOT EXISTS docker_depth_experiment_repositories ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_id INTEGER NOT NULL, + query_ordinal INTEGER NOT NULL, + source TEXT NOT NULL, + query TEXT NOT NULL, + repository_queue_id INTEGER NOT NULL, + eligibility_page_id INTEGER NOT NULL, + repository_rank INTEGER NOT NULL, + is_deep_probe INTEGER NOT NULL DEFAULT 0, + planned_is_deep_probe INTEGER NOT NULL DEFAULT 0, + work_state TEXT NOT NULL DEFAULT 'pending', + resolver_generation INTEGER NOT NULL DEFAULT 0, + resolver_owner TEXT, + resolver_token TEXT, + resolver_expires_at TEXT, + resolver_attempts INTEGER NOT NULL DEFAULT 0, + resolver_due_at TEXT, + candidate_distinct_graph_count INTEGER NOT NULL DEFAULT 0, + selected_image_count INTEGER NOT NULL DEFAULT 0, + replacement_repository_queue_id INTEGER, + replacement_eligibility_page_id INTEGER, + replacement_count INTEGER NOT NULL DEFAULT 0, + replacement_evidence_sha256 TEXT, + last_error_code TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + resolved_at TEXT, + UNIQUE(experiment_id, query_ordinal, id), + {_docker_depth_check_clause('docker_depth_experiment_repositories', 'docker_depth_experiment_repositories_range_check')}, + {_docker_depth_check_clause('docker_depth_experiment_repositories', 'docker_depth_experiment_repositories_boolean_check')}, + {_docker_depth_check_clause('docker_depth_experiment_repositories', 'docker_depth_experiment_repositories_state_check')}, + {_docker_depth_check_clause('docker_depth_experiment_repositories', 'docker_depth_experiment_repositories_fence_check')}, + {_docker_depth_check_clause('docker_depth_experiment_repositories', 'docker_depth_experiment_repositories_resolver_check')}, + {_docker_depth_check_clause('docker_depth_experiment_repositories', 'docker_depth_experiment_repositories_replacement_check')}, + FOREIGN KEY(experiment_id, query_ordinal, source, query) + REFERENCES docker_depth_experiment_queries( + experiment_id, query_ordinal, source, query + ), + FOREIGN KEY(repository_queue_id) REFERENCES target_queue(id), + FOREIGN KEY(source, query, repository_queue_id) + REFERENCES docker_repository_query_provenance(source, query, repository_queue_id), + FOREIGN KEY(eligibility_page_id, repository_queue_id, source, query) + REFERENCES docker_repository_query_observations( + page_id, repository_queue_id, source, query + ), + FOREIGN KEY(replacement_repository_queue_id) REFERENCES target_queue(id), + FOREIGN KEY( + replacement_eligibility_page_id, replacement_repository_queue_id, source, query + ) REFERENCES docker_repository_query_observations( + page_id, repository_queue_id, source, query + ) +); + +CREATE TABLE IF NOT EXISTS docker_depth_experiment_targets ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_id INTEGER NOT NULL, + target_queue_id INTEGER NOT NULL, + manifest_id INTEGER NOT NULL, + counter_ordinal INTEGER NOT NULL, + state TEXT NOT NULL DEFAULT 'pending', + dispatch_wave INTEGER NOT NULL, + dispatch_order INTEGER NOT NULL, + reservation_count INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + terminal_at TEXT, + UNIQUE(experiment_id, id), + {_docker_depth_check_clause('docker_depth_experiment_targets', 'docker_depth_experiment_targets_range_check')}, + {_docker_depth_check_clause('docker_depth_experiment_targets', 'docker_depth_experiment_targets_state_check')}, + FOREIGN KEY(experiment_id) REFERENCES docker_depth_experiments(id), + FOREIGN KEY(target_queue_id) REFERENCES target_queue(id), + FOREIGN KEY(manifest_id, target_queue_id) + REFERENCES docker_image_manifests(id, target_queue_id) +); + +CREATE TABLE IF NOT EXISTS docker_depth_experiment_selections ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_id INTEGER NOT NULL, + query_ordinal INTEGER NOT NULL, + experiment_repository_id INTEGER NOT NULL, + experiment_target_id INTEGER NOT NULL, + image_rank INTEGER NOT NULL, + selection_reason TEXT NOT NULL, + selection_evidence_sha256 TEXT NOT NULL, + graph_sha256 TEXT NOT NULL, + selected_at TEXT NOT NULL, + created_at TEXT NOT NULL, + {_docker_depth_check_clause('docker_depth_experiment_selections', 'docker_depth_experiment_selections_range_check')}, + {_docker_depth_check_clause('docker_depth_experiment_selections', 'docker_depth_experiment_selections_identity_check')}, + FOREIGN KEY(experiment_id, query_ordinal) + REFERENCES docker_depth_experiment_queries(experiment_id, query_ordinal), + FOREIGN KEY(experiment_id, query_ordinal, experiment_repository_id) + REFERENCES docker_depth_experiment_repositories(experiment_id, query_ordinal, id), + FOREIGN KEY(experiment_id, experiment_target_id) + REFERENCES docker_depth_experiment_targets(experiment_id, id) +); + +CREATE TABLE IF NOT EXISTS docker_depth_experiment_candidate_skips ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_id INTEGER NOT NULL, + experiment_repository_id INTEGER NOT NULL, + repository_queue_id INTEGER NOT NULL, + candidate_kind TEXT NOT NULL, + candidate_ordinal INTEGER NOT NULL, + reason_code TEXT NOT NULL, + candidate_identity_sha256 TEXT NOT NULL, + evidence_sha256 TEXT NOT NULL, + created_at TEXT NOT NULL, + {_docker_depth_check_clause('docker_depth_experiment_candidate_skips', 'docker_depth_experiment_candidate_skips_kind_check')}, + {_docker_depth_check_clause('docker_depth_experiment_candidate_skips', 'docker_depth_experiment_candidate_skips_range_check')}, + {_docker_depth_check_clause('docker_depth_experiment_candidate_skips', 'docker_depth_experiment_candidate_skips_identity_check')}, + FOREIGN KEY(experiment_id) REFERENCES docker_depth_experiments(id), + FOREIGN KEY(experiment_repository_id) + REFERENCES docker_depth_experiment_repositories(id), + FOREIGN KEY(repository_queue_id) REFERENCES target_queue(id) +); + +CREATE TABLE IF NOT EXISTS docker_depth_resolver_attempt_refunds ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_id INTEGER NOT NULL, + experiment_repository_id INTEGER NOT NULL, + repository_queue_id INTEGER NOT NULL, + query_ordinal INTEGER NOT NULL, + repository_rank INTEGER NOT NULL, + recovery_kind TEXT NOT NULL, + manifest_sha256 TEXT NOT NULL, + entry_evidence_sha256 TEXT NOT NULL, + log_sha256 TEXT NOT NULL, + target_identity_sha256 TEXT NOT NULL, + prior_error_code_sha256 TEXT NOT NULL, + prior_work_state TEXT NOT NULL, + next_work_state TEXT NOT NULL, + prior_resolver_attempts INTEGER NOT NULL, + refund_attempts INTEGER NOT NULL, + next_resolver_attempts INTEGER NOT NULL, + confirmed_bug_event_count INTEGER NOT NULL, + applied_at TEXT NOT NULL, + created_at TEXT NOT NULL, + {_docker_depth_check_clause('docker_depth_resolver_attempt_refunds', 'docker_depth_resolver_attempt_refunds_kind_check')}, + {_docker_depth_check_clause('docker_depth_resolver_attempt_refunds', 'docker_depth_resolver_attempt_refunds_state_check')}, + {_docker_depth_check_clause('docker_depth_resolver_attempt_refunds', 'docker_depth_resolver_attempt_refunds_range_check')}, + {_docker_depth_check_clause('docker_depth_resolver_attempt_refunds', 'docker_depth_resolver_attempt_refunds_hash_check')}, + FOREIGN KEY(experiment_id) REFERENCES docker_depth_experiments(id), + FOREIGN KEY(experiment_repository_id) + REFERENCES docker_depth_experiment_repositories(id), + FOREIGN KEY(repository_queue_id) REFERENCES target_queue(id) +); + +CREATE TABLE IF NOT EXISTS docker_depth_resolver_dispositions ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_id INTEGER NOT NULL, + experiment_repository_id INTEGER NOT NULL, + prior_repository_queue_id INTEGER NOT NULL, + replacement_repository_queue_id INTEGER, + replacement_eligibility_page_id INTEGER, + replacement_best_search_rank INTEGER, + query_ordinal INTEGER NOT NULL, + repository_rank INTEGER NOT NULL, + disposition_kind TEXT NOT NULL, + outcome TEXT NOT NULL, + manifest_sha256 TEXT NOT NULL, + entry_evidence_sha256 TEXT NOT NULL, + candidate_snapshot_sha256 TEXT NOT NULL, + prior_target_identity_sha256 TEXT NOT NULL, + replacement_target_identity_sha256 TEXT, + prior_error_code_sha256 TEXT NOT NULL, + prior_work_state TEXT NOT NULL, + next_work_state TEXT NOT NULL, + prior_resolver_attempts INTEGER NOT NULL, + next_resolver_attempts INTEGER NOT NULL, + applied_at TEXT NOT NULL, + created_at TEXT NOT NULL, + {_docker_depth_check_clause('docker_depth_resolver_dispositions', 'docker_depth_resolver_dispositions_kind_check')}, + {_docker_depth_check_clause('docker_depth_resolver_dispositions', 'docker_depth_resolver_dispositions_outcome_check')}, + {_docker_depth_check_clause('docker_depth_resolver_dispositions', 'docker_depth_resolver_dispositions_state_check')}, + {_docker_depth_check_clause('docker_depth_resolver_dispositions', 'docker_depth_resolver_dispositions_range_check')}, + {_docker_depth_check_clause('docker_depth_resolver_dispositions', 'docker_depth_resolver_dispositions_hash_check')}, + FOREIGN KEY(experiment_id) REFERENCES docker_depth_experiments(id), + FOREIGN KEY(experiment_repository_id) + REFERENCES docker_depth_experiment_repositories(id), + FOREIGN KEY(prior_repository_queue_id) REFERENCES target_queue(id), + FOREIGN KEY(replacement_repository_queue_id) REFERENCES target_queue(id), + FOREIGN KEY(replacement_eligibility_page_id) REFERENCES docker_discovery_pages(id) +); + +CREATE TABLE IF NOT EXISTS docker_depth_experiment_scan_bindings ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_target_id INTEGER NOT NULL, + reservation_id INTEGER NOT NULL, + target_scan_id INTEGER, + attempt INTEGER NOT NULL, + state TEXT NOT NULL DEFAULT 'reserved', + bound_at TEXT NOT NULL, + scan_bound_at TEXT, + completed_at TEXT, + created_at TEXT NOT NULL, + {_docker_depth_check_clause('docker_depth_experiment_scan_bindings', 'docker_depth_experiment_scan_bindings_range_check')}, + {_docker_depth_check_clause('docker_depth_experiment_scan_bindings', 'docker_depth_experiment_scan_bindings_state_check')}, + FOREIGN KEY(experiment_target_id) REFERENCES docker_depth_experiment_targets(id), + FOREIGN KEY(reservation_id) REFERENCES result_reservations(id), + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id) +); + +CREATE TABLE IF NOT EXISTS docker_finding_layer_attributions ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + scan_binding_id INTEGER NOT NULL, + finding_id INTEGER NOT NULL, + manifest_layer_id INTEGER, + attribution_state TEXT NOT NULL, + reported_layer_digest TEXT, + unattributed_reason TEXT, + position_from_base INTEGER, + position_from_top INTEGER, + created_at TEXT NOT NULL, + {_docker_depth_check_clause('docker_finding_layer_attributions', 'docker_finding_layer_attributions_state_check')}, + {_docker_depth_check_clause('docker_finding_layer_attributions', 'docker_finding_layer_attributions_evidence_check')}, + {_docker_depth_check_clause('docker_finding_layer_attributions', 'docker_finding_layer_attributions_reason_check')}, + FOREIGN KEY(scan_binding_id) REFERENCES docker_depth_experiment_scan_bindings(id), + FOREIGN KEY(finding_id) REFERENCES findings(id), + FOREIGN KEY( + manifest_layer_id, position_from_base, position_from_top, reported_layer_digest + ) REFERENCES docker_manifest_layers( + id, position_from_base, position_from_top, layer_digest + ) +); + +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_depth_experiments_key +ON docker_depth_experiments(experiment_key); +CREATE INDEX IF NOT EXISTS idx_docker_depth_experiments_state +ON docker_depth_experiments(source, state, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_depth_experiment_queries_ordinal +ON docker_depth_experiment_queries(experiment_id, query_ordinal); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_depth_experiment_queries_query +ON docker_depth_experiment_queries(experiment_id, source, query); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_discovery_passes_token +ON docker_discovery_passes(pass_token); +CREATE INDEX IF NOT EXISTS idx_docker_discovery_passes_policy +ON docker_discovery_passes(source, policy_sha256, state, id); +CREATE INDEX IF NOT EXISTS idx_docker_discovery_passes_experiment +ON docker_discovery_passes(experiment_id, state, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_discovery_pages_position +ON docker_discovery_pages(pass_id, query_ordinal, page_number); +CREATE INDEX IF NOT EXISTS idx_docker_discovery_pages_query +ON docker_discovery_pages(pass_id, query, query_complete, page_number); +CREATE INDEX IF NOT EXISTS idx_docker_discovery_pages_retry +ON docker_discovery_pages(retry_work_id, id) WHERE retry_work_id IS NOT NULL; +CREATE INDEX IF NOT EXISTS idx_docker_repository_provenance_queue +ON docker_repository_query_provenance(repository_queue_id, source, query); +CREATE INDEX IF NOT EXISTS idx_docker_repository_provenance_eligible +ON docker_repository_query_provenance( + source, query, fresh_coverage_eligible, best_search_rank, repository_queue_id +); +CREATE INDEX IF NOT EXISTS idx_docker_repository_observations_query +ON docker_repository_query_observations(source, query, repository_queue_id, page_id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_image_manifests_queue +ON docker_image_manifests(target_queue_id); +CREATE INDEX IF NOT EXISTS idx_docker_image_manifests_digest +ON docker_image_manifests(manifest_digest, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_manifest_layers_base_position +ON docker_manifest_layers(manifest_id, position_from_base); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_manifest_layers_top_position +ON docker_manifest_layers(manifest_id, position_from_top); +CREATE INDEX IF NOT EXISTS idx_docker_manifest_layers_digest +ON docker_manifest_layers(layer_digest, manifest_id, position_from_base); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_experiment_repositories_member +ON docker_depth_experiment_repositories(experiment_id, query_ordinal, repository_queue_id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_experiment_repositories_rank +ON docker_depth_experiment_repositories(experiment_id, query_ordinal, repository_rank); +CREATE INDEX IF NOT EXISTS idx_docker_experiment_repository_work +ON docker_depth_experiment_repositories( + experiment_id, work_state, repository_rank, query_ordinal, id +); +CREATE INDEX IF NOT EXISTS idx_docker_experiment_repository_due +ON docker_depth_experiment_repositories( + experiment_id, work_state, resolver_due_at, repository_rank, query_ordinal, id +); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_experiment_targets_queue +ON docker_depth_experiment_targets(experiment_id, target_queue_id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_experiment_targets_counter +ON docker_depth_experiment_targets(experiment_id, counter_ordinal); +CREATE INDEX IF NOT EXISTS idx_docker_experiment_targets_dispatch +ON docker_depth_experiment_targets(experiment_id, state, dispatch_wave, dispatch_order, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_experiment_selections_rank +ON docker_depth_experiment_selections(experiment_repository_id, image_rank); +CREATE INDEX IF NOT EXISTS idx_docker_experiment_selections_query +ON docker_depth_experiment_selections( + experiment_id, query_ordinal, image_rank, experiment_repository_id, id +); +CREATE INDEX IF NOT EXISTS idx_docker_experiment_selections_target +ON docker_depth_experiment_selections(experiment_target_id, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_experiment_candidate_skip +ON docker_depth_experiment_candidate_skips( + experiment_repository_id, candidate_kind, candidate_identity_sha256 +); +CREATE INDEX IF NOT EXISTS idx_docker_experiment_candidate_skips +ON docker_depth_experiment_candidate_skips( + experiment_id, experiment_repository_id, candidate_kind, id +); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_depth_resolver_attempt_refund +ON docker_depth_resolver_attempt_refunds(experiment_repository_id, recovery_kind); +CREATE INDEX IF NOT EXISTS idx_docker_depth_resolver_attempt_refunds +ON docker_depth_resolver_attempt_refunds(experiment_id, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_depth_resolver_disposition +ON docker_depth_resolver_dispositions(experiment_repository_id, disposition_kind); +CREATE INDEX IF NOT EXISTS idx_docker_depth_resolver_dispositions +ON docker_depth_resolver_dispositions(experiment_id, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_experiment_bindings_reservation +ON docker_depth_experiment_scan_bindings(reservation_id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_experiment_bindings_attempt +ON docker_depth_experiment_scan_bindings(experiment_target_id, attempt); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_experiment_bindings_scan +ON docker_depth_experiment_scan_bindings(target_scan_id) WHERE target_scan_id IS NOT NULL; +CREATE INDEX IF NOT EXISTS idx_docker_experiment_bindings_target +ON docker_depth_experiment_scan_bindings(experiment_target_id, state, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_finding_layer_exact +ON docker_finding_layer_attributions(finding_id, manifest_layer_id) WHERE attribution_state = 'exact'; +CREATE UNIQUE INDEX IF NOT EXISTS uq_docker_finding_layer_unattributed +ON docker_finding_layer_attributions(finding_id) WHERE attribution_state = 'unattributed'; +CREATE INDEX IF NOT EXISTS idx_docker_finding_layer_binding +ON docker_finding_layer_attributions(scan_binding_id, finding_id, id); +CREATE INDEX IF NOT EXISTS idx_target_queue_policy_events_experiment +ON target_queue_policy_events(experiment_id, id) WHERE experiment_id IS NOT NULL; +''' + + +PIPELINE_SCHEMA_SQL = f''' +CREATE TABLE IF NOT EXISTS runtime_schema_migrations ( + version TEXT PRIMARY KEY, + applied_at TEXT NOT NULL, + code_sha256 TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS discovery_retry_queue ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + work_key TEXT NOT NULL, + source TEXT NOT NULL, + query TEXT NOT NULL, + source_cycle_id INTEGER, + policy_sha256 TEXT NOT NULL, + pass_kind TEXT NOT NULL, + work_kind TEXT NOT NULL, + page_start INTEGER NOT NULL DEFAULT 1, + page_end INTEGER NOT NULL DEFAULT 1, + next_page INTEGER NOT NULL DEFAULT 1, + status TEXT NOT NULL DEFAULT 'pending', + attempts INTEGER NOT NULL DEFAULT 0, + available_after TEXT, + lease_owner TEXT, + lease_token TEXT, + leased_at TEXT, + lease_expires_at TEXT, + last_error_category TEXT, + held_at TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + CONSTRAINT discovery_retry_queue_identity_check CHECK ( + length(work_key) = 64 AND work_key = lower(work_key) + AND length(source) BETWEEN 1 AND 64 + AND length(query) BETWEEN 1 AND 256 + AND length(policy_sha256) = 64 AND policy_sha256 = lower(policy_sha256) + AND pass_kind IN ('ordinary','deep') + AND work_kind IN ('query','page','range') + ), + CONSTRAINT discovery_retry_queue_page_check CHECK ( + page_start BETWEEN 1 AND 30 + AND page_end BETWEEN page_start AND 30 + AND next_page BETWEEN page_start AND page_end + AND (work_kind <> 'query' OR page_start = 1) + AND (work_kind <> 'page' OR page_start = page_end) + ), + CONSTRAINT discovery_retry_queue_status_check CHECK ( + status IN ('pending','leased','held') + AND attempts BETWEEN 0 AND 1000000 + ), + CONSTRAINT discovery_retry_queue_error_check CHECK ( + last_error_category IS NULL OR last_error_category IN ( + 'account_pool_exhausted','auth_forbidden','auth_invalid','auth_unavailable', + 'invalid_payload','network','page_unavailable','policy_mismatch', + 'provider_cooldown','provider_unavailable','query_removed','rate_limit', + 'remote_transient','request_failed','tail_unavailable','transport' + ) + ), + {_docker_depth_check_clause('discovery_retry_queue', 'discovery_retry_queue_lifecycle_check')}, + {_docker_depth_check_clause('discovery_retry_queue', 'discovery_retry_queue_sha256_check')}, + FOREIGN KEY(source_cycle_id) REFERENCES source_cycles(id) +); + +CREATE UNIQUE INDEX IF NOT EXISTS uq_discovery_retry_queue_work_key +ON discovery_retry_queue(work_key); +CREATE INDEX IF NOT EXISTS idx_discovery_retry_queue_due +ON discovery_retry_queue(source, available_after, id) WHERE status = 'pending'; +CREATE INDEX IF NOT EXISTS idx_discovery_retry_queue_lease +ON discovery_retry_queue(lease_expires_at, id) WHERE status = 'leased'; +CREATE INDEX IF NOT EXISTS idx_discovery_retry_queue_policy +ON discovery_retry_queue(source, query, policy_sha256, status, id); + +CREATE TABLE IF NOT EXISTS runtime_final_cutover ( + id INTEGER PRIMARY KEY CHECK (id = 1), + marker TEXT NOT NULL, + checked_at TEXT NOT NULL, + evidence_sha256 TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS runtime_operations ( + operation_id TEXT PRIMARY KEY CHECK (length(operation_id) = 36), + actor TEXT NOT NULL CHECK (length(actor) BETWEEN 1 AND 256), + action TEXT NOT NULL CHECK (length(action) BETWEEN 1 AND 128), + target_kind TEXT NOT NULL CHECK (length(target_kind) BETWEEN 1 AND 64), + target_ref TEXT NOT NULL DEFAULT '' CHECK (length(target_ref) <= 512), + status TEXT NOT NULL DEFAULT 'requested' CHECK ( + status IN ('requested','running','succeeded','failed','rolled_back','failed_hold','canceled') + ), + safe_category TEXT CHECK (safe_category IS NULL OR length(safe_category) <= 128), + safe_detail TEXT CHECK (safe_detail IS NULL OR length(safe_detail) <= 2048), + expected_revision INTEGER CHECK (expected_revision IS NULL OR expected_revision >= 0), + resulting_revision INTEGER CHECK (resulting_revision IS NULL OR resulting_revision >= 0), + expected_identity_json TEXT NOT NULL DEFAULT '{{}}' + CHECK (length(expected_identity_json) <= 65536), + resulting_identity_json TEXT CHECK ( + resulting_identity_json IS NULL OR length(resulting_identity_json) <= 65536 + ), + agent_state TEXT NOT NULL DEFAULT 'not_required' CHECK ( + agent_state IN ( + 'not_required','pending','running','succeeded','failed','rolled_back','failed_hold' + ) + ), + agent_result_sha256 TEXT CHECK ( + agent_result_sha256 IS NULL OR ( + length(agent_result_sha256) = 64 AND agent_result_sha256 = lower(agent_result_sha256) + ) + ), + requested_at TEXT NOT NULL, + started_at TEXT, + completed_at TEXT, + agent_reconciled_at TEXT, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS runtime_operations_control ( + id INTEGER PRIMARY KEY CHECK (id = 1), + revision INTEGER NOT NULL DEFAULT 0 CHECK (revision >= 0), + discovery_paused INTEGER NOT NULL DEFAULT 0 CHECK (discovery_paused IN (0,1)), + dispatch_paused INTEGER NOT NULL DEFAULT 0 CHECK (dispatch_paused IN (0,1)), + drain_state TEXT NOT NULL DEFAULT 'normal' + CHECK (drain_state IN ('normal','draining','drained')), + actor TEXT NOT NULL CHECK (length(actor) BETWEEN 1 AND 256), + operation_id TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + FOREIGN KEY(operation_id) REFERENCES runtime_operations(operation_id) +); + +CREATE TABLE IF NOT EXISTS runtime_audit_events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + operation_id TEXT, + actor TEXT NOT NULL CHECK (length(actor) BETWEEN 1 AND 256), + action TEXT NOT NULL CHECK (length(action) BETWEEN 1 AND 128), + target_kind TEXT NOT NULL CHECK (length(target_kind) BETWEEN 1 AND 64), + target_ref TEXT NOT NULL DEFAULT '' CHECK (length(target_ref) <= 512), + result TEXT NOT NULL CHECK ( + result IN ( + 'accepted','succeeded','rejected','failed','rolled_back','canceled','failed_hold' + ) + ), + safe_category TEXT CHECK (safe_category IS NULL OR length(safe_category) <= 128), + before_identity_json TEXT CHECK ( + before_identity_json IS NULL OR length(before_identity_json) <= 65536 + ), + after_identity_json TEXT CHECK ( + after_identity_json IS NULL OR length(after_identity_json) <= 65536 + ), + before_bytes INTEGER CHECK (before_bytes IS NULL OR before_bytes >= 0), + after_bytes INTEGER CHECK (after_bytes IS NULL OR after_bytes >= 0), + previous_event_id INTEGER UNIQUE, + previous_event_sha256 TEXT, + event_sha256 TEXT NOT NULL UNIQUE CHECK ( + length(event_sha256) = 64 AND event_sha256 = lower(event_sha256) + ), + created_at TEXT NOT NULL, + CONSTRAINT runtime_audit_events_chain_check CHECK ( + (previous_event_id IS NULL AND previous_event_sha256 IS NULL) + OR (previous_event_id IS NOT NULL AND length(previous_event_sha256) = 64 + AND previous_event_sha256 = lower(previous_event_sha256)) + ), + FOREIGN KEY(operation_id) REFERENCES runtime_operations(operation_id), + FOREIGN KEY(previous_event_id) REFERENCES runtime_audit_events(id) +); + +CREATE TABLE IF NOT EXISTS pipeline_capacity ( + id INTEGER PRIMARY KEY CHECK (id = 1), + bundle_items BIGINT NOT NULL DEFAULT 0 CHECK (bundle_items >= 0), + bundle_bytes BIGINT NOT NULL DEFAULT 0 CHECK (bundle_bytes >= 0), + projection_items BIGINT NOT NULL DEFAULT 0 CHECK (projection_items >= 0), + projection_bytes BIGINT NOT NULL DEFAULT 0 CHECK (projection_bytes >= 0), + keycheck_items BIGINT NOT NULL DEFAULT 0 CHECK (keycheck_items >= 0), + keycheck_bytes BIGINT NOT NULL DEFAULT 0 CHECK (keycheck_bytes >= 0), + quarantine_items BIGINT NOT NULL DEFAULT 0 CHECK (quarantine_items >= 0), + quarantine_bytes BIGINT NOT NULL DEFAULT 0 CHECK (quarantine_bytes >= 0), + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS remote_worker_users ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + user_key TEXT NOT NULL UNIQUE, + active_assignment_cap INTEGER NOT NULL DEFAULT 0 + CHECK (active_assignment_cap BETWEEN 0 AND 10000), + disabled_at TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS remote_worker_devices ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + user_id INTEGER NOT NULL, + device_key TEXT NOT NULL UNIQUE, + token_sha256 TEXT NOT NULL UNIQUE CHECK ( + length(token_sha256) = 64 AND token_sha256 = lower(token_sha256) + ), + revoked_at TEXT, + last_contact_at TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + UNIQUE(id, user_id), + FOREIGN KEY(user_id) REFERENCES remote_worker_users(id) +); + +CREATE TABLE IF NOT EXISTS admission_intents ( + reservation_token TEXT PRIMARY KEY, + intent_sha256 TEXT NOT NULL, + state TEXT NOT NULL CHECK (state IN ('pending','committed','aborted')), + reservation_id BIGINT, + resolution_detail TEXT, + remote_user_id INTEGER, + remote_device_id INTEGER, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + resolved_at TEXT, + FOREIGN KEY(reservation_id) REFERENCES result_reservations(id), + FOREIGN KEY(remote_device_id, remote_user_id) + REFERENCES remote_worker_devices(id, user_id) +); + +CREATE TABLE IF NOT EXISTS admission_intent_retirement ( + id INTEGER PRIMARY KEY CHECK (id = 1), + retired_count BIGINT NOT NULL DEFAULT 0, + chain_sha256 TEXT NOT NULL, + cursor_token TEXT NOT NULL DEFAULT '', + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS pipeline_artifact_retirement ( + id INTEGER PRIMARY KEY CHECK (id = 1), + retired_count BIGINT NOT NULL DEFAULT 0, + chain_sha256 TEXT NOT NULL, + cursor_id BIGINT NOT NULL DEFAULT 0, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS result_reservations ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + reservation_token TEXT NOT NULL UNIQUE, + bundle_id TEXT NOT NULL UNIQUE, + scan_event_id TEXT NOT NULL UNIQUE, + queue_id INTEGER NOT NULL, + run_id INTEGER, + cycle_id INTEGER, + source TEXT NOT NULL, + platform TEXT NOT NULL, + query TEXT, + target TEXT NOT NULL, + normalized_target TEXT NOT NULL, + claim_lease_owner TEXT NOT NULL, + claim_lease_token TEXT NOT NULL, + claim_batch TEXT, + producer_instance_id TEXT NOT NULL, + producer_pid BIGINT NOT NULL, + producer_creation_time TEXT NOT NULL, + producer_executable TEXT NOT NULL, + assignment_kind TEXT NOT NULL DEFAULT 'local' + CHECK (assignment_kind IN ('local','remote')), + remote_user_id INTEGER, + remote_device_id INTEGER, + remote_issued_at TEXT, + remote_expires_at TEXT, + remote_result_upload_body_timeout_seconds INTEGER CHECK ( + remote_result_upload_body_timeout_seconds IS NULL + OR remote_result_upload_body_timeout_seconds BETWEEN 30 AND 86400 + ), + remote_effective_config_sha256 TEXT, + remote_client_compat_sha256 TEXT, + remote_execution_snapshot_json TEXT CHECK ( + remote_execution_snapshot_json IS NULL OR length(remote_execution_snapshot_json) <= 65536 + ), + remote_execution_snapshot_sha256 TEXT, + remote_resolution_kind TEXT CHECK ( + remote_resolution_kind IS NULL OR remote_resolution_kind IN ( + 'bundle_accepted','prebundle_report','expired' + ) + ), + remote_payload_sha256 TEXT, + remote_receipt_id TEXT, + remote_resolution_json TEXT CHECK ( + remote_resolution_json IS NULL OR length(remote_resolution_json) <= 16384 + ), + remote_resolved_at TEXT, + remote_diagnostic_projection_version INTEGER CHECK ( + remote_diagnostic_projection_version IS NULL + OR remote_diagnostic_projection_version IN (0,1) + ), + remote_diagnostic_count INTEGER CHECK ( + remote_diagnostic_count IS NULL OR remote_diagnostic_count >= 0 + ), + remote_diagnostic_uids_sha256 TEXT, + declared_bundle_bytes BIGINT NOT NULL CHECK (declared_bundle_bytes > 0), + reserved_bundle_bytes BIGINT NOT NULL CHECK (reserved_bundle_bytes > 0), + reserved_projection_items BIGINT NOT NULL DEFAULT 1, + reserved_projection_bytes BIGINT NOT NULL, + reserved_candidate_items BIGINT NOT NULL, + reserved_candidate_bytes BIGINT NOT NULL, + ready_relative_path TEXT NOT NULL, + state TEXT NOT NULL CHECK ( + state IN ('scanning','ready','ingesting','db_committed','acknowledged','refunded','quarantined') + ), + producer_lease_expires_at TEXT NOT NULL, + bundle_credit_released INTEGER NOT NULL DEFAULT 0, + projection_credit_transferred INTEGER NOT NULL DEFAULT 0, + candidate_credit_transferred INTEGER NOT NULL DEFAULT 0, + last_error_code TEXT, + last_error_detail TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + refunded_at TEXT, + released_at TEXT, + cleanup_attempts INTEGER NOT NULL DEFAULT 0, + cleanup_available_after TEXT, + git_scan_plan_json TEXT, + git_scan_plan_sha256 TEXT, + docker_layer_plan_json TEXT, + docker_layer_plan_sha256 TEXT, + FOREIGN KEY(queue_id) REFERENCES target_queue(id), + FOREIGN KEY(run_id) REFERENCES runs(id), + FOREIGN KEY(cycle_id) REFERENCES source_cycles(id), + FOREIGN KEY(remote_device_id, remote_user_id) + REFERENCES remote_worker_devices(id, user_id) +); + +CREATE TABLE IF NOT EXISTS worker_progress_events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + reservation_id INTEGER NOT NULL, + remote_device_id INTEGER NOT NULL, + schema_version INTEGER NOT NULL CHECK (schema_version > 0), + sequence INTEGER NOT NULL CHECK (sequence > 0), + event_type TEXT NOT NULL CHECK (length(event_type) BETWEEN 1 AND 128), + phase TEXT NOT NULL CHECK (length(phase) BETWEEN 1 AND 64), + event_timestamp TEXT NOT NULL, + phase_started_at TEXT, + instance_id TEXT NOT NULL CHECK (length(instance_id) BETWEEN 1 AND 256), + slot_id INTEGER NOT NULL CHECK (slot_id >= 0), + source TEXT NOT NULL CHECK (length(source) BETWEEN 1 AND 64), + event_json TEXT NOT NULL CHECK (length(event_json) BETWEEN 2 AND 65536), + event_sha256 TEXT NOT NULL CHECK ( + length(event_sha256) = 64 AND event_sha256 = lower(event_sha256) + ), + received_at TEXT NOT NULL, + FOREIGN KEY(reservation_id) REFERENCES result_reservations(id), + FOREIGN KEY(remote_device_id) REFERENCES remote_worker_devices(id) +); + +CREATE TABLE IF NOT EXISTS worker_diagnostics ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + diagnostic_uid TEXT NOT NULL CHECK (length(diagnostic_uid) BETWEEN 1 AND 256), + reservation_id INTEGER NOT NULL, + target_scan_id INTEGER, + schema_version INTEGER NOT NULL CHECK (schema_version > 0), + scan_event_id TEXT CHECK ( + scan_event_id IS NULL OR ( + length(scan_event_id) BETWEEN 32 AND 64 + AND scan_event_id = lower(scan_event_id) + ) + ), + slot_id INTEGER NOT NULL CHECK (slot_id >= 0), + attempt INTEGER NOT NULL CHECK (attempt > 0), + source TEXT NOT NULL CHECK (length(source) BETWEEN 1 AND 64), + phase TEXT NOT NULL CHECK (length(phase) BETWEEN 1 AND 64), + kind TEXT NOT NULL CHECK (length(kind) BETWEEN 1 AND 64), + category TEXT NOT NULL CHECK (length(category) BETWEEN 1 AND 128), + code TEXT NOT NULL CHECK (length(code) BETWEEN 1 AND 256), + summary TEXT NOT NULL CHECK (length(summary) BETWEEN 1 AND 2048), + retryable INTEGER NOT NULL CHECK (retryable IN (0,1)), + occurred_at TEXT NOT NULL, + captured_at TEXT NOT NULL, + envelope_json TEXT NOT NULL CHECK (length(envelope_json) BETWEEN 2 AND 65536), + envelope_sha256 TEXT NOT NULL CHECK ( + length(envelope_sha256) = 64 AND envelope_sha256 = lower(envelope_sha256) + ), + body_payload_json TEXT CHECK ( + body_payload_json IS NULL OR length(body_payload_json) <= 65536 + ), + log_payload_json TEXT CHECK ( + log_payload_json IS NULL OR length(log_payload_json) <= 65536 + ), + received_at TEXT NOT NULL, + FOREIGN KEY(reservation_id) REFERENCES result_reservations(id), + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id) +); + +CREATE TABLE IF NOT EXISTS docker_content_blobs ( + digest TEXT NOT NULL, + coverage_policy_sha256 TEXT NOT NULL, + descriptor_kind TEXT NOT NULL CHECK (descriptor_kind IN ('config','layer')), + declared_bytes INTEGER NOT NULL CHECK (declared_bytes >= 0), + media_type TEXT NOT NULL, + state TEXT NOT NULL DEFAULT 'pending' + CHECK (state IN ('pending','leased','submitted','covered','failed')), + attempts INTEGER NOT NULL DEFAULT 0 CHECK (attempts >= 0), + max_attempts INTEGER NOT NULL CHECK (max_attempts > 0), + available_after TEXT, + lease_reservation_id INTEGER, + lease_token TEXT, + lease_plan_sha256 TEXT, + lease_expires_at TEXT, + covered_reservation_id INTEGER, + covered_scan_event_id TEXT, + covered_policy_sha256 TEXT, + verified_bytes INTEGER, + covered_at TEXT, + last_error_code TEXT, + last_error_detail TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + PRIMARY KEY(digest, coverage_policy_sha256), + FOREIGN KEY(lease_reservation_id) REFERENCES result_reservations(id), + FOREIGN KEY(covered_reservation_id) REFERENCES result_reservations(id) +); + +CREATE TABLE IF NOT EXISTS docker_image_blob_coverage ( + queue_id INTEGER NOT NULL, + manifest_digest TEXT NOT NULL, + position INTEGER NOT NULL CHECK (position >= 0), + blob_digest TEXT NOT NULL, + coverage_policy_sha256 TEXT NOT NULL, + selection_policy_sha256 TEXT NOT NULL DEFAULT '', + descriptor_kind TEXT NOT NULL CHECK (descriptor_kind IN ('config','layer')), + plan_sha256 TEXT NOT NULL, + reservation_id INTEGER NOT NULL, + selected INTEGER NOT NULL CHECK (selected IN (0,1)), + selection_reason TEXT NOT NULL, + coverage_state TEXT NOT NULL CHECK ( + coverage_state IN ( + 'selected','leased','shared_pending','covered', + 'retryable_failed','terminal_failed','skipped' + ) + ), + covered_at TEXT, + last_error_code TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + PRIMARY KEY(reservation_id, position), + FOREIGN KEY(queue_id) REFERENCES target_queue(id), + FOREIGN KEY(blob_digest, coverage_policy_sha256) + REFERENCES docker_content_blobs(digest, coverage_policy_sha256), + FOREIGN KEY(reservation_id) REFERENCES result_reservations(id) +); + +CREATE TABLE IF NOT EXISTS docker_adaptive_shadow_reports ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + report_token TEXT NOT NULL UNIQUE, + evaluator_version TEXT NOT NULL, + state TEXT NOT NULL DEFAULT 'running' + CHECK (state IN ('running','completed','failed')), + scan_policy_sha256 TEXT NOT NULL, + execution_policy_sha256 TEXT NOT NULL, + selection_policy_sha256 TEXT NOT NULL, + cohort_size INTEGER NOT NULL CHECK (cohort_size >= 50 AND cohort_size <= 100), + completed_pairs INTEGER NOT NULL DEFAULT 0 CHECK (completed_pairs >= 0), + full_routed_count INTEGER NOT NULL DEFAULT 0 CHECK (full_routed_count >= 0), + adaptive_routed_count INTEGER NOT NULL DEFAULT 0 CHECK (adaptive_routed_count >= 0), + routed_intersection_count INTEGER NOT NULL DEFAULT 0 CHECK (routed_intersection_count >= 0), + full_detector_count INTEGER NOT NULL DEFAULT 0 CHECK (full_detector_count >= 0), + adaptive_detector_count INTEGER NOT NULL DEFAULT 0 CHECK (adaptive_detector_count >= 0), + detector_intersection_count INTEGER NOT NULL DEFAULT 0 + CHECK (detector_intersection_count >= 0), + full_slot_ms INTEGER NOT NULL DEFAULT 0 CHECK (full_slot_ms >= 0), + adaptive_slot_ms INTEGER NOT NULL DEFAULT 0 CHECK (adaptive_slot_ms >= 0), + omitted_descriptor_count INTEGER NOT NULL DEFAULT 0 CHECK (omitted_descriptor_count >= 0), + failure_count INTEGER NOT NULL DEFAULT 0 CHECK (failure_count >= 0), + privacy_violation_count INTEGER NOT NULL DEFAULT 0 CHECK (privacy_violation_count >= 0), + safety_regression_count INTEGER NOT NULL DEFAULT 0 CHECK (safety_regression_count >= 0), + selection_metrics_json TEXT NOT NULL DEFAULT '{{}}', + sink_checkpoint_count INTEGER NOT NULL DEFAULT 0 CHECK (sink_checkpoint_count >= 0), + routed_recall_ppm INTEGER NOT NULL DEFAULT 0 CHECK (routed_recall_ppm >= 0), + slot_ratio_ppm INTEGER NOT NULL DEFAULT 0 CHECK (slot_ratio_ppm >= 0), + recall_threshold_ppm INTEGER NOT NULL DEFAULT 850000 + CHECK (recall_threshold_ppm = 850000), + slot_threshold_ppm INTEGER NOT NULL DEFAULT 400000 + CHECK (slot_threshold_ppm = 400000), + passed INTEGER NOT NULL DEFAULT 0 CHECK (passed IN (0,1)), + lease_owner TEXT NOT NULL, + lease_token TEXT NOT NULL, + lease_expires_at TEXT NOT NULL, + started_at TEXT NOT NULL, + completed_at TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS target_queue_policy_events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + queue_id INTEGER NOT NULL, + action TEXT NOT NULL, + prior_status TEXT NOT NULL, + next_status TEXT NOT NULL, + source TEXT NOT NULL, + platform TEXT NOT NULL, + query TEXT NOT NULL, + reason_code TEXT NOT NULL, + config_sha256 TEXT NOT NULL, + policy_sha256 TEXT NOT NULL, + manifest_sha256 TEXT NOT NULL, + review_audit_sha256 TEXT NOT NULL UNIQUE, + reverses_event_id INTEGER UNIQUE, + experiment_id INTEGER, + prior_updated_at TEXT NOT NULL, + created_at TEXT NOT NULL, + {_docker_depth_check_clause('target_queue_policy_events', 'target_queue_policy_events_transition_check')}, + {_docker_depth_check_clause('target_queue_policy_events', 'target_queue_policy_events_hash_check')}, + {_docker_depth_check_clause('target_queue_policy_events', 'target_queue_policy_events_experiment_ownership_check')}, + FOREIGN KEY(queue_id) REFERENCES target_queue(id), + FOREIGN KEY(reverses_event_id) REFERENCES target_queue_policy_events(id), + FOREIGN KEY(experiment_id) REFERENCES docker_depth_experiments(id) +); + +CREATE TABLE IF NOT EXISTS result_bundles ( + reservation_id INTEGER PRIMARY KEY, + bundle_id TEXT NOT NULL UNIQUE, + scan_event_id TEXT NOT NULL UNIQUE, + scan_event_hash TEXT NOT NULL, + format_version INTEGER NOT NULL CHECK (format_version = 2), + relative_path TEXT NOT NULL UNIQUE, + actual_bytes BIGINT NOT NULL CHECK (actual_bytes > 0), + frame_count INTEGER NOT NULL, + finding_count INTEGER NOT NULL, + error_count INTEGER NOT NULL, + candidate_count INTEGER NOT NULL, + state TEXT NOT NULL CHECK (state IN ('ready','ingesting','db_committed','acknowledged','quarantined')), + available_after TEXT, + ingest_attempts INTEGER NOT NULL DEFAULT 0, + ingest_lease_generation BIGINT, + ingest_lease_token TEXT, + ingest_lease_expires_at TEXT, + target_scan_id INTEGER, + ready_at TEXT NOT NULL, + committed_at TEXT, + acknowledged_at TEXT, + updated_at TEXT NOT NULL, + FOREIGN KEY(reservation_id) REFERENCES result_reservations(id), + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id) +); + +CREATE TABLE IF NOT EXISTS pipeline_leases ( + worker_name TEXT PRIMARY KEY, + generation BIGINT NOT NULL DEFAULT 0, + lease_token TEXT, + supervisor_instance_id TEXT, + owner_pid BIGINT, + owner_creation_time TEXT, + owner_executable TEXT, + state TEXT NOT NULL DEFAULT 'released' + CHECK (state IN ('starting','recovering','ready','stopping','released','failed')), + acquired_at TEXT, + heartbeat_at TEXT, + lease_expires_at TEXT, + last_error TEXT, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS scan_result_compat ( + target_scan_id INTEGER PRIMARY KEY, + schema_version INTEGER NOT NULL CHECK (schema_version = 2), + metadata_json TEXT NOT NULL, + metadata_sha256 TEXT NOT NULL, + metadata_bytes BIGINT NOT NULL, + reconstruction_status TEXT NOT NULL CHECK (reconstruction_status IN ('exact','bounded','legacy')), + omitted_payload_sha256 TEXT, + created_at TEXT NOT NULL, + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id) +); + +CREATE TABLE IF NOT EXISTS finding_compat_payloads ( + finding_id INTEGER PRIMARY KEY, + raw_value TEXT, + raw_v2_value TEXT, + structured_data_json TEXT, + extra_data_json TEXT, + analysis_info_json TEXT, + extension_json TEXT, + payload_sha256 TEXT NOT NULL, + payload_bytes BIGINT NOT NULL, + payload_omitted INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + FOREIGN KEY(finding_id) REFERENCES findings(id) +); + +CREATE TABLE IF NOT EXISTS projection_jobs ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + job_kind TEXT NOT NULL CHECK (job_kind IN ('scan_event','keycheck_event','rebuild')), + event_id TEXT NOT NULL, + event_hash TEXT NOT NULL, + target_scan_id INTEGER, + keycheck_result_id INTEGER, + status TEXT NOT NULL DEFAULT 'pending' CHECK (status IN ('pending','leased','completed','quarantined')), + required_stream_mask INTEGER NOT NULL, + capacity_items BIGINT NOT NULL, + capacity_bytes BIGINT NOT NULL, + capacity_released INTEGER NOT NULL DEFAULT 0, + attempts INTEGER NOT NULL DEFAULT 0, + available_after TEXT, + lease_generation BIGINT, + lease_token TEXT, + lease_expires_at TEXT, + last_error_code TEXT, + last_error_detail TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + completed_at TEXT, + UNIQUE(job_kind, event_id), + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id), + FOREIGN KEY(keycheck_result_id) REFERENCES keycheck_results(id) +); + +CREATE TABLE IF NOT EXISTS projection_streams ( + stream_name TEXT PRIMARY KEY, + base_relative_path TEXT NOT NULL UNIQUE, + current_generation BIGINT NOT NULL DEFAULT 0, + rotation_bytes BIGINT NOT NULL, + max_generations INTEGER NOT NULL, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS projection_cursors ( + stream_name TEXT PRIMARY KEY, + generation BIGINT NOT NULL, + committed_offset BIGINT NOT NULL, + last_append_id INTEGER, + last_job_id INTEGER, + last_event_id TEXT, + last_event_hash TEXT, + updated_at TEXT NOT NULL, + FOREIGN KEY(stream_name) REFERENCES projection_streams(stream_name), + FOREIGN KEY(last_append_id) REFERENCES projection_appends(id), + FOREIGN KEY(last_job_id) REFERENCES projection_jobs(id) +); + +CREATE TABLE IF NOT EXISTS projection_appends ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + job_id INTEGER NOT NULL, + stream_name TEXT NOT NULL, + event_id TEXT NOT NULL, + event_hash TEXT NOT NULL, + generation BIGINT NOT NULL, + byte_offset BIGINT NOT NULL, + byte_length BIGINT NOT NULL, + payload_sha256 TEXT NOT NULL, + record_count INTEGER NOT NULL, + state TEXT NOT NULL CHECK (state IN ('prepared','appended')), + prepared_at TEXT NOT NULL, + appended_at TEXT, + UNIQUE(job_id, stream_name), + UNIQUE(stream_name, generation, byte_offset), + FOREIGN KEY(job_id) REFERENCES projection_jobs(id), + FOREIGN KEY(stream_name) REFERENCES projection_streams(stream_name) +); + +CREATE TABLE IF NOT EXISTS projection_append_audit ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + append_id BIGINT NOT NULL, + job_id INTEGER NOT NULL, + stream_name TEXT NOT NULL, + event_id TEXT NOT NULL, + event_hash TEXT NOT NULL, + generation BIGINT NOT NULL, + byte_offset BIGINT NOT NULL, + byte_length BIGINT NOT NULL, + payload_sha256 TEXT NOT NULL, + state TEXT NOT NULL CHECK (state IN ('canceled','quarantined')), + reason_code TEXT NOT NULL, + reason_detail TEXT, + created_at TEXT NOT NULL, + FOREIGN KEY(job_id) REFERENCES projection_jobs(id), + FOREIGN KEY(stream_name) REFERENCES projection_streams(stream_name) +); + +CREATE TABLE IF NOT EXISTS projection_rotations ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + stream_name TEXT NOT NULL, + from_generation BIGINT NOT NULL, + to_generation BIGINT NOT NULL, + source_bytes BIGINT NOT NULL, + segment_relative_path TEXT NOT NULL, + state TEXT NOT NULL CHECK (state IN ('prepared','renamed','completed')), + created_at TEXT NOT NULL, + completed_at TEXT, + UNIQUE(stream_name, to_generation), + FOREIGN KEY(stream_name) REFERENCES projection_streams(stream_name) +); + +CREATE TABLE IF NOT EXISTS keycheck_credentials ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + service TEXT NOT NULL, + credential_hash TEXT NOT NULL, + provider_key_hash TEXT NOT NULL, + candidate_kind TEXT NOT NULL, + secret_text TEXT, + secret_json TEXT, + key_masked TEXT, + endpoint TEXT, + principal TEXT, + metadata_json TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + UNIQUE(service, credential_hash) +); + +CREATE TABLE IF NOT EXISTS keycheck_candidates ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + candidate_uid TEXT NOT NULL UNIQUE, + credential_id INTEGER NOT NULL, + service TEXT NOT NULL, + routed_service TEXT NOT NULL DEFAULT '', + secret_hash TEXT NOT NULL, + finding_id INTEGER, + target_scan_id INTEGER, + scan_event_id TEXT, + finding_uid TEXT, + source TEXT, + query TEXT, + target TEXT, + detector_name TEXT, + found_at TEXT, + metadata_json TEXT, + state TEXT NOT NULL DEFAULT 'pending' + CHECK (state IN ('pending','leased','deferred','completed','quarantined')), + priority INTEGER NOT NULL DEFAULT 0, + attempts INTEGER NOT NULL DEFAULT 0, + available_after TEXT, + lease_owner TEXT, + lease_token TEXT, + lease_expires_at TEXT, + keycheck_result_id INTEGER, + capacity_bytes BIGINT NOT NULL, + capacity_released INTEGER NOT NULL DEFAULT 0, + result_projection_reserved_bytes BIGINT NOT NULL DEFAULT 0, + result_projection_credit_transferred INTEGER NOT NULL DEFAULT 0, + last_error TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + completed_at TEXT, + FOREIGN KEY(credential_id) REFERENCES keycheck_credentials(id), + FOREIGN KEY(finding_id) REFERENCES findings(id), + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id), + FOREIGN KEY(keycheck_result_id) REFERENCES keycheck_results(id) +); + +CREATE TABLE IF NOT EXISTS keycheck_current_state ( + credential_id INTEGER PRIMARY KEY, + service TEXT NOT NULL, + status TEXT NOT NULL, + status_group TEXT NOT NULL, + last_result_id INTEGER NOT NULL, + result_source TEXT NOT NULL, + checked_at TEXT NOT NULL, + recheck_after TEXT, + state_version BIGINT NOT NULL DEFAULT 1, + metadata_json TEXT, + updated_at TEXT NOT NULL, + FOREIGN KEY(credential_id) REFERENCES keycheck_credentials(id), + FOREIGN KEY(last_result_id) REFERENCES keycheck_results(id) +); + +CREATE TABLE IF NOT EXISTS pipeline_quarantine ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + subsystem TEXT NOT NULL, + object_type TEXT NOT NULL, + object_id INTEGER, + reservation_id INTEGER, + projection_job_id INTEGER, + keycheck_candidate_id INTEGER, + event_id TEXT, + payload_sha256 TEXT, + reason_code TEXT NOT NULL, + reason_detail TEXT, + source_relative_path TEXT, + byte_count BIGINT NOT NULL DEFAULT 0, + capacity_items BIGINT NOT NULL DEFAULT 1, + capacity_bytes BIGINT NOT NULL DEFAULT 0, + capacity_credit_applied INTEGER NOT NULL DEFAULT 1, + review_status TEXT NOT NULL DEFAULT 'pending' + CHECK (review_status IN ('pending','approved_retry','approved_rescan','discarded','resolved')), + detected_at TEXT NOT NULL, + resolved_at TEXT, + review_audit_sha256 TEXT, + FOREIGN KEY(reservation_id) REFERENCES result_reservations(id), + FOREIGN KEY(projection_job_id) REFERENCES projection_jobs(id), + FOREIGN KEY(keycheck_candidate_id) REFERENCES keycheck_candidates(id) +); + +CREATE TABLE IF NOT EXISTS pipeline_artifacts ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + subsystem TEXT NOT NULL, + artifact_kind TEXT NOT NULL, + owner_id BIGINT NOT NULL, + owner_key TEXT NOT NULL DEFAULT '', + relative_path TEXT NOT NULL, + payload_sha256 TEXT, + byte_count BIGINT NOT NULL DEFAULT 0, + state TEXT NOT NULL CHECK (state IN ('expected','present','quarantined','deleted')), + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + deleted_at TEXT, + cleanup_attempts INTEGER NOT NULL DEFAULT 0, + cleanup_available_after TEXT, + cleanup_last_error TEXT, + UNIQUE(subsystem, artifact_kind, owner_id, owner_key), + UNIQUE(subsystem, relative_path) +); + +CREATE TABLE IF NOT EXISTS janitor_cursors ( + layout_name TEXT PRIMARY KEY, + last_name TEXT NOT NULL DEFAULT '', + wrap_count BIGINT NOT NULL DEFAULT 0, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS keycheck_recheck_cursors ( + cursor_key TEXT PRIMARY KEY, + service TEXT NOT NULL, + status_scope TEXT NOT NULL, + last_credential_id BIGINT NOT NULL DEFAULT 0, + wrap_count BIGINT NOT NULL DEFAULT 0, + updated_at TEXT NOT NULL +); + +CREATE UNIQUE INDEX IF NOT EXISTS uq_result_reservation_queue_lease +ON result_reservations(queue_id, claim_lease_token); +CREATE UNIQUE INDEX IF NOT EXISTS uq_worker_progress_reservation_sequence +ON worker_progress_events(reservation_id, sequence); +CREATE INDEX IF NOT EXISTS idx_worker_progress_reservation_received +ON worker_progress_events(reservation_id, received_at, id); +CREATE INDEX IF NOT EXISTS idx_worker_progress_device_received +ON worker_progress_events(remote_device_id, received_at, id); +CREATE INDEX IF NOT EXISTS idx_worker_progress_phase_received +ON worker_progress_events(phase, received_at, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_worker_diagnostics_uid +ON worker_diagnostics(diagnostic_uid); +CREATE INDEX IF NOT EXISTS idx_worker_diagnostics_reservation_received +ON worker_diagnostics(reservation_id, received_at, id); +CREATE INDEX IF NOT EXISTS idx_worker_diagnostics_scan_received +ON worker_diagnostics(target_scan_id, received_at, id); +CREATE INDEX IF NOT EXISTS idx_worker_diagnostics_phase_received +ON worker_diagnostics(phase, received_at, id); +CREATE INDEX IF NOT EXISTS idx_worker_diagnostics_category_code +ON worker_diagnostics(category, code, received_at, id); +CREATE INDEX IF NOT EXISTS idx_worker_diagnostics_code_received +ON worker_diagnostics(code, received_at, id); +CREATE INDEX IF NOT EXISTS idx_worker_diagnostics_kind_received +ON worker_diagnostics(kind, received_at, id); +CREATE INDEX IF NOT EXISTS idx_worker_diagnostics_retryable_received +ON worker_diagnostics(retryable, received_at, id); +CREATE INDEX IF NOT EXISTS idx_runtime_operations_status_updated +ON runtime_operations(status, updated_at, operation_id); +CREATE INDEX IF NOT EXISTS idx_runtime_operations_agent_state_updated +ON runtime_operations(agent_state, updated_at, operation_id); +CREATE INDEX IF NOT EXISTS idx_runtime_audit_events_created +ON runtime_audit_events(created_at DESC, id DESC); +CREATE INDEX IF NOT EXISTS idx_runtime_audit_events_operation +ON runtime_audit_events(operation_id, id DESC); +CREATE UNIQUE INDEX IF NOT EXISTS uq_remote_worker_users_key +ON remote_worker_users(user_key); +CREATE UNIQUE INDEX IF NOT EXISTS uq_remote_worker_devices_key +ON remote_worker_devices(device_key); +CREATE UNIQUE INDEX IF NOT EXISTS uq_remote_worker_devices_token +ON remote_worker_devices(token_sha256); +CREATE INDEX IF NOT EXISTS idx_admission_intents_remote_device +ON admission_intents(remote_device_id, created_at) WHERE remote_device_id IS NOT NULL; +CREATE INDEX IF NOT EXISTS idx_result_reservations_remote_active_user +ON result_reservations(remote_user_id, id) +WHERE assignment_kind = 'remote' AND remote_resolved_at IS NULL; +CREATE INDEX IF NOT EXISTS idx_result_reservations_remote_expiry +ON result_reservations(remote_expires_at, id) +WHERE assignment_kind = 'remote' AND remote_resolved_at IS NULL AND state = 'scanning'; +CREATE INDEX IF NOT EXISTS idx_result_reservations_remote_device_history +ON result_reservations(remote_device_id, remote_issued_at, id) +WHERE assignment_kind = 'remote'; +CREATE INDEX IF NOT EXISTS idx_result_reservations_remote_resolved +ON result_reservations(remote_resolved_at, id) +WHERE assignment_kind = 'remote' AND remote_resolved_at IS NOT NULL; +CREATE UNIQUE INDEX IF NOT EXISTS uq_result_reservations_remote_receipt +ON result_reservations(remote_receipt_id) WHERE remote_receipt_id IS NOT NULL; +CREATE INDEX IF NOT EXISTS idx_result_reservations_recovery +ON result_reservations(state, producer_lease_expires_at, id) +WHERE state IN ('scanning','ready','ingesting','db_committed'); +CREATE INDEX IF NOT EXISTS idx_result_reservations_queue_history +ON result_reservations(queue_id, id DESC); +CREATE INDEX IF NOT EXISTS idx_result_bundles_ready +ON result_bundles(state, available_after, ready_at, reservation_id) +WHERE state IN ('ready','ingesting','db_committed'); +CREATE INDEX IF NOT EXISTS idx_result_bundles_ingest_lease +ON result_bundles(state, ingest_lease_expires_at) +WHERE state = 'ingesting'; +CREATE INDEX IF NOT EXISTS idx_docker_content_blobs_reclaim +ON docker_content_blobs(state, available_after, lease_expires_at, digest); +CREATE INDEX IF NOT EXISTS idx_docker_content_blobs_reservation +ON docker_content_blobs(lease_reservation_id, digest); +CREATE INDEX IF NOT EXISTS idx_docker_image_blob_coverage_manifest +ON docker_image_blob_coverage(manifest_digest, coverage_state, position); +CREATE INDEX IF NOT EXISTS idx_docker_image_blob_coverage_blob +ON docker_image_blob_coverage(blob_digest, coverage_state, queue_id); +CREATE INDEX IF NOT EXISTS idx_docker_image_blob_coverage_reservation +ON docker_image_blob_coverage(reservation_id, position); +CREATE INDEX IF NOT EXISTS idx_docker_image_blob_coverage_selection +ON docker_image_blob_coverage( + queue_id, manifest_digest, selection_policy_sha256, position, reservation_id +); +CREATE INDEX IF NOT EXISTS idx_docker_adaptive_shadow_reports_gate +ON docker_adaptive_shadow_reports( + scan_policy_sha256, execution_policy_sha256, selection_policy_sha256, + state, completed_at, id +); +CREATE INDEX IF NOT EXISTS idx_target_queue_policy_events_queue +ON target_queue_policy_events(queue_id, id DESC); +CREATE INDEX IF NOT EXISTS idx_target_queue_policy_events_manifest +ON target_queue_policy_events(manifest_sha256, id); +CREATE INDEX IF NOT EXISTS idx_projection_jobs_claim +ON projection_jobs(status, available_after, id) WHERE status = 'pending'; +CREATE INDEX IF NOT EXISTS idx_projection_jobs_lease +ON projection_jobs(status, lease_expires_at, id) WHERE status = 'leased'; +CREATE INDEX IF NOT EXISTS idx_keycheck_candidates_pending +ON keycheck_candidates(service, priority DESC, id) WHERE state = 'pending'; +CREATE INDEX IF NOT EXISTS idx_keycheck_candidates_deferred +ON keycheck_candidates(service, available_after, id) WHERE state = 'deferred'; +CREATE INDEX IF NOT EXISTS idx_keycheck_candidates_lease +ON keycheck_candidates(service, lease_expires_at, id) WHERE state = 'leased'; +CREATE INDEX IF NOT EXISTS idx_keycheck_candidates_credential +ON keycheck_candidates(credential_id, state, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_keycheck_candidate_finding_credential +ON keycheck_candidates(service, finding_id, credential_id) WHERE finding_id IS NOT NULL; +CREATE INDEX IF NOT EXISTS idx_pipeline_quarantine_pending +ON pipeline_quarantine(subsystem, review_status, id) WHERE review_status = 'pending'; +CREATE UNIQUE INDEX IF NOT EXISTS uq_pipeline_quarantine_open_object +ON pipeline_quarantine(subsystem, object_type, object_id) +WHERE review_status = 'pending' AND object_id IS NOT NULL; +CREATE INDEX IF NOT EXISTS idx_pipeline_artifacts_owner +ON pipeline_artifacts(subsystem, owner_id, state, id); +CREATE INDEX IF NOT EXISTS idx_keycheck_results_candidate ON keycheck_results(candidate_id); +CREATE INDEX IF NOT EXISTS idx_keycheck_results_credential_checked +ON keycheck_results(credential_id, checked_at DESC, id DESC); +''' + DOCKER_DEPTH_EXPERIMENT_SCHEMA_SQL + + +SCHEMA_SQL = r''' +CREATE TABLE IF NOT EXISTS runs ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + started_at TEXT NOT NULL, + ended_at TEXT, + duration_sec REAL, + status TEXT NOT NULL, + invocation_mode TEXT, + command_line TEXT, + argv_json TEXT, + selected_source TEXT, + selected_platform TEXT, + config_path TEXT, + config_hash TEXT, + enabled_sources_json TEXT, + db_path TEXT, + total_fetched INTEGER DEFAULT 0, + total_queued_new INTEGER DEFAULT 0, + total_scan_requested INTEGER DEFAULT 0, + total_scanned INTEGER DEFAULT 0, + total_clean INTEGER DEFAULT 0, + total_found INTEGER DEFAULT 0, + total_skipped INTEGER DEFAULT 0, + total_errors INTEGER DEFAULT 0, + total_findings INTEGER DEFAULT 0, + total_verified_findings INTEGER DEFAULT 0, + total_unique_secrets INTEGER DEFAULT 0, + total_unique_findings INTEGER DEFAULT 0, + total_staged INTEGER NOT NULL DEFAULT 0, + total_quarantined INTEGER NOT NULL DEFAULT 0, + error TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS source_cycles ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + run_id INTEGER, + source TEXT, + platform TEXT, + mode TEXT, + query TEXT, + query_index INTEGER, + query_count INTEGER, + auth_name TEXT, + started_at TEXT NOT NULL, + ended_at TEXT, + duration_sec REAL, + status TEXT NOT NULL, + message TEXT, + config_json TEXT, + queue_todo_before INTEGER, + queue_checked_before INTEGER, + queue_todo_after INTEGER, + queue_checked_after INTEGER, + fetched_count INTEGER DEFAULT 0, + queued_new_count INTEGER DEFAULT 0, + queued_updated_count INTEGER NOT NULL DEFAULT 0, + scan_requested_count INTEGER DEFAULT 0, + scanned_count INTEGER DEFAULT 0, + clean_count INTEGER DEFAULT 0, + found_count INTEGER DEFAULT 0, + skipped_count INTEGER DEFAULT 0, + error_count INTEGER DEFAULT 0, + findings_count INTEGER DEFAULT 0, + verified_findings_count INTEGER DEFAULT 0, + unique_secrets_count INTEGER DEFAULT 0, + unique_findings_count INTEGER DEFAULT 0, + targets_per_hour REAL DEFAULT 0, + hit_rate REAL DEFAULT 0, + verified_hit_rate REAL DEFAULT 0, + error_rate REAL DEFAULT 0, + staged_count INTEGER NOT NULL DEFAULT 0, + ingested_count INTEGER NOT NULL DEFAULT 0, + quarantined_count INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + FOREIGN KEY(run_id) REFERENCES runs(id) +); + +CREATE TABLE IF NOT EXISTS target_scans ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + scan_event_id TEXT, + scan_event_hash TEXT, + queue_id INTEGER, + claim_lease_token TEXT, + queue_completion_applied INTEGER NOT NULL DEFAULT 0, + queue_completion_disposition TEXT, + run_id INTEGER, + cycle_id INTEGER, + source TEXT, + query TEXT, + target TEXT, + normalized_target TEXT, + scan_type TEXT, + status TEXT, + started_at TEXT, + ended_at TEXT, + duration_sec REAL, + scan_options_json TEXT, + package_name TEXT, + package_version TEXT, + package_artifact TEXT, + package_date TEXT, + package_filename TEXT, + package_type TEXT, + package_size INTEGER, + findings_count INTEGER DEFAULT 0, + verified_findings_count INTEGER DEFAULT 0, + error_count INTEGER DEFAULT 0, + skipped_reason TEXT, + first_error_summary TEXT, + raw_result_json TEXT, + result_reservation_id INTEGER, + compat_schema_version INTEGER NOT NULL DEFAULT 2, + raw_result_storage TEXT NOT NULL DEFAULT 'legacy', + created_at TEXT NOT NULL, + FOREIGN KEY(run_id) REFERENCES runs(id), + FOREIGN KEY(cycle_id) REFERENCES source_cycles(id), + FOREIGN KEY(queue_id) REFERENCES target_queue(id), + FOREIGN KEY(result_reservation_id) REFERENCES result_reservations(id) +); + +CREATE TABLE IF NOT EXISTS findings ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + run_id INTEGER, + cycle_id INTEGER, + target_scan_id INTEGER, + source TEXT, + query TEXT, + target TEXT, + normalized_target TEXT, + detector_name TEXT, + detector_type TEXT, + verified INTEGER DEFAULT 0, + raw_secret TEXT, + redacted_secret TEXT, + secret_hash TEXT, + detector_secret_hash TEXT, + finding_fingerprint TEXT, + finding_uid TEXT, + file_path TEXT, + line_number TEXT, + commit_hash TEXT, + source_timestamp TEXT, + source_metadata_type TEXT, + source_metadata_json TEXT, + raw_finding_json TEXT, + provider TEXT, + credential_kind TEXT, + credential_confidence TEXT, + required_context_missing INTEGER DEFAULT 0, + principal TEXT, + username TEXT, + email TEXT, + project_id TEXT, + tenant_id TEXT, + organization TEXT, + registry TEXT, + endpoint TEXT, + scope TEXT, + resource TEXT, + enrichment_json TEXT, + raw_payload_sha256 TEXT, + raw_payload_bytes INTEGER, + raw_payload_omitted INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + FOREIGN KEY(run_id) REFERENCES runs(id), + FOREIGN KEY(cycle_id) REFERENCES source_cycles(id), + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id) +); + +CREATE TABLE IF NOT EXISTS finding_uid_map ( + finding_uid TEXT PRIMARY KEY, + finding_id INTEGER NOT NULL, + created_at TEXT NOT NULL, + FOREIGN KEY(finding_id) REFERENCES findings(id) +); + +CREATE TABLE IF NOT EXISTS errors ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + run_id INTEGER, + cycle_id INTEGER, + target_scan_id INTEGER, + source TEXT, + query TEXT, + target TEXT, + normalized_target TEXT, + category TEXT, + summary TEXT, + raw_error TEXT, + created_at TEXT NOT NULL, + FOREIGN KEY(run_id) REFERENCES runs(id), + FOREIGN KEY(cycle_id) REFERENCES source_cycles(id), + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id) +); + +CREATE TABLE IF NOT EXISTS queue_snapshots ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + run_id INTEGER, + cycle_id INTEGER, + source TEXT, + phase TEXT, + todo_count INTEGER, + checked_count INTEGER, + todo_file TEXT, + checked_file TEXT, + captured_at TEXT NOT NULL, + FOREIGN KEY(run_id) REFERENCES runs(id), + FOREIGN KEY(cycle_id) REFERENCES source_cycles(id) +); + +CREATE TABLE IF NOT EXISTS config_snapshots ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + run_id INTEGER, + cycle_id INTEGER, + scope TEXT, + source TEXT, + config_json TEXT, + captured_at TEXT NOT NULL, + FOREIGN KEY(run_id) REFERENCES runs(id), + FOREIGN KEY(cycle_id) REFERENCES source_cycles(id) +); + +CREATE TABLE IF NOT EXISTS package_repo_candidates ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + package_source TEXT NOT NULL, + package_name TEXT NOT NULL, + package_version TEXT NOT NULL, + query TEXT, + repo_url TEXT NOT NULL, + provider TEXT, + evidence_json TEXT, + confidence TEXT, + first_seen_at TEXT NOT NULL, + last_seen_at TEXT NOT NULL, + last_run_id INTEGER, + last_cycle_id INTEGER, + UNIQUE(package_source, package_name, package_version, repo_url), + FOREIGN KEY(last_run_id) REFERENCES runs(id), + FOREIGN KEY(last_cycle_id) REFERENCES source_cycles(id) +); + +CREATE TABLE IF NOT EXISTS target_queue ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + source TEXT NOT NULL, + platform TEXT NOT NULL, + query TEXT, + target TEXT NOT NULL, + normalized_target TEXT NOT NULL, + status TEXT NOT NULL DEFAULT 'pending', + attempts INTEGER DEFAULT 0, + lease_owner TEXT, + lease_token TEXT, + claim_batch TEXT, + leased_at TEXT, + lease_expires_at TEXT, + available_after TEXT, + target_scan_id INTEGER, + last_error TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + completed_at TEXT, + resolver_state TEXT, + resolver_due_at TEXT, + resolver_attempts INTEGER NOT NULL DEFAULT 0, + resolver_token TEXT, + current_result_reservation_id INTEGER, + claim_event_id TEXT, + remote_modified_at TEXT, + scan_remote_modified_at TEXT, + covered_ref TEXT, + covered_head TEXT, + UNIQUE(source, normalized_target), + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id), + FOREIGN KEY(current_result_reservation_id) REFERENCES result_reservations(id) +); + +CREATE TABLE IF NOT EXISTS scan_publication_outbox ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + target_scan_id INTEGER NOT NULL UNIQUE, + payload_json TEXT NOT NULL, + status TEXT NOT NULL DEFAULT 'pending', + attempts INTEGER DEFAULT 0, + last_error TEXT, + lease_owner TEXT, + lease_expires_at TEXT, + available_after TEXT, + created_at TEXT NOT NULL, + delivered_at TEXT, + updated_at TEXT NOT NULL, + FOREIGN KEY(target_scan_id) REFERENCES target_scans(id) +); + +CREATE TABLE IF NOT EXISTS target_queue_reconciliation_cursors ( + source_file TEXT PRIMARY KEY, + file_identity TEXT NOT NULL, + file_size INTEGER NOT NULL DEFAULT 0, + file_mtime_ns INTEGER NOT NULL DEFAULT 0, + source TEXT NOT NULL, + platform TEXT NOT NULL, + byte_offset INTEGER NOT NULL DEFAULT 0, + line_number INTEGER NOT NULL DEFAULT 0, + discarding_oversized INTEGER NOT NULL DEFAULT 0, + oversized_line_start INTEGER, + cumulative_rows INTEGER NOT NULL DEFAULT 0, + cumulative_bytes INTEGER NOT NULL DEFAULT 0, + cumulative_inserted INTEGER NOT NULL DEFAULT 0, + cumulative_rejected INTEGER NOT NULL DEFAULT 0, + completed_at TEXT, + last_report_json TEXT, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS target_queue_reconciliation_issues ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + source_file TEXT NOT NULL, + file_identity TEXT NOT NULL, + source TEXT NOT NULL, + platform TEXT NOT NULL, + line_number INTEGER NOT NULL, + byte_offset INTEGER NOT NULL, + reason TEXT NOT NULL, + target_preview TEXT, + created_at TEXT NOT NULL, + resolved_at TEXT +); + +''' + KEYCHECK_RESULTS_SQL + PIPELINE_SCHEMA_SQL + r''' + +CREATE INDEX IF NOT EXISTS idx_runs_started_at ON runs(started_at); +CREATE INDEX IF NOT EXISTS idx_runs_status ON runs(status); +CREATE INDEX IF NOT EXISTS idx_runs_selected_source_status ON runs(selected_source, status, id); +CREATE INDEX IF NOT EXISTS idx_source_cycles_run_id ON source_cycles(run_id); +CREATE INDEX IF NOT EXISTS idx_source_cycles_source_started ON source_cycles(source, started_at); +CREATE INDEX IF NOT EXISTS idx_source_cycles_source_query ON source_cycles(source, query); +CREATE INDEX IF NOT EXISTS idx_source_cycles_source_status ON source_cycles(source, status, id); +CREATE INDEX IF NOT EXISTS idx_target_scans_cycle_id ON target_scans(cycle_id); +CREATE INDEX IF NOT EXISTS idx_target_scans_queue_id ON target_scans(queue_id); +CREATE INDEX IF NOT EXISTS idx_target_scans_result_reservation ON target_scans(result_reservation_id, id); +CREATE INDEX IF NOT EXISTS idx_target_scans_source_status ON target_scans(source, status); +CREATE INDEX IF NOT EXISTS idx_target_scans_source_ended ON target_scans(source, ended_at DESC); +CREATE INDEX IF NOT EXISTS idx_target_scans_source_ended_id ON target_scans(source, ended_at DESC, id DESC); +CREATE INDEX IF NOT EXISTS idx_target_scans_normalized_target ON target_scans(normalized_target); +CREATE INDEX IF NOT EXISTS idx_target_scans_source_skip_ended ON target_scans(source, skipped_reason, ended_at DESC); +CREATE INDEX IF NOT EXISTS idx_target_scans_cooldown_recent ON target_scans(source, ended_at DESC, id DESC) WHERE status = 'skipped' AND ((source = 'github_actions' AND (skipped_reason = 'no downloadable workflow logs or artifacts' OR skipped_reason = 'no recent workflow runs')) OR (source = 'gitlab_ci' AND (skipped_reason = 'no downloadable job traces or artifacts' OR skipped_reason = 'no recent pipelines'))); +CREATE UNIQUE INDEX IF NOT EXISTS uq_target_scans_scan_event_id ON target_scans(scan_event_id) WHERE scan_event_id IS NOT NULL; +CREATE INDEX IF NOT EXISTS idx_target_scans_event_hash ON target_scans(scan_event_id, scan_event_hash); +CREATE INDEX IF NOT EXISTS idx_findings_cycle_id ON findings(cycle_id); +CREATE INDEX IF NOT EXISTS idx_findings_target_scan_id_id ON findings(target_scan_id, id); +CREATE INDEX IF NOT EXISTS idx_findings_source ON findings(source); +CREATE INDEX IF NOT EXISTS idx_findings_detector ON findings(detector_name); +CREATE INDEX IF NOT EXISTS idx_findings_secret_hash ON findings(secret_hash); +CREATE INDEX IF NOT EXISTS idx_findings_detector_secret_hash ON findings(detector_secret_hash); +CREATE INDEX IF NOT EXISTS idx_findings_fingerprint ON findings(finding_fingerprint); +CREATE INDEX IF NOT EXISTS idx_findings_finding_uid ON findings(finding_uid); +CREATE INDEX IF NOT EXISTS idx_errors_cycle_id ON errors(cycle_id); +CREATE INDEX IF NOT EXISTS idx_errors_source_category ON errors(source, category); +CREATE INDEX IF NOT EXISTS idx_errors_target_scan_id ON errors(target_scan_id); +CREATE INDEX IF NOT EXISTS idx_queue_snapshots_source_time ON queue_snapshots(source, captured_at); +CREATE INDEX IF NOT EXISTS idx_package_repo_candidates_query ON package_repo_candidates(query); +CREATE INDEX IF NOT EXISTS idx_package_repo_candidates_repo ON package_repo_candidates(repo_url); +CREATE INDEX IF NOT EXISTS idx_package_repo_candidates_provider_seen ON package_repo_candidates(LOWER(provider), last_seen_at DESC); +CREATE INDEX IF NOT EXISTS idx_package_repo_candidates_source_seen ON package_repo_candidates(package_source, last_seen_at); +CREATE INDEX IF NOT EXISTS idx_package_repo_candidates_query_seen ON package_repo_candidates(query, last_seen_at DESC, id DESC) WHERE repo_url IS NOT NULL AND repo_url <> ''; +CREATE INDEX IF NOT EXISTS idx_package_repo_candidates_recent_lookup ON package_repo_candidates(last_seen_at DESC, id DESC, package_source, query, package_name) WHERE repo_url IS NOT NULL AND repo_url <> ''; +CREATE INDEX IF NOT EXISTS idx_target_queue_source_status ON target_queue(source, status, updated_at); +CREATE INDEX IF NOT EXISTS idx_target_queue_observe_source_status ON target_queue(source, status); +CREATE INDEX IF NOT EXISTS idx_target_queue_lease ON target_queue(source, status, lease_expires_at); +CREATE INDEX IF NOT EXISTS idx_target_queue_platform_status ON target_queue(platform, status); +CREATE INDEX IF NOT EXISTS idx_target_queue_claim_batch ON target_queue(claim_batch, lease_owner); +CREATE INDEX IF NOT EXISTS idx_target_queue_claim ON target_queue(source, platform, status, available_after, lease_expires_at, id); +CREATE INDEX IF NOT EXISTS idx_target_queue_resolver_claim ON target_queue(source, platform, status, resolver_state, resolver_due_at, id); +CREATE INDEX IF NOT EXISTS idx_target_queue_source_platform_normalized ON target_queue(source, platform, normalized_target); +CREATE INDEX IF NOT EXISTS idx_target_queue_cold ON target_queue(source, platform, query, id) WHERE status = 'cold'; +CREATE INDEX IF NOT EXISTS idx_target_queue_current_reservation ON target_queue(current_result_reservation_id); +CREATE INDEX IF NOT EXISTS idx_target_queue_claimable_v2 ON target_queue(source, platform, status, available_after, id) WHERE current_result_reservation_id IS NULL AND (status = 'pending' OR status = 'deferred'); +CREATE INDEX IF NOT EXISTS idx_target_queue_claim_pending ON target_queue(source, platform, id, attempts, available_after) WHERE status = 'pending' AND current_result_reservation_id IS NULL AND (resolver_state IS NULL OR resolver_state = 'resolved'); +CREATE INDEX IF NOT EXISTS idx_target_queue_claim_deferred ON target_queue(source, platform, available_after, id, attempts) WHERE status = 'deferred' AND current_result_reservation_id IS NULL AND (resolver_state IS NULL OR resolver_state = 'resolved'); +CREATE INDEX IF NOT EXISTS idx_target_queue_claim_in_progress ON target_queue(source, platform, id, attempts, available_after, lease_expires_at, resolver_state) WHERE status = 'in_progress' AND (resolver_state IS NULL OR resolver_state = 'resolved'); +CREATE INDEX IF NOT EXISTS idx_target_queue_active_lease_owner_token ON target_queue(lease_owner, lease_token) WHERE status = 'in_progress'; +CREATE INDEX IF NOT EXISTS idx_target_queue_exhausted_attempts ON target_queue(source, platform, status, attempts, id, lease_expires_at) WHERE status = 'pending' OR status = 'deferred' OR status = 'in_progress'; +CREATE INDEX IF NOT EXISTS idx_target_queue_updated_rescan ON target_queue(source, platform, completed_at, remote_modified_at, scan_remote_modified_at, id) WHERE status = 'done' AND remote_modified_at IS NOT NULL; +CREATE INDEX IF NOT EXISTS idx_scan_publication_outbox_status ON scan_publication_outbox(status, available_after, id); +CREATE INDEX IF NOT EXISTS idx_scan_publication_outbox_age ON scan_publication_outbox(status, created_at, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_scan_publication_outbox_target_scan_id ON scan_publication_outbox(target_scan_id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_keycheck_event_map_event_id ON keycheck_event_map(event_id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_finding_uid_map_finding_uid ON finding_uid_map(finding_uid); +CREATE INDEX IF NOT EXISTS idx_reconciliation_issues_open ON target_queue_reconciliation_issues(source_file, resolved_at, id); +CREATE UNIQUE INDEX IF NOT EXISTS uq_reconciliation_issue_identity ON target_queue_reconciliation_issues(source_file, file_identity, line_number, byte_offset, reason); +''' diff --git a/app/scanner_error_policy_smoke.py b/app/scanner_error_policy_smoke.py new file mode 100644 index 0000000..569af75 --- /dev/null +++ b/app/scanner_error_policy_smoke.py @@ -0,0 +1,423 @@ +import sys + +sys.dont_write_bytecode = True + +import json +import gzip +import os +import tarfile +import tempfile +import uuid +import zipfile +from datetime import datetime, timedelta, timezone +from types import SimpleNamespace + +import scanner as scanner_module +from console_runner import queue_error_disposition, target_retry_delay_sec +from db_backend import redact_database_url +from scanner import ( + apply_trufflehog_diagnostics, + build_authenticated_git_url, + cached_gharchive_hour, + convert_package_git_unavailable_to_skip, + docker_tag_platform_support, + fetch_dockerhub_images, + normalize_git_repo_candidate, + redact_command_args, + run_command, + safe_extract_package_zip, + safe_extract_tar, +) +from scanner_db import ScannerDB, normalize_target, target_status +from runtime_security import ensure_private_directory, harden_private_file + + +def diagnostic(message, error, **extra): + return json.dumps({'level': 'error', 'msg': message, 'error': error, **extra}) + + +def assert_diagnostic_policy(): + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics( + result, + diagnostic('non-critical error processing chunk', 'error reading chunk: brotli: PADDING_2'), + 0, + 'pypi', + ) + assert not result.get('errors') + assert result.get('warnings') and result.get('degraded') + assert target_status(result) == 'degraded' + + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics(result, diagnostic('Skipping result: invalid', 'empty raw'), 0, 'npm') + assert not result.get('errors') and result.get('warnings') + + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics( + result, + diagnostic('error reading chunk', 'read tcp: connection reset by peer'), + 0, + 'pypi', + ) + assert result.get('errors') and result.get('retryable') is True + assert result.get('error_class') == 'network' + + for detail in ('unexpected EOF', 'permission denied', 'no space left on device'): + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics(result, diagnostic('error reading chunk', detail), 0, 'pypi') + assert result.get('errors'), detail + + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics(result, diagnostic('error reading chunk', 'brotli: PADDING_2'), 0, 'git') + assert result.get('errors') + + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics(result, diagnostic('space', 'no repo found for repo'), 0, 'huggingface') + assert result.get('skipped') and not result.get('errors') + assert target_status(result) == 'skipped' + + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics( + result, + diagnostic('error processing image', 'no child with platform linux/amd64 in index image:tag'), + 1, + 'docker', + ) + assert result.get('skipped') and not result.get('errors') + + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics( + result, + diagnostic('error processing layer', 'gzip: invalid header') + '\n' + json.dumps({'level': 'info-0', 'msg': 'finished scanning'}), + 0, + 'docker', + ) + assert result.get('degraded') and not result.get('errors') + + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics( + result, + diagnostic('a detector ignored the context timeout', 'context deadline exceeded') + '\n' + json.dumps({'level': 'info-0', 'msg': 'finished scanning'}), + 0, + 'docker', + ) + assert result.get('degraded') and not result.get('errors') + assert result.get('warning_classes') == ['detector_timeout'] + + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics( + result, + diagnostic('error processing layer', 'gzip: invalid header'), + 1, + 'docker', + ) + assert result.get('errors') and not result.get('degraded') + + result = {'findings': [], 'errors': []} + apply_trufflehog_diagnostics(result, '', 2, 'filesystem') + assert result.get('errors') and result.get('retryable') is True + + result['findings'] = [{'DetectorName': 'Example'}] + assert target_status(result) == 'error' + + result = {'errors': [diagnostic('error running scan', 'remote: Repository not found.')], 'findings': []} + convert_package_git_unavailable_to_skip(result) + assert result.get('skipped') and not result.get('errors') + + +def assert_docker_platform_policy(): + assert docker_tag_platform_support({'images': [{'os': 'linux', 'architecture': 'amd64'}]}) is True + assert docker_tag_platform_support({'images': [{'os': 'linux', 'architecture': 'arm64'}]}) is False + assert docker_tag_platform_support({'images': []}) is None + assert docker_tag_platform_support({'images': [{'os': 'linux'}]}) is None + assert docker_tag_platform_support({'images': [{'os': 'unknown', 'architecture': 'unknown'}]}) is None + assert docker_tag_platform_support({'images': [{'os': 'linux', 'architecture': 'arm64'}, {'os': 'unknown', 'architecture': 'unknown'}]}) is None + assert docker_tag_platform_support({}) is None + assert normalize_target('Owner/Repo:Prod', 'docker') == 'owner/repo:prod' + assert normalize_git_repo_candidate('https://[invalid url, do not cite]/repo') is None + + +def assert_docker_partial_pagination_policy(): + class Response: + def __init__(self, page): + self.page = page + + def raise_for_status(self): + if self.page > 1: + raise RuntimeError('page outside result set') + + def json(self): + return {'count': 1, 'results': [{'repo_name': 'owner/repo'}]} + + original = scanner_module.api_request + try: + def fake_request(method, url, **kwargs): + page = int(url.split('page=', 1)[1].split('&', 1)[0]) + return Response(page) + + scanner_module.api_request = fake_request + assert fetch_dockerhub_images('test', 3, per_page=10, resolve_tags=False) == ['owner/repo'] + finally: + scanner_module.api_request = original + + +def assert_retry_policy(): + assert target_retry_delay_sec(1, 60, 3600) == 60 + assert target_retry_delay_sec(2, 60, 3600) == 120 + assert target_retry_delay_sec(10, 60, 3600) == 3600 + + args = SimpleNamespace(target_retry_max_attempts=3, target_timeout_retry_delay_sec=21600) + status, available_after, attempts, max_attempts = queue_error_disposition( + None, 'github', 'github', 'https://github.com/o/r', + {'errors': ['timeout'], 'error_class': 'timeout', 'scan_meta': {'command_timed_out': True}}, + args, {'attempts': 3}, + ) + assert status == 'failed' and available_after is None and attempts == 3 and max_attempts == 3 + + +def assert_timeout_output_policy(): + marker = '{"DetectorName":"PartialFinding"}' + old_work_dir = scanner_module.scan_config.work_dir + old_min_free = scanner_module.scan_config.min_free_gb + old_authority_check = scanner_module.require_trufflehog_launch_authority + with tempfile.TemporaryDirectory() as temp_dir: + work_dir = os.path.join(temp_dir, 'work') + ensure_private_directory(work_dir, reject_reparse=True) + scanner_module.initialize_scanner_runtime(preflight_complete=True, register_cleanup=False) + scanner_module.scan_config.work_dir = work_dir + scanner_module.scan_config.min_free_gb = 0 + scanner_module.require_trufflehog_launch_authority = lambda _command: None + try: + stdout, stderr, returncode = run_command( + [sys.executable, '-c', f'import time; print({marker!r}, flush=True); time.sleep(5)'], + 1, + ) + finally: + scanner_module.scan_config.work_dir = old_work_dir + scanner_module.scan_config.min_free_gb = old_min_free + scanner_module.require_trufflehog_launch_authority = old_authority_check + assert returncode == -1 and marker in stdout and 'timed out' in stderr.lower() + + +def assert_archive_and_secret_safety(): + url, secrets = build_authenticated_git_url('https://attacker.example/repo.git', 'github', 'sentinel-token') + assert url == 'https://attacker.example/repo.git' and not secrets and 'sentinel-token' not in url + redacted = redact_command_args(['git', 'https://x-access-token:sentinel-token@github.com/org/repo.git', '--token', 'sentinel-token']) + assert all('sentinel-token' not in value for value in redacted) + + with tempfile.TemporaryDirectory() as temp_dir: + tar_path = os.path.join(temp_dir, 'unsafe.tar') + with tarfile.open(tar_path, 'w') as archive: + link = tarfile.TarInfo('link') + link.type = tarfile.SYMTYPE + link.linkname = '..' + archive.addfile(link) + try: + safe_extract_tar(tar_path, os.path.join(temp_dir, 'tar-out')) + raise AssertionError('unsafe tar link was accepted') + except ValueError: + pass + + zip_path = os.path.join(temp_dir, 'large.zip') + with zipfile.ZipFile(zip_path, 'w') as archive: + archive.writestr('large.txt', b'x' * 2048) + try: + safe_extract_package_zip( + zip_path, os.path.join(temp_dir, 'zip-out'), + max_total_size_mb=0, max_file_size_mb=0, + ) + except ValueError: + raise AssertionError('disabled package zip bounds rejected a valid archive') + + crowded_tar = os.path.join(temp_dir, 'crowded.tar') + with tarfile.open(crowded_tar, 'w') as archive: + archive.addfile(tarfile.TarInfo('one')) + archive.addfile(tarfile.TarInfo('two')) + try: + safe_extract_tar(crowded_tar, os.path.join(temp_dir, 'crowded-out'), max_files=1) + raise AssertionError('tar member bound was not enforced') + except ValueError: + pass + + assert 'query-secret' not in redact_database_url('postgresql://u@localhost/db?password=query-secret') + malformed = ScannerDB(db_path=os.path.join(tempfile.gettempdir(), 'must-not-open.db'), db_url='postgreql://bad') + assert malformed.postgres_required and not malformed.enabled and malformed.path is None + + with tempfile.TemporaryDirectory() as temp_dir: + runtime_dir = os.path.join(temp_dir, 'runtime') + cache_dir = os.path.join(runtime_dir, 'gharchive') + ensure_private_directory(runtime_dir) + ensure_private_directory(cache_dir) + hour = datetime(2026, 7, 11, 12, tzinfo=timezone.utc) + cache_path = os.path.join(cache_dir, '2026-07-11-12.json.gz') + with gzip.open(cache_path, 'wb') as archive: + archive.write(b'{"type":"PushEvent"}\n') + harden_private_file(cache_path) + old_runtime = scanner_module.scan_config.runtime_dir + old_free = scanner_module.scan_config.gharchive_cache_min_free_bytes + try: + scanner_module.scan_config.runtime_dir = runtime_dir + scanner_module.scan_config.gharchive_cache_min_free_bytes = 0 + assert scanner_module.canonical_path( + cached_gharchive_hour(hour, cache_dir, request_timeout=1, retries=1) + ) == scanner_module.canonical_path(cache_path) + finally: + scanner_module.scan_config.runtime_dir = old_runtime + scanner_module.scan_config.gharchive_cache_min_free_bytes = old_free + + +def assert_queue_policy(): + old_urls = {key: os.environ.pop(key, None) for key in ('SCANNER_DB_URL', 'DATABASE_URL')} + try: + with tempfile.TemporaryDirectory() as temp_dir: + db = ScannerDB(db_path=os.path.join(temp_dir, 'scanner.db')) + try: + assert db.enabled + target = 'owner/repo:tag' + db.enqueue_targets('dockerhub', 'docker', 'q', [target]) + claimed = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 3600, return_rows=True) + assert [row['target'] for row in claimed] == [target] + first_claim = claimed[0] + row = db.target_queue_item('dockerhub', 'docker', target) + assert row['status'] == 'in_progress' and int(row['attempts']) == 1 + assert db.reclaim_target_leases('dockerhub', 'docker', 'owner-b') == 0 + + future = (datetime.now(timezone.utc) + timedelta(hours=1)).isoformat(timespec='seconds') + db.complete_target_queue_item( + 'dockerhub', 'docker', target, None, 'deferred', 'network', future, + queue_id=first_claim['id'], lease_token=first_claim['lease_token'], + ) + db.sync_target_queue_from_files('dockerhub', 'docker', [target], [], 'q') + row = db.target_queue_item('dockerhub', 'docker', target) + assert row['status'] == 'deferred' and row['available_after'] == future + updated_at = db.conn.execute( + 'SELECT updated_at FROM target_queue WHERE source = ? AND normalized_target = ?', + ('dockerhub', normalize_target(target, 'docker')), + ).fetchone()['updated_at'] + db.sync_target_queue_from_files('dockerhub', 'docker', [target], [], 'q') + assert db.conn.execute( + 'SELECT updated_at FROM target_queue WHERE source = ? AND normalized_target = ?', + ('dockerhub', normalize_target(target, 'docker')), + ).fetchone()['updated_at'] == updated_at + assert db.claim_targets('dockerhub', 'docker', 1, 'owner-b', 3600) == [] + + db.conn.execute( + "UPDATE target_queue SET available_after = ? WHERE source = ?", + ('2000-01-01T00:00:00+00:00', 'dockerhub'), + ) + db.conn.commit() + reclaimed = db.claim_targets('dockerhub', 'docker', 1, 'owner-b', 3600, return_rows=True) + assert [item['target'] for item in reclaimed] == [target] + db.complete_target_queue_item( + 'dockerhub', 'docker', target, None, 'failed', 'permanent', + queue_id=reclaimed[0]['id'], lease_token=reclaimed[0]['lease_token'], + ) + db.sync_target_queue_from_files('dockerhub', 'docker', [], [target], 'q') + row = db.target_queue_item('dockerhub', 'docker', target) + assert row['status'] == 'failed' and row['available_after'] is None + + done_target = 'owner/other:tag' + db.enqueue_targets('dockerhub', 'docker', 'q', [done_target]) + done_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 3600, return_rows=True)[0] + db.complete_target_queue_item( + 'dockerhub', 'docker', done_target, None, 'done', + queue_id=done_claim['id'], lease_token=done_claim['lease_token'], + ) + db.enqueue_targets('dockerhub', 'docker', 'q', [done_target], requeue_done=True) + row = db.target_queue_item('dockerhub', 'docker', done_target) + assert row['status'] == 'pending' and int(row['attempts']) == 0 + done_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 3600, return_rows=True)[0] + db.complete_target_queue_item( + 'dockerhub', 'docker', done_target, None, 'done', + queue_id=done_claim['id'], lease_token=done_claim['lease_token'], + ) + + legacy_target = 'owner/legacy:tag' + db.enqueue_targets('dockerhub', 'docker', 'q', [legacy_target]) + db.conn.execute( + "UPDATE target_queue SET status = 'pending', available_after = ? WHERE normalized_target = ?", + (future, normalize_target(legacy_target, 'docker')), + ) + db.conn.commit() + db.sync_target_queue_from_files('dockerhub', 'docker', [legacy_target], [], 'q') + row = db.target_queue_item('dockerhub', 'docker', legacy_target) + assert row['status'] == 'deferred' and row['available_after'] == future + + stale_target = 'owner/stale:tag' + db.enqueue_targets('dockerhub', 'docker', 'q', [stale_target]) + stale_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 60, max_attempts=3, return_rows=True)[0] + assert stale_claim['target'] == stale_target + db.conn.execute( + "UPDATE target_queue SET lease_expires_at = ? WHERE normalized_target = ?", + ('2000-01-01T00:00:00+00:00', normalize_target(stale_target, 'docker')), + ) + db.conn.commit() + newer_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-b', 60, max_attempts=3, return_rows=True)[0] + assert newer_claim['target'] == stale_target + assert newer_claim['lease_token'] != stale_claim['lease_token'] + assert not db.complete_target_queue_item( + 'dockerhub', 'docker', stale_target, None, 'done', + queue_id=stale_claim['id'], lease_token=stale_claim['lease_token'], + ) + row = db.target_queue_item('dockerhub', 'docker', stale_target) + assert row['status'] == 'in_progress' and row['lease_owner'] == 'owner-b' + + run_id = db.start_run('smoke', ['smoke']) + cycle_id = db.start_source_cycle(run_id, 'dockerhub', 'docker', 'search', 'q', 1, 1, None, {}, {}) + result = { + 'target': stale_target, 'scan_type': 'docker', 'findings': [], 'errors': [], + 'scan_event_id': str(uuid.uuid4()), + 'timestamp': datetime.now(timezone.utc).isoformat(timespec='seconds'), + } + before = db.conn.execute('SELECT COUNT(*) AS count FROM target_scans').fetchone()['count'] + stale = db.record_and_complete_target_result( + run_id, cycle_id, 'dockerhub', 'q', stale_target, result, {}, + row['id'], 'owner-a', 'done', + lease_token=stale_claim['lease_token'], + ) + assert stale and stale['stale'] + assert db.conn.execute('SELECT COUNT(*) AS count FROM target_scans').fetchone()['count'] == before + 1 + row = db.target_queue_item('dockerhub', 'docker', stale_target) + assert row['status'] == 'in_progress' and row['lease_token'] == newer_claim['lease_token'] + result = dict(result, scan_event_id=str(uuid.uuid4())) + owned = db.record_and_complete_target_result( + run_id, cycle_id, 'dockerhub', 'q', stale_target, result, {}, + row['id'], 'owner-b', 'done', + lease_token=newer_claim['lease_token'], + ) + assert owned and not owned['stale'] + assert db.conn.execute('SELECT COUNT(*) AS count FROM target_scans').fetchone()['count'] == before + 2 + + exhausted_target = 'owner/exhausted:tag' + db.enqueue_targets('dockerhub', 'docker', 'q', [exhausted_target]) + assert db.claim_targets('dockerhub', 'docker', 1, 'owner-x', 60, max_attempts=1) == [exhausted_target] + db.conn.execute( + "UPDATE target_queue SET lease_expires_at = ? WHERE normalized_target = ?", + ('2000-01-01T00:00:00+00:00', normalize_target(exhausted_target, 'docker')), + ) + db.conn.commit() + assert db.claim_targets('dockerhub', 'docker', 1, 'owner-y', 60, max_attempts=1) == [] + assert db.target_queue_item('dockerhub', 'docker', exhausted_target)['status'] == 'failed' + finally: + db.close() + finally: + for key, value in old_urls.items(): + if value is not None: + os.environ[key] = value + + +def main(): + for key in ('SCANNER_DB_URL', 'DATABASE_URL', 'TRUF_MANAGED_POSTGRES_DSN', 'KEYCHECK_DB_URL'): + os.environ.pop(key, None) + assert_diagnostic_policy() + assert_docker_platform_policy() + assert_docker_partial_pagination_policy() + assert_retry_policy() + assert_timeout_output_policy() + assert_archive_and_secret_safety() + assert_queue_policy() + print('scanner error policy smoke: OK') + + +if __name__ == '__main__': + main() diff --git a/app/supervisor.py b/app/supervisor.py new file mode 100644 index 0000000..a4d4b75 --- /dev/null +++ b/app/supervisor.py @@ -0,0 +1,6562 @@ +import sys +import os + +if __name__ == '__main__': + if sys.platform != 'linux' or not os.path.isfile('/.dockerenv') or os.path.abspath(__file__) != '/opt/truf/app/supervisor.py': + raise SystemExit('Docker development copy: runtime control is disabled outside the prepared container. See DOCKER_MIGRATION.md.') + import runpy + runpy.run_path('/opt/truf/app/container_runtime.py')['require_container']() + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('supervisor could not disable bytecode writes') + +import argparse +import json +import queue +import re +import selectors +import secrets +import shlex +import shutil +import signal +import socket +import socketserver +import sqlite3 +import stat +import subprocess +import threading +import time +import urllib.request +from concurrent.futures import ThreadPoolExecutor +from contextlib import redirect_stdout +from datetime import datetime, timezone +from io import StringIO +from urllib.parse import quote + + +if os.name == 'nt': + import ctypes + from ctypes import wintypes + + _SUPERVISOR_KERNEL32 = ctypes.WinDLL('kernel32', use_last_error=True) + _SUPERVISOR_OPEN_PROCESS = _SUPERVISOR_KERNEL32.OpenProcess + _SUPERVISOR_OPEN_PROCESS.argtypes = [wintypes.DWORD, wintypes.BOOL, wintypes.DWORD] + _SUPERVISOR_OPEN_PROCESS.restype = wintypes.HANDLE + _SUPERVISOR_GET_EXIT_CODE_PROCESS = _SUPERVISOR_KERNEL32.GetExitCodeProcess + _SUPERVISOR_GET_EXIT_CODE_PROCESS.argtypes = [ + wintypes.HANDLE, ctypes.POINTER(wintypes.DWORD), + ] + _SUPERVISOR_GET_EXIT_CODE_PROCESS.restype = wintypes.BOOL + _SUPERVISOR_CLOSE_HANDLE = _SUPERVISOR_KERNEL32.CloseHandle + _SUPERVISOR_CLOSE_HANDLE.argtypes = [wintypes.HANDLE] + _SUPERVISOR_CLOSE_HANDLE.restype = wintypes.BOOL +else: + _SUPERVISOR_KERNEL32 = None + _SUPERVISOR_OPEN_PROCESS = None + _SUPERVISOR_GET_EXIT_CODE_PROCESS = None + _SUPERVISOR_CLOSE_HANDLE = None + + +def _preimport_runtime_launch_requested(arguments): + flags = {str(value).split('=', 1)[0] for value in arguments if str(value).startswith('--')} + if flags.intersection({'--background', '--background-child'}): + return True + return not flags.intersection({'--dry-run', '--stop-background', '--background-status', '--attach', '--cmd'}) + + +def _preimport_is_reparse_point(path): + details = os.lstat(path) + if stat.S_ISLNK(details.st_mode): + return True + attributes = getattr(details, 'st_file_attributes', 0) + reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0) + return bool(attributes & reparse_attribute) or getattr(os.path, 'isjunction', lambda _path: False)(path) + + +def _preimport_reject_cached_bytecode(app_dir): + def raise_walk_error(exc): + raise SystemExit(f'Unable to inspect application root: {exc}') from exc + + try: + root_details = os.lstat(app_dir) + except OSError as exc: + raise SystemExit(f'Application root is unavailable: {app_dir}') from exc + if _preimport_is_reparse_point(app_dir): + raise SystemExit(f'Application root reparse point is forbidden: {app_dir}') + if not stat.S_ISDIR(root_details.st_mode): + raise SystemExit(f'Application root is not a directory: {app_dir}') + canonical_root = os.path.normcase(os.path.realpath(os.path.abspath(app_dir))) + for current, directories, files in os.walk(app_dir, followlinks=False, onerror=raise_walk_error): + for name in directories: + candidate = os.path.join(current, name) + if _preimport_is_reparse_point(candidate): + if name.lower() == '__pycache__': + raise SystemExit(f'Application bytecode cache link is forbidden: {candidate}') + raise SystemExit(f'Application directory reparse point is forbidden: {candidate}') + relative = os.path.relpath(current, app_dir) + in_cache = any(part.lower() == '__pycache__' for part in relative.split(os.sep)) + for name in files: + candidate = os.path.join(current, name) + if _preimport_is_reparse_point(candidate): + raise SystemExit(f'Application file reparse point is forbidden: {candidate}') + if name.lower().endswith(('.py', '.pyw', '.pyc', '.pyd')): + path = os.path.normcase(os.path.realpath(os.path.abspath(candidate))) + try: + contained = os.path.commonpath((canonical_root, path)) == canonical_root + except ValueError: + contained = False + if not contained: + raise SystemExit(f'Application Python authority escapes its root: {candidate}') + if in_cache and name.lower().endswith('.pyc'): + raise SystemExit( + f'Application __pycache__ bytecode is forbidden: ' + f'{os.path.relpath(candidate, app_dir)}' + ) + + +if __name__ == '__main__' and _preimport_runtime_launch_requested(sys.argv[1:]): + if '--with-postgres' not in sys.argv[1:]: + raise SystemExit('Supervisor unmanaged PostgreSQL mutation is retired; use --with-postgres.') + if not (sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode): + raise SystemExit('Mutating supervisor runtime requires python -I -S -B via runtime_bootstrap.py.') + if os.getenv('TRUF_RUNTIME_BOOTSTRAP') != '1': + raise SystemExit('Mutating supervisor runtime requires the canonical runtime bootstrap.') + _preimport_reject_cached_bytecode(os.path.dirname(os.path.abspath(__file__))) + + +from paths import apply_path_config, resolve_optional_path +from docker_depth_experiment import validate_docker_depth_config +from db_backend import connect_postgres +from owned_process import OwnedProcess +from process_identity import exact_process_identity_state, open_process +from postgres_runtime import PostgresState, canonical_database_url, controller_from_config +from lifecycle_authority import ( + DISCOVERY_PRODUCER_ROLE, + DISCOVERY_PRODUCER_SOURCES, + PHASE_ACTIVE, + PHASE_ACTIVATING, + PHASE_FAILED_HOLD, + PHASE_STOPPING, + LifecycleAuthorityError, + build_code_manifest, + code_manifest_sha256, + dsn_sha256, + strip_supervisor_credentials, + supervised_child_environment, + verify_code_manifest, +) +from runtime_security import ( + ClusterAuthorityLock, + durable_unlink, + harden_private_file, + preflight_lifecycle_paths, + private_directory_ready, + private_file_ready, + read_private_json, + reject_reparse_components, + require_private_directory, + sha256_file, +) +from supervisor_instance import ( + CONTROL_SCHEMA, + InstanceMetadataError, + InstanceLockError, + SupervisorInstanceLock, + authenticate_request, + build_instance_metadata, + load_instance_metadata, + load_shutdown_receipt, + is_loopback_host, + remove_instance_if_matches, + remove_shutdown_receipt, + update_instance_activation, + verify_instance_process, + write_shutdown_receipt, + write_instance_metadata, +) + +SOURCE_ALIASES = {'docker': 'dockerhub'} +KEYCHECK_SERVICE_NAMES = { + 'anthropic', 'aws', 'azure', 'deepseek', 'dockerhub', 'gcp', 'gemini', 'github', 'gitlab', + 'groq', 'huggingface', 'kimi', 'openai', 'openrouter', 'provider_resolver', 'qwen', + 'replicate', 'xai', 'zai', +} +RECHECK_TYPE_FLAGS = { + 'network': '--retry-network', + 'net': '--retry-network', + 'limited': '--retry-limited', + 'limit': '--retry-limited', + 'ratelimited': '--retry-limited', + 'rate-limited': '--retry-limited', + 'rate-limit': '--retry-limited', + 'aliveratelimited': '--retry-limited', + 'alive-rate-limited': '--retry-limited', + 'alive-limited': '--retry-limited', + 'validratelimited': '--retry-limited', + 'valid-rate-limited': '--retry-limited', + 'valid-limited': '--retry-limited', + 'unknown': '--retry-unknown', + 'restricted': '--retry-restricted', + 'restriction': '--retry-restricted', + 'nobalance': '--retry-no-balance', + 'no-balance': '--retry-no-balance', + 'noquota': '--retry-no-balance', + 'no-quota': '--retry-no-balance', + 'quota': '--retry-no-balance', + 'valid': '--retry-valid', + 'alive': '--retry-valid', + 'legacy-vertex': '--import-legacy-gcp-vertex', + 'legacyvertex': '--import-legacy-gcp-vertex', + 'all': '--recheck-all', + 'everything': '--recheck-all', +} +RECHECK_BOOL_OPTIONS = { + 'no-resource-probe': '--no-resource-probe', + 'no-summary': '--no-summary', + 'summary-only': '--summary-only', +} +RECHECK_VALUE_OPTIONS = { + 'max-keys': '--max-keys', + 'max': '--max-keys', + 'limit': '--max-keys', + 'input': '--input', + 'proxy-file': '--proxy-file', + 'proxy': '--proxy-file', +} +DEFAULT_REFRESH_SEC = 5 +DEFAULT_RESTART_DELAY = 30 +SOURCE_INFRASTRUCTURE_HOLD_EXIT = 75 +DEFAULT_MAX_RESTART_DELAY = 600 +DEFAULT_RESTART_RESET_AFTER = 300 +DEFAULT_TEMP_CLEANUP_INTERVAL = 900 +ANSI_ALT_SCREEN = '\x1b[?1049h' +ANSI_MAIN_SCREEN = '\x1b[?1049l' +ANSI_HOME = '\x1b[H' +ANSI_CLEAR_SCREEN = '\x1b[2J' +DETACHED_PROCESS = 0x00000008 +CREATE_NO_WINDOW = 0x08000000 +MAX_CONTROL_REQUEST_BYTES = 64 * 1024 +MAX_CONTROL_RESPONSE_BYTES = 2 * 1024 * 1024 +MAX_CONTROL_PENDING_SOCKETS = 32 +MAX_CONTROL_WORKERS = 4 +CONTROL_READ_TIMEOUT_SEC = 1.0 +MAX_LOG_TAIL_BYTES = 1024 * 1024 +MAX_LOG_TAIL_LINES = 5000 +RUNTIME_SNAPSHOT_SCHEMA = 2 +MAX_MANAGED_SOURCE_DELAY_SECONDS = 365 * 24 * 60 * 60 +MANAGED_SOURCE_LIFECYCLE_ACTIONS = ('start', 'stop', 'restart', 'pause', 'resume') +MANAGED_SOURCE_SETTING_ACTIONS = ( + 'once', 'set-mode', 'set-interval', 'set-restart', 'set-restart-delay', +) +DISCOVERY_CYCLE_STATUSES = frozenset({ + 'running', 'completed', 'completed_with_retries', 'query_invalid', 'failed', + 'source_failed', 'backlog_only', 'paused', 'auth_failed', 'rate_limited', +}) +DISCOVERY_ERROR_CATEGORIES = frozenset({ + 'auth_failed', 'failed', 'paused', 'query_invalid', 'rate_limited', + 'source_failed', 'runtime_error', +}) +RUNTIME_BOOTSTRAP_ENV = 'TRUF_RUNTIME_BOOTSTRAP' +RUNTIME_BOOTSTRAP_VALUE = '1' + + +def child_bootstrap_command(kind, arguments, provider_entrypoint=None): + command = [ + sys.executable, + '-I', + '-S', + '-B', + os.path.join(os.path.dirname(os.path.abspath(__file__)), 'child_bootstrap.py'), + str(kind), + ] + if provider_entrypoint: + command.append(str(provider_entrypoint).replace('\\', '/')) + command.append('--') + command.extend(str(value) for value in arguments) + return command + + +def load_yaml(path, *, managed_postgres=None, final_cutover=None): + try: + import yaml + except ImportError as e: + raise SystemExit('PyYAML is required. Run: python -m pip install -r requirements.txt') from e + with open(path, 'r', encoding='utf-8') as f: + loaded = yaml.safe_load(f) + if loaded is None: + loaded = {} + validated = validate_docker_depth_config( + loaded, + managed_postgres=managed_postgres, + final_cutover=final_cutover, + ) + return apply_path_config(validated.config, path) + + +def resolve_path(base_file, value): + if not value: + return value + if os.path.isabs(str(value)): + return str(value) + return os.path.join(os.path.dirname(os.path.abspath(base_file)), str(value)) + + +def load_postgres_env(config_path=None, layout=None, enforce_canonical=False): + candidates = [] + root_dir = (layout or {}).get('root_dir') + if root_dir: + candidates.append(os.path.join(root_dir, '.env.postgres')) + if config_path: + config_dir = os.path.dirname(os.path.abspath(config_path)) + candidates.append(os.path.join(config_dir, '..', '.env.postgres')) + candidates.append(os.path.join(config_dir, '.env.postgres')) + seen = set() + loaded = None + for path in candidates: + path = os.path.abspath(path) + if path in seen: + continue + seen.add(path) + if not os.path.exists(path): + continue + try: + with open(path, 'r', encoding='utf-8') as f: + for line in f: + text = line.strip() + if not text or text.startswith('#') or '=' not in text: + continue + key, value = text.split('=', 1) + key = key.strip() + value = value.strip().strip('"').strip("'") + if key and value and not os.getenv(key): + os.environ[key] = value + except OSError: + continue + loaded = path + if os.getenv('SCANNER_DB_URL') or os.getenv('DATABASE_URL'): + break + password = os.getenv('TRUF_POSTGRES_PASSWORD') + if password: + user = os.getenv('TRUF_POSTGRES_USER') or 'truf' + database = os.getenv('TRUF_POSTGRES_DB') or 'truf' + port = os.getenv('TRUF_POSTGRES_PORT') or '5432' + os.environ['SCANNER_DB_URL'] = ( + f'postgresql://{quote(user, safe="")}:{quote(password, safe="")}@127.0.0.1:{port}/{quote(database, safe="")}' + ) + break + if enforce_canonical: + try: + url = canonical_database_url() + except ValueError as exc: + raise SystemExit(f'Managed PostgreSQL database URL failed closed: {exc}') from exc + if url: + for key in list(os.environ): + if key.upper().startswith('PG'): + os.environ.pop(key, None) + os.environ['SCANNER_DB_URL'] = url + os.environ['DATABASE_URL'] = url + os.environ['SCANNER_DASHBOARD_DB_URL'] = url + os.environ['TRUF_MANAGED_POSTGRES_DSN'] = url + return loaded + + +def normalize_source_name(source): + source = str(source or '').strip().lower() + return SOURCE_ALIASES.get(source, source) + + +def parse_source_list(value): + if not value: + return None + if isinstance(value, str): + parts = value.split(',') + else: + parts = value + return [normalize_source_name(item) for item in parts if str(item).strip()] + + +def normalize_recheck_token(value): + return str(value or '').strip().lower().lstrip('-').replace('_', '-') + + +def append_unique(items, value): + if value not in items: + items.append(value) + + +def parse_recheck_command(parts): + if len(parts) < 2: + return None, 'Usage: recheck [network|ratelimited|unknown|restricted|nobalance|valid|legacy-vertex|all] [--force] [--max-keys N] [--input PATH] [--proxy-file PATH]' + + first = normalize_source_name(parts[1]) + first_norm = normalize_recheck_token(first) + if first_norm in RECHECK_TYPE_FLAGS and first_norm != 'all': + service = 'all' + tokens = parts[1:] + else: + service = 'all' if first_norm == 'all' else first + tokens = parts[2:] + + if service != 'all' and service not in KEYCHECK_SERVICE_NAMES: + return None, f'Unknown keycheck service: {service}. Available: all, {", ".join(sorted(KEYCHECK_SERVICE_NAMES))}' + + runner_args = [] + retry_flag_seen = False + force = False + index = 0 + while index < len(tokens): + token = tokens[index] + norm = normalize_recheck_token(token) + if not norm: + index += 1 + continue + if norm in ('force', 'f'): + force = True + index += 1 + continue + if norm in RECHECK_TYPE_FLAGS: + append_unique(runner_args, RECHECK_TYPE_FLAGS[norm]) + retry_flag_seen = True + index += 1 + continue + if norm in RECHECK_BOOL_OPTIONS: + append_unique(runner_args, RECHECK_BOOL_OPTIONS[norm]) + index += 1 + continue + matched_value_option = None + matched_value = None + for option_name, flag in RECHECK_VALUE_OPTIONS.items(): + prefix = option_name + '=' + if norm.startswith(prefix): + matched_value_option = flag + matched_value = token.split('=', 1)[1] + break + if matched_value_option: + if matched_value == '': + return None, f'{token} requires a value' + runner_args.extend([matched_value_option, matched_value]) + index += 1 + continue + if norm in RECHECK_VALUE_OPTIONS: + if index + 1 >= len(tokens): + return None, f'{token} requires a value' + runner_args.extend([RECHECK_VALUE_OPTIONS[norm], tokens[index + 1]]) + index += 2 + continue + return None, f'Unknown recheck option: {token}' + + if not retry_flag_seen: + append_unique(runner_args, '--recheck-all') + return {'service': service, 'runner_args': runner_args, 'force': force}, None + + +def bool_value(value, default=False): + if value is None: + return default + if isinstance(value, bool): + return value + return str(value).strip().lower() in ('1', 'true', 'yes', 'on') + + +def list_value(value): + if not value: + return [] + if isinstance(value, str): + return [item for item in value.split() if item] + return [str(item) for item in value] + + +PROXY_ENV_KEYS = ( + 'HTTP_PROXY', 'HTTPS_PROXY', 'ALL_PROXY', + 'http_proxy', 'https_proxy', 'all_proxy', +) + + +def safe_source_filename(source): + return ''.join(ch if ch.isalnum() or ch in ('-', '_') else '_' for ch in source) + + +def int_value(value, default): + try: + return int(value) + except (TypeError, ValueError): + return default + + +def next_rotated_path(path): + base, ext = os.path.splitext(path) + seq = 1 + while True: + candidate = f'{base}.{seq:06d}{ext or ".log"}' + if not os.path.exists(candidate): + return candidate + seq += 1 + + +def prune_rotated_logs(path, keep): + keep = max(0, int_value(keep, 0)) + base, ext = os.path.splitext(path) + parent = os.path.dirname(path) or '.' + prefix = os.path.basename(base) + '.' + suffix = ext or '.log' + try: + rotated = [] + for name in os.listdir(parent): + if name.startswith(prefix) and name.endswith(suffix): + full = os.path.join(parent, name) + if os.path.isfile(full): + rotated.append(full) + rotated.sort(key=lambda item: os.path.getmtime(item), reverse=True) + for old in rotated[keep:]: + try: + os.remove(old) + except OSError: + pass + except OSError: + pass + + +def rotate_log_if_needed(path, max_mb=64, keep=5): + max_bytes = max(0, int_value(max_mb, 64)) * 1024 * 1024 + if max_bytes <= 0 or not path or not os.path.exists(path): + return + try: + if os.path.getsize(path) < max_bytes: + return + rotated_path = next_rotated_path(path) + os.replace(path, rotated_path) + prune_rotated_logs(path, keep) + except OSError: + pass + + +def open_private_append(path): + parent = os.path.dirname(os.path.abspath(path)) + require_private_directory(parent, create=False) + reject_reparse_components(path) + flags = os.O_WRONLY | os.O_APPEND | os.O_CREAT + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + if hasattr(os, 'O_NOFOLLOW'): + flags |= os.O_NOFOLLOW + created = False + if os.path.lexists(path): + if not private_file_ready(path): + raise OSError(f'private log file ACL is not ready; run offline hardening: {path}') + descriptor = os.open(path, flags, 0o600) + else: + try: + descriptor = os.open(path, flags | os.O_EXCL, 0o600) + created = True + except FileExistsError: + if not private_file_ready(path): + raise OSError(f'private log file ACL is not ready; run offline hardening: {path}') + descriptor = os.open(path, flags, 0o600) + try: + if created: + harden_private_file(path) + return os.fdopen(descriptor, 'a', encoding='utf-8', buffering=1) + except BaseException: + os.close(descriptor) + raise + + +def open_private_append_binary(path): + parent = os.path.dirname(os.path.abspath(path)) + require_private_directory(parent, create=False) + reject_reparse_components(path) + flags = os.O_WRONLY | os.O_APPEND | os.O_CREAT | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0) + created = False + if os.path.lexists(path): + if not private_file_ready(path): + raise OSError(f'private log file ACL is not ready; run offline hardening: {path}') + descriptor = os.open(path, flags, 0o600) + else: + try: + descriptor = os.open(path, flags | os.O_EXCL, 0o600) + created = True + except FileExistsError: + if not private_file_ready(path): + raise OSError(f'private log file ACL is not ready; run offline hardening: {path}') + descriptor = os.open(path, flags, 0o600) + try: + if created: + harden_private_file(path) + return os.fdopen(descriptor, 'ab', buffering=0) + except BaseException: + os.close(descriptor) + raise + + +class BoundedRotatingLogPump: + """Drain one child pipe into privately-owned bounded log segments.""" + + def __init__(self, path, max_bytes, keep): + self.path = os.path.abspath(path) + self.max_bytes = max(1, int(max_bytes)) + self.keep = max(1, int(keep)) + self._lock = threading.RLock() + self.handle = open_private_append_binary(self.path) + self.thread = None + self.error = '' + + def _rotate(self): + self.handle.flush() + opened = os.fstat(self.handle.fileno()) + current = os.stat(self.path, follow_symlinks=False) + if (opened.st_dev, opened.st_ino) != (current.st_dev, current.st_ino): + raise OSError('active source log path changed while its bounded writer was open') + if opened.st_size > self.max_bytes: + os.ftruncate(self.handle.fileno(), self.max_bytes) + self.handle.flush() + os.fsync(self.handle.fileno()) + self.handle.close() + self.handle = None + rotated = next_rotated_path(self.path) + os.replace(self.path, rotated) + harden_private_file(rotated) + prune_rotated_logs(self.path, self.keep) + self.handle = open_private_append_binary(self.path) + + def write(self, data): + data = data.encode('utf-8', errors='replace') if isinstance(data, str) else bytes(data or b'') + with self._lock: + view = memoryview(data) + while view: + size = os.fstat(self.handle.fileno()).st_size + if size >= self.max_bytes: + self._rotate() + size = 0 + portion = view[:max(1, self.max_bytes - size)] + self.handle.write(portion) + view = view[len(portion):] + + def flush(self): + with self._lock: + if self.handle is not None: + self.handle.flush() + + def attach(self, stream): + def drain(): + try: + while True: + chunk = stream.read(64 * 1024) + if not chunk: + break + if not self.error: + try: + self.write(chunk) + except Exception as exc: + self.error = f'{type(exc).__name__}: {exc}' + finally: + try: + stream.close() + except OSError: + pass + self.close() + + self.thread = threading.Thread(target=drain, name='source-log-pump', daemon=True) + self.thread.start() + + def join(self, timeout=5): + if self.thread is not None and self.thread.ident is not None: + self.thread.join(timeout=max(0.0, float(timeout))) + if self.thread.is_alive() and not self.error: + self.error = 'log pipe did not close after child exit' + else: + self.close() + + def close(self): + with self._lock: + handle = self.handle + self.handle = None + if handle is not None: + try: + handle.flush() + os.fsync(handle.fileno()) + except OSError: + pass + handle.close() + + +class BoundedRotatingTextWriter: + encoding = 'utf-8' + + def __init__(self, path, max_bytes, keep): + self.sink = BoundedRotatingLogPump(path, max_bytes, keep) + self.closed = False + + def write(self, value): + if self.closed: + return 0 + text = str(value or '') + self.sink.write(text) + return len(text) + + def flush(self): + if not self.closed: + self.sink.flush() + + def isatty(self): + return False + + def close(self): + if not self.closed: + self.closed = True + self.sink.close() + + +def install_bounded_background_output(path, max_bytes, keep): + writer = BoundedRotatingTextWriter(path, max_bytes, keep) + previous = (sys.stdout, sys.stderr) + sys.stdout = writer + sys.stderr = writer + for stream in dict.fromkeys(previous): + try: + stream.flush() + stream.close() + except Exception: + pass + return writer + + +def append_bounded_log_record(path, max_bytes, keep, text): + sink = BoundedRotatingLogPump(path, max_bytes, keep) + try: + sink.write(str(text).encode('utf-8', errors='replace')) + finally: + sink.close() + + +def bounded_tail_lines(path, limit=40, max_bytes=MAX_LOG_TAIL_BYTES): + if not path or not os.path.exists(path): + return [] + limit = min(MAX_LOG_TAIL_LINES, max(1, int(limit or 40))) + max_bytes = max(1, int(max_bytes)) + try: + with open(path, 'rb') as handle: + handle.seek(0, os.SEEK_END) + size = handle.tell() + start = max(0, size - max_bytes) + handle.seek(start) + data = handle.read(max_bytes) + if start and data: + newline = data.find(b'\n') + data = data[newline + 1:] if newline >= 0 else b'' + return [line.decode('utf-8', errors='replace') for line in data.splitlines()[-limit:]] + except OSError: + return [] + + +def console_safe_text(value, stream=None): + text = str(value) + encoding = getattr(stream or sys.stdout, 'encoding', None) or 'utf-8' + try: + text.encode(encoding, errors='strict') + return text + except (LookupError, UnicodeEncodeError): + try: + return text.encode(encoding, errors='backslashreplace').decode(encoding, errors='replace') + except LookupError: + return text.encode('ascii', errors='backslashreplace').decode('ascii') + + +def format_duration(seconds): + if seconds is None: + return '-' + seconds = max(0, int(seconds)) + hours, rem = divmod(seconds, 3600) + minutes, sec = divmod(rem, 60) + if hours: + return f'{hours}h{minutes:02d}m' + if minutes: + return f'{minutes}m{sec:02d}s' + return f'{sec}s' + + +def format_exit_code(code): + if code is None: + return '-' + if os.name == 'nt' and (code < 0 or code > 255): + return f'0x{code & 0xFFFFFFFF:08X}' + return str(code) + + +def now_iso(): + return datetime.now().isoformat(timespec='seconds') + + +def safe_state_timestamp(value): + if type(value) is not str or 'T' not in value or len(value) > 64: + return None + try: + datetime.fromisoformat(value) + except ValueError: + return None + return value + + +def finalize_source_runs(database_url, source, status, reason): + if not database_url: + return 0, 0 + connection = connect_postgres( + database_url, + connect_timeout_sec=5, + statement_timeout_ms=10000, + lock_timeout_ms=5000, + idle_in_transaction_timeout_ms=10000, + tcp_user_timeout_ms=5000, + ) + timestamp = datetime.now().astimezone().isoformat(timespec='seconds') + try: + cycles = connection.execute( + '''UPDATE source_cycles SET ended_at = ?, status = ?, + message = COALESCE(NULLIF(message, ''), ?) + WHERE source = ? AND status = 'running' ''', + (timestamp, status, reason, source), + ) + runs = connection.execute( + '''UPDATE runs SET ended_at = ?, status = ?, + error = COALESCE(NULLIF(error, ''), ?), updated_at = ? + WHERE selected_source = ? AND status = 'running' ''', + (timestamp, status, reason, timestamp, source), + ) + connection.commit() + return int(cycles.rowcount or 0), int(runs.rowcount or 0) + except Exception: + connection.rollback() + raise + finally: + connection.close() + + +def terminal_width(default=120): + try: + return max(80, shutil.get_terminal_size((default, 24)).columns) + except OSError: + return default + + +def truncate_text(value, width): + value = str(value or '') + if width <= 0: + return '' + if len(value) <= width: + return value + if width <= 1: + return value[:width] + return value[:width - 1] + '~' + + +def enable_ansi_terminal(): + if os.name != 'nt': + return True + try: + import ctypes + kernel32 = ctypes.windll.kernel32 + handle = kernel32.GetStdHandle(-11) + mode = ctypes.c_uint32() + if not kernel32.GetConsoleMode(handle, ctypes.byref(mode)): + return False + return bool(kernel32.SetConsoleMode(handle, mode.value | 0x0004)) + except Exception: + return False + + +def parse_time(value): + if not value: + return None + try: + parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) + return parsed + except ValueError: + return None + + +def auth_item_kind(item): + if not isinstance(item, dict): + return 'ok' + if item.get('disabled_until') == 'manual' or item.get('status') == 'dead' or item.get('disabled_reason') == 'auth_invalid': + return 'dead' + disabled_until = parse_time(item.get('disabled_until')) + if disabled_until: + now = datetime.now(disabled_until.tzinfo) if disabled_until.tzinfo else datetime.now() + if disabled_until > now: + return 'limited' + return 'ok' + + +def auth_counts_from_status(auth_status): + counts = {'ok': 0, 'dead': 0, 'limited': 0, 'rate_limit_errors': 0, 'auth_invalid_errors': 0} + for item in (auth_status or {}).values(): + kind = auth_item_kind(item) + counts[kind] += 1 + counts['rate_limit_errors'] += int((item or {}).get('rate_limit_count', 0) or 0) + counts['rate_limit_errors'] += int((item or {}).get('secondary_rate_limit_count', 0) or 0) + counts['auth_invalid_errors'] += int((item or {}).get('auth_invalid_count', 0) or 0) + if (item or {}).get('disabled_reason') == 'auth_invalid' and not (item or {}).get('auth_invalid_count'): + counts['auth_invalid_errors'] += int((item or {}).get('failures', 1) or 1) + return counts + + +def source_options(source, supervisor_config, source_config): + defaults = supervisor_config.get('defaults') or {} + per_source = (supervisor_config.get('sources') or {}).get(source, {}) or {} + options = dict(defaults) + options.update(per_source) + if 'enabled' not in options: + options['enabled'] = source_config.get('enabled', False) + return options + + +def get_enabled_sources(config, selected_sources=None, supervisor_config=None): + sources = config.get('sources') or {} + selected = parse_source_list(selected_sources) + if selected and 'all' in selected: + selected = None + if selected: + selected_sources_only = [source for source in selected if source != 'keychecks'] + missing = [source for source in selected_sources_only if source not in sources] + if missing: + raise SystemExit(f'Source(s) not present in config: {", ".join(missing)}') + return selected_sources_only + supervisor_sources = (supervisor_config or {}).get('sources') or {} + enabled = [] + for name, source in sources.items(): + supervisor_source = supervisor_sources.get(name) or {} + if source.get('enabled', False) or bool_value(supervisor_source.get('enabled'), False): + enabled.append(name) + return enabled + + +class DependencyGate: + def __init__(self, ready=True, database_url=''): + self.ready = bool(ready) + self.database_url = str(database_url or '') + self._dependents = [] + self.stop_failures = [] + + def register(self, dependent): + if dependent not in self._dependents: + self._dependents.append(dependent) + + def unregister(self, dependent): + if dependent in self._dependents: + self._dependents.remove(dependent) + + def set_ready(self, ready): + ready = bool(ready) + if ready == self.ready and not (not ready and self.stop_failures): + return + if ready and self.stop_failures: + raise DependencyStopError(self.stop_failures) + self.ready = ready + failures = [] + for dependent in list(self._dependents): + try: + result = dependent.dependency_available() if ready else dependent.dependency_unavailable() + if not ready and result is False: + failures.append(getattr(dependent, 'source', dependent.__class__.__name__)) + except Exception as exc: + failures.append(f'{getattr(dependent, "source", dependent.__class__.__name__)}: {exc}') + self.stop_failures = failures if not ready else [] + if failures: + raise DependencyStopError(failures) + return True + + def force_database_environment(self, env): + if not self.database_url: + return env + for key in list(env): + if key.upper().startswith('PG'): + env.pop(key, None) + env['SCANNER_DB_URL'] = self.database_url + env['DATABASE_URL'] = self.database_url + env['SCANNER_DASHBOARD_DB_URL'] = self.database_url + env['KEYCHECK_DB_URL'] = self.database_url + env['TRUF_MANAGED_POSTGRES_DSN'] = self.database_url + return env + + +class DependencyStopError(RuntimeError): + def __init__(self, failures): + self.failures = [str(item) for item in failures] + super().__init__('database-dependent child stop failed: ' + ', '.join(self.failures)) + + +class ManagedSource: + def __init__( + self, + source, + config_path, + project_dir, + results_dir, + supervisor_config, + source_config, + global_force_once=False, + dependency_gate=None, + authority_check=None, + start_gate=None, + child_environment=None, + ): + self.source = source + self.config_path = os.path.abspath(config_path) + self.project_dir = project_dir + self.results_dir = results_dir + self.options = source_options(source, supervisor_config, source_config) + self.once = bool_value(self.options.get('once'), False) or global_force_once + self.repeat = bool_value(self.options.get('repeat'), self.once) + self.restart = bool_value(self.options.get('restart'), True) + self.enabled = bool_value(self.options.get('enabled'), True) + self.interval = int(self.options.get('interval', self.options.get('cooldown', supervisor_config.get('interval', 0))) or 0) + self.restart_delay = int(self.options.get('restart_delay', supervisor_config.get('restart_delay', DEFAULT_RESTART_DELAY)) or DEFAULT_RESTART_DELAY) + self.max_restart_delay = int(self.options.get('max_restart_delay', supervisor_config.get('max_restart_delay', DEFAULT_MAX_RESTART_DELAY)) or DEFAULT_MAX_RESTART_DELAY) + self.restart_reset_after = int(self.options.get('restart_reset_after', supervisor_config.get('restart_reset_after', DEFAULT_RESTART_RESET_AFTER)) or DEFAULT_RESTART_RESET_AFTER) + self.extra_args = list_value(self.options.get('extra_args')) + self.env_overrides = self.options.get('env') if isinstance(self.options.get('env'), dict) else {} + self.use_per_source_state = bool_value(supervisor_config.get('per_source_state'), True) + self.log_max_mb = int_value(self.options.get('log_max_mb', supervisor_config.get('log_max_mb', 64)), 64) + self.log_keep = int_value(self.options.get('log_keep', supervisor_config.get('log_keep', 5)), 5) + + log_dir = resolve_path(config_path, supervisor_config.get('log_dir') or os.path.join(results_dir, 'logs')) + state_dir = resolve_path(config_path, supervisor_config.get('state_dir') or os.path.join(results_dir, 'state')) + require_private_directory(log_dir, create=False) + require_private_directory(state_dir, create=False) + safe_name = safe_source_filename(source) + self.log_path = os.path.join(log_dir, f'{safe_name}.log') + self.state_path = os.path.join(state_dir, f'runner_state_{safe_name}.json') + + self.process = None + self.payload_identity = None + self.startup_cleanup_pending = False + self.log_handle = None + self.log_pump = None + self.started_at = None + self.last_exit_code = None + self.last_exit_at = None + self.restarts = 0 + self.restart_streak = 0 + self.next_start_at = 0 + self.status = 'disabled' if not self.enabled else 'stopped' + self.stop_requested = False + self.manual_stop = True + self.paused = False + self.desired_state = 'stopped' + self.runtime_blocked = False + self._resume_immediately = False + self.dependency_gate = dependency_gate + self.authority_check = authority_check + self.start_gate = start_gate + self.child_environment = dict(child_environment or {}) + self.manual_only = False + self.last_action_error = '' + if self.dependency_gate is not None: + self.dependency_gate.register(self) + + def is_running(self): + return self.process is not None and self.process.poll() is None + + def reconfigure(self, results_dir, supervisor_config, source_config, global_force_once=False): + self.results_dir = results_dir + self.options = source_options(self.source, supervisor_config, source_config) + self.once = bool_value(self.options.get('once'), False) or global_force_once + self.repeat = bool_value(self.options.get('repeat'), self.once) + self.restart = bool_value(self.options.get('restart'), True) + self.enabled = bool_value(self.options.get('enabled'), True) + self.interval = int(self.options.get('interval', self.options.get('cooldown', supervisor_config.get('interval', 0))) or 0) + self.restart_delay = int(self.options.get('restart_delay', supervisor_config.get('restart_delay', DEFAULT_RESTART_DELAY)) or DEFAULT_RESTART_DELAY) + self.max_restart_delay = int(self.options.get('max_restart_delay', supervisor_config.get('max_restart_delay', DEFAULT_MAX_RESTART_DELAY)) or DEFAULT_MAX_RESTART_DELAY) + self.restart_reset_after = int(self.options.get('restart_reset_after', supervisor_config.get('restart_reset_after', DEFAULT_RESTART_RESET_AFTER)) or DEFAULT_RESTART_RESET_AFTER) + self.extra_args = list_value(self.options.get('extra_args')) + self.env_overrides = self.options.get('env') if isinstance(self.options.get('env'), dict) else {} + self.use_per_source_state = bool_value(supervisor_config.get('per_source_state'), True) + self.log_max_mb = int_value(self.options.get('log_max_mb', supervisor_config.get('log_max_mb', 64)), 64) + self.log_keep = int_value(self.options.get('log_keep', supervisor_config.get('log_keep', 5)), 5) + + log_dir = resolve_path(self.config_path, supervisor_config.get('log_dir') or os.path.join(results_dir, 'logs')) + state_dir = resolve_path(self.config_path, supervisor_config.get('state_dir') or os.path.join(results_dir, 'state')) + require_private_directory(log_dir, create=False) + require_private_directory(state_dir, create=False) + safe_name = safe_source_filename(self.source) + self.log_path = os.path.join(log_dir, f'{safe_name}.log') + self.state_path = os.path.join(state_dir, f'runner_state_{safe_name}.json') + if not self.enabled: + self.status = 'disabled' + + def build_command(self): + arguments = [ + '--config', + self.config_path, + '--source', + self.source, + ] + if self.once: + arguments.append('--once') + arguments.extend(self.extra_args) + return child_bootstrap_command('scanner', arguments) + + def build_env(self): + env = os.environ.copy() + use_system_proxy = bool_value(self.options.get('use_system_proxy'), False) + if not use_system_proxy: + for key in PROXY_ENV_KEYS: + env.pop(key, None) + env['NO_PROXY'] = '*' + env['no_proxy'] = '*' + else: + env.pop('NO_PROXY', None) + env.pop('no_proxy', None) + env['PYTHONUNBUFFERED'] = '1' + env['PYTHONIOENCODING'] = 'utf-8' + env['SCANNER_SOURCE'] = self.source + env['SCANNER_SKIP_STARTUP_CLEANUP'] = '1' + if self.use_per_source_state: + env['RUNNER_STATE_FILE'] = self.state_path + for key, value in self.env_overrides.items(): + env[str(key)] = str(value) + if self.source == 'janitor': + for key in list(env): + if key.upper().startswith('PG') or key.upper() in { + 'TRUF_MANAGED_POSTGRES_DSN', 'SCANNER_DB_URL', 'DATABASE_URL', + 'SCANNER_DASHBOARD_DB_URL', 'KEYCHECK_DB_URL', + }: + env.pop(key, None) + elif self.dependency_gate is not None: + self.dependency_gate.force_database_environment(env) + env.update(self.child_environment) + if self.source == 'janitor': + env.pop('TRUF_MANAGED_POSTGRES_DSN', None) + env.pop('SCANNER_DB_URL', None) + env.pop('DATABASE_URL', None) + return env + + def mode_label(self): + if self.once and self.repeat: + return 'once+repeat' + if self.once: + return 'once' + return 'loop' + + def finalize_database_runs(self, status, reason): + database_url = self.child_environment.get('SCANNER_DB_URL') + if not database_url and self.dependency_gate is not None: + database_url = self.dependency_gate.database_url + return finalize_source_runs(database_url, self.source, status, reason) + + def schedule_start(self, delay=0): + if self.startup_cleanup_pending: + return + if not self.enabled: + return + if self.start_gate is not None and not self.start_gate(): + self.last_action_error = 'supervisor lifecycle start gate is closed' + self.status = 'stopped' + return + self.desired_state = 'running' + self.manual_stop = False + self.paused = False + self.stop_requested = False + self.next_start_at = time.time() + max(0, int(delay or 0)) + if self.dependency_gate is not None and not self.dependency_gate.ready: + self.runtime_blocked = True + self.status = 'blocked' + else: + self.status = 'waiting' + + def start(self, force=False, record_intent=True): + if self.startup_cleanup_pending: + self.status = 'failed' + return False + self.last_action_error = '' + if self.start_gate is not None and not self.start_gate(): + self.last_action_error = 'supervisor lifecycle start gate is closed' + self.status = 'stopped' + return False + if self.authority_check is not None and not self.authority_check(): + self.last_action_error = 'runtime script/config authority drifted' + self.status = 'failed' + return False + if not self.enabled: + return False + if record_intent: + self.desired_state = 'running' + self.manual_stop = False + self.paused = False + self.stop_requested = False + if self.desired_state != 'running': + return False + if self.dependency_gate is not None and not self.dependency_gate.ready: + self.runtime_blocked = True + self._resume_immediately = self._resume_immediately or force or self.status != 'waiting' + self.status = 'blocked' + return False + if self.process and self.process.poll() is None: + return True + now = time.time() + if not force and now < self.next_start_at: + self.status = 'waiting' + return False + + if force: + self.restart_streak = 0 + self.last_exit_code = None + + self.runtime_blocked = False + self._resume_immediately = False + self.manual_stop = False + self.paused = False + self.stop_requested = False + self.next_start_at = 0 + + try: + log_max_bytes = max(1, self.log_max_mb) * 1024 * 1024 + self.log_pump = BoundedRotatingLogPump(self.log_path, log_max_bytes, self.log_keep) + self.log_pump.write(f'\n=== supervisor start {now_iso()} source={self.source} once={self.once} ===\n'.encode('utf-8')) + self.log_pump.write(('command: ' + ' '.join(command_for_log(self.build_command())) + '\n').encode('utf-8')) + if self.use_per_source_state: + self.log_pump.write(f'RUNNER_STATE_FILE={self.state_path}\n'.encode('utf-8')) + + creationflags = (subprocess.CREATE_NEW_PROCESS_GROUP | CREATE_NO_WINDOW) if os.name == 'nt' else 0 + self.process = OwnedProcess( + self.build_command(), + cwd=self.project_dir, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + stdin=subprocess.DEVNULL, + env=self.build_env(), + creationflags=creationflags, + ) + try: + with open_process(self.process.pid) as retained: + self.payload_identity = retained.identity + except OSError: + self.payload_identity = None + stream = getattr(self.process, 'stdout', None) + if stream is not None: + self.log_pump.attach(stream) + else: + self.log_pump.close() + except BaseException as exc: + self.startup_cleanup_pending = self.process is not None + self.started_at = None + self.status = 'failed' + self.last_action_error = f'owned process launch failed: {type(exc).__name__}: {exc}' + cleanup_error = None + if self.process is not None: + try: + if self.process.poll() is None: + self.process.terminate() + self.process.wait(timeout=5) + if self.process.poll() is not None: + self.process = None + self.payload_identity = None + self.startup_cleanup_pending = False + except BaseException as cleanup_exc: + cleanup_error = cleanup_exc + if self.startup_cleanup_pending: + # Only the retained owner can confirm rollback; a failed wait + # must not make this source (or its peers) eligible to start. + self.desired_state = 'stopped' + self.manual_stop = True + self.stop_requested = True + self.next_start_at = 0 + self.last_action_error += '; owned child exit is unconfirmed; shutdown retry required' + elif self.log_pump is not None: + try: + try: + if self.log_pump.thread is None and self.log_pump.handle is not None: + self.log_pump.write( + f'=== supervisor launch failed {now_iso()} source={self.source}: {type(exc).__name__} ===\n'.encode('utf-8') + ) + finally: + self.log_pump.join(timeout=1) + except BaseException as cleanup_exc: + cleanup_error = cleanup_exc + else: + self.log_pump = None + if not isinstance(exc, Exception): + raise + if cleanup_error is not None and not isinstance(cleanup_error, Exception): + raise cleanup_error + return False + self.started_at = now + self.status = 'running' + return True + + def poll(self): + if self.startup_cleanup_pending: + return + if not self.enabled: + return + if self.dependency_gate is not None and not self.dependency_gate.ready: + if self.desired_state == 'running' and not self.runtime_blocked: + self.dependency_unavailable() + return + if not self.process: + if self.status == 'waiting' and self.desired_state == 'running' and not self.stop_requested: + self.start(record_intent=False) + return + code = self.process.poll() + if code is None: + if self.log_pump is not None and self.log_pump.error: + self.last_action_error = f'bounded source log writer failed: {self.log_pump.error}' + try: + self.process.terminate() + self.process.wait(timeout=5) + except Exception: + return + code = self.process.poll() + else: + self.status = 'running' + if self.started_at and time.time() - self.started_at >= self.restart_reset_after: + self.restart_streak = 0 + self.last_exit_code = None + return + + + self._handle_process_exit(code) + + def _handle_process_exit(self, code): + run_duration = time.time() - self.started_at if self.started_at else 0 + log_error = False + if self.log_pump is not None: + self.log_pump.join(timeout=5) + if self.log_pump.error: + log_error = True + self.last_action_error = f'bounded source log writer failed: {self.log_pump.error}' + if code == 0: + code = -1 + self.log_pump = None + if not log_error: + append_bounded_log_record( + self.log_path, + max(1, self.log_max_mb) * 1024 * 1024, + self.log_keep, + f'=== supervisor exit {now_iso()} source={self.source} code={code} ===\n', + ) + + self.last_exit_code = code + self.last_exit_at = time.time() + self.process = None + self.payload_identity = None + self.started_at = None + if self.stop_requested or self.manual_stop or self.paused: + self.status = 'paused' if self.paused else 'stopped' + return + if code == SOURCE_INFRASTRUCTURE_HOLD_EXIT: + self.status = 'failed' + self.desired_state = 'stopped' + self.manual_stop = True + self.last_action_error = ( + 'source entered infrastructure hold after unresolved durable handoff cleanup' + ) + return + + # A failed periodic one-shot keeps its normal cadence when crash restarts are disabled. + should_repeat = self.once and self.repeat and (code == 0 or not self.restart) + should_restart = self.restart and (code != 0 or not self.once) + if should_repeat or should_restart: + scheduled_repeat = should_repeat + if scheduled_repeat or run_duration >= self.restart_reset_after: + self.restart_streak = 0 + delay = self.interval if scheduled_repeat else min(self.restart_delay * max(1, 2 ** min(self.restart_streak, 6)), self.max_restart_delay) + self.restarts += 1 + if not scheduled_repeat: + self.restart_streak += 1 + self.next_start_at = time.time() + max(0, delay) + self.status = 'waiting' + else: + self.status = 'done' if code == 0 else 'failed' + self.desired_state = 'stopped' + self.manual_stop = True + + def stop(self, timeout=15, final=False, paused=False, preserve_desired=False, dependency_block=False): + self.last_action_error = '' + preserve_wait_until = ( + self.next_start_at + if dependency_block and self.status == 'waiting' and not self.is_running() + else 0 + ) + if not preserve_desired: + self.desired_state = 'paused' if paused else 'stopped' + self.stop_requested = bool(final) + self.manual_stop = self.desired_state != 'running' + self.paused = self.desired_state == 'paused' + if dependency_block and self.desired_state == 'running': + self.runtime_blocked = True + self._resume_immediately = self._resume_immediately or self.is_running() or self.status != 'waiting' + self.next_start_at = preserve_wait_until + if not self.process or self.process.poll() is not None: + self.process = None + self.payload_identity = None + self.startup_cleanup_pending = False + if self.log_pump is not None: + self.log_pump.join(timeout=5) + if self.log_pump.error: + self.status = 'failed' + self.last_action_error = f'bounded source log writer failed: {self.log_pump.error}' + self.log_pump = None + return False + self.log_pump = None + if dependency_block and self.desired_state == 'running': + self.status = 'blocked' + else: + self.status = 'paused' if self.desired_state == 'paused' else 'stopped' + return True + self.status = 'stopping' + try: + self.process.terminate() + try: + self.process.wait(timeout=timeout) + except subprocess.TimeoutExpired: + self.process.kill() + self.process.wait(timeout=5) + except Exception as exc: + self.status = 'failed' + self.last_action_error = f'process stop failed: {exc}' + return False + if self.process.poll() is None: + self.status = 'failed' + self.last_action_error = 'process remained live after bounded stop' + return False + if self.startup_cleanup_pending: + self.process = None + self.payload_identity = None + self.startup_cleanup_pending = False + if self.log_pump is not None: + self.log_pump.join(timeout=5) + if self.log_pump.error: + self.status = 'failed' + self.last_action_error = f'bounded source log writer failed: {self.log_pump.error}' + self.log_pump = None + return False + self.log_pump = None + self.payload_identity = None + if not dependency_block: + try: + self.finalize_database_runs('stopped', 'supervisor controlled source stop') + except Exception as exc: + self.status = 'failed' + self.last_action_error = f'database run finalization failed: {exc}' + return False + append_bounded_log_record( + self.log_path, + max(1, self.log_max_mb) * 1024 * 1024, + self.log_keep, + f'=== supervisor stopped {now_iso()} source={self.source} ===\n', + ) + if dependency_block and self.desired_state == 'running': + self.status = 'blocked' + else: + self.status = 'paused' if self.desired_state == 'paused' else 'stopped' + return True + + def pause(self): + return self.stop(paused=True) + + def resume(self): + return self.start(force=True) + + def restart_now(self): + if not self.stop(timeout=10): + if not self.last_action_error: + self.last_action_error = 'current process did not stop' + return False + return self.start(force=True) + + def dependency_unavailable(self): + if self.process is not None: + code = self.process.poll() + if code is None: + return self.stop(timeout=15, preserve_desired=True, dependency_block=True) + self._handle_process_exit(code) + if self.desired_state != 'running': + return True + return self.stop(timeout=15, preserve_desired=True, dependency_block=True) + + def dependency_available(self): + if not self.runtime_blocked: + return True + self.runtime_blocked = False + if self.desired_state != 'running': + return True + force = self._resume_immediately + self.status = 'waiting' + return self.start(force=force, record_intent=False) + + def last_log_line(self): + lines = [line.strip() for line in bounded_tail_lines(self.log_path, 40, 64 * 1024) if line.strip()] + return lines[-1] if lines else '' + + def tail_log_lines(self, limit=40): + return bounded_tail_lines(self.log_path, limit, MAX_LOG_TAIL_BYTES) + + def state_data(self): + if not self.use_per_source_state or not os.path.exists(self.state_path): + return {} + try: + with open(self.state_path, 'r', encoding='utf-8') as f: + return json.load(f) + except (OSError, json.JSONDecodeError): + return {} + + def auth_summary(self): + state = self.state_data() + source_state = (state.get('sources') or {}).get(self.source) or {} + summary = source_state.get('auth_summary') or {} + auth_status = source_state.get('auth_status') or {} + if not summary and auth_status: + counts = auth_counts_from_status(auth_status) + summary = { + 'pool': '', + 'current': source_state.get('last_auth') or 'none', + 'total': len(auth_status), + **counts, + } + return summary + + def auth_label(self): + summary = self.auth_summary() + if not summary: + return '-' + current = str(summary.get('current') or 'none') + if current == 'none' and not summary.get('total'): + return '-' + return ( + f"{current} ok={int(summary.get('ok', 0) or 0)} " + f"dead={int(summary.get('dead', 0) or 0)} " + f"lim={int(summary.get('limited', 0) or 0)} " + f"rl={int(summary.get('rate_limit_errors', 0) or 0)}" + ) + + def structured_state(self): + pid = self.process.pid if self.process is not None and self.process.poll() is None else None + next_run = timestamp_iso(self.next_start_at) + auth = self.auth_summary() + safe_error = '' + if self.startup_cleanup_pending: + safe_error = 'cleanup_pending' + elif self.last_action_error: + safe_error = 'runtime_error' + elif self.last_exit_code not in (None, 0): + safe_error = 'child_exit' + return { + 'id': managed_source_id(self), + 'source': self.source, + 'role': managed_source_role(self), + 'lifecycle_state': self.status, + 'desired_state': self.desired_state, + 'process_state': 'running' if pid is not None else 'stopped', + 'pid': pid, + 'enabled': bool(self.enabled), + 'dependency_blocked': bool(self.runtime_blocked), + 'startup_cleanup_pending': bool(self.startup_cleanup_pending), + 'mode': self.mode_label(), + 'interval_seconds': max(0, int(self.interval or 0)), + 'restart_enabled': bool(self.restart), + 'restart_delay_seconds': max(0, int(self.restart_delay or 0)), + 'restart_count': max(0, int(self.restarts or 0)), + 'restart_streak': max(0, int(self.restart_streak or 0)), + 'last_exit_code': self.last_exit_code, + 'last_exit_at': timestamp_iso(self.last_exit_at), + 'next_scheduled_run_at': next_run, + 'safe_error_category': safe_error, + 'auth_summary': { + key: max(0, int(auth.get(key, 0) or 0)) + for key in ( + 'total', 'ok', 'dead', 'limited', 'rate_limit_errors', + 'auth_invalid_errors', + ) + } if auth else {}, + 'allowed_actions': list(managed_source_allowed_actions(self)), + } + + def row(self): + pid = self.process.pid if self.process and self.process.poll() is None else '-' + uptime = format_duration(time.time() - self.started_at) if self.started_at else '-' + wait_for = format_duration(self.next_start_at - time.time()) if self.status == 'waiting' else '-' + return [ + self.source, + self.status, + str(pid), + self.mode_label(), + uptime, + format_exit_code(self.last_exit_code), + wait_for, + f'{self.restart_streak}/{self.restarts}', + self.auth_label(), + self.log_path, + self.last_log_line()[:90], + self.desired_state, + ] + + +class ManagedDiscoveryProducer(ManagedSource): + role = DISCOVERY_PRODUCER_ROLE + + def __init__( + self, + source, + config_path, + project_dir, + results_dir, + supervisor_config, + source_config, + global_force_once=False, + dependency_gate=None, + authority_check=None, + start_gate=None, + child_environment=None, + ): + if source not in DISCOVERY_PRODUCER_SOURCES: + raise ValueError('discovery producer source is outside the canonical allowlist') + super().__init__( + source, config_path, project_dir, results_dir, supervisor_config, + source_config, True, dependency_gate, authority_check, start_gate, + child_environment, + ) + self.once = True + self.repeat = True + self._set_producer_log_path() + + @property + def producer_id(self): + return f'{DISCOVERY_PRODUCER_ROLE}:{self.source}' + + def _set_producer_log_path(self): + self.log_path = os.path.join( + os.path.dirname(self.log_path), + f'{DISCOVERY_PRODUCER_ROLE}-{safe_source_filename(self.source)}.log', + ) + + def reconfigure(self, results_dir, supervisor_config, source_config, global_force_once=False): + super().reconfigure(results_dir, supervisor_config, source_config, True) + self.once = True + self.repeat = True + self._set_producer_log_path() + + def build_command(self): + return child_bootstrap_command(DISCOVERY_PRODUCER_ROLE, [ + '--config', self.config_path, + '--source', self.source, + '--once', + ]) + + def build_env(self): + env = os.environ.copy() + use_system_proxy = bool_value(self.options.get('use_system_proxy'), False) + if not use_system_proxy: + for key in PROXY_ENV_KEYS: + env.pop(key, None) + env['NO_PROXY'] = '*' + env['no_proxy'] = '*' + else: + env.pop('NO_PROXY', None) + env.pop('no_proxy', None) + env['PYTHONUNBUFFERED'] = '1' + env['PYTHONIOENCODING'] = 'utf-8' + if self.use_per_source_state: + env['RUNNER_STATE_FILE'] = self.state_path + for key, value in self.env_overrides.items(): + env[str(key)] = str(value) + strip_supervisor_credentials(env) + env.pop('SCANNER_SKIP_STARTUP_CLEANUP', None) + database_url = self.dependency_gate.database_url if self.dependency_gate else '' + if not database_url: + raise RuntimeError('discovery producer requires managed PostgreSQL authority') + env['SCANNER_DB_URL'] = database_url + env['DATABASE_URL'] = database_url + env['TRUF_MANAGED_POSTGRES_DSN'] = database_url + env['TRUF_DISCOVERY_SOURCE'] = self.source + env.update(self.child_environment) + return env + + def structured_state(self): + state = self.state_data() + source_state = (state.get('sources') or {}).get(self.source) or {} + raw_result = source_state.get('last_cycle_result') or {} + raw_status = raw_result.get('status') or source_state.get('last_status') or '' + status = raw_status if type(raw_status) is str and raw_status in DISCOVERY_CYCLE_STATUSES else 'unknown' + result = { + 'status': status, + } + for key in ('fetched_count', 'queued_new_count', 'queued_updated_count'): + try: + result[key] = max(0, int(raw_result.get(key, 0) or 0)) + except (TypeError, ValueError): + result[key] = 0 + raw_error = source_state.get('last_error_category') or '' + if type(raw_error) is str and raw_error in DISCOVERY_ERROR_CATEGORIES: + safe_error = raw_error + else: + safe_error = 'runtime_error' if raw_error else '' + if not safe_error and self.last_action_error: + safe_error = 'runtime_error' + elif not safe_error and self.last_exit_code not in (None, 0): + safe_error = 'child_exit' + structured = super().structured_state() + structured.update({ + 'last_cycle_result': result, + 'last_successful_discovery_at': safe_state_timestamp( + source_state.get('last_discovery_success_at') + ), + 'safe_error_category': safe_error, + }) + return structured + + +class ManagedKeychecks(ManagedSource): + def __init__( + self, + config_path, + project_dir, + results_dir, + supervisor_config, + keychecks_config, + global_force_once=False, + dependency_gate=None, + authority_check=None, + start_gate=None, + child_environment=None, + ): + self.keychecks_config = dict(keychecks_config or {}) + self.command_override = None + self.recheck_restore = None + self.recheck_label = '' + super().__init__( + 'keychecks', config_path, project_dir, results_dir, supervisor_config, + {'enabled': self.keychecks_config.get('enabled', False)}, global_force_once, + dependency_gate, authority_check, start_gate, child_environment, + ) + self.once = True + self.repeat = bool_value(self.keychecks_config.get('repeat'), True) + self.restart = bool_value(self.keychecks_config.get('restart'), False) + self.enabled = bool_value(self.keychecks_config.get('enabled'), False) + self.interval = int(self.keychecks_config.get('interval', self.keychecks_config.get('cooldown', 3600)) or 3600) + self.status = 'disabled' if not self.enabled else 'stopped' + self.summary_path = self.keychecks_config.get('summary_tsv') or os.path.join( + (self.keychecks_config.get('keycheck_dir') or os.path.join(os.path.dirname(results_dir), 'keychecks')), + 'summary.tsv', + ) + self.state_path = self.summary_path + + def reconfigure(self, results_dir, supervisor_config, source_config, global_force_once=False): + self.keychecks_config = dict(source_config or {}) + super().reconfigure(results_dir, supervisor_config, {'enabled': self.keychecks_config.get('enabled', False)}, global_force_once) + self.once = True + self.repeat = bool_value(self.keychecks_config.get('repeat'), True) + self.restart = bool_value(self.keychecks_config.get('restart'), False) + self.enabled = bool_value(self.keychecks_config.get('enabled'), False) + self.interval = int(self.keychecks_config.get('interval', self.keychecks_config.get('cooldown', 3600)) or 3600) + self.summary_path = self.keychecks_config.get('summary_tsv') or os.path.join( + (self.keychecks_config.get('keycheck_dir') or os.path.join(os.path.dirname(results_dir), 'keychecks')), + 'summary.tsv', + ) + self.state_path = self.summary_path + if not self.enabled: + self.status = 'disabled' + + def build_keycheck_command(self, services=None, extra_args=None): + services = services if services is not None else self.keychecks_config.get('services', 'all') + if isinstance(services, (list, tuple)): + services = ','.join(str(item) for item in services) + arguments = [ + '--config', + self.config_path, + '--service', + str(services or 'all'), + ] + for key, flag in ( + ('input', '--input'), + ('proxy_file', '--proxy-file'), + ('summary_tsv', '--summary-tsv'), + ('summary_json', '--summary-json'), + ('alive_summary_tsv', '--alive-summary-tsv'), + ): + value = self.keychecks_config.get(key) + if value: + arguments.extend([flag, str(value)]) + max_keys = int(self.keychecks_config.get('max_keys', 0) or 0) + if max_keys: + arguments.extend(['--max-keys', str(max_keys)]) + for key, flag in ( + ('retry_network', '--retry-network'), + ('retry_limited', '--retry-limited'), + ('retry_unknown', '--retry-unknown'), + ('retry_restricted', '--retry-restricted'), + ('retry_no_balance', '--retry-no-balance'), + ('retry_valid', '--retry-valid'), + ('recheck_all', '--recheck-all'), + ('summary_only', '--summary-only'), + ('no_summary', '--no-summary'), + ): + if bool_value(self.keychecks_config.get(key), False): + arguments.append(flag) + arguments.extend(list_value(self.keychecks_config.get('extra_args'))) + arguments.extend(extra_args or []) + return child_bootstrap_command('keycheck', arguments) + + def build_command(self): + if self.command_override: + return list(self.command_override) + return self.build_keycheck_command() + + def build_env(self): + env = os.environ.copy() + env['PYTHONUNBUFFERED'] = '1' + for key, value in (self.keychecks_config.get('env') or {}).items() if isinstance(self.keychecks_config.get('env'), dict) else []: + env[str(key)] = str(value) + if self.dependency_gate is not None: + self.dependency_gate.force_database_environment(env) + env.update(self.child_environment) + return env + + def mode_label(self): + if self.command_override: + return 'recheck' + return 'hourly' if self.repeat else 'once' + + def start_recheck(self, service, runner_args, force=False): + if self.is_running(): + if not force: + return False, 'keychecks is already running; use `recheck ... --force` to stop it and start the requested recheck.' + if not self.stop(timeout=10): + return False, 'keychecks recheck refused because the current process did not stop.' + self.recheck_restore = { + 'repeat': self.repeat, + 'restart': self.restart, + 'interval': self.interval, + } + self.command_override = self.build_keycheck_command(service, runner_args) + self.recheck_label = f"{service} {' '.join(runner_args)}".strip() + self.once = True + self.repeat = False + self.restart = False + self.restarts = 0 + started = self.start(force=True) + if started: + return True, f'keychecks: recheck started: {self.recheck_label}' + if self.runtime_blocked: + return True, f'keychecks: recheck intent recorded; blocked until PostgreSQL is stably ready: {self.recheck_label}' + detail = self.last_action_error or 'owned keycheck process did not start' + self.restore_recheck_mode(schedule_next=False) + return False, f'keychecks: recheck failed: {detail}' + + def restore_recheck_mode(self, schedule_next=False): + if not self.recheck_restore: + self.command_override = None + self.recheck_label = '' + return + restore = self.recheck_restore + self.command_override = None + self.recheck_restore = None + self.recheck_label = '' + self.once = True + self.repeat = bool_value(restore.get('repeat'), True) + self.restart = bool_value(restore.get('restart'), False) + self.interval = int(restore.get('interval', self.interval) or self.interval or 3600) + if schedule_next and self.repeat and self.enabled and not self.manual_stop and not self.paused: + self.next_start_at = time.time() + max(0, self.interval) + self.status = 'waiting' + + def poll(self): + recheck_was_active = bool(self.command_override) + desired_before_poll = self.desired_state + super().poll() + if recheck_was_active and not self.is_running() and not self.runtime_blocked: + if desired_before_poll == 'running': + self.desired_state = 'running' + self.manual_stop = False + self.restore_recheck_mode(schedule_next=self.status in ('done', 'failed')) + + def stop(self, timeout=15, final=False, paused=False, preserve_desired=False, dependency_block=False): + stopped = super().stop( + timeout=timeout, + final=final, + paused=paused, + preserve_desired=preserve_desired, + dependency_block=dependency_block, + ) + if self.command_override and not preserve_desired: + self.restore_recheck_mode(schedule_next=False) + return stopped + + +class ManagedPipelineWorker(ManagedSource): + CHILD_KINDS = { + 'result-ingester': 'result-ingester', + 'jsonl-projector': 'jsonl-projector', + 'janitor': 'janitor', + 'worker-api': 'worker-api', + } + + def __init__( + self, worker_name, config_path, project_dir, results_dir, + supervisor_config, worker_config, dependency_gate=None, + authority_check=None, start_gate=None, child_environment=None, + ): + default_enabled = worker_name != 'worker-api' + options = { + 'enabled': bool_value((worker_config or {}).get('enabled'), default_enabled), + 'restart': True, + 'once': False, + 'repeat': False, + 'interval': 0, + 'env': (worker_config or {}).get('env') or {}, + } + super().__init__( + worker_name, config_path, project_dir, results_dir, + supervisor_config, options, False, dependency_gate, + authority_check, start_gate, child_environment, + ) + self.worker_config = dict(worker_config or {}) + self.enabled = bool_value(self.worker_config.get('enabled'), default_enabled) + self.status = 'disabled' if not self.enabled else 'stopped' + self.once = False + self.repeat = False + self.restart = True + self.use_per_source_state = False + + def build_command(self): + return child_bootstrap_command( + self.CHILD_KINDS[self.source], ['--config', self.config_path], + ) + + def build_env(self): + env = os.environ.copy() + env['PYTHONUNBUFFERED'] = '1' + env['PYTHONIOENCODING'] = 'utf-8' + if self.source == 'janitor': + for key in list(env): + if key.upper().startswith('PG') or key.upper() in { + 'TRUF_MANAGED_POSTGRES_DSN', 'SCANNER_DB_URL', 'DATABASE_URL', + 'SCANNER_DASHBOARD_DB_URL', 'KEYCHECK_DB_URL', + }: + env.pop(key, None) + elif self.dependency_gate is not None: + self.dependency_gate.force_database_environment(env) + env.update(self.child_environment) + if self.source == 'janitor': + env.pop('TRUF_MANAGED_POSTGRES_DSN', None) + env.pop('SCANNER_DB_URL', None) + env.pop('DATABASE_URL', None) + return env + + def finalize_database_runs(self, status, reason): + return 0, 0 + + def state_data(self): + return {} + + def mode_label(self): + return 'singleton' + + +class ManagedDockerShadow(ManagedSource): + def __init__( + self, config_path, project_dir, results_dir, supervisor_config, + shadow_config, dependency_gate=None, authority_check=None, + start_gate=None, child_environment=None, + ): + options = {'enabled': bool_value((shadow_config or {}).get('enabled'), False)} + super().__init__( + 'docker-shadow', config_path, project_dir, results_dir, + supervisor_config, options, False, dependency_gate, + authority_check, start_gate, child_environment, + ) + self.enabled = bool_value((shadow_config or {}).get('enabled'), False) + self.status = 'disabled' if not self.enabled else 'stopped' + self.manual_only = True + self.once = True + self.repeat = False + self.restart = False + self.interval = 0 + self.use_per_source_state = False + + def build_command(self): + return child_bootstrap_command( + 'docker-shadow', ['--config', self.config_path], + ) + + def start(self, force=False, record_intent=True): + if self.dependency_gate is not None and not self.dependency_gate.ready: + self.last_action_error = 'managed PostgreSQL dependency is unavailable' + self.desired_state = 'stopped' + self.manual_stop = True + self.runtime_blocked = False + self.status = 'stopped' + return False + return super().start(force=force, record_intent=record_intent) + + def dependency_unavailable(self): + self.desired_state = 'stopped' + self.manual_stop = True + self.runtime_blocked = False + return self.stop(timeout=15) + + def dependency_available(self): + self.runtime_blocked = False + return True + + def finalize_database_runs(self, status, reason): + return 0, 0 + + def state_data(self): + return {} + + def mode_label(self): + return 'manual-once' + + +def build_table_lines(managed_sources, max_width=None): + max_width = max_width or terminal_width() + headers = ['source', 'status', 'desired', 'pid', 'mode', 'up', 'exit', 'next', 'rs', 'auth', 'last_log'] + raw_rows = [source.row() for source in managed_sources] + rows = [ + [row[0], row[1], row[11], row[2], row[3], row[4], row[5], row[6], row[7], row[8], row[10]] + for row in raw_rows + ] + + fixed_widths = [] + for index, header in enumerate(headers[:-1]): + values = [row[index] for row in rows] + cap = 36 if header == 'auth' else 20 if header == 'source' else 12 if header in ('mode', 'desired') else 10 if header == 'exit' else 9 + fixed_widths.append(min(max([len(header)] + [len(str(value)) for value in values]) if values else len(header), cap)) + separator_width = 3 * (len(headers) - 1) + last_width = max(20, max_width - sum(fixed_widths) - separator_width) + widths = fixed_widths + [last_width] + + def format_row(values): + return ' | '.join(truncate_text(values[index], widths[index]).ljust(widths[index]) for index in range(len(headers))) + + lines = [ + 'Supervisor status ' + now_iso(), + format_row(headers), + '-+-'.join('-' * width for width in widths), + ] + for row in rows: + lines.append(format_row(row)) + lines.append('') + lines.append('Type `help` for commands. Use `command ` for full log/state paths.') + return lines + + +def scan_worker_snapshot(config): + global_config = (config or {}).get('global') or {} + base_limit = max(0, int_value(global_config.get('max_active_scans'), 0)) + bonus_limit = max(0, min(1, int_value(global_config.get('opportunistic_scan_slots'), 0))) + limit = base_limit + bonus_limit + snapshot = { + 'active': 0, 'limit': limit, 'base_active': 0, 'base_limit': base_limit, + 'bonus_active': 0, 'bonus_limit': bonus_limit, + 'trufflehog': 0, 'sources': {}, 'detail': '', + } + if limit <= 0: + return snapshot + db_path = str(global_config.get('scan_limiter_db') or '') + if not db_path: + state_dir = str(global_config.get('state_dir') or '') + if state_dir: + db_path = os.path.join(state_dir, 'scan_limiter.db') + if not db_path or not os.path.isfile(db_path): + return snapshot + connection = None + try: + reject_reparse_components(db_path) + normalized = os.path.abspath(db_path).replace('\\', '/') + uri = f'file:{quote(normalized, safe="/:")}?mode=ro' + connection = sqlite3.connect(uri, uri=True, timeout=0.2) + connection.execute('PRAGMA query_only=ON') + connection.execute('PRAGMA busy_timeout=200') + columns = { + row[1] for row in connection.execute('PRAGMA table_info(scan_slots)').fetchall() + } + slot_kind = "COALESCE(slot_kind, 'base')" if 'slot_kind' in columns else "'base'" + rows = connection.execute( + f"""SELECT COALESCE(NULLIF(owner_source, ''), 'unknown'), child_executable, + {slot_kind} + FROM scan_slots ORDER BY 1""" + ).fetchall() + sources = {} + for source, child_executable, kind in rows: + source = str(source) + sources[source] = sources.get(source, 0) + 1 + if kind == 'bonus': + snapshot['bonus_active'] += 1 + else: + snapshot['base_active'] += 1 + if os.path.basename(str(child_executable or '')).lower() == 'trufflehog.exe': + snapshot['trufflehog'] += 1 + snapshot['sources'] = sources + snapshot['active'] = len(rows) + except (OSError, ValueError, sqlite3.Error) as exc: + snapshot['detail'] = str(exc)[:200] + finally: + if connection is not None: + connection.close() + return snapshot + + +def current_scan_worker_snapshot(context, refresh_sec=2): + context = context or {} + now = time.monotonic() + cached = context.get('_scan_worker_snapshot') + if cached is not None and now < float(context.get('_scan_worker_snapshot_refresh_at') or 0): + return cached + snapshot = scan_worker_snapshot(context.get('config')) + context['_scan_worker_snapshot'] = snapshot + context['_scan_worker_snapshot_refresh_at'] = now + max(0.1, float(refresh_sec)) + return snapshot + + +def scan_worker_status_line(context): + snapshot = current_scan_worker_snapshot(context) + if snapshot['limit'] <= 0: + return '' + staging = max(0, snapshot['active'] - snapshot['trufflehog']) + line = ( + f"Scan workers: active={snapshot['active']}/{snapshot['limit']} " + f"base={snapshot['base_active']}/{snapshot['base_limit']} " + f"bonus={snapshot['bonus_active']}/{snapshot['bonus_limit']} " + f"trufflehog={snapshot['trufflehog']} staging={staging}" + ) + if snapshot['sources']: + sources = ', '.join(f'{source}:{count}' for source, count in snapshot['sources'].items()) + line += ' sources=' + truncate_text(sources, 160) + if snapshot['detail']: + line += ' detail=' + snapshot['detail'] + return line + + +def pipeline_status_snapshot(context, refresh_sec=None, allow_refresh=True): + context = context or {} + if context.get('_defer_pipeline_status_until_next_tick'): + cached = context.get('_pipeline_status_snapshot') + if cached is not None: + return cached + return { + 'ingester_ready': False, 'projector_ready': False, + 'cutover_ready': False, + 'ingester_state': 'starting', 'projector_state': 'starting', + 'bundle_items': 0, 'bundle_bytes': 0, 'projection_items': 0, + 'projection_bytes': 0, 'keycheck_items': 0, 'keycheck_bytes': 0, + 'quarantine_items': 0, 'quarantine_bytes': 0, + 'detail': 'Pipeline workers are starting', + } + if refresh_sec is None: + refresh_sec = (context.get('supervisor_config') or {}).get( + 'pipeline_status_refresh_sec', 2, + ) or 2 + now = time.monotonic() + cached = context.get('_pipeline_status_snapshot') + if cached is not None and now < float(context.get('_pipeline_status_refresh_at') or 0): + return cached + if not allow_refresh: + if cached is not None: + return cached + return { + 'ingester_ready': False, 'projector_ready': False, + 'cutover_ready': False, + 'ingester_state': 'starting', 'projector_state': 'starting', + 'bundle_items': 0, 'bundle_bytes': 0, 'projection_items': 0, + 'projection_bytes': 0, 'keycheck_items': 0, 'keycheck_bytes': 0, + 'quarantine_items': 0, 'quarantine_bytes': 0, + 'detail': 'Pipeline status refresh is pending', + } + snapshot = { + 'ingester_ready': False, 'projector_ready': False, + 'cutover_ready': False, + 'ingester_state': 'missing', 'projector_state': 'missing', + 'bundle_items': 0, 'bundle_bytes': 0, 'projection_items': 0, + 'projection_bytes': 0, 'keycheck_items': 0, 'keycheck_bytes': 0, + 'quarantine_items': 0, 'quarantine_bytes': 0, 'detail': '', + } + gate = context.get('dependency_gate') + database_url = getattr(gate, 'database_url', '') if gate is not None else '' + if not database_url or (gate is not None and not gate.ready): + snapshot['detail'] = 'PostgreSQL is not stably ready' + else: + connection = None + try: + connection = connect_postgres( + database_url, connect_timeout_sec=2, statement_timeout_ms=3000, + lock_timeout_ms=1000, idle_in_transaction_timeout_ms=3000, + tcp_user_timeout_ms=3000, + ) + leases = connection.execute( + '''SELECT worker_name, state, lease_expires_at + FROM pipeline_leases + WHERE worker_name IN ('result_ingester','jsonl_projector')''' + ).fetchall() + capacity = connection.execute( + 'SELECT * FROM pipeline_capacity WHERE id = 1' + ).fetchone() + cutover = connection.execute( + 'SELECT marker, evidence_sha256 FROM runtime_final_cutover WHERE id = 1' + ).fetchone() + connection.commit() + lease_map = {row['worker_name']: row for row in leases} + current = datetime.now(timezone.utc).isoformat(timespec='seconds') + ingester = lease_map.get('result_ingester') + projector = lease_map.get('jsonl_projector') + snapshot['ingester_state'] = str(ingester['state']) if ingester else 'missing' + snapshot['projector_state'] = str(projector['state']) if projector else 'missing' + snapshot['cutover_ready'] = bool( + cutover + and cutover['marker'] == 'postgres-normalized-v2-authority' + and re.fullmatch(r'[a-f0-9]{64}', str(cutover['evidence_sha256'] or '')) + ) + snapshot['ingester_ready'] = bool( + ingester and ingester['state'] == 'ready' + and str(ingester['lease_expires_at'] or '') > current + and snapshot['cutover_ready'] + ) + snapshot['projector_ready'] = bool( + projector and projector['state'] == 'ready' + and str(projector['lease_expires_at'] or '') > current + and snapshot['cutover_ready'] + ) + if not snapshot['cutover_ready']: + snapshot['detail'] = 'Final PostgreSQL v2 cutover marker is absent or invalid' + if capacity: + for key in ( + 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'keycheck_items', 'keycheck_bytes', 'quarantine_items', 'quarantine_bytes', + ): + snapshot[key] = int(capacity[key] or 0) + except Exception as exc: + snapshot['detail'] = f'{type(exc).__name__}: {exc}'[:200] + if connection is not None: + try: + connection.rollback() + except Exception: + pass + finally: + if connection is not None: + connection.close() + context['_pipeline_status_snapshot'] = snapshot + context['_pipeline_status_refresh_at'] = now + max(0.2, float(refresh_sec)) + return snapshot + + +def pipeline_status_line(context): + snapshot = pipeline_status_snapshot(context) + line = ( + f"Pipeline: ingester={snapshot['ingester_state']} projector={snapshot['projector_state']} " + f"bundles={snapshot['bundle_items']}/{snapshot['bundle_bytes']}B " + f"projection={snapshot['projection_items']}/{snapshot['projection_bytes']}B " + f"candidates={snapshot['keycheck_items']}/{snapshot['keycheck_bytes']}B " + f"quarantine={snapshot['quarantine_items']}/{snapshot['quarantine_bytes']}B" + ) + if snapshot['detail']: + line += ' detail=' + snapshot['detail'] + return line + + +def build_runtime_table_lines(managed_sources, context=None, max_width=None): + lines = [] + scan_line = scan_worker_status_line(context) + if scan_line: + lines.extend((scan_line, '')) + lines.extend((pipeline_status_line(context), '')) + lines.extend(build_table_lines(managed_sources, max_width=max_width)) + return lines + + +def table_signature(managed_sources): + signature = [] + for source in managed_sources: + process = source.process + pid = process.pid if process is not None and process.poll() is None else None + signature.append(( + source.source, + source.status, + source.desired_state, + pid, + source.last_exit_code, + source.restarts, + source.restart_streak, + source.runtime_blocked, + source.last_action_error, + )) + return tuple(signature) + + +def print_table(managed_sources, clear=True, context=None): + if clear: + os.system('cls' if os.name == 'nt' else 'clear') + for line in build_runtime_table_lines(managed_sources, context=context): + print(line) + + +HELP_TEXT = r''' +Commands: + help | h | ? + Show this help. + + status | s + Print a fresh status table once. + + auth + Print auth pool health from runner state. + Example: auth github + + watch | w + Open live status view in an alternate screen. Press q to return to prompt. + This keeps scrollback clean and avoids table spam. + + reload | r + Live reload is disabled by config authority binding. Use coordinated shutdown and restart. + + recheck [types...] [options] + Start a one-shot keycheck_runner recheck using configured keycheck probes/service_args. + If no type is given, defaults to --recheck-all. + Types: network, ratelimited/aliveratelimited, unknown, restricted, nobalance, valid, legacy-vertex, all. + Options: --force, --max-keys N, --input PATH, --proxy-file PATH, --no-resource-probe. + Examples: recheck all network + recheck gemini --network --ratelimited + recheck replicate --valid --max-keys 5 + + start + Start a stopped source using its current mode from the table. + Example: start pypi + + stop + Stop the child process and keep it stopped. Supervisor will not auto-restart it. + Example: stop dockerhub + + restart + Stop then start immediately. + Example: restart github + + pause + Stop and mark as paused. Same process behavior as stop, but visually distinct. + Example: pause npm + + resume + Unpause and start immediately. + Example: resume npm + + once + Switch source to once mode and start it with --once. It will not repeat unless mode is changed. + Example: once pypi + + mode loop|once|repeat + loop = no --once; child console_runner loops internally using config cooldown. + once = pass --once; one source cycle, then stop when child exits. + repeat = pass --once; supervisor repeats one-shot cycles after interval. + Example: mode pypi repeat + + set interval + Set repeat interval for once+repeat mode. + Example: set pypi interval 120 + + set restart on|off + Enable/disable restart after crash or loop child exit. + Example: set dockerhub restart off + + set restart_delay + Set initial failure restart delay. + Example: set github restart_delay 60 + + logs [lines] + Print the last N log lines from the configured runtime logs directory. Default: 40. + Example: logs pypi 80 + + command + Print child command, log file, and state file. + Example: command github + + dashboard status|stop|start|restart + Manage the dashboard process without stopping supervisor or source processes. + Example: dashboard stop + + shutdown + Request coordinated shutdown of sources, keychecks, dashboard, and identity-verified PostgreSQL. + Authenticated shutdown remains available after config hash drift; other mutations are rejected. + + quit | exit | q + In foreground supervisor: stop all children and exit. + In --attach: detach only; background supervisor keeps running. + +Statuses: + stopped Not running. This is the default state. + running Child console_runner.py is currently running. + waiting Supervisor will start/restart after the `next` countdown. + blocked Desired state is running, but stable PostgreSQL readiness is unavailable. + paused Explicitly paused by command. + done One-shot child exited successfully and will not repeat. + failed Child exited with non-zero code and restart is disabled. + +The `rs` column is current failure streak / total automatic restarts. + +Modes: + loop Child runs without --once and handles its own loop/cooldown. + once Child runs with --once and stops after one source cycle. + once+repeat Child runs with --once; supervisor restarts it after `interval` seconds. +'''.strip() + + +def timestamp_iso(value): + if not value: + return None + try: + return datetime.fromtimestamp(float(value), timezone.utc).isoformat(timespec='seconds') + except (OSError, OverflowError, TypeError, ValueError): + return None + + +def managed_source_id(source): + if isinstance(source, ManagedDiscoveryProducer): + return source.producer_id + return str(source.source) + + +def managed_source_role(source): + if isinstance(source, ManagedDiscoveryProducer): + return DISCOVERY_PRODUCER_ROLE + if isinstance(source, ManagedKeychecks): + return 'keycheck' + if isinstance(source, ManagedPipelineWorker): + return source.CHILD_KINDS.get(source.source, 'pipeline-worker') + if isinstance(source, ManagedDockerShadow): + return 'docker-shadow' + return 'scanner' + + +def managed_source_allowed_actions(source): + if getattr(source, 'manual_only', False): + return ('start', 'stop') + actions = list(MANAGED_SOURCE_LIFECYCLE_ACTIONS) + if isinstance(source, ManagedDiscoveryProducer): + actions.append('set-interval') + elif isinstance(source, ManagedPipelineWorker): + actions.extend(('set-restart', 'set-restart-delay')) + else: + actions.extend(MANAGED_SOURCE_SETTING_ACTIONS) + return tuple(actions) + + +def managed_source_registry(managed_sources): + registry = {} + for source in managed_sources: + source_id = managed_source_id(source) + if not source_id or source_id in registry: + raise ValueError('managed source IDs are not unique') + registry[source_id] = source + return registry + + +def source_map(managed_sources): + return {source.source: source for source in managed_sources} + + +def select_sources(managed_sources, selector): + selector = normalize_source_name(selector) + if selector == 'all': + return managed_sources + sources = source_map(managed_sources) + item = sources.get(selector) + if not item: + print(f'Unknown source: {selector}. Available: {", ".join(sorted(sources))}') + return [] + return [item] + + +def autostart_sources(managed_sources): + return [source for source in managed_sources if not getattr(source, 'manual_only', False)] + + +def set_mode(source, mode): + mode = str(mode or '').lower() + if mode == 'loop': + source.once = False + source.repeat = True + elif mode == 'once': + source.once = True + source.repeat = False + elif mode == 'repeat': + source.once = True + source.repeat = True + else: + print(f'Unknown mode: {mode}. Use loop, once, or repeat.') + return False + return True + + +def print_source_command(source): + print(f'[{source.source}]') + print('command:', ' '.join(command_for_log(source.build_command()))) + print('mode:', source.mode_label()) + print('log:', source.log_path) + print('state:', source.state_path if source.use_per_source_state else '(config default)') + print('desired:', source.desired_state, 'dependency_blocked:', source.runtime_blocked) + print('restart:', source.restart, 'interval:', source.interval, 'restart_delay:', source.restart_delay) + + +def print_auth_status(source): + state = source.state_data() + source_state = (state.get('sources') or {}).get(source.source) or {} + summary = source_state.get('auth_summary') or source.auth_summary() + auth_status = source_state.get('auth_status') or {} + if not summary and not auth_status: + print(f'[{source.source}] auth: no auth pool state') + print('state:', source.state_path if source.use_per_source_state else '(config default)') + return + print( + f"[{source.source}] pool={summary.get('pool') or '-'} current={summary.get('current') or '-'} " + f"total={int(summary.get('total', len(auth_status)) or 0)} " + f"ok={int(summary.get('ok', 0) or 0)} dead={int(summary.get('dead', 0) or 0)} " + f"lim={int(summary.get('limited', 0) or 0)} rl={int(summary.get('rate_limit_errors', 0) or 0)} " + f"auth_invalid={int(summary.get('auth_invalid_errors', 0) or 0)}" + ) + print('state:', source.state_path if source.use_per_source_state else '(config default)') + if not auth_status: + return + print('name\tstatus\tdisabled_until\treason\trl\tauth_invalid\tfailures\tlast_error') + for name in sorted(auth_status): + item = auth_status.get(name) or {} + status = item.get('status') or auth_item_kind(item) + rl = int(item.get('rate_limit_count', 0) or 0) + int(item.get('secondary_rate_limit_count', 0) or 0) + auth_invalid = int(item.get('auth_invalid_count', 0) or 0) + if item.get('disabled_reason') == 'auth_invalid' and not auth_invalid: + auth_invalid = int(item.get('failures', 1) or 1) + print('\t'.join([ + str(name), + str(status), + str(item.get('disabled_until') or '-'), + str(item.get('disabled_reason') or '-'), + str(rl), + str(auth_invalid), + str(int(item.get('failures', 0) or 0)), + str(item.get('last_error') or '').replace('\t', ' ')[:180], + ])) + + +def runtime_authority( + config_path, + supervisor_path=None, + trufflehog_path=None, + expected_config_sha256=None, + expected_supervisor_sha256=None, + expected_code_manifest_sha256=None, + code_manifest=None, + policy_paths=None, + include_trufflehog=True, +): + supervisor_path = os.path.abspath(supervisor_path or __file__) + config_sha256 = sha256_file(config_path) + if expected_config_sha256 and config_sha256 != expected_config_sha256: + raise RuntimeError('supervisor config changed after it was loaded') + supervisor_sha256 = sha256_file(supervisor_path) + if expected_supervisor_sha256 and supervisor_sha256 != expected_supervisor_sha256: + raise RuntimeError('supervisor script changed after parent authority capture') + code_manifest = code_manifest or build_code_manifest( + trufflehog_path=trufflehog_path, + policy_paths=policy_paths, + include_trufflehog=include_trufflehog, + ) + manifest_sha256 = code_manifest_sha256(code_manifest) + if expected_code_manifest_sha256 and manifest_sha256 != expected_code_manifest_sha256: + raise RuntimeError('supervisor code manifest changed after parent authority capture') + verify_code_manifest(code_manifest, manifest_sha256) + return { + 'supervisor_path': supervisor_path, + 'supervisor_sha256': supervisor_sha256, + 'config_path': os.path.abspath(config_path), + 'config_sha256': config_sha256, + 'code_manifest': code_manifest, + 'code_manifest_sha256': manifest_sha256, + } + + +def configured_policy_paths(config): + global_config = (config or {}).get('global') or {} + values = [global_config.get('trufflehog_config')] + for source in ((config or {}).get('sources') or {}).values(): + if isinstance(source, dict) and source.get('trufflehog_config'): + values.append(resolve_optional_path(source['trufflehog_config'], global_config)) + return list(dict.fromkeys(value for value in values if value)) + + +def runtime_authority_error(context, require_private_acl=False): + authority = (context or {}).get('authority') + if not authority: + return '' + try: + verify_code_manifest( + authority['code_manifest'], + authority['code_manifest_sha256'], + require_private_acl=require_private_acl, + ) + if sha256_file(authority['supervisor_path']) != authority['supervisor_sha256']: + return 'supervisor script authority hash drifted' + if sha256_file(authority['config_path']) != authority['config_sha256']: + return 'supervisor config authority hash drifted' + except (OSError, ValueError, LifecycleAuthorityError) as exc: + return f'runtime code authority validation failed: {exc}' + return '' + + +def inhibit_for_authority_drift(context, detail): + context = context or {} + context['runtime_failed'] = True + if not context.get('authority_drift'): + context['authority_drift'] = str(detail) + context['start_gate_open'] = False + controller = context.get('postgres_controller') + if controller is not None: + controller.inhibit_lifecycle(detail) + gate = context.get('dependency_gate') + if gate is not None: + try: + gate.set_ready(False) + except DependencyStopError as exc: + context['fatal_child_stop_error'] = str(exc) + grace = max(0.0, float((context.get('supervisor_config') or {}).get('authority_drift_shutdown_grace_sec', 5) or 0)) + context['authority_drift_shutdown_at'] = time.monotonic() + grace + shutdown_event = context.get('shutdown_event') + if ( + shutdown_event is not None + and time.monotonic() >= float(context.get('authority_drift_shutdown_at') or 0) + ): + shutdown_event.set() + return False + + +def check_runtime_authority(context, trigger_shutdown=True, require_private_acl=False): + detail = runtime_authority_error(context, require_private_acl=require_private_acl) + if not detail: + return True + if trigger_shutdown: + return inhibit_for_authority_drift(context, detail) + return False + + +def lifecycle_phase(context): + return str((context or {}).get('lifecycle_phase') or (context or {}).get('activation_state') or PHASE_ACTIVE).upper() + + +def lifecycle_start_allowed(context): + context = context or {} + return ( + not context.get('shutdown_requested', False) + and lifecycle_phase(context) == PHASE_ACTIVE + and bool(context.get('start_gate_open', True)) + and not any( + getattr(source, 'startup_cleanup_pending', False) is True + for source in context.get('managed_sources', ()) + ) + ) + + +def shutdown_checkpoint(context): + """Consume shutdown flags and uncertain rollback under the control lock.""" + context = context or {} + if lifecycle_phase(context) not in (PHASE_STOPPING, PHASE_FAILED_HOLD): + pending = next(( + source for source in context.get('managed_sources', ()) + if getattr(source, 'startup_cleanup_pending', False) is True + ), None) + if pending is not None: + try: + begin_stopping(context) + finally: + enter_failed_hold(context, pending.last_action_error) + elif context.get('shutdown_requested'): + begin_stopping(context) + shutdown_event = context.get('shutdown_event') + return ( + bool(context.get('shutdown_requested')) + or lifecycle_phase(context) in (PHASE_STOPPING, PHASE_FAILED_HOLD) + or (shutdown_event is not None and shutdown_event.is_set()) + ) + + +def begin_stopping(context): + """Close every lifecycle start gate before a shutdown acknowledgement.""" + context = {} if context is None else context + phase = lifecycle_phase(context) + already_stopping = phase in (PHASE_STOPPING, PHASE_FAILED_HOLD) + context['lifecycle_phase'] = phase if already_stopping else PHASE_STOPPING + context['activation_state'] = context['lifecycle_phase'] + context['start_gate_open'] = False + if not already_stopping: + context['authority_release_safe'] = False + try: + shutdown_event = context.get('shutdown_event') + if shutdown_event is not None: + shutdown_event.set() + if already_stopping: + retry_event = context.get('shutdown_retry_event') + if retry_event is not None: + retry_event.set() + return + controller = context.get('postgres_controller') + if controller is not None and hasattr(controller, 'inhibit_lifecycle'): + controller.inhibit_lifecycle('supervisor lifecycle is STOPPING') + instance_file = context.get('instance_file') + instance_id = context.get('instance_id') + if instance_file and instance_id: + update_instance_activation(instance_file, instance_id, PHASE_STOPPING) + except BaseException: + context['runtime_failed'] = True + raise + + +def enter_failed_hold(context, detail='coordinated shutdown remains unsafe'): + """Keep process and OS authority live after any unconfirmed child stop.""" + context = {} if context is None else context + context['lifecycle_phase'] = PHASE_FAILED_HOLD + context['activation_state'] = PHASE_FAILED_HOLD + context['start_gate_open'] = False + context['authority_release_safe'] = False + context['runtime_failed'] = True + context['shutdown_failure'] = str(detail or 'coordinated shutdown remains unsafe') + instance_file = context.get('instance_file') + instance_id = context.get('instance_id') + if instance_file and instance_id: + update_instance_activation(instance_file, instance_id, PHASE_FAILED_HOLD) + + +def shutdown_authority_error(context, instance_id=None, token=None, control_address_value=None): + """Use private credentials/process identity even when on-disk code drifted.""" + context = context or {} + authority = context.get('authority') + instance_file = context.get('instance_file') + if not instance_file: + return '' + try: + metadata = load_instance_metadata(instance_file) + if not secrets.compare_digest(metadata['instance_id'], str(instance_id or '')): + return 'private shutdown metadata instance mismatch' + if not secrets.compare_digest(metadata['token'], str(token or '')): + return 'private shutdown metadata credential mismatch' + expected_control = metadata['control'] + actual_host, actual_port = control_address_value or ('', 0) + if str(expected_control['host']) != str(actual_host) or int(expected_control['port']) != int(actual_port): + return 'private shutdown metadata control endpoint mismatch' + retained = verify_instance_process( + metadata, + authority.get('supervisor_path') if authority else os.path.abspath(__file__), + authority.get('config_path') if authority else context.get('config_path'), + allow_config_drift=True, + allow_code_drift=True, + ) + retained.close() + except (OSError, ValueError) as exc: + return f'private shutdown authority verification failed: {exc}' + return '' + + +def command_failure(context, message): + if context is not None: + context['_command_failed'] = True + print(message) + + +def command_is_mutating(parts): + if not parts: + return False + action = parts[0].lower() + if action == 'dashboard': + return len(parts) < 2 or parts[1].lower() != 'status' + return action in { + 'reload', 'r', 'recheck', 'start', 'stop', 'restart', 'pause', 'resume', + 'once', 'mode', 'set', 'shutdown', + } + + +def handle_recheck_command(parts, managed_sources, context=None): + parsed, error = parse_recheck_command(parts) + if error: + command_failure(context, error) + return True + keychecks = source_map(managed_sources).get('keychecks') + if not keychecks: + command_failure(context, 'keychecks source is not managed. Enable keychecks.enabled or run supervisor with --sources keychecks/all.') + return True + if not isinstance(keychecks, ManagedKeychecks): + command_failure(context, 'keychecks source is not a ManagedKeychecks instance') + return True + ok, message = keychecks.start_recheck(parsed['service'], parsed['runner_args'], force=parsed['force']) + print(message) + if not ok: + if context is not None: + context['_command_failed'] = True + if ok: + print('command:', ' '.join(keychecks.build_command())) + print('log:', keychecks.log_path) + return True + + +class ManagedDashboard: + def __init__( + self, + config_path, + project_dir, + supervisor_config, + results_dir, + queue_dir, + dependency_gate=None, + force=False, + process_factory=OwnedProcess, + health_probe=None, + executor=None, + clock=None, + authority_check=None, + start_gate=None, + child_environment=None, + ): + self.source = 'dashboard' + enabled, self.host, self.port, self.url = dashboard_settings(supervisor_config) + self.enabled = bool(enabled or force) + self.config_path = os.path.abspath(config_path) + self.project_dir = project_dir + self.supervisor_config = supervisor_config + self.results_dir = results_dir + self.queue_dir = queue_dir + self.dependency_gate = dependency_gate + self.process_factory = process_factory + self.health_probe = health_probe or self._http_health_probe + self._executor = executor or ThreadPoolExecutor(max_workers=1, thread_name_prefix='dashboard-health') + self._owns_executor = executor is None + self._clock = clock or time.monotonic + self.authority_check = authority_check + self.start_gate = start_gate + self.child_environment = dict(child_environment or {}) + dashboard_config = supervisor_config.get('dashboard') if isinstance(supervisor_config.get('dashboard'), dict) else {} + self.startup_grace_sec = max(0.1, float(dashboard_config.get('startup_grace_sec', 30) or 30)) + self.health_interval_sec = max(0.1, float(dashboard_config.get('health_interval_sec', 2) or 2)) + self.health_timeout_sec = max(0.1, float(dashboard_config.get('health_timeout_sec', 1) or 1)) + self.restart_base_sec = max(0.1, float(dashboard_config.get('restart_base_sec', 2) or 2)) + self.restart_max_sec = max(self.restart_base_sec, float(dashboard_config.get('restart_max_sec', 60) or 60)) + stable_health = dashboard_config.get('stable_health_sec', 60) + self.stable_health_sec = max(0.0, float(60 if stable_health is None else stable_health)) + self.env_overrides = dashboard_config.get('env') if isinstance(dashboard_config.get('env'), dict) else {} + log_dir = resolve_path(config_path, supervisor_config.get('log_dir') or os.path.join(results_dir, 'logs')) + self.log_path = resolve_path(config_path, supervisor_config.get('dashboard_log') or os.path.join(log_dir, 'dashboard.log')) + self.process = None + self.log_pump = None + self.desired_state = 'running' if self.enabled else 'stopped' + self.status = ( + 'blocked' if self.enabled and dependency_gate and not dependency_gate.ready + else 'pending' if self.enabled else 'disabled' + ) + self.detail = 'waiting for dependency readiness' if self.enabled and dependency_gate and not dependency_gate.ready else '' + self.failures = 0 + self.started_at = None + self.healthy_since = None + self.next_health_at = 0.0 + self.next_start_at = 0.0 + self._future = None + self._future_generation = None + self._generation = 0 + self.fatal_stop_failure = False + if dependency_gate is not None: + dependency_gate.register(self) + + @property + def healthy(self): + return self.status == 'healthy' and self.process is not None and self.process.poll() is None + + def _http_health_probe(self, url, timeout): + request = urllib.request.Request(url.rstrip('/') + '/_stcore/health', method='GET') + opener = urllib.request.build_opener(urllib.request.ProxyHandler({})) + with opener.open(request, timeout=max(0.1, float(timeout))) as response: + return int(getattr(response, 'status', response.getcode())) == 200 + + def build_command(self): + return child_bootstrap_command('dashboard', [ + '--server.headless', + 'true', + '--server.address', + self.host, + '--server.port', + str(self.port), + '--', + '--config', + self.config_path, + '--results-dir', + self.results_dir, + '--queue-dir', + self.queue_dir, + ]) + + def build_env(self): + env = os.environ.copy() + env['PYTHONUNBUFFERED'] = '1' + env['PYTHONIOENCODING'] = 'utf-8' + for key, value in self.env_overrides.items(): + env[str(key)] = str(value) + if self.dependency_gate is not None: + self.dependency_gate.force_database_environment(env) + env.update(self.child_environment) + env['TRUF_DASHBOARD_CANONICAL_LAUNCH'] = '1' + env['TRUF_DASHBOARD_HOST'] = self.host + return env + + def _launch(self, now): + if self.start_gate is not None and not self.start_gate(): + self.detail = 'supervisor lifecycle start gate is closed' + self.status = 'stopped' + return False + if self.authority_check is not None and not self.authority_check(): + self.detail = 'runtime script/config authority drifted' + self.status = 'failed' + return False + require_private_directory(os.path.dirname(self.log_path), create=False) + self._generation += 1 + if self._future is not None: + self._future.cancel() + self._future = None + self._future_generation = None + log_max_bytes = max(1, int(self.supervisor_config.get('log_max_mb', 64) or 64)) * 1024 * 1024 + self.log_pump = BoundedRotatingLogPump( + self.log_path, + log_max_bytes, + self.supervisor_config.get('log_keep', 5), + ) + try: + command = self.build_command() + self.log_pump.write(f'\n=== dashboard start {now_iso()} ===\n') + self.log_pump.write(f'url: {self.url}\n') + self.log_pump.write('command: ' + ' '.join(command) + '\n') + creationflags = (subprocess.CREATE_NEW_PROCESS_GROUP | CREATE_NO_WINDOW) if os.name == 'nt' else 0 + self.process = self.process_factory( + command, + cwd=self.project_dir, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + stdin=subprocess.DEVNULL, + env=self.build_env(), + creationflags=creationflags, + ) + stream = getattr(self.process, 'stdout', None) + if stream is not None: + self.log_pump.attach(stream) + else: + self.log_pump.close() + except BaseException: + if self.process is not None and self.process.poll() is None: + self._stop_current() + elif self.log_pump is not None: + self.log_pump.close() + self.log_pump = None + raise + self.started_at = now + self.healthy_since = None + self.next_health_at = now + self.next_start_at = 0.0 + self.status = 'pending' + self.detail = 'process started; health has not been confirmed' + return True + + def start(self, force=False, record_intent=True): + if self.start_gate is not None and not self.start_gate(): + self.detail = 'supervisor lifecycle start gate is closed' + self.status = 'stopped' + return False + if record_intent: + self.desired_state = 'running' + if force: + self.enabled = True + if not self.enabled or self.desired_state != 'running': + self.status = 'disabled' if not self.enabled else 'stopped' + return False + if self.dependency_gate is not None and not self.dependency_gate.ready: + self.status = 'blocked' + self.detail = 'waiting for stable PostgreSQL readiness' + return False + if self.process is not None and self.process.poll() is None: + return True + now = self._clock() + if not force and now < self.next_start_at: + self.status = 'backoff' + return False + if force: + self.next_start_at = 0.0 + try: + return self._launch(now) + except Exception as exc: + if self.process is not None and self.process.poll() is None: + self.status = 'failed' + self.detail = 'dashboard process launch cleanup failed' + self.fatal_stop_failure = True + return False + self.process = None + self._schedule_failure(now, f'dashboard process launch failed: {type(exc).__name__}') + return False + + def _stop_current(self, timeout=15): + process = self.process + if process is None or process.poll() is not None: + self.process = None + if self.log_pump is not None: + self.log_pump.join(timeout=5) + self.log_pump = None + return True + try: + process.terminate() + try: + process.wait(timeout=timeout) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=5) + except Exception as exc: + self.detail = f'dashboard process stop failed: {type(exc).__name__}' + self.fatal_stop_failure = True + return False + if process.poll() is None: + self.detail = 'dashboard process remained live after bounded stop' + self.fatal_stop_failure = True + return False + if self.log_pump is not None: + self.log_pump.join(timeout=5) + if self.log_pump.error: + self.detail = f'bounded dashboard log writer failed: {self.log_pump.error}' + self.log_pump = None + self.fatal_stop_failure = True + return False + self.log_pump = None + self.process = None + return True + + def _schedule_failure(self, now, detail): + self.failures += 1 + delay = min(self.restart_base_sec * (2 ** min(self.failures - 1, 20)), self.restart_max_sec) + self.next_start_at = now + delay + self.status = 'backoff' + self.detail = detail + self.started_at = None + self.healthy_since = None + + def _consume_health(self, now): + if self._future is None or not self._future.done(): + return + future = self._future + generation = self._future_generation + self._future = None + self._future_generation = None + if generation != self._generation: + return + try: + healthy = bool(future.result()) + except BaseException: + healthy = False + if self.process is None or self.process.poll() is not None: + return + if healthy: + if self.healthy_since is None: + self.healthy_since = now + self.status = 'healthy' + self.detail = 'Streamlit health endpoint is ready' + if now - self.healthy_since >= self.stable_health_sec: + self.failures = 0 + self.next_health_at = now + self.health_interval_sec + return + self.healthy_since = None + if self.started_at is not None and now - self.started_at < self.startup_grace_sec: + self.status = 'pending' + self.detail = 'waiting for Streamlit health during startup grace' + self.next_health_at = now + self.health_interval_sec + return + stopped = self._stop_current() + detail = 'dashboard health check failed after startup grace' + if not stopped: + self.status = 'failed' + self.detail = detail + '; current owned process did not stop' + raise RuntimeError(self.detail) + self._schedule_failure(now, detail) + + def poll(self): + now = self._clock() + self._consume_health(now) + if self.start_gate is not None and not self.start_gate(): + return self.status + if self.desired_state != 'running': + return self.status + if self.dependency_gate is not None and not self.dependency_gate.ready: + if self.process is not None: + self.dependency_unavailable() + return self.status + if self.process is None: + self.start(record_intent=False) + return self.status + code = self.process.poll() + if code is None and self.log_pump is not None and self.log_pump.error: + self.detail = f'bounded dashboard log writer failed: {self.log_pump.error}' + if not self._stop_current(): + self.status = 'failed' + raise RuntimeError(self.detail) + self._schedule_failure(now, self.detail) + return self.status + if code is not None: + if self.log_pump is not None: + self.log_pump.join(timeout=5) + if self.log_pump.error: + self.detail = f'bounded dashboard log writer failed: {self.log_pump.error}' + self.log_pump = None + self.process = None + self._generation += 1 + if self._future is not None: + self._future.cancel() + self._future = None + self._future_generation = None + self._schedule_failure(now, f'dashboard process exited with code {format_exit_code(code)}') + return self.status + if self._future is None and now >= self.next_health_at: + self._future = self._executor.submit(self.health_probe, self.url, self.health_timeout_sec) + self._future_generation = self._generation + return self.status + + def stop(self, timeout=15, final=False, preserve_desired=False): + if not preserve_desired: + self.desired_state = 'stopped' + if self._future is not None: + self._future.cancel() + self._future = None + self._future_generation = None + self._generation += 1 + stopped = self._stop_current(timeout=timeout) + if stopped: + self.fatal_stop_failure = False + self.status = 'blocked' if preserve_desired and self.desired_state == 'running' else 'stopped' + self.detail = 'waiting for stable PostgreSQL readiness' if self.status == 'blocked' else '' + else: + self.status = 'failed' + return stopped + + def dependency_unavailable(self): + if self.process is not None and self.process.poll() is None: + return self.stop(preserve_desired=True) + if self.desired_state != 'running': + return True + return self.stop(preserve_desired=True) + + def dependency_available(self): + if self.desired_state == 'running': + return self.start(force=True, record_intent=False) + return True + + def snapshot(self): + pid = self.process.pid if self.process is not None and self.process.poll() is None else None + return { + 'status': self.status, + 'desired': self.desired_state, + 'healthy': self.healthy, + 'pid': pid, + 'failures': self.failures, + 'detail': self.detail, + 'url': self.url, + } + + def close(self): + if self._owns_executor: + self._executor.shutdown(wait=False, cancel_futures=True) + + +def handle_dashboard_command(parts, context): + manager = context.get('dashboard_manager') if context else None + if manager is None: + command_failure(context, 'Dashboard control is unavailable in this supervisor context') + return True + action = parts[1].lower() if len(parts) > 1 else 'status' + if action == 'status': + snapshot = manager.snapshot() + pid = f" pid={snapshot['pid']}" if snapshot.get('pid') else '' + print(f"dashboard: {snapshot['status']}{pid}: {snapshot['detail']}") + return True + if action == 'stop': + if manager.stop(): + print('dashboard: stopped') + else: + command_failure(context, 'dashboard: stop failed') + return True + if action == 'start': + started = manager.start(force=True) + snapshot = manager.snapshot() + print(f"dashboard: {snapshot['status']}: {snapshot['detail']}") + if not started and snapshot['status'] != 'blocked': + if context is not None: + context['_command_failed'] = True + return True + if action == 'restart': + if not manager.stop(): + command_failure(context, 'dashboard: restart refused because the current process did not stop') + return True + started = manager.start(force=True) + snapshot = manager.snapshot() + print(f"dashboard: {snapshot['status']}: {snapshot['detail']}") + if not started and snapshot['status'] != 'blocked': + if context is not None: + context['_command_failed'] = True + return True + command_failure(context, 'Usage: dashboard status|stop|start|restart') + return True + + +def load_supervisor_runtime( + config_path, selected_sources_arg=None, *, managed_postgres=None, + final_cutover=None, +): + config = load_yaml( + config_path, + managed_postgres=managed_postgres, + final_cutover=final_cutover, + ) + return supervisor_runtime_from_config( + config_path, config, selected_sources_arg, + ) + + +def supervisor_runtime_from_config(config_path, config, selected_sources_arg=None): + global_config = config.get('global') or {} + project_dir = global_config.get('project_dir') or os.path.dirname(os.path.abspath(config_path)) + results_dir = global_config.get('results_dir') + supervisor_config = dict(config.get('supervisor') or {}) + supervisor_config.setdefault('interval', int(global_config.get('cooldown', 300) or 300)) + selected_sources = get_enabled_sources( + config, + selected_sources_arg or supervisor_config.get('enabled_sources') or supervisor_config.get('sources_enabled'), + supervisor_config, + ) + return config, project_dir, results_dir, supervisor_config, selected_sources + + +def validate_managed_runtime_startup(config_path): + from runtime_document_io import validate_managed_runtime_files + + return validate_managed_runtime_files(config_path) + + +def keychecks_config_for(config_path, config): + global_config = config.get('global') or {} + keychecks_config = dict(config.get('keychecks') or {}) + keycheck_dir = keychecks_config.get('keycheck_dir') or global_config.get('keycheck_dir') or os.path.join(os.path.dirname(global_config.get('results_dir') or ''), 'keychecks') + keychecks_config['keycheck_dir'] = keycheck_dir + keychecks_config.setdefault('summary_tsv', os.path.join(keycheck_dir, 'summary.tsv')) + keychecks_config.setdefault('summary_json', os.path.join(keycheck_dir, 'summary.json')) + keychecks_config.setdefault('alive_summary_tsv', os.path.join(keycheck_dir, 'alive_summary.tsv')) + for key in ('input', 'proxy_file', 'summary_tsv', 'summary_json', 'alive_summary_tsv', 'keycheck_dir'): + if keychecks_config.get(key): + keychecks_config[key] = resolve_optional_path(keychecks_config[key], global_config) + return keychecks_config + + +def should_manage_keychecks(config, supervisor_config, selected_sources_arg=None): + keychecks_config = config.get('keychecks') or {} + if not bool_value(keychecks_config.get('enabled'), False): + return False + return True + + +def reload_supervisor_config(managed_sources, context): + config_path = context['config_path'] + selected_sources_arg = context.get('selected_sources_arg') + global_force_once = bool_value(context.get('global_force_once'), False) + dependency_gate = context.get('dependency_gate') + try: + config, project_dir, results_dir, supervisor_config, selected_sources = load_supervisor_runtime( + config_path, + selected_sources_arg, + managed_postgres=bool(context.get('with_postgres')), + ) + except Exception as e: + print(f'Reload failed: {e}') + return False + + current = source_map(managed_sources) + include_keychecks = should_manage_keychecks(config, supervisor_config, selected_sources_arg) + selected_set = set(selected_sources) + if include_keychecks: + selected_set.add('keychecks') + next_sources = [] + added = [] + updated = [] + kept_running = [] + removed = [] + + for source_name in selected_sources: + source_config = (config.get('sources') or {}).get(source_name, {}) + options = source_options(source_name, supervisor_config, source_config) + enabled = bool_value(options.get('enabled'), True) + existing = current.get(source_name) + if existing: + if not enabled and existing.is_running(): + kept_running.append(source_name) + next_sources.append(existing) + elif not enabled: + if dependency_gate is not None: + dependency_gate.unregister(existing) + removed.append(source_name) + else: + was_running = existing.is_running() + existing.reconfigure(results_dir, supervisor_config, source_config, global_force_once) + updated.append(source_name + (' (restart to apply to running child)' if was_running else '')) + next_sources.append(existing) + elif enabled: + item = ManagedSource( + source_name, config_path, project_dir, results_dir, supervisor_config, + source_config, global_force_once, dependency_gate, + ) + added.append(source_name) + next_sources.append(item) + + if include_keychecks: + keychecks_config = keychecks_config_for(config_path, config) + existing = current.get('keychecks') + if existing: + was_running = existing.is_running() + existing.reconfigure(results_dir, supervisor_config, keychecks_config, global_force_once) + updated.append('keychecks' + (' (restart to apply to running child)' if was_running else '')) + next_sources.append(existing) + else: + item = ManagedKeychecks( + config_path, project_dir, results_dir, supervisor_config, + keychecks_config, global_force_once, dependency_gate, + ) + added.append('keychecks') + next_sources.append(item) + + for source in managed_sources: + if source.source in selected_set: + continue + if source.is_running(): + kept_running.append(source.source) + next_sources.append(source) + else: + if dependency_gate is not None: + dependency_gate.unregister(source) + removed.append(source.source) + + managed_sources[:] = next_sources + context['config'] = config + context['project_dir'] = project_dir + context['results_dir'] = results_dir + context['supervisor_config'] = supervisor_config + context['selected_sources'] = selected_sources + context['keychecks_config'] = keychecks_config_for(config_path, config) + + print('Reloaded config.yaml') + if added: + print('Added:', ', '.join(added)) + if updated: + print('Updated:', ', '.join(updated)) + if removed: + print('Removed:', ', '.join(removed)) + if kept_running: + print('Kept running until stopped/restarted:', ', '.join(kept_running)) + if not any((added, updated, removed, kept_running)): + print('No source changes') + return True + + +def handle_command(command, managed_sources, context=None): + if context is not None: + context['_command_failed'] = False + command = command.strip() + if not command: + return True + try: + parts = shlex.split(command) + except ValueError as e: + print(f'Invalid command: {e}') + return True + if not parts: + return True + + action = parts[0].lower() + if command_is_mutating(parts) and action != 'shutdown' and not lifecycle_start_allowed(context): + command_failure(context, f'supervisor lifecycle is {lifecycle_phase(context)}; mutation is refused') + return True + if command_is_mutating(parts) and action != 'shutdown' and not check_runtime_authority(context): + command_failure(context, (context or {}).get('authority_drift') or 'runtime authority drifted') + return True + if action in ('help', 'h', '?'): + print(HELP_TEXT) + return True + if action in ('quit', 'exit', 'q'): + return False + if action == 'shutdown': + shutdown_event = context.get('shutdown_event') if context else None + if shutdown_event is None: + command_failure(context, 'Shutdown control is unavailable in this supervisor context') + else: + begin_stopping(context) + print('Coordinated shutdown requested') + return True + if action in ('status', 's'): + poll_managed_sources(managed_sources, context) + print_table(managed_sources, clear=False, context=context) + return True + if action == 'auth': + if len(parts) < 2: + print('Usage: auth ') + return True + targets = select_sources(managed_sources, parts[1]) + for source in targets: + print_auth_status(source) + return True + if action in ('watch', 'w'): + print('watch is only available in foreground supervisor or --attach prompt.') + return True + if action in ('reload', 'r'): + command_failure(context, 'Live reload is disabled by config authority binding; use authenticated coordinated shutdown and restart the supervisor.') + return True + + if action == 'dashboard': + return handle_dashboard_command(parts, context) + + if action == 'recheck': + return handle_recheck_command(parts, managed_sources, context) + + if action in ('start', 'stop', 'restart', 'pause', 'resume', 'once', 'command'): + if len(parts) < 2: + print(f'Usage: {action} ') + return True + targets = select_sources(managed_sources, parts[1]) + for source in targets: + if isinstance(source, ManagedDiscoveryProducer) and action == 'once': + command_failure(context, f'{source.source}: producer mode is fixed to once+repeat') + continue + if getattr(source, 'manual_only', False) and parts[1].lower() == 'all' and action == 'start': + continue + if getattr(source, 'manual_only', False) and action not in ('start', 'stop', 'command'): + command_failure( + context, + f'{source.source}: only explicit start, stop, command, and logs are supported', + ) + continue + if ( + isinstance(source, ManagedPipelineWorker) + and source.source in ('result-ingester', 'jsonl-projector') + and action in ('stop', 'restart', 'pause', 'once') + and any( + item.is_running() for item in managed_sources + if not isinstance(item, ManagedPipelineWorker) and item.source != 'keychecks' + ) + ): + command_failure( + context, + f'{source.source}: manual lifecycle change is refused while scanner sources are running', + ) + continue + if action == 'start': + started = source.start(force=True) + if started: + print(f'{source.source}: started') + elif source.runtime_blocked: + print(f'{source.source}: running intent recorded; dependency blocked') + else: + command_failure(context, f'{source.source}: start failed: {source.last_action_error or "process did not start"}') + elif action == 'stop': + if source.stop(): + print(f'{source.source}: stopped') + else: + command_failure(context, f'{source.source}: stop failed: {source.last_action_error or "process remained live"}') + elif action == 'restart': + started = source.restart_now() + if started: + print(f'{source.source}: restarted') + elif source.last_action_error: + command_failure(context, f'{source.source}: restart failed: {source.last_action_error}') + else: + print(f'{source.source}: restart intent recorded; dependency blocked') + elif action == 'pause': + if source.pause(): + print(f'{source.source}: paused') + else: + command_failure(context, f'{source.source}: pause failed: {source.last_action_error or "process remained live"}') + elif action == 'resume': + started = source.resume() + if started: + print(f'{source.source}: resumed') + elif source.runtime_blocked: + print(f'{source.source}: resume intent recorded; dependency blocked') + else: + command_failure(context, f'{source.source}: resume failed: {source.last_action_error or "process did not start"}') + elif action == 'once': + set_mode(source, 'once') + started = source.start(force=True) + if started: + print(f'{source.source}: once started') + elif source.runtime_blocked: + print(f'{source.source}: once intent recorded; dependency blocked') + else: + command_failure(context, f'{source.source}: once start failed: {source.last_action_error or "process did not start"}') + elif action == 'command': + print_source_command(source) + return True + + if action == 'mode': + if len(parts) < 3: + print('Usage: mode loop|once|repeat') + return True + for source in select_sources(managed_sources, parts[1]): + if isinstance(source, ManagedDiscoveryProducer): + command_failure(context, f'{source.source}: producer mode changes are forbidden') + continue + if getattr(source, 'manual_only', False): + command_failure(context, f'{source.source}: mode changes are forbidden') + continue + if set_mode(source, parts[2]): + print(f'{source.source}: mode={source.mode_label()}') + return True + + if action == 'set': + if len(parts) < 4: + print('Usage: set interval|restart|restart_delay ') + return True + key = parts[2].lower() + value = parts[3] + for source in select_sources(managed_sources, parts[1]): + if isinstance(source, ManagedDiscoveryProducer) and key != 'interval': + command_failure( + context, + f'{source.source}: only producer interval may be changed at runtime', + ) + continue + if getattr(source, 'manual_only', False): + command_failure(context, f'{source.source}: runtime option changes are forbidden') + continue + if key == 'interval': + try: + source.interval = max(0, int(value)) + except ValueError: + print('interval must be an integer number of seconds') + continue + print(f'{source.source}: interval={source.interval}') + elif key == 'restart_delay': + try: + source.restart_delay = max(0, int(value)) + except ValueError: + print('restart_delay must be an integer number of seconds') + continue + print(f'{source.source}: restart_delay={source.restart_delay}') + elif key == 'restart': + source.restart = bool_value(value, source.restart) + print(f'{source.source}: restart={source.restart}') + else: + print(f'Unknown set key: {key}') + return True + + if action in ('logs', 'tail'): + if len(parts) < 2: + print('Usage: logs [lines]') + return True + targets = select_sources(managed_sources, parts[1]) + if not targets: + return True + try: + limit = int(parts[2]) if len(parts) > 2 else 40 + except ValueError: + print('lines must be an integer') + return True + source = targets[0] + print(f'--- {source.log_path} (last {limit}) ---') + for line in source.tail_log_lines(limit): + print(console_safe_text(line)) + print('--- end log ---') + return True + + command_failure(context, f'Unknown command: {action}. Type `help`.') + return True + + +def read_single_key(): + if os.name == 'nt': + import msvcrt + if not msvcrt.kbhit(): + return None + ch = msvcrt.getwch() + if ch in ('\x00', '\xe0'): + if msvcrt.kbhit(): + msvcrt.getwch() + return None + return ch + import select + ready, _, _ = select.select([sys.stdin], [], [], 0) + if ready: + return sys.stdin.read(1) + return None + + +def poll_managed_sources(managed_sources, context): + for source in managed_sources: + if not lifecycle_start_allowed(context): + return + source.poll() + if context is not None and not context.get('background_child') and source.status == 'failed': + context['runtime_failed'] = True + + +def watch_local(managed_sources, poll_sec=0.5, context=None): + if not sys.stdin.isatty() or not sys.stdout.isatty() or not enable_ansi_terminal(): + print('watch requires an interactive ANSI terminal') + return + sys.stdout.write(ANSI_ALT_SCREEN) + sys.stdout.flush() + try: + while True: + guard = (context or {}).get('control_lock') or nullcontext() + pipeline_snapshot = pipeline_status_snapshot(context) + with guard: + if shutdown_checkpoint(context): + return + tick_supervisor_runtime(context, pipeline_snapshot=pipeline_snapshot) + poll_managed_sources(managed_sources, context) + sys.stdout.write(ANSI_HOME + ANSI_CLEAR_SCREEN) + print('\n'.join(build_runtime_table_lines(managed_sources, context=context))) + print('\nwatch mode: press q to return') + sys.stdout.flush() + deadline = time.time() + max(0.1, float(poll_sec or 0.5)) + while time.time() < deadline: + with guard: + if shutdown_checkpoint(context): + return + key = read_single_key() + if key and key.lower() == 'q': + return + time.sleep(0.05) + finally: + sys.stdout.write(ANSI_MAIN_SCREEN) + sys.stdout.flush() + + +def watch_remote(metadata, poll_sec=0.5): + if not sys.stdin.isatty() or not sys.stdout.isatty() or not enable_ansi_terminal(): + print('watch requires an interactive ANSI terminal') + return + sys.stdout.write(ANSI_ALT_SCREEN) + sys.stdout.flush() + try: + while True: + snapshot = get_control_snapshot(metadata) + sys.stdout.write(ANSI_HOME + ANSI_CLEAR_SCREEN) + print((snapshot.get('table') or '').rstrip()) + print('\nwatch mode: press q to return') + sys.stdout.flush() + deadline = time.time() + max(0.1, float(poll_sec or 0.5)) + while time.time() < deadline: + key = read_single_key() + if key and key.lower() == 'q': + return + time.sleep(0.05) + finally: + sys.stdout.write(ANSI_MAIN_SCREEN) + sys.stdout.flush() + + +def interactive_loop_blocking(managed_sources, autostart=False, clear=True, context=None, poll_sec=0.5): + guard = (context or {}).get('control_lock') or nullcontext() + with guard: + if shutdown_checkpoint(context): + return + if autostart: + for source in autostart_sources(managed_sources): + if shutdown_checkpoint(context) or not lifecycle_start_allowed(context): + break + source.start(force=True) + print_table(managed_sources, clear=clear, context=context) + print('\n' + HELP_TEXT + '\n') + commands = queue.Queue() + allow_read = threading.Event() + allow_read.set() + + def read_commands(): + while True: + allow_read.wait() + allow_read.clear() + try: + commands.put(input('supervisor> ')) + except (EOFError, KeyboardInterrupt): + commands.put(None) + return + + threading.Thread(target=read_commands, daemon=True).start() + while True: + pipeline_snapshot = pipeline_status_snapshot(context) + with guard: + if shutdown_checkpoint(context): + break + tick_supervisor_runtime(context, pipeline_snapshot=pipeline_snapshot) + poll_managed_sources(managed_sources, context) + if shutdown_checkpoint(context): + break + try: + command = commands.get(timeout=max(0.05, float(poll_sec or 0.2))) + except queue.Empty: + continue + if command is None: + print() + break + if command.strip().lower() in ('watch', 'w'): + watch_local(managed_sources, poll_sec, context) + allow_read.set() + continue + with guard: + if shutdown_checkpoint(context): + break + keep_running = handle_command(command, managed_sources, context) + if not keep_running: + break + allow_read.set() + print('Stopping child processes...') + + +def interactive_loop(managed_sources, autostart=False, clear=True, context=None, poll_sec=0.2): + interactive_loop_blocking(managed_sources, autostart, clear, context, poll_sec) + + +def tick_supervisor_runtime(context, pipeline_snapshot=None): + context = context or {} + context.pop('_defer_pipeline_status_until_next_tick', None) + if shutdown_checkpoint(context): + return + if not lifecycle_start_allowed(context): + if context.get('authority_drift'): + inhibit_for_authority_drift(context, context['authority_drift']) + controller = context.get('postgres_controller') + if controller is not None: + controller.tick() + return + now = time.monotonic() + if now >= float(context.get('next_authority_check_at') or 0): + interval = max(0.2, float((context.get('supervisor_config') or {}).get('authority_check_interval_sec', 5) or 5)) + context['next_authority_check_at'] = now + interval + if not check_runtime_authority(context): + return + elif context.get('authority_drift'): + return + controller = context.get('postgres_controller') + gate = context.get('dependency_gate') + if controller is None: + if gate is not None: + gate.set_ready(True) + else: + previous = context.get('postgres_reported_state') + controller.tick() + snapshot = controller.snapshot() + current = snapshot['state'] + if current != previous: + detail = f": {snapshot['detail']}" if snapshot.get('detail') else '' + print(f'PostgreSQL controller: {current}{detail}') + context['postgres_reported_state'] = current + if gate is not None: + try: + gate_was_ready = gate.ready + gate.set_ready(controller.ready) + if not gate_was_ready and gate.ready: + context['_defer_pipeline_status_until_next_tick'] = True + context['_pipeline_status_refresh_at'] = 0 + except DependencyStopError as exc: + controller.inhibit_lifecycle(str(exc)) + context['fatal_child_stop_error'] = str(exc) + shutdown_event = context.get('shutdown_event') + if shutdown_event is not None: + shutdown_event.set() + raise + source_gate = context.get('source_dependency_gate') + if ( + source_gate is not None + and not source_gate.ready + and not context.get('_defer_pipeline_status_until_next_tick') + ): + snapshot = ( + pipeline_snapshot + if pipeline_snapshot is not None + else pipeline_status_snapshot(context) + ) + if snapshot.get('ingester_ready'): + source_gate.set_ready(True) + context['pipeline_initial_ready'] = True + print('Result ingester reported ready; source admission gate is open.') + dashboard = context.get('dashboard_manager') + if dashboard is not None: + try: + dashboard.poll() + except Exception as exc: + if controller is not None: + controller.inhibit_lifecycle(f'dashboard stop failure: {exc}') + context['fatal_child_stop_error'] = str(exc) + shutdown_event = context.get('shutdown_event') + if shutdown_event is not None: + shutdown_event.set() + raise + + +def non_interactive_loop(managed_sources, refresh_sec=DEFAULT_REFRESH_SEC, status_file=None, poll_sec=1.0, context=None, lock=None, stay_alive=False, status_heartbeat_sec=60): + last_status_signature = None + last_status_write = 0 + guard = lock or (context or {}).get('control_lock') or nullcontext() + while True: + pipeline_snapshot = pipeline_status_snapshot(context) + with guard: + if shutdown_checkpoint(context): + return + tick_supervisor_runtime(context, pipeline_snapshot=pipeline_snapshot) + poll_managed_sources(managed_sources, context) + if shutdown_checkpoint(context): + return + + if context.get('_defer_pipeline_status_until_next_tick'): + time.sleep(max(0.2, float(poll_sec or 1.0))) + continue + + with guard: + if shutdown_checkpoint(context): + return + controller = context.get('postgres_controller') if context else None + postgres_signature = None + if controller is not None: + snapshot = controller.snapshot() + postgres_signature = (snapshot['state'], snapshot['failures'], snapshot['detail']) + dashboard = context.get('dashboard_manager') if context else None + dashboard_signature = None + if dashboard is not None: + dashboard_snapshot = dashboard.snapshot() + dashboard_signature = tuple(dashboard_snapshot.get(key) for key in ('status', 'desired', 'healthy', 'pid', 'failures', 'detail')) + scan_snapshot = current_scan_worker_snapshot(context) + scan_signature = ( + scan_snapshot['active'], scan_snapshot['limit'], scan_snapshot['trufflehog'], + tuple(scan_snapshot['sources'].items()), scan_snapshot['detail'], + ) + pipeline_signature = tuple( + pipeline_snapshot.get(key) for key in ( + 'ingester_state', 'projector_state', 'bundle_items', 'bundle_bytes', + 'projection_items', 'projection_bytes', 'keycheck_items', + 'keycheck_bytes', 'quarantine_items', 'quarantine_bytes', 'detail', + ) + ) + current_signature = ( + table_signature(managed_sources), postgres_signature, + dashboard_signature, scan_signature, pipeline_signature, + ) + now = time.time() + heartbeat_due = status_heartbeat_sec and now - last_status_write >= max(1, int(status_heartbeat_sec)) + if current_signature != last_status_signature or heartbeat_due: + write_status_file(status_file, managed_sources, context) + last_status_signature = current_signature + last_status_write = now + + with guard: + all_done = all(source.status in ('done', 'failed', 'disabled', 'stopped') for source in managed_sources) + dashboard = context.get('dashboard_manager') if context else None + if dashboard is not None and dashboard.desired_state == 'running': + all_done = False + if all_done and not stay_alive: + break + time.sleep(max(0.2, float(poll_sec or 1.0))) + + +def _coordinated_shutdown_locked(managed_sources, context): + context = {} if context is None else context + context['coordinated_shutdown_complete'] = False + context['authority_release_safe'] = False + children_complete = True + drain_timeout = max( + 0.0, + float((context.get('supervisor_config') or {}).get('source_handoff_drain_timeout_sec', 30) or 0), + ) + deadline = time.monotonic() + drain_timeout + while time.monotonic() < deadline: + snapshot = scan_worker_snapshot(context.get('config')) + if snapshot['active'] == 0: + break + time.sleep(0.1) + + ordered = sorted(managed_sources, key=lambda source: { + 'worker-api': 1, 'keychecks': 2, 'jsonl-projector': 3, + 'result-ingester': 4, 'janitor': 5, + }.get(source.source, 0)) + for source in ordered: + try: + if not source.stop(final=True): + children_complete = False + print(f'{source.source}: shutdown did not confirm process exit') + except Exception as exc: + children_complete = False + print(f'{source.source}: shutdown failed: {exc}') + dashboard = context.get('dashboard_manager') + if dashboard is not None: + try: + if not dashboard.stop(final=True): + children_complete = False + print('dashboard: shutdown did not confirm process exit') + except Exception as exc: + children_complete = False + print(f'dashboard: shutdown failed: {exc}') + + for source in ordered: + try: + if source.is_running() is not False: + children_complete = False + print(f'{source.source}: process exit is unconfirmed after bounded shutdown') + except Exception: + children_complete = False + if dashboard is not None: + process = dashboard.process + try: + if process is not None and process.poll() is None: + children_complete = False + print('dashboard: process is still live after bounded shutdown') + except Exception: + children_complete = False + + controller = context.get('postgres_controller') + if controller is None: + if dashboard is not None and children_complete: + dashboard.close() + context['coordinated_shutdown_complete'] = children_complete + context['authority_release_safe'] = children_complete + return children_complete + if not children_complete: + print('PostgreSQL controller close/stop was deferred because one or more managed children did not stop.') + return False + supervisor_config = context.get('supervisor_config') or {} + timeout = max(5.0, float(supervisor_config.get('postgres_shutdown_timeout_sec', 120) or 120)) + deadline = time.monotonic() + timeout + if bool(getattr(controller, 'lifecycle_action_required', True)): + controller.request_stop() + while not controller.terminal and time.monotonic() < deadline: + controller.tick() + time.sleep(min(0.05, max(0.0, deadline - time.monotonic()))) + if not controller.terminal: + print(f'PostgreSQL controller stop did not complete within {timeout:g}s; no unverified process action was taken.') + try: + close_result = controller.close(wait=False, timeout_sec=max(0.0, deadline - time.monotonic())) + except TypeError: + close_result = controller.close(wait=False) + if controller.detail: + print(f'PostgreSQL controller: {controller.state.value}: {controller.detail}') + close_safe = bool(close_result) and bool(getattr(controller, 'authority_release_safe', False)) + if dashboard is not None and close_safe: + dashboard.close() + context['coordinated_shutdown_complete'] = ( + children_complete + and bool(getattr(controller, 'terminal', False)) + and bool(getattr(controller, 'stop_succeeded', False)) + and close_safe + ) + context['authority_release_safe'] = context['coordinated_shutdown_complete'] + return context['coordinated_shutdown_complete'] + + +def coordinated_shutdown(managed_sources, context): + context = {} if context is None else context + guard = context.get('control_lock') or nullcontext() + with guard: + context['authority_release_safe'] = False + try: + begin_stopping(context) + complete = _coordinated_shutdown_locked(managed_sources, context) + if not complete: + enter_failed_hold(context, 'one or more owned processes did not confirm stopped disposition') + return complete + except BaseException: + try: + enter_failed_hold(context, 'coordinated shutdown was interrupted or failed') + except BaseException: + pass + raise + + +def retain_unsafe_authority(managed_sources, context, max_attempts=None): + """Retry bounded stops while retaining locks, metadata, and authenticated control.""" + context = {} if context is None else context + if context.get('authority_release_safe', False): + return True + guard = context.get('control_lock') or nullcontext() + try: + with guard: + enter_failed_hold(context, context.get('shutdown_failure')) + except BaseException: + pass + retry_delay = 30.0 + attempts = 0 + retry_event = context.get('shutdown_retry_event') + while True: + try: + attempts += 1 + supervisor_config = context.get('supervisor_config') or {} + retry_delay = max(1.0, float(supervisor_config.get('postgres_stop_failed_retry_sec', 30) or 30)) + # FAILED_HOLD permits only read-only status and another shutdown + # request, so retries need not monopolize authenticated control. + if _coordinated_shutdown_locked(managed_sources, context): + context['shutdown_failure'] = '' + context['authority_release_safe'] = True + return True + with guard: + enter_failed_hold(context, 'bounded shutdown retry did not confirm every owned process stopped') + print('FAILED_HOLD: retaining all OS locks and authenticated control until every owned process is stopped.') + write_status_file(context.get('status_file'), managed_sources, context) + if max_attempts is not None and attempts >= max(1, int(max_attempts)): + return False + if retry_event is not None: + retry_event.wait(retry_delay) + retry_event.clear() + else: + time.sleep(retry_delay) + except BaseException as exc: + context['authority_release_safe'] = False + try: + with guard: + enter_failed_hold(context, f'shutdown retry failed: {type(exc).__name__}: {exc}') + print(f'Authority retention ignored shutdown interruption: {type(exc).__name__}: {exc}') + except BaseException: + pass + if max_attempts is not None and attempts >= max(1, int(max_attempts)): + return False + try: + if retry_event is not None: + retry_event.wait(retry_delay) + retry_event.clear() + else: + time.sleep(retry_delay) + except BaseException: + pass + + +def retain_unsafe_postgres_authority(context): + """Compatibility wrapper for pre-activation PostgreSQL compensation.""" + if (context or {}).get('authority_release_safe', False): + return True + return retain_unsafe_authority([], context) + + +def dashboard_settings(supervisor_config): + dashboard_config = supervisor_config.get('dashboard') + if not dashboard_config: + return False, '127.0.0.1', 5000, None + if isinstance(dashboard_config, dict): + enabled = bool_value(dashboard_config.get('enabled'), False) + port = int(dashboard_config.get('port', 5000)) + host = str(dashboard_config.get('address') or dashboard_config.get('host') or '127.0.0.1') + else: + enabled = bool_value(dashboard_config, False) + port = 5000 + host = '127.0.0.1' + if not is_loopback_host(host): + raise ValueError(f'dashboard address must be loopback-only: {host}') + if not 0 < port <= 65535: + raise ValueError(f'invalid dashboard port: {port}') + return enabled, host, port, f'http://{host}:{port}' + + +def background_paths(config_path, results_dir, supervisor_config, explicit_instance_file=None, explicit_pid_file=None): + log_dir = resolve_path(config_path, supervisor_config.get('log_dir') or os.path.join(results_dir, 'logs')) + require_private_directory(log_dir, create=False) + control_dir = resolve_path( + config_path, + supervisor_config.get('control_dir') or os.path.join(os.path.dirname(log_dir), 'control'), + ) + if os.path.normcase(os.path.abspath(control_dir)) == os.path.normcase(os.path.abspath(log_dir)): + raise ValueError('private supervisor control directory must be separate from the log directory') + require_private_directory(control_dir, create=False) + if explicit_instance_file: + instance_file = resolve_path(config_path, explicit_instance_file) + if os.path.normcase(os.path.dirname(os.path.abspath(instance_file))) != os.path.normcase(os.path.abspath(control_dir)): + raise ValueError('custom supervisor instance metadata must remain in the private control directory') + else: + instance_file = resolve_path(config_path, supervisor_config.get('instance_file') or os.path.join(control_dir, 'supervisor.instance.json')) + if os.path.normcase(os.path.dirname(os.path.abspath(instance_file))) != os.path.normcase(os.path.abspath(control_dir)): + raise ValueError('supervisor instance metadata must reside directly in the private control directory') + reject_reparse_components(os.path.dirname(os.path.abspath(instance_file))) + if not private_directory_ready(os.path.dirname(os.path.abspath(instance_file))): + raise ValueError('supervisor instance parent directory is not private') + status_file = resolve_path(config_path, supervisor_config.get('status_file') or os.path.join(log_dir, 'supervisor.status.txt')) + log_file = resolve_path(config_path, supervisor_config.get('supervisor_log') or os.path.join(log_dir, 'supervisor.log')) + legacy_pid_file = resolve_path(config_path, explicit_pid_file) if explicit_pid_file else os.path.join(log_dir, 'supervisor.pid') + background_lock_path(config_path, results_dir, supervisor_config) + return instance_file, log_file, status_file, legacy_pid_file + + +def background_lock_path(config_path, results_dir, supervisor_config): + log_dir = resolve_path(config_path, supervisor_config.get('log_dir') or os.path.join(results_dir, 'logs')) + control_dir = resolve_path( + config_path, + supervisor_config.get('control_dir') or os.path.join(os.path.dirname(log_dir), 'control'), + ) + if os.path.normcase(os.path.abspath(control_dir)) == os.path.normcase(os.path.abspath(log_dir)): + raise ValueError('private supervisor control directory must be separate from the log directory') + require_private_directory(control_dir, create=False) + lock_file = resolve_path(config_path, supervisor_config.get('lock_file') or os.path.join(control_dir, 'supervisor.lock')) + if os.path.normcase(os.path.dirname(os.path.abspath(lock_file))) != os.path.normcase(os.path.abspath(control_dir)): + raise ValueError('supervisor singleton lock must reside directly in the private control directory') + reject_reparse_components(os.path.dirname(os.path.abspath(lock_file))) + return lock_file + + +def legacy_log_instance_path(config_path, results_dir, supervisor_config): + log_dir = resolve_path(config_path, supervisor_config.get('log_dir') or os.path.join(results_dir, 'logs')) + return os.path.join(log_dir, 'supervisor.instance.json') + + +def read_pid_file(path): + try: + with open(path, 'r', encoding='utf-8') as f: + return int(f.read().strip()) + except (OSError, ValueError): + return None + + +def write_status_file(status_file, managed_sources, context=None): + if not status_file: + return + parent = os.path.dirname(status_file) + if parent: + require_private_directory(parent, create=False) + if os.path.lexists(status_file) and not private_file_ready(status_file): + raise OSError(f'private status file ACL is not ready; run offline hardening: {status_file}') + tmp_path = f'{status_file}.{os.getpid()}.{threading.get_ident()}.tmp' + with open(tmp_path, 'x', encoding='utf-8') as f: + controller = context.get('postgres_controller') if context else None + if controller is not None: + snapshot = controller.snapshot() + f.write(f"PostgreSQL: {snapshot['state']} failures={snapshot['failures']} detail={snapshot['detail']}\n\n") + dashboard = context.get('dashboard_manager') if context else None + if dashboard is not None: + snapshot = dashboard.snapshot() + f.write( + f"Dashboard: {snapshot['status']} desired={snapshot['desired']} " + f"healthy={snapshot['healthy']} detail={snapshot['detail']}\n\n" + ) + f.write('\n'.join(build_runtime_table_lines(managed_sources, context=context))) + f.write('\n') + harden_private_file(tmp_path) + for attempt in range(5): + try: + os.replace(tmp_path, status_file) + if not private_file_ready(status_file): + raise OSError(f'private status file ACL changed during update: {status_file}') + return + except PermissionError as e: + if attempt == 4: + print(f'Unable to update status file {status_file}: {e}') + break + time.sleep(0.1 * (attempt + 1)) + try: + if os.path.exists(tmp_path): + os.remove(tmp_path) + except OSError: + pass + + +def control_address(supervisor_config): + host = str(supervisor_config.get('control_host') or '127.0.0.1') + port = int(supervisor_config.get('control_port') or 8765) + if not is_loopback_host(host): + raise ValueError(f'control address must be loopback-only: {host}') + if not 0 <= port <= 65535: + raise ValueError(f'invalid control port: {port}') + return host, port + + +def send_control_request(metadata, action, command=None, timeout=60, **extra): + control = metadata['control'] + request = { + 'schema': CONTROL_SCHEMA, + 'instance_id': metadata['instance_id'], + 'token': metadata['token'], + 'action': str(action), + } + if command is not None: + request['command'] = str(command) + request.update(extra) + payload = json.dumps(request, ensure_ascii=True, separators=(',', ':')).encode('utf-8') + b'\n' + if len(payload) > MAX_CONTROL_REQUEST_BYTES: + raise ValueError('control request is too large') + timeout_seconds = max(0.1, float(timeout)) + deadline = time.monotonic() + timeout_seconds + + def remaining_timeout(): + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError('control request timed out') + return remaining + + with socket.create_connection( + (control['host'], control['port']), timeout=remaining_timeout(), + ) as sock: + sock.settimeout(remaining_timeout()) + sock.sendall(payload) + sock.shutdown(socket.SHUT_WR) + chunks = [] + total = 0 + while True: + sock.settimeout(remaining_timeout()) + chunk = sock.recv(65536) + if not chunk: + break + total += len(chunk) + if total > MAX_CONTROL_RESPONSE_BYTES: + raise ValueError('control response is too large') + chunks.append(chunk) + raw = b''.join(chunks) + try: + response = json.loads(raw.decode('utf-8')) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ValueError('invalid control response') from exc + if not isinstance(response, dict) or response.get('schema') != CONTROL_SCHEMA: + raise ValueError('invalid control response schema') + if response.get('instance_id') != metadata['instance_id']: + raise ValueError('control response instance mismatch') + if type(response.get('ok')) is not bool: + raise ValueError('invalid control response status') + expected_fields = ( + {'schema', 'instance_id', 'ok', 'result'} + if response['ok'] else {'schema', 'instance_id', 'ok', 'error'} + ) + if set(response) != expected_fields: + raise ValueError('invalid control response fields') + if not response['ok']: + if type(response.get('error')) is not str: + raise ValueError('invalid control response error') + raise RuntimeError(str(response.get('error') or 'control request failed')) + return response.get('result') + + +def send_control_command(metadata, command, timeout=60): + return send_control_request(metadata, 'command', command=command, timeout=timeout) + + +def get_control_snapshot(metadata): + result = send_control_request(metadata, 'snapshot') + if not isinstance(result, dict): + raise ValueError('invalid supervisor snapshot') + return result + + +_STRUCTURED_SOURCE_FIELDS = frozenset({ + 'id', 'source', 'role', 'lifecycle_state', 'desired_state', 'process_state', + 'pid', 'enabled', 'dependency_blocked', 'startup_cleanup_pending', 'mode', + 'interval_seconds', 'restart_enabled', 'restart_delay_seconds', + 'restart_count', 'restart_streak', 'last_exit_code', 'last_exit_at', + 'next_scheduled_run_at', 'safe_error_category', 'auth_summary', + 'allowed_actions', +}) +_STRUCTURED_PRODUCER_FIELDS = frozenset({ + 'last_cycle_result', 'last_successful_discovery_at', +}) +_STRUCTURED_AUTH_FIELDS = frozenset({ + 'total', 'ok', 'dead', 'limited', 'rate_limit_errors', + 'auth_invalid_errors', +}) +_STRUCTURED_DASHBOARD_FIELDS = frozenset({ + 'id', 'status', 'desired_state', 'process_state', 'healthy', 'pid', + 'restart_count', 'safe_error_category', 'allowed_actions', +}) + + +def _nonnegative_control_integer(value): + return type(value) is int and value >= 0 + + +def _valid_structured_source_state(source, source_id=None): + if not isinstance(source, dict): + return False + fields = frozenset(source) + producer = fields == _STRUCTURED_SOURCE_FIELDS.union(_STRUCTURED_PRODUCER_FIELDS) + if fields != _STRUCTURED_SOURCE_FIELDS and not producer: + return False + if source_id is not None and source.get('id') != source_id: + return False + if any( + type(source.get(key)) is not str + for key in ( + 'id', 'source', 'role', 'lifecycle_state', 'desired_state', + 'process_state', 'mode', 'safe_error_category', + ) + ): + return False + if source['process_state'] not in ('running', 'stopped'): + return False + if not ( + source['pid'] is None + or (type(source['pid']) is int and source['pid'] > 0) + ): + return False + if any( + type(source.get(key)) is not bool + for key in ( + 'enabled', 'dependency_blocked', 'startup_cleanup_pending', + 'restart_enabled', + ) + ): + return False + if any( + not _nonnegative_control_integer(source.get(key)) + for key in ( + 'interval_seconds', 'restart_delay_seconds', 'restart_count', + 'restart_streak', + ) + ): + return False + if source['last_exit_code'] is not None and type(source['last_exit_code']) is not int: + return False + if any( + source[key] is not None and type(source[key]) is not str + for key in ('last_exit_at', 'next_scheduled_run_at') + ): + return False + auth = source['auth_summary'] + if not isinstance(auth, dict) or ( + auth and ( + frozenset(auth) != _STRUCTURED_AUTH_FIELDS + or any(not _nonnegative_control_integer(value) for value in auth.values()) + ) + ): + return False + if type(source['allowed_actions']) is not list or any( + type(action) is not str for action in source['allowed_actions'] + ): + return False + if not producer: + return True + cycle = source['last_cycle_result'] + return ( + source['role'] == DISCOVERY_PRODUCER_ROLE + and isinstance(cycle, dict) + and set(cycle) == { + 'status', 'fetched_count', 'queued_new_count', 'queued_updated_count', + } + and type(cycle.get('status')) is str + and all( + _nonnegative_control_integer(cycle.get(key)) + for key in ('fetched_count', 'queued_new_count', 'queued_updated_count') + ) + and ( + source['last_successful_discovery_at'] is None + or type(source['last_successful_discovery_at']) is str + ) + ) + + +def _valid_structured_dashboard_state(dashboard): + return ( + isinstance(dashboard, dict) + and frozenset(dashboard) == _STRUCTURED_DASHBOARD_FIELDS + and dashboard.get('id') == 'dashboard' + and all( + type(dashboard.get(key)) is str + for key in ( + 'id', 'status', 'desired_state', 'process_state', + 'safe_error_category', + ) + ) + and dashboard.get('process_state') in ('running', 'stopped') + and type(dashboard.get('healthy')) is bool + and ( + dashboard.get('pid') is None + or (type(dashboard.get('pid')) is int and dashboard['pid'] > 0) + ) + and _nonnegative_control_integer(dashboard.get('restart_count')) + and type(dashboard.get('allowed_actions')) is list + and all(type(action) is str for action in dashboard['allowed_actions']) + ) + + +def get_runtime_snapshot(metadata, timeout=60): + result = send_control_request(metadata, 'runtime-snapshot', timeout=timeout) + runtime_fields = { + 'pid', 'phase', 'manages_postgres', 'start_gate_open', + 'shutdown_requested', 'runtime_failed', + } + postgres_base_fields = {'state', 'ready', 'failures', 'safe_error_category'} + postgres_live_fields = postgres_base_fields.union({ + 'stop_succeeded', 'lifecycle_inert', 'automatic_inhibited', + 'authority_release_safe', 'inflight_start', 'lifecycle_action_required', + }) + pipeline_fields = { + 'ingester_ready', 'projector_ready', 'cutover_ready', + 'ingester_state', 'projector_state', + 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'keycheck_items', 'keycheck_bytes', 'quarantine_items', 'quarantine_bytes', + } + scan_worker_fields = { + 'active', 'limit', 'base_active', 'base_limit', 'bonus_active', + 'bonus_limit', 'trufflehog', 'sources', + } + valid = isinstance(result, dict) and set(result) == { + 'snapshot_schema', 'runtime', 'postgres', 'dashboard', 'sources', + 'pipeline', 'scan_workers', + } + if valid: + runtime = result['runtime'] + postgres = result['postgres'] + pipeline = result['pipeline'] + scan_workers = result['scan_workers'] + valid = ( + result['snapshot_schema'] == RUNTIME_SNAPSHOT_SCHEMA + and isinstance(runtime, dict) and set(runtime) == runtime_fields + and type(runtime.get('pid')) is int and runtime['pid'] > 0 + and type(runtime.get('phase')) is str + and all( + type(runtime.get(key)) is bool + for key in ( + 'manages_postgres', 'start_gate_open', 'shutdown_requested', + 'runtime_failed', + ) + ) + and isinstance(postgres, dict) + and set(postgres) in (postgres_base_fields, postgres_live_fields) + and type(postgres.get('state')) is str + and type(postgres.get('ready')) is bool + and _nonnegative_control_integer(postgres.get('failures')) + and type(postgres.get('safe_error_category')) is str + ) + if valid and set(postgres) == postgres_live_fields: + valid = all( + type(postgres.get(key)) is bool + for key in postgres_live_fields - postgres_base_fields + ) + valid = valid and _valid_structured_dashboard_state(result['dashboard']) + valid = valid and type(result['sources']) is list and all( + _valid_structured_source_state(source) for source in result['sources'] + ) + valid = valid and isinstance(pipeline, dict) and set(pipeline) == pipeline_fields + if valid: + valid = ( + type(pipeline['ingester_ready']) is bool + and type(pipeline['projector_ready']) is bool + and type(pipeline['cutover_ready']) is bool + and type(pipeline['ingester_state']) is str + and type(pipeline['projector_state']) is str + and all( + _nonnegative_control_integer(pipeline[key]) + for key in pipeline_fields - { + 'ingester_ready', 'projector_ready', 'cutover_ready', + 'ingester_state', + 'projector_state', + } + ) + and isinstance(scan_workers, dict) + and set(scan_workers) == scan_worker_fields + and all( + _nonnegative_control_integer(scan_workers[key]) + for key in scan_worker_fields - {'sources'} + ) + and isinstance(scan_workers['sources'], dict) + and all( + type(key) is str and _nonnegative_control_integer(value) + for key, value in scan_workers['sources'].items() + ) + ) + if not valid: + raise ValueError('invalid structured supervisor snapshot') + return result + + +def send_managed_source_action(metadata, source_id, source_action, timeout=60, **parameters): + result = send_control_request( + metadata, + 'managed-source-action', + timeout=timeout, + source_id=source_id, + source_action=source_action, + **parameters, + ) + source = result.get('source') if isinstance(result, dict) else None + if ( + not isinstance(result, dict) + or set(result) != {'source_action', 'outcome', 'source'} + or result.get('source_action') != source_action + or result.get('outcome') not in ('completed', 'dependency-blocked') + or not _valid_structured_source_state(source, source_id=source_id) + ): + raise ValueError('invalid managed source action response') + return result + + +def send_managed_source_log_tail(metadata, source_id, line_count, timeout=60): + result = send_control_request( + metadata, + 'managed-source-log-tail', + timeout=timeout, + source_id=source_id, + line_count=line_count, + ) + if ( + not isinstance(result, dict) + or set(result) != {'source_id', 'line_count', 'lines', 'response_truncated'} + or result.get('source_id') != source_id + or type(result.get('lines')) is not list + or any(type(line) is not str for line in result['lines']) + or type(result.get('line_count')) is not int + or result['line_count'] != len(result['lines']) + or result['line_count'] > line_count + or type(result.get('response_truncated')) is not bool + or len(json.dumps( + result['lines'], ensure_ascii=True, separators=(',', ':'), + ).encode('utf-8')) > MAX_LOG_TAIL_BYTES + ): + raise ValueError('invalid managed source log tail') + return result + + +def send_dashboard_action(metadata, dashboard_action, timeout=60): + result = send_control_request( + metadata, 'dashboard-action', timeout=timeout, dashboard_action=dashboard_action, + ) + dashboard = result.get('dashboard') if isinstance(result, dict) else None + if ( + not isinstance(result, dict) + or set(result) != {'dashboard_action', 'outcome', 'dashboard'} + or result.get('dashboard_action') != dashboard_action + or result.get('outcome') not in ('completed', 'dependency-blocked') + or not _valid_structured_dashboard_state(dashboard) + ): + raise ValueError('invalid dashboard action response') + return result + + +_CONTROL_BASE_FIELDS = frozenset({'schema', 'instance_id', 'token', 'action'}) + + +def require_exact_control_fields(request, fields): + if set(request) != _CONTROL_BASE_FIELDS.union(fields): + raise ValueError('control request fields are invalid for this action') + + +def structured_postgres_state(controller): + if controller is None: + return { + 'state': PostgresState.DISABLED.value, + 'ready': True, + 'failures': 0, + 'safe_error_category': '', + } + snapshot = controller.snapshot() + state = str(snapshot.get('state') or PostgresState.DISABLED.value) + return { + 'state': state, + 'ready': bool(snapshot.get('ready')), + 'failures': max(0, int(snapshot.get('failures', 0) or 0)), + 'stop_succeeded': bool(snapshot.get('stop_succeeded')), + 'lifecycle_inert': bool(snapshot.get('lifecycle_inert')), + 'automatic_inhibited': bool(snapshot.get('automatic_inhibited')), + 'authority_release_safe': bool(snapshot.get('authority_release_safe')), + 'inflight_start': bool(snapshot.get('inflight_start')), + 'lifecycle_action_required': bool(snapshot.get('lifecycle_action_required')), + 'safe_error_category': 'postgres_error' if 'FAILED' in state.upper() else '', + } + + +def structured_dashboard_state(manager): + if manager is None: + return { + 'id': 'dashboard', + 'status': 'disabled', + 'desired_state': 'stopped', + 'process_state': 'stopped', + 'healthy': False, + 'pid': None, + 'restart_count': 0, + 'safe_error_category': '', + 'allowed_actions': ['start', 'stop', 'restart'], + } + snapshot = manager.snapshot() + status = str(snapshot.get('status') or 'disabled') + pid = snapshot.get('pid') if type(snapshot.get('pid')) is int else None + safe_error = '' + if status in ('failed', 'backoff'): + safe_error = 'dashboard_error' + return { + 'id': 'dashboard', + 'status': status, + 'desired_state': str(snapshot.get('desired') or 'stopped'), + 'process_state': 'running' if pid is not None else 'stopped', + 'healthy': bool(snapshot.get('healthy')), + 'pid': pid, + 'restart_count': max(0, int(snapshot.get('failures', 0) or 0)), + 'safe_error_category': safe_error, + 'allowed_actions': ['start', 'stop', 'restart'], + } + + +def structured_runtime_snapshot(managed_sources, context): + context = context or {} + pipeline = pipeline_status_snapshot(context, allow_refresh=False) + scan_workers = current_scan_worker_snapshot(context) + return { + 'snapshot_schema': RUNTIME_SNAPSHOT_SCHEMA, + 'runtime': { + 'pid': os.getpid(), + 'phase': lifecycle_phase(context), + 'manages_postgres': bool(context.get('with_postgres')), + 'start_gate_open': bool(context.get('start_gate_open', True)), + 'shutdown_requested': bool(context.get('shutdown_requested')), + 'runtime_failed': bool(context.get('runtime_failed')), + }, + 'postgres': structured_postgres_state(context.get('postgres_controller')), + 'dashboard': structured_dashboard_state(context.get('dashboard_manager')), + 'sources': [source.structured_state() for source in managed_sources], + 'pipeline': { + key: pipeline[key] + for key in ( + 'ingester_ready', 'projector_ready', 'cutover_ready', + 'ingester_state', 'projector_state', + 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'keycheck_items', 'keycheck_bytes', 'quarantine_items', 'quarantine_bytes', + ) + }, + 'scan_workers': { + key: scan_workers[key] + for key in ( + 'active', 'limit', 'base_active', 'base_limit', 'bonus_active', + 'bonus_limit', 'trufflehog', 'sources', + ) + }, + } + + +def _pipeline_lifecycle_change_is_blocked(source, source_action, managed_sources): + return ( + isinstance(source, ManagedPipelineWorker) + and source.source in ('result-ingester', 'jsonl-projector') + and source_action in ('stop', 'restart', 'pause', 'once') + and any( + item.is_running() for item in managed_sources + if not isinstance(item, ManagedPipelineWorker) and item.source != 'keychecks' + ) + ) + + +def run_managed_source_action(request, managed_sources, context): + source_action = request.get('source_action') + expected_fields = {'source_id', 'source_action'} + if source_action == 'set-mode': + expected_fields.add('mode') + elif source_action == 'set-interval': + expected_fields.add('interval_seconds') + elif source_action == 'set-restart': + expected_fields.add('restart_enabled') + elif source_action == 'set-restart-delay': + expected_fields.add('restart_delay_seconds') + require_exact_control_fields(request, expected_fields) + if source_action not in MANAGED_SOURCE_LIFECYCLE_ACTIONS + MANAGED_SOURCE_SETTING_ACTIONS: + raise ValueError('unsupported managed source action') + source_id = request.get('source_id') + if not isinstance(source_id, str) or not source_id or len(source_id) > 128: + raise ValueError('managed source ID is invalid') + source = managed_source_registry(managed_sources).get(source_id) + if source is None: + raise ValueError('managed source ID is unknown') + if source_action not in managed_source_allowed_actions(source): + raise ValueError('managed source action is unavailable for this source') + if _pipeline_lifecycle_change_is_blocked(source, source_action, managed_sources): + raise RuntimeError('managed pipeline lifecycle change is refused while scanner sources are running') + + outcome = 'completed' + try: + if source_action == 'start': + success = source.start(force=True) + elif source_action == 'stop': + success = source.stop() + elif source_action == 'restart': + success = source.restart_now() + elif source_action == 'pause': + success = source.pause() + elif source_action == 'resume': + success = source.resume() + elif source_action == 'once': + if source.is_running() and not source.stop(timeout=10): + success = False + else: + set_mode(source, 'once') + success = source.start(force=True) + elif source_action == 'set-mode': + mode = request.get('mode') + if mode not in ('loop', 'once', 'repeat'): + raise ValueError('managed source mode is invalid') + if isinstance(source, ManagedKeychecks) and mode == 'loop': + raise ValueError('managed source mode is unavailable for this source') + if source.is_running(): + raise RuntimeError('managed source mode change requires a stopped source') + set_mode(source, mode) + success = True + elif source_action == 'set-interval': + value = request.get('interval_seconds') + if ( + type(value) is not int or value < 1 + or value > MAX_MANAGED_SOURCE_DELAY_SECONDS + ): + raise ValueError('managed source interval is invalid') + source.interval = value + success = True + elif source_action == 'set-restart': + value = request.get('restart_enabled') + if type(value) is not bool: + raise ValueError('managed source restart setting is invalid') + source.restart = value + success = True + else: + value = request.get('restart_delay_seconds') + if ( + type(value) is not int or value < 1 + or value > MAX_MANAGED_SOURCE_DELAY_SECONDS + ): + raise ValueError('managed source restart delay is invalid') + source.restart_delay = value + success = True + except (ValueError, RuntimeError): + raise + except Exception as exc: + raise RuntimeError('managed source action failed') from exc + + if not success: + if source.runtime_blocked and source.desired_state == 'running': + outcome = 'dependency-blocked' + else: + raise RuntimeError('managed source action failed') + status_file = (context or {}).get('status_file') + if status_file: + try: + write_status_file(status_file, managed_sources, context) + except Exception: + pass + return { + 'source_action': source_action, + 'outcome': outcome, + 'source': source.structured_state(), + } + + +def run_dashboard_action(request, context): + require_exact_control_fields(request, {'dashboard_action'}) + dashboard_action = request.get('dashboard_action') + if dashboard_action not in ('start', 'stop', 'restart'): + raise ValueError('unsupported dashboard action') + manager = (context or {}).get('dashboard_manager') + if manager is None: + raise RuntimeError('dashboard control is unavailable') + try: + if dashboard_action == 'start': + success = manager.start(force=True) + elif dashboard_action == 'stop': + success = manager.stop() + else: + success = manager.stop() and manager.start(force=True) + except Exception as exc: + raise RuntimeError('dashboard action failed') from exc + snapshot = structured_dashboard_state(manager) + if not success and snapshot['status'] != 'blocked': + raise RuntimeError('dashboard action failed') + return { + 'dashboard_action': dashboard_action, + 'outcome': 'dependency-blocked' if snapshot['status'] == 'blocked' else 'completed', + 'dashboard': snapshot, + } + + +def run_managed_source_log_tail(request, managed_sources): + require_exact_control_fields(request, {'source_id', 'line_count'}) + source_id = request.get('source_id') + line_count = request.get('line_count') + if not isinstance(source_id, str) or not source_id or len(source_id) > 128: + raise ValueError('managed source ID is invalid') + if type(line_count) is not int or not 1 <= line_count <= MAX_LOG_TAIL_LINES: + raise ValueError('managed source log line count is invalid') + source = managed_source_registry(managed_sources).get(source_id) + if source is None: + raise ValueError('managed source ID is unknown') + try: + lines = source.tail_log_lines(line_count) + except Exception as exc: + raise RuntimeError('managed source log tail failed') from exc + if type(lines) is not list or any(type(line) is not str for line in lines): + raise RuntimeError('managed source log tail failed') + lines = lines[-line_count:] + + selected = [] + encoded_bytes = 2 + for line in reversed(lines): + item_bytes = len(json.dumps(line, ensure_ascii=True).encode('utf-8')) + separator_bytes = 1 if selected else 0 + if encoded_bytes + separator_bytes + item_bytes > MAX_LOG_TAIL_BYTES: + break + selected.append(line) + encoded_bytes += separator_bytes + item_bytes + selected.reverse() + return { + 'source_id': source_id, + 'line_count': len(selected), + 'lines': selected, + 'response_truncated': len(selected) != len(lines), + } + + +def run_control_command(command, managed_sources, context, lock=None): + command = (command or '').strip() + if not command: + return {'success': True, 'output': ''} + if command in ('__status__', 'status-raw'): + guard = lock or nullcontext() + with guard: + poll_managed_sources(managed_sources, context) + return {'success': True, 'output': '\n'.join(build_runtime_table_lines(managed_sources, context=context)) + '\n'} + + if command.lower() in ('quit', 'exit', 'q'): + return {'success': True, 'output': 'Detached from background supervisor. Use --stop-background to stop it.\n'} + + output = StringIO() + guard = lock or nullcontext() + with guard: + with redirect_stdout(output): + keep_running = handle_command(command, managed_sources, context) + if not keep_running: + print('Ignored quit/exit for background supervisor. Use --stop-background to stop it.') + status_file = context.get('status_file') if context else None + write_status_file(status_file, managed_sources, context) + failed = bool((context or {}).pop('_command_failed', False)) + return {'success': not failed, 'output': output.getvalue()} + + +class nullcontext: + def __enter__(self): + return None + + def __exit__(self, exc_type, exc, tb): + return False + + +class SupervisorControlHandler(socketserver.StreamRequestHandler): + def handle(self): + self.connection.settimeout(5.0) + try: + raw = self.rfile.readline(MAX_CONTROL_REQUEST_BYTES + 1) + except OSError: + raw = b'' + if len(raw) > MAX_CONTROL_REQUEST_BYTES or not raw.endswith(b'\n'): + response = self.server.error_response('invalid or oversized control request') + else: + try: + request = json.loads(raw.decode('utf-8')) + response = self.server.run_request(request) + except (UnicodeDecodeError, json.JSONDecodeError): + response = self.server.error_response('invalid JSON control request') + except Exception as exc: + response = self.server.error_response( + f'control request failed: {type(exc).__name__}: {exc}' + ) + signal_shutdown = bool(response.pop('_signal_shutdown', False)) if isinstance(response, dict) else False + encoded = json.dumps(response, ensure_ascii=True, default=str, separators=(',', ':')).encode('utf-8') + b'\n' + if len(encoded) > MAX_CONTROL_RESPONSE_BYTES: + encoded = json.dumps(self.server.error_response('control response is too large'), separators=(',', ':')).encode('utf-8') + b'\n' + try: + self.wfile.write(encoded) + self.wfile.flush() + except (BrokenPipeError, ConnectionAbortedError, ConnectionResetError): + pass + finally: + if signal_shutdown: + shutdown_event = self.server.context.get('shutdown_event') + if shutdown_event is not None: + shutdown_event.set() + + +class SupervisorControlServer(socketserver.TCPServer): + allow_reuse_address = False + request_queue_size = 16 + + def server_bind(self): + if os.name == 'nt' and hasattr(socket, 'SO_EXCLUSIVEADDRUSE'): + self.socket.setsockopt(socket.SOL_SOCKET, socket.SO_EXCLUSIVEADDRUSE, 1) + return super().server_bind() + + def __init__(self, server_address, managed_sources, context, lock, instance_id, token): + super().__init__(server_address, SupervisorControlHandler) + self.managed_sources = managed_sources + self.context = context + self.lock = lock + self.instance_id = str(instance_id) + self.token = str(token) + self._intake_selector = selectors.DefaultSelector() + self._pending_lock = threading.Lock() + self._pending = {} + self._intake_stopping = threading.Event() + self._worker_slots = threading.BoundedSemaphore(MAX_CONTROL_WORKERS) + self._worker_executor = ThreadPoolExecutor(max_workers=MAX_CONTROL_WORKERS, thread_name_prefix='supervisor-control') + self._active_workers = 0 + self._intake_thread = threading.Thread( + target=self._control_intake_loop, + name='supervisor-control-intake', + daemon=True, + ) + self._intake_thread.start() + + @property + def pending_control_connections(self): + with self._pending_lock: + return len(self._pending) + + @property + def active_control_workers(self): + with self._pending_lock: + return self._active_workers + + @staticmethod + def _close_control_socket(request): + try: + request.shutdown(socket.SHUT_RDWR) + except OSError: + pass + try: + request.close() + except OSError: + pass + + def _remove_pending_locked(self, request): + self._pending.pop(request, None) + try: + self._intake_selector.unregister(request) + except (KeyError, OSError, ValueError): + pass + + def process_request(self, request, client_address): + request.setblocking(False) + evicted = None + with self._pending_lock: + if len(self._pending) >= MAX_CONTROL_PENDING_SOCKETS: + evicted = next(iter(self._pending)) + self._remove_pending_locked(evicted) + self._pending[request] = { + 'address': client_address, + 'buffer': bytearray(), + 'deadline': time.monotonic() + CONTROL_READ_TIMEOUT_SEC, + } + try: + self._intake_selector.register(request, selectors.EVENT_READ) + except BaseException: + self._pending.pop(request, None) + self._close_control_socket(request) + raise + if evicted is not None: + self._close_control_socket(evicted) + + def _take_pending(self, request): + with self._pending_lock: + state = self._pending.get(request) + if state is not None: + self._remove_pending_locked(request) + return state + + def _send_control_response(self, request, response): + signal_shutdown = bool(response.pop('_signal_shutdown', False)) if isinstance(response, dict) else False + encoded = json.dumps(response, ensure_ascii=True, default=str, separators=(',', ':')).encode('utf-8') + b'\n' + if len(encoded) > MAX_CONTROL_RESPONSE_BYTES: + encoded = json.dumps(self.error_response('control response is too large'), separators=(',', ':')).encode('utf-8') + b'\n' + try: + request.setblocking(True) + request.settimeout(1.0) + request.sendall(encoded) + except OSError: + pass + finally: + self._close_control_socket(request) + if signal_shutdown: + shutdown_event = self.context.get('shutdown_event') + if shutdown_event is not None: + shutdown_event.set() + + def _run_authenticated_request(self, request, value): + try: + try: + response = self.run_request(value) + except Exception as exc: + if value.get('action') in ( + 'runtime-snapshot', 'managed-source-action', 'dashboard-action', + 'managed-source-log-tail', + ): + response = self.error_response('control request failed') + else: + response = self.error_response( + f'control request failed: {type(exc).__name__}: {exc}' + ) + self._send_control_response(request, response) + finally: + with self._pending_lock: + self._active_workers -= 1 + self._worker_slots.release() + + def _dispatch_control_request(self, request, raw): + try: + value = json.loads(raw.decode('utf-8')) + except (UnicodeDecodeError, json.JSONDecodeError): + self._send_control_response(request, self.error_response('invalid JSON control request')) + return + if not authenticate_request(value, self.instance_id, self.token): + self._send_control_response(request, self.error_response('authentication failed')) + return + if not self._worker_slots.acquire(blocking=False): + self._send_control_response(request, self.error_response('control worker limit reached')) + return + with self._pending_lock: + self._active_workers += 1 + try: + self._worker_executor.submit(self._run_authenticated_request, request, value) + except BaseException: + with self._pending_lock: + self._active_workers -= 1 + self._worker_slots.release() + self._close_control_socket(request) + raise + + def _read_pending_control(self, request): + with self._pending_lock: + state = self._pending.get(request) + if state is None: + return + try: + chunk = request.recv(min(65536, MAX_CONTROL_REQUEST_BYTES + 1 - len(state['buffer']))) + except BlockingIOError: + return + except OSError: + chunk = b'' + if not chunk: + self._take_pending(request) + self._close_control_socket(request) + return + state['buffer'].extend(chunk) + newline = state['buffer'].find(b'\n') + if newline < 0 and len(state['buffer']) <= MAX_CONTROL_REQUEST_BYTES: + return + self._take_pending(request) + if newline < 0 or newline + 1 > MAX_CONTROL_REQUEST_BYTES: + self._send_control_response(request, self.error_response('invalid or oversized control request')) + return + self._dispatch_control_request(request, bytes(state['buffer'][:newline])) + + def _control_intake_loop(self): + while not self._intake_stopping.is_set(): + with self._pending_lock: + has_pending = bool(self._pending) + if not has_pending: + self._intake_stopping.wait(0.05) + continue + try: + events = self._intake_selector.select(0.05) + except (OSError, ValueError): + break + for key, _ in events: + self._read_pending_control(key.fileobj) + now = time.monotonic() + with self._pending_lock: + expired = [request for request, state in self._pending.items() if state['deadline'] <= now] + for request in expired: + self._remove_pending_locked(request) + for request in expired: + self._close_control_socket(request) + + def server_close(self): + self._intake_stopping.set() + with self._pending_lock: + pending = list(self._pending) + for request in pending: + self._remove_pending_locked(request) + for request in pending: + self._close_control_socket(request) + if self._intake_thread.is_alive(): + self._intake_thread.join(timeout=2) + try: + self._intake_selector.close() + except OSError: + pass + self._worker_executor.shutdown(wait=True, cancel_futures=False) + super().server_close() + + def response(self, ok, result=None, error=None): + value = {'schema': CONTROL_SCHEMA, 'instance_id': self.instance_id, 'ok': bool(ok)} + if ok: + value['result'] = result + else: + value['error'] = str(error or 'request failed') + return value + + def error_response(self, error): + return self.response(False, error=error) + + def run_request(self, request): + if not authenticate_request(request, self.instance_id, self.token): + return self.error_response('authentication failed') + action = str(request.get('action') or '') + if action == 'handshake': + guard = self.lock or nullcontext() + with guard: + dashboard = self.context.get('dashboard_manager') + authority = self.context.get('authority') or {} + return self.response(True, { + 'instance_id': self.instance_id, + 'pid': os.getpid(), + 'manages_postgres': bool(self.context.get('with_postgres')), + 'activation_state': lifecycle_phase(self.context), + 'config_sha256': authority.get('config_sha256', ''), + 'supervisor_sha256': authority.get('supervisor_sha256', ''), + 'code_manifest_sha256': authority.get('code_manifest_sha256', ''), + 'canonical_dsn_sha256': self.context.get('canonical_dsn_sha256', ''), + 'dashboard': dashboard.snapshot() if dashboard else {'status': 'disabled', 'healthy': False}, + }) + if action == 'activation': + guard = self.lock or nullcontext() + with guard: + return self.response(True, { + 'activation_state': lifecycle_phase(self.context), + }) + if action == 'activate': + guard = self.lock or nullcontext() + with guard: + if shutdown_checkpoint(self.context): + return self.error_response('supervisor shutdown is already requested') + if lifecycle_phase(self.context) == PHASE_ACTIVE: + return self.response(True, {'activation_state': PHASE_ACTIVE}) + if lifecycle_phase(self.context) != PHASE_ACTIVATING: + return self.error_response(f'supervisor cannot activate from {lifecycle_phase(self.context)}') + if not check_runtime_authority(self.context, trigger_shutdown=False): + self.context['runtime_failed'] = True + detail = runtime_authority_error(self.context) or 'runtime authority drifted before activation' + shutdown_event = self.context.get('shutdown_event') + if shutdown_event is not None: + shutdown_event.set() + return self.error_response(detail) + callback = self.context.get('activation_callback') + try: + if callback is not None: + callback() + except BaseException as exc: + self.context['runtime_failed'] = True + self.context['start_gate_open'] = False + shutdown_event = self.context.get('shutdown_event') + if shutdown_event is not None: + shutdown_event.set() + return self.error_response(f'activation metadata update failed: {exc}') + if shutdown_checkpoint(self.context): + return self.error_response('supervisor shutdown is already requested') + if lifecycle_phase(self.context) != PHASE_ACTIVE: + self.context['lifecycle_phase'] = PHASE_ACTIVE + self.context['activation_state'] = PHASE_ACTIVE + self.context['start_gate_open'] = True + activation_event = self.context.get('activation_event') + if activation_event is not None: + activation_event.set() + return self.response(True, {'activation_state': PHASE_ACTIVE}) + if action == 'shutdown': + guard = self.lock or nullcontext() + with guard: + authority_error = shutdown_authority_error( + self.context, self.instance_id, self.token, self.server_address[:2], + ) + if authority_error: + return self.error_response(authority_error) + if bool(request.get('with_postgres')) and not self.context.get('with_postgres'): + return self.error_response('supervisor does not manage PostgreSQL') + if self.context.get('shutdown_event') is None: + return self.error_response('shutdown event is unavailable') + try: + begin_stopping(self.context) + except Exception as exc: + return self.error_response(f'unable to enter STOPPING phase: {exc}') + return self.response(True, 'coordinated shutdown requested') + command_shutdown = action == 'command' and str(request.get('command') or '').strip().lower() == 'shutdown' + command_status = action == 'command' and str(request.get('command') or '').strip().lower() in ('status', 's', 'status-raw', '__status__') + typed_mutation = action in ('managed-source-action', 'dashboard-action') + hold_status = lifecycle_phase(self.context) in (PHASE_STOPPING, PHASE_FAILED_HOLD) and ( + action in ('snapshot', 'runtime-snapshot', 'managed-source-log-tail', 'status') + or command_status + ) + if lifecycle_phase(self.context) != PHASE_ACTIVE and not command_shutdown and not hold_status: + return self.error_response(f'supervisor runtime is {lifecycle_phase(self.context)}; request is refused') + if ( + lifecycle_phase(self.context) == PHASE_ACTIVE + and not command_shutdown and not typed_mutation + and not check_runtime_authority(self.context) + ): + if action in ( + 'runtime-snapshot', 'managed-source-action', 'dashboard-action', + 'managed-source-log-tail', + ): + return self.error_response('runtime authority drifted') + return self.error_response(self.context.get('authority_drift') or 'runtime authority drifted') + if action == 'snapshot': + guard = self.lock or nullcontext() + with guard: + poll_managed_sources(self.managed_sources, self.context) + controller = self.context.get('postgres_controller') + dashboard = self.context.get('dashboard_manager') + return self.response(True, { + 'activation_state': lifecycle_phase(self.context), + 'signature': table_signature(self.managed_sources), + 'discovery_producers': [ + producer.structured_state() + for producer in self.context.get('discovery_producers', ()) + ], + 'table': '\n'.join(build_runtime_table_lines(self.managed_sources, context=self.context)) + '\n', + 'postgres': controller.snapshot() if controller else {'state': PostgresState.DISABLED.value, 'ready': True}, + 'dashboard': dashboard.snapshot() if dashboard else {'status': 'disabled', 'healthy': False}, + }) + if action == 'runtime-snapshot': + try: + require_exact_control_fields(request, set()) + except ValueError as exc: + return self.error_response(str(exc)) + guard = self.lock or nullcontext() + with guard: + poll_managed_sources(self.managed_sources, self.context) + return self.response(True, structured_runtime_snapshot( + self.managed_sources, self.context, + )) + if action == 'managed-source-action': + guard = self.lock or nullcontext() + with guard: + if lifecycle_phase(self.context) != PHASE_ACTIVE: + return self.error_response( + f'supervisor runtime is {lifecycle_phase(self.context)}; request is refused' + ) + if not check_runtime_authority(self.context): + return self.error_response('runtime authority drifted') + try: + result = run_managed_source_action( + request, self.managed_sources, self.context, + ) + except (ValueError, RuntimeError) as exc: + return self.error_response(str(exc)) + return self.response(True, result) + if action == 'managed-source-log-tail': + guard = self.lock or nullcontext() + with guard: + try: + result = run_managed_source_log_tail(request, self.managed_sources) + except (ValueError, RuntimeError) as exc: + return self.error_response(str(exc)) + return self.response(True, result) + if action == 'dashboard-action': + guard = self.lock or nullcontext() + with guard: + if lifecycle_phase(self.context) != PHASE_ACTIVE: + return self.error_response( + f'supervisor runtime is {lifecycle_phase(self.context)}; request is refused' + ) + if not check_runtime_authority(self.context): + return self.error_response('runtime authority drifted') + try: + result = run_dashboard_action(request, self.context) + except (ValueError, RuntimeError) as exc: + return self.error_response(str(exc)) + return self.response(True, result) + if action == 'status': + outcome = run_control_command('__status__', self.managed_sources, self.context, self.lock) + return self.response(True, outcome['output']) + if action == 'command': + command = str(request.get('command') or '').strip() + if command.lower() == 'shutdown': + guard = self.lock or nullcontext() + with guard: + authority_error = shutdown_authority_error( + self.context, self.instance_id, self.token, self.server_address[:2], + ) + if authority_error: + return self.error_response(authority_error) + try: + begin_stopping(self.context) + except Exception as exc: + return self.error_response(f'unable to enter STOPPING phase: {exc}') + return self.response(True, 'coordinated shutdown requested\n') + outcome = run_control_command(command, self.managed_sources, self.context, self.lock) + if not outcome['success']: + return self.error_response(outcome['output'].strip() or 'supervisor command failed') + return self.response(True, outcome['output']) + return self.error_response('unknown control action') + + +def start_control_server(supervisor_config, managed_sources, context, lock, instance_id, token, start_thread=True): + host, port = control_address(supervisor_config) + server = SupervisorControlServer((host, port), managed_sources, context, lock, instance_id, token) + if start_thread: + threading.Thread(target=server.serve_forever, daemon=True).start() + print(f'Control server listening on {server.server_address[0]}:{server.server_address[1]}') + return server + + +def process_running_status(pid): + if not pid: + return False + if os.name == 'nt': + try: + process_query_limited_information = 0x1000 + handle = _SUPERVISOR_OPEN_PROCESS(process_query_limited_information, False, int(pid)) + if not handle: + error = ctypes.get_last_error() + return False if error in (87, 1168) else None + try: + exit_code = wintypes.DWORD() + ok = _SUPERVISOR_GET_EXIT_CODE_PROCESS(handle, ctypes.byref(exit_code)) + return (exit_code.value == 259) if ok else None + finally: + _SUPERVISOR_CLOSE_HANDLE(handle) + except Exception: + return None + try: + os.kill(pid, 0) + return True + except PermissionError: + return None + except OSError: + return False + + +def is_pid_running(pid): + return process_running_status(pid) is True + + +def background_child_command(args, config_path, launch_nonce, instance_file, expected_authority=None): + app_dir = os.path.dirname(os.path.abspath(__file__)) + command = [ + sys.executable, + '-I', + '-S', + '-B', + os.path.join(app_dir, 'runtime_bootstrap.py'), + 'supervisor', + '--', + '--runtime-bootstrap-entrypoint', + os.path.abspath(__file__), + '--config', + config_path, + '--non-interactive', + '--background-child', + '--no-clear', + ] + command.extend(['--instance-file', instance_file, f'--launch-nonce={launch_nonce}']) + if expected_authority: + command.extend([ + '--expected-config-sha256', expected_authority['config_sha256'], + '--expected-supervisor-sha256', expected_authority['supervisor_sha256'], + '--expected-code-manifest-sha256', expected_authority['code_manifest_sha256'], + ]) + if args.sources: + command.extend(['--sources', args.sources]) + if args.once: + command.append('--once') + if args.autostart: + command.append('--autostart') + if args.status_interval: + command.extend(['--status-interval', str(args.status_interval)]) + if args.dashboard: + command.append('--dashboard') + if args.no_dashboard: + command.append('--no-dashboard') + if args.with_postgres: + command.append('--with-postgres') + return command + + +def command_for_log(command): + hidden_value_flags = { + '--launch-nonce', '--expected-config-sha256', '--expected-supervisor-sha256', + '--expected-code-manifest-sha256', '--token', '--docker-token', '--password', + '--api-key', '--secret', '--db-url', '--database-url', '--scanner-db-url', + } + output = [] + hide_next = False + for value in command: + text = str(value) + if hide_next: + output.append('') + hide_next = False + else: + flag = text.split('=', 1)[0].lower() + if flag in hidden_value_flags and '=' in text: + output.append(text.split('=', 1)[0] + '=') + else: + output.append(text) + hide_next = flag in hidden_value_flags + return output + + +def inspect_existing_instance(instance_file, config_path): + if not os.path.exists(instance_file): + return 'absent', None, None + try: + metadata = load_instance_metadata(instance_file) + except (OSError, ValueError) as exc: + try: + raw = read_private_json(instance_file) + if ( + raw.get('schema') == 2 + and raw.get('instance_id') + and os.path.normcase(os.path.abspath(raw.get('instance_file') or '')) + == os.path.normcase(os.path.abspath(instance_file)) + and exact_process_identity_state( + raw.get('pid'), raw.get('process_creation_time'), raw.get('executable'), + ) in ('dead', 'reused') + ): + return 'stale', raw, f'legacy manifest is invalid and exact process identity is dead: {exc}' + except (OSError, TypeError, ValueError): + pass + return 'invalid', None, str(exc) + try: + process = verify_instance_process(metadata, os.path.abspath(__file__), config_path) + except (OSError, ValueError) as exc: + try: + drift_process = verify_instance_process( + metadata, + os.path.abspath(__file__), + config_path, + allow_config_drift=True, + ) + except (OSError, ValueError): + drift_process = None + if drift_process is not None: + drift_process.close() + return 'config_drift', metadata, 'live supervisor permits authenticated shutdown only' + identity_state = exact_process_identity_state( + metadata.get('pid'), metadata.get('process_creation_time'), metadata.get('executable'), + ) + if identity_state in ('dead', 'reused'): + return 'stale', metadata, str(exc) + return 'identity_mismatch', metadata, f'{identity_state}: {exc}' + try: + handshake = send_control_request(metadata, 'handshake', timeout=3) + if not isinstance(handshake, dict) or handshake.get('instance_id') != metadata['instance_id']: + raise ValueError('invalid authenticated handshake') + return 'verified', metadata, process + except (OSError, ValueError, RuntimeError) as exc: + process.close() + return 'unreachable', metadata, str(exc) + + +def _terminate_unpublished_child(process): + if process.poll() is not None: + return True + try: + process.terminate() + except OSError: + pass + try: + process.wait(timeout=5) + except subprocess.TimeoutExpired: + process.kill() + process.wait() + return process.poll() is not None + + +def _rollback_background_start( + process, + candidate, + instance_file, + launch_nonce, + with_postgres, + lock_path=None, + activation_attempted=False, + shutdown_timeout=180, +): + exact_candidate = ( + isinstance(candidate, dict) + and candidate.get('launch_nonce') == launch_nonce + and candidate.get('pid') == process.pid + ) + if process.poll() is None and exact_candidate: + try: + send_control_request(candidate, 'activation', timeout=2) + except (OSError, ValueError, RuntimeError): + pass + + if not exact_candidate: + _terminate_unpublished_child(process) + outcome = 'unpublished child was reaped by its retained parent handle' + else: + if process.poll() is None: + try: + send_control_request(candidate, 'shutdown', timeout=5, with_postgres=bool(with_postgres)) + except (OSError, ValueError, RuntimeError): + pass + try: + process.wait(timeout=max(5.0, float(shutdown_timeout))) + except subprocess.TimeoutExpired: + pass + if process.poll() is None: + return 'activation authority was uncertain; the exact child was left running because authenticated shutdown was not confirmed' + outcome = 'exact candidate exited after authenticated coordinated shutdown' + + if ( + process.poll() == 0 + and exact_candidate + ): + remove_instance_if_matches(instance_file, candidate.get('instance_id'), lock_path=lock_path) + remove_shutdown_receipt(instance_file, candidate.get('instance_id')) + return outcome + + +def start_background(args, config_path, results_dir, supervisor_config, expected_authority=None): + expected_authority = expected_authority or runtime_authority(config_path) + runtime_authority( + config_path, + expected_config_sha256=expected_authority['config_sha256'], + expected_supervisor_sha256=expected_authority['supervisor_sha256'], + expected_code_manifest_sha256=expected_authority['code_manifest_sha256'], + code_manifest=expected_authority['code_manifest'], + ) + instance_file, log_file, status_file, legacy_pid_file = background_paths( + config_path, results_dir, supervisor_config, args.instance_file, args.pid_file, + ) + singleton_lock_file = background_lock_path(config_path, results_dir, supervisor_config) + legacy_instance_file = legacy_log_instance_path(config_path, results_dir, supervisor_config) + if ( + os.path.normcase(os.path.abspath(legacy_instance_file)) != os.path.normcase(os.path.abspath(instance_file)) + and os.path.exists(legacy_instance_file) + ): + raise SystemExit( + f'Refusing background launch while legacy log-directory instance metadata exists: {legacy_instance_file}' + ) + existing_state, existing_metadata, existing_detail = inspect_existing_instance(instance_file, config_path) + if existing_state == 'verified': + existing_process = existing_detail + existing_process.close() + raise SystemExit(f'Refusing duplicate verified background supervisor instance: {existing_metadata["instance_id"]}') + if existing_state not in ('absent', 'stale'): + raise SystemExit(f'Refusing to replace {existing_state} supervisor metadata at {instance_file}: {existing_detail}') + legacy_pid = read_pid_file(legacy_pid_file) + if legacy_pid: + legacy_status = process_running_status(legacy_pid) + label = 'running' if legacy_status is True else 'not running' if legacy_status is False else 'unknown' + raise SystemExit( + f'Refusing background launch while legacy status-only PID metadata exists ' + f'({legacy_pid}, {label}): {legacy_pid_file}' + ) + + rotate_log_if_needed(log_file, supervisor_config.get('log_max_mb', 64), supervisor_config.get('log_keep', 5)) + log_handle = open_private_append(log_file) + log_handle.write(f'\n=== background supervisor launch {now_iso()} ===\n') + launch_nonce = secrets.token_urlsafe(32) + command = background_child_command(args, config_path, launch_nonce, instance_file, expected_authority) + log_handle.write('command: ' + ' '.join(command_for_log(command)) + '\n') + log_handle.write(f'instance_file: {instance_file}\n') + log_handle.write(f'status_file: {status_file}\n') + dashboard_enabled, dashboard_host, dashboard_port, dashboard_url = dashboard_settings(supervisor_config) + if dashboard_enabled: + log_handle.write(f'dashboard_url: {dashboard_url}\n') + log_handle.flush() + + creationflags = 0 + if os.name == 'nt': + creationflags = DETACHED_PROCESS | subprocess.CREATE_NEW_PROCESS_GROUP | CREATE_NO_WINDOW + try: + process = subprocess.Popen( + command, + cwd=os.path.dirname(config_path), + stdin=subprocess.DEVNULL, + stdout=log_handle, + stderr=subprocess.STDOUT, + env={ + **os.environ, + 'PYTHONUNBUFFERED': '1', + 'TRUF_SUPERVISOR_LAUNCH_NONCE': launch_nonce, + }, + creationflags=creationflags, + close_fds=True, + ) + finally: + log_handle.close() + timeout = max(1.0, float(supervisor_config.get('background_start_timeout_sec', 20) or 20)) + deadline = time.monotonic() + timeout + last_error = 'instance metadata was not written' + metadata = None + candidate = None + handshake = None + activation_attempted = False + while time.monotonic() < deadline: + if process.poll() is not None: + last_error = f'child exited with code {process.returncode}' + break + try: + candidate = load_instance_metadata(instance_file) + if candidate['launch_nonce'] != launch_nonce or candidate['pid'] != process.pid: + raise InstanceMetadataError('launch nonce or child PID mismatch') + if candidate['manages_postgres'] != bool(args.with_postgres): + raise InstanceMetadataError('PostgreSQL management mode mismatch') + if candidate.get('instance_file') != os.path.normcase(os.path.realpath(os.path.abspath(instance_file))): + raise InstanceMetadataError('child instance metadata path mismatch') + for key in ('config_sha256', 'supervisor_sha256', 'code_manifest_sha256'): + if candidate.get(key) != expected_authority[key]: + raise InstanceMetadataError(f'parent/child {key} authority mismatch') + retained = verify_instance_process(candidate, os.path.abspath(__file__), config_path) + try: + handshake = send_control_request(candidate, 'handshake', timeout=2) + finally: + retained.close() + if not isinstance(handshake, dict) or handshake.get('instance_id') != candidate['instance_id']: + raise InstanceMetadataError('authenticated startup handshake mismatch') + if candidate.get('activation_state') != PHASE_ACTIVATING or handshake.get('activation_state') != PHASE_ACTIVATING: + raise InstanceMetadataError('new background child was not published in ACTIVATING state') + for key in ('config_sha256', 'supervisor_sha256', 'code_manifest_sha256'): + if handshake.get(key) != expected_authority[key]: + raise InstanceMetadataError(f'authenticated handshake {key} mismatch') + if handshake.get('canonical_dsn_sha256', '') != candidate.get('canonical_dsn_sha256', ''): + raise InstanceMetadataError('authenticated handshake DSN authority mismatch') + metadata = candidate + break + except (OSError, ValueError, RuntimeError) as exc: + last_error = str(exc) + time.sleep(0.05) + if metadata is not None: + activation_attempted = True + activation_timeout = max( + 5.0, + float(supervisor_config.get('background_activation_timeout_sec', timeout) or timeout), + ) + activation_deadline = time.monotonic() + activation_timeout + try: + result = send_control_request(metadata, 'activate', timeout=2) + if not isinstance(result, dict) or result.get('activation_state') != PHASE_ACTIVE: + last_error = 'authenticated activate response was invalid' + except (OSError, ValueError, RuntimeError) as exc: + last_error = f'activate response was unavailable: {exc}' + confirmed = False + while time.monotonic() < activation_deadline and process.poll() is None: + try: + handshake = send_control_request(metadata, 'handshake', timeout=2) + if ( + isinstance(handshake, dict) + and handshake.get('instance_id') == metadata['instance_id'] + and handshake.get('activation_state') == PHASE_ACTIVE + ): + metadata = load_instance_metadata(instance_file) + if metadata.get('activation_state') != PHASE_ACTIVE: + raise InstanceMetadataError('active handshake disagreed with private metadata') + for key in ('config_sha256', 'supervisor_sha256', 'code_manifest_sha256'): + if metadata.get(key) != expected_authority[key] or handshake.get(key) != expected_authority[key]: + raise InstanceMetadataError(f'parent activation {key} recheck failed') + if handshake.get('canonical_dsn_sha256', '') != metadata.get('canonical_dsn_sha256', ''): + raise InstanceMetadataError('parent activation DSN authority recheck failed') + retained = verify_instance_process(metadata, os.path.abspath(__file__), config_path) + retained.close() + confirmed = True + break + last_error = 'activation handshake remained ACTIVATING' + except (OSError, ValueError, RuntimeError) as exc: + last_error = f'activation confirmation failed: {exc}' + time.sleep(0.05) + if not confirmed: + metadata = None + + if metadata is None: + rollback = _rollback_background_start( + process, + candidate, + instance_file, + launch_nonce, + args.with_postgres, + lock_path=singleton_lock_file, + activation_attempted=activation_attempted, + shutdown_timeout=supervisor_config.get('background_shutdown_timeout_sec', 180), + ) + raise SystemExit( + f'Background supervisor startup failed within {timeout:g}s: {last_error}; {rollback}. Check {log_file}' + ) + + print(f'Started background supervisor: PID {metadata["pid"]}') + print(f'Instance file: {instance_file}') + print(f'Log file: {log_file}') + print(f'Status file: {status_file}') + if dashboard_enabled: + dashboard = handshake.get('dashboard') if isinstance(handshake, dict) else {} + print(f'Dashboard: {dashboard.get("status", "pending")} ({dashboard.get("detail", "health pending")})') + print(f'Dashboard URL: {dashboard_url}') + + +def stop_background(config_path, results_dir, supervisor_config, explicit_instance_file=None, explicit_pid_file=None, with_postgres=False): + instance_file, log_file, status_file, legacy_pid_file = background_paths( + config_path, results_dir, supervisor_config, explicit_instance_file, explicit_pid_file, + ) + singleton_lock_file = background_lock_path(config_path, results_dir, supervisor_config) + try: + metadata = load_instance_metadata(instance_file) + process = verify_instance_process( + metadata, + os.path.abspath(__file__), + config_path, + allow_config_drift=True, + allow_code_drift=True, + ) + except (OSError, ValueError) as exc: + print(f'Refusing background stop without verified instance metadata: {exc}') + legacy_instance_file = legacy_log_instance_path(config_path, results_dir, supervisor_config) + if os.path.exists(legacy_instance_file) and os.path.abspath(legacy_instance_file) != os.path.abspath(instance_file): + print(f'Legacy instance metadata at {legacy_instance_file} is detection-only and does not grant control authority.') + legacy_pid = read_pid_file(legacy_pid_file) + if legacy_pid: + print(f'Legacy PID {legacy_pid} is status-only and will not be terminated.') + return False + try: + if with_postgres and not metadata['manages_postgres']: + print('Refusing --with-postgres stop because this verified supervisor does not manage PostgreSQL.') + return False + result = send_control_request(metadata, 'shutdown', timeout=5, with_postgres=bool(with_postgres)) + print(str(result)) + timeout = max(5.0, float(supervisor_config.get('background_shutdown_timeout_sec', 180) or 180)) + if not process.wait(timeout): + print(f'Coordinated shutdown did not finish within {timeout:g}s; no PID-based termination was attempted.') + return False + try: + exit_code = process.exit_code() + except (AttributeError, OSError, ValueError): + exit_code = None + try: + receipt_code = load_shutdown_receipt(instance_file, metadata['instance_id']) + except (OSError, ValueError): + receipt_code = None + if exit_code is not None and receipt_code is not None and exit_code != receipt_code: + print('Background supervisor exit status disagreed with its authenticated shutdown receipt.') + return False + confirmed_code = exit_code if exit_code is not None else receipt_code + if confirmed_code != 0: + label = 'unavailable' if confirmed_code is None else str(confirmed_code) + print(f'Background supervisor coordinated shutdown exit status was {label}; metadata was retained.') + return False + if not remove_instance_if_matches( + instance_file, + metadata['instance_id'], + lock_path=singleton_lock_file, + ): + print('Background supervisor exited successfully, but matching metadata could not be removed safely.') + return False + remove_shutdown_receipt(instance_file, metadata['instance_id']) + print('Background supervisor exited after coordinated shutdown.') + return True + except (OSError, ValueError, RuntimeError) as exc: + print(f'Authenticated coordinated shutdown failed: {exc}') + return False + finally: + process.close() + + +def background_status(config_path, results_dir, supervisor_config, explicit_instance_file=None, explicit_pid_file=None): + instance_file, log_file, status_file, legacy_pid_file = background_paths( + config_path, results_dir, supervisor_config, explicit_instance_file, explicit_pid_file, + ) + dashboard_enabled, dashboard_host, dashboard_port, dashboard_url = dashboard_settings(supervisor_config) + print(f'Instance file: {instance_file}') + print(f'Log file: {log_file}') + print(f'Status file: {status_file}') + if dashboard_enabled: + print(f'Dashboard URL: {dashboard_url}') + verified = False + try: + metadata = load_instance_metadata(instance_file) + process = verify_instance_process(metadata, os.path.abspath(__file__), config_path) + try: + handshake = send_control_request(metadata, 'handshake', timeout=3) + print(f'PID: {metadata["pid"]}') + print(f'Instance: {metadata["instance_id"]}') + print('Status: verified and authenticated') + dashboard = handshake.get('dashboard') if isinstance(handshake, dict) else None + if isinstance(dashboard, dict): + print( + f"Dashboard status: {dashboard.get('status', 'unknown')} " + f"healthy={dashboard.get('healthy', False)} detail={dashboard.get('detail', '')}" + ) + verified = True + finally: + process.close() + except (OSError, ValueError, RuntimeError) as exc: + print(f'Status: no verified authenticated instance ({exc})') + legacy_instance_file = legacy_log_instance_path(config_path, results_dir, supervisor_config) + if os.path.exists(legacy_instance_file) and os.path.abspath(legacy_instance_file) != os.path.abspath(instance_file): + print(f'Legacy instance metadata (detection-only): {legacy_instance_file}') + legacy_pid = read_pid_file(legacy_pid_file) + if legacy_pid: + legacy_status = process_running_status(legacy_pid) + label = 'running' if legacy_status is True else 'not running' if legacy_status is False else 'unknown' + print(f'Legacy PID (status-only): {legacy_pid} ({label})') + if os.path.exists(status_file): + print('\nLast supervisor table:') + try: + with open(status_file, 'r', encoding='utf-8') as f: + print(f.read().rstrip()) + except OSError as e: + print(f'Unable to read status file: {e}') + return verified + + +def verified_instance_for_control(instance_file, config_path): + metadata = load_instance_metadata(instance_file) + process = verify_instance_process(metadata, os.path.abspath(__file__), config_path) + try: + send_control_request(metadata, 'handshake', timeout=3) + return metadata, process + except BaseException: + process.close() + raise + + +def send_background_command(config_path, results_dir, supervisor_config, command, explicit_instance_file=None, explicit_pid_file=None): + instance_file, log_file, status_file, legacy_pid_file = background_paths( + config_path, results_dir, supervisor_config, explicit_instance_file, explicit_pid_file, + ) + try: + metadata, process = verified_instance_for_control(instance_file, config_path) + except (OSError, ValueError, RuntimeError) as exc: + print(f'Refusing command without verified authenticated instance metadata: {exc}') + return False + try: + if str(command).strip().lower() == 'shutdown': + response = send_control_request(metadata, 'shutdown') + else: + response = send_control_command(metadata, command) + print(str(response).rstrip()) + return True + except (OSError, ValueError, RuntimeError) as exc: + print(f'Authenticated control command failed: {exc}') + return False + finally: + process.close() + + +def attach_background(config_path, results_dir, supervisor_config, explicit_instance_file=None, explicit_pid_file=None): + instance_file, log_file, status_file, legacy_pid_file = background_paths( + config_path, results_dir, supervisor_config, explicit_instance_file, explicit_pid_file, + ) + try: + metadata, process = verified_instance_for_control(instance_file, config_path) + initial_snapshot = get_control_snapshot(metadata) + except (OSError, ValueError, RuntimeError) as exc: + print(f'Unable to verify and authenticate background supervisor: {exc}') + return False + finally: + if 'process' in locals(): + process.close() + + print((initial_snapshot.get('table') or '').rstrip()) + print('\nAttached to background supervisor. Type `quit` to detach, `help` for commands, `watch` for live view.') + poll_sec = float(supervisor_config.get('attach_poll_sec', supervisor_config.get('poll_sec', 0.5)) or 0.5) + while True: + try: + command = input('attach> ').strip() + except (EOFError, KeyboardInterrupt): + print() + break + if not command: + continue + if command.lower() in ('quit', 'exit', 'q'): + break + if command.lower() in ('watch', 'w'): + try: + watch_remote(metadata, poll_sec) + except (OSError, ValueError, RuntimeError) as e: + print(f'Connection failed: {e}') + break + continue + try: + if command.lower() == 'shutdown': + response = send_control_request(metadata, 'shutdown') + else: + response = send_control_command(metadata, command) + print(str(response).rstrip()) + except (OSError, ValueError, RuntimeError) as e: + print(f'Connection failed: {e}') + break + print('Detached. Background supervisor is still running.') + return True + + +def parse_args(): + parser = argparse.ArgumentParser(description='Supervisor for running configured scanner sources in one console.') + parser.add_argument('--config', default='config.yaml') + parser.add_argument('--sources', help='Comma-separated source list. Defaults to enabled sources from config.yaml') + parser.add_argument('--once', action='store_true', help='Force --once for every managed source') + actions = parser.add_mutually_exclusive_group() + actions.add_argument('--dry-run', action='store_true', help='Print child commands and exit') + parser.add_argument('--status-interval', type=int, help='Seconds between status redraws') + parser.add_argument('--no-clear', action='store_true', help='Do not clear console before status redraws') + parser.add_argument('--autostart', action='store_true', help='Start selected sources immediately') + parser.add_argument('--non-interactive', action='store_true', help='Run status loop without command prompt') + parser.add_argument('--background-child', action='store_true', help=argparse.SUPPRESS) + parser.add_argument('--launch-nonce', help=argparse.SUPPRESS) + parser.add_argument('--expected-config-sha256', help=argparse.SUPPRESS) + parser.add_argument('--expected-supervisor-sha256', help=argparse.SUPPRESS) + parser.add_argument('--expected-code-manifest-sha256', help=argparse.SUPPRESS) + parser.add_argument('--runtime-bootstrap-entrypoint', help=argparse.SUPPRESS) + actions.add_argument('--background', action='store_true', help='Launch supervisor in the background and exit') + actions.add_argument('--attach', action='store_true', help='Attach a foreground command prompt to running background supervisor') + actions.add_argument('--stop-background', action='store_true', help='Request authenticated coordinated background shutdown') + actions.add_argument('--background-status', action='store_true', help='Show verified background supervisor status') + actions.add_argument('--cmd', help='Send one supervisor prompt command to a running background supervisor') + parser.add_argument('--instance-file', help='Override instance metadata path; its parent must be the configured private control directory') + parser.add_argument('--pid-file', help='Legacy status-only PID file path; never grants control authority') + parser.add_argument('--dashboard', action='store_true', help='Launch dashboard regardless of config supervisor.dashboard') + parser.add_argument('--no-dashboard', action='store_true', help='Disable dashboard launch') + parser.add_argument('--with-postgres', action='store_true', help='Let the child controller manage verified bundled PostgreSQL') + return parser.parse_args() + + +def main(): + args = parse_args() + command_parts = [] + if getattr(args, 'cmd', None): + try: + command_parts = shlex.split(args.cmd) + except ValueError: + command_parts = ['invalid'] + read_only = ( + getattr(args, 'dry_run', False) + or getattr(args, 'background_status', False) + or (getattr(args, 'cmd', None) and not command_is_mutating(command_parts)) + ) + control_requested = ( + getattr(args, 'stop_background', False) + or getattr(args, 'background_status', False) + or getattr(args, 'attach', False) + or bool(getattr(args, 'cmd', None)) + ) + runtime_launch = getattr(args, 'background', False) or not control_requested + if not read_only and runtime_launch and not getattr(args, 'with_postgres', False): + raise SystemExit('Supervisor unmanaged PostgreSQL mutation is retired; use --with-postgres.') + if not read_only and os.getenv(RUNTIME_BOOTSTRAP_ENV) != RUNTIME_BOOTSTRAP_VALUE: + raise SystemExit('Mutating supervisor runtime requires the canonical runtime bootstrap.') + bootstrap_entrypoint = getattr(args, 'runtime_bootstrap_entrypoint', None) + if getattr(args, 'background_child', False): + expected_entrypoint = os.path.normcase(os.path.realpath(os.path.abspath(__file__))) + actual_entrypoint = ( + os.path.normcase(os.path.realpath(os.path.abspath(bootstrap_entrypoint))) + if bootstrap_entrypoint and os.path.isabs(bootstrap_entrypoint) + else '' + ) + if actual_entrypoint != expected_entrypoint: + raise SystemExit('Background child requires its canonical bootstrap entrypoint binding.') + config_path = os.path.abspath(args.config) + config_hash_before = sha256_file(config_path) + managed_runtime = ( + not read_only + and runtime_launch + and getattr(args, 'with_postgres', False) + ) + if managed_runtime: + documents = validate_managed_runtime_startup(config_path) + validated_config = documents.config + validated_config_sha256 = documents.config_sha256 + documents = None + if validated_config_sha256 != config_hash_before: + raise SystemExit( + 'Supervisor config changed while it was being validated; ' + 'retry with a stable private config file.' + ) + validated_depth = validate_docker_depth_config( + validated_config, + managed_postgres=bool(getattr(args, 'with_postgres', False)), + ) + validated_config = apply_path_config(validated_depth.config, config_path) + config, project_dir, results_dir, supervisor_config, selected_sources = ( + supervisor_runtime_from_config( + config_path, + validated_config, + getattr(args, 'sources', None), + ) + ) + loaded_config_sha256 = sha256_file(config_path) + if loaded_config_sha256 != validated_config_sha256: + raise SystemExit( + 'Supervisor config changed while it was being validated; ' + 'retry with a stable private config file.' + ) + else: + config, project_dir, results_dir, supervisor_config, selected_sources = load_supervisor_runtime( + config_path, + getattr(args, 'sources', None), + managed_postgres=bool(getattr(args, 'with_postgres', False)), + ) + loaded_config_sha256 = sha256_file(config_path) + if loaded_config_sha256 != config_hash_before: + raise SystemExit( + 'Supervisor config changed while it was being loaded; ' + 'retry with a stable private config file.' + ) + try: + preflight_lifecycle_paths( + config_path, + config, + authority_profile='server' if managed_runtime else 'full', + ) + except (OSError, ValueError) as exc: + raise SystemExit(str(exc)) from exc + if getattr(args, 'dashboard', False): + dashboard_config = supervisor_config.get('dashboard') if isinstance(supervisor_config.get('dashboard'), dict) else {} + supervisor_config['dashboard'] = {**dashboard_config, 'enabled': True} + if getattr(args, 'no_dashboard', False): + dashboard_config = supervisor_config.get('dashboard') if isinstance(supervisor_config.get('dashboard'), dict) else {} + supervisor_config['dashboard'] = {**dashboard_config, 'enabled': False} + dashboard_settings(supervisor_config) + authority = runtime_authority( + config_path, + trufflehog_path=(config.get('global') or {}).get('trufflehog_path'), + policy_paths=configured_policy_paths(config), + include_trufflehog=not managed_runtime, + expected_config_sha256=loaded_config_sha256, + ) + + if getattr(args, 'background', False): + if not getattr(args, 'with_postgres', False): + raise SystemExit('Supervisor unmanaged PostgreSQL mutation is retired; use --with-postgres.') + start_background(args, config_path, results_dir, supervisor_config, authority) + return 0 + if getattr(args, 'stop_background', False): + stopped = stop_background( + config_path, + results_dir, + supervisor_config, + getattr(args, 'instance_file', None), + getattr(args, 'pid_file', None), + with_postgres=getattr(args, 'with_postgres', False), + ) + return 0 if stopped else 1 + if getattr(args, 'background_status', False): + return 0 if background_status(config_path, results_dir, supervisor_config, getattr(args, 'instance_file', None), getattr(args, 'pid_file', None)) else 1 + if getattr(args, 'attach', False): + return 0 if attach_background(config_path, results_dir, supervisor_config, getattr(args, 'instance_file', None), getattr(args, 'pid_file', None)) else 1 + if getattr(args, 'cmd', None): + return 0 if send_background_command(config_path, results_dir, supervisor_config, args.cmd, getattr(args, 'instance_file', None), getattr(args, 'pid_file', None)) else 1 + + keychecks_config = keychecks_config_for(config_path, config) + + def create_sources( + gate, authority_check=None, start_gate=None, child_environments=None, + ): + child_environments = child_environments or {} + items = [] + for source in selected_sources: + producer = source in DISCOVERY_PRODUCER_SOURCES + source_class = ManagedDiscoveryProducer if producer else ManagedSource + items.append(source_class( + source, + config_path, + project_dir, + results_dir, + supervisor_config, + (config.get('sources') or {}).get(source, {}), + getattr(args, 'once', False), + gate, + authority_check, + start_gate, + child_environments.get(DISCOVERY_PRODUCER_ROLE if producer else 'scanner'), + )) + if should_manage_keychecks(config, supervisor_config, getattr(args, 'sources', None)): + keychecks = ManagedKeychecks( + config_path, + project_dir, + results_dir, + supervisor_config, + keychecks_config, + getattr(args, 'once', False), + gate, + authority_check, + start_gate, + child_environments.get('keycheck'), + ) + items.append(keychecks) + return [source for source in items if source.enabled] + + if getattr(args, 'dry_run', False): + dry_sources = create_sources(DependencyGate(ready=True)) + if not dry_sources: + raise SystemExit('No enabled supervisor sources selected') + for source in dry_sources: + print(f'[{source.source}]') + print('command:', ' '.join(source.build_command())) + print('log:', source.log_path) + print('state:', source.state_path if source.use_per_source_state else '(config default)') + print('once:', source.once, 'repeat:', source.repeat, 'restart:', source.restart, 'interval:', source.interval) + print() + return 0 + + if not getattr(args, 'with_postgres', False): + raise SystemExit('Supervisor unmanaged PostgreSQL mutation is retired; use --with-postgres.') + + load_postgres_env( + config_path, + config.get('global') or {}, + enforce_canonical=getattr(args, 'with_postgres', False), + ) + if getattr(args, 'with_postgres', False): + managed_database_url = canonical_database_url() + if not managed_database_url: + raise RuntimeError('managed PostgreSQL lifecycle requires a canonical DSN') + else: + try: + managed_database_url = canonical_database_url() + except ValueError: + managed_database_url = '' + if not managed_database_url: + raise RuntimeError('supervised mutation requires one caller-selected canonical PostgreSQL DSN') + + instance_file, supervisor_log_file, status_file, _ = background_paths( + config_path, results_dir, supervisor_config, getattr(args, 'instance_file', None), getattr(args, 'pid_file', None), + ) + singleton_lock_file = background_lock_path(config_path, results_dir, supervisor_config) + background_child = bool(getattr(args, 'background_child', False)) + background_log_writer = None + try: + cluster_lock = ClusterAuthorityLock(config, endpoint_dsn=managed_database_url).acquire() + except (BlockingIOError, OSError) as exc: + raise SystemExit(f'Refusing duplicate lifecycle-owning supervisor or maintenance authority for this PostgreSQL cluster: {exc}') from exc + try: + instance_lock = SupervisorInstanceLock(instance_file, lock_path=singleton_lock_file).acquire() + except InstanceLockError as exc: + cluster_lock.release() + raise SystemExit(f'Refusing duplicate lifecycle-owning supervisor: {exc}') from exc + + if background_child: + try: + background_log_writer = install_bounded_background_output( + supervisor_log_file, + max(1, int(supervisor_config.get('log_max_mb', 64) or 64)) * 1024 * 1024, + supervisor_config.get('log_keep', 5), + ) + except BaseException: + instance_lock.release() + cluster_lock.release() + raise + + refresh_sec = getattr(args, 'status_interval', None) or int(supervisor_config.get('refresh_sec', DEFAULT_REFRESH_SEC) or DEFAULT_REFRESH_SEC) + heartbeat_sec = int(supervisor_config.get('heartbeat_sec', 60) or 0) + poll_sec = float(supervisor_config.get('poll_sec', 0.2) or 0.2) + autostart = getattr(args, 'autostart', False) or bool_value(supervisor_config.get('autostart'), False) + interactive = not getattr(args, 'non_interactive', False) and bool_value(supervisor_config.get('interactive'), True) + control_lock = threading.RLock() + managed_sources = [] + shutdown_event = threading.Event() + shutdown_retry_event = threading.Event() + activation_event = threading.Event() + context = { + 'config_path': config_path, + 'selected_sources_arg': getattr(args, 'sources', None), + 'global_force_once': getattr(args, 'once', False), + 'config': config, + 'project_dir': project_dir, + 'results_dir': results_dir, + 'supervisor_config': supervisor_config, + 'selected_sources': selected_sources, + 'managed_sources': managed_sources, + 'discovery_producers': [], + 'scanner_sources': [], + 'keychecks_config': keychecks_config, + 'queue_dir': (config.get('global') or {}).get('queue_dir') or os.path.join(os.path.dirname(results_dir), 'queues'), + 'dashboard_manager': None, + 'with_postgres': bool(getattr(args, 'with_postgres', False)), + 'control_lock': control_lock, + 'dependency_gate': None, + 'source_dependency_gate': None, + 'postgres_controller': None, + 'background_child': background_child, + 'shutdown_requested': False, + 'runtime_failed': False, + 'shutdown_event': shutdown_event, + 'shutdown_retry_event': shutdown_retry_event, + 'status_file': status_file, + 'authority': authority, + 'instance_file': instance_file, + 'activation_state': PHASE_ACTIVATING, + 'lifecycle_phase': PHASE_ACTIVATING, + 'start_gate_open': False, + 'activation_event': activation_event, + 'activation_lock': threading.RLock(), + } + control_server = None + control_server_started = False + instance_metadata = None + dashboard_manager = None + postgres_controller = None + runtime_activated = False + exit_code = 0 + sigterm_installed = False + previous_sigterm_handler = None + + def request_shutdown(_signum, _frame): + context['shutdown_requested'] = True + + try: + if os.name == 'posix': + previous_sigterm_handler = signal.signal(signal.SIGTERM, request_shutdown) + sigterm_installed = True + if os.path.exists(instance_file): + try: + existing = load_instance_metadata(instance_file) + except (OSError, ValueError) as exc: + existing = read_private_json(instance_file) + identity_state = exact_process_identity_state( + existing.get('pid'), existing.get('process_creation_time'), + existing.get('executable'), + ) + if ( + existing.get('schema') != 2 + or not existing.get('instance_id') + or os.path.normcase(os.path.abspath(existing.get('instance_file') or '')) + != os.path.normcase(os.path.abspath(instance_file)) + or identity_state not in ('dead', 'reused') + ): + raise InstanceLockError( + f'refusing invalid supervisor metadata with {identity_state} identity: {exc}' + ) from exc + durable_unlink(instance_file) + remove_shutdown_receipt(instance_file, existing['instance_id']) + existing = None + if existing is None: + pass + elif ( + os.path.normcase(os.path.abspath(existing.get('instance_file') or '')) + != os.path.normcase(os.path.abspath(instance_file)) + ): + raise InstanceLockError('refusing supervisor metadata bound to another instance path') + elif exact_process_identity_state( + existing.get('pid'), existing.get('process_creation_time'), existing.get('executable'), + ) not in ('dead', 'reused'): + raise InstanceLockError('refusing to replace supervisor metadata whose exact process identity is live or unknown') + elif not remove_instance_if_matches(instance_file, existing['instance_id'], instance_lock=instance_lock): + raise InstanceLockError('unable to remove matching stale supervisor metadata under the lifetime lock') + elif existing is not None: + remove_shutdown_receipt(instance_file, existing['instance_id']) + + if background_child: + launch_nonce = getattr(args, 'launch_nonce', None) + if not launch_nonce: + raise SystemExit('Background child requires a launch nonce') + expected = { + 'config_sha256': getattr(args, 'expected_config_sha256', None), + 'supervisor_sha256': getattr(args, 'expected_supervisor_sha256', None), + 'code_manifest_sha256': getattr(args, 'expected_code_manifest_sha256', None), + } + if not all(expected.values()): + raise SystemExit('Background child requires immutable parent authority hashes') + authority = runtime_authority( + config_path, + trufflehog_path=(config.get('global') or {}).get('trufflehog_path'), + policy_paths=configured_policy_paths(config), + include_trufflehog=not managed_runtime, + expected_config_sha256=expected['config_sha256'], + expected_supervisor_sha256=expected['supervisor_sha256'], + expected_code_manifest_sha256=expected['code_manifest_sha256'], + ) + context['authority'] = authority + else: + launch_nonce = secrets.token_urlsafe(32) + + # All fallible non-lifecycle initialization is complete before ACTIVE. + context['canonical_dsn_sha256'] = dsn_sha256(managed_database_url) + dependency_gate = DependencyGate( + ready=not getattr(args, 'with_postgres', False), + database_url=managed_database_url, + ) + context['dependency_gate'] = dependency_gate + source_dependency_gate = DependencyGate( + ready=False, + database_url=managed_database_url, + ) + context['source_dependency_gate'] = source_dependency_gate + authority_check = lambda: check_runtime_authority(context, require_private_acl=True) + start_gate = lambda: lifecycle_start_allowed(context) + postgres_controller = controller_from_config(config, supervisor_config) if getattr(args, 'with_postgres', False) else None + if postgres_controller is not None: + postgres_controller.authority_check = lambda: lifecycle_start_allowed(context) and authority_check() + context['postgres_controller'] = postgres_controller + + instance_id = secrets.token_urlsafe(24) + token = secrets.token_urlsafe(48) + child_metadata = { + 'instance_file': instance_file, + 'instance_id': instance_id, + 'token': token, + 'config_sha256': authority['config_sha256'], + 'supervisor_sha256': authority['supervisor_sha256'], + 'code_manifest_sha256': authority['code_manifest_sha256'], + 'canonical_dsn_sha256': context['canonical_dsn_sha256'], + } + child_environments = { + 'scanner': supervised_child_environment(child_metadata, managed_database_url, 'scanner'), + DISCOVERY_PRODUCER_ROLE: supervised_child_environment( + child_metadata, managed_database_url, DISCOVERY_PRODUCER_ROLE, + ), + 'keycheck': supervised_child_environment(child_metadata, managed_database_url, 'keycheck'), + 'dashboard': supervised_child_environment(child_metadata, managed_database_url, 'dashboard'), + 'result-ingester': supervised_child_environment(child_metadata, managed_database_url, 'result-ingester'), + 'jsonl-projector': supervised_child_environment(child_metadata, managed_database_url, 'jsonl-projector'), + 'worker-api': supervised_child_environment(child_metadata, managed_database_url, 'worker-api'), + 'janitor': supervised_child_environment(child_metadata, '', 'janitor'), + 'docker-shadow': supervised_child_environment(child_metadata, managed_database_url, 'docker-shadow'), + } + pipeline_workers = [ + ManagedPipelineWorker( + 'result-ingester', config_path, project_dir, results_dir, + supervisor_config, supervisor_config.get('result_ingester') or {}, + dependency_gate, authority_check, start_gate, + child_environments['result-ingester'], + ), + ManagedPipelineWorker( + 'jsonl-projector', config_path, project_dir, results_dir, + supervisor_config, supervisor_config.get('jsonl_projector') or {}, + dependency_gate, authority_check, start_gate, + child_environments['jsonl-projector'], + ), + ManagedPipelineWorker( + 'janitor', config_path, project_dir, results_dir, + supervisor_config, supervisor_config.get('janitor') or {}, + None, authority_check, start_gate, child_environments['janitor'], + ), + ManagedPipelineWorker( + 'worker-api', config_path, project_dir, results_dir, + supervisor_config, supervisor_config.get('worker_api') or {}, + dependency_gate, authority_check, start_gate, + child_environments['worker-api'], + ), + ] + managed_sources.extend(worker for worker in pipeline_workers if worker.enabled) + docker_shadow = ManagedDockerShadow( + config_path, project_dir, results_dir, supervisor_config, + supervisor_config.get('docker_shadow') or {}, dependency_gate, + authority_check, start_gate, child_environments['docker-shadow'], + ) + if docker_shadow.enabled: + managed_sources.append(docker_shadow) + configured_sources = create_sources( + source_dependency_gate, + authority_check, + start_gate, + child_environments, + ) + managed_sources.extend(configured_sources) + context['discovery_producers'] = [ + source for source in configured_sources + if isinstance(source, ManagedDiscoveryProducer) + ] + context['scanner_sources'] = [ + source for source in configured_sources + if not isinstance(source, ManagedDiscoveryProducer) + and not isinstance(source, ManagedKeychecks) + ] + if not managed_sources: + raise SystemExit('No enabled supervisor sources selected') + dashboard_manager = ManagedDashboard( + config_path, + project_dir, + supervisor_config, + results_dir, + context['queue_dir'], + dependency_gate=dependency_gate, + authority_check=authority_check, + start_gate=start_gate, + child_environment=child_environments['dashboard'], + ) + context['dashboard_manager'] = dashboard_manager + + control_server = start_control_server( + supervisor_config, + managed_sources, + context, + control_lock, + instance_id, + token, + start_thread=False, + ) + host, port = control_server.server_address[:2] + instance_metadata = build_instance_metadata( + launch_nonce, + os.path.abspath(__file__), + config_path, + host, + port, + getattr(args, 'with_postgres', False), + instance_id=instance_id, + token=token, + activation_state=PHASE_ACTIVATING, + expected_config_sha256=authority['config_sha256'], + expected_supervisor_sha256=authority['supervisor_sha256'], + code_manifest=authority['code_manifest'], + expected_code_manifest_sha256=authority['code_manifest_sha256'], + canonical_dsn_sha256=context['canonical_dsn_sha256'], + lifecycle_mode='background' if background_child else 'foreground', + instance_file=instance_file, + ) + write_instance_metadata(instance_file, instance_metadata) + context['instance_id'] = instance_id + + def activate_runtime(): + nonlocal runtime_activated + if shutdown_checkpoint(context): + return + current = load_instance_metadata(instance_file) + if current['instance_id'] != instance_id or current['activation_state'] != PHASE_ACTIVATING: + raise InstanceMetadataError('activation metadata no longer names the exact ACTIVATING candidate') + for key in ('config_sha256', 'supervisor_sha256', 'code_manifest_sha256'): + if current.get(key) != authority[key]: + raise InstanceMetadataError(f'activation metadata {key} mismatch') + detail = runtime_authority_error(context) + if detail: + raise InstanceMetadataError(detail) + if shutdown_checkpoint(context): + return + # ACTIVE may reach disk even if publication or event delivery raises. + runtime_activated = True + update_instance_activation(instance_file, instance_id, PHASE_ACTIVE) + if shutdown_checkpoint(context): + return + context['lifecycle_phase'] = PHASE_ACTIVE + context['activation_state'] = PHASE_ACTIVE + context['start_gate_open'] = True + activation_event.set() + + context['activation_callback'] = activate_runtime + threading.Thread(target=control_server.serve_forever, daemon=True).start() + control_server_started = True + print(f'Control server listening on {host}:{port}') + + if background_child: + activation_timeout = max( + 5.0, + float(supervisor_config.get( + 'background_activation_timeout_sec', + max(30.0, float(supervisor_config.get('background_start_timeout_sec', 20) or 20) * 2), + )), + ) + activation_deadline = time.monotonic() + activation_timeout + while time.monotonic() < activation_deadline: + with control_lock: + if shutdown_checkpoint(context) or activation_event.is_set(): + break + time.sleep(0.02) + with control_lock: + if not shutdown_checkpoint(context) and not activation_event.is_set(): + exit_code = 1 + print(f'Background child activation timed out after {activation_timeout:g}s; no lifecycle action was taken.') + else: + with control_lock: + activate_runtime() + + with control_lock: + shutdown_checkpoint(context) + if runtime_activated and lifecycle_start_allowed(context): + keychecks_auto_started = False + with control_lock: + for source in managed_sources: + if shutdown_checkpoint(context) or not lifecycle_start_allowed(context): + break + if isinstance(source, ManagedPipelineWorker): + source.start(force=True) + if not shutdown_checkpoint(context) and lifecycle_start_allowed(context): + dashboard_manager.start(record_intent=False) + if not autostart and bool_value(keychecks_config.get('autostart'), False): + for source in managed_sources: + if shutdown_checkpoint(context) or not lifecycle_start_allowed(context): + break + if source.source == 'keychecks': + source.start(force=True) + keychecks_auto_started = True + + if interactive: + interactive_loop( + managed_sources, + autostart=autostart, + clear=not getattr(args, 'no_clear', False), + context=context, + poll_sec=poll_sec, + ) + elif not autostart and not keychecks_auto_started and not background_child: + with control_lock: + if not shutdown_checkpoint(context): + exit_code = 1 + print('Non-interactive mode requires --autostart or supervisor.autostart=true') + else: + if autostart: + with control_lock: + for source in autostart_sources(managed_sources): + if shutdown_checkpoint(context) or not lifecycle_start_allowed(context): + break + source.start(force=True) + non_interactive_loop( + managed_sources, + refresh_sec=refresh_sec, + status_file=status_file, + poll_sec=poll_sec, + context=context, + lock=control_lock, + stay_alive=background_child, + status_heartbeat_sec=heartbeat_sec, + ) + except KeyboardInterrupt: + try: + print('\nStopping child processes...') + except BaseException: + exit_code = exit_code or 1 + except SystemExit as exc: + exit_code = exit_code or (0 if exc.code is None else int(exc.code) if isinstance(exc.code, int) else 1) + if exc.code and not isinstance(exc.code, int): + print(str(exc.code)) + except BaseException as exc: + exit_code = 1 + print(f'Supervisor runtime failed closed: {type(exc).__name__}: {exc}') + finally: + context['shutdown_requested'] = True + context['authority_release_safe'] = False + try: + with control_lock: + if any( + getattr(source, 'startup_cleanup_pending', False) is True + or (not background_child and source.status == 'failed') + for source in managed_sources + ): + context['runtime_failed'] = True + if runtime_activated: + if not coordinated_shutdown(managed_sources, context): + exit_code = exit_code or 1 + else: + begin_stopping(context) + if dashboard_manager is not None: + dashboard_manager.close() + context['authority_release_safe'] = ( + postgres_controller is None or bool(postgres_controller.close(wait=False)) + ) + except BaseException as exc: + exit_code = exit_code or 1 + context['authority_release_safe'] = False + try: + with control_lock: + enter_failed_hold(context, f'initial shutdown failed: {type(exc).__name__}: {exc}') + print(f'Coordinated shutdown failed closed: {exc}') + except BaseException: + pass + while not context.get('authority_release_safe', False): + exit_code = exit_code or 1 + retain_unsafe_authority(managed_sources, context) + try: + try: + if control_server and control_server_started: + control_server.shutdown() + finally: + if control_server: + control_server.server_close() + except BaseException: + exit_code = exit_code or 1 + try: + print('Supervisor stopped.') + sys.stdout.flush() + sys.stderr.flush() + except BaseException: + exit_code = exit_code or 1 + try: + if background_log_writer is not None: + background_log_writer.close() + except BaseException: + exit_code = exit_code or 1 + exit_code = exit_code or int(bool(context.get('runtime_failed') or context.get('authority_drift'))) + if instance_metadata: + try: + write_shutdown_receipt(instance_file, instance_metadata['instance_id'], exit_code) + except BaseException: + exit_code = exit_code or 1 + # The waiting stopper (or locked stale reconciliation) removes both files. + instance_lock.release() + cluster_lock.release() + if sigterm_installed: + signal.signal(signal.SIGTERM, previous_sigterm_handler) + return exit_code + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/app/supervisor_instance.py b/app/supervisor_instance.py new file mode 100644 index 0000000..46c20cd --- /dev/null +++ b/app/supervisor_instance.py @@ -0,0 +1,378 @@ +import hmac +import ipaddress +import os +import re +import secrets +from datetime import datetime, timezone + +from process_identity import current_process_identity, verify_retained_process +from lifecycle_authority import ( + LIFECYCLE_PHASES, + PHASE_ACTIVATING, + LifecycleAuthorityError, + build_code_manifest, + code_manifest_sha256, + normalize_code_manifest, + verify_code_manifest, + verify_supervisor_command_line, +) +from runtime_security import ( + atomic_write_private_json, + canonical_path, + durable_unlink, + PrivateFileLock, + read_private_json, + sha256_file, + write_private_json_exclusive, +) + + +INSTANCE_SCHEMA = 2 +CONTROL_SCHEMA = 1 +SHUTDOWN_RECEIPT_SCHEMA = 1 + + +class InstanceMetadataError(ValueError): + pass + + +class InstanceLockError(OSError): + pass + + +def instance_lock_path(instance_path): + return os.path.splitext(os.path.abspath(instance_path))[0] + '.lock' + + +def shutdown_receipt_path(instance_path): + return os.path.splitext(os.path.abspath(instance_path))[0] + '.exit.json' + + +class SupervisorInstanceLock: + """Lifetime singleton lock keyed by the canonical private instance path.""" + + def __init__(self, instance_path, lock_path=None): + self.instance_path = canonical_path(instance_path) + self.path = os.path.normcase(os.path.abspath(lock_path or instance_lock_path(instance_path))) + self._lock = PrivateFileLock(self.path) + self._acquired = False + + @property + def acquired(self): + return self._acquired + + def acquire(self): + if self._acquired: + return self + try: + self._lock.acquire() + except BlockingIOError as exc: + raise InstanceLockError('another supervisor owns this private runtime control lock') from exc + except OSError as exc: + raise InstanceLockError(str(exc)) from exc + self._acquired = True + return self + + def release(self): + if not self._acquired: + return + self._acquired = False + self._lock.release() + + def __enter__(self): + return self.acquire() + + def __exit__(self, exc_type, value, traceback): + self.release() + + +def utc_now_iso(): + return datetime.now(timezone.utc).isoformat(timespec='seconds') + + +def is_loopback_host(host): + try: + return ipaddress.ip_address(str(host)).is_loopback + except ValueError: + return str(host).strip().lower() == 'localhost' + + +def build_instance_metadata( + launch_nonce, + supervisor_path, + config_path, + control_host, + control_port, + manages_postgres, + identity=None, + instance_id=None, + token=None, + activation_state=PHASE_ACTIVATING, + expected_config_sha256=None, + expected_supervisor_sha256=None, + code_manifest=None, + expected_code_manifest_sha256=None, + canonical_dsn_sha256='', + lifecycle_mode='background', + instance_file=None, +): + if not launch_nonce: + raise InstanceMetadataError('launch nonce is required') + if not is_loopback_host(control_host): + raise InstanceMetadataError('control endpoint must be loopback-only') + identity = identity or current_process_identity() + supervisor_path = canonical_path(supervisor_path) + config_path = canonical_path(config_path) + actual_supervisor_sha256 = sha256_file(supervisor_path) + actual_config_sha256 = sha256_file(config_path) + if expected_supervisor_sha256 and not hmac.compare_digest(actual_supervisor_sha256, str(expected_supervisor_sha256)): + raise InstanceMetadataError('supervisor script changed after parent authority capture') + if expected_config_sha256 and not hmac.compare_digest(actual_config_sha256, str(expected_config_sha256)): + raise InstanceMetadataError('supervisor config changed after parent authority capture') + code_manifest = normalize_code_manifest(code_manifest or build_code_manifest()) + manifest_sha256 = code_manifest_sha256(code_manifest) + if expected_code_manifest_sha256 and not hmac.compare_digest(manifest_sha256, str(expected_code_manifest_sha256)): + raise InstanceMetadataError('code manifest changed after parent authority capture') + lifecycle_mode = str(lifecycle_mode) + if lifecycle_mode not in ('background', 'foreground'): + raise InstanceMetadataError('invalid supervisor lifecycle mode') + return { + 'schema': INSTANCE_SCHEMA, + 'instance_id': instance_id or secrets.token_urlsafe(24), + 'token': token or secrets.token_urlsafe(48), + 'launch_nonce': str(launch_nonce), + 'pid': int(identity.pid), + 'process_creation_time': str(identity.creation_time), + 'executable': canonical_path(identity.executable), + 'supervisor_path': supervisor_path, + 'supervisor_sha256': actual_supervisor_sha256, + 'config_path': config_path, + 'config_sha256': actual_config_sha256, + 'code_manifest': code_manifest, + 'code_manifest_sha256': manifest_sha256, + 'canonical_dsn_sha256': str(canonical_dsn_sha256 or ''), + 'instance_file': canonical_path(instance_file) if instance_file else '', + 'lifecycle_mode': lifecycle_mode, + 'control': {'host': str(control_host), 'port': int(control_port)}, + 'startup_time': utc_now_iso(), + 'manages_postgres': bool(manages_postgres), + 'activation_state': str(activation_state), + 'private_file_ready': True, + } + + +def validate_instance_metadata(value): + if not isinstance(value, dict) or value.get('schema') != INSTANCE_SCHEMA: + raise InstanceMetadataError('unsupported supervisor instance schema') + required_strings = ( + 'instance_id', 'token', 'launch_nonce', 'process_creation_time', 'executable', + 'supervisor_path', 'supervisor_sha256', 'config_path', 'config_sha256', 'startup_time', + 'code_manifest_sha256', 'lifecycle_mode', + ) + for key in required_strings: + if not isinstance(value.get(key), str) or not value[key]: + raise InstanceMetadataError(f'invalid supervisor instance field: {key}') + for key in ('supervisor_sha256', 'config_sha256', 'code_manifest_sha256'): + if not re.fullmatch(r'[0-9a-f]{64}', value[key]): + raise InstanceMetadataError(f'invalid supervisor instance hash: {key}') + canonical_dsn_sha256 = str(value.get('canonical_dsn_sha256') or '') + if canonical_dsn_sha256 and not re.fullmatch(r'[0-9a-f]{64}', canonical_dsn_sha256): + raise InstanceMetadataError('invalid supervisor instance DSN authority') + try: + manifest = normalize_code_manifest(value.get('code_manifest')) + except (OSError, ValueError) as exc: + raise InstanceMetadataError(str(exc)) from exc + if not hmac.compare_digest(code_manifest_sha256(manifest), value['code_manifest_sha256']): + raise InstanceMetadataError('supervisor instance code manifest digest mismatch') + if len(value['instance_id']) > 256 or len(value['token']) < 32 or len(value['token']) > 512: + raise InstanceMetadataError('invalid supervisor instance credentials') + try: + pid = int(value.get('pid')) + except (TypeError, ValueError) as exc: + raise InstanceMetadataError('invalid supervisor instance PID') from exc + if pid <= 0: + raise InstanceMetadataError('invalid supervisor instance PID') + control = value.get('control') + if not isinstance(control, dict) or not is_loopback_host(control.get('host')): + raise InstanceMetadataError('invalid supervisor control endpoint') + try: + port = int(control.get('port')) + except (TypeError, ValueError) as exc: + raise InstanceMetadataError('invalid supervisor control port') from exc + if not 0 < port <= 65535: + raise InstanceMetadataError('invalid supervisor control port') + if value.get('private_file_ready') is not True: + raise InstanceMetadataError('supervisor instance is not marked private-file-ready') + activation_state = str(value.get('activation_state') or PHASE_ACTIVATING).upper() + if activation_state not in LIFECYCLE_PHASES: + raise InstanceMetadataError('invalid supervisor activation state') + lifecycle_mode = str(value.get('lifecycle_mode') or '') + if lifecycle_mode not in ('background', 'foreground'): + raise InstanceMetadataError('invalid supervisor lifecycle mode') + normalized = dict(value) + normalized['pid'] = pid + normalized['control'] = {'host': str(control['host']), 'port': port} + normalized['executable'] = canonical_path(value['executable']) + normalized['supervisor_path'] = canonical_path(value['supervisor_path']) + normalized['config_path'] = canonical_path(value['config_path']) + normalized['code_manifest'] = manifest + normalized['code_manifest_sha256'] = value['code_manifest_sha256'] + normalized['canonical_dsn_sha256'] = canonical_dsn_sha256 + normalized['instance_file'] = canonical_path(value['instance_file']) if value.get('instance_file') else '' + normalized['lifecycle_mode'] = lifecycle_mode + normalized['manages_postgres'] = bool(value.get('manages_postgres')) + normalized['activation_state'] = activation_state + return normalized + + +def write_instance_metadata(path, metadata): + write_private_json_exclusive(path, validate_instance_metadata(metadata)) + + +def load_instance_metadata(path): + return validate_instance_metadata(read_private_json(path)) + + +def update_instance_activation(path, instance_id, activation_state): + activation_state = str(activation_state).upper() + if activation_state not in LIFECYCLE_PHASES: + raise InstanceMetadataError('invalid supervisor activation state') + current = load_instance_metadata(path) + if not hmac.compare_digest(current['instance_id'], str(instance_id)): + raise InstanceMetadataError('supervisor activation instance mismatch') + current['activation_state'] = activation_state + atomic_write_private_json(path, validate_instance_metadata(current)) + return current + + +def remove_instance_if_matches(path, instance_id, instance_lock=None, lock_path=None): + owned_lock = None + if instance_lock is None: + try: + owned_lock = SupervisorInstanceLock(path, lock_path=lock_path).acquire() + instance_lock = owned_lock + except OSError: + return False + try: + current = load_instance_metadata(path) + except (OSError, ValueError): + if owned_lock: + owned_lock.release() + return False + if not hmac.compare_digest(current['instance_id'], str(instance_id)): + if owned_lock: + owned_lock.release() + return False + try: + before = os.stat(path, follow_symlinks=False) + confirmed = load_instance_metadata(path) + after = os.stat(path, follow_symlinks=False) + identity_before = (before.st_dev, before.st_ino, before.st_size, before.st_mtime_ns) + identity_after = (after.st_dev, after.st_ino, after.st_size, after.st_mtime_ns) + if identity_before != identity_after or not hmac.compare_digest(confirmed['instance_id'], str(instance_id)): + return False + durable_unlink(path) + return True + except (OSError, ValueError): + return False + finally: + if owned_lock: + owned_lock.release() + + +def verify_instance_process( + metadata, + supervisor_path=None, + config_path=None, + allow_config_drift=False, + allow_code_drift=False, +): + metadata = validate_instance_metadata(metadata) + if supervisor_path and metadata['supervisor_path'] != canonical_path(supervisor_path): + raise InstanceMetadataError('supervisor path does not match instance metadata') + if config_path and metadata['config_path'] != canonical_path(config_path): + raise InstanceMetadataError('config path does not match instance metadata') + if not allow_code_drift: + try: + verify_code_manifest(metadata['code_manifest'], metadata['code_manifest_sha256']) + if sha256_file(metadata['supervisor_path']) != metadata['supervisor_sha256']: + raise InstanceMetadataError('supervisor script hash does not match instance metadata') + except OSError as exc: + raise InstanceMetadataError(f'unable to recompute supervisor code authority: {exc}') from exc + except ValueError as exc: + raise InstanceMetadataError(str(exc)) from exc + try: + config_matches = sha256_file(metadata['config_path']) == metadata['config_sha256'] + except OSError as exc: + if not allow_config_drift: + raise InstanceMetadataError(f'unable to recompute supervisor config authority hash: {exc}') from exc + config_matches = False + if not config_matches and not allow_config_drift: + raise InstanceMetadataError('supervisor config hash drifted; only authenticated shutdown is allowed') + process = verify_retained_process( + metadata['pid'], + metadata['process_creation_time'], + metadata['executable'], + ) + try: + arguments = process.command_line() + if metadata['lifecycle_mode'] == 'background' and '--background-child' not in arguments: + raise InstanceMetadataError('retained Python process is not a background supervisor child') + if metadata['lifecycle_mode'] == 'foreground' and '--background-child' in arguments: + raise InstanceMetadataError('retained Python process lifecycle mode mismatch') + try: + verify_supervisor_command_line( + arguments, + metadata['supervisor_path'], + metadata['config_path'], + ) + except LifecycleAuthorityError as exc: + raise InstanceMetadataError(str(exc)) from exc + except BaseException: + process.close() + raise + return process + + +def write_shutdown_receipt(instance_path, instance_id, exit_code): + value = { + 'schema': SHUTDOWN_RECEIPT_SCHEMA, + 'instance_id': str(instance_id), + 'exit_code': int(exit_code), + 'completed_at': utc_now_iso(), + } + atomic_write_private_json(shutdown_receipt_path(instance_path), value) + + +def load_shutdown_receipt(instance_path, instance_id): + value = read_private_json(shutdown_receipt_path(instance_path)) + if value.get('schema') != SHUTDOWN_RECEIPT_SCHEMA: + raise InstanceMetadataError('unsupported supervisor shutdown receipt schema') + if not hmac.compare_digest(str(value.get('instance_id') or ''), str(instance_id)): + raise InstanceMetadataError('supervisor shutdown receipt instance mismatch') + try: + code = int(value.get('exit_code')) + except (TypeError, ValueError) as exc: + raise InstanceMetadataError('invalid supervisor shutdown receipt exit code') from exc + return code + + +def remove_shutdown_receipt(instance_path, instance_id=None): + path = shutdown_receipt_path(instance_path) + try: + if instance_id is not None: + load_shutdown_receipt(instance_path, instance_id) + durable_unlink(path) + return True + except (OSError, ValueError): + return False + + +def authenticate_request(request, instance_id, token): + if not isinstance(request, dict) or request.get('schema') != CONTROL_SCHEMA: + return False + request_instance = request.get('instance_id') + request_token = request.get('token') + if not isinstance(request_instance, str) or not isinstance(request_token, str): + return False + return hmac.compare_digest(request_instance, str(instance_id)) and hmac.compare_digest(request_token, str(token)) diff --git a/app/sync_alive_github_tokens.py b/app/sync_alive_github_tokens.py new file mode 100644 index 0000000..c05ff8d --- /dev/null +++ b/app/sync_alive_github_tokens.py @@ -0,0 +1,356 @@ +import sys + +sys.dont_write_bytecode = True + +import argparse +import os +import re +import threading + +import yaml + +from migrate_runtime_safety import require_runtime_hardening_stopped +from db_backend import database_url_from_env, is_postgres_url +from paths import apply_path_config +from postgres_runtime import load_postgres_environment +from runtime_security import ( + ClusterAuthorityLock, + canonical_path, + durable_replace, + harden_private_file, + PrivateFileLock, + private_file_ready, + require_private_directory, + require_private_file, +) + + +GITHUB_TOKEN_PREFIXES = ('ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_') +ACCEPTED_ALIVE_STATUSES = {'ALIVE', 'VALID', 'VALID_2FA'} +DOCKERHUB_TOKEN_RE = re.compile(r'^dckr_pat_[A-Za-z0-9_-]{27}$') +PROVIDER_DEFAULTS = { + 'github': { + 'alive_file': os.path.join('runtime', 'keychecks', 'github', 'githubAlive.txt'), + 'pool': 'github_main', + 'name_prefix': 'gh', + }, + 'dockerhub': { + 'alive_file': os.path.join('runtime', 'keychecks', 'dockerhub', 'dockerhubAlive.txt'), + 'pool': 'dockerhub_main', + 'name_prefix': 'dockerhub', + }, +} + + +def read_alive_tokens(path): + tokens = [] + seen = set() + with open(path, 'r', encoding='utf-8', errors='replace') as f: + for line in f: + # Status files are TSV-like: token, status, message, extra. + fields = line.rstrip('\r\n').split('\t') + token = fields[0].strip() if fields else '' + if not token or not token.startswith(GITHUB_TOKEN_PREFIXES): + continue + if any(character.isspace() for character in token): + raise ValueError('alive token input contains whitespace in a token field') + status = fields[1].strip().upper() if len(fields) > 1 else '' + if status not in ACCEPTED_ALIVE_STATUSES: + raise ValueError(f'alive token input contains an unaccepted or missing status: {status or "(missing)"}') + if token in seen: + continue + seen.add(token) + tokens.append(token) + return tokens + + +def read_alive_credentials(path, provider): + provider = str(provider or 'github').strip().lower() + if provider == 'github': + return [{'token': token} for token in read_alive_tokens(path)], 0 + if provider != 'dockerhub': + raise ValueError(f'unsupported alive credential provider: {provider}') + + credentials = [] + seen = {} + skipped_missing_username = 0 + with open(path, 'r', encoding='utf-8', errors='replace') as handle: + for line in handle: + fields = line.rstrip('\r\n').split('\t') + identity = fields[0].strip() if fields else '' + if not identity: + continue + if ':' in identity: + username, token = identity.rsplit(':', 1) + username = username.strip() + token = token.strip() + else: + username = '' + token = identity + if not DOCKERHUB_TOKEN_RE.fullmatch(token): + continue + status = fields[1].strip().upper() if len(fields) > 1 else '' + if status not in {'VALID', 'VALID_2FA'}: + raise ValueError(f'alive DockerHub input contains an unaccepted or missing status: {status or "(missing)"}') + if not username: + skipped_missing_username += 1 + continue + if len(username) > 256 or ':' in username or any(character.isspace() for character in username): + raise ValueError('alive DockerHub input contains an invalid username field') + previous = seen.get(token) + if previous is not None: + if previous.casefold() != username.casefold(): + raise ValueError('alive DockerHub input contains conflicting usernames for one token') + continue + seen[token] = username + credentials.append({'username': username, 'token': token}) + return credentials, skipped_missing_username + + +def next_name(existing_names, prefix): + pattern = re.compile(rf'^{re.escape(prefix)}_(\d+)$') + max_index = 0 + for name in existing_names: + match = pattern.match(str(name or '')) + if match: + max_index = max(max_index, int(match.group(1))) + return f'{prefix}_{max_index + 1}' + + +def _atomic_write_private_yaml(path, value): + parent = require_private_directory(os.path.dirname(os.path.abspath(path)), create=False) + payload = yaml.safe_dump(value, allow_unicode=True, sort_keys=False, width=120).encode('utf-8') + temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.tmp' + flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0) + descriptor = os.open(temporary, flags, 0o600) + try: + os.close(descriptor) + descriptor = None + harden_private_file(temporary) + with open(temporary, 'wb') as handle: + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + if not private_file_ready(temporary): + raise OSError(f'private temporary secrets ACL changed: {temporary}') + durable_replace(temporary, path) + if not private_file_ready(path): + raise OSError(f'private secrets ACL changed during publication: {path}') + finally: + if descriptor is not None: + os.close(descriptor) + try: + if os.path.exists(temporary): + os.remove(temporary) + except OSError: + pass + + +def sync_tokens( + secrets_path, + alive_path, + pool_name, + name_prefix, + apply=False, + *, + provider='github', + replace_conflicting_usernames=False, + canonical_secrets_path=None, + authority_lock=None, + stopped_verified=False, +): + if authority_lock is None or not getattr(authority_lock, 'acquired', False): + raise RuntimeError('alive-token sync requires an acquired cluster authority lock') + if stopped_verified is not True: + raise RuntimeError('alive-token sync requires verified stopped runtime proof') + if not canonical_secrets_path or canonical_path(secrets_path) != canonical_path(canonical_secrets_path): + raise RuntimeError('alive-token sync secrets path must exactly match canonical global.secrets_file') + if not os.path.exists(alive_path): + raise SystemExit(f'alive token file not found: {alive_path}') + if not os.path.exists(secrets_path): + raise SystemExit(f'secrets file not found: {secrets_path}') + + require_private_file(secrets_path) + require_private_file(alive_path) + lock = PrivateFileLock(f'{secrets_path}.sync.lock').acquire() + try: + require_private_file(secrets_path) + require_private_file(alive_path) + with open(secrets_path, 'r', encoding='utf-8') as f: + secrets = yaml.safe_load(f) or {} + + auth_pools = secrets.setdefault('auth_pools', {}) + pool = auth_pools.setdefault(pool_name, []) + if not isinstance(pool, list): + raise SystemExit(f'auth_pools.{pool_name} must be a list') + existing_before = len(pool) + + provider = str(provider or 'github').strip().lower() + if provider not in PROVIDER_DEFAULTS: + raise ValueError(f'unsupported alive credential provider: {provider}') + existing_tokens = set() + existing_entries = {} + existing_names = set() + normalized_existing = 0 + normalized_usernames = 0 + for entry in pool: + if not isinstance(entry, dict): + continue + if entry.get('name'): + existing_names.add(str(entry.get('name'))) + token = str(entry.get('token') or '') + stripped = token.strip() + if stripped != token: + normalized_existing += 1 + if apply: + entry['token'] = stripped + if stripped: + existing_tokens.add(stripped) + existing_entries.setdefault(stripped, []).append(entry) + if provider == 'dockerhub': + username = str(entry.get('username') or '') + stripped_username = username.strip() + if stripped_username != username: + normalized_usernames += 1 + if apply: + entry['username'] = stripped_username + + alive_credentials, skipped_missing_username = read_alive_credentials(alive_path, provider) + added = [] + username_filled = 0 + username_conflicts = 0 + username_replaced = 0 + for credential in alive_credentials: + token = credential['token'] + if token in existing_tokens: + if provider == 'dockerhub': + entries = existing_entries.get(token, []) + usernames = { + str(entry.get('username') or '').strip().casefold() + for entry in entries if str(entry.get('username') or '').strip() + } + expected = credential['username'].casefold() + if len(usernames) > 1 or (usernames and expected not in usernames): + if replace_conflicting_usernames: + username_replaced += 1 + if apply: + for entry in entries: + entry['username'] = credential['username'] + else: + username_conflicts += 1 + continue + if not usernames: + username_filled += 1 + if apply: + for entry in entries: + entry['username'] = credential['username'] + continue + name = next_name(existing_names, name_prefix) + existing_names.add(name) + existing_tokens.add(token) + new_entry = {'name': name} + if provider == 'dockerhub': + new_entry['username'] = credential['username'] + new_entry['token'] = token + added.append(new_entry) + + if apply and ( + added or normalized_existing or normalized_usernames + or username_filled or username_replaced + ): + pool.extend(added) + _atomic_write_private_yaml(secrets_path, secrets) + + return { + 'alive_unique': len(alive_credentials), + 'existing_before': existing_before, + 'added': len(added), + 'normalized_existing': normalized_existing, + 'normalized_usernames': normalized_usernames, + 'username_filled': username_filled, + 'username_conflicts': username_conflicts, + 'username_replaced': username_replaced, + 'skipped_missing_username': skipped_missing_username, + 'pool_after': len(pool) + (0 if apply else len(added)), + } + finally: + lock.release() + + +def load_config(path): + with open(path, 'r', encoding='utf-8') as handle: + return apply_path_config(yaml.safe_load(handle) or {}, path) + + +def parse_args(): + parser = argparse.ArgumentParser(description='Sync alive provider credentials into a private auth pool without printing them.') + parser.add_argument('--provider', choices=sorted(PROVIDER_DEFAULTS), default='github') + parser.add_argument('--secrets', help='Must exactly match global.secrets_file from --config') + parser.add_argument('--alive-file') + parser.add_argument('--pool') + parser.add_argument('--name-prefix') + parser.add_argument('--config', default=os.path.join('app', 'config.yaml')) + parser.add_argument('--dry-run', action='store_true') + parser.add_argument('--apply', action='store_true', help='Apply under verified offline maintenance authority') + parser.add_argument( + '--replace-conflicting-usernames', action='store_true', + help='Replace an existing DockerHub username only when the same token has an authoritative alive pair', + ) + return parser.parse_args() + + +def main(): + args = parse_args() + if args.apply and args.dry_run: + raise SystemExit('--apply and --dry-run are mutually exclusive') + config_path = canonical_path(args.config) + require_private_file(config_path) + config = load_config(config_path) + configured_value = (config.get('global') or {}).get('secrets_file') + if not configured_value: + raise SystemExit('global.secrets_file is required') + configured_secrets = canonical_path(configured_value) + requested_secrets = canonical_path(args.secrets) if args.secrets else configured_secrets + if requested_secrets != configured_secrets: + raise SystemExit('--secrets must exactly match canonical global.secrets_file') + load_postgres_environment(config_path, config) + endpoint_dsn = database_url_from_env() or (config.get('global') or {}).get('database_url') + if not is_postgres_url(endpoint_dsn): + raise SystemExit('A caller-selected canonical PostgreSQL DSN is required for maintenance authority') + provider = str(getattr(args, 'provider', 'github') or 'github').strip().lower() + defaults = PROVIDER_DEFAULTS.get(provider) + if defaults is None: + raise SystemExit(f'unsupported provider: {provider}') + alive_file = args.alive_file or defaults['alive_file'] + pool_name = args.pool or defaults['pool'] + name_prefix = args.name_prefix or defaults['name_prefix'] + with ClusterAuthorityLock(config, endpoint_dsn=endpoint_dsn) as authority_lock: + require_runtime_hardening_stopped(config) + result = sync_tokens( + configured_secrets, + alive_file, + pool_name, + name_prefix, + apply=args.apply, + provider=provider, + replace_conflicting_usernames=bool(getattr(args, 'replace_conflicting_usernames', False)), + canonical_secrets_path=configured_secrets, + authority_lock=authority_lock, + stopped_verified=True, + ) + mode = 'updated' if args.apply else 'dry_run' + print( + f"{mode}: provider={provider} pool={pool_name} alive_unique={result['alive_unique']} " + f"existing_before={result['existing_before']} added={result['added']} " + f"username_filled={result.get('username_filled', 0)} " + f"username_conflicts={result.get('username_conflicts', 0)} " + f"username_replaced={result.get('username_replaced', 0)} " + f"skipped_missing_username={result.get('skipped_missing_username', 0)} " + f"normalized_existing={result['normalized_existing']} " + f"normalized_usernames={result.get('normalized_usernames', 0)} pool_after={result['pool_after']}" + ) + return 0 + + +if __name__ == '__main__': + main() diff --git a/app/target_identity.py b/app/target_identity.py new file mode 100644 index 0000000..ed78daa --- /dev/null +++ b/app/target_identity.py @@ -0,0 +1,199 @@ +import hashlib +import json +import os +import re + + +SHA256_RE = re.compile(r'^[0-9a-f]{64}$') +DOCKER_DIGEST_RE = re.compile(r'^sha256:[0-9a-f]{64}$') +DOCKER_TAG_RE = r'[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}' +DOCKER_NAME_COMPONENT_RE = r'[a-z0-9]+(?:(?:[._]|__|[-]+)[a-z0-9]+)*' +DOCKER_DOMAIN_COMPONENT_RE = r'[A-Za-z0-9](?:[A-Za-z0-9-]{0,61}[A-Za-z0-9])?' +DOCKER_PORT_RE = r'[0-9]{1,5}' +DOCKER_REGISTRY_RE = ( + rf'(?:\[[0-9A-Fa-f:.]+\](?::{DOCKER_PORT_RE})?' + rf'|(?:localhost|{DOCKER_DOMAIN_COMPONENT_RE}(?:\.{DOCKER_DOMAIN_COMPONENT_RE})*)(?::{DOCKER_PORT_RE})?)' +) +DOCKER_IMAGE_RE = re.compile( + rf'^(?=.{{1,512}}$)(?:{DOCKER_REGISTRY_RE}/)?' + rf'{DOCKER_NAME_COMPONENT_RE}(?:/{DOCKER_NAME_COMPONENT_RE})*' + rf'(?::{DOCKER_TAG_RE})?(?:@sha256:[0-9a-f]{{64}})?$' +) +DOCKER_REVISION_RE = re.compile(r'^[A-Za-z0-9:+._-]{1,128}$') +DOCKER_TAG_TARGET_SCHEMA = 'docker-tag-v1' +DOCKERHUB_REGISTRIES = frozenset(( + 'docker.io', 'index.docker.io', 'registry-1.docker.io', +)) +HUGGINGFACE_SPACE_COMPONENT_RE = re.compile( + r'^[A-Za-z0-9_](?:[A-Za-z0-9._-]{0,94}[A-Za-z0-9_])?$' +) + + +def _postman_data(target): + if isinstance(target, dict): + return dict(target) + text = str(target or '').strip() + if not text: + return {} + if text.startswith('{'): + value = json.loads(text) + if not isinstance(value, dict): + raise ValueError('Postman target JSON must be an object') + return value + if text.lower().startswith('file:'): + return {'source': 'local_file', 'local_path': text[5:]} + if '://' in text: + return {'source': 'url', 'url': text} + return {'source': 'local_file', 'local_path': text} + + +def postman_target_identity(target): + """Return artifact identity independent of discovery-origin metadata.""" + try: + data = _postman_data(target) + except (TypeError, ValueError, json.JSONDecodeError): + return str(target or '').strip().lower() + digest = str(data.get('sha256') or data.get('hash') or data.get('cache_key') or '').strip().lower() + if SHA256_RE.fullmatch(digest): + return f'postman:sha256:{digest}' + cache_path = str(data.get('cache_path') or '') + cache_name = os.path.basename(cache_path).lower() + cache_digest = cache_name.split('.', 1)[0] + if SHA256_RE.fullmatch(cache_digest): + return f'postman:sha256:{cache_digest}' + source = str(data.get('source') or '').lower() + if source == 'github_code': + return 'postman:github_code:{}:{}:{}'.format( + str(data.get('repo') or '').lower(), + str(data.get('path') or '').replace('\\', '/').lower(), + str(data.get('sha') or '').lower(), + ) + if source == 'url': + return f'postman:url:{str(data.get("url") or "").rstrip("/").lower()}' + local_path = data.get('local_path') or data.get('path') or cache_path + if local_path: + canonical = os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(local_path)))) + return f'postman:file:{canonical}' + stable = { + key: value for key, value in data.items() + if key not in {'origin', 'cached_at', 'discovered_at', 'size', 'bytes'} + } + payload = json.dumps(stable, ensure_ascii=True, sort_keys=True, separators=(',', ':'), default=str) + return 'postman:metadata:' + hashlib.sha256(payload.encode('utf-8')).hexdigest() + + +def normalize_docker_digest(value): + digest = str(value or '').strip().lower() + return digest if DOCKER_DIGEST_RE.fullmatch(digest) else '' + + +def validate_docker_image_reference(value, require_digest=True): + image = str(value or '').strip() + if not image or image.startswith('{') or '://' in image or not DOCKER_IMAGE_RE.fullmatch(image): + raise ValueError('invalid Docker image reference') + digest = image.rsplit('@', 1)[1] if '@' in image else '' + if require_digest and not normalize_docker_digest(digest): + raise ValueError('Docker image reference must include a sha256 digest') + return image + + +def serialize_docker_tag_target(image, revision): + image = validate_docker_image_reference(image) + revision = str(revision or '').strip() + if not DOCKER_REVISION_RE.fullmatch(revision): + raise ValueError('invalid Docker tag revision') + return json.dumps( + { + 'image': image, + 'revision': revision, + 'schema': DOCKER_TAG_TARGET_SCHEMA, + }, + ensure_ascii=True, + sort_keys=True, + separators=(',', ':'), + ) + + +def parse_docker_target(target): + structured = isinstance(target, dict) + if structured: + data = dict(target) + else: + text = str(target or '').strip() + if not text.startswith('{'): + image = validate_docker_image_reference(text) + return {'image': image, 'revision': '', 'structured': False, 'target': image} + try: + data = json.loads(text) + except json.JSONDecodeError as exc: + raise ValueError('invalid Docker target JSON') from exc + if not isinstance(data, dict): + raise ValueError('Docker target JSON must be an object') + structured = True + + if set(data) != {'schema', 'image', 'revision'} or data.get('schema') != DOCKER_TAG_TARGET_SCHEMA: + raise ValueError('unsupported Docker target structure') + canonical = serialize_docker_tag_target(data.get('image'), data.get('revision')) + return { + 'image': validate_docker_image_reference(data.get('image')), + 'revision': str(data.get('revision')).strip(), + 'structured': structured, + 'target': canonical, + } + + +def parse_dockerhub_digest_target(target): + """Return canonical parts for a public immutable DockerHub image target.""" + parsed = parse_docker_target(target) + image = parsed['image'] + if image != image.lower(): + raise ValueError('DockerHub image reference must be canonical lowercase') + image_name, separator, digest = image.rpartition('@') + if not separator or not normalize_docker_digest(digest): + raise ValueError('DockerHub image reference must include a sha256 digest') + repository_with_registry = image_name.rsplit(':', 1)[0] if ':' in image_name.rsplit('/', 1)[-1] else image_name + parts = repository_with_registry.split('/') + first = parts[0] + has_registry = first == 'localhost' or '.' in first or ':' in first or first.startswith('[') + if has_registry: + registry = first + if registry not in DOCKERHUB_REGISTRIES or len(parts) < 2: + raise ValueError('Docker direct execution requires a public DockerHub image') + repository = '/'.join(parts[1:]) + else: + registry = 'docker.io' + repository = repository_with_registry + if '/' not in repository: + repository = 'library/' + repository + return { + **parsed, + 'image': image, + 'registry': registry, + 'repository': repository, + 'manifest_digest': digest, + 'normalized_target': image, + } + + +def normalize_huggingface_space_id(value): + """Validate and return one canonical case-preserved HuggingFace Space ID.""" + if not isinstance(value, str) or value != value.strip() or not 3 <= len(value) <= 96: + raise ValueError('invalid HuggingFace Space identifier') + if value.count('/') != 1 or '\\' in value or '..' in value or '--' in value: + raise ValueError('invalid HuggingFace Space identifier') + namespace, name = value.split('/', 1) + if ( + HUGGINGFACE_SPACE_COMPONENT_RE.fullmatch(namespace) is None + or HUGGINGFACE_SPACE_COMPONENT_RE.fullmatch(name) is None + or name.lower().endswith('.git') + ): + raise ValueError('invalid HuggingFace Space identifier') + return value + + +def docker_target_identity(target): + try: + data = parse_docker_target(target) + except (TypeError, ValueError): + return str(target or '').strip().lower() + return data['image'].lower() diff --git a/app/trufflehog-custom-detectors.yaml b/app/trufflehog-custom-detectors.yaml new file mode 100644 index 0000000..b7e6d52 --- /dev/null +++ b/app/trufflehog-custom-detectors.yaml @@ -0,0 +1,125 @@ +detectors: +- name: GoogleAIStudio + description: Google AI Studio API key in AQ-prefixed format. + keywords: + - AQ. + regex: + google_ai_studio_aq: '(?:^|[^A-Za-z0-9_-])(AQ\.[A-Za-z0-9_-]{50})(?:$|[^A-Za-z0-9_-])' + entropy: 4.0 +- name: QwenDashScope + description: Alibaba Cloud Model Studio / DashScope Qwen API key in an explicit environment or JSON assignment. + keywords: + - DASHSCOPE_API_KEY + - QWEN_API_KEY + regex: + qwen_dashscope_assignment: '(?:^|[^A-Za-z0-9_])(?:DASHSCOPE_API_KEY|QWEN_API_KEY)\s*["'']?\s*[:=]\s*["'']?(sk-(?:sp-)?[A-Za-z0-9][A-Za-z0-9_-]{20,505})(?:$|[^A-Za-z0-9_-])' + entropy: 3.5 +- name: DeepSeekApiKey + description: DeepSeek API key in an explicit environment or JSON assignment. + keywords: + - DEEPSEEK_API_KEY + regex: + deepseek_api_key_assignment: '(?:^|[^A-Za-z0-9_])DEEPSEEK_API_KEY\s*["'']?\s*[:=]\s*["'']?(sk-[a-z0-9]{32})(?:$|[^A-Za-z0-9_-])' + entropy: 3.5 +- name: KimiMoonshot + description: Kimi / Moonshot AI API key in an explicit environment or JSON assignment. + keywords: + - MOONSHOT_API_KEY + - KIMI_API_KEY + regex: + kimi_moonshot_assignment: '(?:^|[^A-Za-z0-9_])(?:MOONSHOT_API_KEY|KIMI_API_KEY)\s*["'']?\s*[:=]\s*["'']?(sk-[A-Za-z0-9][A-Za-z0-9_-]{20,505})(?:$|[^A-Za-z0-9_-])' + entropy: 3.5 +- name: Xai + description: xAI / Grok API key near xAI-specific context. + keywords: + - XAI_API_KEY + - api.x.ai + - xai- + - xAI + - grok + - Grok + regex: + xai_context_before: '(?i)(?:XAI_API_KEY|api\.x\.ai|xAI|grok)[\s\S]{0,120}\b(xai-[A-Za-z0-9_-]{20,})\b' + entropy: 3.5 +- name: XaiContextAfter + description: xAI / Grok API key before xAI-specific context. + keywords: + - XAI_API_KEY + - api.x.ai + - xai- + - xAI + - grok + - Grok + regex: + xai_context_after: '(?i)\b(xai-[A-Za-z0-9_-]{20,})\b[\s\S]{0,120}(?:XAI_API_KEY|api\.x\.ai|xAI|grok)' + entropy: 3.5 +- name: ZaiGLM + description: Z.ai / ZhipuAI / GLM API key near GLM-specific context. + keywords: + - ZAI_API_KEY + - GLM_API_KEY + - ZHIPUAI_API_KEY + - BIGMODEL_API_KEY + - api.z.ai + - z.ai + - Z.ai + - bigmodel.cn + - open.bigmodel.cn + - zhipuai + - ZhipuAI + - GLM + - glm + - ChatGLM + - chatglm + - glm- + regex: + zai_glm_context_before: '(?i)(?:ZAI_API_KEY|GLM_API_KEY|ZHIPUAI_API_KEY|BIGMODEL_API_KEY|api\.z\.ai|bigmodel\.cn|open\.bigmodel\.cn|zhipuai|chatglm|glm)[\s\S]{0,120}\b((?:zai|sk)-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128})\b' + entropy: 3.5 +- name: ZaiGLMContextAfter + description: Z.ai / ZhipuAI / GLM API key before GLM-specific context. + keywords: + - ZAI_API_KEY + - GLM_API_KEY + - ZHIPUAI_API_KEY + - BIGMODEL_API_KEY + - api.z.ai + - z.ai + - Z.ai + - bigmodel.cn + - open.bigmodel.cn + - zhipuai + - ZhipuAI + - GLM + - glm + - ChatGLM + - chatglm + - glm- + regex: + zai_glm_context_after: '(?i)\b((?:zai|sk)-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128})\b[\s\S]{0,120}(?:ZAI_API_KEY|GLM_API_KEY|ZHIPUAI_API_KEY|BIGMODEL_API_KEY|api\.z\.ai|bigmodel\.cn|open\.bigmodel\.cn|zhipuai|chatglm|glm)' + entropy: 3.5 +- name: AzureFoundryEndpointBeforeKey + description: Azure AI Foundry / model inference endpoint before an API key or bearer token. + keywords: + - services.ai.azure.com + - models.ai.azure.com + - inference.ai.azure.com + - AZURE_AI_FOUNDRY + - AZURE_FOUNDRY + - AI_FOUNDRY + - foundry + regex: + azure_foundry_endpoint_before_key: '(?i)((?:https?://)?[a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models\.ai\.azure\.com|services\.ai\.azure\.com|inference\.ai\.azure\.com)(?:/[^\s"''<>\\]*)?[\s\S]{0,240}(?:authorization|api[_-]?key|key|token|secret|credential|bearer)[^\n:=]{0,80}[:=]\s*["'']?(?:bearer\s+)?[A-Za-z0-9_./+=\-]{20,512})' + entropy: 3.5 +- name: AzureFoundryKeyBeforeEndpoint + description: Azure AI Foundry / model inference API key or bearer token before an endpoint. + keywords: + - services.ai.azure.com + - models.ai.azure.com + - inference.ai.azure.com + - AZURE_AI_FOUNDRY + - AZURE_FOUNDRY + - AI_FOUNDRY + - foundry + regex: + azure_foundry_key_before_endpoint: '(?i)((?:authorization|api[_-]?key|key|token|secret|credential|bearer)[^\n:=]{0,80}[:=]\s*["'']?(?:bearer\s+)?[A-Za-z0-9_./+=\-]{20,512}[\s\S]{0,240}(?:https?://)?[a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models\.ai\.azure\.com|services\.ai\.azure\.com|inference\.ai\.azure\.com)(?:/[^\s"''<>\\]*)?)' + entropy: 3.5 diff --git a/app/ui_components.py b/app/ui_components.py new file mode 100644 index 0000000..70dfa6e --- /dev/null +++ b/app/ui_components.py @@ -0,0 +1,12 @@ +"""Retired legacy Streamlit helpers. + +The unauthenticated UI no longer exposes finding payloads or credentials. Use +dashboard.py for redacted observability metadata. +""" + + +def extract_secret_value(finding): + """Return only an existing redacted representation for legacy callers.""" + if not isinstance(finding, dict): + return 'N/A' + return str(finding.get('Redacted') or finding.get('redacted_secret') or '***REDACTED***') diff --git a/app/worker_api.py b/app/worker_api.py new file mode 100644 index 0000000..b36f442 --- /dev/null +++ b/app/worker_api.py @@ -0,0 +1,1137 @@ +import argparse +import asyncio +import hashlib +import ipaddress +import json +import logging +import os +import re +import sys +from contextlib import asynccontextmanager + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('worker API could not disable bytecode writes') + +from starlette.applications import Starlette +from starlette.concurrency import run_in_threadpool +from starlette.requests import Request +from starlette.responses import JSONResponse, Response +from starlette.routing import Route + +from admin_api import AdminService, admin_routes +from capacity_model import MAX_RESULT_BUNDLE_BYTES, validate_remote_assignment_capacity +from host_agent_client import HostAgentClient +from host_agent_reconcile import ( + fixed_result_directory_is_safe, + reconcile_pending_host_results, +) +from managed_files import ManagedFileTraversal, managed_file_root_registry_from_config +from result_bundle import ( + BundleReservation, + ResultBundleError, + ResultBundleReader, + bundle_partial_path, + bundle_ready_path, + ensure_bundle_reservation_paths, +) +from runtime_security import ( + durable_publish, + harden_private_file, + private_file_ready, + reject_reparse_components, + require_private_directory, +) +from scan_execution import ( + WorkerBuildCompatibility, + validate_protocol1_remote_assignment, validate_protocol2_remote_assignment, +) +from scanner_db import ( + PipelineCapacityUnavailable, + ScanEventConflictError, + ScannerDB, + WorkerProgressInactiveError, + utc_now_iso, +) +from worker_contracts import ( + AssignmentOutcome, + DIAGNOSTIC_PROJECTION_VERSION, + MAX_DIAGNOSTICS_PER_ASSIGNMENT, + ScanOutcome, + decode_diagnostic_envelope, + encode_diagnostic_envelope, + ordered_diagnostic_uid_set_sha256, +) + + +REQUEST_ID_RE = re.compile(r'^[a-f0-9]{32,64}$') +DIGEST_RE = re.compile(r'^[a-f0-9]{64}$') +DEFAULT_MAX_JSON_BYTES = 16 * 1024 +DEFAULT_MAX_BUNDLE_BYTES = MAX_RESULT_BUNDLE_BYTES +DEFAULT_BODY_IDLE_TIMEOUT_SECONDS = 30 +DEFAULT_JSON_BODY_TIMEOUT_SECONDS = 60 +DEFAULT_BUNDLE_BODY_TIMEOUT_SECONDS = 30 * 60 +DEFAULT_CLAIM_RETRY_AFTER_SECONDS = 5 +NO_WORK_REASONS = frozenset(( + 'empty_queue', 'assignment_cap', 'dispatch_paused', 'capacity', + 'compatibility', +)) +TERMINAL_FAILURE_CODES = frozenset(( + 'client_process_failed', 'client_storage_failed', 'client_cancelled', +)) +logger = logging.getLogger(__name__) + + +class WorkerAPIError(RuntimeError): + def __init__(self, status_code, code, message): + super().__init__(message) + self.status_code = int(status_code) + self.code = str(code) + + +def _error(status_code, code, message): + return JSONResponse( + {'error': {'code': str(code), 'message': str(message)}}, + status_code=int(status_code), + headers=( + {'WWW-Authenticate': 'Bearer'} if int(status_code) == 401 else None + ), + ) + + +def _db_instance(db_factory, db_url): + db = db_factory(db_url=db_url, initialize=False) + if not db.enabled: + db.close() + raise RuntimeError('worker API PostgreSQL connection is unavailable') + return db + + +def _hash_file(path): + digest = hashlib.sha256() + with open(path, 'rb', buffering=0) as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b''): + digest.update(chunk) + return digest.hexdigest() + + +def _log_remote_result_event(event, outcome, identity, reservation_id, transport): + target = str( + transport.get('normalized_target') or transport.get('target') or '' + ) + correlation = { + 'event': str(event), + 'outcome': str(outcome), + 'reservation_id': int(reservation_id), + 'device_id': int(identity.get('device_id') or 0), + 'queue_id': int(transport.get('queue_id') or 0), + 'source': str(transport.get('source') or '')[:32], + 'bundle_id': str(transport.get('bundle_id') or '')[:64], + 'scan_event_id': str(transport.get('scan_event_id') or '')[:64], + 'target_sha256': hashlib.sha256(target.encode('utf-8')).hexdigest(), + } + logger.info( + 'remote result event %s', + json.dumps(correlation, ensure_ascii=True, sort_keys=True, separators=(',', ':')), + ) + + +class WorkerService: + def __init__( + self, db_url, bundle_root, assignment_builder, *, db_factory=ScannerDB, + max_bundle_bytes=DEFAULT_MAX_BUNDLE_BYTES, reaper_batch_size=1000, + bundle_capacity_bytes=3 * 1024 * 1024 * 1024, + body_idle_timeout_seconds=DEFAULT_BODY_IDLE_TIMEOUT_SECONDS, + json_body_timeout_seconds=DEFAULT_JSON_BODY_TIMEOUT_SECONDS, + bundle_body_timeout_seconds=DEFAULT_BUNDLE_BODY_TIMEOUT_SECONDS, + claim_retry_after_seconds=DEFAULT_CLAIM_RETRY_AFTER_SECONDS, + ): + self.db_url = str(db_url or '') + self.bundle_root = require_private_directory( + os.path.abspath(bundle_root), create=False, + ) + if not callable(assignment_builder): + raise TypeError('worker assignment builder must be callable') + self.assignment_builder = assignment_builder + self.db_factory = db_factory + self.max_bundle_bytes = max(1, min( + DEFAULT_MAX_BUNDLE_BYTES, int(max_bundle_bytes), + )) + self.bundle_capacity_bytes = max(1, int(bundle_capacity_bytes)) + self.reaper_batch_size = max(1, min(1000, int(reaper_batch_size))) + self.body_idle_timeout_seconds = float(body_idle_timeout_seconds) + self.json_body_timeout_seconds = float(json_body_timeout_seconds) + self.bundle_body_timeout_seconds = float(bundle_body_timeout_seconds) + self.claim_retry_after_seconds = int(claim_retry_after_seconds) + if ( + not 1 <= self.body_idle_timeout_seconds <= 120 + or not 1 <= self.json_body_timeout_seconds <= 300 + or not 30 <= self.bundle_body_timeout_seconds <= 24 * 60 * 60 + or not 1 <= self.claim_retry_after_seconds <= 300 + ): + raise ValueError('worker API request body time bounds are invalid') + self._upload_locks = {} + self._upload_locks_guard = asyncio.Lock() + + def authenticate(self, authorization): + prefix, separator, token = str(authorization or '').partition(' ') + if prefix.lower() != 'bearer' or not separator or not 16 <= len(token) <= 512: + raise WorkerAPIError(401, 'unauthorized', 'worker credentials are invalid') + token_sha256 = hashlib.sha256(token.encode('utf-8')).hexdigest() + db = _db_instance(self.db_factory, self.db_url) + try: + identity = db.authenticate_remote_worker(token_sha256) + finally: + db.close() + if not identity: + raise WorkerAPIError(401, 'unauthorized', 'worker credentials are invalid') + identity = dict(identity) + identity['token_sha256'] = token_sha256 + return identity + + def claim(self, identity, payload): + payload = dict(payload or {}) + if set(payload) != {'request_id', 'build'}: + raise WorkerAPIError(400, 'invalid_request', 'claim request shape is invalid') + request_id = str(payload.get('request_id') or '').lower() + if not REQUEST_ID_RE.fullmatch(request_id): + raise WorkerAPIError(400, 'invalid_request', 'request_id must be a 128-bit or stronger lowercase hex value') + try: + compatibility = WorkerBuildCompatibility.from_mapping(payload.get('build')) + except (TypeError, ValueError) as exc: + raise WorkerAPIError(400, 'invalid_compatibility', str(exc)) from exc + assignment = self.assignment_builder( + dict(identity), request_id, compatibility.as_dict(), + ) + if assignment is None: + if compatibility.protocol_version == 1: + raise WorkerAPIError( + 409, 'incompatible_protocol', + 'protocol-1 packages cannot receive new assignments', + ) + return None + assignment = dict(assignment) + if set(assignment) == {'no_assignment'}: + if compatibility.protocol_version == 1: + raise WorkerAPIError( + 409, 'incompatible_protocol', + 'protocol-1 packages cannot receive new assignments', + ) + no_assignment = assignment['no_assignment'] + if ( + not isinstance(no_assignment, dict) + or set(no_assignment) != {'reason'} + or no_assignment.get('reason') not in NO_WORK_REASONS + ): + raise RuntimeError('assignment builder returned an invalid no-work reason') + return {'no_assignment': {'reason': no_assignment['reason']}} + if set(assignment) == {'resolution'}: + resolution = dict(assignment['resolution'] or {}) + if ( + int(resolution.get('reservation_id') or 0) <= 0 + or resolution.get('resolution') not in { + 'bundle_accepted', 'prebundle_report', 'expired', + } + or not DIGEST_RE.fullmatch(str(resolution.get('receipt_id') or '')) + or not REQUEST_ID_RE.fullmatch(str(resolution.get('bundle_id') or '')) + or not REQUEST_ID_RE.fullmatch(str(resolution.get('scan_event_id') or '')) + ): + raise RuntimeError('assignment builder returned an invalid resolution receipt') + return {'resolution': resolution} + allowed = { + 'reservation', 'deadlines', 'compatibility', 'scan_kwargs', 'event_scan_options', + 'queue_policy', 'limits', 'scan_policy', 'execution_snapshot', + 'execution_snapshot_sha256', 'execution_plan', + } + if set(assignment) != allowed: + raise RuntimeError('assignment builder returned an invalid payload shape') + reservation = dict(assignment['reservation']) + validator = ( + validate_protocol1_remote_assignment + if compatibility.protocol_version == 1 + else validate_protocol2_remote_assignment + ) + validated = validator(assignment, compatibility) + required = validated['compatibility'] + if ( + int(reservation.get('remote_device_id') or 0) != int(identity['device_id']) + or str(reservation.get('assignment_kind') or '') != 'remote' + or not REQUEST_ID_RE.fullmatch( + str(reservation.get('reservation_token') or ''), + ) + or str(reservation.get('remote_effective_config_sha256') or '') + != required.effective_config_sha256 + ): + raise RuntimeError('assignment builder returned a conflicting remote identity') + return assignment + + def status(self, identity, reservation_id): + db = _db_instance(self.db_factory, self.db_url) + try: + result = db.remote_assignment_status( + int(reservation_id), int(identity['device_id']), + str(identity['token_sha256']), + ) + return result + finally: + db.close() + + def progress(self, identity, reservation_id, payload): + db = _db_instance(self.db_factory, self.db_url) + try: + try: + stored = db.record_remote_worker_progress_event( + int(reservation_id), int(identity['device_id']), + str(identity['token_sha256']), payload, + ) + except ValueError as exc: + raise WorkerAPIError( + 400, 'invalid_progress', 'worker progress event is invalid' + ) from exc + except WorkerProgressInactiveError as exc: + raise WorkerAPIError( + 410, 'progress_stale', + 'owned assignment is inactive for new progress', + ) from exc + return { + 'accepted': True, + 'reservation_id': int(reservation_id), + 'sequence': int(stored['sequence']), + 'received_at': str(stored['received_at']), + 'replayed': bool(stored.get('replayed')), + } + finally: + db.close() + + def transport(self, identity, reservation_id): + db = _db_instance(self.db_factory, self.db_url) + try: + return db.remote_assignment_transport( + int(reservation_id), int(identity['device_id']), + str(identity['token_sha256']), + ) + finally: + db.close() + + def accept_ready( + self, identity, transport, effective_diagnostics, metadata, payload_sha256, + ): + values = metadata.as_dict() + values['effective_diagnostic_count'] = len(effective_diagnostics) + values['effective_diagnostic_projection_version'] = ( + DIAGNOSTIC_PROJECTION_VERSION + ) + values['effective_diagnostic_uids_sha256'] = ( + ordered_diagnostic_uid_set_sha256(effective_diagnostics) + ) + values['relative_path'] = str(transport['ready_relative_path']).replace('\\', '/') + db = _db_instance(self.db_factory, self.db_url) + try: + try: + return db.mark_result_bundle_ready( + int(transport['reservation_id']), values, + remote_acceptance={ + 'device_id': int(identity['device_id']), + 'payload_sha256': payload_sha256, + 'token_sha256': str(identity['token_sha256']), + }, + bundle_capacity_bytes=self.bundle_capacity_bytes, + ) + except PipelineCapacityUnavailable as exc: + raise WorkerAPIError( + 503, 'capacity_backpressure', + 'bundle capacity is temporarily unavailable; retry the identical upload', + ) from exc + finally: + db.close() + + def report_terminal(self, identity, reservation_id, payload): + payload = dict(payload or {}) + failure_code = str(payload.get('failure_code') or '') + detail = str(payload.get('detail') or '') + if ( + set(payload) not in ( + {'failure_code', 'detail'}, + {'failure_code', 'detail', 'diagnostics'}, + ) + or failure_code not in TERMINAL_FAILURE_CODES or len(detail) > 1000 + or not isinstance(payload.get('detail'), str) + ): + raise WorkerAPIError(400, 'invalid_report', 'terminal report fields are invalid') + normalized = {'failure_code': failure_code, 'detail': detail} + if 'diagnostics' in payload: + diagnostics = payload['diagnostics'] + if ( + not isinstance(diagnostics, list) + or len(diagnostics) > MAX_DIAGNOSTICS_PER_ASSIGNMENT + ): + raise WorkerAPIError(400, 'invalid_report', 'terminal diagnostics are invalid') + try: + normalized_diagnostics = [] + diagnostic_uids = set() + for diagnostic in diagnostics: + envelope = decode_diagnostic_envelope(json.dumps( + diagnostic, ensure_ascii=True, sort_keys=True, + separators=(',', ':'), + ).encode('ascii')) + if ( + envelope.assignment_outcome + is not AssignmentOutcome.PREBUNDLE_FAILED + or envelope.scan_outcome is not ScanOutcome.UNAVAILABLE + ): + raise ValueError('terminal diagnostic outcome is invalid') + if envelope.diagnostic_uid in diagnostic_uids: + raise ValueError('terminal diagnostic UID is duplicated') + diagnostic_uids.add(envelope.diagnostic_uid) + normalized_diagnostics.append(json.loads( + encode_diagnostic_envelope(envelope).decode('ascii') + )) + normalized['diagnostics'] = normalized_diagnostics + except (TypeError, ValueError, UnicodeError) as exc: + raise WorkerAPIError( + 400, 'invalid_report', 'terminal diagnostics are invalid' + ) from exc + db = _db_instance(self.db_factory, self.db_url) + try: + return db.report_remote_prebundle_failure( + int(reservation_id), int(identity['device_id']), + str(identity['token_sha256']), normalized, + ) + finally: + db.close() + + def reap(self): + db = _db_instance(self.db_factory, self.db_url) + try: + receipts = db.reap_expired_remote_assignments(self.reaper_batch_size) + db.reconcile_runtime_drain() + try: + reconcile_pending_host_results(db) + except Exception as exc: + logger.warning( + 'host operation result reconciliation unavailable (%s)', + type(exc).__name__, + ) + return receipts + finally: + db.close() + + @asynccontextmanager + async def upload_scope(self, reservation_id): + reservation_id = int(reservation_id) + async with self._upload_locks_guard: + entry = self._upload_locks.get(reservation_id) + if entry is None: + entry = [asyncio.Lock(), 0] + self._upload_locks[reservation_id] = entry + entry[1] += 1 + await entry[0].acquire() + try: + yield + finally: + entry[0].release() + async with self._upload_locks_guard: + entry[1] -= 1 + if entry[1] == 0: + self._upload_locks.pop(reservation_id, None) + + +async def _bounded_body_chunks(request, *, absolute_timeout, idle_timeout): + iterator = request.stream().__aiter__() + loop = asyncio.get_running_loop() + deadline = loop.time() + float(absolute_timeout) + while True: + remaining = deadline - loop.time() + if remaining <= 0: + raise WorkerAPIError(408, 'request_timeout', 'request body deadline elapsed') + try: + chunk = await asyncio.wait_for( + anext(iterator), timeout=min(float(idle_timeout), remaining), + ) + except StopAsyncIteration: + return + except asyncio.TimeoutError as exc: + raise WorkerAPIError(408, 'request_timeout', 'request body deadline elapsed') from exc + yield chunk + + +async def _bounded_json( + request, max_bytes=DEFAULT_MAX_JSON_BYTES, *, + absolute_timeout=DEFAULT_JSON_BODY_TIMEOUT_SECONDS, + idle_timeout=DEFAULT_BODY_IDLE_TIMEOUT_SECONDS, +): + content_type = request.headers.get('content-type', '').split(';', 1)[0].strip().lower() + if content_type != 'application/json': + raise WorkerAPIError(415, 'unsupported_media_type', 'application/json is required') + body = bytearray() + async for chunk in _bounded_body_chunks( + request, absolute_timeout=absolute_timeout, idle_timeout=idle_timeout, + ): + if len(body) + len(chunk) > max_bytes: + raise WorkerAPIError(413, 'request_too_large', 'JSON request exceeds its byte bound') + body.extend(chunk) + try: + def reject_duplicate_fields(pairs): + value = {} + for key, item in pairs: + if key in value: + raise ValueError('duplicate JSON field') + value[key] = item + return value + + value = json.loads( + bytes(body).decode('utf-8', errors='strict'), + object_pairs_hook=reject_duplicate_fields, + ) + except (UnicodeDecodeError, json.JSONDecodeError, ValueError) as exc: + raise WorkerAPIError(400, 'invalid_json', 'request body is not valid UTF-8 JSON') from exc + if not isinstance(value, dict): + raise WorkerAPIError(400, 'invalid_json', 'request body must be a JSON object') + return value + + +def _reservation_id(request): + try: + value = int(request.path_params['reservation_id']) + except (TypeError, ValueError, OverflowError): + raise WorkerAPIError(404, 'not_found', 'assignment was not found') from None + if value <= 0: + raise WorkerAPIError(404, 'not_found', 'assignment was not found') + return value + + +async def _identity(request): + return await run_in_threadpool( + request.app.state.worker_service.authenticate, + request.headers.get('authorization'), + ) + + +async def claim(request): + identity = await _identity(request) + service = request.app.state.worker_service + payload = await _bounded_json( + request, absolute_timeout=getattr( + service, 'json_body_timeout_seconds', DEFAULT_JSON_BODY_TIMEOUT_SECONDS, + ), + idle_timeout=getattr( + service, 'body_idle_timeout_seconds', DEFAULT_BODY_IDLE_TIMEOUT_SECONDS, + ), + ) + assignment = await run_in_threadpool( + request.app.state.worker_service.claim, identity, payload, + ) + if assignment is None: + return Response( + status_code=204, + headers={'Retry-After': str(service.claim_retry_after_seconds)}, + ) + if set(assignment) == {'no_assignment'}: + return Response( + status_code=204, + headers={ + 'Retry-After': str(service.claim_retry_after_seconds), + 'X-Truf-No-Work-Reason': assignment['no_assignment']['reason'], + }, + ) + if set(assignment) == {'resolution'}: + return JSONResponse(assignment, status_code=200) + return JSONResponse({'assignment': assignment}, status_code=201) + + +async def assignment_status(request): + identity = await _identity(request) + result = await run_in_threadpool( + request.app.state.worker_service.status, identity, _reservation_id(request), + ) + if result is None: + raise WorkerAPIError(404, 'not_found', 'assignment was not found') + return JSONResponse(result) + + +async def assignment_progress(request): + identity = await _identity(request) + service = request.app.state.worker_service + payload = await _bounded_json( + request, absolute_timeout=getattr( + service, 'json_body_timeout_seconds', DEFAULT_JSON_BODY_TIMEOUT_SECONDS, + ), + idle_timeout=getattr( + service, 'body_idle_timeout_seconds', DEFAULT_BODY_IDLE_TIMEOUT_SECONDS, + ), + ) + result = await run_in_threadpool( + service.progress, identity, _reservation_id(request), payload, + ) + return JSONResponse(result) + + +async def terminal_report(request): + identity = await _identity(request) + service = request.app.state.worker_service + payload = await _bounded_json( + request, absolute_timeout=getattr( + service, 'json_body_timeout_seconds', DEFAULT_JSON_BODY_TIMEOUT_SECONDS, + ), + idle_timeout=getattr( + service, 'body_idle_timeout_seconds', DEFAULT_BODY_IDLE_TIMEOUT_SECONDS, + ), + ) + result = await run_in_threadpool( + request.app.state.worker_service.report_terminal, + identity, _reservation_id(request), payload, + ) + if result is None: + raise WorkerAPIError(409, 'assignment_not_active', 'assignment is no longer active') + return JSONResponse(result) + + +def _remove_private_regular(path): + if not os.path.lexists(path): + return + reject_reparse_components(path) + if not os.path.isfile(path) or os.path.islink(path): + raise ResultBundleError('bundle transport path is not a regular file') + os.remove(path) + + +async def upload_bundle(request): + identity = await _identity(request) + service = request.app.state.worker_service + reservation_id = _reservation_id(request) + supplied_digest = str(request.headers.get('x-truf-payload-sha256') or '').lower() + if not DIGEST_RE.fullmatch(supplied_digest): + raise WorkerAPIError(400, 'invalid_digest', 'X-Truf-Payload-SHA256 is required') + if request.headers.get('content-type', '').split(';', 1)[0].strip().lower() != 'application/octet-stream': + raise WorkerAPIError(415, 'unsupported_media_type', 'application/octet-stream is required') + try: + content_length = int(request.headers.get('content-length') or '') + except (TypeError, ValueError, OverflowError): + raise WorkerAPIError(411, 'length_required', 'a valid Content-Length is required') from None + + async with service.upload_scope(reservation_id): + transport = await run_in_threadpool(service.transport, identity, reservation_id) + if transport is None: + raise WorkerAPIError(404, 'not_found', 'assignment was not found') + byte_bound = min( + int(transport['declared_bundle_bytes']), service.max_bundle_bytes, + ) + if content_length <= 0 or content_length > byte_bound: + raise WorkerAPIError(413, 'bundle_too_large', 'bundle exceeds its assigned byte bound') + + async def receive(handle=None): + digest = hashlib.sha256() + received = 0 + async for chunk in _bounded_body_chunks( + request, + absolute_timeout=getattr( + service, 'bundle_body_timeout_seconds', + DEFAULT_BUNDLE_BODY_TIMEOUT_SECONDS, + ), + idle_timeout=getattr( + service, 'body_idle_timeout_seconds', + DEFAULT_BODY_IDLE_TIMEOUT_SECONDS, + ), + ): + received += len(chunk) + if received > content_length or received > byte_bound: + raise WorkerAPIError(413, 'bundle_too_large', 'bundle exceeds its assigned byte bound') + if handle is not None: + handle.write(chunk) + digest.update(chunk) + if received != content_length: + raise WorkerAPIError(400, 'length_mismatch', 'bundle length does not match Content-Length') + if digest.hexdigest() != supplied_digest: + raise WorkerAPIError(400, 'digest_mismatch', 'bundle digest does not match its declaration') + + receipt = transport.get('receipt') + if receipt is not None: + if ( + str(transport.get('remote_resolution_kind')) != 'bundle_accepted' + or str(transport.get('remote_payload_sha256') or '') != supplied_digest + ): + _log_remote_result_event( + 'remote_result_rejected', 'resolution_conflict', + identity, reservation_id, transport, + ) + raise WorkerAPIError(409, 'resolution_conflict', 'assignment already has a conflicting resolution') + await receive() + _log_remote_result_event( + 'remote_result_replayed', 'original_acceptance_returned', + identity, reservation_id, transport, + ) + return JSONResponse(receipt) + if str(transport.get('state')) != 'scanning' or str( + transport.get('remote_expires_at') or '' + ) <= utc_now_iso(): + _log_remote_result_event( + 'remote_result_rejected', 'stale_assignment', + identity, reservation_id, transport, + ) + raise WorkerAPIError(410, 'assignment_expired', 'assignment deadline has passed') + reservation = BundleReservation.from_mapping(transport) + await run_in_threadpool(ensure_bundle_reservation_paths, service.bundle_root, reservation) + partial_path = bundle_partial_path( + service.bundle_root, reservation.bundle_id, reservation.reservation_token, + ) + ready_path = bundle_ready_path(service.bundle_root, reservation.bundle_id) + if os.path.lexists(ready_path): + await receive() + reader = ResultBundleReader(ready_path, max_event_bytes=byte_bound) + metadata = await run_in_threadpool(reader.validate) + effective_diagnostics = await run_in_threadpool( + reader.effective_diagnostics + ) + existing_digest = await run_in_threadpool(_hash_file, ready_path) + if existing_digest != supplied_digest: + _log_remote_result_event( + 'remote_result_rejected', 'bundle_conflict', + identity, reservation_id, transport, + ) + raise WorkerAPIError(409, 'bundle_conflict', 'a different bundle already occupies the assigned path') + result = await run_in_threadpool( + service.accept_ready, identity, transport, effective_diagnostics, metadata, + supplied_digest, + ) + if not result: + await run_in_threadpool(_remove_private_regular, ready_path) + _log_remote_result_event( + 'remote_result_rejected', 'stale_assignment', + identity, reservation_id, transport, + ) + raise WorkerAPIError(410, 'assignment_expired', 'assignment deadline has passed') + return JSONResponse(result) + + await run_in_threadpool(_remove_private_regular, partial_path) + descriptor = os.open( + partial_path, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), + 0o600, + ) + published = False + try: + os.close(descriptor) + descriptor = None + harden_private_file(partial_path) + with open(partial_path, 'wb', buffering=0) as handle: + await receive(handle) + handle.flush() + os.fsync(handle.fileno()) + if not private_file_ready(partial_path): + raise ResultBundleError('bundle transport file is not private') + reader = ResultBundleReader(partial_path, max_event_bytes=byte_bound) + metadata = await run_in_threadpool(reader.validate) + effective_diagnostics = await run_in_threadpool( + reader.effective_diagnostics + ) + await run_in_threadpool(durable_publish, partial_path, ready_path) + published = True + result = await run_in_threadpool( + service.accept_ready, identity, transport, effective_diagnostics, metadata, + supplied_digest, + ) + if not result: + await run_in_threadpool(_remove_private_regular, ready_path) + published = False + _log_remote_result_event( + 'remote_result_rejected', 'stale_assignment', + identity, reservation_id, transport, + ) + raise WorkerAPIError(410, 'assignment_expired', 'assignment deadline has passed') + return JSONResponse(result, status_code=201) + except (WorkerAPIError, ResultBundleError): + if not published: + await run_in_threadpool(_remove_private_regular, partial_path) + raise + except ScanEventConflictError: + _log_remote_result_event( + 'remote_result_rejected', 'ownership_conflict', + identity, reservation_id, transport, + ) + await run_in_threadpool( + _remove_private_regular, ready_path if published else partial_path, + ) + raise + except Exception: + if not published: + await run_in_threadpool(_remove_private_regular, partial_path) + raise + finally: + if descriptor is not None: + os.close(descriptor) + + +async def worker_api_error_handler(request, exc): + return _error(exc.status_code, exc.code, str(exc)) + + +async def conflict_error_handler(request, exc): + return _error(409, 'reservation_conflict', str(exc)) + + +async def bundle_error_handler(request, exc): + return _error(400, 'invalid_bundle', str(exc)) + + +def create_worker_app(service, *, reaper_interval_seconds=60, admin_service=None): + interval = max(1, int(reaper_interval_seconds)) + + @asynccontextmanager + async def lifespan(app): + stopping = asyncio.Event() + traversal = None + app.state.managed_file_traversal = None + + if admin_service is not None: + try: + traversal = await run_in_threadpool( + ManagedFileTraversal, admin_service.managed_file_roots, + ) + app.state.managed_file_traversal = traversal + except Exception as exc: + logger.error( + 'Managed files are unavailable: %s', + getattr(exc, 'category', type(exc).__name__), + ) + + async def reap_loop(): + while not stopping.is_set(): + try: + await run_in_threadpool(service.reap) + except Exception as exc: + logger.error('Remote assignment reaper failed: %s', type(exc).__name__) + try: + await asyncio.wait_for(stopping.wait(), timeout=interval) + except asyncio.TimeoutError: + continue + + task = asyncio.create_task(reap_loop()) + try: + yield + finally: + stopping.set() + try: + await task + finally: + app.state.managed_file_traversal = None + if traversal is not None: + try: + await run_in_threadpool(traversal.close) + except Exception as exc: + logger.error( + 'Managed file traversal shutdown failed: %s', + type(exc).__name__, + ) + + routes = [ + Route('/api/v1/worker/claim', claim, methods=['POST']), + Route('/api/v1/worker/assignments/{reservation_id:int}', assignment_status, methods=['GET']), + Route('/api/v1/worker/assignments/{reservation_id:int}/progress', assignment_progress, methods=['POST']), + Route('/api/v1/worker/assignments/{reservation_id:int}/bundle', upload_bundle, methods=['PUT']), + Route('/api/v1/worker/assignments/{reservation_id:int}/terminal', terminal_report, methods=['POST']), + ] + if admin_service is not None: + routes.extend(admin_routes()) + app = Starlette( + routes=routes, + exception_handlers={ + WorkerAPIError: worker_api_error_handler, + ScanEventConflictError: conflict_error_handler, + ResultBundleError: bundle_error_handler, + }, + lifespan=lifespan, + ) + app.state.worker_service = service + if admin_service is not None: + app.state.admin_service = admin_service + return app + + +def _configured_auth_entry(source, source_config, secrets, configured_names): + from console_runner import auth_pool_entries + + entries, _ = auth_pool_entries(source_config, secrets) + selected_name = str(configured_names.get(source) or '') + if selected_name: + matches = [entry for entry in entries if str(entry.get('name') or '') == selected_name] + if len(matches) != 1: + raise ValueError(f'worker API auth entry for {source} is unavailable or ambiguous') + return matches[0] + if len(entries) > 1: + raise ValueError(f'worker API requires an explicit auth entry for {source}') + return entries[0] if entries else None + + +def build_configured_worker_service(config_path, metadata, *, db_factory=ScannerDB): + from console_runner import ( + apply_global_config, + build_args_from_source_config, + load_config, + load_secrets, + ) + from db_backend import database_url_from_env + from worker_assignment import ( + CORE_ASSIGNMENT_SOURCE_ADAPTERS, + PROTOCOL2_NEW_CLAIM_SOURCES, + RemoteAssignmentBuilder, + assignment_source_adapter, + ) + + config = load_config(config_path, managed_postgres=True, final_cutover=True) + global_config = config.get('global') or {} + supervisor_config = config.get('supervisor') or {} + worker_config = supervisor_config.get('worker_api') or {} + allowed_keys = { + 'enabled', 'address', 'port', 'sources', 'auth_entries', + 'compatibility_profiles', 'assignment_ttl_seconds', + 'assignment_ttl_seconds_by_source', + 'max_bundle_bytes', 'reaper_interval_seconds', 'reaper_batch_size', + 'limit_concurrency', 'body_idle_timeout_seconds', + 'json_body_timeout_seconds', 'bundle_body_timeout_seconds', 'admin', + } + unknown = sorted(set(worker_config) - allowed_keys) + if unknown: + raise ValueError('worker API configuration has unsupported keys: ' + ', '.join(unknown)) + if worker_config.get('enabled') is not True: + raise ValueError('worker API runtime is not explicitly enabled') + admin_config = worker_config.get('admin', {}) + if admin_config is None: + admin_config = {} + if not isinstance(admin_config, dict): + raise ValueError('worker API admin configuration must be a mapping') + admin_allowed_keys = { + 'enabled', 'origin', 'edge_marker', 'max_body_bytes', + 'snapshot_limit', 'requeue_limit', 'managed_file_roots', + } + admin_unknown = sorted(set(admin_config) - admin_allowed_keys) + if admin_unknown: + raise ValueError( + 'worker API admin configuration has unsupported keys: ' + + ', '.join(admin_unknown) + ) + if not isinstance(admin_config.get('enabled', False), bool): + raise ValueError('worker API admin enabled flag must be boolean') + managed_file_roots = managed_file_root_registry_from_config(config) + + raw_sources = worker_config.get('sources') or [] + if not isinstance(raw_sources, list): + raise ValueError('worker API sources must be a list') + sources = tuple(dict.fromkeys( + str(value or '').strip().lower() for value in raw_sources + )) if raw_sources else tuple(CORE_ASSIGNMENT_SOURCE_ADAPTERS) + if not sources or not set(sources) <= PROTOCOL2_NEW_CLAIM_SOURCES: + raise ValueError('worker API sources must use exact protocol-2 core sources') + + assignment_ttl = worker_config.get('assignment_ttl_seconds', 86400) + if ( + type(assignment_ttl) is not int + or not 60 <= assignment_ttl <= 7 * 24 * 60 * 60 + ): + raise ValueError('worker assignment lifetime must be between one minute and seven days') + assignment_ttl_by_source = worker_config.get( + 'assignment_ttl_seconds_by_source', {}, + ) + if ( + type(assignment_ttl_by_source) is not dict + or set(assignment_ttl_by_source) - set(PROTOCOL2_NEW_CLAIM_SOURCES) + ): + raise ValueError('worker assignment lifetime overrides contain unsupported sources') + assignment_ttl_by_source = dict(assignment_ttl_by_source) + if any( + type(value) is not int or not 60 <= value <= 7 * 24 * 60 * 60 + for value in assignment_ttl_by_source.values() + ): + raise ValueError('worker assignment lifetime override is outside its bounds') + + configured_names = worker_config.get('auth_entries') or {} + if ( + not isinstance(configured_names, dict) + or set(configured_names) - (set(sources) | {'github'}) + ): + raise ValueError( + 'worker API auth_entries must map only configured sources or legacy GitHub' + ) + if set(configured_names) - {'github', 'gitlab'}: + raise ValueError( + 'worker API auth_entries may select only GitHub or GitLab credentials' + ) + profiles = worker_config.get('compatibility_profiles') or {} + if not isinstance(profiles, dict) or not profiles or len(profiles) > 16: + raise ValueError('worker API requires between one and sixteen compatibility profiles') + profiles = {name: dict(value or {}) for name, value in profiles.items()} + config_dir = os.path.dirname(os.path.abspath(config_path)) + for profile in profiles.values(): + package_manifest = profile.get('package_manifest') + if isinstance(package_manifest, str) and not os.path.isabs(package_manifest): + profile['package_manifest'] = os.path.join(config_dir, package_manifest) + + address_text = str(worker_config.get('address') or '127.0.0.1').strip() + try: + address = ipaddress.ip_address(address_text) + except ValueError as exc: + raise ValueError('worker API address must be an IP literal') from exc + if ( + address.is_unspecified or address.is_multicast + or not (address.is_loopback or address.is_private) + ): + raise ValueError('worker API address must be loopback or private') + port = int(worker_config.get('port', 8766)) + if not 1024 <= port <= 65535: + raise ValueError('worker API port must be between 1024 and 65535') + + db_url = database_url_from_env() + if not db_url or str(global_config.get('database_url') or '') != db_url: + raise ValueError('worker API requires the canonical managed PostgreSQL DSN') + bundle_root = str(global_config.get('result_bundle_dir') or '') + if not bundle_root: + raise ValueError('worker API requires the canonical result bundle root') + instance_id = str((metadata or {}).get('instance_id') or '') + if not instance_id: + raise ValueError('worker API requires the authenticated supervisor identity') + + apply_global_config(global_config) + secrets = load_secrets(config, config_path) + configured_sources = config.get('sources') or {} + source_args = {} + credential_refs = {} + runtime_sources = sources + ( + ('github',) if 'github' in configured_names and 'github' not in sources else () + ) + for source in runtime_sources: + adapter = assignment_source_adapter(source) + source_config = configured_sources.get(source) + if not isinstance(source_config, dict) or source_config.get('enabled') is not True: + raise ValueError(f'worker API source {source} is not explicitly enabled') + auth_entry = None + if adapter.planning_kind == 'exact_git_v1': + auth_entry = _configured_auth_entry( + source, source_config, secrets, configured_names, + ) + args = build_args_from_source_config( + source, source_config, global_config, '', auth_entry=auth_entry, + ) + if adapter.planning_kind != 'exact_git_v1': + args.token = '' + args.docker_username = '' + args.docker_token = '' + args.auth_name = None + adapter.validate_source_args(args) + source_args[source] = args + credential_refs[source] = str((auth_entry or {}).get('name') or '') + + global_bundle_limit = int(global_config.get( + 'result_bundle_max_event_bytes', DEFAULT_MAX_BUNDLE_BYTES, + )) + validate_remote_assignment_capacity(global_config) + max_bundle_bytes = int(worker_config.get('max_bundle_bytes', global_bundle_limit)) + if not 1024 * 1024 <= max_bundle_bytes <= min(DEFAULT_MAX_BUNDLE_BYTES, global_bundle_limit): + raise ValueError('worker API bundle limit exceeds the canonical event bound') + reaper_interval = int(worker_config.get('reaper_interval_seconds', 60)) + if not 5 <= reaper_interval <= 3600: + raise ValueError('worker API reaper interval must be between 5 and 3600 seconds') + reaper_batch_size = int(worker_config.get('reaper_batch_size', 1000)) + if not 1 <= reaper_batch_size <= 1000: + raise ValueError('worker API reaper batch size must be between 1 and 1000') + limit_concurrency = int(worker_config.get('limit_concurrency', 64)) + if not 1 <= limit_concurrency <= 1024: + raise ValueError('worker API concurrency limit must be between 1 and 1024') + body_idle_timeout = int(worker_config.get( + 'body_idle_timeout_seconds', DEFAULT_BODY_IDLE_TIMEOUT_SECONDS, + )) + json_body_timeout = int(worker_config.get( + 'json_body_timeout_seconds', DEFAULT_JSON_BODY_TIMEOUT_SECONDS, + )) + bundle_body_timeout = worker_config.get( + 'bundle_body_timeout_seconds', DEFAULT_BUNDLE_BODY_TIMEOUT_SECONDS, + ) + if type(bundle_body_timeout) is not int or not 30 <= bundle_body_timeout <= 86400: + raise ValueError('worker result upload body timeout is outside its bounds') + for source in sources: + scan_timeout = getattr(source_args[source], 'timeout', 0) + if isinstance(scan_timeout, bool): + raise ValueError('worker source scan timeout is invalid') + try: + scan_timeout = int(scan_timeout) + except (TypeError, ValueError, OverflowError) as exc: + raise ValueError('worker source scan timeout is invalid') from exc + effective_ttl = assignment_ttl_by_source.get(source, assignment_ttl) + if effective_ttl < scan_timeout + bundle_body_timeout + 60: + raise ValueError( + f'worker assignment lifetime for {source} cannot cover scan and upload deadlines' + ) + + assignment_builder = RemoteAssignmentBuilder( + db_url, bundle_root, source_args, profiles, instance_id, + assignment_ttl_seconds=assignment_ttl, + assignment_ttl_seconds_by_source=assignment_ttl_by_source, + result_upload_body_timeout_seconds=bundle_body_timeout, + credential_refs=credential_refs, + db_factory=db_factory, + ) + service = WorkerService( + db_url, bundle_root, assignment_builder, db_factory=db_factory, + max_bundle_bytes=max_bundle_bytes, reaper_batch_size=reaper_batch_size, + bundle_capacity_bytes=int(global_config['result_bundle_max_total_bytes']), + body_idle_timeout_seconds=body_idle_timeout, + json_body_timeout_seconds=json_body_timeout, + bundle_body_timeout_seconds=bundle_body_timeout, + ) + service.admin_service = None + if admin_config.get('enabled') is True: + runtime_apply_provider = None + if HostAgentClient.is_available() and fixed_result_directory_is_safe(): + runtime_apply_provider = HostAgentClient().dispatch + service.admin_service = AdminService( + db_url, admin_config.get('origin'), admin_config.get('edge_marker'), + db_factory=db_factory, + max_body_bytes=admin_config.get('max_body_bytes', 8 * 1024), + snapshot_limit=admin_config.get('snapshot_limit', 200), + requeue_limit=admin_config.get('requeue_limit', 100), + supervisor_metadata=metadata, + package_compatibility_provider=assignment_builder.compatibility_snapshot, + runtime_config_path=config_path, + managed_file_roots=managed_file_roots, + runtime_apply_provider=runtime_apply_provider, + ) + return service, { + 'address': str(address), 'port': port, + 'reaper_interval_seconds': reaper_interval, + 'limit_concurrency': limit_concurrency, + } + + +def parse_args(): + parser = argparse.ArgumentParser(description='Authenticated remote scan Worker API') + parser.add_argument('--config', required=True) + return parser.parse_args() + + +def main(): + from lifecycle_authority import require_active_supervisor_child + + args = parse_args() + metadata = require_active_supervisor_child( + args.config, child_kind='worker-api', require_dsn=True, + ) + service, runtime = build_configured_worker_service(args.config, metadata) + app = create_worker_app( + service, reaper_interval_seconds=runtime['reaper_interval_seconds'], + admin_service=getattr(service, 'admin_service', None), + ) + import uvicorn + + uvicorn.run( + app, + host=runtime['address'], + port=runtime['port'], + access_log=False, + proxy_headers=False, + server_header=False, + limit_concurrency=runtime['limit_concurrency'], + timeout_keep_alive=5, + workers=1, + ) + + +if __name__ == '__main__': + main() diff --git a/app/worker_assignment.py b/app/worker_assignment.py new file mode 100644 index 0000000..5c12741 --- /dev/null +++ b/app/worker_assignment.py @@ -0,0 +1,1001 @@ +import hashlib +import os +import re +from dataclasses import asdict, dataclass +from datetime import datetime, timedelta +from types import MappingProxyType, SimpleNamespace + +from console_runner import ( + _GitResolutionFailure, + git_resolution_failure_result, + prepare_scan_options, + reserve_v2_admission_with_recovery, + resolve_and_bind_git_claim, + validate_v2_capacity_model, +) +from process_identity import current_process_identity +from result_bundle import BundleReservation, ResultBundleReader, bundle_ready_path +from lifecycle_authority import DISCOVERY_PRODUCER_SOURCES +from scan_execution import ( + PROTOCOL_VERSION, + QueueDispositionPolicy, + ScanCompatibility, + ScanExecutionError, + WorkerBuildCompatibility, + canonical_json_sha256, + normalize_docker_direct_execution_snapshot, + normalize_docker_direct_execution_target, + normalize_exact_git_execution_snapshot, + normalize_huggingface_space_execution_snapshot, + normalize_huggingface_space_execution_target, + normalize_remote_scan_policy, + remote_execution_identity, + remote_execution_snapshot_sha256, + stage_scan_result_in_scope, + validate_scan_kwargs, + validate_worker_build_compatibility, +) +from scanner_db import ScannerDB, ScanEventConflictError +from scanner import scan_config +from runtime_security import sha256_file +from worker_contracts import ( + DIAGNOSTIC_PROJECTION_VERSION, + ordered_diagnostic_uid_set_sha256, +) +from worker_package import ( + KNOWN_WORKER_PACKAGE_CAPABILITIES, + PACKAGE_DETECTOR_POLICY, + load_worker_package_manifest, + normalize_worker_package_manifest, + worker_package_build_compatibility, +) + + +DEFAULT_ASSIGNMENT_TTL_SECONDS = 24 * 60 * 60 +DEFAULT_RESULT_UPLOAD_BODY_TIMEOUT_SECONDS = 30 * 60 +PROTOCOL1_NEW_CLAIM_SOURCES = frozenset(('github', 'gitlab')) +SUPPORTED_REMOTE_GIT_SOURCES = PROTOCOL1_NEW_CLAIM_SOURCES +_EXACT_GIT_ASSIGNMENT_FLOW = object() +_DIRECT_ASSIGNMENT_FLOW = object() +NO_WORK_REASONS = frozenset(( + 'empty_queue', 'assignment_cap', 'dispatch_paused', 'capacity', + 'compatibility', +)) +_ADMISSION_NO_WORK_REASONS = { + 'no_claimable_target': 'empty_queue', + 'remote_user_quota_closed': 'assignment_cap', + 'remote_global_quota_closed': 'assignment_cap', + 'dispatch_gate_closed': 'dispatch_paused', + 'pipeline_capacity_closed': 'capacity', + 'quarantine_capacity_conflict': 'capacity', + 'quarantine_admission_closed': 'capacity', + 'ingester_not_ready': 'capacity', +} +_CANONICAL_CAPABILITY_NAME = re.compile(r'[a-z][a-z0-9_]{0,63}\Z') + + +@dataclass(frozen=True) +class AssignmentCapability: + source: str + platform: str + planning_kind: str + + def as_dict(self): + return { + 'source': self.source, + 'platform': self.platform, + 'planning_kind': self.planning_kind, + } + + +@dataclass(frozen=True) +class AssignmentSourceAdapter: + queue_source: str + worker_platform: str + planning_kind: str + package_capability: AssignmentCapability + snapshot_validator: object + assignment_flow: object = None + + def validate_source_args(self, args): + if str(getattr(args, 'platform', '') or '').strip().lower() != self.worker_platform: + raise ValueError('remote assignment source platform does not match its adapter') + if self.assignment_flow is None: + raise ValueError('remote assignment source is not available for new claims') + if self.assignment_flow is _EXACT_GIT_ASSIGNMENT_FLOW and ( + self.planning_kind != 'exact_git_v1' + or not bool(getattr(args, 'exact_git_planning_enabled', False)) + ): + raise ValueError('remote Git sources require enabled exact Git planning') + if self.assignment_flow is _DIRECT_ASSIGNMENT_FLOW: + if self.planning_kind not in { + 'docker_direct_v1', 'huggingface_space_v1', + } or any(str(getattr(args, name, '') or '') for name in ( + 'token', 'docker_username', 'docker_token', 'auth_name', + )): + raise ValueError('remote direct sources must be credential-free') + elif self.assignment_flow is not _EXACT_GIT_ASSIGNMENT_FLOW: + raise ValueError('remote assignment source flow is invalid') + + def validate_snapshot(self, value): + normalized = self.snapshot_validator(value) + if ( + normalized['execution']['source'] != self.worker_platform + or normalized['planning']['kind'] != self.planning_kind + or normalized['credential_ref']['source'] != self.queue_source + ): + raise ScanExecutionError('remote execution snapshot does not match its source adapter') + return normalized + + +def _exact_git_snapshot(value): + return normalize_exact_git_execution_snapshot(value) + + +def _adapter(source, platform, planning_kind, *, assignment_flow=None, validator=None): + capability = AssignmentCapability(source, platform, planning_kind) + return AssignmentSourceAdapter( + source, platform, planning_kind, capability, + validator, assignment_flow, + ) + + +LEGACY_GITHUB_ASSIGNMENT_ADAPTER = _adapter( + 'github', 'github', 'exact_git_v1', + assignment_flow=_EXACT_GIT_ASSIGNMENT_FLOW, + validator=_exact_git_snapshot, +) +CORE_ASSIGNMENT_SOURCE_ADAPTERS = MappingProxyType({ + 'gitlab': _adapter( + 'gitlab', 'gitlab', 'exact_git_v1', + assignment_flow=_EXACT_GIT_ASSIGNMENT_FLOW, + validator=_exact_git_snapshot, + ), + 'dockerhub': _adapter( + 'dockerhub', 'docker', 'docker_direct_v1', + assignment_flow=_DIRECT_ASSIGNMENT_FLOW, + validator=normalize_docker_direct_execution_snapshot, + ), + 'huggingface': _adapter( + 'huggingface', 'huggingface', 'huggingface_space_v1', + assignment_flow=_DIRECT_ASSIGNMENT_FLOW, + validator=normalize_huggingface_space_execution_snapshot, + ), +}) +ASSIGNMENT_SOURCE_ADAPTERS = MappingProxyType({ + 'github': LEGACY_GITHUB_ASSIGNMENT_ADAPTER, + **CORE_ASSIGNMENT_SOURCE_ADAPTERS, +}) +PROTOCOL2_NEW_CLAIM_SOURCES = frozenset(CORE_ASSIGNMENT_SOURCE_ADAPTERS) + + +def _validate_assignment_source_adapters(): + if tuple(CORE_ASSIGNMENT_SOURCE_ADAPTERS) != tuple(DISCOVERY_PRODUCER_SOURCES): + raise RuntimeError('core assignment adapters do not match discovery producers') + pairs = set() + for key, adapter in ASSIGNMENT_SOURCE_ADAPTERS.items(): + capability = adapter.package_capability + identities = ( + key, adapter.queue_source, adapter.worker_platform, + adapter.planning_kind, + ) + if any(_CANONICAL_CAPABILITY_NAME.fullmatch(value) is None for value in identities): + raise RuntimeError('remote assignment adapter identity is invalid') + if key != adapter.queue_source or capability.as_dict() != { + 'source': adapter.queue_source, + 'platform': adapter.worker_platform, + 'planning_kind': adapter.planning_kind, + }: + raise RuntimeError('remote assignment adapter capability is inconsistent') + pair = (adapter.queue_source, adapter.worker_platform) + if pair in pairs: + raise RuntimeError('remote assignment adapter source/platform is duplicated') + pairs.add(pair) + if { + source for source, adapter in ASSIGNMENT_SOURCE_ADAPTERS.items() + if adapter.assignment_flow is _EXACT_GIT_ASSIGNMENT_FLOW + } != set(PROTOCOL1_NEW_CLAIM_SOURCES): + raise RuntimeError('protocol-1 assignment source registry is inconsistent') + if { + source for source, adapter in ASSIGNMENT_SOURCE_ADAPTERS.items() + if adapter.assignment_flow is _DIRECT_ASSIGNMENT_FLOW + } != {'dockerhub', 'huggingface'}: + raise RuntimeError('direct assignment source registry is inconsistent') + if { + ( + adapter.package_capability.source, + adapter.package_capability.platform, + adapter.package_capability.planning_kind, + ) + for adapter in ASSIGNMENT_SOURCE_ADAPTERS.values() + } != set(KNOWN_WORKER_PACKAGE_CAPABILITIES): + raise RuntimeError('worker package capabilities do not match source adapters') + + +_validate_assignment_source_adapters() + + +def assignment_source_adapter(source): + source = str(source or '').strip().lower() + adapter = ASSIGNMENT_SOURCE_ADAPTERS.get(source) + if adapter is None: + raise ValueError('unsupported remote assignment source') + return adapter + + +def _stable_id(device_id, request_id, label): + payload = f'truf-worker-v1\0{int(device_id)}\0{request_id}\0{label}'.encode('ascii') + return hashlib.sha256(payload).hexdigest()[:32] + + +def _hash_file(path): + digest = hashlib.sha256() + with open(path, 'rb', buffering=0) as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b''): + digest.update(chunk) + return digest.hexdigest() + + +class RemoteAssignmentBuilder: + """Central remote admission builder routed through closed source adapters.""" + + def __init__( + self, db_url, bundle_root, source_args, compatibility_profiles, + supervisor_instance_id, *, assignment_ttl_seconds=DEFAULT_ASSIGNMENT_TTL_SECONDS, + assignment_ttl_seconds_by_source=None, + result_upload_body_timeout_seconds=DEFAULT_RESULT_UPLOAD_BODY_TIMEOUT_SECONDS, + credential_refs=None, + db_factory=ScannerDB, admission=reserve_v2_admission_with_recovery, + planner=resolve_and_bind_git_claim, + ): + self.db_url = str(db_url or '') + self.bundle_root = os.path.abspath(bundle_root) + self.supervisor_instance_id = str(supervisor_instance_id or '') + if not self.db_url or not self.supervisor_instance_id: + raise ValueError('remote assignment builder requires database and supervisor identities') + if ( + type(assignment_ttl_seconds) is not int + or not 60 <= assignment_ttl_seconds <= 7 * 24 * 60 * 60 + ): + raise ValueError('remote assignment lifetime must be between one minute and seven days') + self.assignment_ttl_seconds = assignment_ttl_seconds + raw_ttl_overrides = assignment_ttl_seconds_by_source + if raw_ttl_overrides is None: + raw_ttl_overrides = {} + if ( + type(raw_ttl_overrides) is not dict + or set(raw_ttl_overrides) - set(CORE_ASSIGNMENT_SOURCE_ADAPTERS) + ): + raise ValueError('remote assignment lifetime overrides contain unsupported sources') + self.assignment_ttl_seconds_by_source = {} + for source, value in raw_ttl_overrides.items(): + if type(value) is not int or not 60 <= value <= 7 * 24 * 60 * 60: + raise ValueError('remote assignment lifetime override is outside its bounds') + self.assignment_ttl_seconds_by_source[source] = value + if ( + type(result_upload_body_timeout_seconds) is not int + or not 30 <= result_upload_body_timeout_seconds <= 24 * 60 * 60 + ): + raise ValueError('remote result upload body timeout is outside its bounds') + self.result_upload_body_timeout_seconds = result_upload_body_timeout_seconds + self.db_factory = db_factory + self.admission = admission + self.planner = planner + + self.source_args = {} + self.source_adapters = {} + for source, args in dict(source_args or {}).items(): + source = str(source or '').strip().lower() + adapter = assignment_source_adapter(source) + adapter.validate_source_args(args) + self.source_args[source] = args + self.source_adapters[source] = adapter + if not self.source_args: + raise ValueError('remote assignment builder has no supported sources') + raw_credential_refs = dict(credential_refs or {}) + if set(raw_credential_refs) - set(self.source_args): + raise ValueError('remote credential references contain unsupported sources') + self.credential_refs = {} + for source in self.source_args: + reference = str(raw_credential_refs.get(source) or '') + if len(reference) > 128 or '\x00' in reference: + raise ValueError('remote credential reference is invalid') + if ( + self.source_adapters[source].assignment_flow + is _DIRECT_ASSIGNMENT_FLOW and reference + ): + raise ValueError('remote direct source credential reference must be empty') + self.credential_refs[source] = reference + + self.compatibility_profiles = {} + for profile_name, value in dict(compatibility_profiles or {}).items(): + profile_name = str(profile_name or '').strip() + profile = dict(value or {}) + if set(profile) - {'package_manifest', 'sources'}: + raise ValueError('remote compatibility profile shape is invalid') + package_value = profile.get('package_manifest') + package = ( + load_worker_package_manifest(package_value) + if isinstance(package_value, (str, os.PathLike)) + else normalize_worker_package_manifest(package_value) + ) + required_build = WorkerBuildCompatibility.from_mapping( + worker_package_build_compatibility(package), + ) + package_capabilities = frozenset( + ( + item['source'], item['platform'], item['planning_kind'], + ) + for item in package['capabilities'] + ) + package_sources = frozenset( + source for source, _platform, _planning in package_capabilities + ) + allowed_sources = frozenset( + str(item or '').strip().lower() + for item in (profile.get('sources') or package_sources) + ) + if ( + not profile_name or not allowed_sources + or not allowed_sources <= set(self.source_args) + or any( + ( + self.source_adapters[source].package_capability.source, + self.source_adapters[source].package_capability.platform, + self.source_adapters[source].package_capability.planning_kind, + ) not in package_capabilities + for source in allowed_sources + ) + ): + raise ValueError('remote compatibility profile has invalid sources') + key = tuple(required_build.as_dict().values()) + if key in self.compatibility_profiles: + raise ValueError('remote compatibility profile identity is duplicated') + self.compatibility_profiles[key] = { + 'profile_name': profile_name, + 'build': required_build, + 'package_manifest': package, + 'sources': allowed_sources, + 'capabilities': package_capabilities, + } + if not self.compatibility_profiles: + raise ValueError('remote assignment builder has no compatibility profiles') + + def compatibility_snapshot(self): + profiles = [] + for profile in sorted( + self.compatibility_profiles.values(), + key=lambda item: item['profile_name'], + ): + profiles.append({ + 'profile_name': profile['profile_name'], + **profile['build'].as_dict(), + 'sources': sorted(profile['sources']), + 'capabilities': [ + { + 'source': source, + 'platform': platform, + 'planning_kind': planning_kind, + } + for source, platform, planning_kind in sorted( + profile['capabilities'], + ) + ], + }) + required_capabilities = [ + self.source_adapters[source].package_capability.as_dict() + for source in sorted(self.source_adapters) + ] + return { + 'profiles': profiles, + 'required_capabilities': required_capabilities, + } + + @staticmethod + def _queue_policy(args): + return QueueDispositionPolicy( + target_retry_max_attempts=int(getattr(args, 'target_retry_max_attempts', 3) or 3), + target_retry_base_delay_sec=int(getattr(args, 'target_retry_base_delay_sec', 3600) or 3600), + target_retry_max_delay_sec=int(getattr(args, 'target_retry_max_delay_sec', 86400) or 86400), + target_timeout_retry_delay_sec=int(getattr(args, 'target_timeout_retry_delay_sec', 21600) or 21600), + ) + + @staticmethod + def _capacity(args): + return { + 'bundle_items': int(getattr(args, 'result_bundle_max_items', 10000)), + 'bundle_bytes': int(getattr(args, 'result_bundle_max_total_bytes', 3 << 30)), + 'projection_items': int(getattr(args, 'projection_backlog_max_items', 10000)), + 'projection_bytes': int(getattr(args, 'projection_backlog_max_bytes', 2 << 30)), + 'projection_headroom_bytes': int(getattr( + args, 'projection_backlog_headroom_bytes', 0, + )), + 'keycheck_items': int(getattr(args, 'keycheck_queue_max_items', 100000)), + 'keycheck_bytes': int(getattr(args, 'keycheck_queue_max_bytes', 512 << 20)), + 'quarantine_items': int(getattr(args, 'pipeline_quarantine_max_items', 10000)), + 'quarantine_bytes': int(getattr(args, 'pipeline_quarantine_max_bytes', 1 << 30)), + } + + @staticmethod + def _scan_policy(args): + def setting(name, default): + return getattr(args, name, getattr(scan_config, name, default)) + + return normalize_remote_scan_policy({ + 'drop_detectors': setting('drop_detectors', ()), + 'strict_git_provider_token_filter': bool(setting( + 'strict_git_provider_token_filter', True, + )), + 'trufflehog_stdout_max_mb': int(setting('trufflehog_stdout_max_mb', 32)), + 'trufflehog_stderr_max_mb': int(setting('trufflehog_stderr_max_mb', 8)), + 'result_bundle_max_event_bytes': int(setting( + 'result_bundle_max_event_bytes', 64 << 20, + )), + 'trufflehog_max_findings_per_target': int(setting( + 'trufflehog_max_findings_per_target', 20000, + )), + 'trufflehog_job_memory_limit_bytes': int(setting( + 'trufflehog_job_memory_limit_bytes', 0, + )), + 'trufflehog_windows_job_cpu_weight': int(setting( + 'trufflehog_windows_job_cpu_weight', 0, + )), + 'trufflehog_windows_memory_priority': int(setting( + 'trufflehog_windows_memory_priority', 0, + )), + 'trufflehog_diagnostic_max_lines': int(setting( + 'trufflehog_diagnostic_max_lines', 2000, + )), + 'trufflehog_diagnostic_max_line_chars': int(setting( + 'trufflehog_diagnostic_max_line_chars', 8192, + )), + 'trufflehog_diagnostic_max_line_bytes': int(setting( + 'trufflehog_diagnostic_max_line_bytes', 8192, + )), + 'trufflehog_diagnostic_max_errors': int(setting( + 'trufflehog_diagnostic_max_errors', 200, + )), + 'trufflehog_diagnostic_max_warnings': int(setting( + 'trufflehog_diagnostic_max_warnings', 200, + )), + 'trufflehog_diagnostic_max_unclassified': int(setting( + 'trufflehog_diagnostic_max_unclassified', 20, + )), + }) + + def _db(self, application_name): + db = self.db_factory(db_url=self.db_url, initialize=False) + if not db.enabled: + db.close() + raise RuntimeError('remote assignment PostgreSQL connection is unavailable') + db.set_application_name(application_name) + return db + + def _accept_staged(self, identity, claim, staged): + ready_path = bundle_ready_path(self.bundle_root, claim['bundle_id']) + relative_path = os.path.relpath(ready_path, self.bundle_root).replace(os.sep, '/') + staged_relative_path = str(getattr(staged, 'relative_path', '') or '').replace( + '\\', '/' + ) + if staged_relative_path and staged_relative_path != relative_path: + raise RuntimeError('published remote bundle path conflicts with its reservation') + digest = _hash_file(ready_path) + reader = ResultBundleReader( + ready_path, max_event_bytes=int(claim['declared_bundle_bytes']), + ) + metadata = reader.validate() + effective_diagnostics = reader.effective_diagnostics() + values = metadata.as_dict() + values['effective_diagnostic_count'] = len(effective_diagnostics) + values['effective_diagnostic_projection_version'] = ( + DIAGNOSTIC_PROJECTION_VERSION + ) + values['effective_diagnostic_uids_sha256'] = ( + ordered_diagnostic_uid_set_sha256(effective_diagnostics) + ) + values['relative_path'] = relative_path + db = self._db('truf-worker-planning-result') + try: + return db.mark_result_bundle_ready( + int(claim['reservation_id']), values, + remote_acceptance={ + 'device_id': int(identity['device_id']), + 'payload_sha256': digest, + 'token_sha256': str(identity['token_sha256']), + }, + bundle_capacity_bytes=self._capacity( + self.source_args[claim['source']] + )['bundle_bytes'], + ) + finally: + db.close() + + def _recover_published(self, identity, claim): + path = bundle_ready_path(self.bundle_root, claim['bundle_id']) + if not os.path.lexists(path): + return False + metadata = ResultBundleReader( + path, max_event_bytes=int(claim['declared_bundle_bytes']), + ).validate() + if metadata.header != BundleReservation.from_mapping(claim).header(): + raise RuntimeError('published remote bundle conflicts with its reservation') + self._accept_staged(identity, claim, metadata) + return True + + def _bound_plan(self, identity, claim): + db = self._db('truf-worker-plan-reconcile') + try: + return db.remote_bound_git_scan_plan( + int(claim['reservation_id']), int(identity['device_id']), + str(claim['claim_lease_token']), str(identity['token_sha256']), + ) + finally: + db.close() + + def _reconcile_request(self, identity, request_id): + db = self._db('truf-worker-request-reconcile') + try: + return db.reconcile_remote_assignment_request( + request_id, int(identity['device_id']), str(identity['token_sha256']), + ) + finally: + db.close() + + def _snapshot(self, adapter, args, compatibility, execution): + planning = {'kind': adapter.planning_kind} + if adapter.assignment_flow is _EXACT_GIT_ASSIGNMENT_FLOW: + planning.update({ + 'git_baseline_depth': max( + 1, int(getattr(args, 'git_baseline_depth', 100) or 100), + ), + 'git_ref_resolution_attempts': max( + 1, int(getattr(args, 'git_ref_resolution_attempts', 2) or 2), + ), + 'git_ref_resolution_timeout_sec': max( + 0.1, float(getattr(args, 'git_ref_resolution_timeout_sec', 10) or 10), + ), + 'git_ref_resolution_max_bytes': max( + 1024, int(getattr(args, 'git_ref_resolution_max_bytes', 1 << 20) or (1 << 20)), + ), + }) + return adapter.validate_snapshot({ + 'schema': 1, + 'compatibility': compatibility.as_dict(), + 'execution': execution, + 'planning': planning, + 'credential_ref': { + 'source': adapter.queue_source, + 'auth_entry': self.credential_refs[adapter.queue_source], + }, + }) + + def _rehydrate(self, client, claim, snapshot, snapshot_sha256=None): + adapter = assignment_source_adapter(claim.get('source')) + if ( + self.source_adapters.get(adapter.queue_source) is not adapter + or str(claim.get('platform') or '') != adapter.worker_platform + or adapter.assignment_flow not in { + _EXACT_GIT_ASSIGNMENT_FLOW, _DIRECT_ASSIGNMENT_FLOW, + } + ): + raise RuntimeError('remote assignment source adapter is unavailable') + snapshot = adapter.validate_snapshot(snapshot) + if ( + snapshot_sha256 is not None + and str(snapshot_sha256) != remote_execution_snapshot_sha256(snapshot) + ): + raise ScanEventConflictError('remote execution snapshot changed after admission') + required = ScanCompatibility.from_mapping(snapshot['compatibility']) + required_build = WorkerBuildCompatibility.from_mapping({ + 'protocol_version': required.protocol_version, + 'bundle_format_version': required.bundle_format_version, + 'platform_tag': required.platform_tag, + 'code_manifest_sha256': required.code_manifest_sha256, + 'detector_policy_sha256': required.detector_policy_sha256, + }) + if required.protocol_version not in {1, PROTOCOL_VERSION}: + raise ScanEventConflictError( + 'remote execution snapshot protocol is unsupported' + ) + validate_worker_build_compatibility( + required_build, client, + expected_protocol_version=required.protocol_version, + ) + source = adapter.queue_source + args = self.source_args.get(source) + if args is None or snapshot['credential_ref'] != { + 'source': source, 'auth_entry': self.credential_refs.get(source), + }: + raise RuntimeError('remote assignment credential reference is unavailable') + values = vars(args).copy() + values.update(snapshot['planning']) + values.pop('kind', None) + planning_args = SimpleNamespace(**values) + event_scan_options = dict(snapshot['execution']['scan_kwargs']) + scan_kwargs = dict(event_scan_options) + if adapter.assignment_flow is _EXACT_GIT_ASSIGNMENT_FLOW: + scan_kwargs['token'] = str(getattr(args, 'token', '') or '') + return ( + adapter, planning_args, snapshot, scan_kwargs, event_scan_options, + QueueDispositionPolicy(**snapshot['execution']['queue_policy']), + dict(snapshot['execution']['limits']), + dict(snapshot['execution']['scan_policy']), required, + ) + + def _complete_claim( + self, identity, client, claim, snapshot, *, snapshot_sha256=None, + recovered_execution_plan=None, + ): + ( + adapter, args, snapshot, scan_kwargs, event_scan_options, queue_policy, + limits, scan_policy, required, + ) = self._rehydrate(client, claim, snapshot, snapshot_sha256) + device_id = int(identity['device_id']) + status_db = self._db('truf-worker-claim-reconcile') + try: + status = status_db.remote_assignment_status( + int(claim['reservation_id']), device_id, + str(identity['token_sha256']), + ) + finally: + status_db.close() + if status and status.get('receipt_id'): + return None + if self._recover_published(identity, claim): + return None + + if adapter.assignment_flow is _EXACT_GIT_ASSIGNMENT_FLOW: + execution_target = str(claim.get('target') or '') + elif adapter.planning_kind == 'docker_direct_v1': + execution_target = normalize_docker_direct_execution_target( + claim.get('target'), + )['image'] + elif adapter.planning_kind == 'huggingface_space_v1': + execution_target = normalize_huggingface_space_execution_target( + claim.get('target'), + ) + else: + raise RuntimeError('remote assignment execution plan is unavailable') + + bound_plan = None + if recovered_execution_plan is not None: + if ( + not isinstance(recovered_execution_plan, dict) + or set(recovered_execution_plan) != { + 'kind', 'execution_target', 'bound_plan', + } + or recovered_execution_plan['kind'] != adapter.planning_kind + or str(recovered_execution_plan['execution_target']) != execution_target + or ( + adapter.assignment_flow is _EXACT_GIT_ASSIGNMENT_FLOW + and not isinstance(recovered_execution_plan['bound_plan'], dict) + ) + or ( + adapter.assignment_flow is _DIRECT_ASSIGNMENT_FLOW + and recovered_execution_plan['bound_plan'] is not None + ) + ): + raise ScanEventConflictError( + 'recovered execution plan conflicts with its assignment' + ) + bound_plan = recovered_execution_plan['bound_plan'] + if ( + adapter.assignment_flow is _EXACT_GIT_ASSIGNMENT_FLOW + and bound_plan is None + ): + bound_plan = self._bound_plan(identity, claim) + if ( + adapter.assignment_flow is _EXACT_GIT_ASSIGNMENT_FLOW + and bound_plan is None + ): + try: + bound_plan = self.planner( + args, self.db_url, adapter.queue_source, claim, scan_kwargs, + remote_credential={ + 'device_id': device_id, + 'token_sha256': str(identity['token_sha256']), + }, + ) + except ScanEventConflictError: + bound_plan = self._bound_plan(identity, claim) + if bound_plan is None: + raise + except _GitResolutionFailure as failure: + bound_plan = self._bound_plan(identity, claim) + if bound_plan is None: + result = git_resolution_failure_result(args, claim, scan_kwargs, failure) + staged = stage_scan_result_in_scope( + result, claim, self.bundle_root, event_scan_options, queue_policy, + attempts=int(claim.get('attempts') or 0), + candidate_max_items=limits['candidate_max_items'], + candidate_max_bytes=limits['candidate_max_bytes'], + require_s_drive=False, + ) + self._accept_staged(identity, claim, staged) + return None + + assigned_scan_kwargs = dict(scan_kwargs) + if adapter.assignment_flow is _EXACT_GIT_ASSIGNMENT_FLOW: + assigned_scan_kwargs['git_plan'] = bound_plan + execution_plan = { + 'kind': adapter.planning_kind, + 'execution_target': execution_target, + 'bound_plan': bound_plan, + } + issued_at = str(claim.get('remote_issued_at') or '') + deadline_at = str(claim.get('remote_expires_at') or '') + try: + issued = datetime.fromisoformat(issued_at) + deadline = datetime.fromisoformat(deadline_at) + except ValueError as exc: + raise RuntimeError('remote assignment reservation deadline is invalid') from exc + if ( + issued.tzinfo is None + or deadline.tzinfo is None + or issued.utcoffset() != timedelta(0) + or deadline.utcoffset() != timedelta(0) + or issued.isoformat(timespec='seconds') != issued_at + or deadline.isoformat(timespec='seconds') != deadline_at + ): + raise RuntimeError('remote assignment reservation deadline is invalid') + assignment_ttl = (deadline - issued).total_seconds() + if ( + not assignment_ttl.is_integer() + or not 60 <= assignment_ttl <= 7 * 24 * 60 * 60 + ): + raise RuntimeError('remote assignment reservation lifetime is invalid') + target_scan_timeout = scan_kwargs.get('timeout_sec') + if ( + isinstance(target_scan_timeout, bool) + or not isinstance(target_scan_timeout, (int, float)) + or not float(target_scan_timeout).is_integer() + or target_scan_timeout <= 0 + ): + raise RuntimeError('remote assignment target scan timeout is invalid') + persisted_upload_timeout = claim.get( + 'remote_result_upload_body_timeout_seconds' + ) + if persisted_upload_timeout is None: + persisted_upload_timeout = self.result_upload_body_timeout_seconds + if ( + isinstance(persisted_upload_timeout, bool) + or not isinstance(persisted_upload_timeout, int) + or not 30 <= persisted_upload_timeout <= 24 * 60 * 60 + ): + raise RuntimeError('remote assignment result upload timeout is invalid') + reservation = dict(claim) + reservation.pop('remote_result_upload_body_timeout_seconds', None) + return { + 'reservation': reservation, + 'deadlines': { + 'target_scan_timeout_seconds': int(target_scan_timeout), + 'result_upload_body_timeout_seconds': persisted_upload_timeout, + 'assignment_ttl_seconds': int(assignment_ttl), + 'assignment_issued_at': issued_at, + 'assignment_deadline_at': deadline_at, + }, + 'compatibility': required.as_dict(), + 'scan_kwargs': assigned_scan_kwargs, + 'event_scan_options': event_scan_options, + 'queue_policy': asdict(queue_policy), + 'limits': limits, + 'scan_policy': scan_policy, + 'execution_snapshot': snapshot, + 'execution_snapshot_sha256': remote_execution_snapshot_sha256(snapshot), + 'execution_plan': execution_plan, + } + + def __call__(self, identity, request_id, client_compatibility): + client = WorkerBuildCompatibility.from_mapping(client_compatibility) + reconciled = self._reconcile_request(identity, request_id) + if reconciled is not None: + if reconciled.get('state') != 'committed': + return None + if reconciled.get('receipt'): + return {'resolution': dict(reconciled['receipt'])} + execution_plan = reconciled.get('execution_plan') + if execution_plan is None: + if assignment_source_adapter( + reconciled['claim'].get('source') + ).assignment_flow is not _EXACT_GIT_ASSIGNMENT_FLOW: + raise ScanEventConflictError( + 'direct assignment execution plan is unavailable' + ) + execution_plan = { + 'kind': 'exact_git_v1', + 'execution_target': reconciled['claim']['target'], + 'bound_plan': reconciled.get('git_plan'), + } + return self._complete_claim( + identity, client, reconciled['claim'], + reconciled['execution_snapshot'], + snapshot_sha256=reconciled['execution_snapshot_sha256'], + recovered_execution_plan=execution_plan, + ) + profile = self.compatibility_profiles.get(tuple(client.as_dict().values())) + if profile is None: + return {'no_assignment': {'reason': 'compatibility'}} + required_build = validate_worker_build_compatibility(profile['build'], client) + + candidates = [] + for source in sorted(profile['sources']): + adapter = self.source_adapters[source] + capability = adapter.package_capability + if ( + capability.source, capability.platform, capability.planning_kind, + ) not in profile['capabilities']: + raise RuntimeError( + 'remote package capability changed after profile validation' + ) + values = vars(self.source_args[source]).copy() + central_policy = str(values.get('trufflehog_config') or '') + if ( + not central_policy or not os.path.isfile(central_policy) + or sha256_file(central_policy) + != required_build.detector_policy_sha256 + ): + raise RuntimeError('remote package detector policy does not match central authority') + values['trufflehog_config'] = PACKAGE_DETECTOR_POLICY + source_args = SimpleNamespace(**values) + source_scan_kwargs = validate_scan_kwargs( + adapter.worker_platform, + prepare_scan_options(source_args, 1, quiet=True), + ) + source_event_options = { + key: value for key, value in source_scan_kwargs.items() if key != 'token' + } + source_queue_policy = self._queue_policy(source_args) + source_limits = { + 'candidate_max_items': int(getattr(source_args, 'keycheck_candidates_per_event', 2000)), + 'candidate_max_bytes': int(getattr(source_args, 'keycheck_candidate_bytes_per_event', 2 << 20)), + } + source_scan_policy = self._scan_policy(source_args) + effective_identity, execution = remote_execution_identity( + adapter.worker_platform, source_scan_kwargs, source_event_options, + source_queue_policy, source_limits, source_scan_policy, + ) + required = ScanCompatibility( + protocol_version=required_build.protocol_version, + bundle_format_version=required_build.bundle_format_version, + platform_tag=required_build.platform_tag, + code_manifest_sha256=required_build.code_manifest_sha256, + effective_config_sha256=effective_identity, + detector_policy_sha256=required_build.detector_policy_sha256, + ) + snapshot = self._snapshot(adapter, source_args, required, execution) + candidates.append(( + adapter, source_args, source_scan_kwargs, source_event_options, + source_queue_policy, source_limits, source_scan_policy, required, + snapshot, + )) + if not candidates: + raise RuntimeError('remote compatibility profile has no central source configuration') + device_id = int(identity['device_id']) + producer = current_process_identity().as_dict() + start = int(request_id[:16], 16) % len(candidates) + ordered_candidates = candidates[start:] + candidates[:start] + multiple_sources = len(ordered_candidates) > 1 + no_work_reasons = [] + for candidate in ordered_candidates: + ( + adapter, args, scan_kwargs, event_scan_options, queue_policy, limits, + scan_policy, required, snapshot, + ) = candidate + admission_token = ( + _stable_id( + device_id, request_id, + f'admission:{adapter.queue_source}', + ) + if multiple_sources else request_id + ) + if multiple_sources: + recovered = self._reconcile_request(identity, admission_token) + if recovered is not None: + if recovered.get('state') != 'committed': + continue + if recovered.get('receipt'): + return {'resolution': dict(recovered['receipt'])} + execution_plan = recovered.get('execution_plan') + if execution_plan is None: + if assignment_source_adapter( + recovered['claim'].get('source') + ).assignment_flow is not _EXACT_GIT_ASSIGNMENT_FLOW: + raise ScanEventConflictError( + 'direct assignment execution plan is unavailable' + ) + execution_plan = { + 'kind': 'exact_git_v1', + 'execution_target': recovered['claim']['target'], + 'bound_plan': recovered.get('git_plan'), + } + return self._complete_claim( + identity, client, recovered['claim'], + recovered['execution_snapshot'], + snapshot_sha256=recovered[ + 'execution_snapshot_sha256' + ], + recovered_execution_plan=execution_plan, + ) + max_event_bytes = int(getattr( + args, 'result_bundle_max_event_bytes', 64 << 20, + )) + remote_reserve_bytes = int(getattr( + args, 'remote_assignment_reserve_bytes', 2 << 20, + )) + remote_max_active = int(getattr( + args, 'remote_assignment_max_active', 50, + )) + validate_v2_capacity_model( + int(getattr(args, 'max_active_scans', 1) or 1), + max_event_bytes, + int(getattr(args, 'projection_backlog_max_bytes', 2 << 30)), + int(getattr( + args, 'projection_backlog_headroom_bytes', + max_event_bytes * 2, + )), + ) + outcome = self.admission( + self.db_url, adapter.queue_source, adapter.worker_platform, + producer, self.supervisor_instance_id, + max_event_bytes, remote_reserve_bytes, + limits['candidate_max_items'], limits['candidate_max_bytes'], + lease_seconds=self.assignment_ttl_seconds_by_source.get( + adapter.queue_source, self.assignment_ttl_seconds, + ), + max_attempts=int(getattr( + args, 'target_retry_max_attempts', 3, + ) or 3), + capacity_limits=self._capacity(args), run_id=None, cycle_id=None, + reservation_token=admission_token, + bundle_id=_stable_id(device_id, admission_token, 'bundle'), + scan_event_id=_stable_id(device_id, admission_token, 'event'), + resolution_attempts=int(getattr( + args, 'admission_resolution_attempts', 8, + ) or 8), + resolution_seconds=float(getattr( + args, 'admission_resolution_seconds', 30, + ) or 30), + retry_delay=float(getattr( + args, 'admission_resolution_retry_delay_sec', 0.2, + ) or 0.2), + claim_order=str(getattr( + args, 'target_claim_order', 'oldest', + ) or 'oldest'), + final_cutover=True, + reserved_bundle_bytes=remote_reserve_bytes, + remote_max_active=remote_max_active, + db_factory=self.db_factory, + remote_assignment={ + 'user_id': int(identity['user_id']), + 'device_id': device_id, + 'effective_config_sha256': required.effective_config_sha256, + 'client_compat_sha256': canonical_json_sha256( + client.as_dict() + ), + 'token_sha256': str(identity['token_sha256']), + 'result_upload_body_timeout_seconds': ( + self.result_upload_body_timeout_seconds + ), + 'execution_snapshot': snapshot, + }, + ) + if outcome.claim is not None: + return self._complete_claim( + identity, client, outcome.claim, snapshot, + ) + reason = _ADMISSION_NO_WORK_REASONS.get(str( + getattr(outcome, 'reason', None) or '' + )) + if reason: + no_work_reasons.append(reason) + if no_work_reasons: + priority = ( + 'dispatch_paused', 'assignment_cap', 'capacity', 'empty_queue', + ) + return {'no_assignment': {'reason': next( + reason for reason in priority if reason in no_work_reasons + )}} + return None + + +RemoteGitAssignmentBuilder = RemoteAssignmentBuilder diff --git a/app/worker_assignment_runner.py b/app/worker_assignment_runner.py new file mode 100644 index 0000000..4500ff9 --- /dev/null +++ b/app/worker_assignment_runner.py @@ -0,0 +1,1166 @@ +"""Strict file protocol and package-local process for one remote assignment.""" + +import hashlib +import json +import os +import re +import time +from datetime import datetime, timezone + +from janitor import JanitorBudget, bounded_remove_tree, run_janitor_pass +from process_identity import current_process_identity, serialize_process_identity +from result_bundle import ( + BundleReservation, + ResultBundleReader, + bundle_ready_path, + ensure_bundle_reservation_paths, +) +from runtime_security import ( + atomic_write_private_json, + canonical_path, + durable_publish_directory, + durable_publish, + ensure_private_directory, + fsync_directory, + harden_private_directory, + harden_private_file, + private_file_ready, + read_private_json, + reject_reparse_components, + require_private_directory, + write_private_json_exclusive, +) +from scan_execution import ( + WorkerBuildCompatibility, + execute_protocol2_remote_claim, + stage_scan_result_in_scope, + validate_protocol2_remote_assignment, +) +import scanner +from worker_contracts import WorkerPhase, validate_phase_transition +from worker_package import PACKAGE_DETECTOR_POLICY + + +RUNNER_PROTOCOL_SCHEMA = 1 +RUNNER_INPUT_NAME = 'input.json' +RUNNER_EVENTS_NAME = 'events.jsonl' +RUNNER_START_NAME = 'start.json' +RUNNER_TERMINAL_NAME = 'terminal.json' +RUNNER_ROOT_PREFIX = 'worker-assignment-' +MAX_RUNNER_INPUT_BYTES = 4 * 1024 * 1024 +MAX_RUNNER_EVENT_BYTES = 64 * 1024 +MAX_RUNNER_EVENTS_BYTES = 4 * 1024 * 1024 +MAX_RUNNER_OUTCOME_BYTES = 1024 * 1024 +_GENERATION_RE = re.compile(r'^[a-f0-9]{32}$') +_ROOT_RE = re.compile(r'^worker-assignment-(0|[1-9][0-9]*)-([1-9][0-9]*)-([a-f0-9]{32})$') +_DIGEST_RE = re.compile(r'^[a-f0-9]{64}$') +_ASSIGNMENT_FIELDS = { + 'reservation', 'deadlines', 'compatibility', 'scan_kwargs', + 'event_scan_options', 'queue_policy', 'limits', 'scan_policy', + 'execution_snapshot', 'execution_snapshot_sha256', 'execution_plan', +} +_COMMIT_FIELDS = { + 'target', 'scan_event_id', 'bundle_id', 'reservation_id', + 'scan_event_hash', 'actual_bytes', 'relative_path', 'frame_count', + 'finding_count', 'error_count', 'candidate_count', 'queue_status', + 'source_failure', 'source_failure_category', + 'source_failure_auth_related', 'first_error', +} + + +class RunnerProtocolError(RuntimeError): + pass + + +class RunnerFencedError(RunnerProtocolError): + pass + + +class RunnerDeadlineElapsed(RunnerProtocolError): + pass + + +def utc_now(): + return datetime.now(timezone.utc).isoformat(timespec='milliseconds').replace('+00:00', 'Z') + + +def canonical_json_bytes(value, *, newline=False): + try: + payload = json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + allow_nan=False, + ).encode('utf-8') + except (TypeError, ValueError) as exc: + raise RunnerProtocolError('runner protocol value is not canonical JSON') from exc + return payload + (b'\n' if newline else b'') + + +def runner_root_name(slot_id, reservation_id, generation): + slot_id = int(slot_id) + reservation_id = int(reservation_id) + generation = str(generation or '') + if slot_id < 0 or reservation_id <= 0 or not _GENERATION_RE.fullmatch(generation): + raise RunnerProtocolError('runner root identity is invalid') + return f'{RUNNER_ROOT_PREFIX}{slot_id}-{reservation_id}-{generation}' + + +def runner_paths(work_root, root_name): + work_root = require_private_directory(os.path.abspath(work_root), create=False) + match = _ROOT_RE.fullmatch(str(root_name or '')) + if match is None: + raise RunnerProtocolError('runner root name is invalid') + root = os.path.join(work_root, root_name) + if os.path.dirname(os.path.abspath(root)) != os.path.abspath(work_root): + raise RunnerProtocolError('runner root escapes worker work storage') + return { + 'root': root, + 'input': os.path.join(root, RUNNER_INPUT_NAME), + 'events': os.path.join(root, RUNNER_EVENTS_NAME), + 'start': os.path.join(root, RUNNER_START_NAME), + 'terminal': os.path.join(root, RUNNER_TERMINAL_NAME), + 'bundle_root': os.path.join(root, 'bundle'), + 'scanner_work': os.path.join(root, 'scanner-work'), + } + + +def _timestamp(value, field): + if type(value) is not str: + raise RunnerProtocolError(f'runner {field} is invalid') + try: + parsed = datetime.fromisoformat(value.replace('Z', '+00:00')) + except ValueError as exc: + raise RunnerProtocolError(f'runner {field} is invalid') from exc + if parsed.tzinfo is None: + raise RunnerProtocolError(f'runner {field} is invalid') + return parsed.astimezone(timezone.utc) + + +def validate_runner_input(value): + if not isinstance(value, dict) or set(value) != { + 'schema', 'generation', 'slot_id', 'created_at', 'scan_started_at', + 'scan_deadline_at', 'watchdog_deadline_at', 'operation', + 'timeout_phase', 'assignment', + }: + raise RunnerProtocolError('runner input shape is invalid') + if value.get('schema') != RUNNER_PROTOCOL_SCHEMA: + raise RunnerProtocolError('runner input schema is invalid') + generation = str(value.get('generation') or '') + if _GENERATION_RE.fullmatch(generation) is None: + raise RunnerProtocolError('runner generation is invalid') + if type(value.get('slot_id')) is not int or value['slot_id'] < 0: + raise RunnerProtocolError('runner slot identity is invalid') + created = _timestamp(value.get('created_at'), 'created timestamp') + started = _timestamp(value.get('scan_started_at'), 'scan start timestamp') + deadline = _timestamp(value.get('scan_deadline_at'), 'scan deadline timestamp') + watchdog_deadline = _timestamp( + value.get('watchdog_deadline_at'), 'watchdog deadline timestamp', + ) + if deadline <= started or created < started: + raise RunnerProtocolError('runner scan deadline ordering is invalid') + if watchdog_deadline <= created: + raise RunnerProtocolError('runner watchdog deadline ordering is invalid') + operation = value.get('operation') + timeout_phase = value.get('timeout_phase') + if operation not in {'execute', 'timeout_bundle'}: + raise RunnerProtocolError('runner operation is invalid') + if operation == 'execute' and timeout_phase is not None: + raise RunnerProtocolError('scan runner cannot carry a timeout phase') + if operation == 'timeout_bundle': + try: + timeout_phase = WorkerPhase(timeout_phase).value + except ValueError as exc: + raise RunnerProtocolError('timeout runner phase is invalid') from exc + if timeout_phase in { + WorkerPhase.IDLE.value, WorkerPhase.CLAIMING.value, + WorkerPhase.UPLOADING.value, WorkerPhase.AWAITING_RECEIPT.value, + WorkerPhase.BACKOFF.value, WorkerPhase.DRAINING.value, + WorkerPhase.STOPPED.value, + }: + raise RunnerProtocolError('timeout runner phase is outside the scan stage') + if ( + not isinstance(value.get('assignment'), dict) + or set(value['assignment']) != _ASSIGNMENT_FIELDS + ): + raise RunnerProtocolError('runner assignment is invalid') + reservation = dict(value['assignment'].get('reservation') or {}) + if int(reservation.get('reservation_id') or 0) <= 0: + raise RunnerProtocolError('runner reservation identity is invalid') + runner_root_name( + value['slot_id'], reservation['reservation_id'], generation, + ) + return dict(value) + + +def build_runner_input( + assignment, *, generation, slot_id, scan_started_at, scan_deadline_at, + watchdog_deadline_at=None, created_at=None, operation='execute', + timeout_phase=None, +): + return validate_runner_input({ + 'schema': RUNNER_PROTOCOL_SCHEMA, + 'generation': str(generation), + 'slot_id': int(slot_id), + 'created_at': created_at or utc_now(), + 'scan_started_at': str(scan_started_at), + 'scan_deadline_at': str(scan_deadline_at), + 'watchdog_deadline_at': str(watchdog_deadline_at or scan_deadline_at), + 'operation': str(operation), + 'timeout_phase': timeout_phase, + 'assignment': dict(assignment), + }) + + +def _owner_marker(work_root, root, owner_identity, parent_identity): + relative = os.path.relpath(root, work_root) + if relative.startswith('..' + os.sep) or os.path.isabs(relative): + raise RunnerProtocolError('runner root escapes worker work storage') + owner = dict(owner_identity) + parent = dict(parent_identity) + fields = ('pid', 'creation_time', 'executable') + if any(not owner.get(field) or not parent.get(field) for field in fields): + raise RunnerProtocolError('runner process identity is incomplete') + return { + 'schema': 2, + **{f'owner_{field}': owner[field] for field in fields}, + **{f'parent_{field}': parent[field] for field in fields}, + 'created_at': datetime.now(timezone.utc).isoformat(timespec='seconds'), + 'root_kind': 'work', + 'relative_path': relative.replace(os.sep, '/'), + 'command': ['worker-assignment-runner'], + } + + +def create_runner_root(work_root, root_name, runner_input): + paths = runner_paths(work_root, root_name) + if os.path.lexists(paths['root']): + raise RunnerProtocolError('runner root already exists') + os.mkdir(paths['root'], 0o700) + harden_private_directory(paths['root']) + try: + for name in ('bundle', 'scanner-work'): + ensure_private_directory(os.path.join(paths['root'], name), reject_reparse=True) + for name in ('tmp', 'ready', 'quarantine'): + ensure_private_directory(os.path.join(paths['bundle_root'], name), reject_reparse=True) + current = serialize_process_identity(current_process_identity()) + atomic_write_private_json( + os.path.join(paths['root'], '.scanner-owner.json'), + _owner_marker(work_root, paths['root'], current, current), + ) + atomic_write_private_json( + paths['input'], validate_runner_input(runner_input), + max_bytes=MAX_RUNNER_INPUT_BYTES, + ) + with open(paths['input'], 'rb') as handle: + input_sha256 = hashlib.sha256(handle.read(MAX_RUNNER_INPUT_BYTES + 1)).hexdigest() + return paths, input_sha256 + except BaseException: + try: + bounded_remove_tree(paths['root'], JanitorBudget( + max_candidates=1, max_entries=4000, + max_bytes=512 * 1024 * 1024, max_seconds=2.0, + max_depth=64, + )) + except OSError: + pass + raise + + +def bind_runner_owner(work_root, root_name, payload_identity): + paths = runner_paths(work_root, root_name) + parent = serialize_process_identity(current_process_identity()) + atomic_write_private_json( + os.path.join(paths['root'], '.scanner-owner.json'), + _owner_marker(work_root, paths['root'], payload_identity, parent), + ) + + +def bind_transferred_runner_owner(work_root, root_name, payload_identity): + work_root = require_private_directory(os.path.abspath(work_root), create=False) + root = os.path.join(work_root, 'abandoned', root_name) + if not os.path.isdir(root): + return False + parent = serialize_process_identity(current_process_identity()) + atomic_write_private_json( + os.path.join(root, '.scanner-owner.json'), + _owner_marker(work_root, root, payload_identity, parent), + ) + return True + + +def _load_canonical_object(path, maximum, label): + reject_reparse_components(path) + if not private_file_ready(path): + raise RunnerProtocolError(f'runner {label} is not an exact private file') + with open(path, 'rb') as handle: + payload = handle.read(maximum + 1) + if len(payload) > maximum: + raise RunnerProtocolError(f'runner {label} exceeds its byte bound') + try: + value = json.loads(payload.decode('utf-8', errors='strict')) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise RunnerProtocolError(f'runner {label} is invalid JSON') from exc + if canonical_json_bytes(value, newline=True) != payload: + raise RunnerProtocolError(f'runner {label} is not canonical JSON') + return value, hashlib.sha256(payload).hexdigest() + + +def load_runner_input(path): + value, digest = _load_canonical_object(path, MAX_RUNNER_INPUT_BYTES, 'input') + return validate_runner_input(value), digest + + +def validate_runner_event(value, *, generation=None, input_sha256=None): + if not isinstance(value, dict) or set(value) != { + 'schema', 'generation', 'input_sha256', 'sequence', 'timestamp', + 'phase', 'phase_started_at', 'progress', + } or value.get('schema') != RUNNER_PROTOCOL_SCHEMA: + raise RunnerProtocolError('runner event shape is invalid') + if _GENERATION_RE.fullmatch(str(value.get('generation') or '')) is None: + raise RunnerProtocolError('runner event generation is invalid') + if generation is not None and value['generation'] != generation: + raise RunnerProtocolError('runner event generation conflicts with slot authority') + if _DIGEST_RE.fullmatch(str(value.get('input_sha256') or '')) is None: + raise RunnerProtocolError('runner event input hash is invalid') + if input_sha256 is not None and value['input_sha256'] != input_sha256: + raise RunnerProtocolError('runner event input hash conflicts with slot authority') + if type(value.get('sequence')) is not int or value['sequence'] <= 0: + raise RunnerProtocolError('runner event sequence is invalid') + _timestamp(value.get('timestamp'), 'event timestamp') + _timestamp(value.get('phase_started_at'), 'phase start timestamp') + try: + WorkerPhase(value.get('phase')) + except ValueError as exc: + raise RunnerProtocolError('runner event phase is invalid') from exc + if not isinstance(value.get('progress'), dict): + raise RunnerProtocolError('runner event progress is invalid') + return dict(value) + + +def read_runner_events( + path, *, generation, input_sha256, operation=None, after_sequence=0, +): + if not os.path.exists(path): + return [] + reject_reparse_components(path) + if not private_file_ready(path): + raise RunnerProtocolError('runner event journal is not an exact private file') + if os.path.getsize(path) > MAX_RUNNER_EVENTS_BYTES: + raise RunnerProtocolError('runner event journal exceeds its byte bound') + events = [] + previous = None + with open(path, 'rb') as handle: + while True: + payload = handle.readline(MAX_RUNNER_EVENT_BYTES + 2) + if not payload: + break + if len(payload) > MAX_RUNNER_EVENT_BYTES + 1: + raise RunnerProtocolError('runner event exceeds its byte bound') + if not payload.endswith(b'\n'): + break + try: + value = json.loads(payload.decode('utf-8', errors='strict')) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise RunnerProtocolError('runner event journal contains invalid JSON') from exc + if canonical_json_bytes(value, newline=True) != payload: + raise RunnerProtocolError('runner event is not canonical JSON') + event = validate_runner_event( + value, generation=generation, input_sha256=input_sha256, + ) + if previous is not None: + if event['sequence'] != previous['sequence'] + 1: + raise RunnerProtocolError('runner event sequence is not contiguous') + validate_phase_transition(previous['phase'], event['phase']) + previous_time = _timestamp(previous['timestamp'], 'event timestamp') + current_time = _timestamp(event['timestamp'], 'event timestamp') + if current_time < previous_time: + raise RunnerProtocolError('runner event timestamps are not monotonic') + expected_phase_start = ( + previous['phase_started_at'] + if event['phase'] == previous['phase'] else event['timestamp'] + ) + if event['phase_started_at'] != expected_phase_start: + raise RunnerProtocolError('runner phase start timestamp is inconsistent') + elif event['sequence'] != 1: + raise RunnerProtocolError('runner event journal does not begin at sequence one') + elif operation is not None and event['phase'] != ( + WorkerPhase.PREPARING.value + if operation == 'execute' else WorkerPhase.BUNDLING.value + ): + raise RunnerProtocolError('runner event journal begins with the wrong operation phase') + elif event['phase_started_at'] != event['timestamp']: + raise RunnerProtocolError('runner first phase start timestamp is inconsistent') + previous = event + if event['sequence'] > int(after_sequence or 0): + events.append(event) + return events + + +def phase_durations_from_events(events, completed_at): + completed = _timestamp(completed_at, 'outcome timestamp') + durations = {} + if not events: + return durations + for index, event in enumerate(events): + started = _timestamp(event['timestamp'], 'event timestamp') + ended = ( + _timestamp(events[index + 1]['timestamp'], 'event timestamp') + if index + 1 < len(events) else completed + ) + if ended < started: + raise RunnerProtocolError('runner duration timestamps are inconsistent') + name = event['phase'] + durations[name] = durations.get(name, 0.0) + (ended - started).total_seconds() + return {name: round(value, 6) for name, value in durations.items()} + + +def validate_terminal_against_journal(terminal, events, runner_input): + terminal = validate_generation_terminal( + terminal, generation=runner_input['generation'], + ) + if terminal['input_sha256'] != hashlib.sha256( + canonical_json_bytes(runner_input, newline=True), + ).hexdigest(): + raise RunnerProtocolError('runner terminal does not match canonical input') + if terminal['decision'] != 'completed': + return terminal + outcome = terminal['outcome'] + if not events: + raise RunnerProtocolError('completed runner has no durable phase events') + expected_sequence = events[-1]['sequence'] if events else 0 + expected_phase = events[-1]['phase'] if events else None + expected_durations = phase_durations_from_events( + events, outcome['completed_at'], + ) + if ( + outcome['last_event_sequence'] != expected_sequence + or outcome['final_phase'] != expected_phase + or outcome['phase_durations'] != expected_durations + ): + raise RunnerProtocolError('runner outcome conflicts with its durable event journal') + expected_first = ( + WorkerPhase.PREPARING.value + if runner_input['operation'] == 'execute' else WorkerPhase.BUNDLING.value + ) + if events and events[0]['phase'] != expected_first: + raise RunnerProtocolError('runner operation began with an invalid phase') + return terminal + + +def _duration_map(value): + if not isinstance(value, dict) or set(value) - {phase.value for phase in WorkerPhase}: + raise RunnerProtocolError('runner phase durations are invalid') + result = {} + for name, duration in value.items(): + if isinstance(duration, bool) or not isinstance(duration, (int, float)) or not 0 <= duration < 10 ** 9: + raise RunnerProtocolError('runner phase duration is invalid') + result[name] = round(float(duration), 6) + return result + + +def validate_runner_outcome(value, *, generation=None, input_sha256=None): + if not isinstance(value, dict) or set(value) != { + 'schema', 'generation', 'input_sha256', 'identity', 'status', + 'completed_at', 'last_event_sequence', 'final_phase', + 'phase_durations', 'bundle', 'error', + } or value.get('schema') != RUNNER_PROTOCOL_SCHEMA: + raise RunnerProtocolError('runner outcome shape is invalid') + if _GENERATION_RE.fullmatch(str(value.get('generation') or '')) is None: + raise RunnerProtocolError('runner outcome generation is invalid') + if generation is not None and value['generation'] != generation: + raise RunnerProtocolError('runner outcome generation conflicts with slot authority') + if _DIGEST_RE.fullmatch(str(value.get('input_sha256') or '')) is None: + raise RunnerProtocolError('runner outcome input hash is invalid') + if input_sha256 is not None and value['input_sha256'] != input_sha256: + raise RunnerProtocolError('runner outcome input hash conflicts with slot authority') + identity = value.get('identity') + if not isinstance(identity, dict) or set(identity) != { + 'slot_id', 'reservation_id', 'bundle_id', 'scan_event_id', + 'execution_snapshot_sha256', + }: + raise RunnerProtocolError('runner outcome identity shape is invalid') + if type(identity.get('slot_id')) is not int or identity['slot_id'] < 0 or type(identity.get('reservation_id')) is not int or identity['reservation_id'] <= 0: + raise RunnerProtocolError('runner outcome identity is invalid') + if not re.fullmatch(r'[a-f0-9]{32,64}', str(identity.get('bundle_id') or '')) or not re.fullmatch(r'[a-f0-9]{32,64}', str(identity.get('scan_event_id') or '')) or _DIGEST_RE.fullmatch(str(identity.get('execution_snapshot_sha256') or '')) is None: + raise RunnerProtocolError('runner outcome immutable identity is invalid') + if value.get('status') not in {'succeeded', 'failed'}: + raise RunnerProtocolError('runner outcome status is invalid') + _timestamp(value.get('completed_at'), 'outcome timestamp') + if type(value.get('last_event_sequence')) is not int or value['last_event_sequence'] < 0: + raise RunnerProtocolError('runner outcome event sequence is invalid') + if value.get('final_phase') is not None: + try: + WorkerPhase(value['final_phase']) + except ValueError as exc: + raise RunnerProtocolError('runner outcome final phase is invalid') from exc + value = dict(value) + value['phase_durations'] = _duration_map(value.get('phase_durations')) + if value['status'] == 'succeeded': + bundle = value.get('bundle') + if not isinstance(bundle, dict) or set(bundle) != { + 'commit', 'payload_sha256', 'ready_relative_path', + } or value.get('error') is not None: + raise RunnerProtocolError('successful runner outcome is invalid') + commit = bundle.get('commit') + if ( + not isinstance(commit, dict) or set(commit) != _COMMIT_FIELDS + or _DIGEST_RE.fullmatch(str(bundle.get('payload_sha256') or '')) is None + ): + raise RunnerProtocolError('runner bundle outcome is invalid') + if ( + type(commit.get('reservation_id')) is not int + or commit['reservation_id'] <= 0 + or not re.fullmatch(r'[a-f0-9]{32,64}', str(commit.get('bundle_id') or '')) + or not re.fullmatch(r'[a-f0-9]{32,64}', str(commit.get('scan_event_id') or '')) + or _DIGEST_RE.fullmatch(str(commit.get('scan_event_hash') or '')) is None + or any( + type(commit.get(name)) is not int or commit[name] < 0 + for name in ( + 'actual_bytes', 'frame_count', 'finding_count', + 'error_count', 'candidate_count', + ) + ) + or type(commit.get('source_failure')) is not bool + or type(commit.get('source_failure_auth_related')) is not bool + or any( + type(commit.get(name)) is not str + for name in ( + 'target', 'relative_path', 'queue_status', + 'source_failure_category', 'first_error', + ) + ) + ): + raise RunnerProtocolError('runner bundle commit is invalid') + relative = str(bundle.get('ready_relative_path') or '') + if not relative.startswith('ready/') or '\\' in relative or '..' in relative.split('/'): + raise RunnerProtocolError('runner bundle reference is invalid') + else: + error = value.get('error') + if value.get('bundle') is not None or not isinstance(error, dict) or set(error) != { + 'code', 'type', 'summary', + }: + raise RunnerProtocolError('failed runner outcome is invalid') + if not re.fullmatch(r'[a-z0-9_]{1,64}', str(error.get('code') or '')) or any(type(error.get(name)) is not str for name in ('type', 'summary')): + raise RunnerProtocolError('runner error outcome is invalid') + error['summary'] = error['summary'][:1000] + return value + + +def validate_generation_terminal(value, *, generation=None, input_sha256=None): + if not isinstance(value, dict) or set(value) != { + 'schema', 'generation', 'input_sha256', 'decision', 'decided_at', + 'reason', 'outcome', + } or value.get('schema') != RUNNER_PROTOCOL_SCHEMA: + raise RunnerProtocolError('runner terminal-generation record shape is invalid') + if _GENERATION_RE.fullmatch(str(value.get('generation') or '')) is None: + raise RunnerProtocolError('runner terminal generation is invalid') + if generation is not None and value['generation'] != generation: + raise RunnerProtocolError('runner terminal generation conflicts with slot authority') + if _DIGEST_RE.fullmatch(str(value.get('input_sha256') or '')) is None: + raise RunnerProtocolError('runner terminal input hash is invalid') + if input_sha256 is not None and value['input_sha256'] != input_sha256: + raise RunnerProtocolError('runner terminal input hash conflicts with slot authority') + _timestamp(value.get('decided_at'), 'terminal decision timestamp') + decision = value.get('decision') + if decision not in {'completed', 'fenced', 'timed_out'}: + raise RunnerProtocolError('runner terminal decision is invalid') + if decision == 'completed': + if value.get('reason') is not None: + raise RunnerProtocolError('completed runner terminal reason is invalid') + value = dict(value) + value['outcome'] = validate_runner_outcome( + value.get('outcome'), generation=value['generation'], + input_sha256=value['input_sha256'], + ) + elif ( + value.get('outcome') is not None + or type(value.get('reason')) is not str + or not re.fullmatch(r'[a-z0-9_]{1,64}', value['reason']) + ): + raise RunnerProtocolError('controller runner terminal decision is invalid') + return dict(value) + + +def load_generation_terminal(path, *, generation, input_sha256): + value, digest = _load_canonical_object( + path, MAX_RUNNER_OUTCOME_BYTES, 'terminal-generation record', + ) + return validate_generation_terminal( + value, generation=generation, input_sha256=input_sha256, + ), digest + + +def publish_generation_terminal( + path, *, generation, input_sha256, decision, reason=None, outcome=None, +): + value = validate_generation_terminal({ + 'schema': RUNNER_PROTOCOL_SCHEMA, + 'generation': str(generation), + 'input_sha256': str(input_sha256), + 'decision': str(decision), + 'decided_at': utc_now(), + 'reason': reason, + 'outcome': outcome, + }, generation=generation, input_sha256=input_sha256) + try: + write_private_json_exclusive( + path, value, max_bytes=MAX_RUNNER_OUTCOME_BYTES, + ) + return value, True + except FileExistsError: + existing, _digest = load_generation_terminal( + path, generation=generation, input_sha256=input_sha256, + ) + return existing, False + + +def publish_start_gate(path, *, generation, input_sha256, host, payload): + fields = ('pid', 'creation_time', 'executable') + value = { + 'schema': RUNNER_PROTOCOL_SCHEMA, + 'generation': str(generation), + 'input_sha256': str(input_sha256), + 'host': {field: host[field] for field in fields}, + 'payload': {field: payload[field] for field in fields}, + 'released_at': utc_now(), + } + write_private_json_exclusive(path, value) + return value + + +def wait_for_start_gate(paths, runner_input, input_sha256): + deadline = _timestamp( + runner_input['watchdog_deadline_at'], 'watchdog deadline timestamp', + ) + fields = {'pid', 'creation_time', 'executable'} + current = serialize_process_identity(current_process_identity()) + while True: + if os.path.exists(paths['terminal']): + terminal, _digest = load_generation_terminal( + paths['terminal'], generation=runner_input['generation'], + input_sha256=input_sha256, + ) + raise RunnerFencedError( + f"runner startup was closed by {terminal['decision']}" + ) + if os.path.exists(paths['start']): + value, _digest = _load_canonical_object( + paths['start'], MAX_RUNNER_EVENT_BYTES, 'startup gate', + ) + if not isinstance(value, dict) or set(value) != { + 'schema', 'generation', 'input_sha256', 'host', 'payload', + 'released_at', + } or value.get('schema') != RUNNER_PROTOCOL_SCHEMA: + raise RunnerProtocolError('runner startup gate shape is invalid') + if ( + value.get('generation') != runner_input['generation'] + or value.get('input_sha256') != input_sha256 + or not isinstance(value.get('host'), dict) + or set(value['host']) != fields + or not isinstance(value.get('payload'), dict) + or set(value['payload']) != fields + or any(value['payload'][field] != current[field] for field in fields) + ): + raise RunnerProtocolError('runner startup gate identity is invalid') + _timestamp(value.get('released_at'), 'startup gate timestamp') + return value + if datetime.now(timezone.utc) >= deadline: + raise TimeoutError('runner startup gate was not released before its deadline') + time.sleep(0.01) + + +def adopt_runner_bundle(work_root, root_name, outcome, assignment, bundle_root): + outcome = validate_runner_outcome(outcome) + if outcome['status'] != 'succeeded': + raise RunnerProtocolError('failed runner outcome has no adoptable bundle') + paths = runner_paths(work_root, root_name) + reservation = BundleReservation.from_mapping(assignment['reservation']) + identity = outcome['identity'] + root_match = _ROOT_RE.fullmatch(root_name) + if outcome['generation'] != root_match.group(3): + raise RunnerProtocolError('runner outcome generation conflicts with isolated root') + if identity != { + 'slot_id': int(root_match.group(1)), + 'reservation_id': reservation.reservation_id, + 'bundle_id': reservation.bundle_id, + 'scan_event_id': reservation.scan_event_id, + 'execution_snapshot_sha256': assignment['execution_snapshot_sha256'], + }: + raise RunnerProtocolError('runner outcome identity conflicts with assignment') + relative = outcome['bundle']['ready_relative_path'].replace('/', os.sep) + source = os.path.abspath(os.path.join(paths['bundle_root'], relative)) + if os.path.commonpath((os.path.abspath(paths['bundle_root']), source)) != os.path.abspath(paths['bundle_root']): + raise RunnerProtocolError('runner bundle reference escapes its isolated root') + metadata = ResultBundleReader( + source, max_event_bytes=reservation.declared_bytes, + ).validate() + if metadata.header != reservation.header(): + raise RunnerProtocolError('runner bundle header conflicts with assignment') + commit = outcome['bundle']['commit'] + for name in ( + 'reservation_id', 'bundle_id', 'scan_event_id', 'scan_event_hash', + 'actual_bytes', 'frame_count', 'finding_count', 'error_count', + 'candidate_count', + ): + if getattr(metadata, name) != commit.get(name): + raise RunnerProtocolError('runner bundle commit conflicts with durable bundle') + digest = hashlib.sha256() + with open(source, 'rb', buffering=0) as handle: + for block in iter(lambda: handle.read(1024 * 1024), b''): + digest.update(block) + if digest.hexdigest() != outcome['bundle']['payload_sha256']: + raise RunnerProtocolError('runner bundle payload hash is invalid') + + ensure_bundle_reservation_paths(bundle_root, reservation) + destination = bundle_ready_path(bundle_root, reservation.bundle_id) + if os.path.lexists(destination): + adopted = ResultBundleReader( + destination, max_event_bytes=reservation.declared_bytes, + ).validate() + if adopted.header != reservation.header(): + raise RunnerProtocolError('canonical ready bundle conflicts with runner outcome') + return adopted + generation = outcome['generation'] + temporary = os.path.join( + os.path.dirname(os.path.dirname(destination)), '..', 'tmp', + reservation.bundle_id[:2], + f'{reservation.bundle_id}.{generation}.adopt.partial', + ) + temporary = os.path.normpath(temporary) + descriptor = None + try: + if not os.path.lexists(temporary): + descriptor = os.open( + temporary, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), + 0o600, + ) + harden_private_file(temporary) + with os.fdopen(descriptor, 'wb', buffering=0) as target: + descriptor = None + with open(source, 'rb', buffering=0) as source_handle: + for block in iter(lambda: source_handle.read(1024 * 1024), b''): + target.write(block) + target.flush() + os.fsync(target.fileno()) + copied = ResultBundleReader( + temporary, max_event_bytes=reservation.declared_bytes, + ).validate() + if copied.header != reservation.header() or copied.scan_event_hash != metadata.scan_event_hash: + raise RunnerProtocolError('copied runner bundle failed canonical validation') + durable_publish(temporary, destination) + finally: + if descriptor is not None: + os.close(descriptor) + if os.path.exists(temporary): + os.remove(temporary) + return ResultBundleReader( + destination, max_event_bytes=reservation.declared_bytes, + ).validate() + + +def _check_terminal(path, generation, input_sha256): + if not os.path.exists(path): + return + value, _digest = load_generation_terminal( + path, generation=generation, input_sha256=input_sha256, + ) + raise RunnerFencedError( + f"runner generation was closed by {value['decision']}" + ) + + +class RunnerEventJournal: + def __init__(self, path, terminal_path, generation, input_sha256, fault=None): + self.path = path + self.terminal_path = terminal_path + self.generation = generation + self.input_sha256 = input_sha256 + self.fault = fault + self.sequence = 0 + self.phase = None + self.phase_started_at = None + self.phase_started_monotonic = None + self.durations = {} + self.events = [] + + def check_fence(self): + _check_terminal(self.terminal_path, self.generation, self.input_sha256) + + def current_timestamp(self): + now = utc_now() + if ( + self.events + and _timestamp(now, 'event timestamp') + < _timestamp(self.events[-1]['timestamp'], 'event timestamp') + ): + return self.events[-1]['timestamp'] + return now + + def emit(self, phase, progress=None): + self.check_fence() + phase = WorkerPhase(phase) + if self.phase is not None: + validate_phase_transition(self.phase, phase) + now = self.current_timestamp() + monotonic_now = time.monotonic() + measured = dict(progress or {}) + if self.phase is not None and phase != self.phase: + duration = max(0.0, monotonic_now - self.phase_started_monotonic) + self.durations[self.phase.value] = self.durations.get(self.phase.value, 0.0) + duration + measured.update({ + 'previous_phase': self.phase.value, + 'previous_duration_seconds': round(duration, 6), + }) + if phase != self.phase: + self.phase_started_at = now + self.phase_started_monotonic = monotonic_now + event = validate_runner_event({ + 'schema': RUNNER_PROTOCOL_SCHEMA, + 'generation': self.generation, + 'input_sha256': self.input_sha256, + 'sequence': self.sequence + 1, + 'timestamp': now, + 'phase': phase.value, + 'phase_started_at': self.phase_started_at, + 'progress': measured, + }, generation=self.generation, input_sha256=self.input_sha256) + payload = canonical_json_bytes(event, newline=True) + if len(payload) > MAX_RUNNER_EVENT_BYTES: + raise RunnerProtocolError('runner event exceeds its byte bound') + flags = os.O_WRONLY | os.O_APPEND | os.O_CREAT | getattr(os, 'O_BINARY', 0) + created = not os.path.lexists(self.path) + descriptor = os.open(self.path, flags, 0o600) + try: + harden_private_file(self.path) + remaining = memoryview(payload) + while remaining: + written = os.write(descriptor, remaining) + if written <= 0: + raise OSError('runner event append made no progress') + remaining = remaining[written:] + os.fsync(descriptor) + finally: + os.close(descriptor) + if created: + fsync_directory(os.path.dirname(self.path)) + self.sequence = event['sequence'] + self.phase = phase + self.events.append(event) + if self.fault is not None: + self.fault(phase.value) + return event + + def finish_durations(self, completed_at): + return phase_durations_from_events(self.events, completed_at) + + +def _outcome_identity(runner_input): + assignment = runner_input['assignment'] + reservation = assignment['reservation'] + return { + 'slot_id': runner_input['slot_id'], + 'reservation_id': int(reservation['reservation_id']), + 'bundle_id': str(reservation['bundle_id']), + 'scan_event_id': str(reservation['scan_event_id']), + 'execution_snapshot_sha256': str(assignment['execution_snapshot_sha256']), + } + + +def run_assignment(root, package_runtime, *, fault=None, bundle_fault=None): + root = require_private_directory(os.path.abspath(root), create=False) + root_name = os.path.basename(root) + work_root = os.path.dirname(root) + paths = runner_paths(work_root, root_name) + runner_input, input_sha256 = load_runner_input(paths['input']) + reservation_id = int(runner_input['assignment']['reservation']['reservation_id']) + if root_name != runner_root_name( + runner_input['slot_id'], reservation_id, runner_input['generation'], + ): + raise RunnerProtocolError('runner root identity conflicts with input') + journal = RunnerEventJournal( + paths['events'], paths['terminal'], runner_input['generation'], input_sha256, + fault=fault, + ) + identity = _outcome_identity(runner_input) + outcome = None + wait_for_start_gate(paths, runner_input, input_sha256) + try: + journal.emit( + 'preparing' if runner_input['operation'] == 'execute' else 'bundling', + {'operation': runner_input['operation']}, + ) + if set(package_runtime) != { + 'build_compatibility', 'code_manifest', 'code_manifest_sha256', + 'trufflehog_path', 'git_path', 'detector_policy_path', 'capabilities', + }: + raise RunnerProtocolError('verified runner package runtime is invalid') + local_build = WorkerBuildCompatibility.from_mapping( + package_runtime['build_compatibility'], + ) + validated = validate_protocol2_remote_assignment( + runner_input['assignment'], local_build, package_runtime['capabilities'], + ) + scan_kwargs = dict(runner_input['assignment']['scan_kwargs']) + if scan_kwargs.get('trufflehog_config') != PACKAGE_DETECTOR_POLICY: + raise RunnerProtocolError('runner detector policy reference is invalid') + + def phase(phase_value, progress=None): + journal.emit(phase_value, progress) + + def publication_fault(stage, writer): + journal.check_fence() + if bundle_fault is not None: + bundle_fault(stage, writer) + + limits = dict(runner_input['assignment']['limits']) + if runner_input['operation'] == 'timeout_bundle': + phase('bundling', { + 'reason': 'scan_stage_timeout', + 'timed_out_phase': runner_input['timeout_phase'], + }) + reservation = validated['reservation'] + started = _timestamp(runner_input['scan_started_at'], 'scan start timestamp') + result = { + 'findings': [], + 'errors': [ + f"Scan-stage deadline exceeded during {runner_input['timeout_phase']}" + ], + 'target': reservation.target, + 'scan_type': reservation.platform, + 'scan_event_id': reservation.scan_event_id, + 'scan_started_at': runner_input['scan_started_at'], + 'duration_sec': max( + 0.0, (datetime.now(timezone.utc) - started).total_seconds(), + ), + 'timestamp': datetime.now(timezone.utc).isoformat(), + 'error_class': 'timeout', + 'retryable': True, + 'scan_meta': { + 'command_timed_out': True, + 'full_stage_timeout': True, + 'timed_out_phase': runner_input['timeout_phase'], + 'scan_deadline_at': runner_input['scan_deadline_at'], + }, + } + staged = stage_scan_result_in_scope( + result, reservation, paths['bundle_root'], + runner_input['assignment']['event_scan_options'], + runner_input['assignment']['queue_policy'], + attempts=int(runner_input['assignment']['reservation'].get('attempts') or 0), + candidate_max_items=int(limits.get('candidate_max_items') or 2000), + candidate_max_bytes=int(limits.get('candidate_max_bytes') or 2 * 1024 * 1024), + require_s_drive=False, + fault=publication_fault, + diagnostic_slot_id=runner_input['slot_id'], + ) + else: + remaining = ( + _timestamp(runner_input['scan_deadline_at'], 'scan deadline timestamp') + - datetime.now(timezone.utc) + ).total_seconds() + if remaining < 1: + raise RunnerDeadlineElapsed( + 'scan-stage deadline has less than one executable second remaining', + ) + scan_kwargs['trufflehog_config'] = package_runtime['detector_policy_path'] + scanner.scan_config.trufflehog_path = package_runtime['trufflehog_path'] + scanner.scan_config.trufflehog_config = package_runtime['detector_policy_path'] + scanner.scan_config.work_dir = paths['scanner_work'] + scanner.initialize_scanner_runtime(preflight_complete=True, register_cleanup=False) + with scanner.client_scan_launch_authority( + package_runtime['code_manifest'], + expected_sha256=package_runtime['code_manifest_sha256'], + ): + staged = execute_protocol2_remote_claim( + validated, + paths['bundle_root'], + scan_kwargs, + runner_input['assignment']['event_scan_options'], + runner_input['assignment']['queue_policy'], + runner_input['assignment']['scan_policy'], + attempts=int(runner_input['assignment']['reservation'].get('attempts') or 0), + candidate_max_items=int(limits.get('candidate_max_items') or 2000), + candidate_max_bytes=int(limits.get('candidate_max_bytes') or 2 * 1024 * 1024), + require_s_drive=False, + phase_callback=phase, + bundle_fault=publication_fault, + diagnostic_slot_id=runner_input['slot_id'], + ) + journal.check_fence() + ready = bundle_ready_path( + paths['bundle_root'], runner_input['assignment']['reservation']['bundle_id'], + ) + metadata = ResultBundleReader( + ready, + max_event_bytes=int(runner_input['assignment']['reservation']['declared_bundle_bytes']), + ).validate() + staged_value = staged.as_dict() + for name in ( + 'reservation_id', 'bundle_id', 'scan_event_id', 'scan_event_hash', + 'actual_bytes', 'frame_count', 'finding_count', 'error_count', + 'candidate_count', + ): + if getattr(metadata, name) != staged_value[name]: + raise RunnerProtocolError('runner staged bundle commit is inconsistent') + digest = hashlib.sha256() + with open(ready, 'rb', buffering=0) as handle: + for block in iter(lambda: handle.read(1024 * 1024), b''): + digest.update(block) + relative = os.path.relpath(ready, paths['bundle_root']).replace(os.sep, '/') + completed_at = journal.current_timestamp() + outcome = { + 'schema': RUNNER_PROTOCOL_SCHEMA, + 'generation': runner_input['generation'], + 'input_sha256': input_sha256, + 'identity': identity, + 'status': 'succeeded', + 'completed_at': completed_at, + 'last_event_sequence': journal.sequence, + 'final_phase': journal.phase.value if journal.phase is not None else None, + 'phase_durations': journal.finish_durations(completed_at), + 'bundle': { + 'commit': staged.as_dict(), + 'payload_sha256': digest.hexdigest(), + 'ready_relative_path': relative, + }, + 'error': None, + } + except RunnerDeadlineElapsed: + publish_generation_terminal( + paths['terminal'], generation=runner_input['generation'], + input_sha256=input_sha256, decision='timed_out', + reason='scan_stage_deadline', + ) + return 1 + except BaseException as exc: + code = 'runner_fenced' if isinstance(exc, RunnerFencedError) else 'runner_timeout' if isinstance(exc, TimeoutError) else 'runner_failed' + completed_at = journal.current_timestamp() + outcome = { + 'schema': RUNNER_PROTOCOL_SCHEMA, + 'generation': runner_input['generation'], + 'input_sha256': input_sha256, + 'identity': identity, + 'status': 'failed', + 'completed_at': completed_at, + 'last_event_sequence': journal.sequence, + 'final_phase': journal.phase.value if journal.phase is not None else None, + 'phase_durations': journal.finish_durations(completed_at), + 'bundle': None, + 'error': { + 'code': code, + 'type': f'{type(exc).__module__}.{type(exc).__qualname__}', + 'summary': str(exc)[:1000], + }, + } + outcome = validate_runner_outcome( + outcome, generation=runner_input['generation'], input_sha256=input_sha256, + ) + terminal = validate_terminal_against_journal({ + 'schema': RUNNER_PROTOCOL_SCHEMA, + 'generation': runner_input['generation'], + 'input_sha256': input_sha256, + 'decision': 'completed', + 'decided_at': outcome['completed_at'], + 'reason': None, + 'outcome': outcome, + }, journal.events, runner_input) + decided, won = publish_generation_terminal( + paths['terminal'], generation=runner_input['generation'], + input_sha256=input_sha256, decision='completed', outcome=outcome, + ) + if not won or decided['decision'] != 'completed': + return 1 + return 0 if terminal['outcome']['status'] == 'succeeded' else 1 + + +def cleanup_abandoned_runner_roots( + work_root, *, minimum_age_sec=60, budget=None, active_root_names=(), +): + work_root = require_private_directory(os.path.abspath(work_root), create=False) + allowed = {canonical_path(os.path.abspath(os.sys.executable))} + if getattr(os.sys, '_base_executable', None): + allowed.add(canonical_path(os.path.abspath(os.sys._base_executable))) + return run_janitor_pass( + work_root, allowed, minimum_age_sec=minimum_age_sec, + excluded_relative_paths=tuple(active_root_names), + budget=budget or JanitorBudget( + max_candidates=8, max_entries=4000, max_bytes=512 * 1024 * 1024, + max_seconds=2.0, max_depth=64, max_enumerated=256, + ), + ) + + +def cleanup_runner_root(work_root, root_name): + paths = runner_paths(work_root, root_name) + if not os.path.exists(paths['root']): + return True + return bounded_remove_tree(paths['root'], JanitorBudget( + max_candidates=1, max_entries=4000, + max_bytes=512 * 1024 * 1024, max_seconds=2.0, max_depth=64, + )) + + +def transfer_runner_to_janitor(work_root, root_name, generation): + paths = runner_paths(work_root, root_name) + abandoned = ensure_private_directory( + os.path.join(work_root, 'abandoned'), reject_reparse=True, + ) + destination = os.path.join(abandoned, root_name) + destination_relative = f'abandoned/{root_name}' + if not os.path.exists(paths['root']): + if os.path.isdir(destination): + marker = read_private_json( + os.path.join(destination, '.scanner-owner.json'), + max_bytes=64 * 1024, + ) + intent_path = os.path.join(destination, '.janitor-transfer.json') + intent = read_private_json(intent_path, max_bytes=64 * 1024) + if ( + not isinstance(marker, dict) or marker.get('schema') != 2 + or marker.get('root_kind') != 'work' + or marker.get('relative_path') != destination_relative + or not isinstance(intent, dict) or set(intent) != { + 'schema', 'generation', 'source_relative_path', + 'destination_relative_path', 'status', 'prepared_at', + } + or intent.get('schema') != 1 + or intent.get('generation') != str(generation) + or intent.get('source_relative_path') != root_name + or intent.get('destination_relative_path') != destination_relative + or intent.get('status') not in {'prepared', 'transferred'} + ): + raise RunnerProtocolError('janitor runner transfer evidence is invalid') + if intent['status'] == 'prepared': + intent['status'] = 'transferred' + atomic_write_private_json(intent_path, intent) + return destination_relative + raise RunnerProtocolError('runner work tree is unavailable for janitor transfer') + if os.path.lexists(destination): + raise RunnerProtocolError('janitor runner destination already exists') + marker_path = os.path.join(paths['root'], '.scanner-owner.json') + marker = read_private_json(marker_path, max_bytes=64 * 1024) + if ( + not isinstance(marker, dict) or marker.get('schema') != 2 + or marker.get('root_kind') != 'work' + or marker.get('relative_path') not in {root_name, destination_relative} + ): + raise RunnerProtocolError('runner owner marker is invalid for janitor transfer') + marker = dict(marker) + marker['relative_path'] = destination_relative + atomic_write_private_json(marker_path, marker) + intent_path = os.path.join(paths['root'], '.janitor-transfer.json') + intent = { + 'schema': 1, + 'generation': str(generation), + 'source_relative_path': root_name, + 'destination_relative_path': destination_relative, + 'status': 'prepared', + 'prepared_at': utc_now(), + } + atomic_write_private_json(intent_path, intent) + durable_publish_directory(paths['root'], destination) + intent['status'] = 'transferred' + atomic_write_private_json( + os.path.join(destination, '.janitor-transfer.json'), intent, + ) + return destination_relative diff --git a/app/worker_cli.py b/app/worker_cli.py new file mode 100644 index 0000000..6464c7e --- /dev/null +++ b/app/worker_cli.py @@ -0,0 +1,1243 @@ +import argparse +import json +import os +import socket +import ssl +import subprocess +import sys +import tempfile +import threading +import time +import yaml +from datetime import datetime, timezone +from types import SimpleNamespace +from urllib.parse import urlsplit + +sys.dont_write_bytecode = True +if not sys.dont_write_bytecode: + raise RuntimeError('worker CLI could not disable bytecode writes') + +from remote_worker_client import WorkerHTTPClient, default_worker_paths +from runtime_security import ( + atomic_write_private_json, + canonical_path, + durable_unlink, + ensure_private_directory, + private_file_ready, + read_private_json, +) +from worker_contracts import WORKER_EVENT_SCHEMA +from worker_local_state import ( + DEFAULT_LOG_BYTES, + DEFAULT_LOG_FILES, + DEFAULT_RETENTION_BYTES, + DEFAULT_RETENTION_DAYS, + RETENTION_SCHEMA, + WorkerLocalState, + WorkerLocalStateError, +) +from worker_package import verify_worker_package, worker_package_manifest_sha256 +from worker_supervisor import ( + WorkerAlreadyRunning, + WorkerSupervisor, + WorkerSupervisorError, + CONTROL_SCHEMA, + PROJECTION_SCHEMA, + capture_spawned_process_identity, + classify_instance, + detached_command, + load_shutdown_receipt, + send_control_request, + spawn_detached, + terminate_spawned_process, + wait_for_startup, +) +from worker_assignment_runner import run_assignment + + +CLI_SCHEMA = 1 +CONFIG_SCHEMA = 2 +LEGACY_CONFIG_SCHEMA = 1 +LAUNCH_SCHEMA = 1 +MAX_INSTALL_CONFIG_BYTES = 64 * 1024 +EXIT_SUCCESS = 0 +EXIT_NOT_RUNNING = 3 +EXIT_STALE_OR_UNVERIFIABLE = 4 +EXIT_STARTUP_FAILED = 6 +EXIT_STOP_INCOMPLETE = 7 +EXIT_INVALID_INVOCATION = 64 + + +class InvocationError(ValueError): + pass + + +class WorkerArgumentParser(argparse.ArgumentParser): + def error(self, message): + raise InvocationError(message) + + +def _canonical(value): + return json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + allow_nan=False, + ) + + +def _print_json(value): + print(_canonical(value), flush=True) + + +def _config_path(paths): + return os.path.join(paths['state_dir'], 'control', 'worker.config.json') + + +def _validate_config(value): + legacy_fields = { + 'schema', 'server', 'token', 'parallelism', 'installed_at', + } + current_fields = { + 'schema', 'server', 'token', 'parallelism', 'installed_at', + 'retention_days', 'retention_bytes', 'log_bytes', 'log_files', + } + if not isinstance(value, dict): + raise InvocationError('installed worker configuration is invalid') + if value.get('schema') == LEGACY_CONFIG_SCHEMA: + if set(value) == legacy_fields: + value = { + **value, + 'schema': CONFIG_SCHEMA, + 'retention_days': DEFAULT_RETENTION_DAYS, + 'retention_bytes': DEFAULT_RETENTION_BYTES, + 'log_bytes': DEFAULT_LOG_BYTES, + 'log_files': DEFAULT_LOG_FILES, + } + elif set(value) == current_fields: + value = {**value, 'schema': CONFIG_SCHEMA} + else: + raise InvocationError('installed worker configuration is invalid') + elif value.get('schema') == CONFIG_SCHEMA and set(value) == current_fields: + value = dict(value) + else: + raise InvocationError('installed worker configuration is invalid') + if type(value.get('parallelism')) is not int or not 1 <= value['parallelism'] <= 128: + raise InvocationError('installed worker parallelism is invalid') + WorkerHTTPClient(value.get('server'), value.get('token')) + if not isinstance(value.get('installed_at'), str): + raise InvocationError('installed worker timestamp is invalid') + if type(value.get('retention_days')) is not int or not 1 <= value['retention_days'] <= 3650: + raise InvocationError('installed worker retention days are invalid') + if type(value.get('retention_bytes')) is not int or not 16 * 1024 * 1024 <= value['retention_bytes'] <= 1024 ** 4: + raise InvocationError('installed worker retention bytes are invalid') + if type(value.get('log_bytes')) is not int or not 64 * 1024 <= value['log_bytes'] <= 1024 ** 3: + raise InvocationError('installed worker log bytes are invalid') + if type(value.get('log_files')) is not int or not 1 <= value['log_files'] <= 20: + raise InvocationError('installed worker log file count is invalid') + return dict(value) + + +def _load_config(paths): + try: + return _validate_config(read_private_json(_config_path(paths))) + except OSError as exc: + raise InvocationError('worker is not installed; run truf-worker install') from exc + + +def _load_install_config(path): + if path == '-': + payload = sys.stdin.buffer.read(MAX_INSTALL_CONFIG_BYTES + 1) + else: + path = os.path.abspath(os.fspath(path)) + if not private_file_ready(path): + raise InvocationError('worker install configuration must be a private regular file') + with open(path, 'rb') as handle: + payload = handle.read(MAX_INSTALL_CONFIG_BYTES + 1) + if len(payload) > MAX_INSTALL_CONFIG_BYTES: + raise InvocationError('worker install configuration exceeds its byte bound') + try: + class UniqueKeyLoader(yaml.SafeLoader): + def compose_node(self, parent, index): + if self.check_event(yaml.AliasEvent): + event = self.get_event() + raise yaml.composer.ComposerError( + None, None, 'YAML aliases are not supported', event.start_mark, + ) + return super().compose_node(parent, index) + + def construct_mapping(loader, node, deep=False): + loader.flatten_mapping(node) + value = {} + for key_node, value_node in node.value: + key = loader.construct_object(key_node, deep=deep) + if key in value: + raise yaml.constructor.ConstructorError( + 'while constructing a mapping', node.start_mark, + f'duplicate key: {key}', key_node.start_mark, + ) + value[key] = loader.construct_object(value_node, deep=deep) + return value + + UniqueKeyLoader.add_constructor( + yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, construct_mapping, + ) + value = yaml.load(payload.decode('utf-8', errors='strict'), Loader=UniqueKeyLoader) + except (UnicodeDecodeError, yaml.YAMLError, TypeError, ValueError) as exc: + raise InvocationError('worker install configuration is invalid YAML') from exc + if not isinstance(value, dict) or set(value) != {'server', 'token', 'parallelism'}: + raise InvocationError( + 'worker install configuration must contain only server, token, and parallelism' + ) + return value + + +def _configuration(args, paths, *, require_explicit=False): + server = getattr(args, 'server', None) + token = getattr(args, 'token', None) + parallelism = getattr(args, 'parallelism', None) + config_path = getattr(args, 'config', None) + if config_path and any(value is not None for value in (server, token, parallelism)): + raise InvocationError('--config cannot be combined with --server, --token, or --parallelism') + if config_path: + document = _load_install_config(config_path) + server = document['server'] + token = document['token'] + parallelism = document['parallelism'] + if bool(server) != bool(token): + raise InvocationError('--server and --token must be supplied together') + if server: + value = { + 'schema': CONFIG_SCHEMA, + 'server': server, + 'token': token, + 'parallelism': 1 if parallelism is None else parallelism, + 'installed_at': datetime.now(timezone.utc).isoformat().replace('+00:00', 'Z'), + 'retention_days': int(getattr(args, 'retention_days', DEFAULT_RETENTION_DAYS)), + 'retention_bytes': int(getattr(args, 'retention_bytes', DEFAULT_RETENTION_BYTES)), + 'log_bytes': int(getattr(args, 'log_bytes', DEFAULT_LOG_BYTES)), + 'log_files': int(getattr(args, 'log_files', DEFAULT_LOG_FILES)), + } + return _validate_config(value) + if require_explicit: + raise InvocationError('--server and --token are required') + value = _load_config(paths) + if parallelism is not None: + value['parallelism'] = parallelism + return _validate_config(value) + + +def _client_args(config, paths): + return SimpleNamespace( + package_manifest=paths['package_manifest'], + state_dir=paths['state_dir'], + bundle_dir=paths['bundle_dir'], + work_dir=paths['work_dir'], + server=config['server'], + token=config['token'], + parallelism=config['parallelism'], + poll_seconds=5.0, + error_delay_seconds=15.0, + http_timeout=120, + retention_days=config['retention_days'], + retention_bytes=config['retention_bytes'], + log_bytes=config['log_bytes'], + log_files=config['log_files'], + ) + + +def _add_configuration(parser, *, required=False, allow_config=False): + parser.add_argument('--server', required=required) + parser.add_argument('--token', required=required) + parser.add_argument('--parallelism', type=int) + if allow_config: + parser.add_argument('--config') + + +def build_parser(): + parser = WorkerArgumentParser(prog='truf-worker', description='TRUF remote worker operator CLI') + commands = parser.add_subparsers(dest='command', required=True) + + install = commands.add_parser('install') + _add_configuration(install, allow_config=True) + install.add_argument('--retention-days', type=int, default=DEFAULT_RETENTION_DAYS) + install.add_argument('--retention-bytes', type=int, default=DEFAULT_RETENTION_BYTES) + install.add_argument('--log-bytes', type=int, default=DEFAULT_LOG_BYTES) + install.add_argument('--log-files', type=int, default=DEFAULT_LOG_FILES) + run = commands.add_parser('run') + _add_configuration(run) + start = commands.add_parser('start') + _add_configuration(start) + start.add_argument('--startup-timeout', type=float, default=30.0) + + stop = commands.add_parser('stop') + stop.add_argument('--timeout', type=float, default=30.0) + stop.add_argument('--json', action='store_true') + + status = commands.add_parser('status') + status.add_argument('--json', action='store_true') + + attach = commands.add_parser('attach') + attach_output = attach.add_mutually_exclusive_group() + attach_output.add_argument('--json', action='store_true') + attach_output.add_argument('--ndjson', action='store_true') + attach.add_argument('--follow-seconds', type=float) + + watch = commands.add_parser('watch') + watch.add_argument('--follow-seconds', type=float) + watch.set_defaults(json=False, ndjson=False) + + logs = commands.add_parser('logs') + logs.add_argument('--tail', type=int, default=100) + logs.add_argument('--follow', action='store_true') + logs_output = logs.add_mutually_exclusive_group() + logs_output.add_argument('--json', action='store_true') + logs_output.add_argument('--ndjson', action='store_true') + logs.add_argument('--follow-seconds', type=float, default=60.0) + + history = commands.add_parser('history') + history.add_argument('--limit', type=int, default=100) + history.add_argument('--reservation', type=int) + history_output = history.add_mutually_exclusive_group() + history_output.add_argument('--json', action='store_true') + history_output.add_argument('--ndjson', action='store_true') + + doctor = commands.add_parser('doctor') + doctor.add_argument('--json', action='store_true') + + child = commands.add_parser('_supervise', help=argparse.SUPPRESS) + child.add_argument('--launch-file', required=True) + child.add_argument('--startup-file', required=True) + child.add_argument('--launch-nonce', required=True) + runner = commands.add_parser('_assignment_runner', help=argparse.SUPPRESS) + runner.add_argument('--root', required=True) + return parser + + +def parse_args(argv=None): + values = list(sys.argv[1:] if argv is None else argv) + if values and values[0].startswith('-'): + values.insert(0, 'run') + args = build_parser().parse_args(values) + if getattr(args, 'parallelism', None) is not None and not 1 <= args.parallelism <= 128: + raise InvocationError('--parallelism must be between 1 and 128') + if getattr(args, 'startup_timeout', 1) is not None and not 0.1 <= getattr(args, 'startup_timeout', 1) <= 300: + raise InvocationError('--startup-timeout must be between 0.1 and 300 seconds') + if getattr(args, 'timeout', 1) is not None and not 0.1 <= getattr(args, 'timeout', 1) <= 3600: + raise InvocationError('--timeout must be between 0.1 and 3600 seconds') + if getattr(args, 'tail', 1) is not None and not 1 <= getattr(args, 'tail', 1) <= 10000: + raise InvocationError('--tail must be between 1 and 10000') + if getattr(args, 'limit', 1) is not None and not 1 <= getattr(args, 'limit', 1) <= 1000: + raise InvocationError('--limit must be between 1 and 1000') + follow = getattr(args, 'follow_seconds', None) + if follow is not None and not 0 <= follow <= 3600: + raise InvocationError('--follow-seconds must be between 0 and 3600') + if getattr(args, 'reservation', None) is not None and args.reservation <= 0: + raise InvocationError('--reservation must be positive') + return args + + +def _seconds(value, now=None): + if not value: + return None + try: + parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) + except ValueError: + return None + return round(((now or datetime.now(timezone.utc)) - parsed).total_seconds(), 3) + + +def _local_layout_ready(state_dir): + return all(os.path.isdir(os.path.join(state_dir, name)) for name in ( + 'control', 'events', 'history', 'diagnostics', 'logs', + )) + + +def _local_settings(paths): + try: + config = _load_config(paths) + except (InvocationError, ValueError): + config = {} + return { + 'retention_days': config.get('retention_days', DEFAULT_RETENTION_DAYS), + 'retention_bytes': config.get('retention_bytes', DEFAULT_RETENTION_BYTES), + 'log_bytes': config.get('log_bytes', DEFAULT_LOG_BYTES), + 'log_files': config.get('log_files', DEFAULT_LOG_FILES), + } + + +def _open_local_reader(paths): + return WorkerLocalState( + paths['state_dir'], read_only=True, **_local_settings(paths), + ) + + +def _directory_usage(path): + byte_count = 0 + file_count = 0 + if os.path.isdir(path): + for current, directories, files in os.walk(path, followlinks=False): + directories[:] = [ + name for name in directories + if not os.path.islink(os.path.join(current, name)) + ] + for name in files: + candidate = os.path.join(current, name) + try: + if os.path.isfile(candidate) and not os.path.islink(candidate): + byte_count += os.path.getsize(candidate) + file_count += 1 + except OSError: + continue + return {'bytes': byte_count, 'files': file_count} + + +def _root_file_usage(path): + byte_count = 0 + file_count = 0 + if os.path.isdir(path): + with os.scandir(path) as entries: + for entry in entries: + try: + if entry.is_file(follow_symlinks=False): + byte_count += entry.stat(follow_symlinks=False).st_size + file_count += 1 + except OSError: + continue + return {'bytes': byte_count, 'files': file_count} + + +def _fallback_retention(paths): + settings = _local_settings(paths) + def non_evictable(item): + return { + **item, + 'evictable_bytes': 0, + 'evictable_files': 0, + 'non_evictable_bytes': item['bytes'], + 'non_evictable_files': item['files'], + } + categories = { + 'control': non_evictable(_directory_usage( + os.path.join(paths['state_dir'], 'control'), + )), + 'state': non_evictable(_root_file_usage(paths['state_dir'])), + 'events': non_evictable({'bytes': 0, 'files': 0}), + 'history': non_evictable({'bytes': 0, 'files': 0}), + 'diagnostics': non_evictable({'bytes': 0, 'files': 0}), + 'logs': non_evictable({'bytes': 0, 'files': 0}), + 'bundles': non_evictable(_directory_usage(paths['bundle_dir'])), + 'work': non_evictable(_directory_usage(paths['work_dir'])), + } + total_bytes = sum(item['bytes'] for item in categories.values()) + return { + 'schema': RETENTION_SCHEMA, + 'maximum_age_days': settings['retention_days'], + 'maximum_bytes': settings['retention_bytes'], + 'total_bytes': total_bytes, + 'total_files': sum(item['files'] for item in categories.values()), + 'evictable_bytes': 0, + 'evictable_files': 0, + 'non_evictable_bytes': total_bytes, + 'non_evictable_files': sum(item['files'] for item in categories.values()), + 'over_limit': total_bytes > settings['retention_bytes'], + 'categories': categories, + } + + +def _slot_document(slot, now=None): + now = now or datetime.now(timezone.utc) + progress = dict(slot.get('progress') or {}) + assignment_remaining = _seconds(slot.get('assignment_deadline_at'), now) + scan_remaining = _seconds(slot.get('scan_deadline_at'), now) + measured = { + key: value + for key, value in progress.items() + if type(value) in (int, float) and key not in { + 'attempt', 'retry_after_seconds', + } + } + counters = progress.get('counters') + if isinstance(counters, dict): + measured.update({ + key: value for key, value in counters.items() + if type(value) in (int, float) + }) + return { + 'slot_id': int(slot['slot_id']), + 'sequence': int(slot.get('sequence') or 0), + 'phase': str(slot['phase']), + 'phase_age_seconds': _seconds(slot.get('phase_started_at'), now), + 'reservation_id': slot.get('reservation_id'), + 'source': slot.get('source'), + 'attempt': progress.get('attempt'), + 'scan_deadline_at': slot.get('scan_deadline_at'), + 'scan_deadline_remaining_seconds': ( + -scan_remaining if scan_remaining is not None else None + ), + 'assignment_deadline_at': slot.get('assignment_deadline_at'), + 'assignment_remaining_seconds': ( + -assignment_remaining if assignment_remaining is not None else None + ), + 'last_progress_at': slot.get('timestamp'), + 'last_progress_age_seconds': _seconds(slot.get('timestamp'), now), + 'progress': progress, + 'measured_progress': measured, + 'child_state': progress.get('child_state'), + 'idle_reason': progress.get('reason') if slot.get('phase') in {'idle', 'backoff'} else None, + 'next_claim_at': progress.get('next_claim_at'), + } + + +def status_document(paths, classification=None): + classification = classification or classify_instance(paths['state_dir']) + state = classification['state'] + instance = classification.get('instance') + package = instance.get('package') if instance else None + runtime = instance.get('runtime') if instance else None + protocol = instance.get('protocol') if instance else None + if instance is None: + try: + verified = verify_worker_package(paths['package_manifest']) + manifest = verified['manifest'] + package = { + 'schema': manifest['schema'], + 'manifest_sha256': worker_package_manifest_sha256(manifest), + 'code_manifest_sha256': verified['code_manifest_sha256'], + 'platform_tag': manifest['platform_tag'], + } + runtime = { + 'python': '.'.join(str(item) for item in sys.version_info[:3]), + 'platform': sys.platform, + 'executable': canonical_path(sys.executable), + 'mode': 'inactive', + } + protocol = { + 'worker_protocol': manifest['protocol_version'], + 'bundle_format': manifest['bundle_format_version'], + 'event': WORKER_EVENT_SCHEMA, + 'control': CONTROL_SCHEMA, + 'projection': PROJECTION_SCHEMA, + } + except (OSError, ValueError): + pass + worker = None + slots = [] + retention = None + if state in {'running', 'draining'}: + try: + snapshot = send_control_request(classification['record'], 'snapshot', {}) + worker = snapshot['worker'] + slots = [_slot_document(item) for item in snapshot['projection']['slots']] + retention = snapshot['retention'] + except (OSError, ValueError, WorkerSupervisorError, WorkerLocalStateError) as exc: + state = 'unverifiable' + classification = dict(classification) + classification['state'] = state + classification['detail'] = f'live worker snapshot failed: {type(exc).__name__}' + if state not in {'running', 'draining'} and os.path.isdir(paths['state_dir']): + local = None + if _local_layout_ready(paths['state_dir']): + try: + local = _open_local_reader(paths) + except (OSError, ValueError, WorkerLocalStateError): + local = None + projection = ( + local.snapshot() + if local is not None + else {'aggregate': {'slot_count': 0, 'phases': {}}, 'slots': []} + ) + slots = [_slot_document(item) for item in projection['slots']] + retention = ( + local.retention_usage({ + 'bundles': paths['bundle_dir'], 'work': paths['work_dir'], + }) if local is not None else _fallback_retention(paths) + ) + try: + configured_parallelism = _load_config(paths)['parallelism'] + except (InvocationError, ValueError): + configured_parallelism = None + worker = { + 'state': state, + 'parallelism': configured_parallelism, + 'slot_cap': configured_parallelism, + 'configured_slots': configured_parallelism, + 'recovery_slots': None, + 'started_at': None, + 'drain_deadline_at': None, + 'aggregate': projection['aggregate'], + } + return { + 'schema': CLI_SCHEMA, + 'command': 'status', + 'state': state, + 'detail': classification.get('detail'), + 'instance': instance, + 'package': package, + 'runtime': runtime, + 'protocol': protocol, + 'worker': worker, + 'slots': slots, + 'retention': retention, + } + + +def _human_status(document): + print(f"Worker: {document['state']} ({document.get('detail') or 'no detail'})") + instance = document.get('instance') + if instance: + print(f"Instance: {instance['instance_id']} PID: {instance['pid']} Mode: {instance['runtime']['mode']}") + print( + f"Package: {instance['package']['manifest_sha256']} " + f"Protocol: {instance['protocol']['worker_protocol']}" + ) + worker = document.get('worker') + if worker: + print( + f"Slots: {worker['aggregate']['slot_count']} Cap: {worker['slot_cap']} " + f"State: {worker['state']}" + ) + for slot in document.get('slots') or []: + reason = slot['idle_reason'] or '' + visibility = slot['progress'].get('execution_visibility') or '' + counters = _canonical(slot['measured_progress']) if slot['measured_progress'] else '{}' + print( + f"slot {slot['slot_id']}: {slot['phase']} age={slot['phase_age_seconds']}s " + f"source={slot['source'] or '-'} reservation={slot['reservation_id'] or '-'} " + f"attempt={slot['attempt'] or '-'} " + f"scan_deadline={slot['scan_deadline_at'] or '-'} " + f"scan_remaining={slot['scan_deadline_remaining_seconds']}s " + f"assignment_deadline={slot['assignment_deadline_at'] or '-'} " + f"assignment_remaining={slot['assignment_remaining_seconds']}s " + f"progress_age={slot['last_progress_age_seconds']}s " + f"next_claim_at={slot['next_claim_at'] or '-'} " + f"child_state={slot['child_state'] or '-'} counters={counters} " + f"{reason} {visibility}" + ) + retention = document.get('retention') + if retention: + print(f"Local retention: {retention['total_bytes']} bytes in {retention['total_files']} files") + + +def command_install(args, paths): + config = _configuration(args, paths, require_explicit=True) + try: + verify_worker_package(paths['package_manifest']) + except Exception as exc: + print(f'Worker installation failed: {type(exc).__name__}', file=sys.stderr) + return EXIT_STARTUP_FAILED + ensure_private_directory(paths['state_dir'], reject_reparse=True) + ensure_private_directory(os.path.join(paths['state_dir'], 'control'), reject_reparse=True) + ensure_private_directory(paths['bundle_dir'], reject_reparse=True) + ensure_private_directory(paths['work_dir'], reject_reparse=True) + atomic_write_private_json(_config_path(paths), config) + print(f"Installed worker configuration at {_config_path(paths)}") + return EXIT_SUCCESS + + +def command_run(args, paths): + config = _configuration(args, paths) + try: + supervisor = WorkerSupervisor(_client_args(config, paths), foreground=True) + return supervisor.run() + except WorkerAlreadyRunning as exc: + print(str(exc), file=sys.stderr) + return EXIT_STARTUP_FAILED + except Exception as exc: + print(f'Worker startup failed: {type(exc).__name__}', file=sys.stderr) + return EXIT_STARTUP_FAILED + + +def _launch_paths(paths, nonce): + control = ensure_private_directory(os.path.join(paths['state_dir'], 'control'), reject_reparse=True) + return ( + os.path.join(control, f'worker.launch.{nonce}.json'), + os.path.join(control, f'worker.startup.{nonce}.json'), + ) + + +def command_start(args, paths): + current = classify_instance(paths['state_dir'], remove_stale=True) + if current['state'] == 'running': + document = status_document(paths, current) + _human_status(document) + return EXIT_SUCCESS + if current['state'] in {'unverifiable', 'starting', 'draining'} or ( + current['state'] == 'stale' and not current.get('removed') + ): + _human_status(status_document(paths, current)) + return EXIT_STALE_OR_UNVERIFIABLE + config = _configuration(args, paths) + nonce = os.urandom(16).hex() + launch_file, startup_file = _launch_paths(paths, nonce) + launch = { + 'schema': LAUNCH_SCHEMA, + 'nonce': nonce, + 'server': config['server'], + 'token': config['token'], + 'parallelism': config['parallelism'], + 'retention_days': config['retention_days'], + 'retention_bytes': config['retention_bytes'], + 'log_bytes': config['log_bytes'], + 'log_files': config['log_files'], + 'package_manifest': paths['package_manifest'], + 'state_dir': paths['state_dir'], + 'bundle_dir': paths['bundle_dir'], + 'work_dir': paths['work_dir'], + } + atomic_write_private_json(launch_file, launch) + bootstrap = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'remote_worker_bootstrap.py') + process = None + spawned_identity = None + started = False + try: + process = spawn_detached(detached_command(bootstrap, launch_file, startup_file, nonce)) + spawned_identity = capture_spawned_process_identity(process) + result = wait_for_startup(startup_file, nonce, process, timeout=args.startup_timeout) + if result['outcome'] != 'ready': + print(f"Worker startup failed: {result['outcome']}", file=sys.stderr) + return EXIT_STARTUP_FAILED + verified = classify_instance(paths['state_dir']) + if ( + verified['state'] != 'running' + or verified['instance']['instance_id'] != result['instance_id'] + or result['pid'] != spawned_identity['pid'] + or verified['instance']['pid'] != spawned_identity['pid'] + or verified['instance']['process_creation_time'] != spawned_identity['creation_time'] + or canonical_path(verified['instance']['executable']) != canonical_path( + spawned_identity['executable'] + ) + ): + print('Worker startup identity could not be verified', file=sys.stderr) + return EXIT_STARTUP_FAILED + started = True + _human_status(status_document(paths, verified)) + return EXIT_SUCCESS + except (OSError, ValueError, WorkerSupervisorError) as exc: + print(f'Worker startup failed: {exc}', file=sys.stderr) + return EXIT_STARTUP_FAILED + finally: + try: + if process is not None and not started: + if spawned_identity is not None: + terminate_spawned_process(process, spawned_identity) + elif process.poll() is None: + process.terminate() + try: + process.wait(timeout=5) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=5) + classify_instance(paths['state_dir'], remove_stale=True) + finally: + for path in (launch_file, startup_file): + try: + if os.path.exists(path): + durable_unlink(path) + except OSError: + pass + + +def command_stop(args, paths): + classification = classify_instance(paths['state_dir']) + if classification['state'] == 'stopped': + value = {'schema': CLI_SCHEMA, 'command': 'stop', 'state': 'stopped', 'receipt': None} + _print_json(value) if args.json else print('Worker is not running') + return EXIT_NOT_RUNNING + if classification['state'] not in {'running', 'draining'}: + value = { + 'schema': CLI_SCHEMA, 'command': 'stop', + 'state': classification['state'], 'receipt': None, + } + _print_json(value) if args.json else print(f"Worker is {classification['state']}") + return EXIT_STALE_OR_UNVERIFIABLE + record = classification['record'] + try: + result = send_control_request( + record, 'stop', {'timeout_seconds': args.timeout}, timeout=min(args.timeout, 10), + ) + except (OSError, TimeoutError, WorkerSupervisorError): + final = status_document(paths, classify_instance(paths['state_dir'])) + value = { + 'schema': CLI_SCHEMA, 'command': 'stop', + 'state': 'control_disconnected', 'receipt': None, + 'final_status': final, + } + if args.json: + _print_json(value) + else: + print('Worker control disconnected; final state:') + _human_status(final) + return ( + EXIT_NOT_RUNNING if final['state'] == 'stopped' + else EXIT_STALE_OR_UNVERIFIABLE + ) + deadline = time.monotonic() + args.timeout + 5.0 + receipt = None + while time.monotonic() < deadline: + try: + receipt = load_shutdown_receipt(paths['state_dir'], record['instance_id']) + break + except (OSError, WorkerSupervisorError): + time.sleep(0.1) + clean = bool( + receipt is not None + and receipt.get('drained') is True + and receipt.get('exit_code') == 0 + ) + state = 'stopped' if clean else ( + 'non_drained' if receipt is not None else 'stop_timeout' + ) + value = { + 'schema': CLI_SCHEMA, 'command': 'stop', 'state': state, + 'accepted': result['accepted'], 'receipt': receipt, + } + _print_json(value) if args.json else print( + 'Worker stopped cleanly' if clean else ( + 'Worker exited without a clean drain receipt' + if receipt is not None else 'Worker stop timed out without a shutdown receipt' + ) + ) + return EXIT_SUCCESS if clean else EXIT_STOP_INCOMPLETE + + +def command_status(args, paths): + document = status_document(paths) + _print_json(document) if args.json else _human_status(document) + if document['state'] in {'running', 'starting', 'draining'}: + return EXIT_SUCCESS + if document['state'] == 'stopped': + return EXIT_NOT_RUNNING + return EXIT_STALE_OR_UNVERIFIABLE + + +def _follow_events(record, sequence, seconds, emit): + deadline = None if seconds is None else time.monotonic() + seconds + while deadline is None or time.monotonic() < deadline: + response = send_control_request( + record, 'events', {'after_sequence': sequence, 'limit': 256}, + ) + for event in response['events']: + sequence = max(sequence, int(event['sequence'])) + emit(event) + time.sleep(0.2) + return sequence + + +def command_attach(args, paths): + command_name = getattr(args, 'command', 'attach') + classification = classify_instance(paths['state_dir']) + if classification['state'] not in {'running', 'draining'}: + document = status_document(paths, classification) + if args.json: + _print_json({'schema': CLI_SCHEMA, 'command': command_name, 'status': document}) + elif args.ndjson: + _print_json({'schema': CLI_SCHEMA, 'type': 'snapshot', 'status': document}) + else: + _human_status(document) + return ( + EXIT_NOT_RUNNING if classification['state'] == 'stopped' + else EXIT_STALE_OR_UNVERIFIABLE + ) + record = classification['record'] + snapshot = status_document(paths, classification) + if args.json: + _print_json({ + 'schema': CLI_SCHEMA, + 'command': command_name, + 'status': snapshot, + }) + return EXIT_SUCCESS + if args.ndjson: + _print_json({'schema': CLI_SCHEMA, 'type': 'snapshot', 'status': snapshot}) + else: + _human_status(snapshot) + action = 'Watching' if command_name == 'watch' else 'Attached' + print(f'{action}; q, EOF, or Ctrl-C detaches without stopping the worker.') + stop = threading.Event() + if not args.ndjson: + def read_input(): + try: + while not stop.is_set(): + value = sys.stdin.read(1) + if value == '' or value.lower() == 'q': + stop.set() + return + except (EOFError, OSError): + stop.set() + threading.Thread(target=read_input, daemon=True).start() + sequence = max((slot.get('sequence', 0) for slot in snapshot.get('slots', [])), default=0) + deadline = None if args.follow_seconds is None else time.monotonic() + args.follow_seconds + next_refresh = time.monotonic() + 1.0 + disconnected = False + try: + while not stop.is_set() and (deadline is None or time.monotonic() < deadline): + response = send_control_request( + record, 'events', {'after_sequence': sequence, 'limit': 256}, + ) + for event in response['events']: + sequence = max(sequence, int(event['sequence'])) + if args.ndjson: + _print_json(event) + now = time.monotonic() + if not args.ndjson and (response['events'] or now >= next_refresh): + refreshed = status_document(paths, classification) + print('---') + _human_status(refreshed) + next_refresh = now + 1.0 + stop.wait(0.2) + except (OSError, TimeoutError, WorkerSupervisorError): + disconnected = True + final = status_document(paths, classify_instance(paths['state_dir'])) + if args.ndjson: + _print_json({'schema': CLI_SCHEMA, 'type': 'snapshot', 'status': final}) + else: + print('Worker control disconnected; final state:') + _human_status(final) + except (KeyboardInterrupt, EOFError): + pass + finally: + stop.set() + if disconnected: + return ( + EXIT_NOT_RUNNING if final['state'] == 'stopped' + else EXIT_STALE_OR_UNVERIFIABLE + ) + return EXIT_SUCCESS + + +def command_logs(args, paths): + if not os.path.isdir(paths['state_dir']): + value = {'schema': CLI_SCHEMA, 'command': 'logs', 'lines': []} + if args.follow and (args.json or args.ndjson): + print('Worker logs follow unavailable: worker state is not initialized', file=sys.stderr) + elif args.json: + _print_json(value) + elif not args.ndjson: + print('No worker logs') + return EXIT_NOT_RUNNING + layout_ready = _local_layout_ready(paths['state_dir']) + local = _open_local_reader(paths) if layout_ready else None + lines = local.log_tail(args.tail) if local is not None else [] + machine_follow = args.follow and (args.json or args.ndjson) + if machine_follow: + sequence = max(0, (local.snapshot()['sequence'] if local else 0) - args.tail) + events = local.events_after(sequence, args.tail) if local else [] + for event in events: + _print_json(event) + if events: + sequence = events[-1]['sequence'] + elif args.ndjson: + for index, line in enumerate(lines, 1): + _print_json({'schema': CLI_SCHEMA, 'type': 'log', 'index': index, 'line': line}) + elif args.json and not args.follow: + _print_json({'schema': CLI_SCHEMA, 'command': 'logs', 'lines': lines}) + else: + for line in lines: + print(line) + if not args.follow: + return EXIT_SUCCESS + classification = classify_instance(paths['state_dir']) + if classification['state'] not in {'running', 'draining'}: + if machine_follow: + print( + f"Worker logs follow ended: {classification['state']}", + file=sys.stderr, + ) + return EXIT_NOT_RUNNING + if local is None: + local = _open_local_reader(paths) + if not machine_follow: + sequence = local.snapshot()['sequence'] + try: + _follow_events( + classification['record'], sequence, args.follow_seconds, + _print_json if machine_follow else lambda event: print( + f"#{event['sequence']} slot {event['slot_id']} {event['phase']}" + ), + ) + except (OSError, TimeoutError, WorkerSupervisorError): + final = status_document(paths, classify_instance(paths['state_dir'])) + if machine_follow: + print( + f"Worker logs follow control disconnected: {final['state']}", + file=sys.stderr, + ) + else: + print('Worker control disconnected; final state:') + _human_status(final) + return ( + EXIT_NOT_RUNNING if final['state'] == 'stopped' + else EXIT_STALE_OR_UNVERIFIABLE + ) + except KeyboardInterrupt: + pass + return EXIT_SUCCESS + + +def command_history(args, paths): + if not _local_layout_ready(paths['state_dir']): + values = [] + else: + values = _open_local_reader(paths).history(args.limit, args.reservation) + if args.ndjson: + for value in values: + _print_json(value) + elif args.json: + _print_json({'schema': CLI_SCHEMA, 'command': 'history', 'assignments': values}) + elif not values: + print('No terminal assignment history') + else: + for value in values: + print( + f"reservation {value['reservation_id']}: {value['outcome']} " + f"source={value.get('source') or '-'} slot={value['slot_id']} " + f"completed={value['completed_at']} duration={value.get('duration_seconds')}s" + ) + timeline = ' -> '.join( + f"{event['phase']}#{event['sequence']}" + for event in value.get('timeline') or [] + ) + if timeline: + print(f' timeline {timeline}') + for diagnostic in value.get('diagnostics') or []: + print( + f" diagnostic {diagnostic.get('diagnostic_uid')}: " + f"{diagnostic.get('record')} available={diagnostic.get('available')}" + ) + return EXIT_SUCCESS + + +def _doctor_check(name, applicable, status, summary): + return { + 'name': name, + 'applicable': bool(applicable), + 'status': str(status), + 'summary': str(summary), + } + + +def command_doctor(args, paths): + checks = [] + package = None + installed = os.path.isfile(paths['package_manifest']) + checks.append(_doctor_check( + 'installation', True, 'ok' if installed else 'error', + 'worker package manifest is installed' if installed else 'worker package manifest is absent', + )) + try: + package = verify_worker_package(paths['package_manifest']) + checks.append(_doctor_check('package_identity', True, 'ok', 'package identity verified')) + except Exception as exc: + checks.append(_doctor_check('package_identity', True, 'error', f'package verification failed: {type(exc).__name__}')) + config = None + try: + config = _load_config(paths) + checks.append(_doctor_check('configuration', True, 'ok', 'installed configuration is valid')) + checks.append(_doctor_check('credentials', True, 'ok', 'worker credential shape is valid')) + except (InvocationError, ValueError) as exc: + checks.append(_doctor_check('configuration', True, 'error', str(exc))) + checks.append(_doctor_check('credentials', False, 'not_applicable', 'configuration is unavailable')) + path_values = (paths['state_dir'], paths['bundle_dir'], paths['work_dir']) + missing_paths = [path for path in path_values if not os.path.isdir(path)] + if missing_paths: + checks.append(_doctor_check( + 'paths', False, 'not_applicable', + 'runtime paths are not initialized; run install first', + )) + else: + probe_error = None + for path in path_values: + descriptor = None + temporary = None + try: + descriptor, temporary = tempfile.mkstemp(prefix='.doctor-', dir=path) + except OSError as exc: + probe_error = exc + break + finally: + if descriptor is not None: + os.close(descriptor) + if temporary is not None: + try: + os.unlink(temporary) + except FileNotFoundError: + pass + except OSError as exc: + probe_error = probe_error or exc + if probe_error is None: + checks.append(_doctor_check('paths', True, 'ok', 'initialized state and data paths are writable')) + else: + checks.append(_doctor_check( + 'paths', True, 'error', + f'path validation failed: {type(probe_error).__name__}', + )) + for name, key in (('git', 'git_path'), ('trufflehog', 'trufflehog_path')): + if package is None: + checks.append(_doctor_check(name, False, 'not_applicable', 'verified package is unavailable')) + continue + try: + completed = subprocess.run( + [package[key], '--version'], stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + timeout=10, check=False, + ) + status = 'ok' if completed.returncode == 0 else 'error' + checks.append(_doctor_check(name, True, status, f'{name} executable returned {completed.returncode}')) + except (OSError, subprocess.TimeoutExpired) as exc: + checks.append(_doctor_check(name, True, 'error', f'{name} validation failed: {type(exc).__name__}')) + if config is None: + checks.append(_doctor_check('server_reachability', False, 'not_applicable', 'configuration is unavailable')) + else: + parsed = urlsplit(config['server']) + try: + context = ssl.create_default_context() + with socket.create_connection((parsed.hostname, parsed.port or 443), timeout=3) as raw: + with context.wrap_socket(raw, server_hostname=parsed.hostname): + pass + checks.append(_doctor_check('server_reachability', True, 'ok', 'server TLS endpoint is reachable')) + except OSError as exc: + checks.append(_doctor_check('server_reachability', True, 'error', f'server is unreachable: {type(exc).__name__}')) + classification = classify_instance(paths['state_dir']) + status = ( + 'ok' + if classification['state'] in {'running', 'starting', 'draining', 'stopped'} + else 'error' + ) + checks.append(_doctor_check('singleton_process', True, status, classification['detail'])) + if _local_layout_ready(paths['state_dir']): + try: + local = _open_local_reader(paths) + retention = local.retention_usage({ + 'bundles': paths['bundle_dir'], 'work': paths['work_dir'], + }) + except (OSError, ValueError, WorkerLocalStateError): + retention = _fallback_retention(paths) + checks.append(_doctor_check( + 'local_state', True, 'error', 'local state is unreadable', + )) + else: + retention = _fallback_retention(paths) + checks.append(_doctor_check('retention', True, 'ok', f"{retention['total_bytes']} bytes retained")) + overall = 'ok' if all(item['status'] in {'ok', 'not_applicable'} for item in checks) else 'error' + document = { + 'schema': CLI_SCHEMA, + 'command': 'doctor', + 'overall': overall, + 'checks': checks, + 'retention': retention, + } + if args.json: + _print_json(document) + else: + print(f'Doctor: {overall}') + for check in checks: + print(f"[{check['status']}] {check['name']}: {check['summary']}") + return EXIT_SUCCESS if overall == 'ok' else EXIT_STARTUP_FAILED + + +def _load_launch(path, nonce): + value = read_private_json(path) + if not isinstance(value, dict) or set(value) != { + 'schema', 'nonce', 'server', 'token', 'parallelism', 'package_manifest', + 'state_dir', 'bundle_dir', 'work_dir', 'retention_days', + 'retention_bytes', 'log_bytes', 'log_files', + } or value.get('schema') != LAUNCH_SCHEMA or value.get('nonce') != nonce: + raise InvocationError('detached launch record is invalid') + _validate_config({ + 'schema': CONFIG_SCHEMA, + 'server': value['server'], 'token': value['token'], + 'parallelism': value['parallelism'], 'installed_at': '', + 'retention_days': value['retention_days'], + 'retention_bytes': value['retention_bytes'], + 'log_bytes': value['log_bytes'], + 'log_files': value['log_files'], + }) + return value + + +def command_supervise(args): + try: + value = _load_launch(args.launch_file, args.launch_nonce) + durable_unlink(args.launch_file) + client = SimpleNamespace( + package_manifest=value['package_manifest'], + state_dir=value['state_dir'], bundle_dir=value['bundle_dir'], + work_dir=value['work_dir'], server=value['server'], token=value['token'], + parallelism=value['parallelism'], poll_seconds=5.0, + error_delay_seconds=15.0, http_timeout=120, + retention_days=value['retention_days'], + retention_bytes=value['retention_bytes'], + log_bytes=value['log_bytes'], log_files=value['log_files'], + ) + return WorkerSupervisor(client, foreground=False).run( + startup_file=args.startup_file, launch_nonce=args.launch_nonce, + ) + except WorkerAlreadyRunning: + return EXIT_STARTUP_FAILED + except BaseException as exc: + from worker_supervisor import write_startup_result + try: + write_startup_result( + args.startup_file, args.launch_nonce, 'failed', error=type(exc).__name__, + ) + except OSError: + pass + return EXIT_STARTUP_FAILED + + +def command_assignment_runner(args, paths): + work_root = canonical_path(ensure_private_directory( + os.path.abspath(paths['work_dir']), reject_reparse=True, + )) + root = canonical_path(os.path.abspath(args.root)) + if os.path.dirname(root) != work_root or not os.path.basename(root).startswith( + 'worker-assignment-' + ): + raise InvocationError('assignment runner root is outside worker work storage') + package = verify_worker_package(paths['package_manifest']) + manifest = package.pop('manifest') + build_compatibility = package.pop('build_compatibility') + package.pop('runtime_trees') + runtime = { + **package, + 'build_compatibility': build_compatibility, + 'capabilities': tuple( + (item['source'], item['platform'], item['planning_kind']) + for item in manifest['capabilities'] + ), + } + return run_assignment(root, runtime) + + +def main(argv=None): + try: + args = parse_args(argv) + if args.command == '_supervise': + return command_supervise(args) + paths = default_worker_paths(module_path=__file__) + if args.command == '_assignment_runner': + return command_assignment_runner(args, paths) + commands = { + 'install': command_install, + 'run': command_run, + 'start': command_start, + 'stop': command_stop, + 'status': command_status, + 'attach': command_attach, + 'watch': command_attach, + 'logs': command_logs, + 'history': command_history, + 'doctor': command_doctor, + } + return commands[args.command](args, paths) + except (InvocationError, ValueError) as exc: + print(f'truf-worker: {exc}', file=sys.stderr) + return EXIT_INVALID_INVOCATION + except (WorkerSupervisorError, WorkerLocalStateError) as exc: + print(f'truf-worker: {exc}', file=sys.stderr) + return EXIT_STALE_OR_UNVERIFIABLE + except KeyboardInterrupt: + return 130 + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/app/worker_contracts.py b/app/worker_contracts.py new file mode 100644 index 0000000..ae435d0 --- /dev/null +++ b/app/worker_contracts.py @@ -0,0 +1,1165 @@ +import base64 +import binascii +import hashlib +import hmac +import json +import math +import re +from dataclasses import dataclass +from datetime import datetime, timezone +from enum import Enum + + +WORKER_EVENT_SCHEMA = 1 +DIAGNOSTIC_SCHEMA = 1 +WORKER_EVENT_TYPE = 'slot.phase' +PROGRESS_OUTBOX_SCHEMA = 1 +PROGRESS_OUTBOX_RELATIVE_PATH = 'control/progress-outbox.json' + +MAX_DIAGNOSTIC_BODY_BYTES = 16 * 1024 +MAX_DIAGNOSTIC_LOG_BYTES = 32 * 1024 +MAX_DIAGNOSTIC_ENVELOPE_BYTES = 64 * 1024 +MAX_DIAGNOSTICS_PER_ASSIGNMENT = 32 +MAX_DIAGNOSTIC_AGGREGATE_BYTES = 256 * 1024 +LEGACY_ERROR_MATERIAL_BYTES = 2 * 1024 +DIAGNOSTIC_PROJECTION_VERSION = 1 +LEGACY_DIAGNOSTIC_FALLBACK_TIMESTAMP = '1970-01-01T00:00:00.000Z' + +_SHA256_RE = re.compile(r'^[0-9a-f]{64}$') +_SCAN_EVENT_ID_RE = re.compile(r'^[0-9a-f]{32,64}$') +_SOURCE_RE = re.compile(r'^[a-z0-9][a-z0-9_.-]{0,63}$') +_CODE_RE = re.compile(r'^[A-Za-z0-9][A-Za-z0-9._:-]{0,255}$') +_UTC_TIMESTAMP_RE = re.compile( + r'^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{1,6})?Z$' +) + + +class WorkerContractError(ValueError): + def __init__(self, category): + self.category = category + super().__init__('worker contract value is invalid') + + +class WorkerPhase(str, Enum): + IDLE = 'idle' + CLAIMING = 'claiming' + ASSIGNED = 'assigned' + WAITING_PERMIT = 'waiting_permit' + PREPARING = 'preparing' + RESOLVING = 'resolving' + DOWNLOADING = 'downloading' + CLONING = 'cloning' + SCANNING = 'scanning' + FILTERING = 'filtering' + CLEANING = 'cleaning' + BUNDLING = 'bundling' + UPLOADING = 'uploading' + AWAITING_RECEIPT = 'awaiting_receipt' + BACKOFF = 'backoff' + DRAINING = 'draining' + STOPPED = 'stopped' + + +CANONICAL_WORKER_PHASES = tuple(phase.value for phase in WorkerPhase) + +# Same-phase events carry coarse progress. They are accepted by +# validate_phase_transition without obscuring the state-changing edges here. +ALLOWED_PHASE_TRANSITIONS = { + WorkerPhase.IDLE: frozenset(( + WorkerPhase.CLAIMING, WorkerPhase.DRAINING, WorkerPhase.STOPPED, + )), + WorkerPhase.CLAIMING: frozenset(( + WorkerPhase.ASSIGNED, WorkerPhase.IDLE, WorkerPhase.BACKOFF, + WorkerPhase.DRAINING, + )), + WorkerPhase.ASSIGNED: frozenset(( + WorkerPhase.PREPARING, WorkerPhase.WAITING_PERMIT, + WorkerPhase.BACKOFF, WorkerPhase.IDLE, + WorkerPhase.UPLOADING, WorkerPhase.DRAINING, + )), + WorkerPhase.WAITING_PERMIT: frozenset(( + WorkerPhase.RESOLVING, WorkerPhase.DOWNLOADING, + WorkerPhase.CLONING, WorkerPhase.SCANNING, + WorkerPhase.UPLOADING, WorkerPhase.IDLE, WorkerPhase.DRAINING, + )), + WorkerPhase.PREPARING: frozenset(( + WorkerPhase.WAITING_PERMIT, WorkerPhase.RESOLVING, + WorkerPhase.SCANNING, WorkerPhase.CLEANING, + WorkerPhase.BUNDLING, WorkerPhase.UPLOADING, WorkerPhase.IDLE, + WorkerPhase.DRAINING, + )), + WorkerPhase.RESOLVING: frozenset(( + WorkerPhase.DOWNLOADING, WorkerPhase.CLONING, WorkerPhase.SCANNING, + WorkerPhase.FILTERING, WorkerPhase.CLEANING, WorkerPhase.BUNDLING, + WorkerPhase.UPLOADING, WorkerPhase.IDLE, WorkerPhase.WAITING_PERMIT, + WorkerPhase.DRAINING, + )), + WorkerPhase.DOWNLOADING: frozenset(( + WorkerPhase.SCANNING, WorkerPhase.FILTERING, WorkerPhase.CLEANING, + WorkerPhase.BUNDLING, WorkerPhase.UPLOADING, WorkerPhase.IDLE, + WorkerPhase.WAITING_PERMIT, WorkerPhase.DRAINING, + )), + WorkerPhase.CLONING: frozenset(( + WorkerPhase.SCANNING, WorkerPhase.CLEANING, WorkerPhase.BUNDLING, + WorkerPhase.UPLOADING, WorkerPhase.IDLE, WorkerPhase.WAITING_PERMIT, + WorkerPhase.DRAINING, + )), + WorkerPhase.SCANNING: frozenset(( + WorkerPhase.RESOLVING, WorkerPhase.DOWNLOADING, WorkerPhase.CLONING, + WorkerPhase.FILTERING, WorkerPhase.CLEANING, WorkerPhase.BUNDLING, + WorkerPhase.UPLOADING, WorkerPhase.IDLE, WorkerPhase.WAITING_PERMIT, + WorkerPhase.DRAINING, + )), + WorkerPhase.FILTERING: frozenset(( + WorkerPhase.CLEANING, WorkerPhase.BUNDLING, WorkerPhase.UPLOADING, + WorkerPhase.IDLE, WorkerPhase.WAITING_PERMIT, WorkerPhase.DRAINING, + )), + WorkerPhase.CLEANING: frozenset(( + WorkerPhase.BUNDLING, WorkerPhase.UPLOADING, WorkerPhase.IDLE, + WorkerPhase.WAITING_PERMIT, WorkerPhase.DRAINING, + )), + WorkerPhase.BUNDLING: frozenset(( + WorkerPhase.UPLOADING, WorkerPhase.BACKOFF, WorkerPhase.IDLE, + WorkerPhase.WAITING_PERMIT, WorkerPhase.DRAINING, + )), + WorkerPhase.UPLOADING: frozenset(( + WorkerPhase.AWAITING_RECEIPT, WorkerPhase.BACKOFF, WorkerPhase.DRAINING, + )), + WorkerPhase.AWAITING_RECEIPT: frozenset(( + WorkerPhase.IDLE, WorkerPhase.BACKOFF, WorkerPhase.DRAINING, + )), + WorkerPhase.BACKOFF: frozenset(( + WorkerPhase.IDLE, WorkerPhase.CLAIMING, WorkerPhase.UPLOADING, + WorkerPhase.AWAITING_RECEIPT, WorkerPhase.DRAINING, + )), + WorkerPhase.DRAINING: frozenset((WorkerPhase.STOPPED,)), + WorkerPhase.STOPPED: frozenset(), +} + + +class DiagnosticKind(str, Enum): + PROVIDER_HTTP = 'provider_http' + SCANNER_PROCESS = 'scanner_process' + EXCEPTION = 'exception' + STORAGE = 'storage' + PROTOCOL = 'protocol' + ASSIGNMENT = 'assignment' + + +class DiagnosticCategory(str, Enum): + AUTHORIZATION = 'authorization' + RATE_LIMIT = 'rate_limit' + NOT_FOUND = 'not_found' + NETWORK = 'network' + TIMEOUT = 'timeout' + PROVIDER = 'provider' + SCANNER = 'scanner' + STORAGE = 'storage' + PROTOCOL = 'protocol' + ASSIGNMENT_EXPIRED = 'assignment_expired' + INTERNAL = 'internal' + + +class AssignmentOutcome(str, Enum): + ACCEPTED = 'accepted' + PREBUNDLE_FAILED = 'prebundle_failed' + EXPIRED = 'expired' + UNFINISHED = 'unfinished' + + +class ScanOutcome(str, Enum): + CLEAN = 'clean' + FOUND = 'found' + DEGRADED = 'degraded' + ERROR = 'error' + SKIPPED = 'skipped' + UNAVAILABLE = 'unavailable' + + +class MaterialEncoding(str, Enum): + TEXT = 'text' + BASE64 = 'base64' + + +@dataclass(frozen=True, slots=True) +class WorkerEvent: + schema: int + sequence: int + timestamp: str + instance_id: str + slot_id: int + reservation_id: int | None + source: str | None + type: str + phase: WorkerPhase + phase_started_at: str + scan_deadline_at: str | None + assignment_deadline_at: str | None + progress: dict + + +@dataclass(frozen=True, slots=True) +class DiagnosticMaterial: + encoding: MaterialEncoding + head: str + tail: str | None + original_size: int + stored_size: int + sha256: str + truncated: bool + + +@dataclass(frozen=True, slots=True) +class DiagnosticHTTPContext: + operation: str + status_code: int + content_type: str | None + request_id: str | None + body: DiagnosticMaterial | None + headers: DiagnosticMaterial | None = None + + +@dataclass(frozen=True, slots=True) +class DiagnosticProcessContext: + name: str + exit_code: int | None + signal: int | None + timed_out: bool + stdout: DiagnosticMaterial | None + stderr: DiagnosticMaterial | None + + +@dataclass(frozen=True, slots=True) +class DiagnosticExceptionContext: + type: str + message: str + fingerprint: str + + +@dataclass(frozen=True, slots=True) +class DiagnosticEnvelope: + schema: int + diagnostic_uid: str + occurrence_id: str + reservation_id: int + scan_event_id: str | None + slot_id: int + source: str + phase: WorkerPhase + kind: DiagnosticKind + category: DiagnosticCategory + code: str + summary: str + retryable: bool + attempt: int + assignment_outcome: AssignmentOutcome | None + scan_outcome: ScanOutcome | None + occurred_at: str + captured_at: str + received_at: str | None + http: DiagnosticHTTPContext | None + process: DiagnosticProcessContext | None + exception: DiagnosticExceptionContext | None + + +_WORKER_EVENT_FIELDS = frozenset(( + 'schema', 'sequence', 'timestamp', 'instance_id', 'slot_id', + 'reservation_id', 'source', 'type', 'phase', 'phase_started_at', + 'scan_deadline_at', 'assignment_deadline_at', 'progress', +)) +_MATERIAL_FIELDS = frozenset(( + 'encoding', 'head', 'tail', 'original_size', 'stored_size', 'sha256', + 'truncated', +)) +_HTTP_FIELDS = frozenset(( + 'operation', 'status_code', 'content_type', 'request_id', 'body', 'headers', +)) +_LEGACY_HTTP_FIELDS = _HTTP_FIELDS - {'headers'} +_PROCESS_FIELDS = frozenset(( + 'name', 'exit_code', 'signal', 'timed_out', 'stdout', 'stderr', +)) +_EXCEPTION_FIELDS = frozenset(('type', 'message', 'fingerprint')) +_DIAGNOSTIC_FIELDS = frozenset(( + 'schema', 'diagnostic_uid', 'occurrence_id', 'reservation_id', + 'scan_event_id', 'slot_id', 'source', 'phase', 'kind', 'category', 'code', + 'summary', 'retryable', 'attempt', 'assignment_outcome', 'scan_outcome', + 'occurred_at', 'captured_at', 'received_at', 'http', 'process', 'exception', +)) + + +def _canonical_json(value): + try: + return json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + allow_nan=False, + ).encode('ascii') + except (TypeError, ValueError, UnicodeError) as exc: + raise WorkerContractError('json') from exc + + +def _strict_json(payload, *, maximum=None): + if type(payload) is not bytes or not payload: + raise WorkerContractError('json') + if maximum is not None and len(payload) > maximum: + raise WorkerContractError('bounds') + + def reject_duplicate(pairs): + result = {} + for key, value in pairs: + if key in result: + raise WorkerContractError('duplicate_field') + result[key] = value + return result + + try: + value = json.loads( + payload.decode('utf-8', errors='strict'), + object_pairs_hook=reject_duplicate, + parse_constant=lambda _value: (_ for _ in ()).throw( + WorkerContractError('constant') + ), + ) + except WorkerContractError: + raise + except (UnicodeDecodeError, json.JSONDecodeError, TypeError, ValueError) as exc: + raise WorkerContractError('json') from exc + if not isinstance(value, dict): + raise WorkerContractError('shape') + if not hmac.compare_digest(_canonical_json(value), payload): + raise WorkerContractError('canonical') + return value + + +def _enum(value, kind, field): + try: + return kind(value) + except (TypeError, ValueError) as exc: + raise WorkerContractError(field) from exc + + +def _integer(value, field, *, minimum=None, optional=False): + if optional and value is None: + return None + if type(value) is not int or (minimum is not None and value < minimum): + raise WorkerContractError(field) + return value + + +def _string( + value, field, *, optional=False, empty=False, maximum=None, pattern=None, +): + if optional and value is None: + return None + if ( + not isinstance(value, str) + or '\x00' in value + or (not empty and not value.strip()) + or (maximum is not None and len(value) > maximum) + or (pattern is not None and pattern.fullmatch(value) is None) + ): + raise WorkerContractError(field) + return value + + +def validate_utc_timestamp(value, field='timestamp', *, optional=False): + if optional and value is None: + return None + if not isinstance(value, str) or _UTC_TIMESTAMP_RE.fullmatch(value) is None: + raise WorkerContractError(field) + try: + datetime.fromisoformat(value[:-1] + '+00:00') + except ValueError as exc: + raise WorkerContractError(field) from exc + return value + + +def _timestamp_value(value): + return datetime.fromisoformat(value[:-1] + '+00:00') + + +def _validate_json_value(value): + if value is None or type(value) in (bool, int, str): + return + if type(value) is float: + if not math.isfinite(value): + raise WorkerContractError('progress') + return + if isinstance(value, list): + for item in value: + _validate_json_value(item) + return + if isinstance(value, dict): + for key, item in value.items(): + if not isinstance(key, str): + raise WorkerContractError('progress') + _validate_json_value(item) + return + raise WorkerContractError('progress') + + +def _worker_event_value(event): + if not isinstance(event, WorkerEvent): + raise WorkerContractError('event_type') + return { + 'schema': event.schema, + 'sequence': event.sequence, + 'timestamp': event.timestamp, + 'instance_id': event.instance_id, + 'slot_id': event.slot_id, + 'reservation_id': event.reservation_id, + 'source': event.source, + 'type': event.type, + 'phase': event.phase.value if isinstance(event.phase, WorkerPhase) else event.phase, + 'phase_started_at': event.phase_started_at, + 'scan_deadline_at': event.scan_deadline_at, + 'assignment_deadline_at': event.assignment_deadline_at, + 'progress': event.progress, + } + + +def _normalize_worker_event(value): + if not isinstance(value, dict) or set(value) != _WORKER_EVENT_FIELDS: + raise WorkerContractError('shape') + if type(value.get('schema')) is not int or value['schema'] != WORKER_EVENT_SCHEMA: + raise WorkerContractError('schema') + timestamp = validate_utc_timestamp(value.get('timestamp')) + phase_started_at = validate_utc_timestamp(value.get('phase_started_at'), 'phase_started_at') + if _timestamp_value(phase_started_at) > _timestamp_value(timestamp): + raise WorkerContractError('phase_started_at') + progress = value.get('progress') + if not isinstance(progress, dict): + raise WorkerContractError('progress') + _validate_json_value(progress) + return WorkerEvent( + schema=WORKER_EVENT_SCHEMA, + sequence=_integer(value.get('sequence'), 'sequence', minimum=1), + timestamp=timestamp, + instance_id=_string(value.get('instance_id'), 'instance_id'), + slot_id=_integer(value.get('slot_id'), 'slot_id', minimum=0), + reservation_id=_integer( + value.get('reservation_id'), 'reservation_id', minimum=1, optional=True, + ), + source=_string(value.get('source'), 'source', optional=True), + type=( + value.get('type') + if value.get('type') == WORKER_EVENT_TYPE + else (_ for _ in ()).throw(WorkerContractError('type')) + ), + phase=_enum(value.get('phase'), WorkerPhase, 'phase'), + phase_started_at=phase_started_at, + scan_deadline_at=validate_utc_timestamp( + value.get('scan_deadline_at'), 'scan_deadline_at', optional=True, + ), + assignment_deadline_at=validate_utc_timestamp( + value.get('assignment_deadline_at'), 'assignment_deadline_at', optional=True, + ), + progress=progress, + ) + + +def validate_phase_transition(previous, current, *, allow_same_phase=True): + previous = _enum(previous, WorkerPhase, 'previous_phase') + current = _enum(current, WorkerPhase, 'phase') + if current == previous and allow_same_phase: + return current + if current not in ALLOWED_PHASE_TRANSITIONS[previous]: + raise WorkerContractError('phase_transition') + return current + + +def validate_worker_event_sequence(events, *, previous_sequence=None): + sequence = _integer( + previous_sequence, 'previous_sequence', minimum=0, optional=True, + ) + phases = {} + normalized = [] + for event in events: + current = _normalize_worker_event(_worker_event_value(event)) + if sequence is not None and current.sequence <= sequence: + raise WorkerContractError('sequence') + key = (current.instance_id, current.slot_id) + if key in phases: + validate_phase_transition(phases[key], current.phase) + phases[key] = current.phase + sequence = current.sequence + normalized.append(current) + return tuple(normalized) + + +def encode_worker_event(event): + normalized = _normalize_worker_event(_worker_event_value(event)) + return _canonical_json(_worker_event_value(normalized)) + + +def decode_worker_event(payload): + return _normalize_worker_event(_strict_json(payload)) + + +def encode_worker_events_ndjson(events): + normalized = validate_worker_event_sequence(events) + return b''.join(encode_worker_event(event) + b'\n' for event in normalized) + + +def decode_worker_events_ndjson(payload, *, previous_sequence=None): + if type(payload) is not bytes: + raise WorkerContractError('ndjson') + if not payload: + return () + if not payload.endswith(b'\n') or b'\r' in payload: + raise WorkerContractError('ndjson') + events = tuple(decode_worker_event(line) for line in payload[:-1].split(b'\n')) + return validate_worker_event_sequence(events, previous_sequence=previous_sequence) + + +def _material_value(material): + if not isinstance(material, DiagnosticMaterial): + raise WorkerContractError('material_type') + return { + 'encoding': ( + material.encoding.value + if isinstance(material.encoding, MaterialEncoding) + else material.encoding + ), + 'head': material.head, + 'tail': material.tail, + 'original_size': material.original_size, + 'stored_size': material.stored_size, + 'sha256': material.sha256, + 'truncated': material.truncated, + } + + +def _decode_material_segment(value, encoding, field): + if not isinstance(value, str): + raise WorkerContractError(field) + if encoding is MaterialEncoding.TEXT: + return value.encode('utf-8') + try: + decoded = base64.b64decode(value.encode('ascii'), validate=True) + except (UnicodeEncodeError, binascii.Error, ValueError) as exc: + raise WorkerContractError(field) from exc + if base64.b64encode(decoded).decode('ascii') != value: + raise WorkerContractError(field) + return decoded + + +def _normalize_material(value): + if not isinstance(value, dict) or set(value) != _MATERIAL_FIELDS: + raise WorkerContractError('material_shape') + encoding = _enum(value.get('encoding'), MaterialEncoding, 'material_encoding') + head = _string(value.get('head'), 'material_head', empty=True) + tail = _string(value.get('tail'), 'material_tail', optional=True) + head_bytes = _decode_material_segment(head, encoding, 'material_head') + tail_bytes = ( + _decode_material_segment(tail, encoding, 'material_tail') + if tail is not None else b'' + ) + original_size = _integer(value.get('original_size'), 'original_size', minimum=0) + stored_size = _integer(value.get('stored_size'), 'stored_size', minimum=0) + if stored_size != len(head_bytes) + len(tail_bytes) or original_size < stored_size: + raise WorkerContractError('material_size') + truncated = value.get('truncated') + if type(truncated) is not bool: + raise WorkerContractError('truncated') + if truncated != (original_size > stored_size) or (tail is not None and not truncated): + raise WorkerContractError('truncated') + digest = value.get('sha256') + if not isinstance(digest, str) or _SHA256_RE.fullmatch(digest) is None: + raise WorkerContractError('sha256') + if not truncated and not hmac.compare_digest( + hashlib.sha256(head_bytes).hexdigest(), digest, + ): + raise WorkerContractError('sha256') + return DiagnosticMaterial( + encoding=encoding, + head=head, + tail=tail, + original_size=original_size, + stored_size=stored_size, + sha256=digest, + truncated=truncated, + ) + + +def _encode_material_parts(parts): + if any(b'\x00' in part for part in parts): + return MaterialEncoding.BASE64, tuple( + base64.b64encode(part).decode('ascii') for part in parts + ) + try: + text = tuple(part.decode('utf-8', errors='strict') for part in parts) + except UnicodeDecodeError: + return MaterialEncoding.BASE64, tuple( + base64.b64encode(part).decode('ascii') for part in parts + ) + return MaterialEncoding.TEXT, text + + +def make_diagnostic_material(payload, *, maximum, head_tail=False): + if type(payload) is not bytes: + raise WorkerContractError('material_type') + if type(maximum) is not int or maximum < 0: + raise WorkerContractError('bounds') + original_size = len(payload) + truncated = original_size > maximum + if not truncated: + parts = (payload,) + elif head_tail and maximum: + head_size = (maximum + 1) // 2 + tail_size = maximum - head_size + parts = (payload[:head_size], payload[-tail_size:] if tail_size else b'') + else: + parts = (payload[:maximum],) + encoding, encoded = _encode_material_parts(parts) + tail = encoded[1] if len(encoded) == 2 and encoded[1] else None + stored_size = sum(len(part) for part in parts if part) + return DiagnosticMaterial( + encoding=encoding, + head=encoded[0], + tail=tail, + original_size=original_size, + stored_size=stored_size, + sha256=hashlib.sha256(payload).hexdigest(), + truncated=truncated, + ) + + +def make_diagnostic_material_from_chunks(chunks, *, maximum, head_tail=False): + if type(maximum) is not int or maximum < 0: + raise WorkerContractError('bounds') + digest = hashlib.sha256() + total = 0 + complete = bytearray() + head_size = (maximum + 1) // 2 if head_tail else maximum + tail_size = maximum - head_size if head_tail else 0 + head = bytearray() + tail = bytearray() + for chunk in chunks: + if type(chunk) is not bytes: + raise WorkerContractError('material_type') + digest.update(chunk) + total += len(chunk) + if len(complete) <= maximum: + complete.extend(chunk) + if len(complete) > maximum: + complete.clear() + if len(head) < head_size: + head.extend(chunk[:head_size - len(head)]) + if tail_size: + tail.extend(chunk) + if len(tail) > tail_size: + del tail[:-tail_size] + if total <= maximum: + return make_diagnostic_material( + bytes(complete), maximum=maximum, head_tail=head_tail, + ) + parts = (bytes(head), bytes(tail)) if tail_size else (bytes(head),) + encoding, encoded = _encode_material_parts(parts) + return DiagnosticMaterial( + encoding=encoding, + head=encoded[0], + tail=(encoded[1] if len(encoded) == 2 and encoded[1] else None), + original_size=total, + stored_size=sum(len(part) for part in parts), + sha256=digest.hexdigest(), + truncated=True, + ) + + +def make_body_material(payload): + return make_diagnostic_material( + payload, maximum=MAX_DIAGNOSTIC_BODY_BYTES, head_tail=False, + ) + + +def make_log_material(payload, *, maximum=MAX_DIAGNOSTIC_LOG_BYTES): + if type(maximum) is not int or not 0 <= maximum <= MAX_DIAGNOSTIC_LOG_BYTES: + raise WorkerContractError('bounds') + return make_diagnostic_material(payload, maximum=maximum, head_tail=True) + + +def diagnostic_material_bytes(material): + normalized = _normalize_material(_material_value(material)) + head = _decode_material_segment(normalized.head, normalized.encoding, 'material_head') + tail = ( + _decode_material_segment(normalized.tail, normalized.encoding, 'material_tail') + if normalized.tail is not None else b'' + ) + return head + tail + + +def _http_value(context): + if not isinstance(context, DiagnosticHTTPContext): + raise WorkerContractError('http_type') + value = { + 'operation': context.operation, + 'status_code': context.status_code, + 'content_type': context.content_type, + 'request_id': context.request_id, + 'body': _material_value(context.body) if context.body is not None else None, + } + if context.headers is not None: + value['headers'] = _material_value(context.headers) + return value + + +def _normalize_http(value): + if ( + not isinstance(value, dict) + or set(value) not in (_HTTP_FIELDS, _LEGACY_HTTP_FIELDS) + ): + raise WorkerContractError('http_shape') + status = _integer(value.get('status_code'), 'http_status', minimum=100) + if status > 599: + raise WorkerContractError('http_status') + body = value.get('body') + body = _normalize_material(body) if body is not None else None + headers = value.get('headers') + headers = _normalize_material(headers) if headers is not None else None + if body is not None and ( + body.stored_size > MAX_DIAGNOSTIC_BODY_BYTES or body.tail is not None + ): + raise WorkerContractError('body_bounds') + if headers is not None and ( + headers.stored_size > MAX_DIAGNOSTIC_BODY_BYTES + or headers.tail is not None + ): + raise WorkerContractError('header_bounds') + return DiagnosticHTTPContext( + operation=_string( + value.get('operation'), 'http_operation', maximum=128, + pattern=_CODE_RE, + ), + status_code=status, + content_type=_string( + value.get('content_type'), 'content_type', optional=True, maximum=256, + ), + request_id=_string( + value.get('request_id'), 'request_id', optional=True, maximum=512, + ), + body=body, + headers=headers, + ) + + +def _process_value(context): + if not isinstance(context, DiagnosticProcessContext): + raise WorkerContractError('process_type') + return { + 'name': context.name, + 'exit_code': context.exit_code, + 'signal': context.signal, + 'timed_out': context.timed_out, + 'stdout': _material_value(context.stdout) if context.stdout is not None else None, + 'stderr': _material_value(context.stderr) if context.stderr is not None else None, + } + + +def _normalize_process(value): + if not isinstance(value, dict) or set(value) != _PROCESS_FIELDS: + raise WorkerContractError('process_shape') + stdout = value.get('stdout') + stderr = value.get('stderr') + stdout = _normalize_material(stdout) if stdout is not None else None + stderr = _normalize_material(stderr) if stderr is not None else None + if sum(item.stored_size for item in (stdout, stderr) if item is not None) > MAX_DIAGNOSTIC_LOG_BYTES: + raise WorkerContractError('log_bounds') + timed_out = value.get('timed_out') + if type(timed_out) is not bool: + raise WorkerContractError('timed_out') + return DiagnosticProcessContext( + name=_string(value.get('name'), 'process_name', maximum=256), + exit_code=_integer(value.get('exit_code'), 'exit_code', optional=True), + signal=_integer(value.get('signal'), 'signal', minimum=1, optional=True), + timed_out=timed_out, + stdout=stdout, + stderr=stderr, + ) + + +def _exception_value(context): + if not isinstance(context, DiagnosticExceptionContext): + raise WorkerContractError('exception_type') + return { + 'type': context.type, + 'message': context.message, + 'fingerprint': context.fingerprint, + } + + +def _normalize_exception(value): + if not isinstance(value, dict) or set(value) != _EXCEPTION_FIELDS: + raise WorkerContractError('exception_shape') + return DiagnosticExceptionContext( + type=_string(value.get('type'), 'exception_type', maximum=512), + message=_string( + value.get('message'), 'exception_message', empty=True, maximum=4096, + ), + fingerprint=_string( + value.get('fingerprint'), 'exception_fingerprint', maximum=512, + ), + ) + + +def _diagnostic_value(envelope): + if not isinstance(envelope, DiagnosticEnvelope): + raise WorkerContractError('diagnostic_type') + return { + 'schema': envelope.schema, + 'diagnostic_uid': envelope.diagnostic_uid, + 'occurrence_id': envelope.occurrence_id, + 'reservation_id': envelope.reservation_id, + 'scan_event_id': envelope.scan_event_id, + 'slot_id': envelope.slot_id, + 'source': envelope.source, + 'phase': envelope.phase.value if isinstance(envelope.phase, WorkerPhase) else envelope.phase, + 'kind': envelope.kind.value if isinstance(envelope.kind, DiagnosticKind) else envelope.kind, + 'category': ( + envelope.category.value + if isinstance(envelope.category, DiagnosticCategory) + else envelope.category + ), + 'code': envelope.code, + 'summary': envelope.summary, + 'retryable': envelope.retryable, + 'attempt': envelope.attempt, + 'assignment_outcome': ( + envelope.assignment_outcome.value + if isinstance(envelope.assignment_outcome, AssignmentOutcome) + else envelope.assignment_outcome + ), + 'scan_outcome': ( + envelope.scan_outcome.value + if isinstance(envelope.scan_outcome, ScanOutcome) + else envelope.scan_outcome + ), + 'occurred_at': envelope.occurred_at, + 'captured_at': envelope.captured_at, + 'received_at': envelope.received_at, + 'http': _http_value(envelope.http) if envelope.http is not None else None, + 'process': _process_value(envelope.process) if envelope.process is not None else None, + 'exception': ( + _exception_value(envelope.exception) + if envelope.exception is not None else None + ), + } + + +def _diagnostic_uid(value): + identity = dict(value) + identity.pop('diagnostic_uid', None) + identity.pop('received_at', None) + return hashlib.sha256(_canonical_json(identity)).hexdigest() + + +def _normalize_diagnostic(value, *, verify_uid=True): + if not isinstance(value, dict) or set(value) != _DIAGNOSTIC_FIELDS: + raise WorkerContractError('shape') + if type(value.get('schema')) is not int or value['schema'] != DIAGNOSTIC_SCHEMA: + raise WorkerContractError('schema') + uid = value.get('diagnostic_uid') + if not isinstance(uid, str) or _SHA256_RE.fullmatch(uid) is None: + raise WorkerContractError('diagnostic_uid') + retryable = value.get('retryable') + if type(retryable) is not bool: + raise WorkerContractError('retryable') + occurred_at = validate_utc_timestamp(value.get('occurred_at'), 'occurred_at') + captured_at = validate_utc_timestamp(value.get('captured_at'), 'captured_at') + if _timestamp_value(captured_at) < _timestamp_value(occurred_at): + raise WorkerContractError('captured_at') + http = value.get('http') + process = value.get('process') + exception = value.get('exception') + http = _normalize_http(http) if http is not None else None + process = _normalize_process(process) if process is not None else None + exception = _normalize_exception(exception) if exception is not None else None + kind = _enum(value.get('kind'), DiagnosticKind, 'kind') + required_context = { + DiagnosticKind.PROVIDER_HTTP: http, + DiagnosticKind.SCANNER_PROCESS: process, + DiagnosticKind.EXCEPTION: exception, + }.get(kind, True) + if required_context is None: + raise WorkerContractError('diagnostic_context') + scan_event_id = _string( + value.get('scan_event_id'), 'scan_event_id', optional=True, + ) + if scan_event_id is not None and _SCAN_EVENT_ID_RE.fullmatch(scan_event_id) is None: + raise WorkerContractError('scan_event_id') + normalized = DiagnosticEnvelope( + schema=DIAGNOSTIC_SCHEMA, + diagnostic_uid=uid, + occurrence_id=_string( + value.get('occurrence_id'), 'occurrence_id', maximum=512, + ), + reservation_id=_integer(value.get('reservation_id'), 'reservation_id', minimum=1), + scan_event_id=scan_event_id, + slot_id=_integer(value.get('slot_id'), 'slot_id', minimum=0), + source=_string( + value.get('source'), 'source', maximum=64, pattern=_SOURCE_RE, + ), + phase=_enum(value.get('phase'), WorkerPhase, 'phase'), + kind=kind, + category=_enum(value.get('category'), DiagnosticCategory, 'category'), + code=_string( + value.get('code'), 'code', maximum=256, pattern=_CODE_RE, + ), + summary=_string(value.get('summary'), 'summary'), + retryable=retryable, + attempt=_integer(value.get('attempt'), 'attempt', minimum=1), + assignment_outcome=( + _enum(value.get('assignment_outcome'), AssignmentOutcome, 'assignment_outcome') + if value.get('assignment_outcome') is not None else None + ), + scan_outcome=( + _enum(value.get('scan_outcome'), ScanOutcome, 'scan_outcome') + if value.get('scan_outcome') is not None else None + ), + occurred_at=occurred_at, + captured_at=captured_at, + received_at=validate_utc_timestamp( + value.get('received_at'), 'received_at', optional=True, + ), + http=http, + process=process, + exception=exception, + ) + if verify_uid and not hmac.compare_digest( + normalized.diagnostic_uid, _diagnostic_uid(_diagnostic_value(normalized)), + ): + raise WorkerContractError('diagnostic_uid') + return normalized + + +def build_diagnostic_envelope( + *, occurrence_id, reservation_id, scan_event_id, slot_id, source, phase, + kind, category, code, summary, retryable, attempt, assignment_outcome, + scan_outcome, occurred_at, captured_at, received_at=None, http=None, + process=None, exception=None, +): + envelope = DiagnosticEnvelope( + schema=DIAGNOSTIC_SCHEMA, + diagnostic_uid='0' * 64, + occurrence_id=occurrence_id, + reservation_id=reservation_id, + scan_event_id=scan_event_id, + slot_id=slot_id, + source=source, + phase=phase, + kind=kind, + category=category, + code=code, + summary=summary, + retryable=retryable, + attempt=attempt, + assignment_outcome=assignment_outcome, + scan_outcome=scan_outcome, + occurred_at=occurred_at, + captured_at=captured_at, + received_at=received_at, + http=http, + process=process, + exception=exception, + ) + normalized = _normalize_diagnostic(_diagnostic_value(envelope), verify_uid=False) + value = _diagnostic_value(normalized) + value['diagnostic_uid'] = _diagnostic_uid(value) + return _normalize_diagnostic(value) + + +def diagnostic_uid_for(envelope): + normalized = _normalize_diagnostic(_diagnostic_value(envelope), verify_uid=False) + return _diagnostic_uid(_diagnostic_value(normalized)) + + +def encode_diagnostic_envelope(envelope): + normalized = _normalize_diagnostic(_diagnostic_value(envelope)) + payload = _canonical_json(_diagnostic_value(normalized)) + if len(payload) > MAX_DIAGNOSTIC_ENVELOPE_BYTES: + raise WorkerContractError('envelope_bounds') + return payload + + +def decode_diagnostic_envelope(payload): + value = _strict_json(payload, maximum=MAX_DIAGNOSTIC_ENVELOPE_BYTES) + return _normalize_diagnostic(value) + + +def validate_diagnostic_envelopes(envelopes): + normalized = tuple( + _normalize_diagnostic(_diagnostic_value(envelope)) for envelope in envelopes + ) + if len(normalized) > MAX_DIAGNOSTICS_PER_ASSIGNMENT: + raise WorkerContractError('diagnostic_count') + total = sum(len(encode_diagnostic_envelope(item)) + 1 for item in normalized) + if total > MAX_DIAGNOSTIC_AGGREGATE_BYTES: + raise WorkerContractError('diagnostic_aggregate') + return normalized + + +def encode_diagnostic_envelopes_ndjson(envelopes): + normalized = validate_diagnostic_envelopes(envelopes) + return b''.join(encode_diagnostic_envelope(item) + b'\n' for item in normalized) + + +def decode_diagnostic_envelopes_ndjson(payload): + if type(payload) is not bytes: + raise WorkerContractError('ndjson') + if not payload: + return () + if ( + len(payload) > MAX_DIAGNOSTIC_AGGREGATE_BYTES + or not payload.endswith(b'\n') + or b'\r' in payload + ): + raise WorkerContractError('diagnostic_aggregate') + envelopes = tuple( + decode_diagnostic_envelope(line) for line in payload[:-1].split(b'\n') + ) + return validate_diagnostic_envelopes(envelopes) + + +def _legacy_timestamp(value): + try: + parsed = datetime.fromisoformat(str(value).replace('Z', '+00:00')) + except (TypeError, ValueError) as exc: + raise WorkerContractError('timestamp') from exc + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.astimezone(timezone.utc).isoformat( + timespec='milliseconds' + ).replace('+00:00', 'Z') + + +def _legacy_summary(raw): + summary = raw[:256].decode('utf-8', errors='ignore').strip() + return ( + 'legacy E-frame scan error' + if not summary or '\x00' in summary else summary + ) + + +def build_legacy_error_frame_diagnostics( + *, reservation_id, scan_event_id, slot_id, source, timestamp, errors, + retryable=False, attempt=1, +): + values = [str(error).encode('utf-8') for error in errors] + if not values: + return () + occurred_at = _legacy_timestamp( + timestamp or LEGACY_DIAGNOSTIC_FALLBACK_TIMESTAMP + ) + projected = [] + individual = min(len(values), MAX_DIAGNOSTICS_PER_ASSIGNMENT) + if len(values) > MAX_DIAGNOSTICS_PER_ASSIGNMENT: + individual -= 1 + for index, raw in enumerate(values[:individual]): + digest = hashlib.sha256(raw).hexdigest() + projected.append(build_diagnostic_envelope( + occurrence_id=f'{scan_event_id}:legacy-e:{index}:{digest}', + reservation_id=reservation_id, + scan_event_id=scan_event_id, + slot_id=slot_id, + source=source, + phase=WorkerPhase.SCANNING, + kind=DiagnosticKind.SCANNER_PROCESS, + category=DiagnosticCategory.SCANNER, + code='legacy.error_frame', + summary=_legacy_summary(raw), + retryable=bool(retryable), + attempt=attempt, + assignment_outcome=AssignmentOutcome.ACCEPTED, + scan_outcome=ScanOutcome.ERROR, + occurred_at=occurred_at, + captured_at=occurred_at, + process=DiagnosticProcessContext( + name='legacy-bundle-error-frame', + exit_code=None, + signal=None, + timed_out=False, + stdout=None, + stderr=make_log_material( + raw, maximum=LEGACY_ERROR_MATERIAL_BYTES, + ), + ), + exception=DiagnosticExceptionContext( + type='truf.diagnostic.LegacyEFrameProjection', + message=( + 'canonical diagnostic projected from the exact persisted ' + 'legacy E-frame' + ), + fingerprint=digest, + ), + )) + if individual < len(values): + remaining = values[individual:] + material = make_diagnostic_material_from_chunks( + (raw + b'\n' for raw in remaining), + maximum=LEGACY_ERROR_MATERIAL_BYTES, + head_tail=True, + ) + projected.append(build_diagnostic_envelope( + occurrence_id=( + f'{scan_event_id}:legacy-e-aggregate:{individual}:' + f'{material.sha256}' + ), + reservation_id=reservation_id, + scan_event_id=scan_event_id, + slot_id=slot_id, + source=source, + phase=WorkerPhase.SCANNING, + kind=DiagnosticKind.SCANNER_PROCESS, + category=DiagnosticCategory.SCANNER, + code='legacy.error_frame_aggregate', + summary=f'{len(remaining)} additional legacy E-frame scan errors', + retryable=bool(retryable), + attempt=attempt, + assignment_outcome=AssignmentOutcome.ACCEPTED, + scan_outcome=ScanOutcome.ERROR, + occurred_at=occurred_at, + captured_at=occurred_at, + process=DiagnosticProcessContext( + name='legacy-bundle-error-frame-aggregate', + exit_code=None, + signal=None, + timed_out=False, + stdout=None, + stderr=material, + ), + exception=DiagnosticExceptionContext( + type='truf.diagnostic.LegacyEFrameAggregateProjection', + message=( + 'bounded canonical aggregate projected from remaining exact ' + 'persisted legacy E-frames' + ), + fingerprint=material.sha256, + ), + )) + return validate_diagnostic_envelopes(projected) + + +def ordered_diagnostic_uid_set_sha256(diagnostics): + uids = [] + for diagnostic in diagnostics: + uid = ( + diagnostic.diagnostic_uid + if isinstance(diagnostic, DiagnosticEnvelope) + else diagnostic.get('diagnostic_uid') + if isinstance(diagnostic, dict) else diagnostic + ) + if not isinstance(uid, str) or _SHA256_RE.fullmatch(uid) is None: + raise WorkerContractError('diagnostic_uid') + uids.append(uid) + if len(uids) != len(set(uids)): + raise WorkerContractError('diagnostic_uid') + return hashlib.sha256(_canonical_json(uids)).hexdigest() + + +worker_event_to_json = encode_worker_event +worker_event_from_json = decode_worker_event +worker_events_to_ndjson = encode_worker_events_ndjson +worker_events_from_ndjson = decode_worker_events_ndjson +diagnostic_envelope_to_json = encode_diagnostic_envelope +diagnostic_envelope_from_json = decode_diagnostic_envelope +diagnostic_envelopes_to_ndjson = encode_diagnostic_envelopes_ndjson +diagnostic_envelopes_from_ndjson = decode_diagnostic_envelopes_ndjson diff --git a/app/worker_local_state.py b/app/worker_local_state.py new file mode 100644 index 0000000..a610589 --- /dev/null +++ b/app/worker_local_state.py @@ -0,0 +1,1396 @@ +import hashlib +import json +import math +import os +import re +import stat +import threading +import time +from collections import deque +from datetime import datetime, timezone + +from runtime_security import ( + atomic_write_private_json, + durable_replace, + durable_unlink, + ensure_private_directory, + fsync_directory, + harden_private_file, + private_file_ready, + read_private_json, + reject_reparse_components, + require_private_directory, +) +from worker_contracts import ( + WORKER_EVENT_SCHEMA, + WORKER_EVENT_TYPE, + PROGRESS_OUTBOX_RELATIVE_PATH, + PROGRESS_OUTBOX_SCHEMA, + WorkerEvent, + WorkerPhase, + decode_diagnostic_envelope, + decode_worker_event, + diagnostic_material_bytes, + encode_diagnostic_envelope, + encode_worker_event, + validate_utc_timestamp, + validate_phase_transition, +) + + +STATUS_SCHEMA = 1 +HISTORY_SCHEMA = 2 +LEGACY_HISTORY_SCHEMA = 1 +DIAGNOSTIC_ARCHIVE_SCHEMA = 1 +RETENTION_SCHEMA = 1 +MAX_EVENT_LINE_BYTES = 256 * 1024 +MAX_HISTORY_LINE_BYTES = 2 * 1024 * 1024 +MAX_LOG_LINE_BYTES = 256 * 1024 +DEFAULT_LOG_BYTES = 2 * 1024 * 1024 +DEFAULT_LOG_FILES = 5 +DEFAULT_RETENTION_DAYS = 30 +DEFAULT_RETENTION_BYTES = 1024 * 1024 * 1024 +DEFAULT_EVENT_SEGMENT_BYTES = 8 * 1024 * 1024 +DEFAULT_HISTORY_SEGMENT_BYTES = 8 * 1024 * 1024 +EVENT_CURSOR_STRIDE = 128 + +_EVENT_SEGMENT_RE = re.compile( + r'^worker-events\.(0|[1-9][0-9]*)-(0|[1-9][0-9]*)\.jsonl$' +) +_HISTORY_SEGMENT_RE = re.compile(r'^worker-history\.(0|[1-9][0-9]*)\.jsonl$') + +_HISTORY_FIELDS = frozenset({ + 'schema', 'history_id', 'instance_id', 'slot_id', 'reservation_id', + 'source', 'outcome', 'receipt', 'started_at', 'completed_at', + 'duration_seconds', 'first_sequence', 'last_sequence', 'diagnostics', + 'timeline', 'phase_durations', +}) +_TIMELINE_FIELDS = frozenset({ + 'sequence', 'timestamp', 'instance_id', 'phase', 'progress', +}) +_LEGACY_TIMELINE_FIELDS = frozenset({'sequence', 'timestamp', 'phase'}) +_DIAGNOSTIC_REFERENCE_FIELDS = frozenset({ + 'diagnostic_uid', 'record', 'artifacts', +}) +_ARTIFACT_FIELDS = frozenset({ + 'path', 'sha256', 'original_sha256', 'size', 'original_size', 'encoding', + 'truncated', +}) +_SHA256 = re.compile(r'^[0-9a-f]{64}$') + + +class WorkerLocalStateError(RuntimeError): + pass + + +def _progress_cursor_value(path): + value = read_private_json(path, max_bytes=64 * 1024) + if ( + not isinstance(value, dict) + or set(value) != {'schema', 'sequence'} + or value.get('schema') != PROGRESS_OUTBOX_SCHEMA + or type(value.get('sequence')) is not int + or value['sequence'] < 0 + ): + raise WorkerLocalStateError('progress outbox cursor is invalid') + return value['sequence'] + + +def prepare_progress_outbox_cursor(state_root, *, create=True): + root = os.path.abspath(state_root) + ensure_private_directory( + os.path.join(root, 'control'), reject_reparse=True, + ) + old_path = os.path.join(root, 'progress-outbox.json') + path = os.path.join(root, *PROGRESS_OUTBOX_RELATIVE_PATH.split('/')) + old_exists = os.path.exists(old_path) + new_exists = os.path.exists(path) + old_sequence = _progress_cursor_value(old_path) if old_exists else None + new_sequence = _progress_cursor_value(path) if new_exists else None + if old_exists and new_exists: + if old_sequence != new_sequence: + raise WorkerLocalStateError( + 'conflicting progress outbox cursors require operator recovery' + ) + durable_unlink(old_path) + return path, new_sequence + if old_exists: + atomic_write_private_json(path, { + 'schema': PROGRESS_OUTBOX_SCHEMA, + 'sequence': old_sequence, + }, max_bytes=64 * 1024) + durable_unlink(old_path) + return path, old_sequence + if new_exists: + return path, new_sequence + if create: + atomic_write_private_json(path, { + 'schema': PROGRESS_OUTBOX_SCHEMA, + 'sequence': 0, + }, max_bytes=64 * 1024) + return path, 0 + return path, None + + +def utc_now(): + return datetime.now(timezone.utc).isoformat(timespec='milliseconds').replace('+00:00', 'Z') + + +def _canonical_json(value): + return json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + allow_nan=False, + ).encode('ascii') + + +def _append_private(path, payload, maximum): + if type(payload) is not bytes or not payload.endswith(b'\n') or len(payload) > maximum: + raise WorkerLocalStateError('append payload is invalid') + parent = ensure_private_directory(os.path.dirname(os.path.abspath(path)), reject_reparse=True) + existed = os.path.lexists(path) + if existed: + reject_reparse_components(path) + if not private_file_ready(path): + raise WorkerLocalStateError('append target is not private') + flags = os.O_WRONLY | os.O_APPEND | os.O_CREAT + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + descriptor = os.open(path, flags, 0o600) + try: + if not existed: + harden_private_file(path) + with os.fdopen(descriptor, 'ab', buffering=0) as handle: + descriptor = None + handle.write(payload) + os.fsync(handle.fileno()) + fsync_directory(parent) + finally: + if descriptor is not None: + os.close(descriptor) + + +def _write_private_bytes(path, payload, maximum=64 * 1024 * 1024): + if type(payload) is not bytes or len(payload) > maximum: + raise WorkerLocalStateError('artifact payload is invalid') + parent = ensure_private_directory(os.path.dirname(os.path.abspath(path)), reject_reparse=True) + temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.tmp' + descriptor = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + try: + with os.fdopen(descriptor, 'wb') as handle: + descriptor = None + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + harden_private_file(temporary) + durable_replace(temporary, path) + finally: + if descriptor is not None: + os.close(descriptor) + try: + if os.path.exists(temporary): + os.remove(temporary) + except OSError: + pass + if not private_file_ready(path): + raise WorkerLocalStateError('artifact is not private after publication') + fsync_directory(parent) + + +def _event_value(event): + return json.loads(encode_worker_event(event).decode('ascii')) + + +def _relative_path(value, field): + if not isinstance(value, str) or not value or '\\' in value or value.startswith('/'): + raise WorkerLocalStateError(f'history {field} is invalid') + if any(part in ('', '.', '..') for part in value.split('/')): + raise WorkerLocalStateError(f'history {field} is invalid') + return value + + +def _history_timestamp(value, field, optional=False): + if optional and value is None: + return None + try: + return validate_utc_timestamp(value, field) + except ValueError as exc: + raise WorkerLocalStateError(f'history {field} is invalid') from exc + + +def validate_history_record(value): + if not isinstance(value, dict): + raise WorkerLocalStateError('terminal history shape is invalid') + schema = value.get('schema') + legacy_without_instance = _HISTORY_FIELDS - {'instance_id'} + fields = frozenset(value) + if schema == LEGACY_HISTORY_SCHEMA and fields in { + _HISTORY_FIELDS, legacy_without_instance, + }: + legacy = True + value = dict(value) + value.setdefault('instance_id', 'legacy-instance-unavailable') + elif schema == HISTORY_SCHEMA and fields == _HISTORY_FIELDS: + legacy = False + value = dict(value) + else: + raise WorkerLocalStateError('terminal history shape is invalid') + if type(schema) is not int: + raise WorkerLocalStateError('terminal history schema is invalid') + for field in ('history_id', 'instance_id', 'outcome'): + if not isinstance(value.get(field), str) or not value[field] or len(value[field]) > 256: + raise WorkerLocalStateError(f'terminal history {field} is invalid') + if type(value.get('slot_id')) is not int or value['slot_id'] < 0: + raise WorkerLocalStateError('terminal history slot identity is invalid') + if type(value.get('reservation_id')) is not int or value['reservation_id'] <= 0: + raise WorkerLocalStateError('terminal history reservation identity is invalid') + source = value.get('source') + if source is not None and (not isinstance(source, str) or not source): + raise WorkerLocalStateError('terminal history source is invalid') + if not isinstance(value.get('receipt'), dict): + raise WorkerLocalStateError('terminal history receipt is invalid') + try: + _canonical_json(value['receipt']) + except (TypeError, ValueError) as exc: + raise WorkerLocalStateError('terminal history receipt is invalid') from exc + started_at = _history_timestamp(value.get('started_at'), 'started_at', optional=True) + completed_at = _history_timestamp(value.get('completed_at'), 'completed_at') + duration = value.get('duration_seconds') + if duration is not None and ( + type(duration) not in (int, float) or not math.isfinite(duration) or duration < 0 + ): + raise WorkerLocalStateError('terminal history duration is invalid') + first = value.get('first_sequence') + last = value.get('last_sequence') + if first is not None and (type(first) is not int or first <= 0): + raise WorkerLocalStateError('terminal history first sequence is invalid') + if last is not None and (type(last) is not int or last <= 0): + raise WorkerLocalStateError('terminal history last sequence is invalid') + if first is not None and last is not None and last < first: + raise WorkerLocalStateError('terminal history sequence range is invalid') + diagnostics = value.get('diagnostics') + if not isinstance(diagnostics, list) or len(diagnostics) > 32: + raise WorkerLocalStateError('terminal history diagnostics are invalid') + normalized_diagnostics = [] + for reference in diagnostics: + if not isinstance(reference, dict) or set(reference) != _DIAGNOSTIC_REFERENCE_FIELDS: + raise WorkerLocalStateError('terminal history diagnostic reference is invalid') + uid = reference.get('diagnostic_uid') + if not isinstance(uid, str) or _SHA256.fullmatch(uid) is None: + raise WorkerLocalStateError('terminal history diagnostic identity is invalid') + record_path = _relative_path(reference.get('record'), 'diagnostic record path') + artifacts = reference.get('artifacts') + if not isinstance(artifacts, dict) or set(artifacts) != {'body', 'stdout', 'stderr'}: + raise WorkerLocalStateError('terminal history diagnostic artifacts are invalid') + normalized_artifacts = {} + for name, artifact in artifacts.items(): + if artifact is None: + normalized_artifacts[name] = None + continue + if not isinstance(artifact, dict) or set(artifact) != _ARTIFACT_FIELDS: + raise WorkerLocalStateError('terminal history diagnostic artifact is invalid') + if ( + _SHA256.fullmatch(str(artifact.get('sha256') or '')) is None + or _SHA256.fullmatch(str(artifact.get('original_sha256') or '')) is None + or type(artifact.get('size')) is not int or artifact['size'] < 0 + or type(artifact.get('original_size')) is not int or artifact['original_size'] < artifact['size'] + or artifact.get('encoding') not in {'text', 'base64'} + or type(artifact.get('truncated')) is not bool + ): + raise WorkerLocalStateError('terminal history diagnostic artifact fields are invalid') + normalized_artifacts[name] = { + **artifact, + 'path': _relative_path(artifact.get('path'), 'diagnostic artifact path'), + } + normalized_diagnostics.append({ + 'diagnostic_uid': uid, + 'record': record_path, + 'artifacts': normalized_artifacts, + }) + timeline = value.get('timeline') + if not isinstance(timeline, list) or len(timeline) > 1000: + raise WorkerLocalStateError('terminal history timeline is invalid') + normalized_timeline = [] + previous_sequence = 0 + for event in timeline: + if not isinstance(event, dict): + raise WorkerLocalStateError('terminal history timeline event is invalid') + event_fields = frozenset(event) + if legacy and event_fields == _LEGACY_TIMELINE_FIELDS: + legacy_event = True + elif event_fields == _TIMELINE_FIELDS: + legacy_event = False + else: + raise WorkerLocalStateError('terminal history timeline event is invalid') + sequence = event.get('sequence') + if type(sequence) is not int or sequence <= previous_sequence: + raise WorkerLocalStateError('terminal history timeline sequence is invalid') + previous_sequence = sequence + event_instance = value['instance_id'] if legacy_event else event.get('instance_id') + if not isinstance(event_instance, str) or not event_instance: + raise WorkerLocalStateError('terminal history timeline instance is invalid') + try: + phase = WorkerPhase(event.get('phase')).value + except (TypeError, ValueError) as exc: + raise WorkerLocalStateError('terminal history timeline phase is invalid') from exc + event_progress = {} if legacy_event else event.get('progress') + if not isinstance(event_progress, dict): + raise WorkerLocalStateError('terminal history timeline progress is invalid') + try: + _canonical_json(event_progress) + except (TypeError, ValueError) as exc: + raise WorkerLocalStateError('terminal history timeline progress is invalid') from exc + normalized_timeline.append({ + 'sequence': sequence, + 'timestamp': _history_timestamp(event.get('timestamp'), 'timeline timestamp'), + 'instance_id': event_instance, + 'phase': phase, + 'progress': dict(event_progress), + }) + durations = value.get('phase_durations') + if not isinstance(durations, dict): + raise WorkerLocalStateError('terminal history phase durations are invalid') + normalized_durations = {} + for phase, seconds in durations.items(): + try: + phase = WorkerPhase(phase).value + except (TypeError, ValueError) as exc: + raise WorkerLocalStateError('terminal history phase duration key is invalid') from exc + if type(seconds) not in (int, float) or not math.isfinite(seconds) or seconds < 0: + raise WorkerLocalStateError('terminal history phase duration is invalid') + normalized_durations[phase] = seconds + normalized = dict(value) + normalized['schema'] = HISTORY_SCHEMA + normalized['started_at'] = started_at + normalized['completed_at'] = completed_at + normalized['diagnostics'] = normalized_diagnostics + normalized['timeline'] = normalized_timeline + normalized['phase_durations'] = normalized_durations + return normalized + + +class WorkerLocalState: + """Append-only worker event authority with rebuildable local views.""" + + def __init__( + self, state_root, *, log_bytes=DEFAULT_LOG_BYTES, + log_files=DEFAULT_LOG_FILES, retention_days=DEFAULT_RETENTION_DAYS, + retention_bytes=DEFAULT_RETENTION_BYTES, read_only=False, + event_segment_bytes=DEFAULT_EVENT_SEGMENT_BYTES, + history_segment_bytes=DEFAULT_HISTORY_SEGMENT_BYTES, + ): + self.read_only = bool(read_only) + self.root = os.path.abspath(state_root) + directory = require_private_directory if self.read_only else ( + lambda path, create=False: ensure_private_directory(path, reject_reparse=True) + ) + self.root = directory(self.root) + self.control_dir = directory(os.path.join(self.root, 'control')) + self.events_dir = directory(os.path.join(self.root, 'events')) + self.history_dir = directory(os.path.join(self.root, 'history')) + self.diagnostics_dir = directory(os.path.join(self.root, 'diagnostics')) + self.logs_dir = directory(os.path.join(self.root, 'logs')) + self.event_path = os.path.join(self.events_dir, 'worker-events.jsonl') + self.history_path = os.path.join(self.history_dir, 'worker-history.jsonl') + self.status_path = os.path.join(self.control_dir, 'worker.status.json') + self.log_path = os.path.join(self.logs_dir, 'worker.log') + self.log_bytes = max(64 * 1024, int(log_bytes)) + self.log_files = max(1, min(20, int(log_files))) + self.retention_days = max(1, int(retention_days)) + self.retention_bytes = max(1024 * 1024, int(retention_bytes)) + self.event_segment_bytes = max(1024, int(event_segment_bytes)) + self.history_segment_bytes = max(1024, int(history_segment_bytes)) + self._lock = threading.RLock() + self._sequence = 0 + self._events = {} + self._event_segments = [] + self._active_event_first = None + self._active_event_last = None + self._active_sparse_offsets = {} + self._event_end_offset = 0 + self._history_segment_index = 0 + self._active_history_records = 0 + self._segment_readers = {} + self._history_keys = set() + self.recover() + + @property + def next_sequence(self): + with self._lock: + return self._sequence + 1 + + def _require_writer(self): + if self.read_only: + raise WorkerLocalStateError('local worker state reader cannot mutate authority') + + @staticmethod + def _truncate_partial_final_line(path, maximum, label): + if not os.path.exists(path): + return + reject_reparse_components(path) + if not private_file_ready(path): + raise WorkerLocalStateError(f'{label} is not private') + with open(path, 'r+b') as handle: + handle.seek(0, os.SEEK_END) + size = handle.tell() + if size == 0: + return + handle.seek(-1, os.SEEK_END) + if handle.read(1) == b'\n': + return + tail_size = min(size, maximum + 1) + handle.seek(size - tail_size) + tail = handle.read(tail_size) + newline = tail.rfind(b'\n') + if newline < 0 and size > maximum: + raise WorkerLocalStateError(f'truncated {label} line exceeds its bound') + boundary = size - tail_size + newline + 1 + handle.seek(boundary) + handle.truncate() + handle.flush() + os.fsync(handle.fileno()) + + def _closed_event_paths(self): + values = [] + for name in os.listdir(self.events_dir): + match = _EVENT_SEGMENT_RE.fullmatch(name) + if match: + values.append((int(match.group(1)), int(match.group(2)), os.path.join(self.events_dir, name))) + return sorted(values) + + def _closed_history_paths(self): + values = [] + for name in os.listdir(self.history_dir): + match = _HISTORY_SEGMENT_RE.fullmatch(name) + if match: + values.append((int(match.group(1)), os.path.join(self.history_dir, name))) + return sorted(values) + + @staticmethod + def _read_event_line(handle): + raw = handle.readline(MAX_EVENT_LINE_BYTES + 1) + if not raw: + return None + if len(raw) > MAX_EVENT_LINE_BYTES: + raise WorkerLocalStateError('event journal line exceeds its bound') + if not raw.endswith(b'\n'): + return False + return raw + + def recover(self): + with self._lock: + self._sequence = 0 + self._events = {} + self._event_segments = [] + self._active_event_first = None + self._active_event_last = None + self._active_sparse_offsets = {} + self._event_end_offset = 0 + if not self.read_only: + self._truncate_partial_final_line( + self.event_path, MAX_EVENT_LINE_BYTES, 'event journal', + ) + phases = {} + event_paths = self._closed_event_paths() + if os.path.exists(self.event_path): + event_paths.append((None, None, self.event_path)) + for declared_first, declared_last, event_path in event_paths: + observed_first = None + observed_last = None + active = event_path == self.event_path + with open(event_path, 'rb') as handle: + while True: + offset = handle.tell() + raw = self._read_event_line(handle) + if raw is None: + break + if raw is False and active and self.read_only: + break + if raw is False: + raise WorkerLocalStateError('event journal line is invalid') + event = decode_worker_event(raw[:-1]) + if self._sequence and event.sequence != self._sequence + 1: + raise WorkerLocalStateError('event journal sequence is not contiguous') + if not self._sequence and event.sequence < 1: + raise WorkerLocalStateError('event journal sequence is invalid') + key = (event.instance_id, event.slot_id) + previous = phases.get(key) + if previous is not None: + validate_phase_transition(previous, event.phase) + phases[key] = event.phase + self._events[event.slot_id] = event + observed_first = observed_first or event.sequence + observed_last = event.sequence + if active and ( + observed_first == event.sequence + or (event.sequence - observed_first) % EVENT_CURSOR_STRIDE == 0 + ): + self._active_sparse_offsets[event.sequence] = offset + if active: + self._event_end_offset = handle.tell() + self._sequence = event.sequence + if active: + self._active_event_first = observed_first + self._active_event_last = observed_last + elif ( + observed_first != declared_first + or observed_last != declared_last + or observed_first is None + ): + raise WorkerLocalStateError('closed event segment identity is invalid') + else: + self._event_segments.append({ + 'first': observed_first, + 'last': observed_last, + 'path': event_path, + 'bytes': os.path.getsize(event_path), + }) + if not self.read_only: + self._truncate_partial_final_line( + self.history_path, MAX_HISTORY_LINE_BYTES, 'worker history', + ) + self._load_history_keys() + if not self.read_only: + self._write_projection() + return self.snapshot() + + def _load_history_keys(self): + self._history_keys = set() + closed = self._closed_history_paths() + self._history_segment_index = max((index for index, _path in closed), default=0) + paths = [(path, False) for _index, path in closed] + if os.path.exists(self.history_path): + paths.append((self.history_path, True)) + self._active_history_records = 0 + for path, active in paths: + reject_reparse_components(path) + if not private_file_ready(path): + raise WorkerLocalStateError('worker history is not private') + with open(path, 'rb') as handle: + while raw := handle.readline(MAX_HISTORY_LINE_BYTES + 1): + if len(raw) > MAX_HISTORY_LINE_BYTES: + raise WorkerLocalStateError('worker history line exceeds its bound') + if not raw.endswith(b'\n'): + if active and self.read_only: + break + raise WorkerLocalStateError('worker history line is invalid') + value = self._decode_history_line(raw) + key = value['history_id'] + if key in self._history_keys: + raise WorkerLocalStateError('worker history identity is duplicated') + self._history_keys.add(key) + if active: + self._active_history_records += 1 + + @staticmethod + def _decode_history_line(raw): + try: + value = json.loads(raw.decode('ascii')) + if _canonical_json(value) != raw[:-1]: + raise WorkerLocalStateError('worker history is not canonical') + return validate_history_record(value) + except WorkerLocalStateError: + raise + except (UnicodeDecodeError, json.JSONDecodeError, TypeError, ValueError) as exc: + raise WorkerLocalStateError('worker history is invalid') from exc + + def _rotate_event_if_needed(self, incoming): + if ( + self._active_event_first is None + or self._event_end_offset + incoming <= self.event_segment_bytes + ): + return + destination = os.path.join( + self.events_dir, + f'worker-events.{self._active_event_first}-{self._active_event_last}.jsonl', + ) + if os.path.exists(destination): + raise WorkerLocalStateError('event segment destination already exists') + durable_replace(self.event_path, destination) + self._event_segments.append({ + 'first': self._active_event_first, + 'last': self._active_event_last, + 'path': destination, + 'bytes': self._event_end_offset, + }) + self._active_event_first = None + self._active_event_last = None + self._active_sparse_offsets = {} + self._event_end_offset = 0 + + def _rotate_history_if_needed(self, incoming): + current = os.path.getsize(self.history_path) if os.path.exists(self.history_path) else 0 + if self._active_history_records == 0 or current + incoming <= self.history_segment_bytes: + return + self._history_segment_index += 1 + destination = os.path.join( + self.history_dir, + f'worker-history.{self._history_segment_index}.jsonl', + ) + if os.path.exists(destination): + raise WorkerLocalStateError('history segment destination already exists') + durable_replace(self.history_path, destination) + self._active_history_records = 0 + + def _retain_segment_paths(self, paths): + for path in paths: + self._segment_readers[path] = self._segment_readers.get(path, 0) + 1 + + def _release_segment_paths(self, paths): + with self._lock: + for path in paths: + count = self._segment_readers.get(path, 0) - 1 + if count > 0: + self._segment_readers[path] = count + else: + self._segment_readers.pop(path, None) + + def _projection(self): + slots = [] + for slot_id in sorted(self._events): + event = self._events[slot_id] + slots.append({ + 'slot_id': event.slot_id, + 'sequence': event.sequence, + 'timestamp': event.timestamp, + 'instance_id': event.instance_id, + 'reservation_id': event.reservation_id, + 'source': event.source, + 'phase': event.phase.value, + 'phase_started_at': event.phase_started_at, + 'scan_deadline_at': event.scan_deadline_at, + 'assignment_deadline_at': event.assignment_deadline_at, + 'progress': dict(event.progress), + }) + counts = {} + for slot in slots: + counts[slot['phase']] = counts.get(slot['phase'], 0) + 1 + return { + 'schema': STATUS_SCHEMA, + 'sequence': self._sequence, + 'updated_at': max((item['timestamp'] for item in slots), default=None), + 'aggregate': {'slot_count': len(slots), 'phases': counts}, + 'slots': slots, + } + + def _write_projection(self): + atomic_write_private_json(self.status_path, self._projection()) + + def snapshot(self): + with self._lock: + return self._projection() + + def emit_phase( + self, instance_id, slot_id, phase, *, reservation_id=None, + source=None, phase_started_at=None, scan_deadline_at=None, + assignment_deadline_at=None, progress=None, timestamp=None, + ): + self._require_writer() + with self._lock: + phase = WorkerPhase(phase) + previous = self._events.get(int(slot_id)) + if previous is not None and previous.instance_id == str(instance_id): + validate_phase_transition(previous.phase, phase) + if timestamp is None: + timestamp = utc_now() + if ( + previous is not None + and datetime.fromisoformat(timestamp[:-1] + '+00:00') + < datetime.fromisoformat(previous.timestamp[:-1] + '+00:00') + ): + timestamp = previous.timestamp + phase_started_at = ( + previous.phase_started_at + if previous is not None + and previous.instance_id == str(instance_id) + and previous.phase == phase + else (phase_started_at or timestamp) + ) + event = WorkerEvent( + schema=WORKER_EVENT_SCHEMA, + sequence=self._sequence + 1, + timestamp=timestamp, + instance_id=str(instance_id), + slot_id=int(slot_id), + reservation_id=(int(reservation_id) if reservation_id else None), + source=(str(source) if source else None), + type=WORKER_EVENT_TYPE, + phase=phase, + phase_started_at=phase_started_at, + scan_deadline_at=scan_deadline_at, + assignment_deadline_at=assignment_deadline_at, + progress=dict(progress or {}), + ) + encoded = encode_worker_event(event) + self._rotate_event_if_needed(len(encoded) + 1) + offset = self._event_end_offset + _append_private(self.event_path, encoded + b'\n', MAX_EVENT_LINE_BYTES) + self._sequence = event.sequence + self._events[event.slot_id] = event + if self._active_event_first is None: + self._active_event_first = event.sequence + self._active_event_last = event.sequence + if ( + event.sequence == self._active_event_first + or (event.sequence - self._active_event_first) % EVENT_CURSOR_STRIDE == 0 + ): + self._active_sparse_offsets[event.sequence] = offset + self._event_end_offset = offset + len(encoded) + 1 + self._write_projection() + return _event_value(event) + + def events_after(self, sequence=0, limit=256): + sequence = max(0, int(sequence)) + limit = max(1, min(1000, int(limit))) + if not os.path.exists(self.event_path): + with self._lock: + if not self._event_segments: + return [] + with self._lock: + if sequence >= self._sequence: + return [] + segments = [dict(item) for item in self._event_segments] + if self._active_event_first is not None: + segments.append({ + 'first': self._active_event_first, + 'last': self._active_event_last, + 'path': self.event_path, + 'bytes': self._event_end_offset, + 'sparse': dict(self._active_sparse_offsets), + }) + selected = [item for item in segments if item['last'] > sequence] + paths = [item['path'] for item in selected] + self._retain_segment_paths(paths) + results = [] + try: + for segment in selected: + offset = 0 + sparse = segment.get('sparse') or {} + if sparse: + candidates = [item for item in sparse if item <= sequence + 1] + if candidates: + offset = sparse[max(candidates)] + with open(segment['path'], 'rb') as handle: + handle.seek(offset) + while handle.tell() < segment['bytes']: + remaining = segment['bytes'] - handle.tell() + raw = handle.readline(min(MAX_EVENT_LINE_BYTES + 1, remaining + 1)) + if len(raw) > MAX_EVENT_LINE_BYTES: + raise WorkerLocalStateError('event journal line exceeds its bound') + if not raw.endswith(b'\n'): + break + event = decode_worker_event(raw[:-1]) + if event.sequence > sequence: + results.append(_event_value(event)) + if len(results) >= limit: + return results + finally: + self._release_segment_paths(paths) + return results + + def assignment_timeline(self, slot_id, reservation_id, limit=1000): + limit = max(1, min(1000, int(limit))) + selected = deque(maxlen=limit) + if not os.path.exists(self.event_path): + with self._lock: + if not self._event_segments: + return [] + with self._lock: + segments = [dict(item) for item in self._event_segments] + if self._active_event_first is not None: + segments.append({ + 'first': self._active_event_first, + 'last': self._active_event_last, + 'path': self.event_path, + 'bytes': self._event_end_offset, + }) + paths = [item['path'] for item in segments] + self._retain_segment_paths(paths) + try: + for segment in segments: + with open(segment['path'], 'rb') as handle: + while handle.tell() < segment['bytes']: + remaining = segment['bytes'] - handle.tell() + raw = handle.readline(min(MAX_EVENT_LINE_BYTES + 1, remaining + 1)) + if len(raw) > MAX_EVENT_LINE_BYTES: + raise WorkerLocalStateError('event journal line exceeds its bound') + if not raw.endswith(b'\n'): + break + event = decode_worker_event(raw[:-1]) + if ( + event.slot_id == int(slot_id) + and event.reservation_id == int(reservation_id) + ): + selected.append({ + 'sequence': event.sequence, + 'timestamp': event.timestamp, + 'instance_id': event.instance_id, + 'phase': event.phase.value, + 'progress': dict(event.progress), + }) + finally: + self._release_segment_paths(paths) + return list(selected) + + def append_history(self, value): + self._require_writer() + record = validate_history_record({'schema': HISTORY_SCHEMA, **dict(value)}) + history_id = record['history_id'] + payload = _canonical_json(record) + b'\n' + with self._lock: + if history_id in self._history_keys: + return False + self._rotate_history_if_needed(len(payload)) + _append_private(self.history_path, payload, MAX_HISTORY_LINE_BYTES) + self._history_keys.add(history_id) + self._active_history_records += 1 + return True + + def history(self, limit=100, reservation_id=None): + limit = max(1, min(1000, int(limit))) + selected = deque(maxlen=limit) + with self._lock: + paths = [path for _index, path in self._closed_history_paths()] + if os.path.exists(self.history_path): + paths.append(self.history_path) + self._retain_segment_paths(paths) + try: + for path in paths: + with open(path, 'rb') as handle: + while raw := handle.readline(MAX_HISTORY_LINE_BYTES + 1): + if len(raw) > MAX_HISTORY_LINE_BYTES: + raise WorkerLocalStateError('worker history line exceeds its bound') + if not raw.endswith(b'\n'): + break + value = self._decode_history_line(raw) + if reservation_id is None or int(value.get('reservation_id') or 0) == int(reservation_id): + rendered = dict(value) + rendered['diagnostics'] = [] + for diagnostic in value.get('diagnostics') or []: + item = dict(diagnostic) + record = item.get('record') + item['available'] = bool( + record and os.path.isfile(os.path.join(self.root, *str(record).split('/'))) + ) + artifacts = dict(item.get('artifacts') or {}) + item['artifact_availability'] = { + name: bool( + details and details.get('path') and os.path.isfile( + os.path.join(self.root, *str(details['path']).split('/')) + ) + ) + for name, details in artifacts.items() + } + rendered['diagnostics'].append(item) + selected.append(rendered) + finally: + self._release_segment_paths(paths) + return list(selected) + + def _rotate_log(self, incoming): + current = os.path.getsize(self.log_path) if os.path.exists(self.log_path) else 0 + if current + incoming <= self.log_bytes: + return + oldest = f'{self.log_path}.{self.log_files}' + if os.path.exists(oldest): + durable_unlink(oldest) + for index in range(self.log_files - 1, 0, -1): + source = self.log_path if index == 1 else f'{self.log_path}.{index - 1}' + destination = f'{self.log_path}.{index}' + if os.path.exists(source): + os.replace(source, destination) + fsync_directory(self.logs_dir) + + def log(self, message, *, timestamp=None): + self._require_writer() + text = str(message).replace('\r', '\\r').replace('\n', '\\n') + payload = f'{timestamp or utc_now()} {text}\n'.encode('utf-8', errors='replace') + if len(payload) > MAX_LOG_LINE_BYTES: + payload = payload[:MAX_LOG_LINE_BYTES - 1] + b'\n' + with self._lock: + self._rotate_log(len(payload)) + _append_private(self.log_path, payload, MAX_LOG_LINE_BYTES) + + def log_tail(self, limit=100): + limit = max(1, min(10000, int(limit))) + lines = deque(maxlen=limit) + paths = [f'{self.log_path}.{index}' for index in range(self.log_files, 0, -1)] + [self.log_path] + with self._lock: + for path in paths: + if not os.path.exists(path): + continue + with open(path, 'rb') as handle: + while raw := handle.readline(MAX_LOG_LINE_BYTES + 1): + if len(raw) > MAX_LOG_LINE_BYTES: + raise WorkerLocalStateError('worker log line exceeds its bound') + lines.append(raw.decode('utf-8', errors='replace').rstrip('\r\n')) + return list(lines) + + def _artifact(self, directory, name, material, full_payload=None): + if material is None and full_payload is None: + return None + payload = ( + bytes(full_payload) + if full_payload is not None + else diagnostic_material_bytes(material) + ) + path = os.path.join(directory, name) + _write_private_bytes(path, payload) + relative = os.path.relpath(path, self.root).replace(os.sep, '/') + digest = hashlib.sha256(payload).hexdigest() + if full_payload is not None: + try: + payload.decode('utf-8', errors='strict') + encoding = 'text' + except UnicodeDecodeError: + encoding = 'base64' + else: + encoding = material.encoding.value + return { + 'path': relative, + 'sha256': digest, + 'original_sha256': digest if full_payload is not None else material.sha256, + 'size': len(payload), + 'original_size': len(payload) if full_payload is not None else material.original_size, + 'encoding': encoding, + 'truncated': False if full_payload is not None else material.truncated, + } + + def archive_diagnostic(self, envelope, full_materials=None): + self._require_writer() + full_materials = dict(full_materials or {}) + if set(full_materials) - {'body', 'stdout', 'stderr'}: + raise WorkerLocalStateError('diagnostic full material shape is invalid') + encoded = encode_diagnostic_envelope(envelope) + envelope = decode_diagnostic_envelope(encoded) + date = envelope.captured_at[:10] + directory = ensure_private_directory( + os.path.join(self.diagnostics_dir, date, str(envelope.reservation_id)), + reject_reparse=True, + ) + uid = envelope.diagnostic_uid + artifacts = { + 'body': self._artifact( + directory, f'{uid}.body', envelope.http.body if envelope.http else None, + full_materials.get('body'), + ), + 'stdout': self._artifact( + directory, f'{uid}.stdout.log', envelope.process.stdout if envelope.process else None, + full_materials.get('stdout'), + ), + 'stderr': self._artifact( + directory, f'{uid}.stderr.log', envelope.process.stderr if envelope.process else None, + full_materials.get('stderr'), + ), + } + record = { + 'schema': DIAGNOSTIC_ARCHIVE_SCHEMA, + 'envelope': json.loads(encoded.decode('ascii')), + 'artifacts': artifacts, + } + path = os.path.join(directory, f'{uid}.json') + atomic_write_private_json(path, record, max_bytes=2 * 1024 * 1024) + return { + 'diagnostic_uid': uid, + 'record': os.path.relpath(path, self.root).replace(os.sep, '/'), + 'artifacts': artifacts, + } + + def diagnostic_references(self, reservation_id): + reservation_name = str(int(reservation_id)) + references = [] + for date in sorted(os.listdir(self.diagnostics_dir)): + directory = os.path.join(self.diagnostics_dir, date, reservation_name) + if not os.path.isdir(directory) or os.path.islink(directory): + continue + for name in sorted(os.listdir(directory)): + if not name.endswith('.json'): + continue + path = os.path.join(directory, name) + try: + value = read_private_json(path, max_bytes=2 * 1024 * 1024) + except OSError: + continue + if not isinstance(value, dict) or set(value) != { + 'schema', 'envelope', 'artifacts', + } or value.get('schema') != DIAGNOSTIC_ARCHIVE_SCHEMA: + raise WorkerLocalStateError('diagnostic archive record is invalid') + envelope = value.get('envelope') + artifacts = value.get('artifacts') + if not isinstance(envelope, dict) or not isinstance(artifacts, dict): + raise WorkerLocalStateError('diagnostic archive record is invalid') + try: + decoded = decode_diagnostic_envelope(_canonical_json(envelope)) + except ValueError as exc: + raise WorkerLocalStateError('diagnostic archive envelope is invalid') from exc + if decoded.reservation_id != int(reservation_id): + raise WorkerLocalStateError('diagnostic archive reservation is invalid') + reference = { + 'diagnostic_uid': decoded.diagnostic_uid, + 'record': os.path.relpath(path, self.root).replace(os.sep, '/'), + 'artifacts': artifacts, + } + validate_history_record({ + 'schema': HISTORY_SCHEMA, + 'history_id': 'diagnostic-reference-validation', + 'instance_id': 'diagnostic-reference-validation', + 'slot_id': 0, + 'reservation_id': int(reservation_id), + 'source': None, + 'outcome': 'validation', + 'receipt': {}, + 'started_at': None, + 'completed_at': utc_now(), + 'duration_seconds': None, + 'first_sequence': None, + 'last_sequence': None, + 'diagnostics': [reference], + 'timeline': [], + 'phase_durations': {}, + }) + references.append(reference) + return references + + def _active_reservations(self): + reservations = set() + reliable = True + with os.scandir(self.root) as entries: + for entry in entries: + if not re.fullmatch(r'slot-(0|[1-9][0-9]*)\.json', entry.name): + continue + try: + state = read_private_json(entry.path, max_bytes=64 * 1024 * 1024) + reservation = dict((state.get('assignment') or {}).get('reservation') or {}) + reservation_id = int(reservation.get('reservation_id') or 0) + if reservation_id > 0: + reservations.add(reservation_id) + except (OSError, TypeError, ValueError): + reliable = False + return reservations, reliable + + def _diagnostic_is_evictable(self, path, active_reservations, reliable): + if not reliable: + return False + relative = os.path.relpath(path, self.diagnostics_dir).split(os.sep) + if len(relative) < 3: + return True + try: + reservation_id = int(relative[1]) + except ValueError: + return False + return reservation_id not in active_reservations + + def _retained_history_authority(self): + retained_paths = set() + retained_event_sequences = set() + paths = [path for _index, path in self._closed_history_paths()] + if os.path.exists(self.history_path): + paths.append(self.history_path) + for path in paths: + with open(path, 'rb') as handle: + while raw := handle.readline(MAX_HISTORY_LINE_BYTES + 1): + if len(raw) > MAX_HISTORY_LINE_BYTES or not raw.endswith(b'\n'): + break + value = self._decode_history_line(raw) + retained_event_sequences.update( + event['sequence'] for event in value['timeline'] + ) + for reference in value['diagnostics']: + retained_paths.add(os.path.abspath(os.path.join( + self.root, *reference['record'].split('/'), + ))) + for artifact in reference['artifacts'].values(): + if artifact is not None: + retained_paths.add(os.path.abspath(os.path.join( + self.root, *artifact['path'].split('/'), + ))) + return retained_paths, retained_event_sequences + + def _event_segment_is_evictable( + self, path, active_reservations, active_reliable, + retained_event_sequences, + ): + if not active_reliable: + return False + with open(path, 'rb') as handle: + while raw := handle.readline(MAX_EVENT_LINE_BYTES + 1): + if len(raw) > MAX_EVENT_LINE_BYTES or not raw.endswith(b'\n'): + return False + event = decode_worker_event(raw[:-1]) + if event.reservation_id in active_reservations: + return False + if ( + event.reservation_id is not None + and event.sequence not in retained_event_sequences + ): + return False + return True + + def _evictable_event_prefix( + self, active_reservations, active_reliable, + retained_event_sequences, progress_cursor, + ): + paths = set() + for _first, last, path in self._closed_event_paths(): + if progress_cursor is not None and last > progress_cursor: + break + if not self._event_segment_is_evictable( + path, active_reservations, active_reliable, + retained_event_sequences, + ): + break + paths.add(os.path.abspath(path)) + return paths + + def _progress_outbox_cursor(self): + old_path = os.path.join(self.root, 'progress-outbox.json') + path = os.path.join( + self.root, *PROGRESS_OUTBOX_RELATIVE_PATH.split('/'), + ) + try: + old_sequence = ( + _progress_cursor_value(old_path) if os.path.exists(old_path) else None + ) + new_sequence = ( + _progress_cursor_value(path) if os.path.exists(path) else None + ) + except (OSError, ValueError, WorkerLocalStateError): + return 0 + if old_sequence is not None and new_sequence is not None: + return old_sequence if old_sequence == new_sequence else 0 + sequence = new_sequence if new_sequence is not None else old_sequence + if sequence is not None: + return sequence + if os.path.exists(self.event_path) or self._closed_event_paths(): + return 0 + return None + + def _effective_event_prefix( + self, structural_prefix, total_bytes, cutoff, + ): + remaining_total = int(total_bytes) + effective = set() + with self._lock: + readers = dict(self._segment_readers) + for _first, _last, path in self._closed_event_paths(): + absolute = os.path.abspath(path) + if absolute not in structural_prefix or readers.get(path, 0): + break + try: + details = os.stat(path, follow_symlinks=False) + except OSError: + break + if details.st_mtime >= cutoff and remaining_total <= self.retention_bytes: + break + effective.add(absolute) + remaining_total -= details.st_size + return effective + + def _retention_snapshot(self, extra_roots=None, now=None): + now = time.time() if now is None else float(now) + cutoff = now - self.retention_days * 86400 + roots = { + 'control': self.control_dir, + 'events': self.events_dir, + 'history': self.history_dir, + 'diagnostics': self.diagnostics_dir, + 'logs': self.logs_dir, + } + roots.update(dict(extra_roots or {})) + categories = {} + total_bytes = 0 + total_files = 0 + total_evictable_bytes = 0 + total_evictable_files = 0 + active_reservations, active_reliable = self._active_reservations() + retained_diagnostics, retained_event_sequences = self._retained_history_authority() + progress_cursor = self._progress_outbox_cursor() + structural_event_prefix = self._evictable_event_prefix( + active_reservations, active_reliable, retained_event_sequences, + progress_cursor, + ) + for name, root in roots.items(): + byte_count = 0 + file_count = 0 + evictable_bytes = 0 + evictable_files = 0 + if os.path.isdir(root): + for current, directories, files in os.walk(root, followlinks=False): + directories[:] = [item for item in directories if not os.path.islink(os.path.join(current, item))] + for filename in files: + path = os.path.join(current, filename) + try: + details = os.stat(path, follow_symlinks=False) + except OSError: + continue + if stat.S_ISREG(details.st_mode): + byte_count += details.st_size + file_count += 1 + evictable = ( + ( + self._diagnostic_is_evictable( + path, active_reservations, active_reliable, + ) + and os.path.abspath(path) not in retained_diagnostics + ) if name == 'diagnostics' + else ( + (name == 'logs' and path != self.log_path) + ) + ) + if evictable: + evictable_bytes += details.st_size + evictable_files += 1 + categories[name] = { + 'bytes': byte_count, + 'files': file_count, + 'evictable_bytes': evictable_bytes, + 'evictable_files': evictable_files, + 'non_evictable_bytes': byte_count - evictable_bytes, + 'non_evictable_files': file_count - evictable_files, + } + total_bytes += byte_count + total_files += file_count + total_evictable_bytes += evictable_bytes + total_evictable_files += evictable_files + state_bytes = 0 + state_files = 0 + with os.scandir(self.root) as entries: + for entry in entries: + try: + details = entry.stat(follow_symlinks=False) + except OSError: + continue + if stat.S_ISREG(details.st_mode): + state_bytes += details.st_size + state_files += 1 + categories['state'] = { + 'bytes': state_bytes, + 'files': state_files, + 'evictable_bytes': 0, + 'evictable_files': 0, + 'non_evictable_bytes': state_bytes, + 'non_evictable_files': state_files, + } + total_bytes += state_bytes + total_files += state_files + effective_event_prefix = self._effective_event_prefix( + structural_event_prefix, total_bytes, cutoff, + ) + event_evictable_bytes = 0 + event_evictable_files = 0 + for path in effective_event_prefix: + try: + event_evictable_bytes += os.stat(path, follow_symlinks=False).st_size + event_evictable_files += 1 + except OSError: + continue + events = categories['events'] + events['evictable_bytes'] = event_evictable_bytes + events['evictable_files'] = event_evictable_files + events['non_evictable_bytes'] = events['bytes'] - event_evictable_bytes + events['non_evictable_files'] = events['files'] - event_evictable_files + total_evictable_bytes += event_evictable_bytes + total_evictable_files += event_evictable_files + document = { + 'schema': RETENTION_SCHEMA, + 'maximum_age_days': self.retention_days, + 'maximum_bytes': self.retention_bytes, + 'total_bytes': total_bytes, + 'total_files': total_files, + 'evictable_bytes': total_evictable_bytes, + 'evictable_files': total_evictable_files, + 'non_evictable_bytes': total_bytes - total_evictable_bytes, + 'non_evictable_files': total_files - total_evictable_files, + 'over_limit': total_bytes > self.retention_bytes, + 'progress_outbox': { + 'cursor_sequence': progress_cursor, + 'protected_after_sequence': ( + progress_cursor if progress_cursor is not None else None + ), + }, + 'categories': categories, + } + return document, effective_event_prefix + + def retention_usage(self, extra_roots=None): + document, _event_prefix = self._retention_snapshot(extra_roots) + return document + + def cleanup_retention(self, extra_roots=None): + self._require_writer() + now = time.time() + cutoff = now - self.retention_days * 86400 + candidates = [] + usage, evictable_event_prefix = self._retention_snapshot( + extra_roots, now=now, + ) + active_reservations, active_reliable = self._active_reservations() + retained_diagnostics, retained_event_sequences = self._retained_history_authority() + for root in ( + self.diagnostics_dir, self.logs_dir, self.events_dir, + ): + for current, directories, files in os.walk(root, topdown=True, followlinks=False): + directories[:] = [item for item in directories if not os.path.islink(os.path.join(current, item))] + for filename in files: + path = os.path.join(current, filename) + if path in {self.log_path, self.event_path} or os.path.islink(path): + continue + if root == self.events_dir and _EVENT_SEGMENT_RE.fullmatch(filename) is None: + continue + if root == self.diagnostics_dir and not self._diagnostic_is_evictable( + path, active_reservations, active_reliable, + ): + continue + if root == self.diagnostics_dir and os.path.abspath(path) in retained_diagnostics: + continue + if ( + root == self.events_dir + and os.path.abspath(path) not in evictable_event_prefix + ): + continue + try: + details = os.stat(path, follow_symlinks=False) + except OSError: + continue + if stat.S_ISREG(details.st_mode): + event_match = _EVENT_SEGMENT_RE.fullmatch(filename) + if root == self.events_dir and event_match: + kind = 'events' + order = int(event_match.group(1)) + sort_key = (0, order) + else: + kind = None + order = 0 + sort_key = (1, details.st_mtime) + candidates.append(( + sort_key, details.st_mtime, details.st_size, + path, kind, order, + )) + removed_bytes = 0 + removed_files = 0 + total_bytes = usage['total_bytes'] + blocked_segments = set() + with self._lock: + for _sort_key, modified, size, path, kind, _order in sorted(candidates): + if kind in blocked_segments: + continue + if modified >= cutoff and total_bytes <= self.retention_bytes: + if kind is not None: + blocked_segments.add(kind) + continue + if self._segment_readers.get(path, 0): + if kind is not None: + blocked_segments.add(kind) + continue + try: + durable_unlink(path) + except FileNotFoundError: + continue + removed_bytes += size + removed_files += 1 + total_bytes -= size + if root_path := next(( + segment for segment in self._event_segments + if segment['path'] == path + ), None): + self._event_segments.remove(root_path) + return {'removed_bytes': removed_bytes, 'removed_files': removed_files} diff --git a/app/worker_package.py b/app/worker_package.py new file mode 100644 index 0000000..54488a2 --- /dev/null +++ b/app/worker_package.py @@ -0,0 +1,518 @@ +import hashlib +import json +import os +import platform as host_platform +import re +import stat +import sys + +from lifecycle_authority import ( + APPLICATION_IMPORT_SUFFIXES, + CODE_MANIFEST_SCHEMA, + GIT_MANIFEST_NAME, + REMOTE_WORKER_CODE_AUTHORITY_FILES, + TRUFFLEHOG_MANIFEST_NAME, + code_manifest_sha256, + verify_code_manifest, +) +from result_bundle import FORMAT_VERSION +from runtime_security import ( + atomic_write_private_json, + canonical_path, + ensure_private_directory, + reject_reparse_components, + sha256_file, +) +WORKER_PACKAGE_SCHEMA = 3 +PROTOCOL_VERSION = 2 +PACKAGE_DETECTOR_POLICY = '@package/detector_policy' +MAX_WORKER_PACKAGE_MANIFEST_BYTES = 1024 * 1024 +MAX_WORKER_PACKAGE_FILES = 512 +MAX_WORKER_RUNTIME_TREE_FILES = 10000 +MAX_WORKER_PACKAGE_CAPABILITIES = 16 +_DIGEST = re.compile(r'^[a-f0-9]{64}$') +KNOWN_WORKER_PACKAGE_CAPABILITIES = frozenset({ + ('github', 'github', 'exact_git_v1'), + ('gitlab', 'gitlab', 'exact_git_v1'), + ('dockerhub', 'docker', 'docker_direct_v1'), + ('huggingface', 'huggingface', 'huggingface_space_v1'), +}) +DEFAULT_WORKER_PACKAGE_CAPABILITIES = ( + ('gitlab', 'gitlab', 'exact_git_v1'), + ('dockerhub', 'docker', 'docker_direct_v1'), + ('huggingface', 'huggingface', 'huggingface_space_v1'), +) + + +class WorkerPackageError(ValueError): + pass + + +def local_platform_tag(): + machine = host_platform.machine().strip().lower().replace('amd64', 'x86_64') + system = 'windows' if sys.platform == 'win32' else 'linux' if sys.platform.startswith('linux') else '' + if not system or machine not in {'x86_64', 'aarch64', 'arm64'}: + raise WorkerPackageError('unsupported worker platform') + return f'{system}-{machine.replace("arm64", "aarch64")}' + + +def _relative_path(value, label): + value = str(value or '') + if ( + not value or len(value) > 512 or '\\' in value or '\x00' in value + or value.startswith('/') or value.endswith('/') + ): + raise WorkerPackageError(f'invalid worker package {label} path') + parts = value.split('/') + if any(part in ('', '.', '..') for part in parts): + raise WorkerPackageError(f'invalid worker package {label} path') + return '/'.join(parts) + + +def _entry(value, label): + if not isinstance(value, dict) or set(value) != {'path', 'sha256'}: + raise WorkerPackageError(f'invalid worker package {label} entry') + digest = str(value.get('sha256') or '') + if not _DIGEST.fullmatch(digest): + raise WorkerPackageError(f'invalid worker package {label} digest') + return {'path': _relative_path(value.get('path'), label), 'sha256': digest} + + +def _tree_entry(value, label): + if not isinstance(value, dict) or set(value) != {'path', 'sha256', 'file_count'}: + raise WorkerPackageError(f'invalid worker package {label} tree entry') + digest = str(value.get('sha256') or '') + try: + file_count = int(value.get('file_count')) + except (TypeError, ValueError, OverflowError): + raise WorkerPackageError(f'invalid worker package {label} tree file count') from None + if not _DIGEST.fullmatch(digest) or not 1 <= file_count <= MAX_WORKER_RUNTIME_TREE_FILES: + raise WorkerPackageError(f'invalid worker package {label} tree identity') + return { + 'path': _relative_path(value.get('path'), label), + 'sha256': digest, + 'file_count': file_count, + } + + +def _validate_worker_application_files(names): + names = set(names) + required = set(REMOTE_WORKER_CODE_AUTHORITY_FILES) + if not required <= names: + raise WorkerPackageError('worker package file set is incomplete') + for name in names: + parts = name.lower().split('/') + if 'keycheckers' in parts or parts[-1] == 'keycheck_runner.py': + raise WorkerPackageError('worker package detailed keycheck code is forbidden') + application_code = { + name for name in names + if not name.startswith('dependencies/') + and name.lower().endswith(APPLICATION_IMPORT_SUFFIXES) + } + if application_code != required: + raise WorkerPackageError('worker package application code set is unsupported') + + +def normalize_worker_package_capabilities(value): + if ( + not isinstance(value, list) + or not 1 <= len(value) <= MAX_WORKER_PACKAGE_CAPABILITIES + ): + raise WorkerPackageError('worker package capabilities are invalid') + capabilities = [] + seen = set() + for item in value: + if not isinstance(item, dict) or set(item) != { + 'source', 'platform', 'planning_kind', + }: + raise WorkerPackageError('worker package capability shape is invalid') + if any(type(item[name]) is not str for name in item): + raise WorkerPackageError('worker package capability identity is invalid') + capability = ( + item['source'], item['platform'], item['planning_kind'], + ) + if capability not in KNOWN_WORKER_PACKAGE_CAPABILITIES: + raise WorkerPackageError('worker package capability is unsupported') + if capability in seen: + raise WorkerPackageError('worker package capability is duplicated') + seen.add(capability) + capabilities.append({ + 'source': capability[0], + 'platform': capability[1], + 'planning_kind': capability[2], + }) + return sorted( + capabilities, + key=lambda item: ( + item['source'], item['platform'], item['planning_kind'], + ), + ) + + +def normalize_worker_package_manifest(value): + if not isinstance(value, dict) or set(value) != { + 'schema', 'protocol_version', 'bundle_format_version', 'platform_tag', + 'capabilities', 'app_root', 'files', 'executables', 'assets', + 'runtime_trees', + }: + raise WorkerPackageError('worker package manifest shape is invalid') + if type(value.get('schema')) is not int or value['schema'] != WORKER_PACKAGE_SCHEMA: + raise WorkerPackageError('worker package manifest schema is unsupported') + if ( + type(value.get('protocol_version')) is not int + or value['protocol_version'] != PROTOCOL_VERSION + ): + raise WorkerPackageError('worker package protocol version is unsupported') + if ( + type(value.get('bundle_format_version')) is not int + or value['bundle_format_version'] != FORMAT_VERSION + ): + raise WorkerPackageError('worker package bundle format is unsupported') + platform_tag = str(value.get('platform_tag') or '') + if not re.fullmatch(r'(windows|linux)-(x86_64|aarch64)', platform_tag): + raise WorkerPackageError('worker package platform is unsupported') + capabilities = normalize_worker_package_capabilities(value.get('capabilities')) + app_root = _relative_path(value.get('app_root'), 'application root') + + files_value = value.get('files') + if not isinstance(files_value, dict) or not 1 <= len(files_value) <= MAX_WORKER_PACKAGE_FILES: + raise WorkerPackageError('worker package file set is invalid') + _validate_worker_application_files(files_value) + files = {} + for name in sorted(files_value): + normalized_name = _relative_path(name, 'file name') + if normalized_name != name: + raise WorkerPackageError('worker package file name is not canonical') + files[name] = _entry(files_value[name], f'file {name}') + + executables_value = value.get('executables') + if not isinstance(executables_value, dict) or set(executables_value) != { + TRUFFLEHOG_MANIFEST_NAME, GIT_MANIFEST_NAME, + }: + raise WorkerPackageError('worker package executable set is incomplete') + executables = { + name: _entry(executables_value[name], f'executable {name}') + for name in (TRUFFLEHOG_MANIFEST_NAME, GIT_MANIFEST_NAME) + } + assets_value = value.get('assets') + if not isinstance(assets_value, dict) or set(assets_value) != {'detector_policy'}: + raise WorkerPackageError('worker package asset set is incomplete') + assets = {'detector_policy': _entry(assets_value['detector_policy'], 'detector policy')} + runtime_trees_value = value.get('runtime_trees') + required_trees = {'git', 'python'} if platform_tag.startswith('windows-') else {'git'} + if not isinstance(runtime_trees_value, dict) or set(runtime_trees_value) != required_trees: + raise WorkerPackageError('worker package runtime tree set is incomplete') + runtime_trees = { + name: _tree_entry(runtime_trees_value[name], f'{name} runtime') + for name in sorted(required_trees) + } + git_path = executables[GIT_MANIFEST_NAME]['path'] + git_root = runtime_trees['git']['path'] + '/' + if not git_path.startswith(git_root): + raise WorkerPackageError('worker package Git executable escapes its runtime tree') + return { + 'schema': WORKER_PACKAGE_SCHEMA, + 'protocol_version': PROTOCOL_VERSION, + 'bundle_format_version': FORMAT_VERSION, + 'platform_tag': platform_tag, + 'capabilities': capabilities, + 'app_root': app_root, + 'files': files, + 'executables': executables, + 'assets': assets, + 'runtime_trees': runtime_trees, + } + + +def worker_package_manifest_sha256(manifest): + normalized = normalize_worker_package_manifest(manifest) + encoded = json.dumps( + normalized, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('utf-8') + return hashlib.sha256(encoded).hexdigest() + + +def worker_package_build_compatibility(manifest): + normalized = normalize_worker_package_manifest(manifest) + return { + 'protocol_version': normalized['protocol_version'], + 'bundle_format_version': normalized['bundle_format_version'], + 'platform_tag': normalized['platform_tag'], + 'code_manifest_sha256': worker_package_manifest_sha256(normalized), + 'detector_policy_sha256': normalized['assets']['detector_policy']['sha256'], + } + + +def load_worker_package_manifest_bytes(payload): + if type(payload) is not bytes: + raise WorkerPackageError('worker package manifest payload must be bytes') + if len(payload) > MAX_WORKER_PACKAGE_MANIFEST_BYTES: + raise WorkerPackageError('worker package manifest exceeds its byte bound') + try: + value = json.loads(payload.decode('utf-8', errors='strict')) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise WorkerPackageError('worker package manifest is invalid JSON') from exc + return normalize_worker_package_manifest(value) + + +def load_worker_package_manifest(path): + path = os.path.abspath(os.fspath(path)) + reject_reparse_components(path) + flags = os.O_RDONLY + if hasattr(os, 'O_BINARY'): + flags |= os.O_BINARY + if hasattr(os, 'O_NOFOLLOW'): + flags |= os.O_NOFOLLOW + descriptor = os.open(path, flags) + with os.fdopen(descriptor, 'rb') as handle: + before = os.fstat(handle.fileno()) + if not stat.S_ISREG(before.st_mode) or before.st_nlink != 1: + raise WorkerPackageError('worker package manifest is not a regular file') + payload = handle.read(MAX_WORKER_PACKAGE_MANIFEST_BYTES + 1) + after = os.fstat(handle.fileno()) + reject_reparse_components(path) + current = os.stat(path, follow_symlinks=False) + identity = lambda value: (value.st_dev, value.st_ino) + if ( + identity(before) != identity(after) + or identity(after) != identity(current) + or after.st_nlink != 1 + or current.st_nlink != 1 + or not stat.S_ISREG(current.st_mode) + or before.st_size != after.st_size + or after.st_size != current.st_size + or getattr(before, 'st_mtime_ns', None) != getattr(after, 'st_mtime_ns', None) + or getattr(after, 'st_mtime_ns', None) != getattr(current, 'st_mtime_ns', None) + or getattr(before, 'st_ctime_ns', None) != getattr(after, 'st_ctime_ns', None) + or ( + os.name != 'nt' + and getattr(after, 'st_ctime_ns', None) + != getattr(current, 'st_ctime_ns', None) + ) + ): + raise WorkerPackageError('worker package manifest changed while it was being read') + return load_worker_package_manifest_bytes(payload) + + +def _package_path(root, relative): + root = canonical_path(root) + candidate = canonical_path(os.path.join(root, *relative.split('/'))) + try: + contained = os.path.commonpath((root, candidate)) == root and candidate != root + except ValueError: + contained = False + if not contained: + raise WorkerPackageError('worker package path escapes its root') + reject_reparse_components(candidate) + return candidate + + +def _manifest_entry_for_path(package_root, relative, label): + relative = _relative_path(relative, label) + path = _package_path(package_root, relative) + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or os.path.islink(path): + raise WorkerPackageError(f'worker package {label} is not a regular file') + return {'path': relative, 'sha256': sha256_file(path)} + + +def _application_package_files(package_root, app_root): + app_path = _package_path(package_root, app_root) + files = {} + + def raise_walk_error(exc): + raise WorkerPackageError(f'unable to inspect worker package application root: {exc}') from exc + + for current, directories, names in os.walk( + app_path, followlinks=False, onerror=raise_walk_error, + ): + for name in directories: + path = os.path.join(current, name) + reject_reparse_components(path) + if os.path.islink(path) or name.lower() == '__pycache__': + raise WorkerPackageError('worker package application directory is unsupported') + for name in names: + path = os.path.join(current, name) + reject_reparse_components(path) + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or os.path.islink(path): + raise WorkerPackageError('worker package application file is not regular') + relative_name = os.path.relpath(path, app_path).replace(os.sep, '/') + relative_name = _relative_path(relative_name, 'file name') + files[relative_name] = _manifest_entry_for_path( + package_root, f'{app_root}/{relative_name}', f'file {relative_name}', + ) + if len(files) > MAX_WORKER_PACKAGE_FILES: + raise WorkerPackageError('worker package file set exceeds its bound') + return files + + +def _runtime_tree_identity(package_root, relative, label): + relative = _relative_path(relative, label) + tree_root = _package_path(package_root, relative) + if not os.path.isdir(tree_root) or os.path.islink(tree_root): + raise WorkerPackageError(f'worker package {label} runtime tree is not a directory') + files = [] + + def raise_walk_error(exc): + raise WorkerPackageError(f'unable to inspect worker package {label} runtime tree: {exc}') from exc + + for current, directories, names in os.walk( + tree_root, followlinks=False, onerror=raise_walk_error, + ): + for name in directories: + path = os.path.join(current, name) + reject_reparse_components(path) + if os.path.islink(path): + raise WorkerPackageError(f'worker package {label} runtime tree contains a link') + for name in names: + path = os.path.join(current, name) + reject_reparse_components(path) + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or os.path.islink(path): + raise WorkerPackageError(f'worker package {label} runtime file is not regular') + name = _relative_path( + os.path.relpath(path, tree_root).replace(os.sep, '/'), + f'{label} runtime file', + ) + files.append((name, sha256_file(path))) + if len(files) > MAX_WORKER_RUNTIME_TREE_FILES: + raise WorkerPackageError(f'worker package {label} runtime tree exceeds its bound') + if not files: + raise WorkerPackageError(f'worker package {label} runtime tree is empty') + digest = hashlib.sha256(b'truf-worker-runtime-tree-v1\0') + for name, file_digest in sorted(files): + digest.update(name.encode('utf-8', errors='strict')) + digest.update(b'\0') + digest.update(file_digest.encode('ascii')) + digest.update(b'\0') + return {'path': relative, 'sha256': digest.hexdigest(), 'file_count': len(files)} + + +def _runtime_tree_permissions_ready(package_root, entry): + tree_root = _package_path(package_root, entry['path']) + for current, directories, names in os.walk(tree_root, followlinks=False): + for path in [current, *(os.path.join(current, name) for name in directories + names)]: + details = os.stat(path, follow_symlinks=False) + if os.name == 'nt': + from runtime_security import private_directory_ready, private_file_ready + ready = private_directory_ready(path) if stat.S_ISDIR(details.st_mode) else private_file_ready(path) + if not ready: + return False + elif details.st_uid != 0 or details.st_mode & 0o022: + return False + return True + + +def _verify_runtime_tree(package_root, name, entry): + current = _runtime_tree_identity(package_root, entry['path'], name) + if current != entry: + raise WorkerPackageError(f'worker package {name} runtime tree drifted') + if not _runtime_tree_permissions_ready(package_root, entry): + raise WorkerPackageError(f'worker package {name} runtime tree is not trusted') + return _package_path(package_root, entry['path']) + + +def build_worker_package_manifest( + package_root, *, app_root='app', trufflehog_path, git_path, + detector_policy_path, git_root, capabilities, platform_tag=None, + python_root=None, +): + package_root = canonical_path(package_root) + reject_reparse_components(package_root) + if not os.path.isdir(package_root) or os.path.islink(package_root): + raise WorkerPackageError('worker package root is not a directory') + app_root = _relative_path(app_root, 'application root') + app_path = _package_path(package_root, app_root) + if not os.path.isdir(app_path) or os.path.islink(app_path): + raise WorkerPackageError('worker package application root is not a directory') + + files = _application_package_files(package_root, app_root) + + platform_tag = platform_tag or local_platform_tag() + runtime_trees = {'git': _runtime_tree_identity(package_root, git_root, 'git')} + if platform_tag.startswith('windows-'): + if not python_root: + raise WorkerPackageError('Windows worker package requires a Python runtime tree') + runtime_trees['python'] = _runtime_tree_identity(package_root, python_root, 'python') + elif python_root: + raise WorkerPackageError('Linux worker package must use its pinned image Python runtime') + manifest = { + 'schema': WORKER_PACKAGE_SCHEMA, + 'protocol_version': PROTOCOL_VERSION, + 'bundle_format_version': FORMAT_VERSION, + 'platform_tag': platform_tag, + 'capabilities': list(capabilities), + 'app_root': app_root, + 'files': files, + 'executables': { + TRUFFLEHOG_MANIFEST_NAME: _manifest_entry_for_path( + package_root, trufflehog_path, 'TruffleHog executable', + ), + GIT_MANIFEST_NAME: _manifest_entry_for_path( + package_root, git_path, 'Git executable', + ), + }, + 'assets': { + 'detector_policy': _manifest_entry_for_path( + package_root, detector_policy_path, 'detector policy', + ), + }, + 'runtime_trees': runtime_trees, + } + return normalize_worker_package_manifest(manifest) + + +def write_worker_package_manifest(path, manifest): + path = os.path.abspath(os.fspath(path)) + ensure_private_directory(os.path.dirname(path), reject_reparse=True) + normalized = normalize_worker_package_manifest(manifest) + atomic_write_private_json( + path, normalized, max_bytes=MAX_WORKER_PACKAGE_MANIFEST_BYTES, + ) + return path + + +def verify_worker_package(manifest_path): + manifest_path = os.path.abspath(os.fspath(manifest_path)) + package_root = canonical_path(os.path.dirname(manifest_path)) + manifest = load_worker_package_manifest(manifest_path) + if manifest['platform_tag'] != local_platform_tag(): + raise WorkerPackageError('worker package does not match the local platform') + app_root = _package_path(package_root, manifest['app_root']) + files = {} + for name, entry in manifest['files'].items(): + files[name] = {'path': _package_path(package_root, entry['path']), 'sha256': entry['sha256']} + executables = { + name: {'path': _package_path(package_root, entry['path']), 'sha256': entry['sha256']} + for name, entry in manifest['executables'].items() + } + policy = manifest['assets']['detector_policy'] + policy_path = _package_path(package_root, policy['path']) + runtime_trees = { + name: _verify_runtime_tree(package_root, name, entry) + for name, entry in manifest['runtime_trees'].items() + } + code_manifest = { + 'schema': CODE_MANIFEST_SCHEMA, + 'root': app_root, + 'files': files, + 'executables': executables, + 'assets': {policy_path: {'path': policy_path, 'sha256': policy['sha256']}}, + } + verified = verify_code_manifest( + code_manifest, + require_private_acl=True, + required_names=REMOTE_WORKER_CODE_AUTHORITY_FILES, + external_names=(), + ) + return { + 'manifest': manifest, + 'build_compatibility': worker_package_build_compatibility(manifest), + 'code_manifest': verified, + 'code_manifest_sha256': code_manifest_sha256(verified), + 'trufflehog_path': executables[TRUFFLEHOG_MANIFEST_NAME]['path'], + 'git_path': executables[GIT_MANIFEST_NAME]['path'], + 'detector_policy_path': policy_path, + 'runtime_trees': runtime_trees, + } diff --git a/app/worker_package_builder.py b/app/worker_package_builder.py new file mode 100644 index 0000000..1a2506a --- /dev/null +++ b/app/worker_package_builder.py @@ -0,0 +1,537 @@ +"""Build reproducible remote-worker package trees from pinned public inputs.""" + +import argparse +import base64 +import csv +import hashlib +import json +import os +import shutil +import stat +import struct +import subprocess +import sys +import tarfile +import tempfile +import urllib.request +import zipfile + +sys.dont_write_bytecode = True + + +APP_DIR = os.path.abspath(os.path.dirname(__file__)) +PROJECT_DIR = os.path.dirname(APP_DIR) +DEPENDENCIES_DIR = os.path.join(APP_DIR, 'dependencies') +if os.path.isdir(DEPENDENCIES_DIR) and DEPENDENCIES_DIR not in sys.path: + sys.path.insert(0, DEPENDENCIES_DIR) +if APP_DIR not in sys.path: + sys.path.insert(0, APP_DIR) + +from lifecycle_authority import REMOTE_WORKER_CODE_AUTHORITY_FILES +from runtime_security import harden_private_tree, reject_reparse_components, sha256_file +from worker_package import ( + DEFAULT_WORKER_PACKAGE_CAPABILITIES, + build_worker_package_manifest, + verify_worker_package, + worker_package_manifest_sha256, + write_worker_package_manifest, +) + + +class WorkerPackageBuildError(RuntimeError): + pass + + +def _regular_file(path, label): + path = os.path.abspath(os.fspath(path)) + reject_reparse_components(path) + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or os.path.islink(path): + raise WorkerPackageBuildError(f'{label} is not a regular file') + return path + + +def _copy_file(source, destination, label): + source = _regular_file(source, label) + os.makedirs(os.path.dirname(destination), exist_ok=True) + shutil.copyfile(source, destination) + shutil.copymode(source, destination, follow_symlinks=False) + return destination + + +def _copy_tree(source, destination, label): + source = os.path.abspath(os.fspath(source)) + reject_reparse_components(source) + if not os.path.isdir(source) or os.path.islink(source): + raise WorkerPackageBuildError(f'{label} is not a directory') + os.makedirs(destination, exist_ok=False) + copied = 0 + for current, directories, names in os.walk(source, followlinks=False): + relative = os.path.relpath(current, source) + target = destination if relative == '.' else os.path.join(destination, relative) + for name in list(directories): + path = os.path.join(current, name) + reject_reparse_components(path) + if os.path.islink(path) or name.lower() == '__pycache__': + raise WorkerPackageBuildError(f'{label} contains an unsupported directory') + os.makedirs(os.path.join(target, name), exist_ok=False) + for name in names: + if name.lower().endswith(('.pyc', '.pyo')): + raise WorkerPackageBuildError(f'{label} contains cached bytecode') + _copy_file( + os.path.join(current, name), os.path.join(target, name), + f'{label} file', + ) + copied += 1 + if copied == 0: + raise WorkerPackageBuildError(f'{label} is empty') + return copied + + +def _windows_support_files(root): + launcher = ( + '@echo off\n' + '"%~dp0runtime\\python\\python.exe" -u -I -S -B ' + '"%~dp0app\\remote_worker_bootstrap.py" -- %*\n' + ) + with open(os.path.join(root, 'truf-worker.cmd'), 'w', encoding='ascii', newline='\r\n') as handle: + handle.write(launcher) + with open(os.path.join(root, 'run-worker.cmd'), 'w', encoding='ascii', newline='\r\n') as handle: + handle.write( + '@echo off\n' + '"%~dp0runtime\\python\\python.exe" -u -I -S -B ' + '"%~dp0app\\remote_worker_bootstrap.py" -- run %*\n' + ) + with open(os.path.join(root, 'prepare-worker.ps1'), 'w', encoding='ascii', newline='\r\n') as handle: + handle.write( + "$ErrorActionPreference = 'Stop'\n" + "$root = (Resolve-Path -LiteralPath $PSScriptRoot).Path\n" + "$sid = [System.Security.Principal.WindowsIdentity]::GetCurrent().User.Value\n" + "& icacls.exe $root /inheritance:r /grant:r " + "\"*$sid`:(OI)(CI)F\" \"*S-1-5-18`:(OI)(CI)F\" " + "\"*S-1-5-32-544`:(OI)(CI)F\" | Out-Null\n" + "if ($LASTEXITCODE -ne 0) { throw 'worker package ACL preparation failed' }\n" + "& icacls.exe (Join-Path $root '*') /inheritance:d /T /C | Out-Null\n" + "if ($LASTEXITCODE -ne 0) { throw 'worker package child ACL preparation failed' }\n" + ) + + +def _linux_support_files(root): + launcher = ( + '#!/bin/sh\n' + 'set -eu\n' + 'root=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd)\n' + 'exec python3 -u -I -S -B "$root/app/remote_worker_bootstrap.py" -- "$@"\n' + ) + for name, arguments in (('truf-worker', ''), ('run-worker', 'run ')): + path = os.path.join(root, name) + with open(path, 'w', encoding='ascii', newline='\n') as handle: + handle.write(launcher.replace('-- "$@"', f'-- {arguments}"$@"')) + os.chmod(path, 0o755) + prepare = ( + '#!/bin/sh\n' + 'set -eu\n' + 'test "$(id -u)" -eq 0 || { echo "prepare-worker.sh requires sudo" >&2; exit 1; }\n' + 'test -n "${SUDO_UID:-}" && test -n "${SUDO_GID:-}" && test "$SUDO_UID" -ne 0 || ' + '{ echo "run prepare-worker.sh through sudo as the worker user" >&2; exit 1; }\n' + 'root=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd)\n' + 'test -z "$(find "$root" -type l -print -quit)" || ' + '{ echo "worker package contains a symbolic link" >&2; exit 1; }\n' + 'chown 0:0 "$root"\n' + 'chmod 0755 "$root"\n' + 'chown -R "$SUDO_UID:$SUDO_GID" "$root/app"\n' + 'find "$root/app" -type d -exec chmod 0700 {} +\n' + 'find "$root/app" -type f -exec chmod 0600 {} +\n' + 'chown "$SUDO_UID:$SUDO_GID" "$root/worker-package.json"\n' + 'chmod 0600 "$root/worker-package.json"\n' + 'chown -R 0:0 "$root/bin" "$root/runtime"\n' + 'find "$root/bin" "$root/runtime" -type d -exec chmod 0755 {} +\n' + 'find "$root/bin" "$root/runtime" -type f -exec chmod go-w {} +\n' + 'chown 0:0 "$root/truf-worker" "$root/run-worker" "$root/prepare-worker.sh"\n' + 'chmod 0755 "$root/truf-worker" "$root/run-worker" "$root/prepare-worker.sh"\n' + ) + prepare_path = os.path.join(root, 'prepare-worker.sh') + with open(prepare_path, 'w', encoding='ascii', newline='\n') as handle: + handle.write(prepare) + os.chmod(prepare_path, 0o755) + + +def assemble_worker_package( + package_root, *, source_app, dependencies_root, detector_policy_source, + trufflehog_source, git_source_root, git_executable, platform_tag, + operator_readme_source, python_source_root=None, build_inputs_path=None, + operator_cheatsheet_sources=(), + capabilities=DEFAULT_WORKER_PACKAGE_CAPABILITIES, +): + package_root = os.path.abspath(os.fspath(package_root)) + parent = os.path.dirname(package_root) + if os.path.lexists(package_root): + raise WorkerPackageBuildError('worker package destination already exists') + if not os.path.isdir(parent): + raise WorkerPackageBuildError('worker package destination parent is absent') + staging = tempfile.mkdtemp(prefix='.truf-worker-build-', dir=parent) + try: + app_root = os.path.join(staging, 'app') + os.makedirs(app_root) + source_app = os.path.abspath(os.fspath(source_app)) + for name in REMOTE_WORKER_CODE_AUTHORITY_FILES: + _copy_file( + os.path.join(source_app, *name.split('/')), + os.path.join(app_root, *name.split('/')), + f'worker authority {name}', + ) + _copy_tree(dependencies_root, os.path.join(app_root, 'dependencies'), 'worker dependencies') + policy_relative = 'app/trufflehog-custom-detectors.yaml' + _copy_file( + detector_policy_source, + os.path.join(staging, *policy_relative.split('/')), + 'detector policy', + ) + + executable_name = 'trufflehog.exe' if platform_tag.startswith('windows-') else 'trufflehog' + trufflehog_relative = f'bin/{executable_name}' + _copy_file( + trufflehog_source, os.path.join(staging, *trufflehog_relative.split('/')), + 'TruffleHog executable', + ) + if not platform_tag.startswith('windows-'): + os.chmod(os.path.join(staging, *trufflehog_relative.split('/')), 0o755) + + git_root_relative = 'runtime/git' + _copy_tree(git_source_root, os.path.join(staging, 'runtime', 'git'), 'Git runtime') + git_executable = str(git_executable).replace('\\', '/').strip('/') + git_relative = f'{git_root_relative}/{git_executable}' + _regular_file(os.path.join(staging, *git_relative.split('/')), 'Git executable') + + python_root_relative = None + if platform_tag.startswith('windows-'): + if not python_source_root: + raise WorkerPackageBuildError('Windows package requires bundled Python') + python_root_relative = 'runtime/python' + _copy_tree( + python_source_root, os.path.join(staging, 'runtime', 'python'), + 'Python runtime', + ) + _regular_file( + os.path.join(staging, 'runtime', 'python', 'python.exe'), + 'Python executable', + ) + _windows_support_files(staging) + elif python_source_root: + raise WorkerPackageBuildError('Linux package must use image Python') + else: + _linux_support_files(staging) + + if build_inputs_path: + _copy_file( + build_inputs_path, os.path.join(staging, 'worker-build-inputs.json'), + 'worker build inputs', + ) + _copy_file( + operator_readme_source, os.path.join(staging, 'README_RU.md'), + 'Russian worker operator guide', + ) + for source in operator_cheatsheet_sources: + name = os.path.basename(os.fspath(source)) + if not name.startswith('remote-worker-cheatsheet-') or not name.endswith('-ru.md'): + raise WorkerPackageBuildError('worker operator cheatsheet name is invalid') + _copy_file(source, os.path.join(staging, name), 'worker operator cheatsheet') + manifest = build_worker_package_manifest( + staging, + trufflehog_path=trufflehog_relative, + git_path=git_relative, + detector_policy_path=policy_relative, + git_root=git_root_relative, + python_root=python_root_relative, + capabilities=[ + { + 'source': source, + 'platform': platform, + 'planning_kind': planning_kind, + } + for source, platform, planning_kind in capabilities + ], + platform_tag=platform_tag, + ) + manifest_path = write_worker_package_manifest( + os.path.join(staging, 'worker-package.json'), manifest, + ) + if platform_tag.startswith('windows-'): + harden_private_tree(staging) + verify_worker_package(manifest_path) + os.replace(staging, package_root) + staging = None + return manifest + finally: + if staging is not None: + shutil.rmtree(staging, ignore_errors=True) + + +def _download(url, destination, expected_sha256, expected_bytes): + if os.path.exists(destination): + if os.path.getsize(destination) != expected_bytes or sha256_file(destination) != expected_sha256: + raise WorkerPackageBuildError('cached public build input does not match its pin') + return destination + partial = destination + '.partial' + digest = hashlib.sha256() + size = 0 + try: + with urllib.request.urlopen(url, timeout=120) as response, open(partial, 'xb') as output: + while block := response.read(1024 * 1024): + digest.update(block) + size += len(block) + output.write(block) + if size != expected_bytes or digest.hexdigest() != expected_sha256: + raise WorkerPackageBuildError('downloaded public build input does not match its pin') + os.replace(partial, destination) + finally: + if os.path.exists(partial): + os.unlink(partial) + return destination + + +def _safe_unzip(archive_path, destination): + os.makedirs(destination, exist_ok=False) + root = os.path.abspath(destination) + with zipfile.ZipFile(archive_path) as archive: + for item in archive.infolist(): + name = item.filename.replace('\\', '/') + parts = [part for part in name.split('/') if part] + if not parts or name.startswith('/') or any(part in ('.', '..') for part in parts): + raise WorkerPackageBuildError('ZIP build input contains an unsafe path') + mode = item.external_attr >> 16 + if stat.S_ISLNK(mode): + raise WorkerPackageBuildError('ZIP build input contains a link') + target = os.path.abspath(os.path.join(root, *parts)) + if os.path.commonpath((root, target)) != root: + raise WorkerPackageBuildError('ZIP build input escapes its destination') + if item.is_dir(): + os.makedirs(target, exist_ok=True) + continue + os.makedirs(os.path.dirname(target), exist_ok=True) + with archive.open(item) as source, open(target, 'xb') as output: + shutil.copyfileobj(source, output) + + +def _extract_trufflehog(archive_path, destination): + with tarfile.open(archive_path, 'r:gz') as archive: + matches = [item for item in archive.getmembers() if item.name == 'trufflehog.exe'] + if len(matches) != 1 or not matches[0].isfile(): + raise WorkerPackageBuildError('TruffleHog archive executable is unavailable') + with archive.extractfile(matches[0]) as source, open(destination, 'xb') as output: + shutil.copyfileobj(source, output) + + +def _deterministic_zip(root, destination): + if os.path.lexists(destination): + raise WorkerPackageBuildError('worker archive destination already exists') + root = os.path.abspath(root) + with zipfile.ZipFile(destination, 'x', compression=zipfile.ZIP_DEFLATED, compresslevel=9) as archive: + for current, directories, names in os.walk(root, followlinks=False): + directories.sort() + names.sort() + for name in names: + path = _regular_file(os.path.join(current, name), 'worker archive file') + relative = os.path.relpath(path, root).replace(os.sep, '/') + item = zipfile.ZipInfo(relative, (1980, 1, 1, 0, 0, 0)) + item.compress_type = zipfile.ZIP_DEFLATED + item.external_attr = 0o100600 << 16 + with open(path, 'rb') as source: + archive.writestr(item, source.read()) + return destination + + +def _normalize_windows_dependency_artifacts(dependencies): + bin_root = os.path.join(dependencies, 'bin') + normalized = set() + if os.path.isdir(bin_root): + for name in sorted(os.listdir(bin_root)): + if not name.lower().endswith('.exe'): + continue + path = _regular_file( + os.path.join(bin_root, name), 'Windows dependency launcher', + ) + with zipfile.ZipFile(path) as archive: + entries = archive.infolist() + central_offset = archive.start_dir + if not entries: + raise WorkerPackageBuildError( + 'Windows dependency launcher has no embedded ZIP entries' + ) + payload = bytearray(open(path, 'rb').read()) + for entry in entries: + offset = entry.header_offset + if payload[offset:offset + 4] != b'PK\x03\x04': + raise WorkerPackageBuildError( + 'Windows dependency launcher local header is invalid' + ) + payload[offset + 10:offset + 14] = b'\x00\x00\x21\x00' + offset = central_offset + for _entry in entries: + if payload[offset:offset + 4] != b'PK\x01\x02': + raise WorkerPackageBuildError( + 'Windows dependency launcher central header is invalid' + ) + payload[offset + 12:offset + 16] = b'\x00\x00\x21\x00' + name_bytes, extra_bytes, comment_bytes = struct.unpack_from( + ' maximum: + raise WorkerSupervisorError('control JSON is invalid or oversized') + + def reject_duplicate(pairs): + value = {} + for key, item in pairs: + if key in value: + raise WorkerSupervisorError('control JSON contains duplicate fields') + value[key] = item + return value + + try: + value = json.loads( + payload.decode('ascii'), object_pairs_hook=reject_duplicate, + parse_constant=lambda _value: (_ for _ in ()).throw( + WorkerSupervisorError('control JSON constant is invalid') + ), + ) + except WorkerSupervisorError: + raise + except (UnicodeDecodeError, json.JSONDecodeError, TypeError, ValueError) as exc: + raise WorkerSupervisorError('control JSON is invalid') from exc + if not isinstance(value, dict) or not hmac.compare_digest(_canonical_json(value), payload): + raise WorkerSupervisorError('control JSON is not a canonical object') + return value + + +def encode_frame(value, maximum=CONTROL_MAX_RESPONSE_BYTES): + payload = _canonical_json(value) + if not payload or len(payload) > int(maximum): + raise WorkerSupervisorError('control frame exceeds its bound') + return struct.pack('!I', len(payload)) + payload + + +def _receive_exact(sock, size): + chunks = [] + remaining = int(size) + while remaining: + chunk = sock.recv(remaining) + if not chunk: + raise WorkerSupervisorError('control frame ended early') + chunks.append(chunk) + remaining -= len(chunk) + return b''.join(chunks) + + +def receive_frame(sock, maximum=CONTROL_MAX_REQUEST_BYTES): + header = _receive_exact(sock, 4) + length = struct.unpack('!I', header)[0] + if length < 2 or length > int(maximum): + raise WorkerSupervisorError('control frame length is invalid') + return _decode_canonical(_receive_exact(sock, length), int(maximum)) + + +def _package_identity(package_manifest): + verified = verify_worker_package(package_manifest) + manifest = verified['manifest'] + return { + 'schema': manifest['schema'], + 'manifest_sha256': worker_package_manifest_sha256(manifest), + 'code_manifest_sha256': verified['code_manifest_sha256'], + 'platform_tag': manifest['platform_tag'], + }, { + 'worker_protocol': manifest['protocol_version'], + 'bundle_format': manifest['bundle_format_version'], + 'event': WORKER_EVENT_SCHEMA, + 'control': CONTROL_SCHEMA, + 'projection': PROJECTION_SCHEMA, + } + + +def _runtime_identity(foreground): + return { + 'python': '.'.join(str(item) for item in sys.version_info[:3]), + 'platform': sys.platform, + 'executable': canonical_path(sys.executable), + 'mode': 'foreground' if foreground else 'detached', + } + + +def build_instance_record( + *, instance_id, token, identity, package, protocol, control_port, + foreground, started_at=None, lifecycle='running', +): + return validate_instance_record({ + 'schema': INSTANCE_SCHEMA, + 'instance_id': str(instance_id), + 'token': str(token), + 'pid': int(identity.pid), + 'process_creation_time': str(identity.creation_time), + 'executable': canonical_path(identity.executable), + 'package': dict(package), + 'runtime': _runtime_identity(foreground), + 'protocol': dict(protocol), + 'control': {'host': '127.0.0.1', 'port': int(control_port)}, + 'started_at': started_at or utc_now(), + 'lifecycle': str(lifecycle), + }) + + +def validate_instance_record(value): + fields = { + 'schema', 'instance_id', 'token', 'pid', 'process_creation_time', + 'executable', 'package', 'runtime', 'protocol', 'control', 'started_at', + 'lifecycle', + } + if not isinstance(value, dict) or set(value) != fields or value.get('schema') != INSTANCE_SCHEMA: + raise WorkerSupervisorError('worker instance record shape is invalid') + if not isinstance(value.get('instance_id'), str) or not value['instance_id']: + raise WorkerSupervisorError('worker instance identity is invalid') + if not isinstance(value.get('token'), str) or not 32 <= len(value['token']) <= 512: + raise WorkerSupervisorError('worker instance token is invalid') + if type(value.get('pid')) is not int or value['pid'] <= 0: + raise WorkerSupervisorError('worker instance PID is invalid') + if not isinstance(value.get('process_creation_time'), str) or not value['process_creation_time']: + raise WorkerSupervisorError('worker process creation identity is invalid') + if not isinstance(value.get('executable'), str) or not value['executable']: + raise WorkerSupervisorError('worker executable identity is invalid') + package = value.get('package') + if not isinstance(package, dict) or set(package) != { + 'schema', 'manifest_sha256', 'code_manifest_sha256', 'platform_tag', + }: + raise WorkerSupervisorError('worker package identity is invalid') + if any(not isinstance(package[name], str) or not package[name] for name in ( + 'manifest_sha256', 'code_manifest_sha256', 'platform_tag', + )) or type(package['schema']) is not int: + raise WorkerSupervisorError('worker package identity fields are invalid') + if ( + re.fullmatch(r'[0-9a-f]{64}', package['manifest_sha256']) is None + or re.fullmatch(r'[0-9a-f]{64}', package['code_manifest_sha256']) is None + or re.fullmatch(r'(windows|linux)-(x86_64|aarch64)', package['platform_tag']) is None + ): + raise WorkerSupervisorError('worker package identity fields are invalid') + runtime = value.get('runtime') + if not isinstance(runtime, dict) or set(runtime) != { + 'python', 'platform', 'executable', 'mode', + } or runtime.get('mode') not in {'foreground', 'detached'}: + raise WorkerSupervisorError('worker runtime identity is invalid') + if any(not isinstance(runtime.get(name), str) or not runtime[name] for name in ( + 'python', 'platform', 'executable', + )): + raise WorkerSupervisorError('worker runtime identity fields are invalid') + protocol = value.get('protocol') + if not isinstance(protocol, dict) or set(protocol) != { + 'worker_protocol', 'bundle_format', 'event', 'control', 'projection', + } or any(type(item) is not int for item in protocol.values()): + raise WorkerSupervisorError('worker protocol identity is invalid') + if any(item <= 0 for item in protocol.values()): + raise WorkerSupervisorError('worker protocol identity is invalid') + control = value.get('control') + if not isinstance(control, dict) or set(control) != {'host', 'port'}: + raise WorkerSupervisorError('worker control identity is invalid') + if control.get('host') != '127.0.0.1' or type(control.get('port')) is not int or not 0 < control['port'] <= 65535: + raise WorkerSupervisorError('worker control endpoint is invalid') + if value.get('lifecycle') not in {'starting', 'running', 'draining'}: + raise WorkerSupervisorError('worker lifecycle state is invalid') + if not isinstance(value.get('started_at'), str) or not value['started_at'].endswith('Z'): + raise WorkerSupervisorError('worker startup timestamp is invalid') + normalized = dict(value) + normalized['executable'] = canonical_path(value['executable']) + normalized['runtime'] = dict(runtime) + normalized['runtime']['executable'] = canonical_path(runtime['executable']) + if normalized['runtime']['executable'] != normalized['executable']: + raise WorkerSupervisorError('worker runtime executable identity is inconsistent') + normalized['package'] = dict(package) + normalized['protocol'] = dict(protocol) + normalized['control'] = dict(control) + return normalized + + +def load_instance(state_dir): + return validate_instance_record(read_private_json(instance_path(state_dir))) + + +def public_instance(record): + record = validate_instance_record(record) + return { + 'schema': record['schema'], + 'instance_id': record['instance_id'], + 'pid': record['pid'], + 'process_creation_time': record['process_creation_time'], + 'executable': record['executable'], + 'package': record['package'], + 'runtime': record['runtime'], + 'protocol': record['protocol'], + 'control': record['control'], + 'started_at': record['started_at'], + 'lifecycle': record['lifecycle'], + } + + +def _remove_exact_stale(state_dir, record, identity_state=exact_process_identity_state): + if identity_state( + record['pid'], record['process_creation_time'], record['executable'], + ) not in {'dead', 'reused'}: + return False + lock = PrivateFileLock(worker_lock_path(state_dir)) + try: + lock.acquire() + except OSError: + return False + try: + current = load_instance(state_dir) + if not hmac.compare_digest(current['instance_id'], record['instance_id']): + return False + if identity_state( + current['pid'], current['process_creation_time'], current['executable'], + ) not in {'dead', 'reused'}: + return False + durable_unlink(instance_path(state_dir)) + return True + except (OSError, ValueError, WorkerSupervisorError): + return False + finally: + lock.release() + + +def classify_instance( + state_dir, *, identity_state=exact_process_identity_state, + request=None, remove_stale=False, +): + path = instance_path(state_dir) + if not os.path.exists(path): + return {'state': 'stopped', 'instance': None, 'detail': 'no instance record'} + try: + record = load_instance(state_dir) + except (OSError, ValueError, WorkerSupervisorError) as exc: + return {'state': 'unverifiable', 'instance': None, 'detail': str(exc)} + state = identity_state( + record['pid'], record['process_creation_time'], record['executable'], + ) + if state in {'dead', 'reused'}: + removed = ( + _remove_exact_stale(state_dir, record, identity_state=identity_state) + if remove_stale else False + ) + return { + 'state': 'stale', 'instance': public_instance(record), + 'detail': f'process identity is {state}', 'reason': f'process_{state}', + 'removable': True, 'removed': removed, + } + if state != 'alive': + return { + 'state': 'unverifiable', 'instance': public_instance(record), + 'detail': 'process identity could not be verified', + } + try: + response = (request or send_control_request)(record, 'handshake', {}) + except (OSError, TimeoutError, ValueError, WorkerSupervisorError) as exc: + return { + 'state': 'stale', 'instance': public_instance(record), + 'detail': f'live process control handshake failed: {type(exc).__name__}', + 'reason': 'control_handshake_failed', + 'removable': False, + 'removed': False, + } + if not isinstance(response, dict) or set(response) != { + 'schema', 'instance_id', 'lifecycle', 'sequence', + } or ( + response.get('schema') != 1 + or response.get('instance_id') != record['instance_id'] + or response.get('lifecycle') not in {'starting', 'running', 'draining'} + or type(response.get('sequence')) is not int + or response['sequence'] < 0 + ): + return { + 'state': 'stale', 'instance': public_instance(record), + 'detail': 'control handshake response is invalid', + 'reason': 'control_handshake_invalid', + 'removable': False, + 'removed': False, + } + lifecycle = response['lifecycle'] + public = public_instance(record) + public['lifecycle'] = lifecycle + return { + 'state': lifecycle, 'instance': public, + 'detail': f'verified process and {lifecycle} control handshake', 'record': record, + 'handshake': response, + } + + +def capture_spawned_process_identity(process, *, opener=open_process): + if process is None or type(getattr(process, 'pid', None)) is not int or process.pid <= 0: + raise WorkerSupervisorError('spawned worker process handle is invalid') + retained = opener(process.pid) + try: + identity = retained.identity + if process.poll() is not None: + raise WorkerSupervisorError('spawned worker exited during identity capture') + return { + 'pid': identity.pid, + 'creation_time': identity.creation_time, + 'executable': identity.executable, + } + finally: + retained.close() + + +def terminate_spawned_process( + process, identity, *, identity_state=exact_process_identity_state, + timeout=5.0, +): + if process is None or process.poll() is not None: + return True + if not isinstance(identity, dict) or set(identity) != { + 'pid', 'creation_time', 'executable', + }: + raise WorkerSupervisorError('spawned worker exact identity is unavailable') + state = identity_state( + identity['pid'], identity['creation_time'], identity['executable'], + ) + if state != 'alive' or int(process.pid) != int(identity['pid']): + if process.poll() is not None: + return True + raise WorkerSupervisorError('spawned worker exact identity is no longer retained') + process.terminate() + try: + process.wait(timeout=max(0.1, float(timeout))) + except subprocess.TimeoutExpired: + state = identity_state( + identity['pid'], identity['creation_time'], identity['executable'], + ) + if state != 'alive': + return process.poll() is not None + process.kill() + process.wait(timeout=max(0.1, float(timeout))) + return process.poll() is not None + + +def send_control_request(record, action, parameters=None, timeout=CONTROL_TIMEOUT_SECONDS): + record = validate_instance_record(record) + request = { + 'schema': CONTROL_SCHEMA, + 'instance_id': record['instance_id'], + 'token': record['token'], + 'action': str(action), + 'parameters': dict(parameters or {}), + } + deadline = time.monotonic() + max(0.1, float(timeout)) + remaining = lambda: max(0.01, deadline - time.monotonic()) + with socket.create_connection( + (record['control']['host'], record['control']['port']), timeout=remaining(), + ) as connection: + connection.settimeout(remaining()) + connection.sendall(encode_frame(request, CONTROL_MAX_REQUEST_BYTES)) + response = receive_frame(connection, CONTROL_MAX_RESPONSE_BYTES) + expected = ( + {'schema', 'instance_id', 'ok', 'result'} + if response.get('ok') is True + else {'schema', 'instance_id', 'ok', 'error'} + ) + if ( + set(response) != expected + or response.get('schema') != CONTROL_SCHEMA + or response.get('instance_id') != record['instance_id'] + or type(response.get('ok')) is not bool + ): + raise WorkerSupervisorError('control response shape is invalid') + if not response['ok']: + if not isinstance(response.get('error'), str): + raise WorkerSupervisorError('control response error is invalid') + raise WorkerSupervisorError(response['error']) + if not isinstance(response.get('result'), dict): + raise WorkerSupervisorError('control response result is invalid') + return response['result'] + + +class _ControlHandler(socketserver.BaseRequestHandler): + def handle(self): + self.request.settimeout(CONTROL_TIMEOUT_SECONDS) + try: + request = receive_frame(self.request, CONTROL_MAX_REQUEST_BYTES) + response = self.server.runtime.control_request(request) + except Exception: + response = self.server.runtime.control_error('invalid control request') + try: + self.request.sendall(encode_frame(response, CONTROL_MAX_RESPONSE_BYTES)) + except OSError: + pass + + +class _ControlServer(socketserver.ThreadingMixIn, socketserver.TCPServer): + allow_reuse_address = False + daemon_threads = True + request_queue_size = 16 + + def __init__(self, address, runtime): + self.runtime = runtime + self._workers = threading.BoundedSemaphore(CONTROL_MAX_WORKERS) + super().__init__(address, _ControlHandler) + + def server_bind(self): + if os.name == 'nt' and hasattr(socket, 'SO_EXCLUSIVEADDRUSE'): + self.socket.setsockopt(socket.SOL_SOCKET, socket.SO_EXCLUSIVEADDRUSE, 1) + super().server_bind() + + def process_request(self, request, client_address): + if not self._workers.acquire(blocking=False): + try: + request.sendall(encode_frame( + self.runtime.control_error('control worker limit reached'), + CONTROL_MAX_RESPONSE_BYTES, + )) + finally: + self.shutdown_request(request) + return + try: + super().process_request(request, client_address) + except BaseException: + self._workers.release() + raise + + def process_request_thread(self, request, client_address): + try: + super().process_request_thread(request, client_address) + finally: + self._workers.release() + + +class WorkerSupervisor: + def __init__( + self, args, *, foreground=True, local_state_factory=WorkerLocalState, + monotonic=time.monotonic, wall_time=time.time, + retention_interval_seconds=300.0, + hard_exit_hook=os._exit, + ): + self.args = args + self.foreground = bool(foreground) + self.state_dir = ensure_private_directory(os.path.abspath(args.state_dir), reject_reparse=True) + self.control_dir = os.path.join(self.state_dir, 'control') + self.local = None + self._local_state_factory = local_state_factory + self._monotonic = monotonic + self._wall_time = wall_time + self._retention_interval = max(1.0, float(retention_interval_seconds)) + self._hard_exit_hook = hard_exit_hook + self.instance_id = secrets.token_urlsafe(24) + self.token = secrets.token_urlsafe(48) + self.identity = current_process_identity() + self.package, self.protocol = _package_identity(args.package_manifest) + self.started_at = utc_now() + self.drain_event = threading.Event() + self.stop_event = threading.Event() + self.drain_deadline = None + self._drain_deadline_monotonic = None + self._deadline_escalated = False + self.record = None + self.server = None + self.server_thread = None + self._lock = PrivateFileLock(worker_lock_path(self.state_dir)) + self._record_lock = threading.RLock() + self._startup_ready = False + self._maintenance_stop = threading.Event() + self._maintenance_wakeup = threading.Event() + self._maintenance_thread = None + self._maintenance_started = False + self._next_retention_cleanup = None + self._server_started = False + self._hard_exit_invoked = False + + def _initialize_local_state(self): + if self.local is None: + self.local = self._local_state_factory( + self.state_dir, + log_bytes=getattr(self.args, 'log_bytes', 2 * 1024 * 1024), + log_files=getattr(self.args, 'log_files', 5), + retention_days=getattr(self.args, 'retention_days', 30), + retention_bytes=getattr(self.args, 'retention_bytes', 1024 * 1024 * 1024), + ) + return self.local + + def _public_worker(self): + projection = self.local.snapshot() + configured = set(range(int(self.args.parallelism))) + try: + slots = configured | persisted_slot_ids(self.state_dir) + except OSError: + slots = configured + return { + 'state': ( + 'draining' if self.drain_event.is_set() + else (self.record or {}).get('lifecycle', 'starting') + ), + 'parallelism': int(self.args.parallelism), + 'slot_cap': len(configured), + 'configured_slots': len(configured), + 'recovery_slots': len(slots - configured), + 'started_at': self.started_at, + 'drain_deadline_at': self.drain_deadline, + 'aggregate': projection['aggregate'], + } + + def snapshot(self): + return { + 'schema': PROJECTION_SCHEMA, + 'instance': public_instance(self.record), + 'worker': self._public_worker(), + 'projection': self.local.snapshot(), + 'retention': self.local.retention_usage({ + 'bundles': self.args.bundle_dir, + 'work': self.args.work_dir, + }), + } + + def control_error(self, message): + return { + 'schema': CONTROL_SCHEMA, + 'instance_id': self.instance_id, + 'ok': False, + 'error': str(message), + } + + def control_ok(self, result): + return { + 'schema': CONTROL_SCHEMA, + 'instance_id': self.instance_id, + 'ok': True, + 'result': dict(result), + } + + def control_request(self, request): + if not isinstance(request, dict) or set(request) != { + 'schema', 'instance_id', 'token', 'action', 'parameters', + }: + return self.control_error('control request shape is invalid') + if ( + request.get('schema') != CONTROL_SCHEMA + or not isinstance(request.get('parameters'), dict) + or not hmac.compare_digest(str(request.get('instance_id') or ''), self.instance_id) + or not hmac.compare_digest(str(request.get('token') or ''), self.token) + ): + return self.control_error('control authentication failed') + action = request.get('action') + parameters = request['parameters'] + if action == 'handshake' and not parameters: + return self.control_ok({ + 'schema': 1, + 'instance_id': self.instance_id, + 'lifecycle': ( + 'draining' if self.drain_event.is_set() + else (self.record or {}).get('lifecycle', 'starting') + ), + 'sequence': self.local.snapshot()['sequence'], + }) + if action == 'snapshot' and not parameters: + return self.control_ok(self.snapshot()) + if action == 'events' and set(parameters) == {'after_sequence', 'limit'}: + try: + events = self.local.events_after( + parameters['after_sequence'], parameters['limit'], + ) + except (TypeError, ValueError): + return self.control_error('event request bounds are invalid') + return self.control_ok({ + 'schema': 1, + 'events': events, + 'last_sequence': self.local.snapshot()['sequence'], + }) + if action == 'stop' and set(parameters) == {'timeout_seconds'}: + try: + timeout = float(parameters['timeout_seconds']) + except (TypeError, ValueError, OverflowError): + return self.control_error('stop timeout is invalid') + if not 0.1 <= timeout <= 3600: + return self.control_error('stop timeout is invalid') + self.request_drain(timeout) + return self.control_ok({ + 'schema': 1, + 'accepted': True, + 'drain_deadline_at': self.drain_deadline, + 'slots': self.local.snapshot()['slots'], + }) + return self.control_error('control action is invalid') + + def _update_lifecycle(self, lifecycle): + with self._record_lock: + if self.record is None or self.record['lifecycle'] == lifecycle: + return + self.record = dict(self.record) + self.record['lifecycle'] = lifecycle + self.record = validate_instance_record(self.record) + atomic_write_private_json(instance_path(self.state_dir), self.record) + + def request_drain(self, timeout=30.0): + timeout = max(0.1, float(timeout)) + monotonic_deadline = self._monotonic() + timeout + deadline = datetime.fromtimestamp( + self._wall_time() + timeout, timezone.utc, + ).isoformat(timespec='milliseconds').replace('+00:00', 'Z') + if ( + self._drain_deadline_monotonic is None + or monotonic_deadline < self._drain_deadline_monotonic + ): + self._drain_deadline_monotonic = monotonic_deadline + self.drain_deadline = deadline + self.drain_event.set() + self._update_lifecycle('draining') + self.local.log('graceful drain requested') + self._maintenance_wakeup.set() + + def request_interrupt(self, signum): + if self.drain_event.is_set(): + self._deadline_escalated = True + self.stop_event.set() + self.local.log(f'interrupt {signum} forced supervisor exit') + self._maintenance_wakeup.set() + return 'forced' + self.request_drain(30.0) + self.local.log(f'interrupt {signum} requested graceful drain') + return 'draining' + + def _maintenance_tick(self): + now = self._monotonic() + if ( + self._drain_deadline_monotonic is not None + and now >= self._drain_deadline_monotonic + and not self.stop_event.is_set() + ): + self._deadline_escalated = True + self.stop_event.set() + self.local.log('graceful drain deadline expired; stopping controller') + if self._next_retention_cleanup is None or now >= self._next_retention_cleanup: + self.local.cleanup_retention({ + 'bundles': self.args.bundle_dir, + 'work': self.args.work_dir, + }) + if os.path.isdir(self.args.work_dir): + cleanup_abandoned_runner_roots( + self.args.work_dir, minimum_age_sec=60, + active_root_names=persisted_runner_root_names(self.state_dir), + ) + self._next_retention_cleanup = now + self._retention_interval + + def _maintenance_loop(self): + while not self._maintenance_stop.is_set(): + try: + self._maintenance_tick() + except Exception as exc: + self.local.log(f'worker retention maintenance failed: {type(exc).__name__}') + wait_seconds = 1.0 + if self._drain_deadline_monotonic is not None: + wait_seconds = min( + wait_seconds, + max(0.01, self._drain_deadline_monotonic - self._monotonic()), + ) + self._maintenance_wakeup.wait(wait_seconds) + self._maintenance_wakeup.clear() + + def _start_maintenance(self): + self._next_retention_cleanup = self._monotonic() + self._retention_interval + self._maintenance_thread = threading.Thread( + target=self._maintenance_loop, + name='worker-maintenance', + daemon=True, + ) + self._maintenance_thread.start() + self._maintenance_started = True + + def _stop_maintenance(self): + self._maintenance_stop.set() + self._maintenance_wakeup.set() + if self._maintenance_started and self._maintenance_thread is not None: + self._maintenance_thread.join(timeout=5) + self._maintenance_started = False + + def _event(self, value): + return self.local.emit_phase( + self.instance_id, + value['slot_id'], + value['phase'], + reservation_id=value.get('reservation_id'), + source=value.get('source'), + scan_deadline_at=value.get('scan_deadline_at'), + assignment_deadline_at=value.get('assignment_deadline_at'), + progress=value.get('progress'), + timestamp=value.get('timestamp'), + phase_started_at=value.get('phase_started_at'), + ) + + def _diagnostic(self, envelope, full_materials=None): + return self.local.archive_diagnostic(envelope, full_materials) + + def _terminal(self, value): + completed = value.get('completed_at') or utc_now() + timeline = self.local.assignment_timeline( + value['slot_id'], value['reservation_id'], + ) + phase_durations = {} + for index, event in enumerate(timeline): + end = timeline[index + 1]['timestamp'] if index + 1 < len(timeline) else completed + try: + started_value = datetime.fromisoformat(event['timestamp'].replace('Z', '+00:00')) + ended_value = datetime.fromisoformat(end.replace('Z', '+00:00')) + duration = max(0.0, (ended_value - started_value).total_seconds()) + except ValueError: + duration = 0.0 + phase_durations[event['phase']] = round( + phase_durations.get(event['phase'], 0.0) + duration, 6, + ) + diagnostics = { + reference['diagnostic_uid']: reference + for reference in self.local.diagnostic_references(value['reservation_id']) + } + diagnostics.update({ + reference['diagnostic_uid']: reference + for reference in value.get('diagnostics') or [] + }) + record = { + 'history_id': str(value['history_id']), + 'instance_id': self.instance_id, + 'slot_id': int(value['slot_id']), + 'reservation_id': int(value['reservation_id']), + 'source': value.get('source'), + 'outcome': str(value['outcome']), + 'receipt': dict(value.get('receipt') or {}), + 'started_at': value.get('started_at'), + 'completed_at': completed, + 'duration_seconds': value.get('duration_seconds'), + 'first_sequence': timeline[0]['sequence'] if timeline else value.get('first_sequence'), + 'last_sequence': timeline[-1]['sequence'] if timeline else self.local.snapshot()['sequence'], + 'diagnostics': [diagnostics[key] for key in sorted(diagnostics)], + 'timeline': timeline, + 'phase_durations': phase_durations, + } + self.local.append_history(record) + + def _publish_instance(self): + self._initialize_local_state() + path = instance_path(self.state_dir) + if os.path.exists(path): + previous = load_instance(self.state_dir) + state = exact_process_identity_state( + previous['pid'], previous['process_creation_time'], previous['executable'], + ) + if state not in {'dead', 'reused'}: + raise WorkerInstanceUnverifiable('existing worker instance is not exactly stale') + durable_unlink(path) + if os.path.exists(shutdown_path(self.state_dir)): + durable_unlink(shutdown_path(self.state_dir)) + try: + self.server = _ControlServer(('127.0.0.1', 0), self) + self.record = build_instance_record( + instance_id=self.instance_id, + token=self.token, + identity=self.identity, + package=self.package, + protocol=self.protocol, + control_port=self.server.server_address[1], + foreground=self.foreground, + started_at=self.started_at, + lifecycle='starting', + ) + write_private_json_exclusive(path, self.record) + self.server_thread = threading.Thread( + target=self.server.serve_forever, + name='worker-control', + daemon=True, + ) + self.server_thread.start() + self._server_started = True + except BaseException: + self._remove_instance() + if self.server is not None: + self.server.server_close() + self.server = None + self.server_thread = None + self.record = None + raise + + def _remove_instance(self): + path = instance_path(self.state_dir) + try: + current = load_instance(self.state_dir) + if hmac.compare_digest(current['instance_id'], self.instance_id): + durable_unlink(path) + except (OSError, ValueError, WorkerSupervisorError): + pass + + def _write_receipt(self, exit_code, *, drained=None): + if drained is None: + try: + pending_slots = persisted_slot_ids(self.state_dir) + except OSError: + pending_slots = {-1} + drained = not pending_slots + atomic_write_private_json(shutdown_path(self.state_dir), { + 'schema': SHUTDOWN_SCHEMA, + 'instance_id': self.instance_id, + 'completed_at': utc_now(), + 'exit_code': int(exit_code), + 'drained': bool(drained), + 'last_sequence': self.local.snapshot()['sequence'], + }) + + def _shutdown_control(self): + if self.server is not None: + if self._server_started: + self.server.shutdown() + self.server.server_close() + if self._server_started and self.server_thread is not None: + self.server_thread.join(timeout=5) + self._server_started = False + + def run(self, startup_file=None, launch_nonce=None): + try: + self._lock.acquire() + except OSError as exc: + if startup_file: + write_startup_result(startup_file, launch_nonce, 'already_running', error='singleton lock is held') + raise WorkerAlreadyRunning('another worker supervisor is already active') from exc + previous_handlers = {} + exit_code = 1 + try: + self._initialize_local_state() + self._publish_instance() + prepare_progress_outbox_cursor(self.state_dir, create=True) + self._start_maintenance() + self.local.log('worker supervisor started') + + def started(): + self._update_lifecycle('running') + if startup_file: + write_startup_result( + startup_file, launch_nonce, 'ready', + instance_id=self.instance_id, pid=self.identity.pid, + ) + self._startup_ready = True + + def interrupted(signum, _frame): + self.request_interrupt(signum) + + if threading.current_thread() is threading.main_thread(): + for name in ('SIGINT', 'SIGTERM'): + current = getattr(signal, name, None) + if current is not None: + previous_handlers[current] = signal.getsignal(current) + signal.signal(current, interrupted) + client_exit = run_client( + self.args, + drain_event=self.drain_event, + stop_event=self.stop_event, + event_callback=self._event, + terminal_callback=self._terminal, + log_callback=self.local.log, + acquire_lock=False, + started_callback=started, + diagnostic_callback=self._diagnostic, + progress_event_reader=self.local.events_after, + ) + exit_code = int(client_exit or 0) + if self._deadline_escalated and exit_code == 0: + exit_code = 2 + except BaseException as exc: + if self.local is not None: + self.local.log(f'worker supervisor failed: {type(exc).__name__}') + if startup_file and not self._startup_ready: + write_startup_result(startup_file, launch_nonce, 'failed', error=type(exc).__name__) + if isinstance(exc, (KeyboardInterrupt, SystemExit)): + exit_code = int(getattr(exc, 'code', 1) or 0) + elif isinstance(exc, WorkerAlreadyRunning): + raise + finally: + self._stop_maintenance() + for current, previous in previous_handlers.items(): + signal.signal(current, previous) + try: + if self.record is not None: + try: + pending_slots = persisted_slot_ids(self.state_dir) + except OSError: + pending_slots = {-1} + for slot in self.local.snapshot()['slots']: + if slot['phase'] != WorkerPhase.STOPPED.value: + if slot['phase'] != WorkerPhase.DRAINING.value: + try: + self.local.emit_phase( + self.instance_id, slot['slot_id'], WorkerPhase.DRAINING, + reservation_id=slot['reservation_id'], source=slot['source'], + scan_deadline_at=slot['scan_deadline_at'], + assignment_deadline_at=slot['assignment_deadline_at'], + progress={'reason': 'supervisor_shutdown'}, + ) + except ValueError: + pass + if slot['slot_id'] not in pending_slots and not self._deadline_escalated: + try: + self.local.emit_phase( + self.instance_id, slot['slot_id'], WorkerPhase.STOPPED, + progress={'reason': 'supervisor_shutdown'}, + ) + except ValueError: + pass + if self._deadline_escalated: + forced_code = int(exit_code or 2) + if forced_code == 0: + forced_code = 2 + self._hard_exit_invoked = True + self._write_receipt(forced_code, drained=False) + self._hard_exit_hook(forced_code) + return forced_code + try: + self._write_receipt(exit_code) + finally: + self._remove_instance() + self._shutdown_control() + if not self._deadline_escalated: + try: + self.local.cleanup_retention({ + 'bundles': self.args.bundle_dir, + 'work': self.args.work_dir, + }) + except Exception as exc: + self.local.log(f'worker retention cleanup failed: {type(exc).__name__}') + finally: + if not self._hard_exit_invoked: + self._lock.release() + return exit_code + + +def write_startup_result(path, launch_nonce, outcome, *, instance_id=None, pid=None, error=None): + value = { + 'schema': STARTUP_SCHEMA, + 'launch_nonce': str(launch_nonce or ''), + 'outcome': str(outcome), + 'instance_id': str(instance_id) if instance_id else None, + 'pid': int(pid) if pid else None, + 'error': str(error) if error else None, + 'completed_at': utc_now(), + } + if value['outcome'] not in {'ready', 'already_running', 'failed'}: + raise WorkerSupervisorError('startup result outcome is invalid') + atomic_write_private_json(path, value) + return value + + +def load_startup_result(path, launch_nonce): + value = read_private_json(path) + if not isinstance(value, dict) or set(value) != { + 'schema', 'launch_nonce', 'outcome', 'instance_id', 'pid', 'error', + 'completed_at', + } or value.get('schema') != STARTUP_SCHEMA: + raise WorkerSupervisorError('startup result shape is invalid') + if not hmac.compare_digest(str(value.get('launch_nonce') or ''), str(launch_nonce)): + raise WorkerSupervisorError('startup result nonce mismatch') + if value.get('outcome') not in {'ready', 'already_running', 'failed'}: + raise WorkerSupervisorError('startup result outcome is invalid') + if not isinstance(value.get('completed_at'), str) or not value['completed_at'].endswith('Z'): + raise WorkerSupervisorError('startup result timestamp is invalid') + if value['outcome'] == 'ready': + if ( + not isinstance(value.get('instance_id'), str) or not value['instance_id'] + or type(value.get('pid')) is not int or value['pid'] <= 0 + or value.get('error') is not None + ): + raise WorkerSupervisorError('ready startup result is invalid') + elif value.get('instance_id') is not None or value.get('pid') is not None: + raise WorkerSupervisorError('failed startup result is invalid') + if value.get('error') is not None and not isinstance(value['error'], str): + raise WorkerSupervisorError('startup result error is invalid') + return value + + +def detached_command(bootstrap_path, launch_file, startup_file, launch_nonce): + return [ + sys.executable, '-u', '-I', '-S', '-B', os.path.abspath(bootstrap_path), '--', + '_supervise', '--launch-file', os.path.abspath(launch_file), + '--startup-file', os.path.abspath(startup_file), + '--launch-nonce', str(launch_nonce), + ] + + +def spawn_detached(command, *, popen=subprocess.Popen, platform_name=None): + platform_name = os.name if platform_name is None else platform_name + options = { + 'stdin': subprocess.DEVNULL, + 'stdout': subprocess.DEVNULL, + 'stderr': subprocess.DEVNULL, + 'close_fds': True, + } + if platform_name == 'nt': + options['creationflags'] = ( + getattr(subprocess, 'CREATE_NEW_PROCESS_GROUP', 0) + | getattr(subprocess, 'DETACHED_PROCESS', 0) + ) + else: + options['start_new_session'] = True + return popen(command, **options) + + +def wait_for_startup(startup_file, launch_nonce, process, timeout=30.0): + deadline = time.monotonic() + max(0.1, float(timeout)) + while time.monotonic() < deadline: + if os.path.exists(startup_file): + return load_startup_result(startup_file, launch_nonce) + if process.poll() is not None: + raise WorkerSupervisorError('worker supervisor exited before startup handshake') + time.sleep(0.05) + raise WorkerSupervisorError('worker supervisor startup handshake timed out') + + +def load_shutdown_receipt(state_dir, expected_instance_id=None): + value = read_private_json(shutdown_path(state_dir)) + if not isinstance(value, dict) or set(value) != { + 'schema', 'instance_id', 'completed_at', 'exit_code', 'drained', + 'last_sequence', + } or value.get('schema') != SHUTDOWN_SCHEMA: + raise WorkerSupervisorError('shutdown receipt shape is invalid') + if expected_instance_id is not None and not hmac.compare_digest( + str(value.get('instance_id') or ''), str(expected_instance_id), + ): + raise WorkerSupervisorError('shutdown receipt instance mismatch') + if ( + not isinstance(value.get('instance_id'), str) or not value['instance_id'] + or not isinstance(value.get('completed_at'), str) or not value['completed_at'].endswith('Z') + or type(value.get('exit_code')) is not int + or type(value.get('drained')) is not bool + or type(value.get('last_sequence')) is not int or value['last_sequence'] < 0 + ): + raise WorkerSupervisorError('shutdown receipt fields are invalid') + return value diff --git a/attach_runtime.ps1 b/attach_runtime.ps1 new file mode 100644 index 0000000..a624000 --- /dev/null +++ b/attach_runtime.ps1 @@ -0,0 +1,11 @@ +$ErrorActionPreference = 'Stop' +throw 'Docker development copy: runtime control is disabled until container isolation is ready. See DOCKER_MIGRATION.md.' + +$Root = Split-Path -Parent $MyInvocation.MyCommand.Path +$RuntimeBootstrap = Join-Path $Root 'app\runtime_bootstrap.py' +$Supervisor = (Resolve-Path -LiteralPath (Join-Path $Root 'app\supervisor.py')).Path +$Config = Join-Path $Root 'app\config.yaml' + +python -I -S -B $RuntimeBootstrap supervisor -- --runtime-bootstrap-entrypoint $Supervisor --config $Config --attach +$SupervisorExit = $LASTEXITCODE +exit $SupervisorExit diff --git a/build_worker_release.cmd b/build_worker_release.cmd new file mode 100644 index 0000000..435c36d --- /dev/null +++ b/build_worker_release.cmd @@ -0,0 +1,4 @@ +@echo off +setlocal +powershell.exe -NoLogo -NoProfile -ExecutionPolicy Bypass -File "%~dp0build_worker_release.ps1" %* +exit /b %ERRORLEVEL% diff --git a/build_worker_release.ps1 b/build_worker_release.ps1 new file mode 100644 index 0000000..4ec3a89 --- /dev/null +++ b/build_worker_release.ps1 @@ -0,0 +1,396 @@ +[CmdletBinding()] +param( + [ValidatePattern('^release-[A-Za-z0-9][A-Za-z0-9._-]{0,63}$')] + [string]$ReleaseName = (Get-Date -Format "'release-'yyyyMMdd-HHmmss"), + [string]$WslDistro = 'Ubuntu-24.04', + [string]$Python = 'python', + [switch]$SkipTests +) + +Set-StrictMode -Version Latest +$ErrorActionPreference = 'Stop' + +$ProjectRoot = $PSScriptRoot +$DistRoot = Join-Path $ProjectRoot 'dist' +$ReleaseRoot = Join-Path $DistRoot $ReleaseName +$ReleaseId = $ReleaseName.Substring('release-'.Length) +$ImageTag = 'truf-remote-worker:linux-x86_64' +$DockerMode = $null + +function Invoke-Checked { + param( + [Parameter(Mandatory = $true)][string]$Command, + [Parameter(Mandatory = $true)][string[]]$Arguments, + [Parameter(Mandatory = $true)][string]$Label + ) + & $Command @Arguments + if ($LASTEXITCODE -ne 0) { + throw "$Label failed with exit code $LASTEXITCODE" + } +} + +function Invoke-DockerChecked { + param( + [Parameter(Mandatory = $true)][string[]]$Arguments, + [Parameter(Mandatory = $true)][string]$Label + ) + if ($script:DockerMode -eq 'native') { + Invoke-Checked -Command 'docker' -Arguments $Arguments -Label $Label + return + } + $prefix = @( + '-d', $script:WslDistro, '--', 'sudo', '-n', 'docker', + '-H', 'unix:///var/run/docker.sock' + ) + Invoke-Checked -Command 'wsl.exe' -Arguments ($prefix + $Arguments) -Label $Label +} + +function Get-DockerText { + param( + [Parameter(Mandatory = $true)][string[]]$Arguments, + [Parameter(Mandatory = $true)][string]$Label + ) + if ($script:DockerMode -eq 'native') { + $output = @(& docker @Arguments) + } + else { + $prefix = @( + '-d', $script:WslDistro, '--', 'sudo', '-n', 'docker', + '-H', 'unix:///var/run/docker.sock' + ) + $output = @(& wsl.exe @prefix @Arguments) + } + if ($LASTEXITCODE -ne 0) { + throw "$Label failed with exit code $LASTEXITCODE" + } + return ($output -join "`n") +} + +function Get-WslPath { + param([Parameter(Mandatory = $true)][string]$Path) + $fullPath = [IO.Path]::GetFullPath($Path) + if ($fullPath -notmatch '^(?[A-Za-z]):\\(?.*)$') { + throw "Only absolute Windows drive paths can be translated for WSL: $Path" + } + $drive = $Matches['drive'].ToLowerInvariant() + $tail = $Matches['tail'].Replace('\', '/') + return "/mnt/$drive/$tail" +} + +function Get-Sha256 { + param([Parameter(Mandatory = $true)][string]$Path) + return (Get-FileHash -LiteralPath $Path -Algorithm SHA256).Hash.ToLowerInvariant() +} + +function Write-AsciiLines { + param( + [Parameter(Mandatory = $true)][string]$Path, + [Parameter(Mandatory = $true)][string[]]$Lines + ) + [IO.File]::WriteAllText( + $Path, + (($Lines -join "`n") + "`n"), + [Text.Encoding]::ASCII + ) +} + +function Write-Sha256Manifest { + param( + [Parameter(Mandatory = $true)][string]$Path, + [Parameter(Mandatory = $true)][object[]]$Files + ) + $lines = foreach ($file in $Files) { + if ($file -is [string]) { + $source = $file + $name = [IO.Path]::GetFileName($file) + } + else { + $source = [string]$file.Path + $name = [string]$file.Name + } + "$(Get-Sha256 -Path $source) $name" + } + Write-AsciiLines -Path $Path -Lines $lines +} + +function Compress-Gzip { + param( + [Parameter(Mandatory = $true)][string]$InputPath, + [Parameter(Mandatory = $true)][string]$OutputPath + ) + $source = [IO.File]::OpenRead($InputPath) + try { + $destination = [IO.File]::Open( + $OutputPath, [IO.FileMode]::CreateNew, [IO.FileAccess]::Write, + [IO.FileShare]::None + ) + try { + $gzip = New-Object IO.Compression.GZipStream( + $destination, [IO.Compression.CompressionLevel]::Fastest, $true + ) + try { + $source.CopyTo($gzip) + } + finally { + $gzip.Dispose() + } + } + finally { + $destination.Dispose() + } + } + finally { + $source.Dispose() + } +} + +function New-StoredZip { + param( + [Parameter(Mandatory = $true)][string]$Path, + [Parameter(Mandatory = $true)][object[]]$Files + ) + Add-Type -AssemblyName System.IO.Compression + Add-Type -AssemblyName System.IO.Compression.FileSystem + $stream = [IO.File]::Open( + $Path, [IO.FileMode]::CreateNew, [IO.FileAccess]::Write, + [IO.FileShare]::None + ) + try { + $zip = New-Object IO.Compression.ZipArchive( + $stream, [IO.Compression.ZipArchiveMode]::Create, $false + ) + try { + foreach ($file in $Files) { + $entry = $zip.CreateEntry( + [string]$file.Name, + [IO.Compression.CompressionLevel]::NoCompression + ) + $entry.LastWriteTime = [DateTimeOffset]::Parse( + '1980-01-01T00:00:00Z' + ) + $input = [IO.File]::OpenRead([string]$file.Path) + try { + $output = $entry.Open() + try { + $input.CopyTo($output) + } + finally { + $output.Dispose() + } + } + finally { + $input.Dispose() + } + } + } + finally { + $zip.Dispose() + } + } + finally { + $stream.Dispose() + } +} + +if (-not (Test-Path -LiteralPath $DistRoot -PathType Container)) { + throw "Distribution directory is missing: $DistRoot" +} +if (Test-Path -LiteralPath $ReleaseRoot) { + throw "Release destination already exists: $ReleaseRoot" +} +if (-not (Get-Command $Python -ErrorAction SilentlyContinue)) { + throw "Python command is unavailable: $Python" +} + +$nativeDocker = Get-Command docker -ErrorAction SilentlyContinue +if ($null -ne $nativeDocker) { + & docker version --format '{{.Server.Version}}' *> $null + if ($LASTEXITCODE -eq 0) { + $DockerMode = 'native' + } +} +if ($null -eq $DockerMode) { + if (-not (Get-Command wsl.exe -ErrorAction SilentlyContinue)) { + throw 'Neither native Docker nor WSL is available' + } + & wsl.exe -d $WslDistro -- sudo -n docker -H unix:///var/run/docker.sock ` + version --format '{{.Server.Version}}' *> $null + if ($LASTEXITCODE -ne 0) { + throw "Docker is unavailable in WSL distribution $WslDistro" + } + $DockerMode = 'wsl' +} + +Write-Host "Release: $ReleaseRoot" +Write-Host "Docker: $DockerMode" + +if (-not $SkipTests) { + Invoke-Checked -Command $Python -Arguments @( + '-B', '-m', 'pytest', 'tests/test_worker_package.py', '-q' + ) -Label 'Worker package tests' +} + +New-Item -ItemType Directory -Path $ReleaseRoot | Out-Null + +$windowsRoot = Join-Path $ReleaseRoot 'truf-worker-windows-x86_64' +$windowsZip = Join-Path $ReleaseRoot 'truf-worker-windows-x86_64.zip' +$cacheRoot = Join-Path $ProjectRoot 'build\worker-cache' +Invoke-Checked -Command $Python -Arguments @( + '-B', 'app/worker_package_builder.py', 'windows', + '--project-root', $ProjectRoot, + '--output', $windowsRoot, + '--archive', $windowsZip, + '--cache', $cacheRoot +) -Label 'Windows worker package build' + +$windowsManifest = Join-Path $windowsRoot 'worker-package.json' +$windowsManifestSnapshot = Join-Path $ReleaseRoot ( + "windows-worker-package-$ReleaseId.json" +) +Copy-Item -LiteralPath $windowsManifest -Destination $windowsManifestSnapshot + +$buildContext = if ($DockerMode -eq 'wsl') { + Get-WslPath -Path $ProjectRoot +} +else { + $ProjectRoot +} +Invoke-DockerChecked -Arguments @( + 'build', '--provenance=false', '--target', 'worker', + '--tag', $ImageTag, $buildContext +) -Label 'Linux worker image build' + +$linuxManifestSnapshot = Join-Path $ReleaseRoot ( + "linux-worker-package-$ReleaseId.json" +) +$linuxManifestText = Get-DockerText -Arguments @( + 'run', '--rm', '--network', 'none', '--entrypoint', '/bin/cat', + $ImageTag, '/opt/truf-worker/worker-package.json' +) -Label 'Linux worker manifest read' +$null = $linuxManifestText | ConvertFrom-Json +[IO.File]::WriteAllText( + $linuxManifestSnapshot, + ($linuxManifestText.TrimEnd() + "`n"), + (New-Object Text.UTF8Encoding($false)) +) + +$linuxNativeArchive = Join-Path $ReleaseRoot ( + 'truf-worker-native-linux-x86_64.tar.gz' +) +$releaseMount = if ($DockerMode -eq 'wsl') { + Get-WslPath -Path $ReleaseRoot +} +else { + $ReleaseRoot +} +Invoke-DockerChecked -Arguments @( + 'run', '--rm', '--network', 'none', '--user', '0:0', + '--entrypoint', '/bin/sh', + '--mount', "type=bind,source=$releaseMount,target=/release", + $ImageTag, '-c', + 'set -eu; tar --sort=name --mtime=@0 --owner=0 --group=0 --numeric-owner --transform=s,^truf-worker,truf-worker-linux-x86_64, -C /opt -czf /release/truf-worker-native-linux-x86_64.tar.gz truf-worker' +) -Label 'Native Linux worker package export' +if (-not (Test-Path -LiteralPath $linuxNativeArchive -PathType Leaf)) { + throw "Native Linux worker archive is missing: $linuxNativeArchive" +} + +$linuxTar = Join-Path $ReleaseRoot 'truf-worker-linux-x86_64.tar' +$linuxArchive = "$linuxTar.gz" +if ($DockerMode -eq 'wsl') { + $linuxTarDockerPath = Get-WslPath -Path $linuxTar +} +else { + $linuxTarDockerPath = $linuxTar +} +Invoke-DockerChecked -Arguments @( + 'image', 'save', '--output', $linuxTarDockerPath, $ImageTag +) -Label 'Linux worker image export' +if ($DockerMode -eq 'wsl') { + $linuxTarWsl = Get-WslPath -Path $linuxTar + Invoke-Checked -Command 'wsl.exe' -Arguments @( + '-d', $WslDistro, '--', 'gzip', '-1', '-n', '-f', $linuxTarWsl + ) -Label 'Linux worker image compression' +} +else { + Compress-Gzip -InputPath $linuxTar -OutputPath $linuxArchive + Remove-Item -LiteralPath $linuxTar +} +if (-not (Test-Path -LiteralPath $linuxArchive -PathType Leaf)) { + throw "Linux worker archive is missing: $linuxArchive" +} + +$compose = Join-Path $ReleaseRoot 'compose.yaml' +$readme = Join-Path $ReleaseRoot 'README_RU.md' +$workerctlShell = Join-Path $ReleaseRoot 'workerctl.sh' +$workerctlPowerShell = Join-Path $ReleaseRoot 'workerctl.ps1' +Copy-Item -LiteralPath (Join-Path $ProjectRoot 'deploy\worker\compose.yaml') ` + -Destination $compose +$readmeText = [IO.File]::ReadAllText( + (Join-Path $ProjectRoot 'docs\remote-worker-quickstart-ru.md') +).Replace( + '](remote-worker-cheatsheet-', '](cheatsheets/remote-worker-cheatsheet-' +) +[IO.File]::WriteAllText( + $readme, $readmeText, (New-Object Text.UTF8Encoding($false)) +) +Copy-Item -LiteralPath (Join-Path $ProjectRoot 'deploy\worker\workerctl.sh') ` + -Destination $workerctlShell +Copy-Item -LiteralPath (Join-Path $ProjectRoot 'deploy\worker\workerctl.ps1') ` + -Destination $workerctlPowerShell + +$cheatsheetRoot = Join-Path $ReleaseRoot 'cheatsheets' +New-Item -ItemType Directory -Path $cheatsheetRoot | Out-Null +$cheatsheetEntries = @( + foreach ($name in @( + 'remote-worker-cheatsheet-windows-ru.md', + 'remote-worker-cheatsheet-linux-ru.md', + 'remote-worker-cheatsheet-docker-ru.md' + )) { + $destination = Join-Path $cheatsheetRoot $name + Copy-Item -LiteralPath (Join-Path $ProjectRoot "docs\$name") ` + -Destination $destination + [PSCustomObject]@{ + Path = $destination + Name = "cheatsheets/$name" + } + } +) + +$dockerSums = Join-Path $ReleaseRoot 'DOCKER-SHA256SUMS.txt' +$dockerFiles = @( + [PSCustomObject]@{ Path = $linuxArchive; Name = [IO.Path]::GetFileName($linuxArchive) }, + [PSCustomObject]@{ Path = $compose; Name = 'compose.yaml' }, + [PSCustomObject]@{ Path = $readme; Name = 'README_RU.md' }, + [PSCustomObject]@{ Path = $workerctlShell; Name = 'workerctl.sh' }, + [PSCustomObject]@{ Path = $workerctlPowerShell; Name = 'workerctl.ps1' } +) +$dockerFiles += $cheatsheetEntries +Write-Sha256Manifest -Path $dockerSums -Files $dockerFiles +$dockerBundle = Join-Path $ReleaseRoot 'truf-worker-docker-x86_64-bundle.zip' +$dockerBundleFiles = @($dockerFiles) +$dockerBundleFiles += [PSCustomObject]@{ + Path = $dockerSums + Name = 'SHA256SUMS.txt' +} +New-StoredZip -Path $dockerBundle -Files $dockerBundleFiles + +$releaseSums = Join-Path $ReleaseRoot 'SHA256SUMS.txt' +$releaseFiles = @( + $windowsZip, + "$windowsZip.json", + $windowsManifestSnapshot, + $linuxNativeArchive, + $linuxArchive, + $linuxManifestSnapshot, + $compose, + $readme, + $workerctlShell, + $workerctlPowerShell, + $dockerSums, + $dockerBundle +) +$releaseFiles += $cheatsheetEntries +Write-Sha256Manifest -Path $releaseSums -Files $releaseFiles + +Write-Host "Complete release created: $ReleaseRoot" +Write-Host "Checksums: $releaseSums" diff --git a/cleanup_stale_agentui_vite.ps1 b/cleanup_stale_agentui_vite.ps1 new file mode 100644 index 0000000..cbdf15a --- /dev/null +++ b/cleanup_stale_agentui_vite.ps1 @@ -0,0 +1,146 @@ +[CmdletBinding()] +param( + [ValidateRange(15, 10080)] + [int]$MinAgeMinutes = 120, + + [switch]$Apply +) + +$ErrorActionPreference = 'Stop' +throw 'Docker development copy: runtime control is disabled until container isolation is ready. See DOCKER_MIGRATION.md.' +$now = Get-Date +$agentVitePattern = 'AgentUI_2\.01.*frontend-next.*vite\\bin\\vite\.js' +$npxVitePattern = 'npx-cli\.js.*\bvite\b' +$agentEsbuildPattern = 'AgentUI_2\.01.*frontend-next.*@esbuild' + +function Get-VitePort { + param([string]$CommandLine) + + if ($CommandLine -match '--port\s+(\d+)') { + return $matches[1] + } + return $null +} + +$processes = @(Get-CimInstance Win32_Process) +$agentNodes = @( + $processes | Where-Object { + $_.Name -eq 'node.exe' -and [string]$_.CommandLine -match $agentVitePattern + } +) +$ports = @{} +foreach ($process in $agentNodes) { + $port = Get-VitePort ([string]$process.CommandLine) + if ($port) { + $ports[$port] = $true + } +} +$npxNodes = @( + $processes | Where-Object { + if ($_.Name -ne 'node.exe' -or [string]$_.CommandLine -notmatch $npxVitePattern) { + return $false + } + $port = Get-VitePort ([string]$_.CommandLine) + return $port -and $ports.ContainsKey($port) + } +) + +$connectedPids = @{} +try { + Get-NetTCPConnection -State Established -ErrorAction Stop | ForEach-Object { + $connectedPids[[int]$_.OwningProcess] = $true + } +} catch { + throw "Refusing cleanup because established TCP connections could not be inspected: $($_.Exception.Message)" +} + +$protectedPorts = @{} +$newestAgent = $agentNodes | Sort-Object CreationDate -Descending | Select-Object -First 1 +if ($newestAgent) { + $newestPort = Get-VitePort ([string]$newestAgent.CommandLine) + if ($newestPort) { + $protectedPorts[$newestPort] = 'newest AgentUI Vite instance' + } +} +foreach ($process in @($agentNodes + $npxNodes)) { + $port = Get-VitePort ([string]$process.CommandLine) + if (-not $port) { + continue + } + $created = [datetime]$process.CreationDate + $ageMinutes = ($now - $created).TotalMinutes + if ($ageMinutes -lt $MinAgeMinutes) { + $protectedPorts[$port] = "younger than $MinAgeMinutes minutes" + } + if ($connectedPids.ContainsKey([int]$process.ProcessId)) { + $protectedPorts[$port] = 'has an established TCP connection' + } +} + +$targetAgentNodes = @( + $agentNodes | Where-Object { + $port = Get-VitePort ([string]$_.CommandLine) + $port -and -not $protectedPorts.ContainsKey($port) + } +) +$targetPorts = @{} +$targetAgentPids = @{} +foreach ($process in $targetAgentNodes) { + $targetPorts[(Get-VitePort ([string]$process.CommandLine))] = $true + $targetAgentPids[[int]$process.ProcessId] = $true +} +$targetNpxNodes = @( + $npxNodes | Where-Object { + $port = Get-VitePort ([string]$_.CommandLine) + $port -and $targetPorts.ContainsKey($port) + } +) +$targetEsbuild = @( + $processes | Where-Object { + $_.Name -eq 'esbuild.exe' -and + [string]$_.ExecutablePath -match $agentEsbuildPattern -and + $targetAgentPids.ContainsKey([int]$_.ParentProcessId) + } +) +$targetIds = @( + @($targetEsbuild + $targetAgentNodes + $targetNpxNodes).ProcessId | + Where-Object { $null -ne $_ -and [int]$_ -gt 0 } | + ForEach-Object { [int]$_ } | + Sort-Object -Unique +) + +Write-Output "AgentUI Vite instances: $($agentNodes.Count)" +Write-Output "Protected ports: $($protectedPorts.Count)" +Write-Output "Stale ports: $($targetPorts.Count)" +Write-Output "Target processes: $($targetIds.Count)" + +if (-not $Apply) { + Write-Output 'Dry run only. Re-run with -Apply to terminate the exact stale process set.' + exit 0 +} + +if ($targetIds.Count -eq 0) { + Write-Output 'No stale AgentUI Vite processes require cleanup.' + exit 0 +} + +Stop-Process -Id $targetIds -Force -ErrorAction SilentlyContinue +Start-Sleep -Seconds 5 + +$remaining = @( + Get-CimInstance Win32_Process | Where-Object { + $targetIds -contains [int]$_.ProcessId -and ( + ($_.Name -eq 'node.exe' -and ( + [string]$_.CommandLine -match $agentVitePattern -or + [string]$_.CommandLine -match $npxVitePattern + )) -or + ($_.Name -eq 'esbuild.exe' -and [string]$_.ExecutablePath -match $agentEsbuildPattern) + ) + } +) +if ($remaining.Count -gt 0) { + Write-Error "Cleanup left $($remaining.Count) exact target process(es) running." + exit 1 +} + +Write-Output "Terminated $($targetIds.Count) stale AgentUI Vite process(es)." diff --git a/compose.e2e.yaml b/compose.e2e.yaml new file mode 100644 index 0000000..a5a065b --- /dev/null +++ b/compose.e2e.yaml @@ -0,0 +1,202 @@ +# Invoke through python3 docker/verify.py, never with the production Compose file. +# The verifier supplies immutable IDs for these already-built local images. +name: ${TRUF_WORKER_TEST_PROJECT:?Invoke python3 docker/verify.py} + +x-isolated: &isolated + pull_policy: never + read_only: true + user: "10001:10001" + cap_drop: [ALL] + security_opt: [no-new-privileges:true] + network_mode: none + init: false + # Override Docker client proxy defaults without importing host credentials. + environment: + HTTP_PROXY: "" + http_proxy: "" + HTTPS_PROXY: "" + https_proxy: "" + FTP_PROXY: "" + ftp_proxy: "" + ALL_PROXY: "" + all_proxy: "" + NO_PROXY: "*" + no_proxy: "*" + tmpfs: + - /run/truf:rw,nosuid,nodev,noexec,size=64m,mode=0700,uid=10001,gid=10001 + - /tmp:rw,nosuid,nodev,noexec,size=128m,mode=1777 + cpus: 2.0 + mem_limit: 6g + pids_limit: 512 + shm_size: 256m + stop_signal: SIGTERM + stop_grace_period: 600s + restart: "no" + logging: + driver: json-file + options: + max-size: "16m" + max-file: "4" + +x-runtime: &runtime + <<: *isolated + image: ${TRUF_WORKER_TEST_RUNTIME_IMAGE:?Invoke python3 docker/verify.py} + volumes: + - type: volume + source: data + target: /data + +services: + tools: + <<: *isolated + image: ${TRUF_WORKER_TEST_TEST_IMAGE:?Invoke python3 docker/verify.py} + volumes: + # Only the test tree is copied up, never the test image's application. + - type: volume + source: tools + target: /opt/truf/tests + volume: + nocopy: false + entrypoint: [/usr/local/bin/python3, -I, -S, -B, -c] + command: + - | + import hashlib + import json + import os + from pathlib import Path + import stat + + try: + assert os.getuid() == os.geteuid() == os.getgid() == 10001 + root = Path('/opt/truf/tests') + fixture_files = { + root / 'fixtures/worker_tls_cert.pem', + root / 'fixtures/worker_tls_key.pem', + } + pending = [root] + files = [] + size = 0 + entries = 0 + while pending: + path = pending.pop() + entries += 1 + assert entries <= 512 + details = path.lstat() + assert (details.st_uid, details.st_gid) == (10001, 10001) + if stat.S_ISDIR(details.st_mode): + assert stat.S_IMODE(details.st_mode) == 0o700 + pending.extend(path.iterdir()) + assert len(pending) + len(files) <= 512 + else: + assert stat.S_ISREG(details.st_mode) + assert stat.S_IMODE(details.st_mode) == 0o600 + assert (path.suffix == '.py' or path in fixture_files) + assert details.st_size <= 4 * 1024 * 1024 + files.append(path) + size += details.st_size + assert size <= 64 * 1024 * 1024 + assert root / 'container_e2e.py' in files + assert fixture_files <= set(files) + tree = hashlib.sha256() + for path in sorted(files): + tree.update(path.relative_to(root).as_posix().encode('utf-8') + b'\0') + tree.update(hashlib.sha256(path.read_bytes()).digest()) + summary = { + 'counts': {'ok': 1, 'tool_files': len(files), 'tool_bytes': size}, + 'hashes': { + 'tools_tree_sha256': tree.hexdigest(), + 'driver_sha256': hashlib.sha256((root / 'container_e2e.py').read_bytes()).hexdigest(), + }, + } + except Exception: + summary = {'counts': {'ok': 0, 'tools_seed_failed': 1}, 'hashes': {}} + print(json.dumps(summary, sort_keys=True)) + raise SystemExit(0 if summary['counts']['ok'] else 1) + + provision: + <<: *runtime + user: "0:0" + cap_add: [CHOWN, DAC_OVERRIDE, FOWNER] + command: [provision] + + prepare: + <<: *runtime + volumes: + - type: volume + source: data + target: /data + - type: volume + source: tools + target: /opt/truf/tests + read_only: true + volume: + nocopy: true + entrypoint: [/usr/local/bin/python3, -I, -S, -B, /opt/truf/tests/container_e2e.py] + command: [prepare, --config, /data/config/e2e.yaml] + + runtime: + <<: *runtime + volumes: + - type: volume + source: data + target: /data + - type: volume + source: tools + target: /opt/truf/tests + read_only: true + volume: + nocopy: true + # Preserve the production image's tini -> container_runtime entrypoint. + # Do not add Compose init, a shell supervisor, or a test-image server. + command: [run, --config, /data/config/e2e.yaml] + healthcheck: + test: [CMD, /usr/local/bin/python3, -I, -S, -B, /opt/truf/app/container_runtime.py, health, --config, /data/config/e2e.yaml] + interval: 5s + timeout: 15s + start_period: 240s + retries: 3 + + stopped: + <<: *runtime + volumes: + - type: volume + source: data + target: /data + read_only: true + volume: + nocopy: true + entrypoint: [/usr/local/bin/python3, -I, -S, -B, -c] + command: + - | + import json + import os + import runpy + + try: + runtime = runpy.run_path('/opt/truf/app/container_runtime.py') + runtime['require_container']() + runtime['private_path']('/data/postgres-linux', directory=True) + marker = runtime['_read_json'](runtime['INITIALIZED']) + assert marker.get('pg_major') == 16 + try: + os.lstat('/data/postgres-linux/postmaster.pid') + except FileNotFoundError: + pass + else: + raise AssertionError + summary = {'counts': { + 'ok': 1, 'private_postgres_directory': 1, + 'postgres_pid_absent': 1, 'initialized': 1, + }, 'hashes': {}} + except Exception: + summary = {'counts': {'ok': 0, 'stopped_data_check_failed': 1}, 'hashes': {}} + print(json.dumps(summary, sort_keys=True)) + raise SystemExit(0 if summary['counts']['ok'] else 1) + +volumes: + data: + name: ${TRUF_WORKER_TEST_PROJECT:?Invoke python3 docker/verify.py}_data + driver: local + tools: + name: ${TRUF_WORKER_TEST_PROJECT:?Invoke python3 docker/verify.py}_tools + driver: local diff --git a/compose.edge.yaml b/compose.edge.yaml new file mode 100644 index 0000000..abaf1dc --- /dev/null +++ b/compose.edge.yaml @@ -0,0 +1,56 @@ +services: + runtime: + ports: + - target: 443 + published: "443" + protocol: tcp + mode: host + + edge: + image: truf-local:edge + build: + context: . + dockerfile: deploy/edge/Dockerfile + target: edge + depends_on: + runtime: + condition: service_healthy + network_mode: service:runtime + read_only: true + user: "10001:10001" + cap_drop: [ALL] + cap_add: [NET_BIND_SERVICE] + security_opt: [no-new-privileges:true] + environment: + TRUF_EDGE_HOST: ${TRUF_EDGE_HOST:?set TRUF_EDGE_HOST in the protected edge env file} + TRUF_ADMIN_PREFIX: ${TRUF_ADMIN_PREFIX:?set TRUF_ADMIN_PREFIX in the protected edge env file} + TRUF_ADMIN_USER: ${TRUF_ADMIN_USER:?set TRUF_ADMIN_USER in the protected edge env file} + TRUF_ADMIN_PASSWORD_HASH: ${TRUF_ADMIN_PASSWORD_HASH:?set TRUF_ADMIN_PASSWORD_HASH in the protected edge env file} + TRUF_ADMIN_EDGE_MARKER: ${TRUF_ADMIN_EDGE_MARKER:?set TRUF_ADMIN_EDGE_MARKER in the protected edge env file} + volumes: + - edge_data:/data + - edge_config:/config + - type: bind + source: ${TRUF_EDGE_AUTH_LOG_DIR:?set TRUF_EDGE_AUTH_LOG_DIR in the protected edge env file} + target: /var/log/caddy + - type: bind + source: ${TRUF_EDGE_DENYLIST_DIR:?set TRUF_EDGE_DENYLIST_DIR in the protected edge env file} + target: /etc/caddy/denylist + read_only: true + tmpfs: + - /tmp:rw,nosuid,nodev,noexec,size=16m,mode=1777 + - /run:rw,nosuid,nodev,noexec,size=4m,mode=0700,uid=10001,gid=10001 + pids_limit: 128 + mem_limit: 256m + cpus: 1.0 + restart: unless-stopped + stop_grace_period: 30s + logging: + driver: json-file + options: + max-size: "8m" + max-file: "4" + +volumes: + edge_data: + edge_config: diff --git a/compose.shared-host.yaml b/compose.shared-host.yaml new file mode 100644 index 0000000..89a0ad3 --- /dev/null +++ b/compose.shared-host.yaml @@ -0,0 +1,62 @@ +services: + provision: + cpus: 0.9 + mem_limit: 720m + + runtime: + network_mode: host + cpus: 0.9 + mem_limit: 720m + + edge: + image: truf-local:edge + build: + context: . + dockerfile: deploy/edge/Dockerfile + target: edge + depends_on: + runtime: + condition: service_healthy + network_mode: service:runtime + read_only: true + user: "10001:10001" + cap_drop: [ALL] + security_opt: [no-new-privileges:true] + environment: + TRUF_EDGE_MODE: shared-host-edge-v1 + TRUF_EDGE_HOST: ${TRUF_EDGE_HOST:?set TRUF_EDGE_HOST in the protected edge env file} + TRUF_ADMIN_PREFIX: ${TRUF_ADMIN_PREFIX:?set TRUF_ADMIN_PREFIX in the protected edge env file} + TRUF_ADMIN_USER: ${TRUF_ADMIN_USER:?set TRUF_ADMIN_USER in the protected edge env file} + TRUF_ADMIN_PASSWORD_HASH: ${TRUF_ADMIN_PASSWORD_HASH:?set TRUF_ADMIN_PASSWORD_HASH in the protected edge env file} + TRUF_ADMIN_EDGE_MARKER: ${TRUF_ADMIN_EDGE_MARKER:?set TRUF_ADMIN_EDGE_MARKER in the protected edge env file} + TRUF_SHARED_INGRESS_MARKER: ${TRUF_SHARED_INGRESS_MARKER:?set TRUF_SHARED_INGRESS_MARKER in the protected edge env file} + volumes: + - edge_data:/data + - edge_config:/config + - type: bind + source: ${TRUF_EDGE_AUTH_LOG_DIR:?set TRUF_EDGE_AUTH_LOG_DIR in the protected edge env file} + target: /var/log/caddy + - type: bind + source: ${TRUF_EDGE_DENYLIST_DIR:?set TRUF_EDGE_DENYLIST_DIR in the protected edge env file} + target: /etc/caddy/denylist + read_only: true + tmpfs: + - /tmp:rw,nosuid,nodev,noexec,size=16m,mode=1777 + - /run:rw,nosuid,nodev,noexec,size=4m,mode=0700,uid=10001,gid=10001 + pids_limit: 128 + mem_limit: 256m + cpus: 1.0 + restart: unless-stopped + stop_grace_period: 30s + logging: + driver: json-file + options: + max-size: "8m" + max-file: "4" + +volumes: + data: + external: true + name: truf-remote-server-data + edge_data: + edge_config: diff --git a/compose.snapshot-import.yaml b/compose.snapshot-import.yaml new file mode 100644 index 0000000..feab7e7 --- /dev/null +++ b/compose.snapshot-import.yaml @@ -0,0 +1,22 @@ +# Offline import only. Merge after compose.yaml, not compose.windows-import.yaml. +services: + runtime: + network_mode: none + restart: "no" + healthcheck: + disable: true + environment: + TRUF_CONTAINER_CONFIG: /opt/truf/app/config.linux.yaml + volumes: + - type: bind + source: "${TRUF_WINDOWS_SNAPSHOT:?Set TRUF_WINDOWS_SNAPSHOT to the completed Windows snapshot directory}" + target: /import + read_only: true + bind: + create_host_path: false + command: + - import-snapshot + - --manifest-sha256 + - "${TRUF_WINDOWS_SNAPSHOT_SHA256:?Set TRUF_WINDOWS_SNAPSHOT_SHA256 to the pinned manifest SHA-256}" + - --config + - /opt/truf/app/config.linux.yaml diff --git a/compose.windows-import.yaml b/compose.windows-import.yaml new file mode 100644 index 0000000..73a6a87 --- /dev/null +++ b/compose.windows-import.yaml @@ -0,0 +1,7 @@ +# Use with compose.yaml only after a verified, stopped Windows snapshot import. +# This selects the translated private profile for run, health, and status alike. +# Import itself must explicitly use --config /opt/truf/app/config.linux.yaml. +services: + runtime: + environment: + TRUF_CONTAINER_CONFIG: /data/config/windows-import.yaml diff --git a/compose.yaml b/compose.yaml new file mode 100644 index 0000000..9803aba --- /dev/null +++ b/compose.yaml @@ -0,0 +1,109 @@ +name: truf-docker + +x-runtime: &runtime + image: truf-local:runtime + build: + context: . + target: runtime + read_only: true + user: "10001:10001" + cap_drop: [ALL] + security_opt: [no-new-privileges:true] + volumes: + - data:/data + tmpfs: + - /run/truf:rw,nosuid,nodev,noexec,size=64m,mode=0700,uid=10001,gid=10001 + - /tmp:rw,nosuid,nodev,noexec,size=128m,mode=1777 + cpus: 2.0 + mem_limit: 6g + pids_limit: 512 + shm_size: 256m + stop_signal: SIGTERM + stop_grace_period: 10m + logging: + driver: json-file + options: + max-size: "16m" + max-file: "4" + +services: + provision: + <<: *runtime + user: "0:0" + cap_add: [CHOWN, DAC_OVERRIDE, FOWNER] + network_mode: none + command: [provision] + restart: "no" + + runtime: + <<: *runtime + volumes: + - type: volume + source: data + target: /data + - type: bind + source: /etc/truf/runtime + target: /data/config + read_only: true + bind: + create_host_path: false + - type: bind + source: /etc/truf/worker-packages + target: /data/worker-packages + read_only: true + bind: + create_host_path: false + - type: bind + source: /var/lib/truf/runtime-document-candidates + target: /data/runtime-document-candidates + bind: + create_host_path: false + - type: bind + source: /run/truf/host-agent.sock + target: /run/truf/host-agent.sock + read_only: true + bind: + create_host_path: false + - type: bind + source: /var/lib/truf/host-agent/results + target: /data/host-agent-results + read_only: true + bind: + create_host_path: false + - type: bind + source: /run/truf-postgres + target: /run/truf-postgres + bind: + create_host_path: false + depends_on: + provision: + condition: service_completed_successfully + command: [run, --config, /data/config/config.yaml] + restart: "on-failure:3" + healthcheck: + test: [CMD, /usr/local/bin/python3, -I, -S, -B, /opt/truf/app/container_runtime.py, health, --config, /data/config/config.yaml] + interval: 30s + timeout: 15s + start_period: 240s + retries: 3 + + test: + image: truf-local:test + build: + context: . + target: test + profiles: [test] + read_only: true + user: "10001:10001" + cap_drop: [ALL] + security_opt: [no-new-privileges:true] + network_mode: none + tmpfs: + # These tests intentionally execute temporary venv and askpass fixtures. + - /tmp:rw,exec,nosuid,nodev,size=512m,mode=1777 + cpus: 2.0 + mem_limit: 3g + pids_limit: 512 + +volumes: + data: diff --git a/deploy/capacity50/Dockerfile b/deploy/capacity50/Dockerfile new file mode 100644 index 0000000..d0ebbf0 --- /dev/null +++ b/deploy/capacity50/Dockerfile @@ -0,0 +1,41 @@ +ARG BASE_IMAGE=truf-local:runtime +FROM ${BASE_IMAGE} + +COPY --chown=10001:10001 --chmod=0600 payload/app/capacity_model.py /opt/truf/app/capacity_model.py +COPY --chown=10001:10001 --chmod=0600 payload/app/scanner_db.py /opt/truf/app/scanner_db.py +COPY --chown=10001:10001 --chmod=0600 payload/app/worker_assignment.py /opt/truf/app/worker_assignment.py +COPY --chown=10001:10001 --chmod=0600 payload/app/worker_api.py /opt/truf/app/worker_api.py +COPY --chown=10001:10001 --chmod=0600 payload/app/jsonl_projector.py /opt/truf/app/jsonl_projector.py +COPY --chown=10001:10001 --chmod=0600 payload/app/runtime_document.py /opt/truf/app/runtime_document.py +COPY --chown=10001:10001 --chmod=0600 payload/app/lifecycle_authority.py /opt/truf/app/lifecycle_authority.py +COPY --chown=10001:10001 --chmod=0600 payload/app/config.linux.yaml /opt/truf/app/config.linux.yaml + +RUN python3 -I -S -B - <<'PY' +import ast +from pathlib import Path +import sys + +root = Path('/opt/truf/app') +names = ( + 'capacity_model.py', + 'scanner_db.py', + 'worker_assignment.py', + 'worker_api.py', + 'jsonl_projector.py', + 'runtime_document.py', + 'lifecycle_authority.py', +) +for name in names: + ast.parse((root / name).read_text(encoding='utf-8'), filename=name) +sys.path.insert(0, str(root)) +from capacity_model import ( # noqa: E402 + MAX_RESULT_BUNDLE_BYTES, + REMOTE_ASSIGNMENT_BASELINE_BYTES, + REMOTE_ASSIGNMENT_MAX_ACTIVE, +) +assert (MAX_RESULT_BUNDLE_BYTES, REMOTE_ASSIGNMENT_BASELINE_BYTES, REMOTE_ASSIGNMENT_MAX_ACTIVE) == ( + 64 * 1024 * 1024, + 2 * 1024 * 1024, + 50, +) +PY diff --git a/deploy/capacity50/deploy.sh b/deploy/capacity50/deploy.sh new file mode 100644 index 0000000..fe37ae6 --- /dev/null +++ b/deploy/capacity50/deploy.sh @@ -0,0 +1,481 @@ +#!/bin/bash +set -Eeuo pipefail +umask 077 + +MODE="${1:-}" +STAGE="${2:-}" +if [[ "$MODE" != plan && "$MODE" != apply ]]; then + echo 'usage: deploy.sh plan|apply STAGE' >&2 + exit 64 +fi +if [[ ! "$STAGE" =~ ^/var/lib/truf-deploy/stage/capacity50\.[A-Za-z0-9]+$ ]] || [[ ! -d "$STAGE" ]]; then + echo 'invalid deployment stage' >&2 + exit 64 +fi + +readonly RELEASE_ID='capacity50-20260930' +readonly EXPECTED_IMAGE='sha256:46f1cf1b92d1a7d93d06f690309d8c7eca45f64dc45ca310861c70fb419bb035' +readonly EXPECTED_CONFIG_SHA256='12bd9a60cc56c3d6cbad18435523e8229b0cd8fdccc2d73922ea7e580ce7441c' +readonly CANDIDATE_TAG="truf-local:runtime-${RELEASE_ID}" +readonly ROLLBACK_TAG="truf-local:runtime-pre-${RELEASE_ID}" +readonly ACTIVE_CONFIG='/etc/truf/runtime/config.yaml' +readonly ACTIVE_SECRETS='/etc/truf/runtime/secrets.yaml' +readonly SOURCE_ROOT='/opt/truf' +readonly HISTORY="/var/lib/truf-deploy/history/${RELEASE_ID}" +readonly TEST_USER='operator-trace-windows-20260925' +readonly TEST_USER_ORIGINAL_CAP='2' +readonly APP_FILES=( + capacity_model.py + scanner_db.py + worker_assignment.py + worker_api.py + jsonl_projector.py + runtime_document.py + lifecycle_authority.py + config.linux.yaml +) +readonly COMPOSE=( + docker compose + --project-name truf-docker + --project-directory /opt/truf + --env-file /etc/truf-edge/edge.env + --file /opt/truf/compose.yaml + --file /opt/truf/compose.shared-host.yaml +) + +PHASE='preflight' +MUTATED=0 +PHASE_A_HEALTHY=0 +SOURCE_INSTALLED=0 +USER_CAP_CHANGED=0 +DEPLOY_SUCCEEDED=0 +RESUME=0 + +log() { + printf '[%s] %s\n' "$RELEASE_ID" "$*" +} + +runtime_id() { + "${COMPOSE[@]}" ps --quiet runtime +} + +psql() { + local container + container="$(runtime_id)" + [[ -n "$container" ]] || return 1 + docker exec "$container" /usr/lib/postgresql/16/bin/psql \ + -h /run/truf-postgres -U truf -d truf -v ON_ERROR_STOP=1 -At "$@" +} + +current_image() { + docker image inspect --format '{{.Id}}' truf-local:runtime +} + +config_sha256() { + sha256sum "$ACTIVE_CONFIG" | cut -d' ' -f1 +} + +require_baseline() { + [[ "$(current_image)" == "$EXPECTED_IMAGE" ]] || { + echo 'runtime image identity changed' >&2 + return 1 + } + [[ "$(config_sha256)" == "$EXPECTED_CONFIG_SHA256" ]] || { + echo 'active config identity changed' >&2 + return 1 + } + local state + state="$(psql -F '|' -c \ + "SELECT revision, discovery_paused::int, dispatch_paused::int, drain_state FROM runtime_operations_control WHERE id=1;")" + if ((RESUME)); then + [[ "$state" =~ ^[0-9]+\|0\|0\|(normal|drained)$ ]] || { + echo "resumed runtime control is neither open nor drained: $state" >&2 + return 1 + } + elif [[ ! "$state" =~ ^[0-9]+\|0\|0\|normal$ ]]; then + echo "runtime control is not open: $state" >&2 + return 1 + fi + local debt + debt="$(psql -F '|' -c \ + "SELECT (SELECT count(*) FROM result_reservations WHERE assignment_kind='remote' AND remote_resolved_at IS NULL), bundle_items, bundle_bytes, projection_items, projection_bytes, keycheck_items, keycheck_bytes, quarantine_items, quarantine_bytes FROM pipeline_capacity WHERE id=1;")" + [[ "$debt" == '0|0|0|0|0|0|0|0|0' ]] || { + echo "pipeline is not reconciled: $debt" >&2 + return 1 + } + [[ "$(psql -c "SELECT active_assignment_cap FROM remote_worker_users WHERE user_key='${TEST_USER}' AND disabled_at IS NULL;")" == "$TEST_USER_ORIGINAL_CAP" ]] || { + echo 'temporary validation user identity changed' >&2 + return 1 + } +} + +edge_value() { + local name="$1" + sed -n "s/^${name}=//p" /etc/truf-edge/edge.env +} + +admin_material() { + local marker host page token revision + marker="$(edge_value TRUF_ADMIN_EDGE_MARKER)" + host="$(edge_value TRUF_EDGE_HOST)" + [[ "$marker" =~ ^[a-f0-9]{64}$ ]] || return 1 + [[ "$host" =~ ^[A-Za-z0-9.-]+$ ]] || return 1 + page="$(curl --fail --silent --show-error --max-time 20 \ + --header "X-Truf-Admin-Edge: ${marker}" \ + --header 'X-Truf-Admin-Operator: deploy-runtime' \ + http://127.0.0.1:8766/admin-internal/)" + token="$(python3 -c \ + 'import re,sys; values=set(re.findall(r"name=\"csrf_token\" value=\"([^\"]+)\"",sys.stdin.read())); print(values.pop() if len(values)==1 else "")' \ + <<<"$page")" + revision="$(python3 -c \ + 'import re,sys; values=set(re.findall(r"name=\"expected_revision\" value=\"([0-9]+)\"",sys.stdin.read())); print(values.pop() if len(values)==1 else "")' \ + <<<"$page")" + [[ "$token" =~ ^[A-Za-z0-9_-]{32,128}$ && "$revision" =~ ^[0-9]+$ ]] || return 1 + printf '%s|%s|%s|%s\n' "$marker" "$host" "$token" "$revision" +} + +admin_post() { + local route="$1" + shift + local material marker host token revision operation + material="$(admin_material)" + IFS='|' read -r marker host token revision <<<"$material" + operation="$(cat /proc/sys/kernel/random/uuid)" + local arguments=( + --fail --silent --show-error --max-time 30 + --request POST + --header "X-Truf-Admin-Edge: ${marker}" + --header 'X-Truf-Admin-Operator: deploy-runtime' + --header "Origin: https://${host}" + --header 'Content-Type: application/x-www-form-urlencoded' + --data-urlencode "csrf_token=${token}" + --data-urlencode "operation_id=${operation}" + ) + if [[ "$route" == dispatch/* || "$route" == search/discovery/* ]]; then + arguments+=(--data-urlencode "expected_revision=${revision}") + fi + while (($#)); do + arguments+=(--data-urlencode "$1") + shift + done + curl "${arguments[@]}" "http://127.0.0.1:8766/admin-internal/${route}" >/dev/null +} + +wait_for_drain() { + local deadline=$((SECONDS + 600)) state debt + while ((SECONDS < deadline)); do + state="$(psql -c "SELECT drain_state FROM runtime_operations_control WHERE id=1;")" + debt="$(psql -F '|' -c \ + "SELECT (SELECT count(*) FROM result_reservations WHERE assignment_kind='remote' AND remote_resolved_at IS NULL), bundle_items, bundle_bytes, projection_items, projection_bytes, keycheck_items, keycheck_bytes, quarantine_items, quarantine_bytes FROM pipeline_capacity WHERE id=1;")" + if [[ "$state" == drained && "$debt" == '0|0|0|0|0|0|0|0|0' ]]; then + return 0 + fi + sleep 2 + done + echo 'runtime did not drain within 600 seconds' >&2 + return 1 +} + +quiesce_pipeline_workers() { + local source deadline active container + for source in result-ingester jsonl-projector; do + admin_post supervisor/sources/stop "source_id=${source}" + done + container="$(runtime_id)" + [[ -n "$container" ]] || return 1 + docker exec --interactive "$container" /usr/local/bin/python3 -I -S -B - \ + <"$STAGE/release_stopped_pipeline_leases.py" + deadline=$((SECONDS + 120)) + while ((SECONDS < deadline)); do + active="$(psql -c \ + "SELECT count(*) FROM pipeline_leases WHERE state NOT IN ('released','failed');")" + if [[ "$active" == 0 ]]; then + return 0 + fi + sleep 2 + done + echo 'pipeline worker leases did not release within 120 seconds' >&2 + return 1 +} + +install_config() { + local source="$1" temporary + temporary="/etc/truf/runtime/.config.yaml.${RELEASE_ID}.tmp" + install -o root -g root -m 0600 "$source" "$temporary" + chown 10001:10001 "$temporary" + mv -f "$temporary" "$ACTIVE_CONFIG" +} + +stop_stack() { + "${COMPOSE[@]}" stop --timeout 30 edge + "${COMPOSE[@]}" stop --timeout 600 runtime + local container state + container="$("${COMPOSE[@]}" ps --all --quiet runtime)" + state="$(docker inspect --format '{{.State.Status}}|{{.State.ExitCode}}|{{.State.OOMKilled}}' "$container")" + [[ "$state" == 'exited|0|false' ]] || { + echo "runtime stop was not clean: $state" >&2 + return 1 + } +} + +wait_runtime_health() { + local deadline=$((SECONDS + 420)) container state status + while ((SECONDS < deadline)); do + container="$("${COMPOSE[@]}" ps --all --quiet runtime)" + if [[ -n "$container" ]]; then + state="$(docker inspect --format '{{.State.Status}}' "$container")" + [[ "$state" != exited && "$state" != dead ]] || return 1 + status="$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{else}}none{{end}}' "$container")" + if [[ "$status" == healthy ]] && docker exec "$container" \ + /usr/local/bin/python3 -I -S -B /opt/truf/app/container_runtime.py \ + health --config /data/config/config.yaml --require-worker-api >/dev/null; then + return 0 + fi + [[ "$status" != unhealthy ]] || return 1 + fi + sleep 3 + done + echo 'runtime health timed out' >&2 + return 1 +} + +start_stack() { + "${COMPOSE[@]}" up --detach --no-deps --no-build --pull never --force-recreate runtime + wait_runtime_health + "${COMPOSE[@]}" up --detach --no-deps --no-build --pull never --force-recreate edge + sleep 3 + local edge_container edge_state host public_code + edge_container="$("${COMPOSE[@]}" ps --quiet edge)" + edge_state="$(docker inspect --format '{{.State.Status}}|{{.State.Running}}|{{.State.OOMKilled}}' "$edge_container")" + [[ "$edge_state" == 'running|true|false' ]] || return 1 + host="$(edge_value TRUF_EDGE_HOST)" + public_code="$(curl --silent --show-error --output /dev/null --write-out '%{http_code}' \ + --max-time 20 "https://${host}/")" + [[ "$public_code" == 401 || "$public_code" == 404 ]] || { + echo "unexpected public edge response: $public_code" >&2 + return 1 + } +} + +validate_candidate_config() { + local path="$1" + docker run --rm --network none --read-only --user 10001:10001 \ + --cap-drop ALL --security-opt no-new-privileges:true \ + --tmpfs /tmp:rw,nosuid,nodev,noexec,size=16m,mode=1777 \ + --volume "$path:/data/config/config.yaml:ro" \ + --volume "$ACTIVE_SECRETS:/data/config/secrets.yaml:ro" \ + --volume /etc/truf/worker-packages:/data/worker-packages:ro \ + --entrypoint /usr/local/bin/python3 "$CANDIDATE_TAG" -I -S -B -c \ + "import sys,sysconfig;sys.path.append(sysconfig.get_paths()['purelib']);sys.path.insert(0,'/opt/truf/app');from runtime_document_io import validate_managed_runtime_files;print(validate_managed_runtime_files('/data/config/config.yaml').config_sha256)" \ + >/dev/null +} + +install_sources() { + local name destination temporary + for name in "${APP_FILES[@]}"; do + destination="${SOURCE_ROOT}/app/${name}" + temporary="${destination}.${RELEASE_ID}.tmp" + install -o root -g root -m 0644 "$STAGE/payload/app/$name" "$temporary" + mv -f "$temporary" "$destination" + done + SOURCE_INSTALLED=1 +} + +restore_sources() { + local name + for name in "${APP_FILES[@]}"; do + if [[ -f "$HISTORY/source/$name" ]]; then + install -o root -g root -m 0644 "$HISTORY/source/$name" "${SOURCE_ROOT}/app/$name" + else + rm -f "${SOURCE_ROOT}/app/$name" + fi + done +} + +restore_user_cap() { + if ((USER_CAP_CHANGED)); then + admin_post users/cap "user_key=${TEST_USER}" "active_assignment_cap=${TEST_USER_ORIGINAL_CAP}" || true + USER_CAP_CHANGED=0 + fi +} + +cancel_drain() { + local state + state="$(psql -c "SELECT drain_state FROM runtime_operations_control WHERE id=1;" 2>/dev/null || true)" + if [[ "$state" == draining || "$state" == drained ]]; then + admin_post dispatch/drain/cancel || return 1 + fi +} + +rollback() { + set +e + log "rollback from phase ${PHASE}" + restore_user_cap + stop_stack + if ((PHASE_A_HEALTHY)); then + docker image tag "$CANDIDATE_TAG" truf-local:runtime + install_config "$HISTORY/config.conservative.yaml" + else + docker image tag "$EXPECTED_IMAGE" truf-local:runtime + install_config "$HISTORY/config.original.yaml" + fi + ((SOURCE_INSTALLED)) && restore_sources + if start_stack; then + cancel_drain + log 'rollback restored a healthy runtime' + else + log 'rollback could not prove health; runtime remains contained' >&2 + fi + set -e +} + +on_exit() { + local code=$? + trap - EXIT ERR INT TERM + if ((code != 0 && MUTATED && !DEPLOY_SUCCEEDED)); then + rollback + fi + exit "$code" +} +trap on_exit EXIT + +exec 9>/run/lock/truf-runtime-deploy.lock +flock -n 9 || { + echo 'another runtime deployment is active' >&2 + exit 1 +} + +if [[ "$MODE" == apply && -d "$HISTORY" ]] \ + && docker image inspect "$CANDIDATE_TAG" >/dev/null 2>&1 \ + && [[ "$(docker image inspect --format '{{.Id}}' "$ROLLBACK_TAG" 2>/dev/null || true)" == "$EXPECTED_IMAGE" ]]; then + RESUME=1 +fi +require_baseline +available_kb="$(df -Pk /var/lib/docker | awk 'NR==2 {print $4}')" +[[ "$available_kb" =~ ^[0-9]+$ && "$available_kb" -ge 786432 ]] || { + echo 'less than 768 MiB is available for the derived image' >&2 + exit 1 +} +log "plan image=${EXPECTED_IMAGE#sha256:} config=${EXPECTED_CONFIG_SHA256} free_kib=${available_kb}" +if [[ "$MODE" == plan ]]; then + log 'plan passed; no runtime state changed' + exit 0 +fi + +if ((RESUME)); then + log 'resuming a verified pre-cutover release' + [[ "$(sha256sum "$HISTORY/config.original.yaml" | cut -d' ' -f1)" == "$EXPECTED_CONFIG_SHA256" ]] || exit 1 + [[ -f "$HISTORY/config.conservative.yaml" && -f "$HISTORY/config.final.yaml" ]] || exit 1 + validate_candidate_config "$HISTORY/config.conservative.yaml" + validate_candidate_config "$HISTORY/config.final.yaml" + if [[ "$(psql -c "SELECT drain_state FROM runtime_operations_control WHERE id=1;")" == normal ]]; then + admin_post dispatch/drain/start + MUTATED=1 + wait_for_drain + else + MUTATED=1 + fi +else + install -d -o root -g root -m 0700 /var/lib/truf-deploy /var/lib/truf-deploy/history + if [[ -e "$HISTORY" ]]; then + echo 'release history already exists' >&2 + exit 1 + fi + install -d -o root -g root -m 0700 "$HISTORY" "$HISTORY/source" + install -o root -g root -m 0600 "$ACTIVE_CONFIG" "$HISTORY/config.original.yaml" + for name in "${APP_FILES[@]}"; do + if [[ -f "${SOURCE_ROOT}/app/$name" ]]; then + install -o root -g root -m 0600 "${SOURCE_ROOT}/app/$name" "$HISTORY/source/$name" + fi + done + + python3 "$STAGE/render_config.py" --input "$ACTIVE_CONFIG" \ + --output "$HISTORY/config.conservative.yaml" --mode conservative \ + --expected-sha256 "$EXPECTED_CONFIG_SHA256" >/dev/null + python3 "$STAGE/render_config.py" --input "$ACTIVE_CONFIG" \ + --output "$HISTORY/config.final.yaml" --mode final \ + --expected-sha256 "$EXPECTED_CONFIG_SHA256" >/dev/null + chown 10001:10001 "$HISTORY/config.conservative.yaml" "$HISTORY/config.final.yaml" + chmod 0600 "$HISTORY/config.conservative.yaml" "$HISTORY/config.final.yaml" + + if docker image inspect "$CANDIDATE_TAG" >/dev/null 2>&1; then + echo 'candidate image tag already exists' >&2 + exit 1 + fi + PHASE='candidate-build' + docker build --network none --build-arg BASE_IMAGE=truf-local:runtime \ + --file "$STAGE/Dockerfile" --tag "$CANDIDATE_TAG" "$STAGE" + [[ "$(current_image)" == "$EXPECTED_IMAGE" ]] || { + echo 'runtime tag changed during candidate build' >&2 + exit 1 + } + validate_candidate_config "$HISTORY/config.conservative.yaml" + validate_candidate_config "$HISTORY/config.final.yaml" + docker image tag "$EXPECTED_IMAGE" "$ROLLBACK_TAG" + + PHASE='drain' + admin_post dispatch/drain/start + MUTATED=1 + wait_for_drain +fi + +PHASE='conservative-cutover' +quiesce_pipeline_workers +stop_stack +install_config "$HISTORY/config.conservative.yaml" +docker image tag "$CANDIDATE_TAG" truf-local:runtime +start_stack +schema_state="$(psql -F '|' -c \ + "SELECT (SELECT count(*) FROM information_schema.columns WHERE table_name='result_reservations' AND column_name='reserved_bundle_bytes'), (SELECT count(*) FROM runtime_schema_migrations WHERE version='20260930_33_remote_assignment_capacity');")" +[[ "$schema_state" == '1|1' ]] || { + echo "capacity migration was not applied: $schema_state" >&2 + exit 1 +} +PHASE_A_HEALTHY=1 + +PHASE='capacity50-cutover' +quiesce_pipeline_workers +stop_stack +install_config "$HISTORY/config.final.yaml" +start_stack + +final_values="$(psql -F '|' -c \ + "SELECT bundle_items, bundle_bytes, projection_items, projection_bytes, keycheck_items, keycheck_bytes, quarantine_items, quarantine_bytes FROM pipeline_capacity WHERE id=1;")" +[[ "$final_values" == '0|0|0|0|0|0|0|0' ]] || { + echo "post-deploy capacity is not reconciled: $final_values" >&2 + exit 1 +} + +PHASE='temporary-user-cap-validation' +admin_post users/cap "user_key=${TEST_USER}" 'active_assignment_cap=50' +USER_CAP_CHANGED=1 +[[ "$(psql -c "SELECT active_assignment_cap FROM remote_worker_users WHERE user_key='${TEST_USER}';")" == 50 ]] || exit 1 +restore_user_cap +[[ "$(psql -c "SELECT active_assignment_cap FROM remote_worker_users WHERE user_key='${TEST_USER}';")" == "$TEST_USER_ORIGINAL_CAP" ]] || exit 1 + +PHASE='source-install' +install_sources +for name in "${APP_FILES[@]}"; do + cmp -s "$STAGE/payload/app/$name" "${SOURCE_ROOT}/app/$name" || exit 1 +done + +PHASE='resume' +cancel_drain +post_control="$(psql -F '|' -c \ + "SELECT discovery_paused::int, dispatch_paused::int, drain_state FROM runtime_operations_control WHERE id=1;")" +[[ "$post_control" == '0|0|normal' ]] || { + echo "runtime control did not resume: $post_control" >&2 + exit 1 +} + +cat >"$HISTORY/result.txt" </dev/null +USER 10001:10001 diff --git a/deploy/edge/README.md b/deploy/edge/README.md new file mode 100644 index 0000000..94df380 --- /dev/null +++ b/deploy/edge/README.md @@ -0,0 +1,253 @@ +# Production edge deployment + +This opt-in deployment keeps PostgreSQL, supervisor control, the standalone dashboard, +the Worker API, and its typed admin backend on the runtime container's loopback. The +root-owned deployment profile selects one of two exact Caddy topologies. Both keep the +private Caddy admin API on an unpublished Unix socket and preserve the same Worker API, +admin authentication, operator attribution, denylist, and header contract. + +## Deployment profiles + +If `/etc/truf/deployment-profile` is absent, `standalone-edge-v1` is selected. The +standalone profile publishes runtime TCP 443 and gives the managed edge only +`NET_BIND_SERVICE`. + +For a host whose existing root-owned Caddy must remain the sole owner of ports 80/443, +install the shared profile before running the host-agent installer: + +```sh +printf '%s\n' shared-host-edge-v1 | sudo install -m 0444 -o root -g root /dev/stdin /etc/truf/deployment-profile +``` + +`shared-host-edge-v1` runs the runtime in the host network namespace with no Docker +published ports. The managed Truf edge shares that namespace, has no capabilities, and +binds plain HTTP only at `127.0.0.1:18766`. The existing host Caddy imports the fixed +route-only `deploy/edge/host-caddy-shared.caddy` snippet inside the reviewed public site. +Install that import before any catch-all handler. It handles only `/api/v1/worker/*` and +the exact random admin prefix; it does not define a listener, TLS policy, global option, +or route for another application. The host agent never restarts or reconfigures host +Caddy or X-UI. + +Profile changes are maintenance operations: stop the host agent first, require no active +apply or failed hold, install the exact root-owned mode-0444 value, validate the selected +Compose projection and host-Caddy configuration, then restart the agent. Never expose +profile selection through the admin or host-agent request. + +### Choosing a topology + +Use `standalone-edge-v1` on a dedicated host where the managed edge can own public TCP +443. The request path is: + +```text +Internet -> managed Caddy :443 -> private runtime :8766 +``` + +Use `shared-host-edge-v1` only when an existing root-owned Caddy must remain the sole +owner of public ports and TLS. The request path is: + +```text +Internet -> host Caddy :443 -> 127.0.0.1:18766 -> managed Caddy -> private runtime :8766 +``` + +Shared-host mode adds a one-time integration boundary, not a second public edge. The +operator installs the fixed route snippet, places its import before every catch-all, +supplies the independent ingress marker, and validates the complete host Caddy +configuration. After that bootstrap, runtime restart and document apply use the same +host-agent lifecycle as standalone mode. The host agent never owns the host Caddy +configuration or service lifecycle. + +These are the only supported production topologies. Nginx, Traefik, an arbitrary Caddy +layout, or an ad-hoc Compose override is not equivalent to either profile. Add and test a +new exact deployment profile instead of translating private headers approximately. The +host agent validates the selected fixed Compose projection and rejects metadata drift. + +The shared-host projection currently carries the constrained-host runtime limits declared +in `compose.shared-host.yaml`. A materially different CPU or memory envelope also requires +a reviewed profile change; do not hide it in an unvalidated local override. + +### End-to-end host bootstrap + +The repository provides fixed deployment components, not a universal VPS installer, +Ansible role, public image registry, or infrastructure module. Bootstrap a new host from +one reviewed release checkout as follows: + +1. Install the reviewed Linux, Docker Engine and Compose plugin, systemd, Python 3, and + fail2ban prerequisites; provision DNS and the selected TLS ownership boundary. +2. Install the release checkout root-owned at `/opt/truf` and choose exactly one deployment + profile before installing the host agent. +3. In shared-host mode, install the fixed host-Caddy import, validate the complete host + configuration, and prove an unavailable loopback edge cannot fall through to another + application. +4. Create the protected edge directories, denylist state, and mode-0600 edge environment + described below. Generate independent admin, edge, and shared-ingress values rather + than copying values from another host. +5. Install and validate the fixed host agent. Its first install seeds an absent active + config from `app/config.linux.yaml` and an absent secrets document as an empty mapping; + repeat installation never replaces active documents. +6. Install every trusted worker-package manifest referenced by the runtime config beneath + `/etc/truf/worker-packages` with the exact ownership and mode described below. +7. Review the private active config and secrets, then build `runtime` and `edge` from the + same checkout with the exact base and selected profile Compose files. +8. Start the stack, explicitly enable Worker API and admin only after their private + configuration is complete, and verify PostgreSQL, runtime, edge, HTTPS, admin, Worker + API, host-agent, and unrelated host applications. + +Initial host bootstrap is therefore intentionally more manual than later operation. +Normal config apply, restart, rollback, status, and audit are performed through the typed +control plane and fixed host agent after this trust boundary is established. + +## Host agent and fixed runtime paths + +Install the root-owned checkout at `/opt/truf`. Before invoking the host-agent installer, +prepare the edge state below and create the complete protected environment file that its +fixed combined-Compose validation consumes. + +UID/GID 10001 is the numeric edge identity. The denylist directory is mounted, rather than +its file, so atomic replacement remains visible in the container. + +```sh +sudo install -d -o 10001 -g 10001 -m 0700 /var/log/truf-edge +sudo install -d -o root -g 10001 -m 2750 /etc/truf-edge/denylist +sudo install -m 0640 -o root -g 10001 deploy/edge/admin-denylist.caddy /etc/truf-edge/denylist/admin-denylist.caddy +sudo install -d -o root -g root -m 0700 /var/lib/truf-edge +sudo install -m 0750 -o root -g root deploy/fail2ban/truf_caddy_admin_denylist.py /usr/local/sbin/truf-caddy-admin-denylist +``` + +Create `/etc/truf-edge/edge.env` as root with mode 0600. Generate a new admin segment +with `openssl rand -hex 32`. It must be exactly 64 lowercase hex characters (256 random +bits). Generate the bcrypt value interactively with the pinned edge image's +`caddy hash-password` command; never put the plaintext password in a command, file, or +Compose variable. + +```dotenv +TRUF_EDGE_HOST=edge.example.net +TRUF_ADMIN_PREFIX=replace_with_64_lowercase_hex_characters +TRUF_ADMIN_USER=operator +TRUF_ADMIN_PASSWORD_HASH='$2a$14$replace_with_a_real_caddy_bcrypt_hash' +TRUF_ADMIN_EDGE_MARKER=replace_with_a_second_independent_64_character_hex_secret +TRUF_SHARED_INGRESS_MARKER=replace_with_a_third_independent_64_character_hex_secret +TRUF_EDGE_AUTH_LOG_DIR=/var/log/truf-edge +TRUF_EDGE_DENYLIST_DIR=/etc/truf-edge/denylist +``` + +`TRUF_SHARED_INGRESS_MARKER` is required only by `shared-host-edge-v1`. Host Caddy strips +any inbound transit/private headers, injects this marker and its observed client address, +and proxies to loopback. The managed edge rejects a missing marker before trusting that +address and removes the marker before proxying to the application. + +Now install and validate the fixed host agent. The installer creates the fixed candidate, +result, PostgreSQL socket, and active-document paths and enables +`/run/truf/host-agent.sock`; Compose refuses to create missing bind sources. + +```sh +sudo /usr/bin/python3 -I -S -B /opt/truf/deploy/host-agent/truf_host_agent_install.py install +sudo /usr/bin/python3 -I -S -B /opt/truf/deploy/host-agent/truf_host_agent_install.py validate +``` + +The active `/etc/truf/runtime/config.yaml` and `secrets.yaml` are UID/GID 10001 mode 0600 +documents and are never overwritten by repeat installation. Any package manifest referenced +by the config must be installed beneath `/etc/truf/worker-packages` as a root-owned, +root:root mode 0644 regular file before validation. The runtime maps that immutable authority +read-only at `/data/worker-packages`; do not place manifests in the private active-document +directory. + +Automatic TLS remains the default. A deployment that must use operator-provided +certificates can mount a root-owned, non-link `*.caddy` file under `/etc/caddy/tls` and +set `TRUF_EDGE_TLS_INCLUDE` to that absolute container path in a reviewed Compose +override. The include should contain only the site's `tls CERT KEY` directive. Never use +the repository's localhost test certificate or key in a deployment. + +The normal `compose.yaml` remains private and unchanged. Confirm the host-agent socket is +active and rerun installer validation immediately before starting production edge. Always +supply the base file, the exact selected profile file, and the protected environment file. +For standalone: + +```sh +docker compose --env-file /etc/truf-edge/edge.env -f compose.yaml -f compose.edge.yaml build runtime edge +docker compose --env-file /etc/truf-edge/edge.env -f compose.yaml -f compose.edge.yaml up -d +``` + +For shared host: + +```sh +docker compose --env-file /etc/truf-edge/edge.env -f compose.yaml -f compose.shared-host.yaml build runtime edge +docker compose --env-file /etc/truf-edge/edge.env -f compose.yaml -f compose.shared-host.yaml up -d +``` + +Before starting shared host, validate the complete existing host Caddy configuration with +the snippet import in place. A matching request must fail at that Truf route if the +loopback edge is unavailable; it must never fall through to X-UI or another upstream. + +Provisioning creates the private `/data/managed-files` namespace in the named data volume. +Each configured writable root must be a reviewed immediate child such as +`/data/managed-files/exports`, created with UID/GID 10001 and mode 0700 while the runtime is +stopped. Arbitrary host bind paths are not managed-file roots. + +Worker admission remains disabled by `app/config.linux.yaml`. Configure the private +runtime config's worker sources, compatibility profiles, and hashed device credentials +before explicitly enabling `supervisor.worker_api.enabled`. The edge does not enable it. +The typed admin backend is disabled independently under `supervisor.worker_api.admin`. +Set its exact `origin` to `https://TRUF_EDGE_HOST`, set `edge_marker` to the same independent +256-bit value as `TRUF_ADMIN_EDGE_MARKER`, then explicitly enable it. Both authenticated +surfaces share the private runtime loopback port 8766. Caddy strips any inbound +`X-Truf-Admin-Edge` and `X-Truf-Admin-Operator`, sets the configured marker and the +authenticated Basic-auth username only after authentication, and rewrites the public +random prefix to the private `/admin-internal` backend path. The backend accepts the +operator identity only together with the private marker. +Worker API requests receive neither private admin header. + +Build and client bootstrap instructions for Windows and Linux remote workers are in +`docs/remote-worker-operations.md`. Worker executables and images must be produced from a +reviewed release checkout; operators must not assemble Python, Git, TruffleHog, detector +policy, or dependencies manually on each worker. + +## Fail2ban + +Install the host files under their conventional names and enable fail2ban plus the expiry +timer. The jail counts only redacted `admin_auth_failure` JSON records. An initial Basic +challenge without credentials, Worker API authentication failures, and unrelated 404s do +not enter that log. The action changes only the matcher imported inside the secret admin +route; it does not create firewall rules and therefore does not block workers sharing an IP. + +```sh +sudo install -m 0644 deploy/fail2ban/filter.d-truf-admin-auth.conf /etc/fail2ban/filter.d/truf-admin-auth.conf +sudo install -m 0644 deploy/fail2ban/jail.d-truf-admin-auth.local /etc/fail2ban/jail.d/truf-admin-auth.local +sudo install -m 0644 deploy/fail2ban/action.d-truf-caddy-admin-denylist.conf /etc/fail2ban/action.d/truf-caddy-admin-denylist.conf +sudo install -m 0644 deploy/fail2ban/fail2ban.d-truf-persistence.local /etc/fail2ban/fail2ban.d/truf-persistence.local +sudo install -m 0644 deploy/systemd/truf-caddy-admin-denylist-expire.service /etc/systemd/system/ +sudo install -m 0644 deploy/systemd/truf-caddy-admin-denylist-expire.timer /etc/systemd/system/ +sudo systemctl daemon-reload +sudo systemctl enable --now fail2ban truf-caddy-admin-denylist-expire.timer +sudo fail2ban-client status truf-admin-auth +``` + +Fail2ban persists jail state in `/var/lib/fail2ban/fail2ban.sqlite3`. The updater persists +canonical IPs and expiry timestamps in private +`/var/lib/truf-edge/admin-denylist.json`. It validates the complete Caddyfile in the running +edge container, reloads it through the private admin endpoint, and restores/reloads the +previous state if a command fails. Reload uses Caddy's private `/run/caddy-admin.sock` inside the edge +container. The socket is not mounted or published. Shared mode renders denylist matchers +against the marker-authenticated client address; standalone mode uses the direct peer. + +## SSH recovery + +Use fail2ban's normal unban first so its database and Caddy agree: + +```sh +sudo fail2ban-client set truf-admin-auth unbanip 203.0.113.10 +sudo /usr/local/sbin/truf-caddy-admin-denylist status +sudo /usr/local/sbin/truf-caddy-admin-denylist expire +``` + +If fail2ban is unavailable, run the updater's explicit unban over SSH: + +```sh +sudo /usr/local/sbin/truf-caddy-admin-denylist unban 203.0.113.10 +``` + +For recovery from a damaged generated snippet, stop the expiry timer and fail2ban, restore +`deploy/edge/admin-denylist.caddy` to `/etc/truf-edge/denylist/admin-denylist.caddy`, then +run Caddy validation and reload through the private Unix admin socket only after validation +succeeds. Reconcile each +remaining address with the updater before re-enabling the services. Do not use a global +firewall ban as a shortcut. diff --git a/deploy/edge/admin-denylist.caddy b/deploy/edge/admin-denylist.caddy new file mode 100644 index 0000000..dd24d0f --- /dev/null +++ b/deploy/edge/admin-denylist.caddy @@ -0,0 +1 @@ +# Managed by truf-caddy-admin-denylist. Admin-route import only. diff --git a/deploy/edge/automatic-tls.caddy b/deploy/edge/automatic-tls.caddy new file mode 100644 index 0000000..a6915ab --- /dev/null +++ b/deploy/edge/automatic-tls.caddy @@ -0,0 +1 @@ +# Empty by design: Caddy's automatic TLS remains the production default. diff --git a/deploy/edge/entrypoint.sh b/deploy/edge/entrypoint.sh new file mode 100644 index 0000000..39e648c --- /dev/null +++ b/deploy/edge/entrypoint.sh @@ -0,0 +1,71 @@ +#!/bin/sh +set -eu + +fail() { + echo "edge configuration rejected: $1" >&2 + exit 64 +} + +host=${TRUF_EDGE_HOST:-} +prefix=${TRUF_ADMIN_PREFIX:-} +user=${TRUF_ADMIN_USER:-} +password_hash=${TRUF_ADMIN_PASSWORD_HASH:-} +edge_marker=${TRUF_ADMIN_EDGE_MARKER:-} +edge_mode=${TRUF_EDGE_MODE:-standalone-edge-v1} +ingress_marker=${TRUF_SHARED_INGRESS_MARKER:-} +tls_include=${TRUF_EDGE_TLS_INCLUDE:-} + +case "$edge_mode" in + standalone-edge-v1) caddyfile=/etc/caddy/Caddyfile ;; + shared-host-edge-v1) + caddyfile=/etc/caddy/Caddyfile.shared-host + [ "${#ingress_marker}" -eq 64 ] || fail "TRUF_SHARED_INGRESS_MARKER must encode 256 random bits" + printf '%s' "$ingress_marker" | grep -Eq '^[0-9a-f]{64}$' \ + || fail "TRUF_SHARED_INGRESS_MARKER must encode 256 random bits" + ;; + *) fail "TRUF_EDGE_MODE is unsupported" ;; +esac + +if [ "$host" = localhost ]; then + [ -n "$tls_include" ] || fail "localhost requires an explicit static TLS include" +else + case "$host" in + ''|*://*|*/*|*:*|.*|*..*|*.) fail "TRUF_EDGE_HOST must be one DNS hostname" ;; + esac + printf '%s' "$host" | awk -F. ' + length($0) > 253 || NF < 2 { exit 1 } + { for (i = 1; i <= NF; i++) if (length($i) > 63 || $i !~ /^[A-Za-z0-9-]+$/ || $i ~ /^-/ || $i ~ /-$/) exit 1 } + ' \ + || fail "TRUF_EDGE_HOST must be one DNS hostname" +fi + +if [ -n "$tls_include" ]; then + case "$tls_include" in + /etc/caddy/tls/*.caddy) ;; + *) fail "TRUF_EDGE_TLS_INCLUDE must be an absolute Caddy TLS include" ;; + esac + [ -f "$tls_include" ] && [ ! -L "$tls_include" ] \ + || fail "TRUF_EDGE_TLS_INCLUDE must be a regular non-link file" +fi + +[ "${#prefix}" -eq 64 ] || fail "TRUF_ADMIN_PREFIX must encode 256 random bits" +printf '%s' "$prefix" | grep -Eq '^[0-9a-f]{64}$' \ + || fail "TRUF_ADMIN_PREFIX must encode 256 random bits" + +printf '%s' "$user" | grep -Eq '^[A-Za-z0-9_.-]{1,64}$' \ + || fail "TRUF_ADMIN_USER has an unsupported form" +printf '%s' "$password_hash" | grep -Eq '^\$2[aby]\$(0[4-9]|[12][0-9]|3[01])\$[./A-Za-z0-9]{53}$' \ + || fail "TRUF_ADMIN_PASSWORD_HASH must be a supported bcrypt hash" +[ "${#edge_marker}" -eq 64 ] || fail "TRUF_ADMIN_EDGE_MARKER must encode 256 random bits" +printf '%s' "$edge_marker" | grep -Eq '^[0-9a-f]{64}$' \ + || fail "TRUF_ADMIN_EDGE_MARKER must encode 256 random bits" + +[ -f /etc/caddy/denylist/admin-denylist.caddy ] \ + || fail "the managed admin denylist snippet is missing" +[ ! -L /etc/caddy/denylist/admin-denylist.caddy ] \ + || fail "the managed admin denylist snippet must not be a link" +[ -d /var/log/caddy ] && [ -w /var/log/caddy ] \ + || fail "the authentication log directory is not writable" + +umask 077 +exec caddy run --config "$caddyfile" --adapter caddyfile diff --git a/deploy/edge/host-caddy-shared.caddy b/deploy/edge/host-caddy-shared.caddy new file mode 100644 index 0000000..a982fc3 --- /dev/null +++ b/deploy/edge/host-caddy-shared.caddy @@ -0,0 +1,26 @@ +# Import this route-only snippet inside the reviewed public site block. +@truf_worker path /api/v1/worker/* +handle @truf_worker { + reverse_proxy 127.0.0.1:18766 { + header_up X-Truf-Shared-Ingress {$TRUF_SHARED_INGRESS_MARKER} + header_up -X-Truf-Admin-Edge + header_up -X-Truf-Admin-Operator + header_up -Forwarded + header_up -X-Real-IP + header_up -X-Forwarded-For + header_up X-Forwarded-For {http.request.remote.host} + } +} + +@truf_admin path /{$TRUF_ADMIN_PREFIX} /{$TRUF_ADMIN_PREFIX}/* +handle @truf_admin { + reverse_proxy 127.0.0.1:18766 { + header_up X-Truf-Shared-Ingress {$TRUF_SHARED_INGRESS_MARKER} + header_up -X-Truf-Admin-Edge + header_up -X-Truf-Admin-Operator + header_up -Forwarded + header_up -X-Real-IP + header_up -X-Forwarded-For + header_up X-Forwarded-For {http.request.remote.host} + } +} diff --git a/deploy/fail2ban/Dockerfile.edge-e2e b/deploy/fail2ban/Dockerfile.edge-e2e new file mode 100644 index 0000000..f8db42b --- /dev/null +++ b/deploy/fail2ban/Dockerfile.edge-e2e @@ -0,0 +1,61 @@ +FROM caddy:2.10.2-alpine@sha256:4c6e91c6ed0e2fa03efd5b44747b625fec79bc9cd06ac5235a779726618e530d AS caddy-edge-e2e + +FROM debian:bookworm-slim@sha256:88200866dfff7ea7f5cbcb6ec7c8a701889efe6fe859fe64d6990e4b07ea4171 AS fail2ban-edge-e2e + +COPY --from=caddy-edge-e2e --chown=0:0 --chmod=0444 /etc/ssl/certs/ca-certificates.crt /etc/ssl/certs/ca-certificates.crt + +RUN <<'SH' +set -eu +rm -f /etc/apt/sources.list /etc/apt/sources.list.d/debian.sources +printf '%s\n' \ + 'Types: deb' \ + 'URIs: http://snapshot.debian.org/archive/debian/20260914T000000Z/' \ + 'Suites: bookworm bookworm-updates' \ + 'Components: main' \ + 'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \ + 'Check-Valid-Until: no' \ + '' \ + 'Types: deb' \ + 'URIs: http://snapshot.debian.org/archive/debian-security/20260914T000000Z/' \ + 'Suites: bookworm-security' \ + 'Components: main' \ + 'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \ + 'Check-Valid-Until: no' \ + > /etc/apt/sources.list.d/debian.sources +printf '#!/bin/sh\nexit 101\n' > /usr/sbin/policy-rc.d +chmod 0755 /usr/sbin/policy-rc.d +export DEBIAN_FRONTEND=noninteractive +apt-get -o Acquire::Retries=3 -o Acquire::https::Timeout=30 -o APT::Update::Error-Mode=any update +apt-get install -y --no-install-recommends fail2ban=1.0.2-2 +rm -rf /var/lib/apt/lists/* /var/cache/apt/archives/* /var/log/apt/* +find /etc/fail2ban/jail.d -type f -delete +install -d -o 0 -g 0 -m 0755 \ + /etc/caddy /etc/caddy/denylist /etc/caddy/tls /etc/fail2ban/action.d /etc/fail2ban/fail2ban.d \ + /etc/fail2ban/filter.d /etc/fail2ban/jail.d /etc/truf-edge/denylist \ + /run/fail2ban /var/lib/fail2ban /var/lib/truf-edge /var/log/caddy /var/log/truf-edge +SH + +COPY --from=caddy-edge-e2e --chown=0:0 --chmod=0555 /usr/bin/caddy /usr/bin/caddy +RUN cp /usr/bin/caddy /usr/bin/caddy-edge-e2e \ + && rm /usr/bin/caddy \ + && mv /usr/bin/caddy-edge-e2e /usr/bin/caddy \ + && chmod 0555 /usr/bin/caddy +COPY --chown=0:0 --chmod=0444 deploy/edge/Caddyfile /etc/caddy/Caddyfile +RUN chmod 0644 /etc/caddy/Caddyfile \ + && sed -i 's/^[[:space:]]*admin off$/\tadmin 127.0.0.1:2019/' /etc/caddy/Caddyfile \ + && chmod 0444 /etc/caddy/Caddyfile \ + && grep -Fx ' admin 127.0.0.1:2019' /etc/caddy/Caddyfile >/dev/null +COPY --chown=0:0 --chmod=0444 deploy/fail2ban/filter.d-truf-admin-auth.conf /etc/fail2ban/filter.d/truf-admin-auth.conf +COPY --chown=0:0 --chmod=0444 deploy/fail2ban/jail.d-truf-admin-auth.local /etc/fail2ban/jail.d/truf-admin-auth.local +COPY --chown=0:0 --chmod=0444 deploy/fail2ban/action.d-truf-caddy-admin-denylist.conf /etc/fail2ban/action.d/truf-caddy-admin-denylist.conf +COPY --chown=0:0 --chmod=0444 deploy/fail2ban/fail2ban.d-truf-persistence.local /etc/fail2ban/fail2ban.d/truf-persistence.local +COPY --chown=0:0 --chmod=0444 deploy/fail2ban/fail2ban.d-edge-e2e.local /etc/fail2ban/fail2ban.d/edge-e2e.local +COPY --chown=0:0 --chmod=0555 deploy/fail2ban/truf_caddy_admin_denylist.py /usr/local/sbin/truf-caddy-admin-denylist +COPY --chown=0:0 --chmod=0555 deploy/fail2ban/edge_e2e_docker_shim.py /usr/local/bin/docker + +RUN fail2ban-server --version 2>&1 | grep -F 'v1.0.2' >/dev/null \ + && test "$(find /etc/fail2ban/jail.d -type f | wc -l)" -eq 1 + +USER 0:0 +ENTRYPOINT ["/usr/bin/fail2ban-server", "-f", "-x"] +CMD [] diff --git a/deploy/fail2ban/action.d-truf-caddy-admin-denylist.conf b/deploy/fail2ban/action.d-truf-caddy-admin-denylist.conf new file mode 100644 index 0000000..ae9c3fa --- /dev/null +++ b/deploy/fail2ban/action.d-truf-caddy-admin-denylist.conf @@ -0,0 +1,4 @@ +[Definition] +actionstart = /usr/local/sbin/truf-caddy-admin-denylist expire +actionban = /usr/local/sbin/truf-caddy-admin-denylist ban '' +actionunban = /usr/local/sbin/truf-caddy-admin-denylist unban '' diff --git a/deploy/fail2ban/edge_e2e_docker_shim.py b/deploy/fail2ban/edge_e2e_docker_shim.py new file mode 100644 index 0000000..c65cee0 --- /dev/null +++ b/deploy/fail2ban/edge_e2e_docker_shim.py @@ -0,0 +1,194 @@ +#!/usr/bin/python3 +"""Constrain production denylist reload commands to the colocated E2E Caddy.""" + +import os +from pathlib import Path +import re +import stat +import subprocess +import sys + + +ENV_FILE = Path("/etc/truf-edge/edge.env") +AUDIT_PATH = Path("/var/lib/truf-edge/edge-e2e-reload.audit") +VALIDATION_ERROR_PATH = Path("/var/lib/truf-edge/edge-e2e-validation.error") +COMPOSE_PREFIX = ( + "compose", "--ansi", "never", "--env-file", str(ENV_FILE), + "--project-directory", "/opt/truf", + "--file", "/opt/truf/compose.yaml", + "--file", "/opt/truf/compose.edge.yaml", +) +VALIDATE_COMMAND = COMPOSE_PREFIX + ( + "exec", "-T", "edge", "caddy", "validate", + "--config", "/etc/caddy/Caddyfile", "--adapter", "caddyfile", +) +RELOAD_COMMAND = COMPOSE_PREFIX + ("kill", "--signal", "SIGUSR1", "edge") +REQUIRED_ENV = { + "TRUF_EDGE_HOST", + "TRUF_EDGE_TLS_INCLUDE", + "TRUF_ADMIN_PREFIX", + "TRUF_ADMIN_USER", + "TRUF_ADMIN_PASSWORD_HASH", + "TRUF_ADMIN_EDGE_MARKER", +} + + +def classify_command(arguments): + command = tuple(arguments) + if command == VALIDATE_COMMAND: + return "validate" + if command == RELOAD_COMMAND: + return "reload" + raise ValueError("unsupported command") + + +def audit(operation, result): + payload = f"{operation}:{result}\n".encode("ascii") + flags = ( + os.O_WRONLY | os.O_APPEND | os.O_CREAT | getattr(os, "O_NOFOLLOW", 0) + | getattr(os, "O_BINARY", 0) + ) + descriptor = os.open(AUDIT_PATH, flags, 0o600) + try: + details = os.fstat(descriptor) + if not stat.S_ISREG(details.st_mode) or details.st_size + len(payload) > 4096: + raise ValueError("invalid audit file") + os.write(descriptor, payload) + finally: + os.close(descriptor) + + +def record_validation_error(content, environment): + if len(content) > 65536: + content = b"caddy validation error exceeded evidence bound\n" + text = content.decode("utf-8", errors="replace") + for value in environment.values(): + if value: + text = text.replace(value, "[redacted]") + text = re.sub(r"\$2[aby]\$(?:0[4-9]|[12][0-9]|3[01])\$[./A-Za-z0-9]{53}", "[redacted]", text) + text = re.sub(r"(? 8192: + raise ValueError("invalid environment file") + values = {} + for raw_line in path.read_text(encoding="ascii").splitlines(): + if not raw_line or raw_line.startswith("#"): + continue + name, separator, value = raw_line.partition("=") + if not separator or name not in REQUIRED_ENV or name in values or "\x00" in value: + raise ValueError("invalid environment entry") + values[name] = value + if set(values) != REQUIRED_ENV: + raise ValueError("incomplete environment") + if ( + values["TRUF_EDGE_HOST"] != "localhost" + or values["TRUF_EDGE_TLS_INCLUDE"] != "/etc/caddy/tls/static-tls.caddy" + or not re.fullmatch(r"[0-9a-f]{64}", values["TRUF_ADMIN_PREFIX"]) + or not re.fullmatch(r"[A-Za-z0-9_.-]{1,64}", values["TRUF_ADMIN_USER"]) + or not re.fullmatch(r"\$2[aby]\$(?:0[4-9]|[12][0-9]|3[01])\$[./A-Za-z0-9]{53}", values["TRUF_ADMIN_PASSWORD_HASH"]) + or not re.fullmatch(r"[0-9a-f]{64}", values["TRUF_ADMIN_EDGE_MARKER"]) + ): + raise ValueError("unsupported environment") + return { + **values, + "HOME": "/tmp", + "LANG": "C.UTF-8", + "LC_ALL": "C.UTF-8", + "PATH": "/usr/bin:/bin", + } + + +def validate(): + try: + environment = load_environment() + except Exception: + audit("environment", 64) + raise + try: + completed = subprocess.run( + ( + "/usr/bin/caddy", "validate", "--config", "/etc/caddy/Caddyfile", + "--adapter", "caddyfile", + ), + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + env=environment, + timeout=30, + check=False, + ) + except Exception: + audit("caddy-exec", 64) + raise + if completed.returncode: + record_validation_error(completed.stderr, environment) + return completed.returncode + + +def reload_caddy(): + try: + command = Path("/proc/1/cmdline").read_bytes() + except Exception: + audit("reload-proc", 64) + raise + if ( + len(command) > 4096 + or command.rstrip(b"\0").split(b"\0") + not in ( + [b"caddy", b"run", b"--config", b"/etc/caddy/Caddyfile", b"--adapter", b"caddyfile"], + [b"/usr/bin/caddy", b"run", b"--config", b"/etc/caddy/Caddyfile", b"--adapter", b"caddyfile"], + ) + ): + audit("reload-identity", 64) + raise ValueError("unexpected pid namespace") + completed = subprocess.run( + ( + "/usr/bin/caddy", "reload", "--config", "/etc/caddy/Caddyfile", + "--adapter", "caddyfile", "--address", "127.0.0.1:2019", + ), + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + env=load_environment(), + timeout=30, + check=False, + ) + return completed.returncode + + +def main(argv=None): + try: + operation = classify_command((argv or sys.argv)[1:]) + result = validate() if operation == "validate" else reload_caddy() + except Exception: + if "operation" in locals(): + try: + audit(operation, 64) + except Exception: + pass + return 64 + try: + audit(operation, result) + except Exception: + return 64 + return result + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/deploy/fail2ban/fail2ban.d-edge-e2e.local b/deploy/fail2ban/fail2ban.d-edge-e2e.local new file mode 100644 index 0000000..a8a7692 --- /dev/null +++ b/deploy/fail2ban/fail2ban.d-edge-e2e.local @@ -0,0 +1,5 @@ +[Definition] +loglevel = INFO +logtarget = STDOUT +socket = /run/fail2ban/fail2ban.sock +pidfile = /run/fail2ban/fail2ban.pid diff --git a/deploy/fail2ban/fail2ban.d-truf-persistence.local b/deploy/fail2ban/fail2ban.d-truf-persistence.local new file mode 100644 index 0000000..30816cd --- /dev/null +++ b/deploy/fail2ban/fail2ban.d-truf-persistence.local @@ -0,0 +1,3 @@ +[Definition] +dbfile = /var/lib/fail2ban/fail2ban.sqlite3 +dbpurgeage = 7d diff --git a/deploy/fail2ban/filter.d-truf-admin-auth.conf b/deploy/fail2ban/filter.d-truf-admin-auth.conf new file mode 100644 index 0000000..1922ef6 --- /dev/null +++ b/deploy/fail2ban/filter.d-truf-admin-auth.conf @@ -0,0 +1,4 @@ +[Definition] +failregex = ^(?=.{1,1024}$)(?=.*"event"\s*:\s*"admin_auth_failure")(?=.*"status"\s*:\s*401)(?=.*"remote_ip"\s*:\s*"")(?!.*"(?:request|uri|headers|authorization|password|token|prefix)"\s*:).*\s*$ +ignoreregex = +datepattern = "ts":{EPOCH} diff --git a/deploy/fail2ban/jail.d-truf-admin-auth.local b/deploy/fail2ban/jail.d-truf-admin-auth.local new file mode 100644 index 0000000..5ebbf02 --- /dev/null +++ b/deploy/fail2ban/jail.d-truf-admin-auth.local @@ -0,0 +1,9 @@ +[truf-admin-auth] +enabled = true +filter = truf-admin-auth +logpath = /var/log/truf-edge/admin-auth-failures.json +backend = auto +maxretry = 2 +findtime = 10m +bantime = 24h +action = truf-caddy-admin-denylist diff --git a/deploy/fail2ban/truf_caddy_admin_denylist.py b/deploy/fail2ban/truf_caddy_admin_denylist.py new file mode 100644 index 0000000..cabd6ec --- /dev/null +++ b/deploy/fail2ban/truf_caddy_admin_denylist.py @@ -0,0 +1,438 @@ +#!/usr/bin/env python3 +"""Maintain the Caddy admin-only IP denylist with durable expiry state.""" + +import argparse +from contextlib import contextmanager +import hashlib +import ipaddress +import json +import os +from pathlib import Path +import re +import stat +import subprocess +import sys +import tempfile +import time + + +BAN_SECONDS = 24 * 60 * 60 +MAX_BANS = 4096 +MAX_FILE_BYTES = 512 * 1024 +STATE_VERSION = 1 +EMPTY_SNIPPET = "# Managed by truf-caddy-admin-denylist. Admin-route import only.\n" +SHA256_RE = re.compile(r"^[0-9a-f]{64}$") +PROFILE_PATH = Path("/etc/truf/deployment-profile") +STANDALONE_PROFILE = "standalone-edge-v1" +SHARED_HOST_PROFILE = "shared-host-edge-v1" + + +class UpdateError(RuntimeError): + pass + + +class CommandFailure(UpdateError): + pass + + +class RollbackFailure(UpdateError): + pass + + +def canonical_ip(value): + text = str(value or "") + if not text or len(text) > 64 or "%" in text or any(char.isspace() for char in text): + raise UpdateError("invalid IP address") + try: + address = ipaddress.ip_address(text) + except ValueError as exc: + raise UpdateError("invalid IP address") from exc + if address.is_unspecified or address.is_multicast: + raise UpdateError("unsupported IP address") + return address.compressed.lower() + + +def render_snippet(bans, matcher="remote_ip"): + if matcher not in {"remote_ip", "client_ip"}: + raise UpdateError("unsupported denylist matcher") + addresses = sorted( + (ipaddress.ip_address(address) for address in bans), + key=lambda address: (address.version, int(address)), + ) + if not addresses: + return EMPTY_SNIPPET.encode("ascii") + lines = [EMPTY_SNIPPET.rstrip("\n")] + for offset in range(0, len(addresses), 64): + name = f"truf_admin_denied_{offset // 64:04d}" + values = " ".join(address.compressed.lower() for address in addresses[offset:offset + 64]) + lines.append(f"@{name} {matcher} {values}") + lines.append(f'respond @{name} "" 403') + return ("\n".join(lines) + "\n").encode("ascii") + + +def _digest(content): + return hashlib.sha256(content).hexdigest() + + +def _check_parent(path): + parent = path.parent + details = parent.lstat() + if not stat.S_ISDIR(details.st_mode) or stat.S_ISLNK(details.st_mode): + raise UpdateError("managed parent must be a real directory") + if os.name == "posix" and stat.S_IMODE(details.st_mode) & 0o002: + raise UpdateError("managed parent must not be world-writable") + + +def _read_optional(path): + _check_parent(path) + try: + details = path.lstat() + except FileNotFoundError: + return None + if not stat.S_ISREG(details.st_mode) or stat.S_ISLNK(details.st_mode): + raise UpdateError("managed path must be a regular file") + flags = os.O_RDONLY | getattr(os, "O_BINARY", 0) | getattr(os, "O_NOFOLLOW", 0) + descriptor = os.open(path, flags) + try: + current = os.fstat(descriptor) + if not stat.S_ISREG(current.st_mode) or current.st_size > MAX_FILE_BYTES: + raise UpdateError("managed file is invalid or too large") + chunks = [] + remaining = MAX_FILE_BYTES + 1 + while remaining: + chunk = os.read(descriptor, min(65536, remaining)) + if not chunk: + break + chunks.append(chunk) + remaining -= len(chunk) + content = b"".join(chunks) + if len(content) > MAX_FILE_BYTES: + raise UpdateError("managed file is too large") + return content + finally: + os.close(descriptor) + + +def _sync_parent(parent): + if os.name != "posix": + return + descriptor = os.open(parent, os.O_RDONLY | getattr(os, "O_DIRECTORY", 0)) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + + +def _atomic_write(path, content, mode): + _check_parent(path) + if len(content) > MAX_FILE_BYTES: + raise UpdateError("managed content is too large") + try: + existing = path.lstat() + except FileNotFoundError: + existing = None + if existing is not None and (not stat.S_ISREG(existing.st_mode) or stat.S_ISLNK(existing.st_mode)): + raise UpdateError("managed path must be a regular file") + descriptor, temporary = tempfile.mkstemp(prefix=".truf-denylist-", dir=path.parent) + temporary_path = Path(temporary) + try: + if hasattr(os, "fchmod"): + os.fchmod(descriptor, mode) + else: + os.chmod(temporary_path, mode) + with os.fdopen(descriptor, "wb", closefd=True) as handle: + descriptor = -1 + handle.write(content) + handle.flush() + os.fsync(handle.fileno()) + os.replace(temporary_path, path) + _sync_parent(path.parent) + finally: + if descriptor >= 0: + os.close(descriptor) + try: + temporary_path.unlink() + except FileNotFoundError: + pass + + +def _restore(path, content, mode): + if content is not None: + _atomic_write(path, content, mode) + return + try: + details = path.lstat() + except FileNotFoundError: + return + if not stat.S_ISREG(details.st_mode) or stat.S_ISLNK(details.st_mode): + raise UpdateError("managed path changed during rollback") + path.unlink() + _sync_parent(path.parent) + + +@contextmanager +def _exclusive_lock(path): + _check_parent(path) + flags = os.O_RDWR | os.O_CREAT | getattr(os, "O_BINARY", 0) | getattr(os, "O_NOFOLLOW", 0) + descriptor = os.open(path, flags, 0o600) + try: + details = os.fstat(descriptor) + if not stat.S_ISREG(details.st_mode): + raise UpdateError("lock path must be a regular file") + if os.name == "posix": + import fcntl + fcntl.flock(descriptor, fcntl.LOCK_EX) + yield + finally: + os.close(descriptor) + + +def _load_state(content): + if content is None: + return {"version": STATE_VERSION, "bans": {}, "applied_sha256": ""} + try: + value = json.loads(content.decode("ascii")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise UpdateError("denylist state is not valid JSON") from exc + if not isinstance(value, dict) or set(value) != {"version", "bans", "applied_sha256"}: + raise UpdateError("denylist state has an invalid schema") + if value["version"] != STATE_VERSION or not isinstance(value["bans"], dict): + raise UpdateError("denylist state has an unsupported version") + if len(value["bans"]) > MAX_BANS: + raise UpdateError("denylist state exceeds its entry bound") + applied = value["applied_sha256"] + if not isinstance(applied, str) or (applied and not SHA256_RE.fullmatch(applied)): + raise UpdateError("denylist state has an invalid applied digest") + bans = {} + for address, expires_at in value["bans"].items(): + canonical = canonical_ip(address) + if canonical != address or isinstance(expires_at, bool) or not isinstance(expires_at, int): + raise UpdateError("denylist state has a noncanonical entry") + if expires_at <= 0 or expires_at > 253402300799: + raise UpdateError("denylist state has an invalid expiry") + bans[canonical] = expires_at + return {"version": STATE_VERSION, "bans": bans, "applied_sha256": applied} + + +def _encode_state(state): + return (json.dumps(state, sort_keys=True, separators=(",", ":")) + "\n").encode("ascii") + + +def _subprocess_runner(command): + environment = { + "HOME": "/root", + "LANG": "C.UTF-8", + "LC_ALL": "C.UTF-8", + "PATH": "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin", + } + try: + completed = subprocess.run( + command, + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + env=environment, + timeout=45, + check=False, + ) + except (OSError, subprocess.SubprocessError): + return False + return completed.returncode == 0 + + +class DenylistUpdater: + def __init__( + self, + state_path, + snippet_path, + project_directory="/opt/truf", + env_file="/etc/truf-edge/edge.env", + profile=None, + runner=None, + clock=None, + ): + self.state_path = Path(state_path) + self.snippet_path = Path(snippet_path) + self.lock_path = self.state_path.with_suffix(self.state_path.suffix + ".lock") + if profile is None: + try: + profile = PROFILE_PATH.read_text(encoding="ascii").strip() + except FileNotFoundError: + profile = STANDALONE_PROFILE + except (OSError, UnicodeError): + raise UpdateError("deployment profile is unreadable") from None + if profile not in {STANDALONE_PROFILE, SHARED_HOST_PROFILE}: + raise UpdateError("deployment profile is unsupported") + compose_file = ( + "compose.shared-host.yaml" + if profile == SHARED_HOST_PROFILE else "compose.edge.yaml" + ) + caddyfile = ( + "/etc/caddy/Caddyfile.shared-host" + if profile == SHARED_HOST_PROFILE else "/etc/caddy/Caddyfile" + ) + self.matcher = "client_ip" if profile == SHARED_HOST_PROFILE else "remote_ip" + compose = ( + "docker", "compose", "--ansi", "never", "--env-file", str(env_file), + "--project-directory", str(project_directory), + "--file", str(Path(project_directory) / "compose.yaml"), + "--file", str(Path(project_directory) / compose_file), + ) + self.validate_command = compose + ( + "exec", "-T", "edge", "caddy", "validate", + "--config", caddyfile, "--adapter", "caddyfile", + ) + self.reload_command = compose + ( + "exec", "-T", "edge", "caddy", "reload", + "--config", caddyfile, "--adapter", "caddyfile", + "--address", "unix//run/caddy-admin.sock", + ) + self.runner = runner or _subprocess_runner + self.clock = clock or time.time + + def _run(self, command, phase): + try: + succeeded = self.runner(command) + except Exception as exc: + raise CommandFailure(f"{phase} command failed") from exc + if not succeeded: + raise CommandFailure(f"{phase} command failed") + + def update(self, operation, address=None): + if operation not in {"ban", "unban", "expire", "status"}: + raise UpdateError("unsupported operation") + canonical = canonical_ip(address) if operation in {"ban", "unban"} else None + now = int(self.clock()) + if now <= 0: + raise UpdateError("system clock is invalid") + + with _exclusive_lock(self.lock_path): + old_state_content = _read_optional(self.state_path) + old_snippet_content = _read_optional(self.snippet_path) + state = _load_state(old_state_content) + bans = { + ip: expires_at for ip, expires_at in state["bans"].items() + if expires_at > now + } + expired = len(state["bans"]) - len(bans) + + if operation == "ban": + if canonical not in bans and len(bans) >= MAX_BANS: + raise UpdateError("denylist entry bound reached") + bans[canonical] = max(bans.get(canonical, 0), now + BAN_SECONDS) + elif operation == "unban": + bans.pop(canonical, None) + + desired_snippet = render_snippet(bans, self.matcher) + desired_digest = _digest(desired_snippet) + pending_state = { + "version": STATE_VERSION, + "bans": bans, + "applied_sha256": state["applied_sha256"], + } + pending_content = _encode_state(pending_state) + needs_reload = ( + old_snippet_content != desired_snippet + or state["applied_sha256"] != desired_digest + ) + needs_state_write = old_state_content != pending_content + + if needs_reload: + reload_attempted = False + try: + _atomic_write(self.state_path, pending_content, 0o600) + _atomic_write(self.snippet_path, desired_snippet, 0o640) + self._run(self.validate_command, "validation") + reload_attempted = True + self._run(self.reload_command, "reload") + pending_state["applied_sha256"] = desired_digest + _atomic_write(self.state_path, _encode_state(pending_state), 0o600) + except Exception as original: + try: + _restore(self.state_path, old_state_content, 0o600) + _restore(self.snippet_path, old_snippet_content, 0o640) + if reload_attempted: + self._run(self.validate_command, "rollback validation") + self._run(self.reload_command, "rollback reload") + except Exception as rollback: + raise RollbackFailure("denylist rollback failed") from rollback + if isinstance(original, UpdateError): + raise + raise UpdateError("denylist update failed") from original + elif needs_state_write: + pending_state["applied_sha256"] = desired_digest + _atomic_write(self.state_path, _encode_state(pending_state), 0o600) + + return { + "operation": operation, + "ip": canonical, + "expired": expired, + "bans": dict(bans), + } + + +def _emit(result): + bans = result["bans"] + if result["operation"] == "status": + addresses = sorted( + bans, key=lambda value: (ipaddress.ip_address(value).version, int(ipaddress.ip_address(value))) + ) + payload = { + "active": len(addresses), + "bans": [ + {"ip": address, "expires_at": bans[address]} + for address in addresses[:256] + ], + "event": "admin_denylist_status", + "truncated": len(addresses) > 256, + } + else: + payload = { + "active": len(bans), + "event": "admin_denylist_" + result["operation"], + "expired": result["expired"], + } + if result["ip"] is not None: + payload["ip"] = result["ip"] + print(json.dumps(payload, sort_keys=True, separators=(",", ":")), flush=True) + + +def parse_args(argv=None): + parser = argparse.ArgumentParser(allow_abbrev=False) + parser.add_argument("--state-path", default="/var/lib/truf-edge/admin-denylist.json") + parser.add_argument("--snippet-path", default="/etc/truf-edge/denylist/admin-denylist.caddy") + parser.add_argument("--project-directory", default="/opt/truf") + parser.add_argument("--env-file", default="/etc/truf-edge/edge.env") + commands = parser.add_subparsers(dest="operation", required=True) + for name in ("ban", "unban"): + command = commands.add_parser(name, allow_abbrev=False) + command.add_argument("ip") + commands.add_parser("expire", allow_abbrev=False) + commands.add_parser("status", allow_abbrev=False) + return parser.parse_args(argv) + + +def main(argv=None): + args = parse_args(argv) + updater = DenylistUpdater( + args.state_path, + args.snippet_path, + project_directory=args.project_directory, + env_file=args.env_file, + ) + try: + result = updater.update(args.operation, getattr(args, "ip", None)) + except Exception as exc: + payload = { + "event": "admin_denylist_error", + "operation": args.operation, + "reason": type(exc).__name__, + } + print(json.dumps(payload, sort_keys=True, separators=(",", ":")), file=sys.stderr) + return 1 + _emit(result) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/deploy/host-agent/truf-host-agent.conf b/deploy/host-agent/truf-host-agent.conf new file mode 100644 index 0000000..6736d84 --- /dev/null +++ b/deploy/host-agent/truf-host-agent.conf @@ -0,0 +1,8 @@ +d /etc/truf/runtime 0755 root root - +d /etc/truf/worker-packages 0755 root root - +d /var/lib/truf/runtime-document-candidates 0700 10001 10001 - +d /var/lib/truf/host-agent 0700 root root - +d /var/lib/truf/host-agent/backups 0700 root root - +d /var/lib/truf/host-agent/operations 0700 root root - +d /var/lib/truf/host-agent/results 0750 root 10001 - +d /run/truf-postgres 0700 10001 10001 - diff --git a/deploy/host-agent/truf-host-agent.service b/deploy/host-agent/truf-host-agent.service new file mode 100644 index 0000000..6a8177a --- /dev/null +++ b/deploy/host-agent/truf-host-agent.service @@ -0,0 +1,50 @@ +[Unit] +Description=Truf privileged host operations agent +Requires=truf-host-agent.socket docker.service +After=truf-host-agent.socket docker.service + +[Service] +Type=exec +User=root +Group=root +UMask=0077 +WorkingDirectory=/ +ExecStartPre=/usr/bin/python3 -I -B /usr/lib/truf-host-agent/truf_host_agent_install.py validate +ExecStart=/usr/bin/python3 -I -B /usr/lib/truf-host-agent/truf_host_agent.py +RuntimeDirectory=truf-host-agent +RuntimeDirectoryMode=0700 +NoNewPrivileges=yes +CapabilityBoundingSet=CAP_CHOWN CAP_DAC_OVERRIDE CAP_FOWNER +AmbientCapabilities= +PrivateTmp=yes +PrivateDevices=yes +PrivateNetwork=yes +ProtectSystem=strict +ProtectHome=yes +ProtectKernelTunables=yes +ProtectKernelModules=yes +ProtectKernelLogs=yes +ProtectControlGroups=yes +ProtectClock=yes +ProtectHostname=yes +ProtectProc=invisible +ProcSubset=pid +RestrictAddressFamilies=AF_UNIX +RestrictNamespaces=yes +RestrictRealtime=yes +RestrictSUIDSGID=yes +LockPersonality=yes +MemoryDenyWriteExecute=yes +SystemCallArchitectures=native +ReadWritePaths=/etc/truf/runtime +ReadWritePaths=/var/lib/truf/host-agent +ReadWritePaths=/run/truf-host-agent +ReadWritePaths=/run/docker.sock +ReadOnlyPaths=/usr/lib/truf-host-agent +ReadOnlyPaths=/opt/truf +ReadOnlyPaths=/var/lib/truf/runtime-document-candidates +ReadOnlyPaths=/run/truf-postgres +Restart=no +TimeoutStopSec=45min +StandardOutput=null +StandardError=journal diff --git a/deploy/host-agent/truf-host-agent.socket b/deploy/host-agent/truf-host-agent.socket new file mode 100644 index 0000000..444c5bd --- /dev/null +++ b/deploy/host-agent/truf-host-agent.socket @@ -0,0 +1,16 @@ +[Unit] +Description=Truf privileged host operations socket + +[Socket] +ListenStream=/run/truf/host-agent.sock +SocketUser=root +SocketGroup=truf-runtime +SocketMode=0660 +DirectoryMode=0755 +Service=truf-host-agent.service +Accept=no +RemoveOnStop=yes +Backlog=8 + +[Install] +WantedBy=sockets.target diff --git a/deploy/host-agent/truf_host_agent.py b/deploy/host-agent/truf_host_agent.py new file mode 100644 index 0000000..1139e04 --- /dev/null +++ b/deploy/host-agent/truf_host_agent.py @@ -0,0 +1,51 @@ +#!/usr/bin/env python3 +import signal +import sys +import threading +from pathlib import Path + +APP_DIRECTORY = Path('/opt/truf/app') +sys.path.insert(0, str(APP_DIRECTORY)) + +from host_agent_runtime import FixedHostOperationDispatcher +from host_agent_server import ( + HostAgentServerError, + inherited_systemd_listener, + serve_forever, +) + + +def main(): + if len(sys.argv) != 1: + raise HostAgentServerError('arguments_forbidden') + listener = inherited_systemd_listener() + stop_event = threading.Event() + dispatcher = FixedHostOperationDispatcher() + + def request_stop(_signum, _frame): + stop_event.set() + + signal.signal(signal.SIGTERM, request_stop) + signal.signal(signal.SIGINT, request_stop) + try: + with listener: + serve_forever( + listener, handler=dispatcher.handle, stop_event=stop_event, + ) + finally: + dispatcher.close() + return 0 + + +if __name__ == '__main__': + try: + result = main() + except BaseException as exc: + if not isinstance(exc, Exception): + raise + print( + 'host operations agent failed (' + type(exc).__name__ + '); details withheld', + file=sys.stderr, + ) + result = 1 + raise SystemExit(result) diff --git a/deploy/host-agent/truf_host_agent_install.py b/deploy/host-agent/truf_host_agent_install.py new file mode 100644 index 0000000..565f1a0 --- /dev/null +++ b/deploy/host-agent/truf_host_agent_install.py @@ -0,0 +1,571 @@ +#!/usr/bin/env python3 +"""Install or validate the fixed Truf host-agent deployment.""" + +import json +import os +from pathlib import Path +import stat +import subprocess +import sys + + +PROJECT = Path('/opt/truf') +DEPLOY = PROJECT / 'deploy/host-agent' +INSTALL_ROOT = Path('/usr/lib/truf-host-agent') +SYSTEMD = Path('/etc/systemd/system') +TMPFILES = Path('/etc/tmpfiles.d/truf-host-agent.conf') +ACTIVE = Path('/etc/truf/runtime') +WORKER_PACKAGES = Path('/etc/truf/worker-packages') +DEPLOYMENT_PROFILE = Path('/etc/truf/deployment-profile') +AGENT_SOCKET = Path('/run/truf/host-agent.sock') +RUNTIME_UID = RUNTIME_GID = 10001 +RUNTIME_GROUP = 'truf-runtime' +MAX_COPY_BYTES = 4 * 1024 * 1024 +MAX_COMPOSE_OUTPUT_BYTES = 4 * 1024 * 1024 +UNITS = ('truf-host-agent.socket', 'truf-host-agent.service') +SCRIPTS = ('truf_host_agent.py', 'truf_host_agent_install.py') +FIXED_ENV = { + 'PATH': '/usr/sbin:/usr/bin:/sbin:/bin', + 'LANG': 'C', + 'LC_ALL': 'C', +} +STANDALONE_PROFILE = { + 'name': 'standalone-edge-v1', + 'compose_files': ('compose.yaml', 'compose.edge.yaml'), + 'runtime_network': None, + 'runtime_ports': [{ + 'mode': 'host', 'target': 443, 'published': '443', 'protocol': 'tcp', + }], + 'runtime_cpus': 2.0, + 'runtime_mem_limit': str(6 * 1024 ** 3), + 'edge_cap_add': ['NET_BIND_SERVICE'], + 'data_volume': {'name': 'truf-docker_data'}, +} +SHARED_HOST_PROFILE = { + 'name': 'shared-host-edge-v1', + 'compose_files': ('compose.yaml', 'compose.shared-host.yaml'), + 'runtime_network': 'host', + 'runtime_ports': None, + 'runtime_cpus': 0.9, + 'runtime_mem_limit': str(720 * 1024 ** 2), + 'edge_cap_add': None, + 'data_volume': {'name': 'truf-remote-server-data', 'external': True}, +} + + +def _compose_config_command(profile): + return ( + '/usr/bin/docker', 'compose', '--ansi', 'never', '--project-name', + 'truf-docker', '--env-file', '/etc/truf-edge/edge.env', + '--project-directory', '/opt/truf', + *(item for name in profile['compose_files'] for item in ( + '--file', '/opt/truf/' + name, + )), + 'config', '--format', 'json', + ) + + +COMPOSE_CONFIG_COMMAND = _compose_config_command(STANDALONE_PROFILE) +EXPECTED_RUNTIME_MOUNTS = ( + ('volume', 'data', '/data', False, None), + ('bind', '/etc/truf/runtime', '/data/config', True, False), + ('bind', '/etc/truf/worker-packages', '/data/worker-packages', True, False), + ( + 'bind', '/var/lib/truf/runtime-document-candidates', + '/data/runtime-document-candidates', False, False, + ), + ('bind', '/run/truf/host-agent.sock', '/run/truf/host-agent.sock', True, False), + ( + 'bind', '/var/lib/truf/host-agent/results', + '/data/host-agent-results', True, False, + ), + ('bind', '/run/truf-postgres', '/run/truf-postgres', False, False), +) + + +class InstallError(RuntimeError): + def __init__(self, category): + self.category = str(category) + super().__init__('host-agent deployment validation failed') + + +def _deployment_profile(): + descriptor = None + try: + descriptor = os.open( + DEPLOYMENT_PROFILE, + os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) + | getattr(os, 'O_NOFOLLOW', 0), + ) + before = os.fstat(descriptor) + if ( + not stat.S_ISREG(before.st_mode) or before.st_uid != 0 + or before.st_gid != 0 or stat.S_IMODE(before.st_mode) != 0o444 + or before.st_nlink != 1 or before.st_size > 64 + ): + raise InstallError('profile') + payload = os.read(descriptor, 65) + after = os.fstat(descriptor) + if ( + len(payload) > 64 or before.st_dev != after.st_dev + or before.st_ino != after.st_ino or before.st_mode != after.st_mode + or before.st_uid != after.st_uid or before.st_gid != after.st_gid + or before.st_nlink != after.st_nlink or before.st_size != after.st_size + or before.st_mtime_ns != after.st_mtime_ns + ): + raise InstallError('profile') + except FileNotFoundError: + return STANDALONE_PROFILE + except InstallError: + raise + except OSError: + raise InstallError('profile') from None + finally: + if descriptor is not None: + os.close(descriptor) + try: + name = payload.decode('ascii').strip() + except UnicodeDecodeError: + raise InstallError('profile') from None + profiles = { + STANDALONE_PROFILE['name']: STANDALONE_PROFILE, + SHARED_HOST_PROFILE['name']: SHARED_HOST_PROFILE, + } + if name not in profiles: + raise InstallError('profile') + return profiles[name] + + +def _run(command, timeout=120): + try: + subprocess.run( + tuple(command), stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, cwd='/', env=FIXED_ENV, shell=False, + timeout=timeout, check=True, + ) + except Exception: + raise InstallError('command') from None + + +def _capture(command, timeout=120): + try: + result = subprocess.run( + tuple(command), stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, cwd='/', env=FIXED_ENV, shell=False, + timeout=timeout, check=True, + ) + if len(result.stdout) > MAX_COMPOSE_OUTPUT_BYTES: + raise InstallError('command') + return result.stdout + except InstallError: + raise + except Exception: + raise InstallError('command') from None + + +def _unit_active(unit): + if unit not in UNITS: + raise InstallError('command') + try: + result = subprocess.run( + ('/usr/bin/systemctl', 'is-active', '--quiet', unit), + stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, cwd='/', env=FIXED_ENV, shell=False, + timeout=120, check=False, + ) + except Exception: + raise InstallError('command') from None + if result.returncode not in (0, 3, 4): + raise InstallError('command') + return result.returncode == 0 + + +def _validate_compose_projection(payload, profile=None): + profile = profile or STANDALONE_PROFILE + try: + projection = json.loads(payload) + if type(projection) is not dict or projection.get('name') != 'truf-docker': + raise InstallError('compose') + volume_definitions = projection.get('volumes') + if ( + type(volume_definitions) is not dict + or volume_definitions.get('data') != profile['data_volume'] + ): + raise InstallError('compose') + services = projection.get('services') + runtime = services.get('runtime') if type(services) is dict else None + edge = services.get('edge') if type(services) is dict else None + if ( + type(runtime) is not dict or type(edge) is not dict + or runtime.get('network_mode') != profile['runtime_network'] + or runtime.get('ports') != profile['runtime_ports'] + or runtime.get('cpus') != profile['runtime_cpus'] + or runtime.get('mem_limit') != profile['runtime_mem_limit'] + or edge.get('network_mode') != 'service:runtime' + or edge.get('cap_add') != profile['edge_cap_add'] + or edge.get('image') != 'truf-local:edge' + ): + raise InstallError('compose') + mounts = runtime.get('volumes') if type(runtime) is dict else None + if type(mounts) is not list: + raise InstallError('compose') + observed = [] + for mount in mounts: + if type(mount) is not dict or type(mount.get('read_only', False)) is not bool: + raise InstallError('compose') + kind = mount.get('type') + if kind == 'volume': + if set(mount) - {'type', 'source', 'target', 'read_only'}: + raise InstallError('compose') + create_host_path = None + elif kind == 'bind': + if set(mount) - {'type', 'source', 'target', 'read_only', 'bind'}: + raise InstallError('compose') + binding = mount.get('bind') + if binding not in ({}, {'create_host_path': False}): + raise InstallError('compose') + create_host_path = False + else: + raise InstallError('compose') + observed.append(( + kind, mount.get('source'), mount.get('target'), + mount.get('read_only', False), create_host_path, + )) + if tuple(observed) != EXPECTED_RUNTIME_MOUNTS: + raise InstallError('compose') + except InstallError: + raise + except Exception: + raise InstallError('compose') from None + + +def _details(path, *, directory, uid, gid, mode): + try: + value = os.stat(path, follow_symlinks=False) + except OSError: + raise InstallError('metadata') from None + expected = stat.S_ISDIR if directory else stat.S_ISREG + if ( + not expected(value.st_mode) or value.st_uid != uid or value.st_gid != gid + or stat.S_IMODE(value.st_mode) != mode + or (not directory and value.st_nlink != 1) + ): + raise InstallError('metadata') + return value + + +def _read_source(path, maximum=MAX_COPY_BYTES): + descriptor = None + try: + descriptor = os.open( + path, os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) + | getattr(os, 'O_NOFOLLOW', 0), + ) + before = os.fstat(descriptor) + if ( + not stat.S_ISREG(before.st_mode) or before.st_nlink != 1 + or before.st_uid != 0 or stat.S_IMODE(before.st_mode) & 0o022 + ): + raise InstallError('source') + with os.fdopen(descriptor, 'rb') as handle: + descriptor = None + payload = handle.read(maximum + 1) + after = os.fstat(handle.fileno()) + if len(payload) > maximum or (before.st_dev, before.st_ino, before.st_size) != ( + after.st_dev, after.st_ino, after.st_size, + ): + raise InstallError('source') + return payload + except InstallError: + raise + except Exception: + raise InstallError('source') from None + finally: + if descriptor is not None: + os.close(descriptor) + + +def _secure_tree(path): + try: + root = os.stat(path, follow_symlinks=False) + if ( + not stat.S_ISDIR(root.st_mode) + or root.st_uid != 0 + or stat.S_IMODE(root.st_mode) & 0o022 + ): + raise InstallError('project') + for current, directories, files in os.walk(path, topdown=True, followlinks=False): + current_path = Path(current) + entries = ((name, True) for name in directories) + entries = tuple(entries) + tuple((name, False) for name in files) + current_details = os.stat(current_path, follow_symlinks=False) + if ( + not stat.S_ISDIR(current_details.st_mode) + or current_details.st_uid != 0 + or stat.S_IMODE(current_details.st_mode) & 0o022 + ): + raise InstallError('project') + for name, directory in entries: + details = os.stat(current_path / name, follow_symlinks=False) + expected = stat.S_ISDIR if directory else stat.S_ISREG + if ( + not expected(details.st_mode) + or details.st_uid != 0 + or stat.S_IMODE(details.st_mode) & 0o022 + or (not directory and details.st_nlink != 1) + ): + raise InstallError('project') + except InstallError: + raise + except Exception: + raise InstallError('project') from None + + +def _same_file(installed, source): + if not _read_source(installed) == _read_source(source): + raise InstallError('installed_content') + + +def _ensure_install_root(): + try: + INSTALL_ROOT.mkdir(mode=0o755) + except FileExistsError: + pass + except OSError: + raise InstallError('write') from None + _details(INSTALL_ROOT, directory=True, uid=0, gid=0, mode=0o755) + + +def _root_directory(path): + try: + details = os.stat(path, follow_symlinks=False) + except OSError: + raise InstallError('metadata') from None + if ( + not stat.S_ISDIR(details.st_mode) + or details.st_uid != 0 + or stat.S_IMODE(details.st_mode) & 0o022 + ): + raise InstallError('metadata') + + +def _secure_executable(path): + try: + link = os.lstat(path) + resolved = os.path.realpath(path) + target = os.stat(path) + parent = os.stat(Path(path).parent, follow_symlinks=False) + except OSError: + raise InstallError('executable') from None + if ( + link.st_uid != 0 + or not (stat.S_ISREG(link.st_mode) or stat.S_ISLNK(link.st_mode)) + or not resolved.startswith(('/usr/bin/', '/usr/sbin/')) + or not stat.S_ISREG(target.st_mode) or target.st_uid != 0 + or stat.S_IMODE(target.st_mode) & 0o022 + or not stat.S_ISDIR(parent.st_mode) or parent.st_uid != 0 + or stat.S_IMODE(parent.st_mode) & 0o022 + ): + raise InstallError('executable') + + +def _runtime_group_exists(): + if sys.platform != 'linux': + raise InstallError('group') + try: + import grp + named = grp.getgrnam(RUNTIME_GROUP) + numbered = grp.getgrgid(RUNTIME_GID) + except KeyError: + return False + except Exception: + raise InstallError('group') from None + if named.gr_gid != RUNTIME_GID or numbered.gr_name != RUNTIME_GROUP: + raise InstallError('group') + return True + + +def _ensure_runtime_group(): + if _runtime_group_exists(): + return + try: + import grp + grp.getgrgid(RUNTIME_GID) + except KeyError: + pass + except Exception: + raise InstallError('group') from None + else: + raise InstallError('group') + _secure_executable('/usr/sbin/groupadd') + _run(('/usr/sbin/groupadd', '--system', '--gid', str(RUNTIME_GID), RUNTIME_GROUP)) + if not _runtime_group_exists(): + raise InstallError('group') + + +def _validate_agent_socket(): + try: + details = os.stat(AGENT_SOCKET, follow_symlinks=False) + except OSError: + raise InstallError('socket') from None + if ( + not stat.S_ISSOCK(details.st_mode) + or details.st_uid != 0 + or details.st_gid != RUNTIME_GID + or stat.S_IMODE(details.st_mode) != 0o660 + ): + raise InstallError('socket') + + +def _write(path, payload, *, uid, gid, mode, replace): + temporary = path.parent / ('.' + path.name + '.truf-install') + descriptor = None + try: + try: + os.unlink(temporary) + except FileNotFoundError: + pass + descriptor = os.open( + temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL + | getattr(os, 'O_NOFOLLOW', 0), mode, + ) + os.fchmod(descriptor, mode) + os.fchown(descriptor, uid, gid) + view = memoryview(payload) + while view: + written = os.write(descriptor, view) + if written <= 0: + raise OSError('short write') + view = view[written:] + os.fsync(descriptor) + os.close(descriptor) + descriptor = None + if not replace and path.exists(): + os.unlink(temporary) + return + os.replace(temporary, path) + parent = os.open(path.parent, os.O_RDONLY | getattr(os, 'O_DIRECTORY', 0)) + try: + os.fsync(parent) + finally: + os.close(parent) + except Exception: + try: + os.unlink(temporary) + except OSError: + pass + raise InstallError('write') from None + finally: + if descriptor is not None: + os.close(descriptor) + payload = None + + +def validate(*, require_socket=True): + if sys.platform != 'linux' or not hasattr(os, 'geteuid') or os.geteuid() != 0: + raise InstallError('root') + _root_directory(PROJECT.parent) + _root_directory(INSTALL_ROOT.parent) + if not _runtime_group_exists(): + raise InstallError('group') + _details(PROJECT, directory=True, uid=0, gid=0, mode=0o755) + _details(DEPLOY, directory=True, uid=0, gid=0, mode=0o755) + _details(INSTALL_ROOT, directory=True, uid=0, gid=0, mode=0o755) + _secure_tree(PROJECT / 'app') + _details(ACTIVE, directory=True, uid=0, gid=0, mode=0o755) + _details(WORKER_PACKAGES, directory=True, uid=0, gid=0, mode=0o755) + _details(ACTIVE / 'config.yaml', directory=False, uid=RUNTIME_UID, gid=RUNTIME_GID, mode=0o600) + _details(ACTIVE / 'secrets.yaml', directory=False, uid=RUNTIME_UID, gid=RUNTIME_GID, mode=0o600) + layouts = ( + ('/var/lib/truf/runtime-document-candidates', RUNTIME_UID, RUNTIME_GID, 0o700), + ('/var/lib/truf/host-agent', 0, 0, 0o700), + ('/var/lib/truf/host-agent/backups', 0, 0, 0o700), + ('/var/lib/truf/host-agent/operations', 0, 0, 0o700), + ('/var/lib/truf/host-agent/results', 0, RUNTIME_GID, 0o750), + ('/run/truf-postgres', RUNTIME_UID, RUNTIME_GID, 0o700), + ) + for path, uid, gid, mode in layouts: + _details(Path(path), directory=True, uid=uid, gid=gid, mode=mode) + for executable in ( + '/usr/bin/python3', '/usr/bin/docker', '/usr/bin/systemctl', + '/usr/bin/systemd-analyze', '/usr/bin/systemd-tmpfiles', + '/usr/sbin/groupadd', + ): + _secure_executable(executable) + for unit in UNITS: + _details(SYSTEMD / unit, directory=False, uid=0, gid=0, mode=0o644) + _same_file(SYSTEMD / unit, DEPLOY / unit) + _details(TMPFILES, directory=False, uid=0, gid=0, mode=0o644) + _same_file(TMPFILES, DEPLOY / 'truf-host-agent.conf') + for script in SCRIPTS: + _details(INSTALL_ROOT / script, directory=False, uid=0, gid=0, mode=0o755) + _same_file(INSTALL_ROOT / script, DEPLOY / script) + profile = _deployment_profile() + for compose_file in profile['compose_files']: + _read_source(PROJECT / compose_file) + try: + docker_socket = os.stat('/run/docker.sock', follow_symlinks=False) + except OSError: + raise InstallError('socket') from None + if ( + not stat.S_ISSOCK(docker_socket.st_mode) + or docker_socket.st_uid != 0 + or stat.S_IMODE(docker_socket.st_mode) & 0o002 + ): + raise InstallError('socket') + if require_socket: + _validate_agent_socket() + _run(('/usr/bin/python3', '-I', '-B', '-c', 'import psycopg, yaml')) + _run(('/usr/bin/docker', 'compose', 'version')) + _run(('/usr/bin/systemd-analyze', 'verify', *(str(SYSTEMD / unit) for unit in UNITS))) + _validate_compose_projection( + _capture(_compose_config_command(profile)), profile, + ) + + +def install(): + if sys.platform != 'linux' or not hasattr(os, 'geteuid') or os.geteuid() != 0: + raise InstallError('root') + _ensure_runtime_group() + for unit in UNITS: + if _unit_active(unit): + _run(('/usr/bin/systemctl', 'stop', unit)) + _ensure_install_root() + for script in SCRIPTS: + _write(INSTALL_ROOT / script, _read_source(DEPLOY / script), uid=0, gid=0, mode=0o755, replace=True) + for unit in UNITS: + _write(SYSTEMD / unit, _read_source(DEPLOY / unit), uid=0, gid=0, mode=0o644, replace=True) + _write(TMPFILES, _read_source(DEPLOY / 'truf-host-agent.conf'), uid=0, gid=0, mode=0o644, replace=True) + _run(('/usr/bin/systemd-tmpfiles', '--create', str(TMPFILES))) + _write( + ACTIVE / 'config.yaml', _read_source(PROJECT / 'app/config.linux.yaml'), + uid=RUNTIME_UID, gid=RUNTIME_GID, mode=0o600, replace=False, + ) + _write( + ACTIVE / 'secrets.yaml', b'{}\n', uid=RUNTIME_UID, gid=RUNTIME_GID, + mode=0o600, replace=False, + ) + validate(require_socket=False) + _run(('/usr/bin/systemctl', 'daemon-reload')) + _run(('/usr/bin/systemctl', 'enable', 'truf-host-agent.socket')) + _run(('/usr/bin/systemctl', 'restart', 'truf-host-agent.socket')) + _validate_agent_socket() + + +def main(): + if len(sys.argv) != 2 or sys.argv[1] not in ('install', 'validate'): + raise InstallError('arguments') + install() if sys.argv[1] == 'install' else validate() + return 0 + + +if __name__ == '__main__': + try: + result = main() + except Exception as exc: + print( + 'host-agent deployment failed (' + type(exc).__name__ + '); details withheld', + file=sys.stderr, + ) + result = 1 + raise SystemExit(result) diff --git a/deploy/systemd/truf-caddy-admin-denylist-expire.service b/deploy/systemd/truf-caddy-admin-denylist-expire.service new file mode 100644 index 0000000..7ba2344 --- /dev/null +++ b/deploy/systemd/truf-caddy-admin-denylist-expire.service @@ -0,0 +1,18 @@ +[Unit] +Description=Expire Truf Caddy admin-only denylist entries +After=docker.service +Requires=docker.service + +[Service] +Type=oneshot +User=root +Group=root +ExecStart=/usr/local/sbin/truf-caddy-admin-denylist expire +NoNewPrivileges=true +PrivateTmp=true +ProtectHome=true +ProtectSystem=strict +ReadWritePaths=/var/lib/truf-edge /etc/truf-edge/denylist +RestrictAddressFamilies=AF_UNIX +LockPersonality=true +MemoryDenyWriteExecute=true diff --git a/deploy/systemd/truf-caddy-admin-denylist-expire.timer b/deploy/systemd/truf-caddy-admin-denylist-expire.timer new file mode 100644 index 0000000..bb109d7 --- /dev/null +++ b/deploy/systemd/truf-caddy-admin-denylist-expire.timer @@ -0,0 +1,11 @@ +[Unit] +Description=Expire Truf Caddy admin-only denylist entries every minute + +[Timer] +OnBootSec=1min +OnUnitActiveSec=1min +Persistent=true +AccuracySec=10s + +[Install] +WantedBy=timers.target diff --git a/deploy/worker/compose.yaml b/deploy/worker/compose.yaml new file mode 100644 index 0000000..7206a42 --- /dev/null +++ b/deploy/worker/compose.yaml @@ -0,0 +1,27 @@ +name: truf-worker + +services: + worker: + image: truf-remote-worker:linux-x86_64 + container_name: truf-worker + restart: unless-stopped + read_only: true + cap_drop: + - ALL + security_opt: + - no-new-privileges:true + pids_limit: 256 + stop_grace_period: 10m + tmpfs: + - /tmp:rw,nosuid,nodev,noexec,size=128m,mode=1777 + environment: + XDG_DATA_HOME: /data/client + XDG_STATE_HOME: /data/state-base + volumes: + - truf-worker-data:/data + command: + - run + +volumes: + truf-worker-data: + name: truf-worker-data diff --git a/deploy/worker/workerctl.ps1 b/deploy/worker/workerctl.ps1 new file mode 100644 index 0000000..33536c5 --- /dev/null +++ b/deploy/worker/workerctl.ps1 @@ -0,0 +1,3 @@ +$ErrorActionPreference = 'Stop' +& docker compose exec worker /opt/truf-worker/truf-worker @args +exit $LASTEXITCODE diff --git a/deploy/worker/workerctl.sh b/deploy/worker/workerctl.sh new file mode 100644 index 0000000..627351a --- /dev/null +++ b/deploy/worker/workerctl.sh @@ -0,0 +1,3 @@ +#!/bin/sh +set -eu +exec docker compose exec worker /opt/truf-worker/truf-worker "$@" diff --git a/deploy_capacity50.ps1 b/deploy_capacity50.ps1 new file mode 100644 index 0000000..10dc2a6 --- /dev/null +++ b/deploy_capacity50.ps1 @@ -0,0 +1,88 @@ +[CmdletBinding()] +param( + [ValidateSet('Plan', 'Apply')] + [string]$Mode = 'Plan', + [string]$ServerHost = '2.27.25.56', + [string]$RemoteUser = 'root', + [string]$WslDistribution = 'Ubuntu-24.04', + [switch]$PrepareOnly +) + +$ErrorActionPreference = 'Stop' +Set-StrictMode -Version Latest + +if ($ServerHost -notmatch '^[A-Za-z0-9.-]+$' -or $RemoteUser -notmatch '^[a-z_][a-z0-9_-]*$') { + throw 'Invalid SSH target' +} + +$projectRoot = $PSScriptRoot +$releaseRoot = Join-Path $projectRoot 'build\capacity50-release' +$payloadRoot = Join-Path $releaseRoot 'payload\app' +$deploymentRoot = Join-Path $projectRoot 'deploy\capacity50' +$sourceNames = @( + 'capacity_model.py', + 'scanner_db.py', + 'worker_assignment.py', + 'worker_api.py', + 'jsonl_projector.py', + 'runtime_document.py', + 'lifecycle_authority.py', + 'config.linux.yaml' +) + +if (Test-Path -LiteralPath $releaseRoot) { + Remove-Item -LiteralPath $releaseRoot -Recurse -Force +} +New-Item -ItemType Directory -Path $payloadRoot -Force | Out-Null +foreach ($name in $sourceNames) { + $source = Join-Path $projectRoot "app\$name" + if (-not (Test-Path -LiteralPath $source -PathType Leaf)) { + throw "Missing release source: $source" + } + Copy-Item -LiteralPath $source -Destination (Join-Path $payloadRoot $name) +} +foreach ($name in @( + 'Dockerfile', + 'deploy.sh', + 'render_config.py', + 'release_stopped_pipeline_leases.py' +)) { + Copy-Item -LiteralPath (Join-Path $deploymentRoot $name) -Destination (Join-Path $releaseRoot $name) +} + +$checksumLines = Get-ChildItem -LiteralPath $releaseRoot -File -Recurse | + Where-Object Name -ne 'checksums.sha256' | + Sort-Object FullName | + ForEach-Object { + $relative = $_.FullName.Substring($releaseRoot.Length + 1).Replace('\', '/') + $hash = (Get-FileHash -LiteralPath $_.FullName -Algorithm SHA256).Hash.ToLowerInvariant() + "$hash $relative" + } +[IO.File]::WriteAllLines( + (Join-Path $releaseRoot 'checksums.sha256'), + [string[]]$checksumLines, + [Text.UTF8Encoding]::new($false) +) + +Write-Output "Prepared release bundle: $releaseRoot" +if ($PrepareOnly) { + return +} + +$wslPath = (& wsl.exe -d $WslDistribution -- wslpath -a ($releaseRoot -replace '\', '/')).Trim() +if (-not $wslPath.StartsWith('/')) { + throw 'Could not resolve the release path inside WSL' +} +$remoteMode = $Mode.ToLowerInvariant() +$remoteCommand = "set -eu; install -d -m 0700 /var/lib/truf-deploy/stage; stage=`$(mktemp -d /var/lib/truf-deploy/stage/capacity50.XXXXXXXX); trap `"rm -rf `$stage`" EXIT; tar -xf - -C `"`$stage`"; cd `"`$stage`"; sha256sum -c checksums.sha256; bash deploy.sh $remoteMode `"`$stage`"" +if ($wslPath.Contains("'")) { + throw 'Release path cannot contain an apostrophe' +} +$bashPath = "'$wslPath'" +$target = "$RemoteUser@$ServerHost" +$bashCommand = "tar -C $bashPath -cf - . | ssh -o StrictHostKeyChecking=yes $target '$remoteCommand'" + +& wsl.exe -d $WslDistribution -- bash -lc $bashCommand +if ($LASTEXITCODE -ne 0) { + throw "Remote deployment failed with exit code $LASTEXITCODE" +} diff --git a/docker-compose.postgres.yml b/docker-compose.postgres.yml new file mode 100644 index 0000000..6068f2a --- /dev/null +++ b/docker-compose.postgres.yml @@ -0,0 +1,40 @@ +# NONCANONICAL MANUAL RECOVERY FIXTURE ONLY. +# The supervisor-owned PostgreSQL under D:/truf/runtime/postgres is authoritative. +name: truf-noncanonical-recovery + +services: + postgres-noncanonical-recovery: + profiles: ["noncanonical-manual-recovery"] + image: postgres:16 + container_name: truf-postgres-noncanonical-recovery + restart: "no" + environment: + POSTGRES_DB: ${TRUF_POSTGRES_DB:-truf} + POSTGRES_USER: ${TRUF_POSTGRES_USER:-truf} + POSTGRES_PASSWORD: ${TRUF_POSTGRES_PASSWORD:?set TRUF_POSTGRES_PASSWORD in .env.postgres} + PGDATA: /var/lib/postgresql/data/pgdata + ports: + - "127.0.0.1:${TRUF_DOCKER_POSTGRES_PORT:?set an explicit unused noncanonical recovery port}:5432" + volumes: + - D:/truf/runtime/postgres-docker-noncanonical:/var/lib/postgresql/data + - D:/truf/runtime/postgres-backups:/backups + command: + - postgres + - -c + - wal_compression=on + - -c + - max_connections=${TRUF_POSTGRES_MAX_CONNECTIONS:-100} + - -c + - shared_buffers=${TRUF_POSTGRES_SHARED_BUFFERS:-512MB} + - -c + - effective_cache_size=${TRUF_POSTGRES_EFFECTIVE_CACHE_SIZE:-2GB} + - -c + - checkpoint_timeout=${TRUF_POSTGRES_CHECKPOINT_TIMEOUT:-15min} + - -c + - log_min_duration_statement=${TRUF_POSTGRES_LOG_MIN_DURATION_STATEMENT:-2000} + healthcheck: + test: ["CMD-SHELL", "pg_isready -U ${TRUF_POSTGRES_USER:-truf} -d ${TRUF_POSTGRES_DB:-truf}"] + interval: 10s + timeout: 5s + retries: 10 + start_period: 20s diff --git a/docker/Dockerfile.edge-e2e b/docker/Dockerfile.edge-e2e new file mode 100644 index 0000000..ff88146 --- /dev/null +++ b/docker/Dockerfile.edge-e2e @@ -0,0 +1,5 @@ +FROM truf-worker-test:test AS edge-e2e-runtime + +USER 0:0 +COPY --chown=10001:10001 --chmod=0600 tests/edge_e2e_backend.py tests/edge_e2e_client.py /opt/truf/tests/ +USER 10001:10001 diff --git a/docker/build-dependencies/README.md b/docker/build-dependencies/README.md new file mode 100644 index 0000000..31ce79f --- /dev/null +++ b/docker/build-dependencies/README.md @@ -0,0 +1,273 @@ +# Runtime Dependency Build + +This directory records the public build inputs and pip-tools generator. It is not +an application configuration directory and must never contain credentials. + +## Pins + +| Input | Pin | +| --- | --- | +| Python image | `python:3.12-slim-bookworm@sha256:782412e85d0f0984994c290652577d4018aff08145c85b262bb63dc0c7522254` | +| Python version | `3.12.14` | +| Linux amd64 manifest | `sha256:9c47360a2a0355e2da18516d0b1c2126ec22c195d2185e97347c9d98398c5bef` | +| Linux arm64/v8 manifest | `sha256:d04f49f5882f49a3b91f874e75e19f0c265f7222da8659741a9d7eab148f22a9` | +| Debian and Debian security snapshots | `20260914T000000Z` | +| Git and git-man | `1:2.39.5-0+deb12u3` | +| tini | `0.19.0-1+b3` | +| CA certificates (already present in the pinned base) | `20250419~deb12u1` | +| PGDG server, client, libpq | `16.15-1.pgdg12+2` | +| PGDG common and client-common | `293.pgdg12+1` | +| PGDG signing-key fingerprint | `B97B0AFCAA1A47F044F244A07FCC7D46ACCC4CF8` | +| PGDG signing-key SHA-256 | `0144068502a1eddd2a0280ede10ef607d1ec592ce819940991203941564e8e76` | +| TruffleHog | `3.97.4` | +| TruffleHog Linux amd64 archive SHA-256 | `dc24007c2f233bd61c05beabeb44aa27ea9b43288166279209abe0458c5ce76b` | +| TruffleHog Linux amd64 archive size | `34970205` bytes | +| TruffleHog Linux arm64 archive SHA-256 | `7e65e771d2a247964056aa5edba0f8ae3945895e5dce867fe0ffbc7b0128239a` | +| TruffleHog Windows amd64 archive SHA-256 | `6ce9a957ac62bfb19463048333d9e8481327dbbf5bdc0c43f5ab5327b9631fb9` | +| Windows embeddable Python | `3.12.10`, SHA-256 `4acbed6dd1c744b0376e3b1cf57ce906f9dc9e95e68824584c8099a63025a3c3` | +| Windows MinGit | `2.47.1.windows.1`, SHA-256 `50b04b55425b5c465d076cdb184f63a0cd0f86f6ec8bb4d5860114a713d2c29a` | +| pip-tools | `7.6.1` | +| Generator pip | `26.2.1` | +| pytest | `8.4.2` | +| httpx (test target only) | `0.28.1` | +| zstandard | `0.23.0` | +| pandas | `3.0.5` | +| plotly | `7.0.0` | +| streamlit | `1.63.0` | +| psycopg and psycopg-binary | `3.3.5` | +| boto3 and botocore | `1.43.94` | + +TruffleHog's expected hashes were checked against the public release's +[`trufflehog_3.97.4_checksums.txt`](https://github.com/trufflesecurity/trufflehog/releases/download/v3.97.4/trufflehog_3.97.4_checksums.txt). +The PGDG key's primary OpenPGP fingerprint was independently calculated from the +hash-pinned public key and matches the fingerprint above. + +The Dockerfile pins the base index, download digests, PGDG package versions, and +Debian snapshot. Apt verifies Debian signatures with its shipped archive keyring +and PGDG signatures with the separately hash-pinned, repository-scoped armored +key. Full GnuPG is not installed. Expired `Valid-Until` checks are disabled only +for the immutable Debian snapshots, never signature verification. The official +PGDG archive retains older package versions; apt preferences exclude every PGDG +package except the five exact pins above. + +The complete Debian Git package and HTTPS helper are retained. Native executable +symlinks in the Python/Git tool directories, and PG client version-wrapper links, +are replaced with root-owned regular hard links. Dependencies stay in the +interpreter's `/usr/local/lib/python3.12/site-packages`, not a virtual environment +or user site. Installation requires hashes and binary wheels and disables +bytecode generation. The application lock includes both manifests, optional +dashboard packages, and pytest in one environment. The test target layers its +separate hash lock containing Starlette TestClient's `httpx` dependency; the +runtime target does not contain `httpx`. The separate worker lock contains only +Requests, PyYAML, zstandard, and their four transitives; it excludes PostgreSQL, +server, dashboard, test, and detailed-keycheck dependencies. + +`docker/worker-package-pins.json` is the canonical public build-input record for +Linux and Windows worker artifacts. Worker manifests hash every application, +dependency, scanner, detector-policy, Python-runtime, and complete Git-runtime +file. The package grants no provider or detailed keycheck authority and contains +no server or database credentials. + +PostgreSQL service starts and automatic cluster creation are disabled during +installation. No database is initialized by the build. Generated distribution +snakeoil TLS keys are removed in the same layer. Package pins make dependency +selection repeatable; this is not a claim of byte-for-byte identical OCI images +across BuildKit versions or package-maintainer timestamp generation. + +## Generate Locks + +All four requirements lock files are generated by pip-tools on Linux with the pinned +Python 3.12.14 interpreter. Do not edit any generated lock by hand. The initial +compiler bootstrap installs only public `pip==26.2.1` and `pip-tools==7.6.1`, then +resolves and hashes the compiler's entire dependency closure. Subsequent compiler +installs use that generated hash lock. + +The generator also copies the existing locks, so normal regeneration retains +valid pins. Use pip-compile's `--upgrade` only for an intentional dependency +refresh, then rebuild and revalidate the dependency image. + +Run these Docker commands from the isolated source checkout (`D:\truf-workers` for +this change), using WSL's +`sudo -n docker -H unix:///var/run/docker.sock` in place of `docker` on this host. +The generator receives only the three public application manifests and compiler +inputs. There are no host bind mounts. + +```sh +docker build --target lock-generator -t truf-lock-generator:py3.12.14 . +docker run --name truf-runtime-lock truf-lock-generator:py3.12.14 +docker cp truf-runtime-lock:/src/docker/requirements.lock docker/requirements.lock +docker rm truf-runtime-lock +docker run --name truf-test-lock truf-lock-generator:py3.12.14 \ + python3 -m piptools compile --generate-hashes --allow-unsafe \ + --resolver=backtracking --strip-extras --no-emit-index-url \ + --no-emit-trusted-host --index-url=https://pypi.org/simple \ + --pip-args=--only-binary=:all: \ + --output-file=docker/requirements-test.lock docker/requirements-test.in +docker cp truf-test-lock:/src/docker/requirements-test.lock docker/requirements-test.lock +docker rm truf-test-lock +docker run --name truf-worker-lock truf-lock-generator:py3.12.14 \ + python3 -m piptools compile --generate-hashes --allow-unsafe \ + --resolver=backtracking --strip-extras --no-emit-index-url \ + --no-emit-trusted-host --index-url=https://pypi.org/simple \ + --pip-args=--only-binary=:all: \ + --output-file=docker/requirements-worker.lock docker/requirements-worker.in +docker cp truf-worker-lock:/src/docker/requirements-worker.lock docker/requirements-worker.lock +docker rm truf-worker-lock +docker run --name truf-compiler-lock truf-lock-generator:py3.12.14 \ + python3 -m piptools compile --generate-hashes --allow-unsafe \ + --resolver=backtracking --strip-extras --no-emit-index-url \ + --no-emit-trusted-host --index-url=https://pypi.org/simple \ + --pip-args=--only-binary=:all: \ + --output-file=docker/build-dependencies/requirements.lock \ + docker/build-dependencies/requirements.in +docker cp truf-compiler-lock:/src/docker/build-dependencies/requirements.lock docker/build-dependencies/requirements.lock +docker rm truf-compiler-lock +``` + +## Build Targets + +```sh +docker build --target dependencies -t truf-dependencies:py3.12.14-pg16.15-th3.97.4 . +docker build --target runtime -t truf-runtime:local . +docker build --target test -t truf-test:local . +docker build --target worker -t truf-remote-worker:linux-x86_64 . +``` + +## Remote Worker Artifacts + +The `worker` target is independent of the server runtime. It uses the pinned image +Python and includes the worker authority, required shared DB-free modules, worker +dependency lock, detector policy, TruffleHog, CA roots, tini, and a package-local +complete Git helper tree. The final build executes the scanner and Git version +checks and verifies the full schema-3/protocol-2 capability package as unprivileged UID 10001. PostgreSQL +tools, server runtime/control authority, test code, `httpx`, provider and detailed +keycheck authority, server credentials, and database credentials are absent. + +Build the Windows amd64 portable directory and deterministic ZIP from the same +isolated checkout: + +```powershell +New-Item -ItemType Directory -Path dist -Force | Out-Null +python -B app/worker_package_builder.py windows ` + --project-root . ` + --output dist/truf-worker-windows-x86_64 ` + --archive dist/truf-worker-windows-x86_64.zip ` + --cache build/worker-cache +``` + +The builder downloads the hash-and-size-pinned Python 3.12.10 embeddable runtime, +MinGit 2.47.1, and TruffleHog 3.97.4, installs the cross-platform worker lock with +hash checking, verifies the complete package, and writes adjacent release JSON. +After extracting the ZIP, run `prepare-worker.ps1` once to replace inherited ACLs, +then use `run-worker.cmd`. Local files rely on private OS ACLs rather than +application-layer encryption; the client has no TLS-verification bypass. + +Build a complete distributable release (Windows ZIP, Linux image archive, +Docker bundle, manifests, README, Compose file, and SHA-256 lists) with one +PowerShell command: + +```powershell +.\build_worker_release.ps1 -ReleaseName release-YYYYMMDD-vN +``` + +The script uses native Docker when available and otherwise uses the Docker Engine +in `Ubuntu-24.04` through WSL. Use `-WslDistro NAME` for another distribution. +The destination must not already exist; a failed build remains on disk for +inspection and is never published as a partial replacement. + +While the main context allowlist is pending, the dependency build can use this +explicit two-file context from WSL. It does not send any application code, secret +files, findings, state, or original checkout directories to the builder: + +```sh +tar -C /mnt/d/truf-docker -cf - Dockerfile docker/requirements.lock \ + | sudo -n docker -H unix:///var/run/docker.sock build --file Dockerfile \ + --target dependencies -t truf-dependencies:py3.12.14-pg16.15-th3.97.4 - +``` + +`dependencies` stops before copying application code. `runtime-base` holds the +production filesystem and launch configuration; `test` inherits those exact +contents and adds private test sources. The last/default target, `runtime`, is a +direct alias of `runtime-base` and contains no test tree. The test entrypoint uses +the same interpreter isolation flags and the real `child_bootstrap.py` dependency +path loader, without processing `.pth` files or starting the application. + +Application and test copies are owned by `10001:10001`; their directories are +`0700` and regular files `0600`. Build-time checks reject symlinks, special files, +and cached bytecode in these controlled copies. No recursive permission changes +are made to host paths or runtime-mounted data. The default user is `10001:10001`; +the image precreates private `/data` and `/data/home` directories. + +## Verified Results + +Validated on 2026-09-15 using Ubuntu 24.04 under WSL2, stock Docker Engine 29.8.0, +the local Unix socket, and `sudo -n`. Builds and generation were polled without +printing full dependency logs. A temporary WSL keepalive prevented idle shutdown +during detached lock generation; no host configuration changes were made. + +| Result | Value | +| --- | --- | +| Dependency image tag | `truf-dependencies:py3.12.14-pg16.15-th3.97.4` | +| Local dependency image ID | `sha256:3fbdc2ea6fd1199aa742f54fb653e433d79ad3f1a5e4bfeed8b413dc9f704eb5` | +| Built and executed platform | `linux/amd64` | +| Runtime lock | 49 exact package pins, 1059 SHA-256 wheel hashes | +| Runtime lock SHA-256 | `82c69394b116fd8a762c3dea481682045af87c013974fcb78ac0529892188537` | +| Compiler lock | 8 exact package pins, 8 SHA-256 wheel hashes | +| Compiler lock SHA-256 | `0172b08004c6f2b0702ea9a472300cc63243492df2d3d782fd4dbaa612497fd2` | +| Lock replay in the hash-locked generator image | Both files byte-for-byte identical | +| `pip check` | No broken requirements | + +- The `dependencies` and `lock-generator` targets built successfully with hash-required, binary-wheel-only installs. +- All 49 package import checks passed normally, then with `-I -S -B` through the actual `child_bootstrap.py` loader for all 10 child kinds. Application entrypoints and providers were not executed. +- Isolated `sys.path` contained only the interpreter ZIP path, standard library, `lib-dynload`, and `/usr/local/lib/python3.12/site-packages`. Neither the working directory nor the user site was added. +- 14,202 dependency-tree permission checks found root ownership, no group/world write access, and no symlinks or bytecode files. The unprivileged UID could not write these dependencies. +- Eleven selected native executables were regular root-owned files. Version and `ldd` checks passed for Python, Git, the Git HTTPS helper, tini, TruffleHog, and the six PG16 tools. TruffleHog is static; other checked ELF binaries had no missing shared libraries. +- TruffleHog global, Git, filesystem, Docker, and Hugging Face help flags were checked, including the scanner's legacy `--local-dev` and `--log-level` options. No scans or provider requests were made. +- No PostgreSQL `PG_VERSION` file or initialized cluster was present. No PostgreSQL server, application, or provider was started. Full GnuPG is absent. +- Minimal bootstrap-only runtime and test fixtures exercised the actual copy stages: `10001:10001`, directories `0700`, files `0600`. Generated in-memory contexts containing a symlink or `.pyc` file were rejected by the build. +- The test target's real tini/Python/bootstrap entrypoint reported `pytest 8.4.2` with networking disabled, a read-only root, and a UID-10001 private `/tmp` tmpfs. Pytest capture needs writable temporary storage even for `--version`. + +The temporary runtime/test fixtures were deliberately incomplete and were used +only for dependency and filesystem-policy verification. They are not deployable +application images. No application test suite, PostgreSQL initialization test, +provider integration, or arm64 build was run. The arm64 base and TruffleHog assets +are pinned, but that platform still requires a native or emulated build/test. + +## Integration Boundary + +The main integration owns `.dockerignore`, `app/container_runtime.py`, Compose, +and tests. The context allowlist must include the exact Dockerfile, dependency +input/lock paths, this metadata file, the new runtime entrypoint, and the intended +test files. Do not replace the deny-by-default context rules with directory-wide +or wildcard exceptions. Keep `.env`, credentials, findings, original runtime +directories, and generated caches excluded. + +New exact metadata/input exceptions required in the main-owned `.dockerignore`: + +```text +!docker/requirements.in +!docker/requirements.lock +!docker/build-dependencies/requirements.in +!docker/build-dependencies/requirements.lock +!docker/build-dependencies/README.md +``` + +The Dockerfile is already allowlisted. Main must separately allowlist its +`app/container_runtime.py` and intended test source files once they exist. The +test target copies only `app/` and `tests/`; repository tests that read root-level +Compose, Docker, or PowerShell fixtures still need main-owned fixture integration. + +Compose must select the runtime target, enforce a read-only root filesystem, +provide newly initialized Linux writable volumes and a temporary filesystem as +needed, and preserve the unprivileged UID/GID. Runtime startup, PostgreSQL +initialization, provider execution, and existing-data integration are explicitly +outside dependency-build validation. + +For a read-only test image, provide private temporary storage, for example +`--tmpfs /tmp:rw,nosuid,nodev,noexec,size=64m,mode=0700,uid=10001,gid=10001` for +version/import checks. Main must choose temporary-storage size and execution +policy appropriate for its full tests. After the entrypoint, allowlist, fixtures, +and Compose changes are integrated, rebuild complete `runtime` and `test` images +and perform the separately authorized integration checks. No files outside the +Dockerfile and `docker/` build inputs were edited, and nothing was staged, +committed, or pushed. diff --git a/docker/build-dependencies/requirements.in b/docker/build-dependencies/requirements.in new file mode 100644 index 0000000..37a303d --- /dev/null +++ b/docker/build-dependencies/requirements.in @@ -0,0 +1,3 @@ +# Toolchain only; these packages are not copied from the generator into runtime. +pip==26.2.1 +pip-tools==7.6.1 diff --git a/docker/build-dependencies/requirements.lock b/docker/build-dependencies/requirements.lock new file mode 100644 index 0000000..0bdc87c --- /dev/null +++ b/docker/build-dependencies/requirements.lock @@ -0,0 +1,40 @@ +# +# This file is autogenerated by pip-compile with Python 3.12 +# by the following command: +# +# See docker/build-dependencies/README.md for the pinned Python 3.12.14 pip-tools generation command. +# +--only-binary :all: + +build==1.6.1 \ + --hash=sha256:ecd351a4be9d35a9eaaba244a7687143c9c7d4aea6ac964e7e7ddab20cbcf4e7 + # via pip-tools +click==8.5.0 \ + --hash=sha256:255bc9599cf7748b4b1a446ccc735421bd08a2ae529a8b88597d3de5664ee360 + # via pip-tools +packaging==26.3 \ + --hash=sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c + # via + # build + # wheel +pip-tools==7.6.1 \ + --hash=sha256:6111c8b4b07fd14b7223ca921485b0e96cf66e20bf94da95eeed9845f510cb8f + # via -r docker/build-dependencies/requirements.in +pyproject-hooks==1.2.0 \ + --hash=sha256:9e5c6bfa8dcc30091c74b0cf803c81fdd29d94f01992a7707bc97babb1141913 + # via + # build + # pip-tools +wheel==0.48.0 \ + --hash=sha256:3217dcc807155e45db462d7ef2431f5ddda0d7273b700d05a67b271ceb1287ab + # via pip-tools + +# The following packages are considered to be unsafe in a requirements file: +pip==26.2.1 \ + --hash=sha256:71138adf1f4ca900cdb7d289c21b7494329f2332b6d85f0e1c42108c0384ed3e + # via + # -r docker/build-dependencies/requirements.in + # pip-tools +setuptools==84.0.0 \ + --hash=sha256:51a52592b3b99e102b609654876bd65f19f999935166d1352678931132b0c670 + # via pip-tools diff --git a/docker/requirements-test.in b/docker/requirements-test.in new file mode 100644 index 0000000..cee3237 --- /dev/null +++ b/docker/requirements-test.in @@ -0,0 +1 @@ +httpx==0.28.1 diff --git a/docker/requirements-test.lock b/docker/requirements-test.lock new file mode 100644 index 0000000..37e4268 --- /dev/null +++ b/docker/requirements-test.lock @@ -0,0 +1,33 @@ +# +# This file is autogenerated by pip-compile with Python 3.12 +# by the following command: +# +# See docker/build-dependencies/README.md for the pinned Python 3.12.14 pip-tools generation command. +# +--only-binary :all: + +anyio==4.15.1 \ + --hash=sha256:6152fdbbf9a77fdec97731721bebf7c4c44f7c29b424b0065826173efc7ed101 + # via httpx +certifi==2026.7.22 \ + --hash=sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775 + # via + # httpcore + # httpx +h11==0.16.0 \ + --hash=sha256:63cf8bbe7522de3bf65932fda1d9c2772064ffb3dae62d55932da54b31cb6c86 + # via httpcore +httpcore==1.0.9 \ + --hash=sha256:2d400746a40668fc9dec9810239072b40b4484b640a8c38fd654a024c7a1bf55 + # via httpx +httpx==0.28.1 \ + --hash=sha256:d909fcccc110f8c7faf814ca82a9a4d816bc5a6dbfea25d6591d6985b8ba59ad + # via -r docker/requirements-test.in +idna==3.20 \ + --hash=sha256:ab7ae7122974553370f0bdb919e1a960b2cd1bc1ef0276416d896db81c14582c + # via + # anyio + # httpx +typing-extensions==4.16.0 \ + --hash=sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8 + # via anyio diff --git a/docker/requirements-worker.in b/docker/requirements-worker.in new file mode 100644 index 0000000..1d58102 --- /dev/null +++ b/docker/requirements-worker.in @@ -0,0 +1,3 @@ +PyYAML==6.0.3 +requests==2.34.2 +zstandard==0.23.0 diff --git a/docker/requirements-worker.lock b/docker/requirements-worker.lock new file mode 100644 index 0000000..7057e3f --- /dev/null +++ b/docker/requirements-worker.lock @@ -0,0 +1,365 @@ +# +# This file is autogenerated by pip-compile with Python 3.12 +# by the following command: +# +# See docker/build-dependencies/README.md for the pinned Python 3.12.14 pip-tools generation command. +# +--only-binary :all: + +certifi==2026.7.22 \ + --hash=sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775 + # via requests +charset-normalizer==3.5.1 \ + --hash=sha256:00668ebb0609751758682eb0b5857e7c35b9f00e84dfdef062e103244ec94d45 \ + --hash=sha256:012a22b88a77ca2e59b98ac5889b0deb604147666032f45e6d6e217634d2550d \ + --hash=sha256:01e93745f7f219b703b60ba7afead36cfc4242782be5af484673fc500df12da5 \ + --hash=sha256:04368edf83514385ffc3e1cfd4546e595f4f1272dd23ba437a93a9cc3741d47b \ + --hash=sha256:0722590aabf9dc6a6c0343d523c05458fa2b5047dbe6302fd526bb570600753f \ + --hash=sha256:07ffd07412fc5d5e84cd8952acf9ff7e4ed7a708e69d1bada19d8ba91711353f \ + --hash=sha256:09a7bba9f739468c8e78c36a75c33768e53cb1959fc638f510454c14683f00d5 \ + --hash=sha256:0b2b1b3fa5670c127b246df1d0c059defd41f689a868a3b9d79df9b1cac42d22 \ + --hash=sha256:0c6dfb5ca6723eeed15aa8e564a014d69fcb8812f94eef11fe3631e0508199f5 \ + --hash=sha256:0d929fc574b4d6fd9e7c0f5c2ede8716a41911923aa7fa5fce38e0818aa4a1ac \ + --hash=sha256:13e3afe97712e8887cd516e960c63f0b93122971e5b5e4b2622fe7701771e838 \ + --hash=sha256:15f024313246a4ed976c60f440bb8d257815513a681d212ff74fd46f7d715a90 \ + --hash=sha256:195ce897c6153c0700078142cf8efe3e6454ca4cf4357499e4078dfd83396626 \ + --hash=sha256:19a3dd5aa73cef1c99687c4fc57db016a9c17104ae1185da88ba566a5d3bebe4 \ + --hash=sha256:1d1c7a53a6c2103925cdd6d7229f8c567379f211c869793df679f2e9f738c369 \ + --hash=sha256:1f5883d77fd409a261abb5dc8ccbe335720d798b1de4abb3b1d47ccbbc76b53b \ + --hash=sha256:21b82d8082f6f5e7f456ef0bd16323d08de1266efbfeb476e64b2a91d1471a4e \ + --hash=sha256:252d099029bcbea642f2a06c4ed5046bdf8b5a8150b64afa5e027e88b106e5ee \ + --hash=sha256:256dd4d85d9e4dc595e2bc983c980e73f62ddeb3165c58b4c3dfe78c5c8548c1 \ + --hash=sha256:26422d45fd13551cf564c58932f7d72b4f58b93b0fcf18c35ba6be12b46bb102 \ + --hash=sha256:2679de311c7946dde5d3b6f44941844133ff5c7cb86099c0061ab1e8901c20a8 \ + --hash=sha256:29880d17a8eb0b5cfdfd8944b468322928059aa35f1f5fa8ff22b149ec0b42f8 \ + --hash=sha256:2bced4061f000f7187254a02ad3433ae17eaf991747ceea2f478422590a5bba9 \ + --hash=sha256:2e9cf9253119d8e5d111f05d71626786fd3d6193817316eab1ca088cdb8593cf \ + --hash=sha256:2f06b7eae9dbe77fe1d644ca244dad508de8d302870a43f3c559b521270938a0 \ + --hash=sha256:2f293479cce755c75f1697e87c409b7ae4c555c7dfecb6e988ad13abba943031 \ + --hash=sha256:329fc3ccb63ad22d867d84c2adea759a64079a37ba4a343433b02c7a2816871e \ + --hash=sha256:343fb4f2821043bd87095f7b08a1a181febc8e36ac64212143bbfd0a0e1bc235 \ + --hash=sha256:3588e376b3ea2eea84976f67273d679f229e24c66dce7b82ae45aef04ff6e072 \ + --hash=sha256:35aea775dc2bd5f54cd84a1cd2696cc3207c479cb9cf0bd346f0d343e4300ddb \ + --hash=sha256:35fe081843b35aad20ffeccec3eeffbe637b15d14f3fb22cc1b59cd8ec17e93c \ + --hash=sha256:36047af20e17097c3bb9476c2b7655f2f7aa51322c0ba58c07695bedf755a950 \ + --hash=sha256:3617ac3cfd8b9888f145ad89dd6e692285834b0201c6074a5eeaad3fd4d668c2 \ + --hash=sha256:366ec70f5547c640d3ce1985722490f23faf4eb5216a7eeba78277490e78dacb \ + --hash=sha256:394fea06235c8543390050ed5f529187074b029fb027213f6c46ac11ab5d950e \ + --hash=sha256:3d27167433c0d5f18dc850f07d0b3816221984fecdc405d6c157a6f0b8f8e9e6 \ + --hash=sha256:3e5e1224c0a6a90e05843e07adfec669edebec17801c67072f51e59561d63c0b \ + --hash=sha256:41876ee62a3dddf48ff1121ad8f0798032aa03f2fd35f21f34a4cab14f18d8d2 \ + --hash=sha256:433c5a81eade63b47e522303bad236f59dba55ea6951746f5558355eeed8c75d \ + --hash=sha256:4582c27e8c889d64811987b5967fbd3ae0c823fe1fd933b543d55ac20bb475fa \ + --hash=sha256:485a0d363cafefcd2538a73c7c838daa2035f09b2c9f9b5e3133f80c6aeb84c2 \ + --hash=sha256:494b70049a4d69aec6e8137c13af4cf8db8c9f9820a1392ac293b0dd2987a818 \ + --hash=sha256:496846868fea80e479324862fa877f02411f2fd0f83b79ccee2607aa68b2a032 \ + --hash=sha256:4abdc5f9ad448c1ecbfae2974b820535d6bc6e7eef63babbab3d81cf46968c71 \ + --hash=sha256:4b599739b93b2cbeded49645ae3c8d1405c29ddfbceac1545c87a3f9580a9e96 \ + --hash=sha256:4bea7f8ebe90bbd7f0e4a2de42ca6924ba23e3e76418c408ff82f1d46fabd687 \ + --hash=sha256:4c4fb141a727957c93edfe5c32a26ceb6b5f6461d67146e2d39f51e16170bea8 \ + --hash=sha256:4c9548dc78002099910abaebc0a72ac58b7d30931869e0351c09b507dff4ece3 \ + --hash=sha256:4d26f14f041e83dd8edfd61f4cd4fa7285d31798b5bf1f28e70c367ba6c41d61 \ + --hash=sha256:4f298bdadb8f0b9e5672877f647d1be9373ef5320c9e2f049795e26cad28b6a9 \ + --hash=sha256:52ec005752a56ae79547a05c0139ca2501a0c866390b6115008456b9f0e7cde1 \ + --hash=sha256:55261ac0d2941c42f196dd576f543d87a8ee03cd6f5e30dfb4d807b2e3b9121a \ + --hash=sha256:56490c595a28b1bb27dfc583e816152a9767721ef58b2c03b13f954d2f707420 \ + --hash=sha256:58d3e12c88e0950bca850ae1f7c256055c097639c2edb9eb123af9807d8b15e4 \ + --hash=sha256:58d4aa13a59c969dbfdf9e6a9560e242cbfd9e8a8f50c2747714df1a423adf65 \ + --hash=sha256:59171c6e45bf07d0d5cab3b0bf81d945035530f6873398b3b531c31184d46663 \ + --hash=sha256:5b6d1386bf0096d26d3a863dc0a487a5b4eb9aa93cf5ba69683d29dde6b9d60f \ + --hash=sha256:5c0ea61a470e070686aa30892fed79e297d2c8d0ab46b8bcdf027d38c51da591 \ + --hash=sha256:5c84bec0ab5ae0c64bfe73a7d2adcb5ce73b467523fc27fd6a28ab2aa6cbe35a \ + --hash=sha256:5ca0555312ae2fe82715cada7fac375530c2f3349e1eaa1bcb33d0283ac79a18 \ + --hash=sha256:5d8531a6569d025f68e2321e7638fb7978f23db58e5f69f56913837aae03816e \ + --hash=sha256:5e2d0e146dcb57034f8b97dc58d2d512cb90aba253960ce449f695fec6a82c6f \ + --hash=sha256:5fc45d653ea8c9a20479167e11d4a0f8cb2fa3470737ab6f9c827532313187b7 \ + --hash=sha256:6199d5606e2bbf2b096cf64d03f8b6790c91081d5ac866b8e7bb6422738cc60c \ + --hash=sha256:62b55f6722735a6c472f88361cde6640608773d9443cebdbb51abf436a1fcdd3 \ + --hash=sha256:687c9ca3035544b113bea2055e180af96fb63c0c476e22a9180f51925186e7b7 \ + --hash=sha256:6b7430cf5728e68f6c462254009a6ef4086e1bea43cf2f57aa9c55fb4f50ff96 \ + --hash=sha256:6ba32c4d2abf1d2fe7cf27d280f4cca5664233b0f885549c7761719eb977f486 \ + --hash=sha256:6c9cdde8becb25a7fde49924511aa2644d6f8081cc8df8e9452724303348d8e3 \ + --hash=sha256:6df0ec430f9a831772c23ca5a224cba36517a58a84bb32c32bb59a9fa67c47f6 \ + --hash=sha256:6e2912d4babbc65196ac13c2f53468dc57fb8b9c25ef913e8c59ddf7c6dc0e1b \ + --hash=sha256:6e5e4d73d588ca5ed09df1b7dcd1b203d1df3c542e3f50d126c947d432b10731 \ + --hash=sha256:70055ff39b97c99e7ae40ea3e393fb62aa2e44dbd9b29f8d14f42fb0025c3959 \ + --hash=sha256:706bfd38730a5ac7a365793269a00f4e988178cec121391f4248d84ad8c972e9 \ + --hash=sha256:7235dc28fc6dd9d832ac7c7bce95367dedb85929f17368a0c2bee1e080b9acbf \ + --hash=sha256:774d157f112367ff4abd29019f38f023c24e00e56edc7829c20e358a5a913ad8 \ + --hash=sha256:77efcff2b23071c349402ac1066667a3d011f62398d81408c9b88ad991747c9e \ + --hash=sha256:789b8982559ae28dad2356519f841655756cdcd96616410590ae0b17454ee64f \ + --hash=sha256:7ac76cf9afd34929d76eb7fcb63be476a4853d8a96f0dcf2d0db68a0cbdf9885 \ + --hash=sha256:7c0c10730342b0c9b35dd1d619beb8214e520bd96a1f870f452680b238aab3e0 \ + --hash=sha256:823f82903d189af463d7df250ef1f7f696f3cee08cc8d91deb565e8d425f6506 \ + --hash=sha256:838648accb3a7fd9803fd45c87bce8509648eb0c11bc34e216141300977244f2 \ + --hash=sha256:854066be00447fa8de2ccbbe893e2ffc4b123ef16d897af794c1e18bd4a714b0 \ + --hash=sha256:85d5855daafc240cc045c026d7a15fd198a09b0fc8ff6f5ecbb5297b509cb11e \ + --hash=sha256:85de3134b5379856e323ba37c19c9256d39425f7b76a63af52b09fb4664c2e8f \ + --hash=sha256:87e4f41d375c0b9be2fb5251aee4b8a689169e134535aed81bf085c3b647451e \ + --hash=sha256:88ca277405c2d3b71c4e1c2ee0e7966e807bcba86a69d11e19ba199d18ae4491 \ + --hash=sha256:88e85ab89cb822c1e635f51d6d32e488f94e002e70e2f492bdb8b945543f345a \ + --hash=sha256:8ac8c94b6539074e0f40899301273ac8402b9b3e01c7b7ba269ff30340aaaf20 \ + --hash=sha256:8fe532b3c966d1fb794e0698e4589d0444017ae77fc0b31edea13c0e35bcc449 \ + --hash=sha256:9085f87b0e38a2b92b8923059b4e8789fe40d9279712d15dcc670048d77079af \ + --hash=sha256:90b7481fb62fbe172c558bc6fd1c4c98d82004a54a7551f20e11ac9bf0b8708c \ + --hash=sha256:92caef967d287a407085d61176fce4012b1dd62daed4eb6d5ceb26d3d2538712 \ + --hash=sha256:9362dd90aa7dab48c0054a21187791ccf05473f7dba5d92b8033ae62164675e7 \ + --hash=sha256:94d78ecec2605a8d0398b0f365d5f12a63248438516f5dac536a5eff7337df4a \ + --hash=sha256:94fbf1c0c6cc0d3d5e50f9a9313a8cdca90dd696d34b381cd1704f8c9e939f20 \ + --hash=sha256:950f23cb393f85543777b0433f082cddd25b51ab398eac7971146495679efe5f \ + --hash=sha256:96eefc178f8636b9c760c5829345307fd81cfae9ab1e80997dbddeb0f54ee9a3 \ + --hash=sha256:96fef3e886d6a9874b14f27fc193fbdc69d5d8035783d86aa4e1cea594e695f9 \ + --hash=sha256:977cdbd483a9cff38179bea4fd754289a6f2195c7abd414aba85410b3e66cc5e \ + --hash=sha256:978eab16f55b4ab2c2a745be9a0a840bf8f09a7f227d9c76eb30214d078865a5 \ + --hash=sha256:994e883d17c559cdfd38c84003c8b27d25424a1077272a17e7cd27bfe0bf57b2 \ + --hash=sha256:9ac4444d8d4fd4c4bd08bf451ed3167aa9e7ec6cdb41b648794f1d1103652e36 \ + --hash=sha256:9b5db6052055d34d41230fb78d7c439c23dc536a9896f6cb039e8dd92cfc1263 \ + --hash=sha256:9d9a0dc7cbe9bec24c3f767c9122c41fe5a1bc43f47cd099d00d393e09769de4 \ + --hash=sha256:9dbdd9205662134957cf0c324f639bdc5031c0ca056e2369e238db75187c0f11 \ + --hash=sha256:9eea3ab2597a5e65fe65296e2d6a84570845a6b55532d90333d740d48bbc850a \ + --hash=sha256:a2028475ba855475b8b4d3cfeb4994269c967aea8b9892dfba907f4263a863a3 \ + --hash=sha256:a3a370082ce34d0612f421e15fe011c53bb1feff21a26d06ad4fb244dab5a375 \ + --hash=sha256:a545775cfe815855ea32d7c27731d79da358ef2055b4a25830231b1622dd18aa \ + --hash=sha256:a5cbd90ecf0fc62e64726917ad083b73001f0563657a87ec3c0b504e277dc90d \ + --hash=sha256:a6d095662e73e74f0a49988e0593373e243e3a52e27bfeea0a859e88acf4a0f5 \ + --hash=sha256:a6dac12ff6b846103483683f60c5f8fee205121adc58ffd87e90a90a3af69e99 \ + --hash=sha256:a951ad59cad9145664a730d3036b40b844e74d2d3683da40111463cd3a83845d \ + --hash=sha256:aa1099b956fb795e686d073568f6dc002a0bb89765ea6d5b055dd7d9bf1b116c \ + --hash=sha256:aa2bb0b37202dca27175591f761108b5d34096ade1191ffe4808bdf6b1571488 \ + --hash=sha256:aae2ee51122d3ae968a3837d97dc24a0aeebb0dea23694422cd172bd30017cd6 \ + --hash=sha256:ab743e9bc90c1f73552ec33e10e3331315acd2c397b36065b591b0181de533cc \ + --hash=sha256:ac00177c4831ffa650f8609e4bdddd5fe09c03b1c0c47acece7e6ea20421598b \ + --hash=sha256:ac13b004224fb341e1e25a1ed5e19d32f57cdb2a403e01f003b46f051a550f6f \ + --hash=sha256:acaf604462bf330b0d07e7a07c1d6e4adac79e5fb13e9c5140590542cafacc00 \ + --hash=sha256:ae31a1a1db2ee6cc2942fccaf695c934bc7f3db9f2133a3fef1f367cf1a4ab10 \ + --hash=sha256:ae4a097991662cd4fff0ddc74e0fe7874f82e00042fa0ea00855645ed0c79598 \ + --hash=sha256:aea996a6aba25260827c9ea511d1addfde2da9eb686ac961838509086188b7e6 \ + --hash=sha256:b39b69b347e5e47a3b5b8cfc005c68c1ba347474e3960236c4944a8ecd174962 \ + --hash=sha256:b54e7e13267d49ffbfe68e25b3cbd774dab38fa37238f71265e91b36146eb21c \ + --hash=sha256:b9af956078716df40d985fb0dfeb2c2120c5ca92ba4ff4b388acfd01cdc14d08 \ + --hash=sha256:ba2f37ee79e6338845261a3c5b1784e5d1acdff2c0785b284f1b633033d136ab \ + --hash=sha256:ba501e667c17d8411f98e67a022d9604ef179aff0e459b7e292c796837c13573 \ + --hash=sha256:baf3775a2635e5a11fbd5e4e64ee69c7e86875d224a5c72aca4c141064589a90 \ + --hash=sha256:bb57753e36e4855b8ca375069482250a6246372331a3e4f3407eaebb007443f5 \ + --hash=sha256:bd6c173f04743d483881bffa1478d5a4624475b8cd1d2194956a75548e191c18 \ + --hash=sha256:be47f99644b208bff7766314013f9acf57b056b04191d570d68ad14022cf5b1d \ + --hash=sha256:c010f5581d9c612804cc59fcf7b524b707fbcb72828551237ab545bb5c7034af \ + --hash=sha256:c1dcc36dcb96abc02236e182d17e0f71430152a6c2c7447421da2d2dc144edea \ + --hash=sha256:c428c6c31eb5f4277d7f8eccaf767fbd548ddd5ce3c8b4f4cbbfab3d96b5904c \ + --hash=sha256:c658c50ac0c98cd755a2dd50b7977d3bca7df401dcc47fbdfa87db53ef7d4e8b \ + --hash=sha256:c71fb0d56c920c269cd3e2e3fe7c610e3f1fdb21a6ce60efa6430ff63676cea6 \ + --hash=sha256:c7b742bf31c88566b4bb6335a7f393bb322e580b6bb98df7bd0c25e6e3519ce8 \ + --hash=sha256:cc0329df4caaceb950d2f580b5ac716a377f7059624a0bafaeaf8a218c6ed774 \ + --hash=sha256:cc5d36d96478aa9c60654bd932525bf32964c62a7281eafdf16d85003a8d6004 \ + --hash=sha256:ce854f5f478050ade5a238731c4ca985a7d3b3cb53ff600a9b5c3b689b5f0a7a \ + --hash=sha256:ced3fdd71aaa83ce593746c2edb42b7a59cb4c19c8b5c407781c72e493aae55a \ + --hash=sha256:cee5dd7c6fb5dd52a0fe2a740f9bc6e3593f5f8b1788bde49de02086f30182b2 \ + --hash=sha256:cfa1c0cc3a8f9f53f1243a5a99ac36fd003880199383b37672e86ddda9cb07e2 \ + --hash=sha256:d1ee1e296209fdce05b81b663250eefa02213a2da7b41bf26f7829b8ba3545aa \ + --hash=sha256:d59b75732e9b6f27388e10c14b0259cc5f2e48c78627d185e6a177b58ad3cffe \ + --hash=sha256:d63600d620ad0064c3a748b950ac5ea38a80190e5498532efefa4b7b3f1da1f3 \ + --hash=sha256:dd732602a7009217f658d5863d12d79d373a4de0eebc111094bcdd3bb8e0a6cc \ + --hash=sha256:e06efa066f7dbadbc84ebc126a97c452a6451dfcf589d89d788484949e1cf795 \ + --hash=sha256:e199fb99720074809a7720f1c0b4d919eea8b87e88713e0f8f602f7bef543d9d \ + --hash=sha256:e4b018dc5a0eee4676e38fe84a47a427816c590b93b55d9025274ec4d6ffc2dc \ + --hash=sha256:e6621fb2a4988d6e53eedc455e5903e2679f3967b8acb3d639f1b63c14a2e893 \ + --hash=sha256:e71c909f353863b2b89c83de2ebed71ea6d0df8a6ef65a128193c5e650766bef \ + --hash=sha256:e90251c0c7bdd54a100a0dce3c07b7e637278c93af29dbf78ebb89a58c4bac7d \ + --hash=sha256:e9fbdce1e47394b09bc9f26ab117dfc8d6491977a11d86f592bb42c779db2fda \ + --hash=sha256:eb12fb2ba69ffa05f8695f61c69e591dc4b4a12ac3757ac8af8adb259bf56d17 \ + --hash=sha256:eda059b6bc8bc0812d626fd91a7ce01bf583df0a61296eff390fd94141a34e30 \ + --hash=sha256:f03ac127268b43ef4fe9e6ab6794a6794b49485a0cc0c1db79876d2f33f75bc7 \ + --hash=sha256:f298e218441525d3794428b4c8b8fb8662c6d3ea79925d4807ee6b9a96a3bca5 \ + --hash=sha256:f5542f9b941279d82d41eb0aa9f98eba36fe4df5c7086c651df7944935b37182 \ + --hash=sha256:f6f7deae3feb4edfa2efaf7c574fe88cbf055038a6abdb40188e4fff66d5699f \ + --hash=sha256:f9b1e28d0e8dbfa858abdba91d6b547beaf2df1a59bec6da6faae7b96a4991a9 \ + --hash=sha256:f9f8405c2c758532c74fed975dbee57be1f31a6e865c031870c79a6ed3212ada \ + --hash=sha256:fa48b1b63d639f9483e0633e092f5851e2348c352f1f9bb6c8182f87884ef876 \ + --hash=sha256:fb78f6e7fcd8ad785d28cd577168bc1aaee827b25bb8755638f694794ea98f0a \ + --hash=sha256:fbc597639158fd7c14d55e808718848319540f51b0e6746e3eefa59723a4a348 \ + --hash=sha256:fce8cbd4997efeb450bd298b54f755dcdff18d496f7a5ddbb4867c6d7c88fdc3 \ + --hash=sha256:fd0350afdc3aabd5576f60ea109228bd5538139713c7b094c5cd27c73a98bc6f \ + --hash=sha256:fd0a274c0e5f9a21565cd9d3dd749b61f96b7aa1e20a93aa1ba4029518f2e5c0 \ + --hash=sha256:fdb8a068947befafba9952162645dc2fecaeb400e64584829ed5e9b2fbe21a7f + # via requests +idna==3.20 \ + --hash=sha256:ab7ae7122974553370f0bdb919e1a960b2cd1bc1ef0276416d896db81c14582c + # via requests +pyyaml==6.0.3 \ + --hash=sha256:00c4bdeba853cc34e7dd471f16b4114f4162dc03e6b7afcc2128711f0eca823c \ + --hash=sha256:0150219816b6a1fa26fb4699fb7daa9caf09eb1999f3b70fb6e786805e80375a \ + --hash=sha256:02893d100e99e03eda1c8fd5c441d8c60103fd175728e23e431db1b589cf5ab3 \ + --hash=sha256:02ea2dfa234451bbb8772601d7b8e426c2bfa197136796224e50e35a78777956 \ + --hash=sha256:0f29edc409a6392443abf94b9cf89ce99889a1dd5376d94316ae5145dfedd5d6 \ + --hash=sha256:10892704fc220243f5305762e276552a0395f7beb4dbf9b14ec8fd43b57f126c \ + --hash=sha256:16249ee61e95f858e83976573de0f5b2893b3677ba71c9dd36b9cf8be9ac6d65 \ + --hash=sha256:1d37d57ad971609cf3c53ba6a7e365e40660e3be0e5175fa9f2365a379d6095a \ + --hash=sha256:1ebe39cb5fc479422b83de611d14e2c0d3bb2a18bbcb01f229ab3cfbd8fee7a0 \ + --hash=sha256:214ed4befebe12df36bcc8bc2b64b396ca31be9304b8f59e25c11cf94a4c033b \ + --hash=sha256:2283a07e2c21a2aa78d9c4442724ec1eb15f5e42a723b99cb3d822d48f5f7ad1 \ + --hash=sha256:22ba7cfcad58ef3ecddc7ed1db3409af68d023b7f940da23c6c2a1890976eda6 \ + --hash=sha256:27c0abcb4a5dac13684a37f76e701e054692a9b2d3064b70f5e4eb54810553d7 \ + --hash=sha256:28c8d926f98f432f88adc23edf2e6d4921ac26fb084b028c733d01868d19007e \ + --hash=sha256:2e71d11abed7344e42a8849600193d15b6def118602c4c176f748e4583246007 \ + --hash=sha256:34d5fcd24b8445fadc33f9cf348c1047101756fd760b4dacb5c3e99755703310 \ + --hash=sha256:37503bfbfc9d2c40b344d06b2199cf0e96e97957ab1c1b546fd4f87e53e5d3e4 \ + --hash=sha256:3c5677e12444c15717b902a5798264fa7909e41153cdf9ef7ad571b704a63dd9 \ + --hash=sha256:3ff07ec89bae51176c0549bc4c63aa6202991da2d9a6129d7aef7f1407d3f295 \ + --hash=sha256:41715c910c881bc081f1e8872880d3c650acf13dfa8214bad49ed4cede7c34ea \ + --hash=sha256:418cf3f2111bc80e0933b2cd8cd04f286338bb88bdc7bc8e6dd775ebde60b5e0 \ + --hash=sha256:44edc647873928551a01e7a563d7452ccdebee747728c1080d881d68af7b997e \ + --hash=sha256:4a2e8cebe2ff6ab7d1050ecd59c25d4c8bd7e6f400f5f82b96557ac0abafd0ac \ + --hash=sha256:4ad1906908f2f5ae4e5a8ddfce73c320c2a1429ec52eafd27138b7f1cbe341c9 \ + --hash=sha256:501a031947e3a9025ed4405a168e6ef5ae3126c59f90ce0cd6f2bfc477be31b7 \ + --hash=sha256:5190d403f121660ce8d1d2c1bb2ef1bd05b5f68533fc5c2ea899bd15f4399b35 \ + --hash=sha256:5498cd1645aa724a7c71c8f378eb29ebe23da2fc0d7a08071d89469bf1d2defb \ + --hash=sha256:5cf4e27da7e3fbed4d6c3d8e797387aaad68102272f8f9752883bc32d61cb87b \ + --hash=sha256:5e0b74767e5f8c593e8c9b5912019159ed0533c70051e9cce3e8b6aa699fcd69 \ + --hash=sha256:5ed875a24292240029e4483f9d4a4b8a1ae08843b9c54f43fcc11e404532a8a5 \ + --hash=sha256:5fcd34e47f6e0b794d17de1b4ff496c00986e1c83f7ab2fb8fcfe9616ff7477b \ + --hash=sha256:5fdec68f91a0c6739b380c83b951e2c72ac0197ace422360e6d5a959d8d97b2c \ + --hash=sha256:6344df0d5755a2c9a276d4473ae6b90647e216ab4757f8426893b5dd2ac3f369 \ + --hash=sha256:64386e5e707d03a7e172c0701abfb7e10f0fb753ee1d773128192742712a98fd \ + --hash=sha256:652cb6edd41e718550aad172851962662ff2681490a8a711af6a4d288dd96824 \ + --hash=sha256:66291b10affd76d76f54fad28e22e51719ef9ba22b29e1d7d03d6777a9174198 \ + --hash=sha256:66e1674c3ef6f541c35191caae2d429b967b99e02040f5ba928632d9a7f0f065 \ + --hash=sha256:6adc77889b628398debc7b65c073bcb99c4a0237b248cacaf3fe8a557563ef6c \ + --hash=sha256:79005a0d97d5ddabfeeea4cf676af11e647e41d81c9a7722a193022accdb6b7c \ + --hash=sha256:7c6610def4f163542a622a73fb39f534f8c101d690126992300bf3207eab9764 \ + --hash=sha256:7f047e29dcae44602496db43be01ad42fc6f1cc0d8cd6c83d342306c32270196 \ + --hash=sha256:8098f252adfa6c80ab48096053f512f2321f0b998f98150cea9bd23d83e1467b \ + --hash=sha256:850774a7879607d3a6f50d36d04f00ee69e7fc816450e5f7e58d7f17f1ae5c00 \ + --hash=sha256:8d1fab6bb153a416f9aeb4b8763bc0f22a5586065f86f7664fc23339fc1c1fac \ + --hash=sha256:8da9669d359f02c0b91ccc01cac4a67f16afec0dac22c2ad09f46bee0697eba8 \ + --hash=sha256:8dc52c23056b9ddd46818a57b78404882310fb473d63f17b07d5c40421e47f8e \ + --hash=sha256:9149cad251584d5fb4981be1ecde53a1ca46c891a79788c0df828d2f166bda28 \ + --hash=sha256:93dda82c9c22deb0a405ea4dc5f2d0cda384168e466364dec6255b293923b2f3 \ + --hash=sha256:96b533f0e99f6579b3d4d4995707cf36df9100d67e0c8303a0c55b27b5f99bc5 \ + --hash=sha256:9c57bb8c96f6d1808c030b1687b9b5fb476abaa47f0db9c0101f5e9f394e97f4 \ + --hash=sha256:9c7708761fccb9397fe64bbc0395abcae8c4bf7b0eac081e12b809bf47700d0b \ + --hash=sha256:9f3bfb4965eb874431221a3ff3fdcddc7e74e3b07799e0e84ca4a0f867d449bf \ + --hash=sha256:a33284e20b78bd4a18c8c2282d549d10bc8408a2a7ff57653c0cf0b9be0afce5 \ + --hash=sha256:a80cb027f6b349846a3bf6d73b5e95e782175e52f22108cfa17876aaeff93702 \ + --hash=sha256:b30236e45cf30d2b8e7b3e85881719e98507abed1011bf463a8fa23e9c3e98a8 \ + --hash=sha256:b3bc83488de33889877a0f2543ade9f70c67d66d9ebb4ac959502e12de895788 \ + --hash=sha256:b865addae83924361678b652338317d1bd7e79b1f4596f96b96c77a5a34b34da \ + --hash=sha256:b8bb0864c5a28024fac8a632c443c87c5aa6f215c0b126c449ae1a150412f31d \ + --hash=sha256:ba1cc08a7ccde2d2ec775841541641e4548226580ab850948cbfda66a1befcdc \ + --hash=sha256:bdb2c67c6c1390b63c6ff89f210c8fd09d9a1217a465701eac7316313c915e4c \ + --hash=sha256:c1ff362665ae507275af2853520967820d9124984e0f7466736aea23d8611fba \ + --hash=sha256:c2514fceb77bc5e7a2f7adfaa1feb2fb311607c9cb518dbc378688ec73d8292f \ + --hash=sha256:c3355370a2c156cffb25e876646f149d5d68f5e0a3ce86a5084dd0b64a994917 \ + --hash=sha256:c458b6d084f9b935061bc36216e8a69a7e293a2f1e68bf956dcd9e6cbcd143f5 \ + --hash=sha256:d0eae10f8159e8fdad514efdc92d74fd8d682c933a6dd088030f3834bc8e6b26 \ + --hash=sha256:ebc55a14a21cb14062aa4162f906cd962b28e2e9ea38f9b4391244cd8de4ae0b \ + --hash=sha256:eda16858a3cab07b80edaf74336ece1f986ba330fdb8ee0d6c0d68fe82bc96be \ + --hash=sha256:ee2922902c45ae8ccada2c5b501ab86c36525b883eff4255313a253a3160861c \ + --hash=sha256:efd7b85f94a6f21e4932043973a7ba2613b059c4a000551892ac9f1d11f5baf3 \ + --hash=sha256:f7057c9a337546edc7973c0d3ba84ddcdf0daa14533c2065749c9075001090e6 \ + --hash=sha256:fa160448684b4e94d80416c0fa4aac48967a969efe22931448d853ada8baf926 \ + --hash=sha256:fc09d0aa354569bc501d4e787133afc08552722d3ab34836a80547331bb5d4a0 + # via -r docker/requirements-worker.in +requests==2.34.2 \ + --hash=sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0 + # via -r docker/requirements-worker.in +urllib3==2.8.0 \ + --hash=sha256:0cf3cae568d36aa9576b28dfb35f11328f1cb974ca7647d9475ebb86c75ac6e3 + # via requests +zstandard==0.23.0 \ + --hash=sha256:034b88913ecc1b097f528e42b539453fa82c3557e414b3de9d5632c80439a473 \ + --hash=sha256:0a7f0804bb3799414af278e9ad51be25edf67f78f916e08afdb983e74161b916 \ + --hash=sha256:11e3bf3c924853a2d5835b24f03eeba7fc9b07d8ca499e247e06ff5676461a15 \ + --hash=sha256:12a289832e520c6bd4dcaad68e944b86da3bad0d339ef7989fb7e88f92e96072 \ + --hash=sha256:1516c8c37d3a053b01c1c15b182f3b5f5eef19ced9b930b684a73bad121addf4 \ + --hash=sha256:157e89ceb4054029a289fb504c98c6a9fe8010f1680de0201b3eb5dc20aa6d9e \ + --hash=sha256:1bfe8de1da6d104f15a60d4a8a768288f66aa953bbe00d027398b93fb9680b26 \ + --hash=sha256:1e172f57cd78c20f13a3415cc8dfe24bf388614324d25539146594c16d78fcc8 \ + --hash=sha256:1fd7e0f1cfb70eb2f95a19b472ee7ad6d9a0a992ec0ae53286870c104ca939e5 \ + --hash=sha256:203d236f4c94cd8379d1ea61db2fce20730b4c38d7f1c34506a31b34edc87bdd \ + --hash=sha256:27d3ef2252d2e62476389ca8f9b0cf2bbafb082a3b6bfe9d90cbcbb5529ecf7c \ + --hash=sha256:29a2bc7c1b09b0af938b7a8343174b987ae021705acabcbae560166567f5a8db \ + --hash=sha256:2ef230a8fd217a2015bc91b74f6b3b7d6522ba48be29ad4ea0ca3a3775bf7dd5 \ + --hash=sha256:2ef3775758346d9ac6214123887d25c7061c92afe1f2b354f9388e9e4d48acfc \ + --hash=sha256:2f146f50723defec2975fb7e388ae3a024eb7151542d1599527ec2aa9cacb152 \ + --hash=sha256:2fb4535137de7e244c230e24f9d1ec194f61721c86ebea04e1581d9d06ea1269 \ + --hash=sha256:32ba3b5ccde2d581b1e6aa952c836a6291e8435d788f656fe5976445865ae045 \ + --hash=sha256:34895a41273ad33347b2fc70e1bff4240556de3c46c6ea430a7ed91f9042aa4e \ + --hash=sha256:379b378ae694ba78cef921581ebd420c938936a153ded602c4fea612b7eaa90d \ + --hash=sha256:38302b78a850ff82656beaddeb0bb989a0322a8bbb1bf1ab10c17506681d772a \ + --hash=sha256:3aa014d55c3af933c1315eb4bb06dd0459661cc0b15cd61077afa6489bec63bb \ + --hash=sha256:4051e406288b8cdbb993798b9a45c59a4896b6ecee2f875424ec10276a895740 \ + --hash=sha256:40b33d93c6eddf02d2c19f5773196068d875c41ca25730e8288e9b672897c105 \ + --hash=sha256:43da0f0092281bf501f9c5f6f3b4c975a8a0ea82de49ba3f7100e64d422a1274 \ + --hash=sha256:445e4cb5048b04e90ce96a79b4b63140e3f4ab5f662321975679b5f6360b90e2 \ + --hash=sha256:48ef6a43b1846f6025dde6ed9fee0c24e1149c1c25f7fb0a0585572b2f3adc58 \ + --hash=sha256:50a80baba0285386f97ea36239855f6020ce452456605f262b2d33ac35c7770b \ + --hash=sha256:519fbf169dfac1222a76ba8861ef4ac7f0530c35dd79ba5727014613f91613d4 \ + --hash=sha256:53dd9d5e3d29f95acd5de6802e909ada8d8d8cfa37a3ac64836f3bc4bc5512db \ + --hash=sha256:53ea7cdc96c6eb56e76bb06894bcfb5dfa93b7adcf59d61c6b92674e24e2dd5e \ + --hash=sha256:576856e8594e6649aee06ddbfc738fec6a834f7c85bf7cadd1c53d4a58186ef9 \ + --hash=sha256:59556bf80a7094d0cfb9f5e50bb2db27fefb75d5138bb16fb052b61b0e0eeeb0 \ + --hash=sha256:5d41d5e025f1e0bccae4928981e71b2334c60f580bdc8345f824e7c0a4c2a813 \ + --hash=sha256:61062387ad820c654b6a6b5f0b94484fa19515e0c5116faf29f41a6bc91ded6e \ + --hash=sha256:61f89436cbfede4bc4e91b4397eaa3e2108ebe96d05e93d6ccc95ab5714be512 \ + --hash=sha256:62136da96a973bd2557f06ddd4e8e807f9e13cbb0bfb9cc06cfe6d98ea90dfe0 \ + --hash=sha256:64585e1dba664dc67c7cdabd56c1e5685233fbb1fc1966cfba2a340ec0dfff7b \ + --hash=sha256:65308f4b4890aa12d9b6ad9f2844b7ee42c7f7a4fd3390425b242ffc57498f48 \ + --hash=sha256:66b689c107857eceabf2cf3d3fc699c3c0fe8ccd18df2219d978c0283e4c508a \ + --hash=sha256:6a41c120c3dbc0d81a8e8adc73312d668cd34acd7725f036992b1b72d22c1772 \ + --hash=sha256:6f77fa49079891a4aab203d0b1744acc85577ed16d767b52fc089d83faf8d8ed \ + --hash=sha256:72c68dda124a1a138340fb62fa21b9bf4848437d9ca60bd35db36f2d3345f373 \ + --hash=sha256:752bf8a74412b9892f4e5b58f2f890a039f57037f52c89a740757ebd807f33ea \ + --hash=sha256:76e79bc28a65f467e0409098fa2c4376931fd3207fbeb6b956c7c476d53746dd \ + --hash=sha256:774d45b1fac1461f48698a9d4b5fa19a69d47ece02fa469825b442263f04021f \ + --hash=sha256:77da4c6bfa20dd5ea25cbf12c76f181a8e8cd7ea231c673828d0386b1740b8dc \ + --hash=sha256:77ea385f7dd5b5676d7fd943292ffa18fbf5c72ba98f7d09fc1fb9e819b34c23 \ + --hash=sha256:80080816b4f52a9d886e67f1f96912891074903238fe54f2de8b786f86baded2 \ + --hash=sha256:80a539906390591dd39ebb8d773771dc4db82ace6372c4d41e2d293f8e32b8db \ + --hash=sha256:82d17e94d735c99621bf8ebf9995f870a6b3e6d14543b99e201ae046dfe7de70 \ + --hash=sha256:837bb6764be6919963ef41235fd56a6486b132ea64afe5fafb4cb279ac44f259 \ + --hash=sha256:84433dddea68571a6d6bd4fbf8ff398236031149116a7fff6f777ff95cad3df9 \ + --hash=sha256:8c24f21fa2af4bb9f2c492a86fe0c34e6d2c63812a839590edaf177b7398f700 \ + --hash=sha256:8ed7d27cb56b3e058d3cf684d7200703bcae623e1dcc06ed1e18ecda39fee003 \ + --hash=sha256:9206649ec587e6b02bd124fb7799b86cddec350f6f6c14bc82a2b70183e708ba \ + --hash=sha256:983b6efd649723474f29ed42e1467f90a35a74793437d0bc64a5bf482bedfa0a \ + --hash=sha256:98da17ce9cbf3bfe4617e836d561e433f871129e3a7ac16d6ef4c680f13a839c \ + --hash=sha256:9c236e635582742fee16603042553d276cca506e824fa2e6489db04039521e90 \ + --hash=sha256:9da6bc32faac9a293ddfdcb9108d4b20416219461e4ec64dfea8383cac186690 \ + --hash=sha256:a05e6d6218461eb1b4771d973728f0133b2a4613a6779995df557f70794fd60f \ + --hash=sha256:a0817825b900fcd43ac5d05b8b3079937073d2b1ff9cf89427590718b70dd840 \ + --hash=sha256:a4ae99c57668ca1e78597d8b06d5af837f377f340f4cce993b551b2d7731778d \ + --hash=sha256:a8c86881813a78a6f4508ef9daf9d4995b8ac2d147dcb1a450448941398091c9 \ + --hash=sha256:a8fffdbd9d1408006baaf02f1068d7dd1f016c6bcb7538682622c556e7b68e35 \ + --hash=sha256:a9b07268d0c3ca5c170a385a0ab9fb7fdd9f5fd866be004c4ea39e44edce47dd \ + --hash=sha256:ab19a2d91963ed9e42b4e8d77cd847ae8381576585bad79dbd0a8837a9f6620a \ + --hash=sha256:ac184f87ff521f4840e6ea0b10c0ec90c6b1dcd0bad2f1e4a9a1b4fa177982ea \ + --hash=sha256:b0e166f698c5a3e914947388c162be2583e0c638a4703fc6a543e23a88dea3c1 \ + --hash=sha256:b2170c7e0367dde86a2647ed5b6f57394ea7f53545746104c6b09fc1f4223573 \ + --hash=sha256:b4567955a6bc1b20e9c31612e615af6b53733491aeaa19a6b3b37f3b65477094 \ + --hash=sha256:b69bb4f51daf461b15e7b3db033160937d3ff88303a7bc808c67bbc1eaf98c78 \ + --hash=sha256:b8c0bd73aeac689beacd4e7667d48c299f61b959475cdbb91e7d3d88d27c56b9 \ + --hash=sha256:be9b5b8659dff1f913039c2feee1aca499cfbc19e98fa12bc85e037c17ec6ca5 \ + --hash=sha256:bf0a05b6059c0528477fba9054d09179beb63744355cab9f38059548fedd46a9 \ + --hash=sha256:c16842b846a8d2a145223f520b7e18b57c8f476924bda92aeee3a88d11cfc391 \ + --hash=sha256:c363b53e257246a954ebc7c488304b5592b9c53fbe74d03bc1c64dda153fb847 \ + --hash=sha256:c7c517d74bea1a6afd39aa612fa025e6b8011982a0897768a2f7c8ab4ebb78a2 \ + --hash=sha256:d20fd853fbb5807c8e84c136c278827b6167ded66c72ec6f9a14b863d809211c \ + --hash=sha256:d2240ddc86b74966c34554c49d00eaafa8200a18d3a5b6ffbf7da63b11d74ee2 \ + --hash=sha256:d477ed829077cd945b01fc3115edd132c47e6540ddcd96ca169facff28173057 \ + --hash=sha256:d50d31bfedd53a928fed6707b15a8dbeef011bb6366297cc435accc888b27c20 \ + --hash=sha256:dc1d33abb8a0d754ea4763bad944fd965d3d95b5baef6b121c0c9013eaf1907d \ + --hash=sha256:dc5d1a49d3f8262be192589a4b72f0d03b72dcf46c51ad5852a4fdc67be7b9e4 \ + --hash=sha256:e2d1a054f8f0a191004675755448d12be47fa9bebbcffa3cdf01db19f2d30a54 \ + --hash=sha256:e7792606d606c8df5277c32ccb58f29b9b8603bf83b48639b7aedf6df4fe8171 \ + --hash=sha256:ed1708dbf4d2e3a1c5c69110ba2b4eb6678262028afd6c6fbcc5a8dac9cda68e \ + --hash=sha256:f2d4380bf5f62daabd7b751ea2339c1a21d1c9463f1feb7fc2bdcea2c29c3160 \ + --hash=sha256:f3513916e8c645d0610815c257cbfd3242adfd5c4cfa78be514e5a3ebb42a41b \ + --hash=sha256:f8346bfa098532bc1fb6c7ef06783e969d87a99dd1d2a5a18a892c1d7a643c58 \ + --hash=sha256:f83fa6cae3fff8e98691248c9320356971b59678a17f20656a9e59cd32cee6d8 \ + --hash=sha256:fa6ce8b52c5987b3e34d5674b0ab529a4602b632ebab0a93b07bfb4dfc8f8a33 \ + --hash=sha256:fb2b1ecfef1e67897d336de3a0e3f52478182d6a47eda86cbd42504c5cbd009a \ + --hash=sha256:fc9ca1c9718cb3b06634c7c8dec57d24e9438b2aa9a0f02b8bb36bf478538880 \ + --hash=sha256:fd30d9c67d13d891f2360b2a120186729c111238ac63b43dbd37a5a40670b8ca \ + --hash=sha256:fd7699e8fd9969f455ef2926221e0233f81a2542921471382e77a9e2f2b57f4b \ + --hash=sha256:fe3b385d996ee0822fd46528d9f0443b880d4d05528fd26a9119a54ec3f91c69 + # via -r docker/requirements-worker.in diff --git a/docker/requirements.in b/docker/requirements.in new file mode 100644 index 0000000..6b63998 --- /dev/null +++ b/docker/requirements.in @@ -0,0 +1,6 @@ +# Keep the runtime union tied to both application manifests, including the dashboard. +-r ../app/requirements.txt +-r ../app/requirements-keycheckers.txt + +# One hash-locked environment is shared by the runtime and test targets. +pytest==8.4.2 diff --git a/docker/requirements.lock b/docker/requirements.lock new file mode 100644 index 0000000..e92543c --- /dev/null +++ b/docker/requirements.lock @@ -0,0 +1,1216 @@ +# +# This file is autogenerated by pip-compile with Python 3.12 +# by the following command: +# +# See docker/build-dependencies/README.md for the pinned Python 3.12.14 pip-tools generation command. +# +--only-binary :all: + +altair==6.2.2 \ + --hash=sha256:94014f8ad8617c3cb163d1137359cd6db5ba134b9b46d93cfd8b609fd245a583 + # via streamlit +anyio==4.15.1 \ + --hash=sha256:6152fdbbf9a77fdec97731721bebf7c4c44f7c29b424b0065826173efc7ed101 + # via + # starlette + # streamlit +attrs==26.1.0 \ + --hash=sha256:c647aa4a12dfbad9333ca4e71fe62ddc36f4e63b2d260a37a8b83d2f043ac309 + # via + # jsonschema + # referencing +boto3==1.43.94 \ + --hash=sha256:2534bf331acd2f448b9cf8317f4eed453c65d9e0b7de254e77c99d390ac57aec + # via -r app/requirements-keycheckers.txt +botocore==1.43.94 \ + --hash=sha256:1dfb86603a87fdaebda2540db56aef5b226ec58ab72c13ca56737e4d66dea9ab + # via + # -r app/requirements-keycheckers.txt + # boto3 + # s3transfer +certifi==2026.7.22 \ + --hash=sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775 + # via requests +charset-normalizer==3.5.1 \ + --hash=sha256:00668ebb0609751758682eb0b5857e7c35b9f00e84dfdef062e103244ec94d45 \ + --hash=sha256:012a22b88a77ca2e59b98ac5889b0deb604147666032f45e6d6e217634d2550d \ + --hash=sha256:01e93745f7f219b703b60ba7afead36cfc4242782be5af484673fc500df12da5 \ + --hash=sha256:04368edf83514385ffc3e1cfd4546e595f4f1272dd23ba437a93a9cc3741d47b \ + --hash=sha256:0722590aabf9dc6a6c0343d523c05458fa2b5047dbe6302fd526bb570600753f \ + --hash=sha256:07ffd07412fc5d5e84cd8952acf9ff7e4ed7a708e69d1bada19d8ba91711353f \ + --hash=sha256:09a7bba9f739468c8e78c36a75c33768e53cb1959fc638f510454c14683f00d5 \ + --hash=sha256:0b2b1b3fa5670c127b246df1d0c059defd41f689a868a3b9d79df9b1cac42d22 \ + --hash=sha256:0c6dfb5ca6723eeed15aa8e564a014d69fcb8812f94eef11fe3631e0508199f5 \ + --hash=sha256:0d929fc574b4d6fd9e7c0f5c2ede8716a41911923aa7fa5fce38e0818aa4a1ac \ + --hash=sha256:13e3afe97712e8887cd516e960c63f0b93122971e5b5e4b2622fe7701771e838 \ + --hash=sha256:15f024313246a4ed976c60f440bb8d257815513a681d212ff74fd46f7d715a90 \ + --hash=sha256:195ce897c6153c0700078142cf8efe3e6454ca4cf4357499e4078dfd83396626 \ + --hash=sha256:19a3dd5aa73cef1c99687c4fc57db016a9c17104ae1185da88ba566a5d3bebe4 \ + --hash=sha256:1d1c7a53a6c2103925cdd6d7229f8c567379f211c869793df679f2e9f738c369 \ + --hash=sha256:1f5883d77fd409a261abb5dc8ccbe335720d798b1de4abb3b1d47ccbbc76b53b \ + --hash=sha256:21b82d8082f6f5e7f456ef0bd16323d08de1266efbfeb476e64b2a91d1471a4e \ + --hash=sha256:252d099029bcbea642f2a06c4ed5046bdf8b5a8150b64afa5e027e88b106e5ee \ + --hash=sha256:256dd4d85d9e4dc595e2bc983c980e73f62ddeb3165c58b4c3dfe78c5c8548c1 \ + --hash=sha256:26422d45fd13551cf564c58932f7d72b4f58b93b0fcf18c35ba6be12b46bb102 \ + --hash=sha256:2679de311c7946dde5d3b6f44941844133ff5c7cb86099c0061ab1e8901c20a8 \ + --hash=sha256:29880d17a8eb0b5cfdfd8944b468322928059aa35f1f5fa8ff22b149ec0b42f8 \ + --hash=sha256:2bced4061f000f7187254a02ad3433ae17eaf991747ceea2f478422590a5bba9 \ + --hash=sha256:2e9cf9253119d8e5d111f05d71626786fd3d6193817316eab1ca088cdb8593cf \ + --hash=sha256:2f06b7eae9dbe77fe1d644ca244dad508de8d302870a43f3c559b521270938a0 \ + --hash=sha256:2f293479cce755c75f1697e87c409b7ae4c555c7dfecb6e988ad13abba943031 \ + --hash=sha256:329fc3ccb63ad22d867d84c2adea759a64079a37ba4a343433b02c7a2816871e \ + --hash=sha256:343fb4f2821043bd87095f7b08a1a181febc8e36ac64212143bbfd0a0e1bc235 \ + --hash=sha256:3588e376b3ea2eea84976f67273d679f229e24c66dce7b82ae45aef04ff6e072 \ + --hash=sha256:35aea775dc2bd5f54cd84a1cd2696cc3207c479cb9cf0bd346f0d343e4300ddb \ + --hash=sha256:35fe081843b35aad20ffeccec3eeffbe637b15d14f3fb22cc1b59cd8ec17e93c \ + --hash=sha256:36047af20e17097c3bb9476c2b7655f2f7aa51322c0ba58c07695bedf755a950 \ + --hash=sha256:3617ac3cfd8b9888f145ad89dd6e692285834b0201c6074a5eeaad3fd4d668c2 \ + --hash=sha256:366ec70f5547c640d3ce1985722490f23faf4eb5216a7eeba78277490e78dacb \ + --hash=sha256:394fea06235c8543390050ed5f529187074b029fb027213f6c46ac11ab5d950e \ + --hash=sha256:3d27167433c0d5f18dc850f07d0b3816221984fecdc405d6c157a6f0b8f8e9e6 \ + --hash=sha256:3e5e1224c0a6a90e05843e07adfec669edebec17801c67072f51e59561d63c0b \ + --hash=sha256:41876ee62a3dddf48ff1121ad8f0798032aa03f2fd35f21f34a4cab14f18d8d2 \ + --hash=sha256:433c5a81eade63b47e522303bad236f59dba55ea6951746f5558355eeed8c75d \ + --hash=sha256:4582c27e8c889d64811987b5967fbd3ae0c823fe1fd933b543d55ac20bb475fa \ + --hash=sha256:485a0d363cafefcd2538a73c7c838daa2035f09b2c9f9b5e3133f80c6aeb84c2 \ + --hash=sha256:494b70049a4d69aec6e8137c13af4cf8db8c9f9820a1392ac293b0dd2987a818 \ + --hash=sha256:496846868fea80e479324862fa877f02411f2fd0f83b79ccee2607aa68b2a032 \ + --hash=sha256:4abdc5f9ad448c1ecbfae2974b820535d6bc6e7eef63babbab3d81cf46968c71 \ + --hash=sha256:4b599739b93b2cbeded49645ae3c8d1405c29ddfbceac1545c87a3f9580a9e96 \ + --hash=sha256:4bea7f8ebe90bbd7f0e4a2de42ca6924ba23e3e76418c408ff82f1d46fabd687 \ + --hash=sha256:4c4fb141a727957c93edfe5c32a26ceb6b5f6461d67146e2d39f51e16170bea8 \ + --hash=sha256:4c9548dc78002099910abaebc0a72ac58b7d30931869e0351c09b507dff4ece3 \ + --hash=sha256:4d26f14f041e83dd8edfd61f4cd4fa7285d31798b5bf1f28e70c367ba6c41d61 \ + --hash=sha256:4f298bdadb8f0b9e5672877f647d1be9373ef5320c9e2f049795e26cad28b6a9 \ + --hash=sha256:52ec005752a56ae79547a05c0139ca2501a0c866390b6115008456b9f0e7cde1 \ + --hash=sha256:55261ac0d2941c42f196dd576f543d87a8ee03cd6f5e30dfb4d807b2e3b9121a \ + --hash=sha256:56490c595a28b1bb27dfc583e816152a9767721ef58b2c03b13f954d2f707420 \ + --hash=sha256:58d3e12c88e0950bca850ae1f7c256055c097639c2edb9eb123af9807d8b15e4 \ + --hash=sha256:58d4aa13a59c969dbfdf9e6a9560e242cbfd9e8a8f50c2747714df1a423adf65 \ + --hash=sha256:59171c6e45bf07d0d5cab3b0bf81d945035530f6873398b3b531c31184d46663 \ + --hash=sha256:5b6d1386bf0096d26d3a863dc0a487a5b4eb9aa93cf5ba69683d29dde6b9d60f \ + --hash=sha256:5c0ea61a470e070686aa30892fed79e297d2c8d0ab46b8bcdf027d38c51da591 \ + --hash=sha256:5c84bec0ab5ae0c64bfe73a7d2adcb5ce73b467523fc27fd6a28ab2aa6cbe35a \ + --hash=sha256:5ca0555312ae2fe82715cada7fac375530c2f3349e1eaa1bcb33d0283ac79a18 \ + --hash=sha256:5d8531a6569d025f68e2321e7638fb7978f23db58e5f69f56913837aae03816e \ + --hash=sha256:5e2d0e146dcb57034f8b97dc58d2d512cb90aba253960ce449f695fec6a82c6f \ + --hash=sha256:5fc45d653ea8c9a20479167e11d4a0f8cb2fa3470737ab6f9c827532313187b7 \ + --hash=sha256:6199d5606e2bbf2b096cf64d03f8b6790c91081d5ac866b8e7bb6422738cc60c \ + --hash=sha256:62b55f6722735a6c472f88361cde6640608773d9443cebdbb51abf436a1fcdd3 \ + --hash=sha256:687c9ca3035544b113bea2055e180af96fb63c0c476e22a9180f51925186e7b7 \ + --hash=sha256:6b7430cf5728e68f6c462254009a6ef4086e1bea43cf2f57aa9c55fb4f50ff96 \ + --hash=sha256:6ba32c4d2abf1d2fe7cf27d280f4cca5664233b0f885549c7761719eb977f486 \ + --hash=sha256:6c9cdde8becb25a7fde49924511aa2644d6f8081cc8df8e9452724303348d8e3 \ + --hash=sha256:6df0ec430f9a831772c23ca5a224cba36517a58a84bb32c32bb59a9fa67c47f6 \ + --hash=sha256:6e2912d4babbc65196ac13c2f53468dc57fb8b9c25ef913e8c59ddf7c6dc0e1b \ + --hash=sha256:6e5e4d73d588ca5ed09df1b7dcd1b203d1df3c542e3f50d126c947d432b10731 \ + --hash=sha256:70055ff39b97c99e7ae40ea3e393fb62aa2e44dbd9b29f8d14f42fb0025c3959 \ + --hash=sha256:706bfd38730a5ac7a365793269a00f4e988178cec121391f4248d84ad8c972e9 \ + --hash=sha256:7235dc28fc6dd9d832ac7c7bce95367dedb85929f17368a0c2bee1e080b9acbf \ + --hash=sha256:774d157f112367ff4abd29019f38f023c24e00e56edc7829c20e358a5a913ad8 \ + --hash=sha256:77efcff2b23071c349402ac1066667a3d011f62398d81408c9b88ad991747c9e \ + --hash=sha256:789b8982559ae28dad2356519f841655756cdcd96616410590ae0b17454ee64f \ + --hash=sha256:7ac76cf9afd34929d76eb7fcb63be476a4853d8a96f0dcf2d0db68a0cbdf9885 \ + --hash=sha256:7c0c10730342b0c9b35dd1d619beb8214e520bd96a1f870f452680b238aab3e0 \ + --hash=sha256:823f82903d189af463d7df250ef1f7f696f3cee08cc8d91deb565e8d425f6506 \ + --hash=sha256:838648accb3a7fd9803fd45c87bce8509648eb0c11bc34e216141300977244f2 \ + --hash=sha256:854066be00447fa8de2ccbbe893e2ffc4b123ef16d897af794c1e18bd4a714b0 \ + --hash=sha256:85d5855daafc240cc045c026d7a15fd198a09b0fc8ff6f5ecbb5297b509cb11e \ + --hash=sha256:85de3134b5379856e323ba37c19c9256d39425f7b76a63af52b09fb4664c2e8f \ + --hash=sha256:87e4f41d375c0b9be2fb5251aee4b8a689169e134535aed81bf085c3b647451e \ + --hash=sha256:88ca277405c2d3b71c4e1c2ee0e7966e807bcba86a69d11e19ba199d18ae4491 \ + --hash=sha256:88e85ab89cb822c1e635f51d6d32e488f94e002e70e2f492bdb8b945543f345a \ + --hash=sha256:8ac8c94b6539074e0f40899301273ac8402b9b3e01c7b7ba269ff30340aaaf20 \ + --hash=sha256:8fe532b3c966d1fb794e0698e4589d0444017ae77fc0b31edea13c0e35bcc449 \ + --hash=sha256:9085f87b0e38a2b92b8923059b4e8789fe40d9279712d15dcc670048d77079af \ + --hash=sha256:90b7481fb62fbe172c558bc6fd1c4c98d82004a54a7551f20e11ac9bf0b8708c \ + --hash=sha256:92caef967d287a407085d61176fce4012b1dd62daed4eb6d5ceb26d3d2538712 \ + --hash=sha256:9362dd90aa7dab48c0054a21187791ccf05473f7dba5d92b8033ae62164675e7 \ + --hash=sha256:94d78ecec2605a8d0398b0f365d5f12a63248438516f5dac536a5eff7337df4a \ + --hash=sha256:94fbf1c0c6cc0d3d5e50f9a9313a8cdca90dd696d34b381cd1704f8c9e939f20 \ + --hash=sha256:950f23cb393f85543777b0433f082cddd25b51ab398eac7971146495679efe5f \ + --hash=sha256:96eefc178f8636b9c760c5829345307fd81cfae9ab1e80997dbddeb0f54ee9a3 \ + --hash=sha256:96fef3e886d6a9874b14f27fc193fbdc69d5d8035783d86aa4e1cea594e695f9 \ + --hash=sha256:977cdbd483a9cff38179bea4fd754289a6f2195c7abd414aba85410b3e66cc5e \ + --hash=sha256:978eab16f55b4ab2c2a745be9a0a840bf8f09a7f227d9c76eb30214d078865a5 \ + --hash=sha256:994e883d17c559cdfd38c84003c8b27d25424a1077272a17e7cd27bfe0bf57b2 \ + --hash=sha256:9ac4444d8d4fd4c4bd08bf451ed3167aa9e7ec6cdb41b648794f1d1103652e36 \ + --hash=sha256:9b5db6052055d34d41230fb78d7c439c23dc536a9896f6cb039e8dd92cfc1263 \ + --hash=sha256:9d9a0dc7cbe9bec24c3f767c9122c41fe5a1bc43f47cd099d00d393e09769de4 \ + --hash=sha256:9dbdd9205662134957cf0c324f639bdc5031c0ca056e2369e238db75187c0f11 \ + --hash=sha256:9eea3ab2597a5e65fe65296e2d6a84570845a6b55532d90333d740d48bbc850a \ + --hash=sha256:a2028475ba855475b8b4d3cfeb4994269c967aea8b9892dfba907f4263a863a3 \ + --hash=sha256:a3a370082ce34d0612f421e15fe011c53bb1feff21a26d06ad4fb244dab5a375 \ + --hash=sha256:a545775cfe815855ea32d7c27731d79da358ef2055b4a25830231b1622dd18aa \ + --hash=sha256:a5cbd90ecf0fc62e64726917ad083b73001f0563657a87ec3c0b504e277dc90d \ + --hash=sha256:a6d095662e73e74f0a49988e0593373e243e3a52e27bfeea0a859e88acf4a0f5 \ + --hash=sha256:a6dac12ff6b846103483683f60c5f8fee205121adc58ffd87e90a90a3af69e99 \ + --hash=sha256:a951ad59cad9145664a730d3036b40b844e74d2d3683da40111463cd3a83845d \ + --hash=sha256:aa1099b956fb795e686d073568f6dc002a0bb89765ea6d5b055dd7d9bf1b116c \ + --hash=sha256:aa2bb0b37202dca27175591f761108b5d34096ade1191ffe4808bdf6b1571488 \ + --hash=sha256:aae2ee51122d3ae968a3837d97dc24a0aeebb0dea23694422cd172bd30017cd6 \ + --hash=sha256:ab743e9bc90c1f73552ec33e10e3331315acd2c397b36065b591b0181de533cc \ + --hash=sha256:ac00177c4831ffa650f8609e4bdddd5fe09c03b1c0c47acece7e6ea20421598b \ + --hash=sha256:ac13b004224fb341e1e25a1ed5e19d32f57cdb2a403e01f003b46f051a550f6f \ + --hash=sha256:acaf604462bf330b0d07e7a07c1d6e4adac79e5fb13e9c5140590542cafacc00 \ + --hash=sha256:ae31a1a1db2ee6cc2942fccaf695c934bc7f3db9f2133a3fef1f367cf1a4ab10 \ + --hash=sha256:ae4a097991662cd4fff0ddc74e0fe7874f82e00042fa0ea00855645ed0c79598 \ + --hash=sha256:aea996a6aba25260827c9ea511d1addfde2da9eb686ac961838509086188b7e6 \ + --hash=sha256:b39b69b347e5e47a3b5b8cfc005c68c1ba347474e3960236c4944a8ecd174962 \ + --hash=sha256:b54e7e13267d49ffbfe68e25b3cbd774dab38fa37238f71265e91b36146eb21c \ + --hash=sha256:b9af956078716df40d985fb0dfeb2c2120c5ca92ba4ff4b388acfd01cdc14d08 \ + --hash=sha256:ba2f37ee79e6338845261a3c5b1784e5d1acdff2c0785b284f1b633033d136ab \ + --hash=sha256:ba501e667c17d8411f98e67a022d9604ef179aff0e459b7e292c796837c13573 \ + --hash=sha256:baf3775a2635e5a11fbd5e4e64ee69c7e86875d224a5c72aca4c141064589a90 \ + --hash=sha256:bb57753e36e4855b8ca375069482250a6246372331a3e4f3407eaebb007443f5 \ + --hash=sha256:bd6c173f04743d483881bffa1478d5a4624475b8cd1d2194956a75548e191c18 \ + --hash=sha256:be47f99644b208bff7766314013f9acf57b056b04191d570d68ad14022cf5b1d \ + --hash=sha256:c010f5581d9c612804cc59fcf7b524b707fbcb72828551237ab545bb5c7034af \ + --hash=sha256:c1dcc36dcb96abc02236e182d17e0f71430152a6c2c7447421da2d2dc144edea \ + --hash=sha256:c428c6c31eb5f4277d7f8eccaf767fbd548ddd5ce3c8b4f4cbbfab3d96b5904c \ + --hash=sha256:c658c50ac0c98cd755a2dd50b7977d3bca7df401dcc47fbdfa87db53ef7d4e8b \ + --hash=sha256:c71fb0d56c920c269cd3e2e3fe7c610e3f1fdb21a6ce60efa6430ff63676cea6 \ + --hash=sha256:c7b742bf31c88566b4bb6335a7f393bb322e580b6bb98df7bd0c25e6e3519ce8 \ + --hash=sha256:cc0329df4caaceb950d2f580b5ac716a377f7059624a0bafaeaf8a218c6ed774 \ + --hash=sha256:cc5d36d96478aa9c60654bd932525bf32964c62a7281eafdf16d85003a8d6004 \ + --hash=sha256:ce854f5f478050ade5a238731c4ca985a7d3b3cb53ff600a9b5c3b689b5f0a7a \ + --hash=sha256:ced3fdd71aaa83ce593746c2edb42b7a59cb4c19c8b5c407781c72e493aae55a \ + --hash=sha256:cee5dd7c6fb5dd52a0fe2a740f9bc6e3593f5f8b1788bde49de02086f30182b2 \ + --hash=sha256:cfa1c0cc3a8f9f53f1243a5a99ac36fd003880199383b37672e86ddda9cb07e2 \ + --hash=sha256:d1ee1e296209fdce05b81b663250eefa02213a2da7b41bf26f7829b8ba3545aa \ + --hash=sha256:d59b75732e9b6f27388e10c14b0259cc5f2e48c78627d185e6a177b58ad3cffe \ + --hash=sha256:d63600d620ad0064c3a748b950ac5ea38a80190e5498532efefa4b7b3f1da1f3 \ + --hash=sha256:dd732602a7009217f658d5863d12d79d373a4de0eebc111094bcdd3bb8e0a6cc \ + --hash=sha256:e06efa066f7dbadbc84ebc126a97c452a6451dfcf589d89d788484949e1cf795 \ + --hash=sha256:e199fb99720074809a7720f1c0b4d919eea8b87e88713e0f8f602f7bef543d9d \ + --hash=sha256:e4b018dc5a0eee4676e38fe84a47a427816c590b93b55d9025274ec4d6ffc2dc \ + --hash=sha256:e6621fb2a4988d6e53eedc455e5903e2679f3967b8acb3d639f1b63c14a2e893 \ + --hash=sha256:e71c909f353863b2b89c83de2ebed71ea6d0df8a6ef65a128193c5e650766bef \ + --hash=sha256:e90251c0c7bdd54a100a0dce3c07b7e637278c93af29dbf78ebb89a58c4bac7d \ + --hash=sha256:e9fbdce1e47394b09bc9f26ab117dfc8d6491977a11d86f592bb42c779db2fda \ + --hash=sha256:eb12fb2ba69ffa05f8695f61c69e591dc4b4a12ac3757ac8af8adb259bf56d17 \ + --hash=sha256:eda059b6bc8bc0812d626fd91a7ce01bf583df0a61296eff390fd94141a34e30 \ + --hash=sha256:f03ac127268b43ef4fe9e6ab6794a6794b49485a0cc0c1db79876d2f33f75bc7 \ + --hash=sha256:f298e218441525d3794428b4c8b8fb8662c6d3ea79925d4807ee6b9a96a3bca5 \ + --hash=sha256:f5542f9b941279d82d41eb0aa9f98eba36fe4df5c7086c651df7944935b37182 \ + --hash=sha256:f6f7deae3feb4edfa2efaf7c574fe88cbf055038a6abdb40188e4fff66d5699f \ + --hash=sha256:f9b1e28d0e8dbfa858abdba91d6b547beaf2df1a59bec6da6faae7b96a4991a9 \ + --hash=sha256:f9f8405c2c758532c74fed975dbee57be1f31a6e865c031870c79a6ed3212ada \ + --hash=sha256:fa48b1b63d639f9483e0633e092f5851e2348c352f1f9bb6c8182f87884ef876 \ + --hash=sha256:fb78f6e7fcd8ad785d28cd577168bc1aaee827b25bb8755638f694794ea98f0a \ + --hash=sha256:fbc597639158fd7c14d55e808718848319540f51b0e6746e3eefa59723a4a348 \ + --hash=sha256:fce8cbd4997efeb450bd298b54f755dcdff18d496f7a5ddbb4867c6d7c88fdc3 \ + --hash=sha256:fd0350afdc3aabd5576f60ea109228bd5538139713c7b094c5cd27c73a98bc6f \ + --hash=sha256:fd0a274c0e5f9a21565cd9d3dd749b61f96b7aa1e20a93aa1ba4029518f2e5c0 \ + --hash=sha256:fdb8a068947befafba9952162645dc2fecaeb400e64584829ed5e9b2fbe21a7f + # via requests +click==8.5.0 \ + --hash=sha256:255bc9599cf7748b4b1a446ccc735421bd08a2ae529a8b88597d3de5664ee360 + # via + # streamlit + # uvicorn +h11==0.16.0 \ + --hash=sha256:63cf8bbe7522de3bf65932fda1d9c2772064ffb3dae62d55932da54b31cb6c86 + # via uvicorn +httptools==0.8.0 \ + --hash=sha256:0770728beb05094c809b98e814edff5fef69d26ad7d21185f2f6d5884a0ba683 \ + --hash=sha256:0ea897f0c729581ebf72131a438a7932d9b14efef72d75ada966700cac3caaeb \ + --hash=sha256:159e9ab5f701ccd42e555a12f1ad8ff69702910fc1c996cf2bb66e5fcb7a231b \ + --hash=sha256:19d1ee275bb59ba2643ba9a3a1e51cc0c788caf2b8df506368e03f56fdd08527 \ + --hash=sha256:20b4aac66ff65f7db06a375808b78f42a94970aa22e826b3cb2b43eb09174124 \ + --hash=sha256:2a021c3a8e65cc125390d72f59b968afca3bdcaff25bd67965e0a055a14946ca \ + --hash=sha256:2c032fa028f46871ec7e1fc59fc15e8023eab3e6bbe6ece786a1611719a5d081 \ + --hash=sha256:2d689918c15a013c65ef52d9fd495d766893ab831a2c8d89f2ac5940a5df847c \ + --hash=sha256:384c17174464c8e873398b7af24f0b1f44d992c820328413951a625323155d77 \ + --hash=sha256:425f83884fd6343828d8c565f046cb72b6d19063f6924093e11bcd8e1548cd09 \ + --hash=sha256:48774d39cbb70e2b1f71f88852a3087ae1d3a1eb80482bb48c13067ab080c14f \ + --hash=sha256:52dd695b865fe96d9d2b16b64a895f3f57bf3cb064e8383cd3b5713a069e8085 \ + --hash=sha256:57278e6fa0424c42a8a3e454828ab4f0aff27b40cddf9679579b98c6dce6a376 \ + --hash=sha256:5931891fb7b441b8a3853cf1b85c82c903defce084dd5f6771ca46e31bf862c5 \ + --hash=sha256:5d7fa4ba7292c1139c0526f0b5aad507c6263c948206ea1b1cbca015c8af1b62 \ + --hash=sha256:5eb911c515b96ee44bbd861e42cbefc488681d450545b1d02127f6136e3a86f5 \ + --hash=sha256:614ceea8ea606848bece2338ac03b3ce5324bcb4be8dc7d377ed708012fa4db8 \ + --hash=sha256:6a43c9dd399758ccc0531acb0a3c4a6c299ee893ee9400e9c893b7bdcfae0681 \ + --hash=sha256:7685df791fad561384bfb139e77fde27a1ffd93134e016f95a0db424ffbf77b1 \ + --hash=sha256:7b71e7d7031928c650e1006e6c03e911bf967f7c69c011d37d541c3e7bf55005 \ + --hash=sha256:880490234c10f70a9830743097e8958d6e4b9f5a0ffc24515023afeef984054d \ + --hash=sha256:88bdd940f2b5d487b4d032c6afa5489a7dc4694410d43de3c38c4fb3af0dc45d \ + --hash=sha256:88eead8ec8680a9f146c655bc88445a325bd7921cfd8194c7337e9467282427d \ + --hash=sha256:9518c406d7b310f05adb1a37f80acabac40504a575d7c0da6d3e365c695ac20d \ + --hash=sha256:9878eb2785ba5eb70631ad269b37976f73d647955e26c91d490eb8a4edfda4ba \ + --hash=sha256:9fc1644f415372cec4f8a5be3a64183737398f10dbb1263602a036427fe75247 \ + --hash=sha256:a1afd7c9fbff0d9f5d489c4ce2768bd09c84a46ddefc7161e6aa82ae35c85745 \ + --hash=sha256:a1b4c8e7a489a0d750d91894e9a8cdc295838f1924c0ca903ae993456fddec07 \ + --hash=sha256:a3b7387147361c3fd47a0bde763c5c91b5b4cd4dc9989b8ece84ff436c99843b \ + --hash=sha256:a6f21e2a3b0067bbe7f67e34cfd16276af556e5e52f4c7503be0cb5f90e905e4 \ + --hash=sha256:b15fc622b0f869d19207c4089a501d9bcc63ca5e071ffdd2f03f922df882dcb2 \ + --hash=sha256:b205e5f5523fa039679da0dfe5a10132b2a4abeae6a86fdd1ddc035f7f836557 \ + --hash=sha256:bbb8caadb2b742d293169d2b458b5c001ef70e3158704aa3d3ef9597624c5d1d \ + --hash=sha256:bf3b6f807c8541503cecfbb8a8dffb385640d0d96102f3d112aa8740f9b7c826 \ + --hash=sha256:c08ffe3e79756e0963cbc8fe410139f38a5884874b6f2e17761bef6563fdcd9b \ + --hash=sha256:c0d726cc107fceb7d45f978483b4b70dd8caa836f5914d3434bb18628eb73813 \ + --hash=sha256:c4a9f1707e4823d54dfec6c33fa3697d302aed536ed352a7ebb5a061ddb869d0 \ + --hash=sha256:cd96f29b4bab1d42fa6e3d008711c75e0f79e94e06827330160e3a304227f150 \ + --hash=sha256:d76ad7b951387e3632c8716a9bb03ac5b45c5f16119aa409db0459520887944e \ + --hash=sha256:da684f2e1aa2ee9bdcb083f3f3a68c5956750b375bc5df864d3a5f0c42a40b77 \ + --hash=sha256:de1ed58a974e75d56560acc7e7fed01a454994429456f65209789992e41f2568 \ + --hash=sha256:de242a49b5d18e0a8776e654e9f6bf6d89f3875a5c35b425a0e7ce940feb3fd6 \ + --hash=sha256:df31ef5494f406ab6cf827b7e64a22841c6e2d654100e6a116ea15b46d02d5e8 \ + --hash=sha256:e93c227b595c6926c1acee96891dd9da4be338cfbe82e5cd3bb9d8dd7dc4ac0b \ + --hash=sha256:eb3028cca2fc0a6d720e52ef61d8ebb62fcbfeb1de56874546d858d3f25a26b7 \ + --hash=sha256:ed377e64805bdba4943c82717333f8f8603a13b09aff9cead2717c6c817fb168 \ + --hash=sha256:ef7c3c97f4311c7be57e2986629df89d49cb434dbff78eafcd48c2bff986b15a \ + --hash=sha256:f256d6ce930c52ca1cb2a960b7da03548c454e7d28b06059ad41bfe789036ce0 \ + --hash=sha256:fe2a4c95aeba2209434e7b31172da572846cae8ca0bf1e7013e61b99fbbf5e72 + # via streamlit +idna==3.19 \ + --hash=sha256:815e7be7a7806d54abb586dc943addc79e8b2ee16915059658cbeff4b1b43bf4 + # via + # anyio + # requests +iniconfig==2.3.0 \ + --hash=sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12 + # via pytest +itsdangerous==2.2.0 \ + --hash=sha256:c6242fc49e35958c8b15141343aa660db5fc54d4f13a1db01a3f5891b98700ef + # via streamlit +jinja2==3.1.6 \ + --hash=sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67 + # via + # altair + # pydeck +jmespath==1.1.0 \ + --hash=sha256:a5663118de4908c91729bea0acadca56526eb2698e83de10cd116ae0f4e97c64 + # via + # boto3 + # botocore +jsonschema==4.26.0 \ + --hash=sha256:d489f15263b8d200f8387e64b4c3a75f06629559fb73deb8fdfb525f2dab50ce + # via altair +jsonschema-specifications==2025.9.1 \ + --hash=sha256:98802fee3a11ee76ecaca44429fda8a41bff98b00a0f2838151b113f210cc6fe + # via jsonschema +markupsafe==3.0.3 \ + --hash=sha256:0303439a41979d9e74d18ff5e2dd8c43ed6c6001fd40e5bf2e43f7bd9bbc523f \ + --hash=sha256:068f375c472b3e7acbe2d5318dea141359e6900156b5b2ba06a30b169086b91a \ + --hash=sha256:0bf2a864d67e76e5c9a34dc26ec616a66b9888e25e7b9460e1c76d3293bd9dbf \ + --hash=sha256:0db14f5dafddbb6d9208827849fad01f1a2609380add406671a26386cdf15a19 \ + --hash=sha256:0eb9ff8191e8498cca014656ae6b8d61f39da5f95b488805da4bb029cccbfbaf \ + --hash=sha256:0f4b68347f8c5eab4a13419215bdfd7f8c9b19f2b25520968adfad23eb0ce60c \ + --hash=sha256:1085e7fbddd3be5f89cc898938f42c0b3c711fdcb37d75221de2666af647c175 \ + --hash=sha256:116bb52f642a37c115f517494ea5feb03889e04df47eeff5b130b1808ce7c219 \ + --hash=sha256:12c63dfb4a98206f045aa9563db46507995f7ef6d83b2f68eda65c307c6829eb \ + --hash=sha256:133a43e73a802c5562be9bbcd03d090aa5a1fe899db609c29e8c8d815c5f6de6 \ + --hash=sha256:1353ef0c1b138e1907ae78e2f6c63ff67501122006b0f9abad68fda5f4ffc6ab \ + --hash=sha256:15d939a21d546304880945ca1ecb8a039db6b4dc49b2c5a400387cdae6a62e26 \ + --hash=sha256:177b5253b2834fe3678cb4a5f0059808258584c559193998be2601324fdeafb1 \ + --hash=sha256:1872df69a4de6aead3491198eaf13810b565bdbeec3ae2dc8780f14458ec73ce \ + --hash=sha256:1b4b79e8ebf6b55351f0d91fe80f893b4743f104bff22e90697db1590e47a218 \ + --hash=sha256:1b52b4fb9df4eb9ae465f8d0c228a00624de2334f216f178a995ccdcf82c4634 \ + --hash=sha256:1ba88449deb3de88bd40044603fafffb7bc2b055d626a330323a9ed736661695 \ + --hash=sha256:1cc7ea17a6824959616c525620e387f6dd30fec8cb44f649e31712db02123dad \ + --hash=sha256:218551f6df4868a8d527e3062d0fb968682fe92054e89978594c28e642c43a73 \ + --hash=sha256:26a5784ded40c9e318cfc2bdb30fe164bdb8665ded9cd64d500a34fb42067b1c \ + --hash=sha256:2713baf880df847f2bece4230d4d094280f4e67b1e813eec43b4c0e144a34ffe \ + --hash=sha256:2a15a08b17dd94c53a1da0438822d70ebcd13f8c3a95abe3a9ef9f11a94830aa \ + --hash=sha256:2f981d352f04553a7171b8e44369f2af4055f888dfb147d55e42d29e29e74559 \ + --hash=sha256:32001d6a8fc98c8cb5c947787c5d08b0a50663d139f1305bac5885d98d9b40fa \ + --hash=sha256:3524b778fe5cfb3452a09d31e7b5adefeea8c5be1d43c4f810ba09f2ceb29d37 \ + --hash=sha256:3537e01efc9d4dccdf77221fb1cb3b8e1a38d5428920e0657ce299b20324d758 \ + --hash=sha256:35add3b638a5d900e807944a078b51922212fb3dedb01633a8defc4b01a3c85f \ + --hash=sha256:38664109c14ffc9e7437e86b4dceb442b0096dfe3541d7864d9cbe1da4cf36c8 \ + --hash=sha256:3a7e8ae81ae39e62a41ec302f972ba6ae23a5c5396c8e60113e9066ef893da0d \ + --hash=sha256:3b562dd9e9ea93f13d53989d23a7e775fdfd1066c33494ff43f5418bc8c58a5c \ + --hash=sha256:457a69a9577064c05a97c41f4e65148652db078a3a509039e64d3467b9e7ef97 \ + --hash=sha256:4bd4cd07944443f5a265608cc6aab442e4f74dff8088b0dfc8238647b8f6ae9a \ + --hash=sha256:4e885a3d1efa2eadc93c894a21770e4bc67899e3543680313b09f139e149ab19 \ + --hash=sha256:4faffd047e07c38848ce017e8725090413cd80cbc23d86e55c587bf979e579c9 \ + --hash=sha256:509fa21c6deb7a7a273d629cf5ec029bc209d1a51178615ddf718f5918992ab9 \ + --hash=sha256:5678211cb9333a6468fb8d8be0305520aa073f50d17f089b5b4b477ea6e67fdc \ + --hash=sha256:591ae9f2a647529ca990bc681daebdd52c8791ff06c2bfa05b65163e28102ef2 \ + --hash=sha256:5a7d5dc5140555cf21a6fefbdbf8723f06fcd2f63ef108f2854de715e4422cb4 \ + --hash=sha256:69c0b73548bc525c8cb9a251cddf1931d1db4d2258e9599c28c07ef3580ef354 \ + --hash=sha256:6b5420a1d9450023228968e7e6a9ce57f65d148ab56d2313fcd589eee96a7a50 \ + --hash=sha256:729586769a26dbceff69f7a7dbbf59ab6572b99d94576a5592625d5b411576b9 \ + --hash=sha256:77f0643abe7495da77fb436f50f8dab76dbc6e5fd25d39589a0f1fe6548bfa2b \ + --hash=sha256:795e7751525cae078558e679d646ae45574b47ed6e7771863fcc079a6171a0fc \ + --hash=sha256:7be7b61bb172e1ed687f1754f8e7484f1c8019780f6f6b0786e76bb01c2ae115 \ + --hash=sha256:7c3fb7d25180895632e5d3148dbdc29ea38ccb7fd210aa27acbd1201a1902c6e \ + --hash=sha256:7e68f88e5b8799aa49c85cd116c932a1ac15caaa3f5db09087854d218359e485 \ + --hash=sha256:83891d0e9fb81a825d9a6d61e3f07550ca70a076484292a70fde82c4b807286f \ + --hash=sha256:8485f406a96febb5140bfeca44a73e3ce5116b2501ac54fe953e488fb1d03b12 \ + --hash=sha256:8709b08f4a89aa7586de0aadc8da56180242ee0ada3999749b183aa23df95025 \ + --hash=sha256:8f71bc33915be5186016f675cd83a1e08523649b0e33efdb898db577ef5bb009 \ + --hash=sha256:915c04ba3851909ce68ccc2b8e2cd691618c4dc4c4232fb7982bca3f41fd8c3d \ + --hash=sha256:949b8d66bc381ee8b007cd945914c721d9aba8e27f71959d750a46f7c282b20b \ + --hash=sha256:94c6f0bb423f739146aec64595853541634bde58b2135f27f61c1ffd1cd4d16a \ + --hash=sha256:9a1abfdc021a164803f4d485104931fb8f8c1efd55bc6b748d2f5774e78b62c5 \ + --hash=sha256:9b79b7a16f7fedff2495d684f2b59b0457c3b493778c9eed31111be64d58279f \ + --hash=sha256:a320721ab5a1aba0a233739394eb907f8c8da5c98c9181d1161e77a0c8e36f2d \ + --hash=sha256:a4afe79fb3de0b7097d81da19090f4df4f8d3a2b3adaa8764138aac2e44f3af1 \ + --hash=sha256:ad2cf8aa28b8c020ab2fc8287b0f823d0a7d8630784c31e9ee5edea20f406287 \ + --hash=sha256:b8512a91625c9b3da6f127803b166b629725e68af71f8184ae7e7d54686a56d6 \ + --hash=sha256:bc51efed119bc9cfdf792cdeaa4d67e8f6fcccab66ed4bfdd6bde3e59bfcbb2f \ + --hash=sha256:bdc919ead48f234740ad807933cdf545180bfbe9342c2bb451556db2ed958581 \ + --hash=sha256:bdd37121970bfd8be76c5fb069c7751683bdf373db1ed6c010162b2a130248ed \ + --hash=sha256:be8813b57049a7dc738189df53d69395eba14fb99345e0a5994914a3864c8a4b \ + --hash=sha256:c0c0b3ade1c0b13b936d7970b1d37a57acde9199dc2aecc4c336773e1d86049c \ + --hash=sha256:c47a551199eb8eb2121d4f0f15ae0f923d31350ab9280078d1e5f12b249e0026 \ + --hash=sha256:c4ffb7ebf07cfe8931028e3e4c85f0357459a3f9f9490886198848f4fa002ec8 \ + --hash=sha256:ccfcd093f13f0f0b7fdd0f198b90053bf7b2f02a3927a30e63f3ccc9df56b676 \ + --hash=sha256:d2ee202e79d8ed691ceebae8e0486bd9a2cd4794cec4824e1c99b6f5009502f6 \ + --hash=sha256:d53197da72cc091b024dd97249dfc7794d6a56530370992a5e1a08983ad9230e \ + --hash=sha256:d6dd0be5b5b189d31db7cda48b91d7e0a9795f31430b7f271219ab30f1d3ac9d \ + --hash=sha256:d88b440e37a16e651bda4c7c2b930eb586fd15ca7406cb39e211fcff3bf3017d \ + --hash=sha256:de8a88e63464af587c950061a5e6a67d3632e36df62b986892331d4620a35c01 \ + --hash=sha256:df2449253ef108a379b8b5d6b43f4b1a8e81a061d6537becd5582fba5f9196d7 \ + --hash=sha256:e1c1493fb6e50ab01d20a22826e57520f1284df32f2d8601fdd90b6304601419 \ + --hash=sha256:e1cf1972137e83c5d4c136c43ced9ac51d0e124706ee1c8aa8532c1287fa8795 \ + --hash=sha256:e2103a929dfa2fcaf9bb4e7c091983a49c9ac3b19c9061b6d5427dd7d14d81a1 \ + --hash=sha256:e56b7d45a839a697b5eb268c82a71bd8c7f6c94d6fd50c3d577fa39a9f1409f5 \ + --hash=sha256:e8afc3f2ccfa24215f8cb28dcf43f0113ac3c37c2f0f0806d8c70e4228c5cf4d \ + --hash=sha256:e8fc20152abba6b83724d7ff268c249fa196d8259ff481f3b1476383f8f24e42 \ + --hash=sha256:eaa9599de571d72e2daf60164784109f19978b327a3910d3e9de8c97b5b70cfe \ + --hash=sha256:ec15a59cf5af7be74194f7ab02d0f59a62bdcf1a537677ce67a2537c9b87fcda \ + --hash=sha256:f190daf01f13c72eac4efd5c430a8de82489d9cff23c364c3ea822545032993e \ + --hash=sha256:f34c41761022dd093b4b6896d4810782ffbabe30f2d443ff5f083e0cbbb8c737 \ + --hash=sha256:f3e98bb3798ead92273dc0e5fd0f31ade220f59a266ffd8a4f6065e0a3ce0523 \ + --hash=sha256:f42d0984e947b8adf7dd6dde396e720934d12c506ce84eea8476409563607591 \ + --hash=sha256:f71a396b3bf33ecaa1626c255855702aca4d3d9fea5e051b41ac59a9c1c41edc \ + --hash=sha256:f9e130248f4462aaa8e2552d547f36ddadbeaa573879158d721bbd33dfe4743a \ + --hash=sha256:fed51ac40f757d41b7c48425901843666a6677e3e8eb0abcff09e4ba6e664f50 + # via jinja2 +narwhals==2.26.0 \ + --hash=sha256:29326d74f107c347fd1009bd58e38d9f7c7c5b51e6de97bc93dbc325d9038b54 + # via + # altair + # plotly +numpy==2.5.3 \ + --hash=sha256:012e66aca395d795496446e52aeeb5866312a5d4d3f27da270e5a0b43f70dc5c \ + --hash=sha256:09d5a423c71ad5feb5625844ad58050e35df43871004b52ac9c0ad44a56775be \ + --hash=sha256:09ffa5d903faeaa5c4dd05009cf81c8bab9f2cb37c548b8d39b65b4cfa7c97f7 \ + --hash=sha256:0a59a421a32580a009e8a1751345bf829631b990dc1794b80514ab722b435def \ + --hash=sha256:116f96cadd935c6122e9228d676fe7ede19e741f5c8bb1c3cddbe0c51ccebea2 \ + --hash=sha256:1302b90c0e52281681b2975adfe8a860cb7b12216a27b4b0b4207c44bf7bccf0 \ + --hash=sha256:15aa985ac73a8db02db7663381aa109510449d3819d37206caed27b33a65a8a6 \ + --hash=sha256:1aad64d99730d013cfc6debafed22783b4fc5a7f4b8bc744d2d8cf7dcc880551 \ + --hash=sha256:1c80eabb4035ecf4ca9cd49cde8a9fdd69a729e63e6474887d1523ade7aa277f \ + --hash=sha256:1f3ed25271581281f2fccb1adcedfcde4c07362eec69189b50baf6f90e3ae159 \ + --hash=sha256:1fb6f8fb9ff0b3a69f52c66ce397b0246583e9f28616231b0e32ca49259a5fa6 \ + --hash=sha256:214045a5bf00113a146ab9ee9730c44501af6723cdf1f6830932f7b5ef2e7af0 \ + --hash=sha256:26e15e4aecd8617dfbaecb37d223e365d7b39411fba20454be2670a96aa74cb5 \ + --hash=sha256:2c25dfa72943e4336ddb6b0ee4277b47a0c85bede0807530ec68103bf58e2c10 \ + --hash=sha256:2d8240cb4c16fd831074aa2b2cf9fc54664d826341d61c372245b96a74a49a9a \ + --hash=sha256:350ba9783ce969cf9f7ce6e6a9a58e1a6e2a19ca025b7ee448c4db727706212a \ + --hash=sha256:4c8a6d2ebce6305fd82fbefca827775437147052a976ee7c94b36a0c1b52ac6c \ + --hash=sha256:4f8929ee6c96bfbd7b4ed2032e0c03af86fe1826740ab61ddabf9072d06e57ff \ + --hash=sha256:536f963710a4e63934d80ac0dc4f478804a83e9a84b6828018f25d09953ada33 \ + --hash=sha256:54a115e5a73b8fc44f0cebef486365a1894b5c9760685d4558b72b7c3eb846e0 \ + --hash=sha256:595d020938c84e320bcf40ad71089e108eac0d377cd018e14a8c094f39e98d85 \ + --hash=sha256:66a78fe4556c60aceda5916f9eacd638b18e9e681016ec302dcb4682d6d4d034 \ + --hash=sha256:6b05c171afb3aa07adbd20abc00aea86fe375beb0fdb9ef780ec5b7f63bab1c0 \ + --hash=sha256:6cef4bb1706dfec49243c05d921eefb4e190d41e2528b30d8035ea1f36b4c24a \ + --hash=sha256:6f24021b9f22bc6301c37b196974a92c1c18dccedb6fef3dd252e95f2d6adbe4 \ + --hash=sha256:71b39d9f935b6ec0f8753e3e2afb51e3efba6f2e05b68b32a40754d24bcd4a3c \ + --hash=sha256:71cad2b2a7451ab79d8f5e71b453485b6775963d5cf794179144a7463fe6e8ec \ + --hash=sha256:76c2c1e6bfa5c84adc6434dfbf013aa92096a7985221762c8f11fedfd20fff58 \ + --hash=sha256:8617bbfae4486cf99c9f899966699428d19da931d06ca94ad3da986c76e15997 \ + --hash=sha256:86bff898a431c0fb71f7610b75726e75a54d47b37edc9d537f48de63bb3c0b90 \ + --hash=sha256:8e4dd766076855b5ff7ea52fa5f07ce26286726e0f8bff446b7739d02e6ea204 \ + --hash=sha256:92f30e89b8ee0ecf363033576c422b2f58fed6a80bed0aa48dff6d14c654663e \ + --hash=sha256:93e1f5447e2b1e479d7bd74701e84746b86450cff1fc368b132d195e2b8f8211 \ + --hash=sha256:9a37475425b431b4d060f23b4f52cd2f3aef6bc7c654bd760adf0040eec9d435 \ + --hash=sha256:9deb49575e5b0b94ed72c8a64ec4d033381adc27e9060ae842971f697ba96104 \ + --hash=sha256:a5fa86b80fd24bcd1aff83ad23be44ea323de3f787be8f8b15d4a65621e25321 \ + --hash=sha256:a6391fafaba97500887132cd582abc6e19452b1ac775a47caa7b24490e152058 \ + --hash=sha256:a72f874bc9e10e4b8f80426fb49716d5141f64442a0c8418065093ec8017fbb0 \ + --hash=sha256:ac7bb1c52d445bd4f8f7f97fefe6abc3a084dc4d63df50d79b17fa2b78e89297 \ + --hash=sha256:adc1ada2662f8a5f960b8a10d9986897e7499ef07e06d4cfe7197f8cce923c07 \ + --hash=sha256:b00eefbcf0f292945c4b4dec2ae845389ef5bcdcd596e6e4328051db5b5ba694 \ + --hash=sha256:b0521d0f4aebb6e06189451025fa17a913287b13c03d5fe05c017333b654ea5b \ + --hash=sha256:b5d93cf48f687479941d12b69c873ad2cc76bbd487f0091c2200636497f34034 \ + --hash=sha256:b7e18c623bb5c95acb3b3328861272816ba199fb531921c5d6d0b675f1fde9e3 \ + --hash=sha256:bd4cb9ad3c7889b9b3fe0a9a9fb5d2ed26f9879bff2608d9f01aed147a20d231 \ + --hash=sha256:be5a8381859b6da607c84f4f7d6847725f1cf1853ef8a2c9e115b7d58bef47dc \ + --hash=sha256:befa1ae5bd6030b3f512b43ff3fa5290bbed6b84411a44244b14adf835f5b89d \ + --hash=sha256:bf63afbe037eb5d2fe87fbcc7778e61da53ebaf21d938a4515aa73b62532a5d4 \ + --hash=sha256:c00abe94c1a69d75d827dcf1c025b25c8a45d230b3bcd77a9020883a1b047653 \ + --hash=sha256:c2381f82999704f818e2c987a865050e285ec3621262c66d40f5a96c8f899f8e \ + --hash=sha256:c76d5dde9f445058f83d0c02af00557a4db91de9a9a57c0df87d1535001d654b \ + --hash=sha256:cb189f09db39283b26bfd061ec16189e14f71c6755207f72a0f7540867afe5b9 \ + --hash=sha256:ccb32e0525d29e8b0572eb84c9a57af0e7a4e615726927506f55063c62414034 \ + --hash=sha256:ccbc4665079665c3cf3bab4db9f6b095370cd6437d66be549b6c2a1fd19e1958 \ + --hash=sha256:d1c89973648c85069c5046ad460f7b8a00218b29a2e42359ac8cc63e9ab94832 \ + --hash=sha256:e01c918ac3d48e18a927cf7b14a26a3e29ff2bdf2eacb976da0aecd6a43ed034 \ + --hash=sha256:e6ab667ba76450084eb64013762c438ea76d9d29cc676dcd6c2e9892ba37f841 \ + --hash=sha256:e931e4f499e0dc7ef29d269a8e5b35dd722e5d14be07df6240166ea7c6532fae \ + --hash=sha256:f54660b0eb6b0b9f36e7fe1cdfdff472028dd0d14acd9b9b65098efbad059469 \ + --hash=sha256:f59a878c33d6b88122d80d239bb3b845d58708750b0cb06a09aebb9b18ec696c \ + --hash=sha256:f7fabeb6cea87d65f3b926de33d03fb016cfdc29314c90974383b5582ae72891 \ + --hash=sha256:f9579f383d1bf9df80081e72760e84960a7fd4f88cf0c9e535a8597c9bb646f5 \ + --hash=sha256:f9a2353b37a1a9e78fd82b27ad7e2a32a2d036604d18f02b05e3136c62ca3b09 \ + --hash=sha256:fc36dc566135b5eceec4cf89758fcb719266a019ef07dae1754ae7c9f617ef3e \ + --hash=sha256:ffdc76bfcae6b255dff75202c5e7feaf95b40246bc0a17944facc1fecf9f79ab + # via + # pandas + # pydeck + # streamlit +packaging==26.3 \ + --hash=sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c + # via + # altair + # plotly + # pytest + # streamlit +pandas==3.0.5 \ + --hash=sha256:08d24fe11a17dc33bd6e937dc9c665f9cba08fbdc9f657f405713515febe300d \ + --hash=sha256:0d298e951f23016ce4699951d044ae6418dbc91bf68cefca0f77666fcbb4e5c6 \ + --hash=sha256:0fac0010c75e4efb6b99e249c183a8993ce0dc95c240f9b120a5e67c727b7928 \ + --hash=sha256:1c10461f6eeb35d8f05b6184c65c8b9991663b66c46b1d559b682cb34ae7c6ea \ + --hash=sha256:25ff585b972a18ef1fe9ffa3ac6544d9950508aa76832e5147640b6022821e49 \ + --hash=sha256:2946e77e4a53cd248cbde631a12f0e51c8324ce354c3eba4d20147c1ad6f4282 \ + --hash=sha256:2a29c53d85ea98c5e792c59ef82ee9fbe6ca902c0d0adb6b23f45ef894cd7bf6 \ + --hash=sha256:2c0cf1dd9b55a22d105fc46c1b489af3bd42264fcba7c66297bf47a9a1d9c78a \ + --hash=sha256:2f264fc46911cc8131a7322a16199bbf8e353d27c10bb211f5bd0c814324dc36 \ + --hash=sha256:303da736987d481074ca720ada325f8bd80c64ebc2d45ed79b29df3aaa4a26ca \ + --hash=sha256:3b2801bbb049d0136f6c213eae02b5fca969384fc2064dd728d8620552aa49da \ + --hash=sha256:3c5015fd1730fbf883647e88068176c839c102cea883ba1769a6f4593bfc1f8c \ + --hash=sha256:3c5ed2e7c06e91d340dfd091d7934f9bc82e4a36b95f647f090b9d1c9ac649da \ + --hash=sha256:4b11c36e218331d0387cbe3a0a5f75162357a1d92d57b2b08a336ff94b19b2be \ + --hash=sha256:5183427f5a8156d480f30333777bc978be93650a49a7c01db26adffe95b31e85 \ + --hash=sha256:53730687fcd161883b24e10411c06d6a4c0f2275d2faf3bb2bc25deb4ba8007c \ + --hash=sha256:66266d3442a5e8b3c90274c2b8b230bee42dd1c286bc822cc2f9f2c7e12b883e \ + --hash=sha256:679f4e85b30ddb1515458ab1e788d3e260eae369b1f78da7a3aa4cac8ebf4a2a \ + --hash=sha256:71ecc8fb7ed1a7aa4392316b5309a6347e8e7f832f38fd897846b3a1457a9298 \ + --hash=sha256:73fa87b08a7ef706f8aafda39ddaccf2a99047bea62d8c88a0361bcafb2237bc \ + --hash=sha256:80a611068e8a3ac23f7398c6c14eb46dc974e5cc9997f653e2dcfd1da74edd41 \ + --hash=sha256:960d3ebcf249f75206899fcd2c6de53f736b7265759ced0d3e559df0b8b709b0 \ + --hash=sha256:9e94c2c5ca43bd3ca32bf64d32308887b65e5f9bfd8023ea52755107a999f93b \ + --hash=sha256:a5ad3b02ed6bc7d7ae9b70804b2c6aa31827489d150f8e623ce82491b82085d7 \ + --hash=sha256:b1261758dfb6cf12c3cff8300e21cefad30e7ec709abb4c24ac7318e6a52462a \ + --hash=sha256:b173f5951ff6b8b0ec7675e20dff3c97b7e7a57dfcce387c2d7c5afe87cb7899 \ + --hash=sha256:b2acb4650527eec6822c3dadb2b771277b65e7dae7a267d4bccf65fd1bb3fbce \ + --hash=sha256:b58b1b39d46a5862e3fb18f50d1a201398619d16a0f9f73f57eea5583cf0e63c \ + --hash=sha256:b86765f268b56f7e665b93bce9d5df69dee7f99e595cf8fb839483ab315942a3 \ + --hash=sha256:c1c05a767fe8e5b4fe9e1c29806829c582052eaedb9120a3da83ba3f69e24a5b \ + --hash=sha256:c2e26bb46934b8a2ca0c3de1d3d606fc5f6746584791b2db264d58cf370e08dc \ + --hash=sha256:c597ecf5616b5c420372c1d4d4c00dbbfba7398bea857dcc984347e1ea48417b \ + --hash=sha256:cce3a9d11d2b1f82c69a27ec1f4948a170e2c403c4bbfa8cca62e3fdebe2ef3a \ + --hash=sha256:cd8f7c6dc98527058ee6264219343f5392240a6f1bfa654fc5d79023020d0c92 \ + --hash=sha256:cf52e1f61d229496da17dc7ab54acdee627357e7008fd4fecba3d0ba2937fa58 \ + --hash=sha256:d373ce03ffd84010ed9839fa73672a9c8256990532e158440c0085db7d914b34 \ + --hash=sha256:db172144bb56422bd157812f3b021eacc255451470b31e2c633c349490a1cfee \ + --hash=sha256:e2759e890db96dfcffdbd9b86c3c2cb6afaf58def482820317e06163ec1066cd \ + --hash=sha256:e819dd5f62966b481a8cb649d3299ebd886a1ea91ed5a99bf7ce77c98d18ab94 \ + --hash=sha256:ef01af4d8dc6cd2c8d6c7736f149574ef93fe043811eeb5e445f2647154b5040 \ + --hash=sha256:fa290c16964d4963fbfbc358928239cf3bd755b20e988ce944877def2f44471d + # via + # -r app/requirements.txt + # streamlit +pillow==12.3.0 \ + --hash=sha256:00808c5e14ef63ac5161091d242999076604ff74b883423a11e5d7bbb38bf756 \ + --hash=sha256:04f01d28a6aaff387bf842a13be313df23ba0597a44f1a976c9feb3c6ff4711a \ + --hash=sha256:06ff022112bc9cbf83b60f8e028d94ad87b60621706487e65f673de61610ab59 \ + --hash=sha256:0740a512dc522224c77d9aa5a8d70d8b7d73fb91f2c21125d8d025d3b8990e45 \ + --hash=sha256:0847a763afefb695bc912d7c131e7e0632d4edc1d8698f58ddabec8e46b8b6d3 \ + --hash=sha256:0dd2064cbc55aaec028ef5fbb60fa47bb6c3e7918e07ff17935284b227a9d2df \ + --hash=sha256:0feb2e9d6ad6c9e3c06effe9d00f3f1e618a6643273576b016f591e9315a7139 \ + --hash=sha256:10e41f0fbf1eec8cfd234b8fe17a4caac7c9d0db4c204d3c173a8f9f6ef3232b \ + --hash=sha256:1182d52bc2d5e5d7d0949503aa7e36d12f42205dc287e4883f407b1988820d39 \ + --hash=sha256:164b31cd1a0490ab6efae01aa5df49da7061be0af1b30e035b6e9a1bfe34ee6e \ + --hash=sha256:1657923d2d45afb66526e5b933e5b3052e6bdea196c90d3abb2424e18c77dae8 \ + --hash=sha256:186941b6aef820ad110fb01fb06eb925374dc3a21b17e37ec9a53b250c6fe2d1 \ + --hash=sha256:1cca606cd25738df4ed873d5ad46bbdb3d83b5cbca291f6b4ff13a4df6b0bbe8 \ + --hash=sha256:21900ce7ba264168cd50defae43cd75d25c833ad4ad6e73ffc5596d12e25ac89 \ + --hash=sha256:236ff70b9312fb68943c703aa842ca6a758abfa45ac187a5e7c1452e96ef72b5 \ + --hash=sha256:23aceaa007d6172b02c277f0cd359c79492bbb14f7072b4ede9fbcaf20648130 \ + --hash=sha256:23d27a3e0307ec2244cc51e7287b919aa68d097504ebe19df4e76a98a3eea5bd \ + --hash=sha256:24870b09b224f7ae3c39ed07d10e819d06f8720bc551847b1d623832b5b0e28d \ + --hash=sha256:251bf95b67017e27b13d82f5b326234ca62d70f9cf4c2b9032de2358a3b12c7b \ + --hash=sha256:25b9b82bb22e6e2b3cd07b39c68b7b862001226cb3dff7130d1cb914121b39ed \ + --hash=sha256:28ce87c5ab450a9dd970b52e5aca5fe63ed432d18a2eaddd1979a00a1ba24ace \ + --hash=sha256:300557495eb45ebb8aec96c2da9c4be642fbf7cd937278b4013ba894ea8eb0eb \ + --hash=sha256:30f2aa603c41533cc25c05acd0da21636e84a315768feb631c937177db558931 \ + --hash=sha256:331b624368d4f1d069149002f25f44bc61c8919ce8ddb3c45bdad8f6e2d89510 \ + --hash=sha256:37d6d0a00072fd2948eb22bce7e1475f34569d90c87c59f7a2ec59541b77f7a6 \ + --hash=sha256:37dc8f7bbb66efe481bb60defacef820c950c24713fb44962ed6aa2a50966de1 \ + --hash=sha256:3edce1d53195db527e0191f84b71d02022de0540bf43a16ed734ed7537b07385 \ + --hash=sha256:446c34dcc4324b084a53b705127dc15717b22c5e140ae0a3c38349d4efec071e \ + --hash=sha256:4998562bf62a445225f22e07c896bb04b35b1b1f2eb6d760584c9c51d7a5f78c \ + --hash=sha256:4b0a7fe987b14c31ebda6083f74f22b561fd3739bc0ac51e019622e3d72668c7 \ + --hash=sha256:4e8c2a84d977f50b9daed6eeaf3baef67d00d5d74d932288f02cb94518ee3ace \ + --hash=sha256:4f883547d4b7f0495ebe7056b0cc2aea76094e7a4abc8e933540f3271df27d9c \ + --hash=sha256:514435a37670e3e5e08f3945b68718b6ed329bb84367777e16f9f4dfe1e61a0f \ + --hash=sha256:53aa02d20d10c3d814d536aa4e5ac9b84ca0ff5a88377963b085ad6822f93e64 \ + --hash=sha256:5594fc43d548a7ed94949d139aa1341b270f1863f11cfd37f5a6c8b778a6b67f \ + --hash=sha256:571b9fcb07b97ef3a492028fb3d2dc0993ca23a06138b0315286566d29ef718a \ + --hash=sha256:57b3d78c95ba9059768b10e28b813002261d3f3dfc55cc48b0c988f625175827 \ + --hash=sha256:5afb51d599ea772b8365ae807ae557f18bccfe46ab261fd1c2a9ed700fc6eb17 \ + --hash=sha256:6b02afb9b97f65fbca5f31db6a2a3ba21aa93030225f150fa3f249717e938fb4 \ + --hash=sha256:6c0016e7b354317c4e9e525b937ac8596c38d2d232b419529b9cd7a1cd46e39a \ + --hash=sha256:71d6097b330eea8fd15097780c8e89cb1a8ce7838669f48c5bacd6f663dd4701 \ + --hash=sha256:756c768d0c9c2955feb7a56c37ea24aea2e369f8d36a88da270b6a9f19e62b5e \ + --hash=sha256:78cb2c6865a35ab8ff8b75fd122f6033b92a62c82801110e48ddd6c936a45d91 \ + --hash=sha256:7a743ff716f746fc19a9557f60dab1600d4613255f8a7aeb3cdde4db7eb15a66 \ + --hash=sha256:85f998ea1848bc6757289e739cfbdda3a04adfd58b02fc018ce54d754a5ce468 \ + --hash=sha256:8728f216dcdb6e6d555cf971cb34076139ad74b31fc2c14da4fafc741c5f6217 \ + --hash=sha256:877c3f311ff35410f690861c4409e7ccbf0cd2f878e50628a28e5a0bb689e658 \ + --hash=sha256:8cd2f7bdda092d99c9fc2fb7391354f306d01443d22785d0cbfafa2e2c8bb418 \ + --hash=sha256:8e95e1385e4998ae9694eeaa4730ba5457ff61185b3a55e2e7bea0880aef452a \ + --hash=sha256:962864dc93511324d51ddbb5b9f8731bf71675b93ca612a07441896f4688fb8c \ + --hash=sha256:9cf95fe4d0f84c82d282745d9bb08ad9f926efa00be4697e767b814ce40d4330 \ + --hash=sha256:9e881fca225083806662a5c43d627d215f258ff43c890f831966c7d7ba9c7402 \ + --hash=sha256:a2b55dd6b2a4c4b7d87ffa56bdb33fdc5fdb9a462173861a7bc097f17d91cb09 \ + --hash=sha256:a45650e8ce7fafffd731db8550230db6b0d306d181a90b67d3e6bca2f1990930 \ + --hash=sha256:a876864214e136f0eb367788dbd7df045f4806801518e2cfe9e13229cfe06d8f \ + --hash=sha256:ae26d61dfa7a47befdc7572b521024e8745f3d809bd95ca9505a7bba9ef849ec \ + --hash=sha256:af8d94b0db561cf68b88a267c5c44b49e134f525d0dc2cb7ed413a66bc23559a \ + --hash=sha256:b343699e8308bdc51978310e1c959c584e7869cc8c40780058c87da7781a1e94 \ + --hash=sha256:b3c777e849237620b022f7f297dd67705f9f5cf1685f09f02e46f93e92725468 \ + --hash=sha256:b629de27fda84b42cde7edef0d85f13b958b47f6e9bbcbba9b673c562a89bd8b \ + --hash=sha256:ba09209fbe443b4acccebe845d8a138b89a8f4fbaeedd44953490b5315d5e965 \ + --hash=sha256:ba54cfebe86920a559a7c4d6b9050791c20513650a1952ebe3368c7dc70306f8 \ + --hash=sha256:bcb46e2f9feff8d06323983bd83ed00c201fdcab3d74973e7072a889b3979fcd \ + --hash=sha256:bcc33feacfaefce60c12fd500a277533bdc02b10a19f7f6d348763d8140bbba7 \ + --hash=sha256:bf16ba1b4d0b6b7c8e534936632270cf70eb00dbe09005bc345b2677b726855c \ + --hash=sha256:cf1845d02ad822a369a49f2bb9345b1614744267682e7a03527dc3bf6eea1777 \ + --hash=sha256:d69141514cc30b774ceea5e3ed3a6635c8d8a96edf664689b890f4089111fb35 \ + --hash=sha256:d9c7f76c0673154f044e9d78c8655fb4213f6ca31a836df48b40fe5d187717b9 \ + --hash=sha256:dbce0b29841537a2fa4a214c2bbf14de3587c9680caa9b4e217568472490b28f \ + --hash=sha256:dc624f6bc473dacdf7ef7eb8678d0d08edf15cd94fad6ae5c7d6cc67a4e4902f \ + --hash=sha256:e158cb00350dc278f3b91551101aa7d12415a66ebf2c91d8d5ac14e56ddd3ad0 \ + --hash=sha256:e491916b378fba47242221bb9ead245211b70d504f495d105d17b14a24b4907c \ + --hash=sha256:e795b7eb908249c4e43c7c99fac7c2c75dab0c43566e37db472a355f63693d71 \ + --hash=sha256:e7e480451b9fa137494bccd3a7d69adbe8ac65a87d97be61e11f1b1050a5bac3 \ + --hash=sha256:e91206ee562682b51b98ef4b26a6ef48fd84e15fd4c4bc5ec768eb641d206838 \ + --hash=sha256:e9871b1ffbfa9656b60aeee92ed5136a5742696006fa322b29ea3d8da0ecc9cf \ + --hash=sha256:e9aeb04d6aef139de265b29683e119b638208f88cf73cdd1658aa07221165321 \ + --hash=sha256:ebaea975e03d3141d9d3a507df75c9b3ec90fa9d2ffd07567b3a978d9d790b26 \ + --hash=sha256:f0606c8bf2cdefea14a43530f7657cbbb7ecf1c4222512492ef4a4434a9501ec \ + --hash=sha256:f13c32a3abd6079a66d9526e18dad9b6d280384d49d7c54040cd57b6424041d9 \ + --hash=sha256:f7401aebd7f581d7f83a439d87d474999317ee099218e5ad25d125290990ba65 \ + --hash=sha256:fa4ecea169a355be7a3ade2c783e2ed12f0e40d2c5621cda8b3297faf7fbb9f5 \ + --hash=sha256:fbd139c8447d25dd750ab79ee274cc5e1fe80fc56340ab10b18a195e1b6eca3e \ + --hash=sha256:fdafc9cce40277e0f7a0feabce0ee50dd2fa1800f3b38015e51296b5e814048d \ + --hash=sha256:fe3cca2e4e8a592be0f269a1ca4835c25199d9f3ce815c8491048f785b0a0198 \ + --hash=sha256:ffd0c5368496f41b0944be820fcb7a838aa6e623d250b01acf2643939c3f99d7 + # via streamlit +plotly==7.0.0 \ + --hash=sha256:78cbf7bd06d1b05bb3b8ec1b709864695229b55151b6f7530fbf55517ead6fdd + # via -r app/requirements.txt +pluggy==1.6.0 \ + --hash=sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746 + # via pytest +protobuf==7.36.1 \ + --hash=sha256:0b53ce95272aad50ad25d7ff03373743209822e8ba42ea7fad27d2bee1547d00 \ + --hash=sha256:39c518c05586c016d7874ff6079ee115bcec1ea5fbb1d177fbf7867ef4c67e44 \ + --hash=sha256:3cf2ee25d006cee57294a1196ea43b37feb78e0dcd1e8af5c1aeddb777655aca \ + --hash=sha256:43d3d37b1eb24c113b9b7d02008cac44e423f00b611b7781ae998d7623972969 \ + --hash=sha256:51139351435d9b43d88a55eaa49fb6f737fbb478fb0cbf2cf694d1a04a9d3363 \ + --hash=sha256:7d951e46b3f963d6c264c367c437921de9d5aedd9c3f9612b9077736b4e3ad5c \ + --hash=sha256:97198b77e369a0abd8e262b8f6c7266c55ddb796a3a12c76d7b8881188ed83aa + # via streamlit +psycopg==3.3.5 \ + --hash=sha256:ce5aa5cdb4f9379f00f487590e5890bfa7df9a164648c969ffa628505e21af4e + # via + # -r app/requirements-keycheckers.txt + # -r app/requirements.txt +psycopg-binary==3.3.5 \ + --hash=sha256:0249c3e960cdee686000eb77169fb6590105c05bacc37e057ccdffdcd8e6ebde \ + --hash=sha256:04f64b39830887c2c737b522cbfd6ad215d65e67ebfff674aa4cf21c02af487b \ + --hash=sha256:06de14ac978a2d53e864069fb5487075c6e3cfb0740f1bfd7017bc8b9942067f \ + --hash=sha256:0d8a4b7ae47f3381e2ded89891d2455b809f4afb7e5b58086844abb8cfa420ea \ + --hash=sha256:1344fd57a19737554670e67aecabd4fb37cd7937f2925840009645d641117e5a \ + --hash=sha256:14f432430fd9e1a9e7d9ab2fe14956c77f5d074ebdc556a1ad04e9a1bd3fca04 \ + --hash=sha256:14fdfd65a96ecbd8b586d14546105641f4a6ac7cbe335c786830ea4de94bbe60 \ + --hash=sha256:19e5bf9872dbd164c220567fd385ba2309c7d9df1541f78343510c6b0f36a1b7 \ + --hash=sha256:1ef2e498be47800f6202b9a2304c22646325ca6d54001b7c785bcfdb24a1e8ab \ + --hash=sha256:2111f880add40fb03c60556069ad68e884a0908a74d2debafc603caf93b73552 \ + --hash=sha256:25105f9b46bdf2a30fcb67f56976ed66f6855941ae16bc024192609b917d493c \ + --hash=sha256:2719fe19a4da752c4110cc767716d0a5bdb760d1153d89018bb7c9c61717bde5 \ + --hash=sha256:358748fc4c8ccdc0e2bdf55420494930e19c3ade586ea9c3a6de3dad1f897311 \ + --hash=sha256:35885e333020fc152d27bea1a494bef13b2e68f6fd92b6229015e93539152008 \ + --hash=sha256:39e70c8e3b5fad70e2970ea4cc502bf3b612018f128aa1c670f1fa78b9774543 \ + --hash=sha256:40505676b1526b9ea387dace034040a8c8b0bcf984cd6bd4720a2ab15e813586 \ + --hash=sha256:40f8b132c7243ef5f503f0b6f986bf16d38a51b0df1c6ba2577743f128be03e3 \ + --hash=sha256:479b96fd78149cfa10369dc53fbfb89ee729be13146b584a23dbc7e164c0cf1e \ + --hash=sha256:4901e5b9a31c230211a1871263b6594373bacd770b14ad9b14d349716cb69cbe \ + --hash=sha256:553b5443cbc94fdb9b0e31b62acdf615e0780982d6b13752a15eb3d6c0dfd0d2 \ + --hash=sha256:5698ab5941a4d138c30fef858588e651fe7d583280cd6e41832825ad9e747750 \ + --hash=sha256:5816472e3bb05615f33a741e0835043d1f4bf9709ff30d2f4aed71815cfc6b5e \ + --hash=sha256:5b981d25fc2dd13fa7328e40703ea8a03f3e9d855ec946431e96a23206d3b9fd \ + --hash=sha256:682a17a57415c3ca1731eec018ed031f012ffcb81ba74806eb219cb396065672 \ + --hash=sha256:6e85d50b87257fb117675a19ee59daa7bf9a57f6431500adf7059df799232ef4 \ + --hash=sha256:7b443f943abfe35aa5a776630cea27c9348aa66659286cee0b99084332252080 \ + --hash=sha256:88e01aa2e938a45655a8a5213fc3a44ba78cb4cab8a569b3e0bcb3d1d0eaba16 \ + --hash=sha256:893ce86a4b997f6ca1261a7826db2506727332a6ff66646fa7f024b39b5e630e \ + --hash=sha256:8dbd694f3741dd4ac5bc60b70e17f7841aefb3f0f38cef4d2756de270e03af43 \ + --hash=sha256:972cc28e943746e71ede254a4dfd1fdfcdd6dcadd6f375703849859f09377f24 \ + --hash=sha256:98a388509306e5e08a4203253ac52846bc1b034e5cbd0ae6da1211593cc28594 \ + --hash=sha256:9c071bf78e5c2e6efa40bc9089a954d7b41221347a72f35c6bf2d8c96e632f75 \ + --hash=sha256:a5e45e4bb68656253ce5c7a344c0a425c293581eba954d7fd7c2e4b2dc9f3038 \ + --hash=sha256:ab39e2794b95af61a2ff69e33e5ab6ac5df36e9ffea9a3b18e38b2aaca8c5ad5 \ + --hash=sha256:ae67072db949d0c094b747a8ec52ad0fa3c42b27842a5f746f3613d54dde3fba \ + --hash=sha256:af5084124fb2fd16557073822519dfe8c389636a16adef661a4c0c3918733171 \ + --hash=sha256:ba466011569297114449df9d523438e1adeedf3e4f31ffb78e897ec3fef3076b \ + --hash=sha256:c065531e8c1815276f50dbfa283e3a7f022671414cdda6fa9a16794dd53b28f9 \ + --hash=sha256:c09775c549b40b274206e1b043c5e5b5af39666e85c98382a30bd05d23ab677b \ + --hash=sha256:c0cac998b9b1e82dec853d2e53b3d34d56a525cf231f9441a636cfd5992929a9 \ + --hash=sha256:c6bd84e4cf67930f26f015dec33f615472b9c5871d46408efe112dbc1bc021de \ + --hash=sha256:ca8af7c0454cdce235d4aedcb5528857468f1490202d25e24fc7af40e176d563 \ + --hash=sha256:cb3b3bffebfe07110730626e76238161124f35ac87b748d663316a28d22f58b0 \ + --hash=sha256:cd0faa2475ab254ad1b507430131cf7f7f0be927ffdc03c32ad3b33d2ef63f42 \ + --hash=sha256:cf0e5e63ee86098299c673992053d556c489ba9ae6aca6cb6e24d16a8e0b09e6 \ + --hash=sha256:d06da67e9c687c6a6fdac9da4b17cbeb296ddd59bd01f6416ed4294bc57c5faf \ + --hash=sha256:d2a61e8147902771df7efe14062a3c8736347850d0d8befcf048235752504f2e \ + --hash=sha256:d8b66353b20e79bf7ac0a80f03ae97f522ccbbf909d687eec62f112e56c0276d \ + --hash=sha256:df209e64674a34b41662c67fdc8b4e0ffd77d2136393790691d086a09f9a6cab \ + --hash=sha256:df9853b832b7b916e02ef68e0d5403a7dab2d5c1ddfe94f22b1b155eb862622f \ + --hash=sha256:e5becd311f9af8d180bad372f51fb2252fd02cb2073056e2b170c9274f95fe7f \ + --hash=sha256:f45d77e398542ce0937d9fa3cd9d84e9c5fc6b34c50a66404ae840bada312750 \ + --hash=sha256:f7e1e45aad410e20de45df2b159df68ff6c8dbf47a3501f806c4489b27f4ad2b \ + --hash=sha256:fd5b047c9fd887b767d063845413e405f5de8ce1dc7a9d0da0637b77b836b469 \ + --hash=sha256:fdbeb38c9b7ca8fa57a7bda3802bedb62f4494ad3dd46c7dd36dc3f77fd5093f + # via psycopg +pyarrow==25.0.1 \ + --hash=sha256:0b1edbb2f385a6a65e9711b62ba86ac54a7816a3f8d17bb3e8a5929d65fb2485 \ + --hash=sha256:0b726ad7e7b669be982b0c71c07fe4b037d654354130da79a7902a669e93a66b \ + --hash=sha256:0befcf816e45a1af33ac775a9970b749e4868a230c7372f0ae5e932bee27039f \ + --hash=sha256:0fe7c8b6c03969b49c8c66182e4a18e3819ab92d07cfab5d8370c531b9369ef0 \ + --hash=sha256:119297a6dc197e45d9c6d4415f7814a67ffa36c180d26f68c154c58067ae782d \ + --hash=sha256:169d3429d5be7c752125890620f75a60776d38b0035eddae939651640822332e \ + --hash=sha256:25f8720bf6387d5dc2ebd2622112de630760419e4b66134405dd24110d15f37e \ + --hash=sha256:31e49a7888fcdf3a835da33ae777f6bb9a866334e5a789282fc26dcf426f7f15 \ + --hash=sha256:35935cd5de130aa5cf4dea052a63e6bf2e17006c35c3a468194242b9b2bf5956 \ + --hash=sha256:38a9a4b4b9613380e200641891495a56c3d5a98a092db4a870af9975e220471d \ + --hash=sha256:3f89685964f46e4216103c75483aac0c0692a5f72212d7ca835adba5ede56ce3 \ + --hash=sha256:4288f27577352d608ca08553b0865e4a9b3aa14820c5d95b53337218d609835b \ + --hash=sha256:4340f0ba6c1d2e13f21658de1d7c662ca2545018568d0030a1e9afca159d87e3 \ + --hash=sha256:44a9120ce5bd81936b8ab9a88076e3fd47c2c6838e0e43630fed83626aca81d9 \ + --hash=sha256:4facd65742a024a4a366328a1d2292062d72d6e023c1b7dda8d4c37544933a25 \ + --hash=sha256:51093dd9e10325fbdb3c10a2ae7c4806e5c822d94e74ae4938b26524a3323fee \ + --hash=sha256:514ddb60285631af068875550c90eddc181db3e8e63a032b1559be189e82f056 \ + --hash=sha256:5389cdf79447ed1515c9e31620e6e1e2302249564d603f2ad727d4f6d313e4c3 \ + --hash=sha256:59a2de54c0cbd954da861eee4d1d330f8e909c45b53455baef696380f2c55033 \ + --hash=sha256:60e89d8f13861a1f7f8d950fa54aebb8023b30734d0ac51ffa80beabe2df4bba \ + --hash=sha256:6109c94d8b9f3b17a041daca16cacb2f651ad8f1ef70a4232c2c0f37a23da2a8 \ + --hash=sha256:62cd0d785b8aa6675ee355f9fc02252a340f4441257c42674937826fd7594325 \ + --hash=sha256:6943e2fe7954d29d84de45d29d34c8dc36ce96570e67d89aa9976e650a4a9138 \ + --hash=sha256:6a1fdfc6659b6b19022f2e50627fb5cf7156a66c46bf4299379955cbe742382a \ + --hash=sha256:880523be3d29efcf83d3998835d206118ccf35e3871dbd2fb60408cf6b007a80 \ + --hash=sha256:8858d7bfc22e3f51529aeaa4077225029724623e4595dc9eff8c793935c34140 \ + --hash=sha256:9171748cdf796972d85a4b60157c279913e242992e350c90c7450182a9838b2a \ + --hash=sha256:a4d6d5e9a3d1879a97c08ded0c797579b7965eafd0f0c26c30b45ccc06db939b \ + --hash=sha256:a4dd8bf99a8fac133efc0ed6a92f5fddbe2adba0d0f6dd720e39ba9855cea85c \ + --hash=sha256:aa0559502e1cd6254d6814614085dd9c5a3dd0419362978a936a3f68a9e5c3df \ + --hash=sha256:b7a296aac7a71fa0886c08e155ddb6c636a50013f801f6178daafa0f9e726188 \ + --hash=sha256:bddd0c4f7630c2a3ddf6347c1bdaa79d97bcf6bd445f9e60c816b7d77c85a5ae \ + --hash=sha256:bf0b672390cdcb640d7288f96b826d71ff4e9abb254a86c89890baf51a29cee6 \ + --hash=sha256:c7c534ec03c358a76ea3e505e74c1b6aef290af90c444dfd092dbfe23e755b85 \ + --hash=sha256:cab40b1edfef0262e0e5251aa2c58d75630f24d06dd7794480243acc001a1d7d \ + --hash=sha256:cc4aa407fde9fc660be3939e49ea31f50f3e9fec17c0ec63159f7711edd3efc9 \ + --hash=sha256:d51592cb7561e87877c506113e7adbf1342ab579e6c21f0ef44b8ba41cb74c80 \ + --hash=sha256:dda9470024204d7bbf2042b47c6e8a0e47a3eeb8e34405882dfaea6577e0c153 \ + --hash=sha256:df961f2e7ae9cf496459259d798652c70625f6c080650d6952f8c04053c58ee9 \ + --hash=sha256:eb6203482ff3746a5632303a7279ae0b5a304c46985b49ed1378cb350ea6728d \ + --hash=sha256:f3831aaa25c67a99f99dc8b05873cb9d64560390372e2aa197ce9dd4a3f06a44 \ + --hash=sha256:f729cfdbd36fd99d543b67a914d2de044c84ebe45be8b34902b299b608c15c8f + # via streamlit +pydeck==0.9.3 \ + --hash=sha256:d8a47c11c81fb12d51b1feb42427ff4f0e13cb599e48931021b2cba98b6849a6 + # via streamlit +pygments==2.21.0 \ + --hash=sha256:2363c69b61c4a97c838da3b130dcd6468f4848992b21a82f2a63ec34377137d9 + # via pytest +pytest==8.4.2 \ + --hash=sha256:872f880de3fc3a5bdc88a11b39c9710c3497a547cfa9320bc3c5e62fbf272e79 + # via -r docker/requirements.in +python-dateutil==2.9.0.post0 \ + --hash=sha256:a8b2bc7bffae282281c8140a97d3aa9c14da0b136dfe83f850eea9a5f7470427 + # via + # botocore + # pandas +python-multipart==0.0.32 \ + --hash=sha256:ff6d3f776f16878c894e52e107296ffc890e913c611b1a4ec6c44e2821fe2e23 + # via + # -r app/requirements.txt + # streamlit +pyyaml==6.0.3 \ + --hash=sha256:00c4bdeba853cc34e7dd471f16b4114f4162dc03e6b7afcc2128711f0eca823c \ + --hash=sha256:0150219816b6a1fa26fb4699fb7daa9caf09eb1999f3b70fb6e786805e80375a \ + --hash=sha256:02893d100e99e03eda1c8fd5c441d8c60103fd175728e23e431db1b589cf5ab3 \ + --hash=sha256:02ea2dfa234451bbb8772601d7b8e426c2bfa197136796224e50e35a78777956 \ + --hash=sha256:0f29edc409a6392443abf94b9cf89ce99889a1dd5376d94316ae5145dfedd5d6 \ + --hash=sha256:10892704fc220243f5305762e276552a0395f7beb4dbf9b14ec8fd43b57f126c \ + --hash=sha256:16249ee61e95f858e83976573de0f5b2893b3677ba71c9dd36b9cf8be9ac6d65 \ + --hash=sha256:1d37d57ad971609cf3c53ba6a7e365e40660e3be0e5175fa9f2365a379d6095a \ + --hash=sha256:1ebe39cb5fc479422b83de611d14e2c0d3bb2a18bbcb01f229ab3cfbd8fee7a0 \ + --hash=sha256:214ed4befebe12df36bcc8bc2b64b396ca31be9304b8f59e25c11cf94a4c033b \ + --hash=sha256:2283a07e2c21a2aa78d9c4442724ec1eb15f5e42a723b99cb3d822d48f5f7ad1 \ + --hash=sha256:22ba7cfcad58ef3ecddc7ed1db3409af68d023b7f940da23c6c2a1890976eda6 \ + --hash=sha256:27c0abcb4a5dac13684a37f76e701e054692a9b2d3064b70f5e4eb54810553d7 \ + --hash=sha256:28c8d926f98f432f88adc23edf2e6d4921ac26fb084b028c733d01868d19007e \ + --hash=sha256:2e71d11abed7344e42a8849600193d15b6def118602c4c176f748e4583246007 \ + --hash=sha256:34d5fcd24b8445fadc33f9cf348c1047101756fd760b4dacb5c3e99755703310 \ + --hash=sha256:37503bfbfc9d2c40b344d06b2199cf0e96e97957ab1c1b546fd4f87e53e5d3e4 \ + --hash=sha256:3c5677e12444c15717b902a5798264fa7909e41153cdf9ef7ad571b704a63dd9 \ + --hash=sha256:3ff07ec89bae51176c0549bc4c63aa6202991da2d9a6129d7aef7f1407d3f295 \ + --hash=sha256:41715c910c881bc081f1e8872880d3c650acf13dfa8214bad49ed4cede7c34ea \ + --hash=sha256:418cf3f2111bc80e0933b2cd8cd04f286338bb88bdc7bc8e6dd775ebde60b5e0 \ + --hash=sha256:44edc647873928551a01e7a563d7452ccdebee747728c1080d881d68af7b997e \ + --hash=sha256:4a2e8cebe2ff6ab7d1050ecd59c25d4c8bd7e6f400f5f82b96557ac0abafd0ac \ + --hash=sha256:4ad1906908f2f5ae4e5a8ddfce73c320c2a1429ec52eafd27138b7f1cbe341c9 \ + --hash=sha256:501a031947e3a9025ed4405a168e6ef5ae3126c59f90ce0cd6f2bfc477be31b7 \ + --hash=sha256:5190d403f121660ce8d1d2c1bb2ef1bd05b5f68533fc5c2ea899bd15f4399b35 \ + --hash=sha256:5498cd1645aa724a7c71c8f378eb29ebe23da2fc0d7a08071d89469bf1d2defb \ + --hash=sha256:5cf4e27da7e3fbed4d6c3d8e797387aaad68102272f8f9752883bc32d61cb87b \ + --hash=sha256:5e0b74767e5f8c593e8c9b5912019159ed0533c70051e9cce3e8b6aa699fcd69 \ + --hash=sha256:5ed875a24292240029e4483f9d4a4b8a1ae08843b9c54f43fcc11e404532a8a5 \ + --hash=sha256:5fcd34e47f6e0b794d17de1b4ff496c00986e1c83f7ab2fb8fcfe9616ff7477b \ + --hash=sha256:5fdec68f91a0c6739b380c83b951e2c72ac0197ace422360e6d5a959d8d97b2c \ + --hash=sha256:6344df0d5755a2c9a276d4473ae6b90647e216ab4757f8426893b5dd2ac3f369 \ + --hash=sha256:64386e5e707d03a7e172c0701abfb7e10f0fb753ee1d773128192742712a98fd \ + --hash=sha256:652cb6edd41e718550aad172851962662ff2681490a8a711af6a4d288dd96824 \ + --hash=sha256:66291b10affd76d76f54fad28e22e51719ef9ba22b29e1d7d03d6777a9174198 \ + --hash=sha256:66e1674c3ef6f541c35191caae2d429b967b99e02040f5ba928632d9a7f0f065 \ + --hash=sha256:6adc77889b628398debc7b65c073bcb99c4a0237b248cacaf3fe8a557563ef6c \ + --hash=sha256:79005a0d97d5ddabfeeea4cf676af11e647e41d81c9a7722a193022accdb6b7c \ + --hash=sha256:7c6610def4f163542a622a73fb39f534f8c101d690126992300bf3207eab9764 \ + --hash=sha256:7f047e29dcae44602496db43be01ad42fc6f1cc0d8cd6c83d342306c32270196 \ + --hash=sha256:8098f252adfa6c80ab48096053f512f2321f0b998f98150cea9bd23d83e1467b \ + --hash=sha256:850774a7879607d3a6f50d36d04f00ee69e7fc816450e5f7e58d7f17f1ae5c00 \ + --hash=sha256:8d1fab6bb153a416f9aeb4b8763bc0f22a5586065f86f7664fc23339fc1c1fac \ + --hash=sha256:8da9669d359f02c0b91ccc01cac4a67f16afec0dac22c2ad09f46bee0697eba8 \ + --hash=sha256:8dc52c23056b9ddd46818a57b78404882310fb473d63f17b07d5c40421e47f8e \ + --hash=sha256:9149cad251584d5fb4981be1ecde53a1ca46c891a79788c0df828d2f166bda28 \ + --hash=sha256:93dda82c9c22deb0a405ea4dc5f2d0cda384168e466364dec6255b293923b2f3 \ + --hash=sha256:96b533f0e99f6579b3d4d4995707cf36df9100d67e0c8303a0c55b27b5f99bc5 \ + --hash=sha256:9c57bb8c96f6d1808c030b1687b9b5fb476abaa47f0db9c0101f5e9f394e97f4 \ + --hash=sha256:9c7708761fccb9397fe64bbc0395abcae8c4bf7b0eac081e12b809bf47700d0b \ + --hash=sha256:9f3bfb4965eb874431221a3ff3fdcddc7e74e3b07799e0e84ca4a0f867d449bf \ + --hash=sha256:a33284e20b78bd4a18c8c2282d549d10bc8408a2a7ff57653c0cf0b9be0afce5 \ + --hash=sha256:a80cb027f6b349846a3bf6d73b5e95e782175e52f22108cfa17876aaeff93702 \ + --hash=sha256:b30236e45cf30d2b8e7b3e85881719e98507abed1011bf463a8fa23e9c3e98a8 \ + --hash=sha256:b3bc83488de33889877a0f2543ade9f70c67d66d9ebb4ac959502e12de895788 \ + --hash=sha256:b865addae83924361678b652338317d1bd7e79b1f4596f96b96c77a5a34b34da \ + --hash=sha256:b8bb0864c5a28024fac8a632c443c87c5aa6f215c0b126c449ae1a150412f31d \ + --hash=sha256:ba1cc08a7ccde2d2ec775841541641e4548226580ab850948cbfda66a1befcdc \ + --hash=sha256:bdb2c67c6c1390b63c6ff89f210c8fd09d9a1217a465701eac7316313c915e4c \ + --hash=sha256:c1ff362665ae507275af2853520967820d9124984e0f7466736aea23d8611fba \ + --hash=sha256:c2514fceb77bc5e7a2f7adfaa1feb2fb311607c9cb518dbc378688ec73d8292f \ + --hash=sha256:c3355370a2c156cffb25e876646f149d5d68f5e0a3ce86a5084dd0b64a994917 \ + --hash=sha256:c458b6d084f9b935061bc36216e8a69a7e293a2f1e68bf956dcd9e6cbcd143f5 \ + --hash=sha256:d0eae10f8159e8fdad514efdc92d74fd8d682c933a6dd088030f3834bc8e6b26 \ + --hash=sha256:ebc55a14a21cb14062aa4162f906cd962b28e2e9ea38f9b4391244cd8de4ae0b \ + --hash=sha256:eda16858a3cab07b80edaf74336ece1f986ba330fdb8ee0d6c0d68fe82bc96be \ + --hash=sha256:ee2922902c45ae8ccada2c5b501ab86c36525b883eff4255313a253a3160861c \ + --hash=sha256:efd7b85f94a6f21e4932043973a7ba2613b059c4a000551892ac9f1d11f5baf3 \ + --hash=sha256:f7057c9a337546edc7973c0d3ba84ddcdf0daa14533c2065749c9075001090e6 \ + --hash=sha256:fa160448684b4e94d80416c0fa4aac48967a969efe22931448d853ada8baf926 \ + --hash=sha256:fc09d0aa354569bc501d4e787133afc08552722d3ab34836a80547331bb5d4a0 + # via -r app/requirements.txt +referencing==0.37.0 \ + --hash=sha256:381329a9f99628c9069361716891d34ad94af76e461dcb0335825aecc7692231 + # via + # jsonschema + # jsonschema-specifications +requests==2.34.2 \ + --hash=sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0 + # via + # -r app/requirements-keycheckers.txt + # -r app/requirements.txt + # streamlit +rpds-py==2026.6.3 \ + --hash=sha256:0be972be84cfcaf46c8c6edf690ca0f154ac17babf1f6a955a51579b34ad2dc5 \ + --hash=sha256:127565fead0a10943b282957bd5447804ff3160ad79f2ad2635e6d249e380680 \ + --hash=sha256:127e08c0642d880cf32ca47ec2a4a77b901f7e2dd1ad9762adb13955d72ffcc9 \ + --hash=sha256:166cf54d9f44fc6ceb53c7860258dde44a81406646de79f8ed3234fca3b6e538 \ + --hash=sha256:168c733a7112e071bb7a66460e667edfcff06c017a3c523f7a8a8e08d0140804 \ + --hash=sha256:1967debc37f64f2c4dc90a7f563aec558b471966e12adcac4e1c4240496b6ebf \ + --hash=sha256:1cf01971c4f2c5553b772a542e4aaf191789cd331bc2cd4ff0e6e65ba49e1e97 \ + --hash=sha256:1e5822dfc2f0d4ab7e745eaa6d85945069329beeccef965af3f3bb26058fcab6 \ + --hash=sha256:22bffe6042b9bcb0822bcd1955ec00e245daf17b4344e4ed8e9551b976b63e96 \ + --hash=sha256:23a439f31ccbeff1574e24889128821d1f7917470e830cf6544dced1c662262a \ + --hash=sha256:24e9c5386e16669b674a69c156c8eeefcb578f3b3397b713b08e6d60f3c7b187 \ + --hash=sha256:270b293dae9058fc9fcedab50f13cebf46fb8ed1d1d54e0521a9da5d6b211975 \ + --hash=sha256:29dfa0533a5d4c94d4dfa1b694fcb56c9c63aad8330ffdd816fd225d0a7a162f \ + --hash=sha256:2a9c6f195058cb45335e8cc3802745c603d716eb96bc9625950c1aac71c0c703 \ + --hash=sha256:2bfd04c19ddbd6640de0b51894d764bd2758854d5b75bd102d2ef10cb9c293a9 \ + --hash=sha256:2c54a076ca4d370980ab57bc0e31df57bbe8d41340436a90ef8b1219a3cbb127 \ + --hash=sha256:2c958bf94822e9290a40aaf2a822d4bc5c88099093e3948ad6c571eca9272e5f \ + --hash=sha256:2c99f7e8ccb3dd6e3e4bfeac657a7b208c9bac8075f4b078c02d7404c34107fa \ + --hash=sha256:2f7c26fbc5acd2522b95d4177fe4710ffd8e9b20529e703ffbf8db4d93903f05 \ + --hash=sha256:30c6dc199b24a5e3e81d50da0f00858c5bbdb2617a750395687f4339c5818171 \ + --hash=sha256:38a2fea2787428f811719ceb9114cb78964a3138838320c29ac39526c79c16ba \ + --hash=sha256:3a83ae6c67b7676b9878378547ca8e93ed77a580037bcbcd1d32f739e1e6089c \ + --hash=sha256:3cfe765c1da0072636ca06628261e0ea05688e160d5c8a03e0217c3854037223 \ + --hash=sha256:421aba32367055614287a4292b6a17f1939c9452299f7a0209c117e990b646d4 \ + --hash=sha256:425560c6fa0415f27261727bb20bd097568485e5eb0c121f1949417d1c516885 \ + --hash=sha256:4470ce197d4090875cf6affbf1f853338387428df97c4fb7b7106317b8214698 \ + --hash=sha256:4cf2d36a2357e4d07bb5a4f98801265327b48256867816cfd2ceb001e9754a8f \ + --hash=sha256:4f4bca01b63096f606e095734dd56e74e175f94cfbf24ff3d63281cec61f7bb7 \ + --hash=sha256:501f9f04a588d6a09179368c57071301445191767c64e4b52a6aa9871f1ef5ed \ + --hash=sha256:536bceea4fa4acf7e1c61da2b5786304367c816c8895be71b8f537c480b0ea1f \ + --hash=sha256:538949e262e46caa31ac01bdb3c1e8f642622922cacbabbae6a8445d9dc33eaf \ + --hash=sha256:539d75de9e0d536c84ff18dfeb805398e58227001ce09231a26a08b9aed1ee0e \ + --hash=sha256:54f45a148e28767bf343d33a684693c70e451c6f4c0e9904709a723fafbdfc1f \ + --hash=sha256:55927d532399c2c646100ff7feb48eaa940ad70f42cd68e1328f3ded9f81ca24 \ + --hash=sha256:58eadac9cd119677b60e1cf8ac4052f35949d71b8a9e5556efccbe82533cf22a \ + --hash=sha256:5e8d07bddee435a2ff6f1920e18feff28d0bc4533e42f4bf6927fbd073312c41 \ + --hash=sha256:62698275682bf121181861295c9181e789030a2d516071f5b8f3c23c170cd0fc \ + --hash=sha256:639c8929aa0afe81be836b04de888460d6bed38b9c54cfc18da8f6bfabf5af5d \ + --hash=sha256:67e3a721ffc5d8d2210d3671872298c4a84e4b8035cfe42ffd7cde35d772b146 \ + --hash=sha256:6de4744d05bd1aa1be4ed7ea1189e3979196808008113bbbf899a460966b925e \ + --hash=sha256:6e84adbcf4bf841aed8116a8264b9f50b4cb3e7bd89b516122e616ac56ca269e \ + --hash=sha256:7491ee23305ac3eb59e492b6945881f5cd77a6f731061a3f25b77fd40f9e99a4 \ + --hash=sha256:79486287de1730dbaff3dbd124d0ca4d2ef7f9d29bf2544f1f93c09b5bcbbd12 \ + --hash=sha256:7b689145a1485c335569bd056464f3243a29af7ed3871c7be31ad624ba239bc7 \ + --hash=sha256:7f88d653e7b3b779d71ae7454e20dcc9b6bae903f33c269db9f2be41bda3f261 \ + --hash=sha256:8020133a74bd81b4572dd8e4be028a6b1ebcd70e6726edc3918008c08bee6ee6 \ + --hash=sha256:808345f53cb952433ca2816f1604ff3515608a81784954f38d4452acfe8e61d5 \ + --hash=sha256:83e35b57523816c8613fd0776b40cd8bb9f596b37ddd2692eb4a6bb5ab2f8c93 \ + --hash=sha256:842e7b070435622248c7a2c44ae53fa1440e073cc3023bc919fed570884097a7 \ + --hash=sha256:847927daf4cffbd4e90e42bc890069897101edd015f956cb8721b3473372edda \ + --hash=sha256:882076c00c0a608b131187055ddc5ae29f2e7eaf870d6168980420d58528a5c8 \ + --hash=sha256:8b95977e7211527ab0ba576e286d023389fbeeb32a6b7b771665d333c60e5342 \ + --hash=sha256:8bb68f03f395eb793220b45c097bd4d8c32944393da0fad8b999efac0868fc8c \ + --hash=sha256:8c2642a7603ec0b16ed77da4555db3b4b472341904873788327c0b0d7b95f1bb \ + --hash=sha256:8c3d1e9c15b9d51ca0391e13da1a25a0a4df3c58a37c9dc368e0736cf7f69df0 \ + --hash=sha256:8c6e5a2f750cc71c3e3b11d71661f21d6f9bc6cebc6564b1466417a1ec03ec77 \ + --hash=sha256:8d2294a31386bfa251d8c8a39472beee17db67d4f1a6eabea665d35c9a4461c3 \ + --hash=sha256:8e4320744c1ffdd95a603def63344bfab2d33edeab301c5007e7de9f9f5b3885 \ + --hash=sha256:8e65860d238379ed982fd9ba690579b5e95af2f4840f99c772816dbe573cb826 \ + --hash=sha256:8f2e5c5ee828d42cb11760761c0af6507927bec42d0ad5458f97c9203b054617 \ + --hash=sha256:900a67df3fd1660b035a4761c4ce73c382ea6b35f90f9863c36c6fd8bf8b09bb \ + --hash=sha256:913ca42ccad3f8cc6e292b587ae8ae49c8c823e5dce51a736252fc7c7cdfa577 \ + --hash=sha256:9250a9a0a6fd4648b3f868da8d91a4c52b5811a62df58e753d50ae4454a36f80 \ + --hash=sha256:931908d9fc855d8f74783377822be318edb6dcb19e47169dc038f9a1bf60b06e \ + --hash=sha256:9826217f048f620d9a712672818bf231442c1b35d96b227a07eabd11b4bb6945 \ + --hash=sha256:9891e594296ab9dada6551c8e7b387b2721f27a67eecd528412e8906247a7b90 \ + --hash=sha256:9c1255b302953c86a486b81d330d5ee1d5bd937691ce271b6be0ef0e299eaab7 \ + --hash=sha256:a0811d33247c3d6128a3001d763f2aa056bb3425204335400ac54f89eec3a0d0 \ + --hash=sha256:a136d453475ac0fcbda502ef1e6504bd28d6d904700915d278deeab0d00fe140 \ + --hash=sha256:a214c993455f99a89aaeadc9b21241900037adc9d97203e374d75513c5911822 \ + --hash=sha256:a3086b538543802f84c843911242db20447de00d8752dd0efc936dbcf02218ba \ + --hash=sha256:a3450b693fde92133e9f51060568a4c31fcca76d5e53bbd611e689ca446517e9 \ + --hash=sha256:a550fb4950a06dde3beb4721f5ad4b25bf4513784665b0a8522c792e2bd822a4 \ + --hash=sha256:a9f4645593036b81bbdb36b9c8e0ea0d1c3fee968c4d59db0344c14087ef143a \ + --hash=sha256:aca6c1ef08a82bfe327cc156da694660f599923e2e6665b6d81c9c2d0ac9ffc8 \ + --hash=sha256:acac386b453c2516111b50985d60ce46e7fadb5ea71ae7b25f4c946935bf27cf \ + --hash=sha256:acc992ab27b15f852c76755eb2ab7dce86585ddadba6fa5946e58556088845b4 \ + --hash=sha256:ae3d4fe8c0b9213624fdce7279d70e3b148b682ca20719ebd193a23ebfa47324 \ + --hash=sha256:ae50181a047c871561212bb97f7932a2d45fb53e947bd9b57ebad85b529cbc53 \ + --hash=sha256:ae6dd8f10bd17aad820876d24caec9efdafd80a318d16c0a48edb5e136902c6b \ + --hash=sha256:af05d726809bff6b141be124d4c7ce998f9c9c7f30edb1f46c07aa103d540b41 \ + --hash=sha256:afd70d95892096cdb26f15a00c45907b17817577aa8d1c76b2dcc2788391f9e9 \ + --hash=sha256:b5c2dc92304aa48a4a60443b548bb12f12e119d4b72f314015e67b9e1be97fca \ + --hash=sha256:bc0011654b91cc4fb2ae701bec0a0ba1e552c0714247fa7af6c59e0ccfa3a4e1 \ + --hash=sha256:bcfbcf66006befb9fd2aeaa9e01feaf881b4dc330a02ba07d2322b1c11be7b5d \ + --hash=sha256:bdbd97738551fca3917c1bd7188bec1920bb520104f28e7e1007f9ceb17b7690 \ + --hash=sha256:c60924535c75f1566b6eb75b5c31a48a43fef04fa2d0d201acbad8a9969c6107 \ + --hash=sha256:c7b9a2f8f4d8e90af72571d3d495deebdd7e3c75451f5b41719aee166e940fc2 \ + --hash=sha256:ca6546b66be9dc4738b1b043d5ebd5488c66c578c5ff0fd0e8065313fe3afb76 \ + --hash=sha256:ccffae9a092a00deb7efd545fe5e2c33c33b88e7c054337e9a74c179347d0b7d \ + --hash=sha256:cdc7e35386f3847df728fbcb5e887e2d79c19e2fa1eba9e51b6621d23e3243af \ + --hash=sha256:d15fde0e6fb0d88a60d221204873743e5d9f0b7d29165e62cd86d0413ad74ba6 \ + --hash=sha256:d34c20167764fbcf927194d532dd7e0c56772f0a5f943fa5ef9e9afbba8fb9db \ + --hash=sha256:d483fe17f01ad64b7bf7cc38fcefff1ca9fb83f8c2b2542b68f97ffe0611b369 \ + --hash=sha256:d7469697dce35be237db177d42e2a2ee26e6dcc5fc052078a6fefabd288c6edd \ + --hash=sha256:db08f45aecde626498fb3df07bcf6d2ec040af42e859a4f5040d79c200342911 \ + --hash=sha256:dc319e5a1de4b6913aac94bf6a2f9e847371e0a140a43dd4991db1a09bc2d504 \ + --hash=sha256:de3eceba0b683bcbb1ab93da016d0270df1f9ae7be716b40214c5dafac6ea45a \ + --hash=sha256:dfcc8b909769d19db55c7cc9541eb64b9b774b1057ffffb4f1048070475bb9f9 \ + --hash=sha256:e059c5dde6452b44424bd1834557556c226b57781dee1227af23518459722b13 \ + --hash=sha256:e4316bf32babbed84e691e352faf967ce2f0f024174a8643c37c94a1080374fc \ + --hash=sha256:e52655eaf81e32593abedaa4bfe33170c8cfedf3365ed9be6e11e07f148f0278 \ + --hash=sha256:e55d236be29255554da47abe5c577637db7c24a02b8b46f0ca9524c855801868 \ + --hash=sha256:ea7bb13b7c9a29791f87a0387ba7d3ad3a6d783d827e4d3f27b40a0ff44495e2 \ + --hash=sha256:ea964164cc9afa72d4d9b23cc28dafae93693c0a53e0b42acbff15b22c3f9ddd \ + --hash=sha256:ec829541c45bca16e61c7ae50c20501f213605beb75d1aba91a6ee37fbbb56a4 \ + --hash=sha256:ecabd69db66de867690f9797f2f8fa27ba501bbc24540cbdbdc649cd15888ba6 \ + --hash=sha256:ed0c1e5d10cdc7135537988c74a0188da68e2f3c30813ba3744ab1e42e0480f9 \ + --hash=sha256:f0840b5b17057f7fd918b76183a4b5a0635f43e14eb2ce60dce1d4ee4707ea00 \ + --hash=sha256:f4d78253f6996be4901669ad25319f842f740eccf4d58e3c7f3dd39e6dde1d8f \ + --hash=sha256:f56f1695bc5c0871cbc33dc0130fcf503aab0c57dcc5a6700a4f49eba4f2652e \ + --hash=sha256:f826877d462181e5eb1c26a0026b8d0cab05d99844ecb6d8bf3627a2ca0c0442 \ + --hash=sha256:f8f23ead891a3b762f35ab3b04623da7056545b48aa60d59957e6789914545da \ + --hash=sha256:f90938e92afda60266da758ee7d363447f7f0138c9559f9e1811629580582d90 \ + --hash=sha256:faa679d19a6696fd54259ad321251ad77a13e70e03dd834daa762a44fb6196ef + # via + # jsonschema + # referencing +s3transfer==0.19.2 \ + --hash=sha256:d8168eccca828cbb2cd573675333f3bddd254313a9c42494b84c76b539e8ba25 + # via boto3 +six==1.17.0 \ + --hash=sha256:4721f391ed90541fddacab5acf947aa0d3dc7d27b2e1e8eda2be8970586c3274 + # via python-dateutil +starlette==0.52.1 \ + --hash=sha256:0029d43eb3d273bc4f83a08720b4912ea4b071087a3b48db01b7c839f7954d74 + # via + # -r app/requirements.txt + # streamlit +streamlit==1.63.0 \ + --hash=sha256:c24fd38170543ffb749321c801627aaa09068219539d7b81ef77633ea2408b36 + # via -r app/requirements.txt +toml==0.10.2 \ + --hash=sha256:806143ae5bfb6a3c6e736a764057db0e6a0e05e338b5630894a5f779cabb4f9b + # via streamlit +typing-extensions==4.16.0 \ + --hash=sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8 + # via + # altair + # anyio + # psycopg + # referencing + # starlette + # streamlit +urllib3==2.7.0 \ + --hash=sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897 + # via + # botocore + # requests +uvicorn==0.53.0 \ + --hash=sha256:e8dca71ec86dce5f04e333f0d56cdedf942446e6643b9cea1af0d6d3a02cb03e + # via + # -r app/requirements.txt + # streamlit +watchdog==6.0.0 \ + --hash=sha256:07df1fdd701c5d4c8e55ef6cf55b8f0120fe1aef7ef39a1c6fc6bc2e606d517a \ + --hash=sha256:20ffe5b202af80ab4266dcd3e91aae72bf2da48c0d33bdb15c66658e685e94e2 \ + --hash=sha256:212ac9b8bf1161dc91bd09c048048a95ca3a4c4f5e5d4a7d1b1a7d5752a7f96f \ + --hash=sha256:2cce7cfc2008eb51feb6aab51251fd79b85d9894e98ba847408f662b3395ca3c \ + --hash=sha256:490ab2ef84f11129844c23fb14ecf30ef3d8a6abafd3754a6f75ca1e6654136c \ + --hash=sha256:6eb11feb5a0d452ee41f824e271ca311a09e250441c262ca2fd7ebcf2461a06c \ + --hash=sha256:6f10cb2d5902447c7d0da897e2c6768bca89174d0c6e1e30abec5421af97a5b0 \ + --hash=sha256:7607498efa04a3542ae3e05e64da8202e58159aa1fa4acddf7678d34a35d4f13 \ + --hash=sha256:76aae96b00ae814b181bb25b1b98076d5fc84e8a53cd8885a318b42b6d3a5134 \ + --hash=sha256:7a0e56874cfbc4b9b05c60c8a1926fedf56324bb08cfbc188969777940aef3aa \ + --hash=sha256:82dc3e3143c7e38ec49d61af98d6558288c415eac98486a5c581726e0737c00e \ + --hash=sha256:9041567ee8953024c83343288ccc458fd0a2d811d6a0fd68c4c22609e3490379 \ + --hash=sha256:90c8e78f3b94014f7aaae121e6b909674df5b46ec24d6bebc45c44c56729af2a \ + --hash=sha256:9513f27a1a582d9808cf21a07dae516f0fab1cf2d7683a742c498b93eedabb11 \ + --hash=sha256:a175f755fc2279e0b7312c0035d52e27211a5bc39719dd529625b1930917345b \ + --hash=sha256:a1914259fa9e1454315171103c6a30961236f508b9b623eae470268bbcc6a22f \ + --hash=sha256:afd0fe1b2270917c5e23c2a65ce50c2a4abb63daafb0d419fde368e272a76b7c \ + --hash=sha256:bc64ab3bdb6a04d69d4023b29422170b74681784ffb9463ed4870cf2f3e66112 \ + --hash=sha256:bdd4e6f14b8b18c334febb9c4425a878a2ac20efd1e0b231978e7b150f92a948 \ + --hash=sha256:c7ac31a19f4545dd92fc25d200694098f42c9a8e391bc00bdd362c5736dbf881 \ + --hash=sha256:c7c15dda13c4eb00d6fb6fc508b3c0ed88b9d5d374056b239c4ad1611125c860 \ + --hash=sha256:c897ac1b55c5a1461e16dae288d22bb2e412ba9807df8397a635d88f671d36c3 \ + --hash=sha256:cbafb470cf848d93b5d013e2ecb245d4aa1c8fd0504e863ccefa32445359d680 \ + --hash=sha256:d1cdb490583ebd691c012b3d6dae011000fe42edb7a82ece80965b42abd61f26 \ + --hash=sha256:e3df4cbb9a450c6d49318f6d14f4bbc80d763fa587ba46ec86f99f9e6876bb26 \ + --hash=sha256:e6439e374fc012255b4ec786ae3c4bc838cd7309a540e5fe0952d03687d8804e \ + --hash=sha256:e6f0e77c9417e7cd62af82529b10563db3423625c5fce018430b249bf977f9e8 \ + --hash=sha256:e7631a77ffb1f7d2eefa4445ebbee491c720a5661ddf6df3498ebecae5ed375c \ + --hash=sha256:ef810fbf7b781a5a593894e4f439773830bdecb885e6880d957d5b9382a960d2 + # via streamlit +websockets==16.1.1 \ + --hash=sha256:01fbdcbac298efe19360b94bc0039c8f746f0220ba570f327577bfee81059175 \ + --hash=sha256:024193f8551a2b0eafbdd160911012c4e6c228c28430c84433253299a9e42d6a \ + --hash=sha256:04fd29a0e2fe9414a95b00e92c67ae51bf900c50c0f8a4b2dafdad621f49ea1d \ + --hash=sha256:056ae37939ed7e9974f364f5864e76e49182622d8f9751ac1903c0d09b013985 \ + --hash=sha256:0f62863e8a00a6d33c3d6566ec0b89f23787b747ffe0c3bc71ec0e76b82c94b1 \ + --hash=sha256:0ffd3031ea8bda8d61762e84220186105ba3b748b3c8da2ae4f7816fac03e573 \ + --hash=sha256:1214e673c404684b9bf7154f5cf43b45025b1a6160fac3a9e438e9c1a97e22cb \ + --hash=sha256:125f22dbefaf1554fea66fc83851490edb284ce4f501d37ffed2752f418332d9 \ + --hash=sha256:130937b167a52af203c8d58e78d67705874e82759862e3b9671a452fec4abc87 \ + --hash=sha256:1427fb4cf0d72f66333e2cacc3ff5f575bf2d7008166ce991a4a470b21d51a22 \ + --hash=sha256:195c978b065fa40910582464f99d6b15c8b314c68e0546549a55ed83f4735328 \ + --hash=sha256:1d27fa8462ad6a1cb36206a3d0640b2333340def181fae11ed7f9adeaa5c0747 \ + --hash=sha256:1db4de4a0e95673f7545d393c49eeb0c2f18ac1ef93073218c79d5cdb2ee75ab \ + --hash=sha256:1f79c89b5eb034d1722938a891916582f8f7f503f58ca22518a63c3f2cd18499 \ + --hash=sha256:23253dd5bcae3f9aaee0a1d30967a8dbd52e5d3cff93a2e5b84df57b77d4750d \ + --hash=sha256:249116b4a76063d930a46391ad56e135c286e4562a18309029fc2c73f4ed4c62 \ + --hash=sha256:29dfa8114c4a620c69591c5973860f768eac29d3fd6904f37f34266cb219c512 \ + --hash=sha256:2a606d9c24035242a3e256e9d5b77ed9cd6bccfcb7cf993e5ca3c0f6f68fb6a7 \ + --hash=sha256:2a636ff1e7a5c4edf71ef0e79adae7f25dba93b4fcbe3dc958733477ffeb0eaf \ + --hash=sha256:2bb5d041a8307d2e18782e7ce777f6fdb1e8c2f5d09291484b18c294b789d9aa \ + --hash=sha256:2e28e602bb13da44fbe518c1781a88e3b9d4c3d48d02c9bad83e546164336f57 \ + --hash=sha256:30bbe120437b5648a77d3519b7024ea09530e0b5b18d3698c5a0ae536fe0cc2e \ + --hash=sha256:34420aaa64440ebd51ac72ca8a45ef4626429438c9b02e633ae412ed43f925d3 \ + --hash=sha256:38565aca3e01ea8734e578fb2118dade0ecb0250533f29e22b8d1a7a196cf4d0 \ + --hash=sha256:387e8e4aa5df2f90b198fa3cad3478822a89cf905b6a6d6c97dc3664689640cc \ + --hash=sha256:39f2a024af5c345ffe8fcf1ee18c049c024c94df393bb09b044a6917c77bde43 \ + --hash=sha256:3df13f73af9b3b38ab1195eb299ecb67a4330c911c97ae04043ff74085728abe \ + --hash=sha256:414e596c75f74e0994084694189d7dc9229fb278e33064d6784b73ffbba3ca31 \ + --hash=sha256:41c8e77f17294c0ac18008a7309b99b34ee72247ef10b6dff4c3f8b5ac29896b \ + --hash=sha256:42290eb6db4ccaca7012656738214f8514082fb6fa40cdeb61bb9a471b52e383 \ + --hash=sha256:42f599f4d48c7e1a3338fdaac3acd075be3b3cf02d4b274f3bf2767aedd3d217 \ + --hash=sha256:43e3a9fdd7cbf7ba6040c31fae0faf84ca1474fef777c4e37912f1540f854499 \ + --hash=sha256:443aefe96b7fdb132e2a70806cca1f2af49bb3f28e47abcd7c2e9dcf4d8fa1b8 \ + --hash=sha256:46dcaa042cd1de6c59e7d9269fa63ff7572b6df40510600b678f0826b3c7af51 \ + --hash=sha256:496af849a472b531f758dbd4d61338f5000538cb1a7b3d20d9d32a264517f509 \ + --hash=sha256:49ae99bdfcae803a885c926bf14f886196e84925395bb3f568fef5c0f0979d7d \ + --hash=sha256:4b57693728576d84ede0a77987ab16881b783d2cd9f1dc180a8fbbc3f79c4428 \ + --hash=sha256:4e3b680b1e0a27457e727a0d572fd81dffa87b6dbf8b228ab57da64f7d85aead \ + --hash=sha256:4e8d01cc3bcae7bbf8167f944aeafefed590fae5693552bba9794a9df68371cc \ + --hash=sha256:5283810d2646741a0d8da2aa733d6aefa0545809afccb2a5d105a26bc45125f1 \ + --hash=sha256:53260c8930da5771cec89439bff99c20c8cb03ddb9588b980697355a83cd4bd3 \ + --hash=sha256:536676848fc5961aca9d20389951f59169508f765637a172403dc5434d722fa0 \ + --hash=sha256:54509b8e92fee4453e152b7558ddef37ce9705a044922f2095a6105e3f80c96f \ + --hash=sha256:56cd5fc4f10a9ea8aa0804bddb7b42506cf9e136046f3b4c27de8fec9e2ecba5 \ + --hash=sha256:5bfd1ac19b1b9986a9c95a82d5e23a391ebb09e12c34d7be6094b86efcc35731 \ + --hash=sha256:5c31aa7e39ee3e8a358573257f1c0bb5c52430d1b637030dd9c8cc2c282926be \ + --hash=sha256:5e3b7d601f6f84156b08cc4a5e541c2b50ad7b36cfc302b657a12477c904a5df \ + --hash=sha256:61922544a0587a13fd3f53e4c0e5e606510c7b0d9d22c8444e5fae22a06b38cb \ + --hash=sha256:6456ff333092d509127d75a638cb411afae8ff17f092635015d1902efec8a293 \ + --hash=sha256:69159730a823dde3ea8d08783e8d47ef135a6d7e8d44eb127e32b321c9db8e3e \ + --hash=sha256:69e52d175a0a7d1e13b4b67ad41c560b7d98e8c6f6126eb0bda496c784faf8c7 \ + --hash=sha256:6aaface73b9c71974c6497366d8b9628357f6c9749e09c4ea3610176c63f2ae3 \ + --hash=sha256:6abbd3e82c731c8e531714466acd5d87b5e88ac3243465337ba71d68e23ae7e3 \ + --hash=sha256:6ff9417c0ada4d0f7d212f928303e5579bdf3ace4c802fa4afabb30995da58c3 \ + --hash=sha256:7421fad442de870a8cbf2287d1cad7e706ece0dbfeba5e911df132cbdc1cb56a \ + --hash=sha256:7883388947767080f094950b342b30d35a2a06b849cd967c422fa0db72b40ea9 \ + --hash=sha256:79eace538c6a97e96d0d03d4f9d314f9677f5ed85a8a984992ffd90b13cb8a56 \ + --hash=sha256:7b1b19636af86a3c7995d4d028dbe376f39b4bf31541146f9c123582a6c94562 \ + --hash=sha256:7dfcad78ea1492ee3a9ec765cb7f51bbc17d477107aaf6b22abf7b2558d1c5a0 \ + --hash=sha256:8087e82f842609734c9b5a1330464f8e94e346ba0e18c832c08bafa4b0d63c15 \ + --hash=sha256:820fb8450edddae3812fd58cbc08e2bf22812cb248ecb5f06dbb82119a56e869 \ + --hash=sha256:8483c2096363120eea8b07c06ae7304d520f686665fffd4811fad423930a65d7 \ + --hash=sha256:84a2cef8deffbd9ab8ee0ea546a2a6a7030c28f44e6cdd4547dbfeb489eb8999 \ + --hash=sha256:86d7f0f8bdb25d2c632b72527325e4776430fd5bc61b9118de4e2b8ddb5f5b01 \ + --hash=sha256:8fe0b50da2d84535fb4f7b4bfa951280f97ce3d558a0443b541166d609e67b57 \ + --hash=sha256:90001d893bc368e302ef168d82130b4e4fdd27b85fa094682df9b667c2d48838 \ + --hash=sha256:9246a0d063cfcbcc85f2359dd6876d681213f4790832272aa16641b4ed5d64d4 \ + --hash=sha256:92b820d345f7a3fc7b8163949ee92df910f290c3fc517b3d5301c78065adafe1 \ + --hash=sha256:952303a7318d4cbe1011400839bb2051c9f84fa0a35923267f5daba34b15d458 \ + --hash=sha256:97fd3a0e8b53efa41970ac1dff3d8cf0d2884cadeb4caaf95db7ad1526926ee3 \ + --hash=sha256:9c1c5705e314449e3308872fe084b8571ce078ee4fc55a98a769bdefe5917392 \ + --hash=sha256:9c9f23004a3d40e89c01a7955d186a6cc83418d93b749701944ce2de3e95a1f3 \ + --hash=sha256:9f63bcef7f4b02b06b35fc01c93b96c43b5e88e1e8868676caacf493d5a31f3a \ + --hash=sha256:a0eadbbf2c30f01efa58e1f110eb6fa293261f6b0b1aa38f7f48707107690af9 \ + --hash=sha256:a28fcbc9b6baf54a2e23f8655f308e4ccc6afdd7266f8fe7954f320dcda0f785 \ + --hash=sha256:a6a61aff018180c9c50b7b0da33bfd29d378af3497429c95006c589a23a11648 \ + --hash=sha256:aabe464bfd13bd25f4821faf111da6fefdc389f870265a53105580e45b0a2e49 \ + --hash=sha256:ab59169ace05dcb49a1d4118f0bde139557adf45091bd85747e36bf5de984dd1 \ + --hash=sha256:b436f6ec4fc3a6b4237c84d3f83170ed2b40bb584222f0ac47a0c8a5921980c7 \ + --hash=sha256:b6b9dadbef0cccd9f4c4ee96b08898afa73e26803bbe0f6aeb5bb12b0074206d \ + --hash=sha256:b852788aa51764e2d8e4cf5493d559326bcae5e38d16ba25ffa322b034df272a \ + --hash=sha256:bae954c382e013d5ea5b190d2830526bfa45ad121c326da0049b8c769f185db6 \ + --hash=sha256:bcce07e23e5769375158f5efdcdafa8d5cd014b93c6683865b840ed65b96f231 \ + --hash=sha256:cc97814dfb786a83b6e2dc2e79351e1b83e6d715647d6887fcabd83026417a00 \ + --hash=sha256:cd2ca96a082a36964aca83e992f72abeb61b7306c1a6cba4c7d06a7b93750cac \ + --hash=sha256:cfb70b4eb56cac4da0a83588f3ad50d46beb0690391082f3d4e2d488c70b68ea \ + --hash=sha256:d0fcf657e9f13ff4b177960ab2200237b12994232dfb6df16f1cfe1d4339f93c \ + --hash=sha256:d14bfb217eb4701e850f1525c9d29d79c44794cdf1c299ead25f39f8c78dea81 \ + --hash=sha256:d57685547e0060cc6fd90ee6a28405d6bd395e525545f13c8d7cd99c78afd79f \ + --hash=sha256:d6bec75c290fe484a8ba4cacdf838501e17c06ecfbbf31eede81a9e431bd7751 \ + --hash=sha256:d9531d9cbeac99af6f038fb1bc351403531f7d634a2c2e10e2f7c854c6ed5b68 \ + --hash=sha256:da4ca1a9d72f9030b3146b8d7022719a9f3d478f61efe6f7dd51d243f61c51b2 \ + --hash=sha256:dab9eb87869da2d6ed3af3f3adf28414baae6ec9d4df355ffc18889132f3436c \ + --hash=sha256:dc0fad4933f427acd5b1cec210f3ea6dce7089e1724e4b9ec6ef47c6c04d1b3b \ + --hash=sha256:dc385593a42e31cd6fb60c19f0ecb015b386603818fc2c6c274fb42bd2bb4165 \ + --hash=sha256:dcc04fedf83effaeb9cce98abc9469bb1b42ef85f03e01c8c1f4438ef7555737 \ + --hash=sha256:e047dc87ef7ca50f4d309bf775ad4a71711c58556d75d7bd0604b2317f43e94b \ + --hash=sha256:e09f753a169951eb4f28c2c774f71069304f66e7277e0f5a2892423599cfa854 \ + --hash=sha256:ed5bb271084b46530ee2ddc0410537a9961152c5ccba2fc98c5276d992ccba87 \ + --hash=sha256:f0aa4aad3b1b69ad3fd85a0fd0952ec64331c762bd77ec51cc814170873890b2 \ + --hash=sha256:f17dbe07eb3ea7f99e4df9b7e0efefe80fbf30d37a8cc4d561a0aed310bc8847 \ + --hash=sha256:f2769a0344a09e9ccf5b3cce538bc75a51b53eff3275d3896310c8552049195d \ + --hash=sha256:f55f0b01956a094c8587146d9558c91937e78789c333860ffaf35931a6e5dbc4 \ + --hash=sha256:f5d497865f05bb222cab7016c6034542e84e5f29f49c6fd3f4939cda7197b5b8 \ + --hash=sha256:f70541f3104339f59f830522d94ebadb1bf47426287381623443d8bb1cdbf33d \ + --hash=sha256:fb9a0a6dc3d1b3986cb88091b6899f0396651e0f74e2c9766ab8d6ffc3842e29 \ + --hash=sha256:fce6c48559c86d1ac3632ccb1bebc7d5442fbe79bd9bb0e40379ee54be2a4051 \ + --hash=sha256:fd46fff7eb62c24804d234f0051c7a8ea81285ad63e0337d3dcf33ca82aee58a + # via streamlit +zstandard==0.23.0 \ + --hash=sha256:034b88913ecc1b097f528e42b539453fa82c3557e414b3de9d5632c80439a473 \ + --hash=sha256:0a7f0804bb3799414af278e9ad51be25edf67f78f916e08afdb983e74161b916 \ + --hash=sha256:11e3bf3c924853a2d5835b24f03eeba7fc9b07d8ca499e247e06ff5676461a15 \ + --hash=sha256:12a289832e520c6bd4dcaad68e944b86da3bad0d339ef7989fb7e88f92e96072 \ + --hash=sha256:1516c8c37d3a053b01c1c15b182f3b5f5eef19ced9b930b684a73bad121addf4 \ + --hash=sha256:157e89ceb4054029a289fb504c98c6a9fe8010f1680de0201b3eb5dc20aa6d9e \ + --hash=sha256:1bfe8de1da6d104f15a60d4a8a768288f66aa953bbe00d027398b93fb9680b26 \ + --hash=sha256:1e172f57cd78c20f13a3415cc8dfe24bf388614324d25539146594c16d78fcc8 \ + --hash=sha256:1fd7e0f1cfb70eb2f95a19b472ee7ad6d9a0a992ec0ae53286870c104ca939e5 \ + --hash=sha256:203d236f4c94cd8379d1ea61db2fce20730b4c38d7f1c34506a31b34edc87bdd \ + --hash=sha256:27d3ef2252d2e62476389ca8f9b0cf2bbafb082a3b6bfe9d90cbcbb5529ecf7c \ + --hash=sha256:29a2bc7c1b09b0af938b7a8343174b987ae021705acabcbae560166567f5a8db \ + --hash=sha256:2ef230a8fd217a2015bc91b74f6b3b7d6522ba48be29ad4ea0ca3a3775bf7dd5 \ + --hash=sha256:2ef3775758346d9ac6214123887d25c7061c92afe1f2b354f9388e9e4d48acfc \ + --hash=sha256:2f146f50723defec2975fb7e388ae3a024eb7151542d1599527ec2aa9cacb152 \ + --hash=sha256:2fb4535137de7e244c230e24f9d1ec194f61721c86ebea04e1581d9d06ea1269 \ + --hash=sha256:32ba3b5ccde2d581b1e6aa952c836a6291e8435d788f656fe5976445865ae045 \ + --hash=sha256:34895a41273ad33347b2fc70e1bff4240556de3c46c6ea430a7ed91f9042aa4e \ + --hash=sha256:379b378ae694ba78cef921581ebd420c938936a153ded602c4fea612b7eaa90d \ + --hash=sha256:38302b78a850ff82656beaddeb0bb989a0322a8bbb1bf1ab10c17506681d772a \ + --hash=sha256:3aa014d55c3af933c1315eb4bb06dd0459661cc0b15cd61077afa6489bec63bb \ + --hash=sha256:4051e406288b8cdbb993798b9a45c59a4896b6ecee2f875424ec10276a895740 \ + --hash=sha256:40b33d93c6eddf02d2c19f5773196068d875c41ca25730e8288e9b672897c105 \ + --hash=sha256:43da0f0092281bf501f9c5f6f3b4c975a8a0ea82de49ba3f7100e64d422a1274 \ + --hash=sha256:445e4cb5048b04e90ce96a79b4b63140e3f4ab5f662321975679b5f6360b90e2 \ + --hash=sha256:48ef6a43b1846f6025dde6ed9fee0c24e1149c1c25f7fb0a0585572b2f3adc58 \ + --hash=sha256:50a80baba0285386f97ea36239855f6020ce452456605f262b2d33ac35c7770b \ + --hash=sha256:519fbf169dfac1222a76ba8861ef4ac7f0530c35dd79ba5727014613f91613d4 \ + --hash=sha256:53dd9d5e3d29f95acd5de6802e909ada8d8d8cfa37a3ac64836f3bc4bc5512db \ + --hash=sha256:53ea7cdc96c6eb56e76bb06894bcfb5dfa93b7adcf59d61c6b92674e24e2dd5e \ + --hash=sha256:576856e8594e6649aee06ddbfc738fec6a834f7c85bf7cadd1c53d4a58186ef9 \ + --hash=sha256:59556bf80a7094d0cfb9f5e50bb2db27fefb75d5138bb16fb052b61b0e0eeeb0 \ + --hash=sha256:5d41d5e025f1e0bccae4928981e71b2334c60f580bdc8345f824e7c0a4c2a813 \ + --hash=sha256:61062387ad820c654b6a6b5f0b94484fa19515e0c5116faf29f41a6bc91ded6e \ + --hash=sha256:61f89436cbfede4bc4e91b4397eaa3e2108ebe96d05e93d6ccc95ab5714be512 \ + --hash=sha256:62136da96a973bd2557f06ddd4e8e807f9e13cbb0bfb9cc06cfe6d98ea90dfe0 \ + --hash=sha256:64585e1dba664dc67c7cdabd56c1e5685233fbb1fc1966cfba2a340ec0dfff7b \ + --hash=sha256:65308f4b4890aa12d9b6ad9f2844b7ee42c7f7a4fd3390425b242ffc57498f48 \ + --hash=sha256:66b689c107857eceabf2cf3d3fc699c3c0fe8ccd18df2219d978c0283e4c508a \ + --hash=sha256:6a41c120c3dbc0d81a8e8adc73312d668cd34acd7725f036992b1b72d22c1772 \ + --hash=sha256:6f77fa49079891a4aab203d0b1744acc85577ed16d767b52fc089d83faf8d8ed \ + --hash=sha256:72c68dda124a1a138340fb62fa21b9bf4848437d9ca60bd35db36f2d3345f373 \ + --hash=sha256:752bf8a74412b9892f4e5b58f2f890a039f57037f52c89a740757ebd807f33ea \ + --hash=sha256:76e79bc28a65f467e0409098fa2c4376931fd3207fbeb6b956c7c476d53746dd \ + --hash=sha256:774d45b1fac1461f48698a9d4b5fa19a69d47ece02fa469825b442263f04021f \ + --hash=sha256:77da4c6bfa20dd5ea25cbf12c76f181a8e8cd7ea231c673828d0386b1740b8dc \ + --hash=sha256:77ea385f7dd5b5676d7fd943292ffa18fbf5c72ba98f7d09fc1fb9e819b34c23 \ + --hash=sha256:80080816b4f52a9d886e67f1f96912891074903238fe54f2de8b786f86baded2 \ + --hash=sha256:80a539906390591dd39ebb8d773771dc4db82ace6372c4d41e2d293f8e32b8db \ + --hash=sha256:82d17e94d735c99621bf8ebf9995f870a6b3e6d14543b99e201ae046dfe7de70 \ + --hash=sha256:837bb6764be6919963ef41235fd56a6486b132ea64afe5fafb4cb279ac44f259 \ + --hash=sha256:84433dddea68571a6d6bd4fbf8ff398236031149116a7fff6f777ff95cad3df9 \ + --hash=sha256:8c24f21fa2af4bb9f2c492a86fe0c34e6d2c63812a839590edaf177b7398f700 \ + --hash=sha256:8ed7d27cb56b3e058d3cf684d7200703bcae623e1dcc06ed1e18ecda39fee003 \ + --hash=sha256:9206649ec587e6b02bd124fb7799b86cddec350f6f6c14bc82a2b70183e708ba \ + --hash=sha256:983b6efd649723474f29ed42e1467f90a35a74793437d0bc64a5bf482bedfa0a \ + --hash=sha256:98da17ce9cbf3bfe4617e836d561e433f871129e3a7ac16d6ef4c680f13a839c \ + --hash=sha256:9c236e635582742fee16603042553d276cca506e824fa2e6489db04039521e90 \ + --hash=sha256:9da6bc32faac9a293ddfdcb9108d4b20416219461e4ec64dfea8383cac186690 \ + --hash=sha256:a05e6d6218461eb1b4771d973728f0133b2a4613a6779995df557f70794fd60f \ + --hash=sha256:a0817825b900fcd43ac5d05b8b3079937073d2b1ff9cf89427590718b70dd840 \ + --hash=sha256:a4ae99c57668ca1e78597d8b06d5af837f377f340f4cce993b551b2d7731778d \ + --hash=sha256:a8c86881813a78a6f4508ef9daf9d4995b8ac2d147dcb1a450448941398091c9 \ + --hash=sha256:a8fffdbd9d1408006baaf02f1068d7dd1f016c6bcb7538682622c556e7b68e35 \ + --hash=sha256:a9b07268d0c3ca5c170a385a0ab9fb7fdd9f5fd866be004c4ea39e44edce47dd \ + --hash=sha256:ab19a2d91963ed9e42b4e8d77cd847ae8381576585bad79dbd0a8837a9f6620a \ + --hash=sha256:ac184f87ff521f4840e6ea0b10c0ec90c6b1dcd0bad2f1e4a9a1b4fa177982ea \ + --hash=sha256:b0e166f698c5a3e914947388c162be2583e0c638a4703fc6a543e23a88dea3c1 \ + --hash=sha256:b2170c7e0367dde86a2647ed5b6f57394ea7f53545746104c6b09fc1f4223573 \ + --hash=sha256:b4567955a6bc1b20e9c31612e615af6b53733491aeaa19a6b3b37f3b65477094 \ + --hash=sha256:b69bb4f51daf461b15e7b3db033160937d3ff88303a7bc808c67bbc1eaf98c78 \ + --hash=sha256:b8c0bd73aeac689beacd4e7667d48c299f61b959475cdbb91e7d3d88d27c56b9 \ + --hash=sha256:be9b5b8659dff1f913039c2feee1aca499cfbc19e98fa12bc85e037c17ec6ca5 \ + --hash=sha256:bf0a05b6059c0528477fba9054d09179beb63744355cab9f38059548fedd46a9 \ + --hash=sha256:c16842b846a8d2a145223f520b7e18b57c8f476924bda92aeee3a88d11cfc391 \ + --hash=sha256:c363b53e257246a954ebc7c488304b5592b9c53fbe74d03bc1c64dda153fb847 \ + --hash=sha256:c7c517d74bea1a6afd39aa612fa025e6b8011982a0897768a2f7c8ab4ebb78a2 \ + --hash=sha256:d20fd853fbb5807c8e84c136c278827b6167ded66c72ec6f9a14b863d809211c \ + --hash=sha256:d2240ddc86b74966c34554c49d00eaafa8200a18d3a5b6ffbf7da63b11d74ee2 \ + --hash=sha256:d477ed829077cd945b01fc3115edd132c47e6540ddcd96ca169facff28173057 \ + --hash=sha256:d50d31bfedd53a928fed6707b15a8dbeef011bb6366297cc435accc888b27c20 \ + --hash=sha256:dc1d33abb8a0d754ea4763bad944fd965d3d95b5baef6b121c0c9013eaf1907d \ + --hash=sha256:dc5d1a49d3f8262be192589a4b72f0d03b72dcf46c51ad5852a4fdc67be7b9e4 \ + --hash=sha256:e2d1a054f8f0a191004675755448d12be47fa9bebbcffa3cdf01db19f2d30a54 \ + --hash=sha256:e7792606d606c8df5277c32ccb58f29b9b8603bf83b48639b7aedf6df4fe8171 \ + --hash=sha256:ed1708dbf4d2e3a1c5c69110ba2b4eb6678262028afd6c6fbcc5a8dac9cda68e \ + --hash=sha256:f2d4380bf5f62daabd7b751ea2339c1a21d1c9463f1feb7fc2bdcea2c29c3160 \ + --hash=sha256:f3513916e8c645d0610815c257cbfd3242adfd5c4cfa78be514e5a3ebb42a41b \ + --hash=sha256:f8346bfa098532bc1fb6c7ef06783e969d87a99dd1d2a5a18a892c1d7a643c58 \ + --hash=sha256:f83fa6cae3fff8e98691248c9320356971b59678a17f20656a9e59cd32cee6d8 \ + --hash=sha256:fa6ce8b52c5987b3e34d5674b0ab529a4602b632ebab0a93b07bfb4dfc8f8a33 \ + --hash=sha256:fb2b1ecfef1e67897d336de3a0e3f52478182d6a47eda86cbd42504c5cbd009a \ + --hash=sha256:fc9ca1c9718cb3b06634c7c8dec57d24e9438b2aa9a0f02b8bb36bf478538880 \ + --hash=sha256:fd30d9c67d13d891f2360b2a120186729c111238ac63b43dbd37a5a40670b8ca \ + --hash=sha256:fd7699e8fd9969f455ef2926221e0233f81a2542921471382e77a9e2f2b57f4b \ + --hash=sha256:fe3b385d996ee0822fd46528d9f0443b880d4d05528fd26a9119a54ec3f91c69 + # via -r app/requirements.txt diff --git a/docker/test_verify.py b/docker/test_verify.py new file mode 100644 index 0000000..49f88d9 --- /dev/null +++ b/docker/test_verify.py @@ -0,0 +1,196 @@ +"""Pure Compose regressions: python -I -S -B docker/test_verify.py -v.""" + +from copy import deepcopy +import importlib.util +from pathlib import Path +import unittest + + +spec = importlib.util.spec_from_file_location( + 'truf_verify', Path(__file__).with_name('verify.py')) +verify = importlib.util.module_from_spec(spec) +spec.loader.exec_module(verify) + + +class ValidateComposeTests(unittest.TestCase): + def setUp(self): + # Only project/images are needed; skip all host setup and never run main. + self.verifier = object.__new__(verify.Verifier) + self.verifier.project = 'truf-worker-test-' + 'a' * 32 + self.verifier.images = {'runtime': 'sha256:' + '1' * 64, + 'test': 'sha256:' + '2' * 64} + common = { + 'pull_policy': 'never', 'read_only': True, 'network_mode': 'none', + 'environment': dict(verify.PROXY_ENV), 'init': False, + 'user': '10001:10001', 'cap_drop': ['ALL'], + 'security_opt': ['no-new-privileges:true'], 'restart': 'no', + 'cpus': 2, 'mem_limit': 6442450944, 'pids_limit': 512, + 'shm_size': 268435456, 'stop_signal': 'SIGTERM', + 'stop_grace_period': '10m0s', + 'logging': {'driver': 'json-file', + 'options': {'max-size': '16m', 'max-file': '4'}}, + 'tmpfs': [target + ':' + options for target, options in verify.TMPFS.items()], + } + data = {'type': 'volume', 'source': 'data', 'target': '/data', 'volume': {}} + tools = {'type': 'volume', 'source': 'tools', 'target': '/opt/truf/tests', + 'read_only': True, 'volume': {'nocopy': True}} + services = {name: deepcopy(common) + for name in ('tools', 'provision', 'prepare', 'runtime', 'stopped')} + for name, service in services.items(): + service['image'] = self.verifier.images['test' if name == 'tools' else 'runtime'] + service['volumes'] = [deepcopy(data)] + # Compose 5 omits false nocopy/read_only; byte sizes are normalized integers. + services['tools'].update( + volumes=[{'type': 'volume', 'source': 'tools', + 'target': '/opt/truf/tests', 'volume': {}}], + entrypoint=[*verify.PYTHON, '-c'], command=['pass']) + services['provision'].update( + user='0:0', cap_add=['CHOWN', 'DAC_OVERRIDE', 'FOWNER'], + entrypoint=None, command=['provision']) + services['prepare'].update( + volumes=[deepcopy(data), deepcopy(tools)], + entrypoint=[*verify.PYTHON, verify.DRIVER], + command=['prepare', '--config', verify.CONFIG]) + services['runtime'].update( + volumes=[deepcopy(data), deepcopy(tools)], entrypoint=None, + command=['run', '--config', verify.CONFIG], + healthcheck={'test': list(verify.HEALTH), 'interval': '5s', + 'timeout': '15s', 'start_period': '4m0s', 'retries': 3}) + services['stopped'].update( + volumes=[dict(data, read_only=True, volume={'nocopy': True})], + entrypoint=[*verify.PYTHON, '-c'], command=['pass']) + self.value = { + 'name': self.verifier.project, 'services': services, + 'volumes': {name: {'name': self.verifier.project + '_' + name, 'driver': 'local'} + for name in ('data', 'tools')}, + } + + def test_normalized_fixture(self): + self.verifier.validate_compose(self.value) + + def test_rejects_production_project_and_volume_names(self): + value = deepcopy(self.value) + value['name'] = 'truf-docker' + with self.assertRaisesRegex(verify.Failure, '^worker_test_project_guard$'): + self.verifier.validate_compose(value) + value = deepcopy(self.value) + value['volumes']['data']['name'] = 'truf-docker_data' + with self.assertRaisesRegex(verify.Failure, '^production_volume_forbidden$'): + self.verifier.validate_compose(value) + + def test_rejects_bind_mounts_and_published_ports(self): + value = deepcopy(self.value) + value['services']['runtime']['volumes'][0] = { + 'type': 'bind', 'source': r'D:\truf-docker', 'target': '/data', + } + with self.assertRaisesRegex(verify.Failure, '^bind_mount_forbidden$'): + self.verifier.validate_compose(value) + value = deepcopy(self.value) + value['services']['runtime']['ports'] = [{'target': 5432, 'published': '5432'}] + with self.assertRaisesRegex(verify.Failure, '^published_port_contract$'): + self.verifier.validate_compose(value) + + def test_rejects_shared_images_and_nonisolated_networks(self): + value = deepcopy(self.value) + value['services']['runtime']['image'] = 'truf-local:runtime' + with self.assertRaisesRegex(verify.Failure, '^worker_test_image_reference_guard$'): + self.verifier.validate_compose(value) + value = deepcopy(self.value) + value['services']['runtime']['network_mode'] = 'bridge' + with self.assertRaisesRegex(verify.Failure, '^internal_network_contract$'): + self.verifier.validate_compose(value) + + def test_seed_accepts_false_or_omitted_nocopy(self): + for volume in (None, {}, {'nocopy': False}): + with self.subTest(volume=volume): + value = deepcopy(self.value) + mount = value['services']['tools']['volumes'][0] + if volume is None: + del mount['volume'] + else: + mount['volume'] = volume + self.verifier.validate_compose(value) + + def test_seed_rejects_nocopy_other_than_false(self): + for nocopy in (True, None, 0, 'false'): + with self.subTest(nocopy=nocopy): + value = deepcopy(self.value) + value['services']['tools']['volumes'][0]['volume']['nocopy'] = nocopy + with self.assertRaisesRegex(verify.Failure, '^tools_copy_up_contract$'): + self.verifier.validate_compose(value) + + def test_consumers_require_explicit_true_nocopy(self): + for name in ('prepare', 'runtime'): + for volume in (None, {}, {'nocopy': False}, {'nocopy': 1}, {'nocopy': 'true'}): + with self.subTest(service=name, volume=volume): + value = deepcopy(self.value) + mount = value['services'][name]['volumes'][1] + if volume is None: + del mount['volume'] + else: + mount['volume'] = volume + with self.assertRaisesRegex(verify.Failure, '^tools_copy_up_contract$'): + self.verifier.validate_compose(value) + + def test_runtime_and_provision_inherit_entrypoint(self): + for name in ('runtime', 'provision'): + for omitted in (False, True): + with self.subTest(service=name, omitted=omitted): + value = deepcopy(self.value) + if omitted: + del value['services'][name]['entrypoint'] + self.verifier.validate_compose(value) + + def test_runtime_and_provision_reject_entrypoint_overrides(self): + for name, guard in (('runtime', 'production_entrypoint_contract'), + ('provision', 'prepare_command_contract')): + for entrypoint in ([], ['/bin/sh', '-c'], ''): + with self.subTest(service=name, entrypoint=entrypoint): + value = deepcopy(self.value) + value['services'][name]['entrypoint'] = entrypoint + with self.assertRaisesRegex(verify.Failure, '^' + guard + '$'): + self.verifier.validate_compose(value) + + def test_foreign_snapshot_excludes_only_revalidated_owned_identities(self): + owned_container = 'a' * 64 + foreign_container = 'b' * 64 + owned_volume = {'name': 'owned', 'identity': 'captured'} + foreign_volume = {'name': 'foreign', 'identity': 'stable'} + current = {'owned': owned_volume, 'foreign': foreign_volume} + verifier = object.__new__(verify.Verifier) + verifier.owned_containers = {owned_container: {'id': owned_container}} + verifier.owned_volumes = {'owned': owned_volume} + verifier.metadata_names = lambda kind: ( + {owned_container, foreign_container} if kind == 'container' + else {'owned', 'foreign'} + ) + verifier.container_metadata = lambda identifier: {'id': identifier} + verifier.volume_metadata = lambda name: current[name] + inspected = [] + verifier.inspect = lambda identifier, label: inspected.append((identifier, label)) + + snapshot = verifier.metadata_snapshot(exclude_owned=True) + + self.assertEqual(snapshot, { + 'containers': {foreign_container: {'id': foreign_container}}, + 'volumes': {'foreign': foreign_volume}, + }) + self.assertEqual(inspected, [(owned_container, 'foreign_owned_exclusion_guard')]) + + current['owned'] = {'name': 'owned', 'identity': 'replaced'} + self.assertIn('owned', verifier.metadata_snapshot(exclude_owned=True)['volumes']) + + def test_cleanup_uses_exact_resource_operations_and_foreign_guard(self): + source = Path(verify.__file__).read_text(encoding='ascii') + self.assertNotIn("compose_command('down'", source) + self.assertNotIn("'--force-recreate'", source) + self.assertIn("'cleanup_container_remove_guard'", source) + self.assertIn("'cleanup_volume_identity_changed'", source) + self.assertIn('self.assert_foreign_unchanged()', source) + self.assertIn("'failure_stop_ownership_guard'", source) + self.assertIn("'stage': label, 'class': failure_class", source) + self.assertIn("'exit_code': exit_code", source) + + +if __name__ == '__main__': + unittest.main() diff --git a/docker/test_windows_snapshot.py b/docker/test_windows_snapshot.py new file mode 100644 index 0000000..80b0e32 --- /dev/null +++ b/docker/test_windows_snapshot.py @@ -0,0 +1,1097 @@ +"""Synthetic tests only: python -I -S -B docker/test_windows_snapshot.py -v.""" + +import contextlib +import ctypes +from dataclasses import replace +import hashlib +import importlib.util +import io +import json +import os +from pathlib import Path +import stat +import subprocess +import sys +import tarfile +import tempfile +from types import SimpleNamespace +import unittest +from unittest.mock import Mock, patch + + +spec = importlib.util.spec_from_file_location('windows_snapshot', Path(__file__).with_name('windows_snapshot.py')) +snapshot = importlib.util.module_from_spec(spec) +sys.modules[spec.name] = snapshot +spec.loader.exec_module(snapshot) +REAL_SECURE_PATH = snapshot._secure_path +REAL_LOAD_SOURCE = snapshot._load_source + +SECRET = 'synthetic-password-do-not-print' + + +class Fixture(unittest.TestCase): + def setUp(self): + temporary_root = Path(tempfile.gettempdir()) / 'opencode' + self.temp = tempfile.TemporaryDirectory(prefix='truf-snapshot-test-', + dir=temporary_root if temporary_root.is_dir() else None) + self.addCleanup(self.temp.cleanup) + self.base = Path(self.temp.name) + self.root, self.bundles = self.base / 'source', self.base / 'bundles' + self.imports = self.base / 'clone/docker/imports' + for path in (self.root / 'app', self.root / 'runtime', self.bundles, self.imports): + path.mkdir(parents=True) + self.output = self.imports / 'unique' + for name, value in (('SOURCE_ROOT', self.root), ('BUNDLE_ROOT', self.bundles), + ('POSTGRES_DATA', self.base / 'pgdata'), ('IMPORTS_ROOT', self.imports)): + patcher = patch.object(snapshot, name, value) + patcher.start() + self.addCleanup(patcher.stop) + # An accidental integration call fails before reaching any original code + # or real child process. Individual tests substitute synthetic processes. + for name in ('_load_source',): + patcher = patch.object(snapshot, name, side_effect=AssertionError('source access forbidden')) + patcher.start() + self.addCleanup(patcher.stop) + for name in ('Popen', 'run'): + patcher = patch.object(snapshot.subprocess, name, side_effect=AssertionError('process forbidden')) + patcher.start() + self.addCleanup(patcher.stop) + patcher = patch.object(snapshot, '_secure_path') + self.secure = patcher.start() + self.addCleanup(patcher.stop) + self.report = Mock() + self.handlers = {number: snapshot.signal.default_int_handler + for number in (snapshot.signal.SIGINT, getattr(snapshot.signal, 'SIGBREAK', None)) + if number is not None} + self.previous_handlers = self.handlers.copy() + + def register(number, handler): + previous = self.handlers[number] + self.handlers[number] = handler + return previous + + patcher = patch.object(snapshot.signal, 'signal', side_effect=register) + self.signal = patcher.start() + self.addCleanup(patcher.stop) + self.addCleanup(lambda: self.assertEqual(self.handlers, self.previous_handlers)) + + def send_signal(self, number): + # Invoke only this fixture's registered Python handler, not an OS signal. + self.handlers[number](number, None) + + def put(self, relative, content=b'fixture', bundles=False): + path = (self.bundles if bundles else self.root) / relative + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(content) + return path + + def runtime(self, state='STOPPED'): + calls = [] + authority = SimpleNamespace(acquired=False) + + def acquire(): + authority.acquired = True + calls.append('acquire') + + def release(): + calls.append('release') + authority.acquired = False + + authority.acquire, authority.release = Mock(side_effect=acquire), Mock(side_effect=release) + backend = SimpleNamespace(state=state, close=Mock()) + backend.probe = Mock(side_effect=lambda: SimpleNamespace(kind=backend.state, detail=SECRET)) + identity = {'pg_major': 16, 'system_identifier': '1234567890123456789', + 'data_directory': str(snapshot.POSTGRES_DATA), 'database': 'fixture', 'user': 'fixture', 'port': 5432} + + def start(config, backend): + self.assertTrue(authority.acquired) + calls.append('start') + backend.state = 'READY' + print(SECRET) + return SimpleNamespace(kind='READY') + + def stop(config, backend): + self.assertTrue(authority.acquired) + calls.append('stop') + backend.state = 'STOPPED' + print(SECRET, file=sys.stderr) + return SimpleNamespace(completed=True, stopped=True) + + pg = SimpleNamespace( + ProbeKind=SimpleNamespace(STOPPED='STOPPED', READY='READY'), + configured_cluster_values=Mock(return_value={'database': 'fixture', 'user': 'fixture', 'port': 5432}), + verify_cluster_identity=Mock(return_value=identity), PostgresBackend=Mock(return_value=backend), + maintenance_start=Mock(side_effect=start), maintenance_stop=Mock(side_effect=stop)) + source = SimpleNamespace(pg=pg, security=SimpleNamespace(ClusterAuthorityLock=Mock(return_value=authority)), + config={}, dsn='postgresql://fixture:' + SECRET + '@127.0.0.1:5432/fixture', inputs={}) + self.calls, self.authority, self.backend, self.identity, self.source = calls, authority, backend, identity, source + return source + + @contextlib.contextmanager + def capture_context(self, database=None): + source = self.runtime() + + def dump(*args): + self.assertTrue(self.authority.acquired) + self.assertEqual(self.backend.state, 'READY') + self.calls.append('database') + self.output.joinpath('database.dump').write_bytes(b'PGDMPfixture') + return {'version_num': 160004, 'system_identifier': self.identity['system_identifier'], + 'database_name': 'fixture', 'table_counts': {'fixture': 2}, 'bytes': 12, 'sha256': '0' * 64, + 'sequence_states': {}, 'sequence_count': 0} + + with patch.object(snapshot, '_load_source', return_value=source), \ + patch.object(snapshot, '_capture_database', side_effect=database or dump), \ + patch.object(snapshot, '_verify_private_acl'), \ + contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()): + yield source + + +class SelectionTests(Fixture): + def test_active_mappings_and_projection_sidecars(self): + expected = {} + for folder in snapshot.ACTIVE_DIRS: + name = 'runtime/' + folder + '/data.jsonl' + self.put(name) + expected['runtime-linux/' + folder + '/data.jsonl'] = self.root / name + for name in ('secrets.yaml', 'trufflehog-custom-detectors.yaml'): + self.put('app/' + name) + expected['config/' + name] = self.root / 'app' / name + self.put('runtime/proxy.txt', b'not-a-real-proxy') + expected['runtime-linux/proxy.txt'] = self.root / 'runtime/proxy.txt' + for folder in ('results', 'keychecks'): + for name in ('segment-00001.jsonl', 'scan_errors.log', 'scan_errors.log.22', 'stream.log.3', + 'data.publication-ledger.sqlite3', 'data.publication-ledger.sqlite3-wal', + 'data.publication-ledger.sqlite3-shm', 'data.sqlite-journal'): + relative = 'runtime/' + folder + '/' + name + self.put(relative) + expected['runtime-linux/' + folder + '/' + name] = self.root / relative + self.put('ready/id/segment.jsonl', bundles=True) + expected['scanner-result-bundles/ready/id/segment.jsonl'] = self.bundles / 'ready/id/segment.jsonl' + self.assertEqual({k: v.source for k, v in snapshot._inventory().files.items()}, expected) + + def test_all_reviewed_archive_mappings(self): + names = ( + 'app/config.yaml', 'app/config.yaml.old', 'app/secrets.yaml.backup', 'app/.streamlit/config.toml', + 'app/secrets.yaml.backup.log', 'scan_errors.log.1', 'scan_results.jsonl.segments/000001', + 'found_secrets.jsonl.publication-ledger.sqlite3-wal', + '.env.postgres', 'docker-compose.postgres.yml', 'found_secrets.jsonl', 'scan_results.jsonl', + 'scan_errors.log', 'checked_provider.txt', 'todo_provider.txt', 'state/legacy.json', + 'runner_state.json', 'scanner.db', 'scanner.db-wal', 'app/scanner.db-shm', + 'runtime/backups/old/scanner.db', 'runtime/' + snapshot.KEYCHECK_COPY + '/stream.log.1', + 'runtime/keychecks.7z', 'runtime/orkey.txt', 'runtime/imports/old.dump', 'runtime/notes.md', + 'runtime/check-openrouter-keys.ps1', 'runtime/control/migration-report.json', + ) + for name in names: + self.put(name) + self.assertEqual(set(snapshot._inventory().files), {'windows-archive/' + name for name in names}) + + def test_state_scratch_archival_not_active(self): + names = ('scan_limiter.db', 'scan_limiter-copy.db-wal', 'scan_limiter.db-shm', + 'nested/file.tmp.123', 'janitor.cursor.json', 'tmpdir.tmp/retained.json') + for name in names: + self.put('runtime/state/' + name) + self.put('runtime/state/useful.json') + self.assertEqual(set(snapshot._inventory().files), + {'windows-archive/runtime/state/' + name for name in names} + | {'runtime-linux/state/useful.json'}) + + def test_excludes_unreviewed_data_logs_locks_and_code_caches(self): + for name in ( + 'runtime/state/gharchive_cache/file.json.gz', 'runtime/state/debug.log.4', + 'runtime/state/writer.lock.old', 'runtime/state/process.pid', 'runtime/queues/__pycache__/x.pyc', + 'runtime/backups/ordinary.log.9', 'runtime/results/file.lock', 'runtime/results/__pycache__/x.pyc', + 'runtime/downloads/archive.gz', 'runtime/git/repo/file', 'runtime/traces/t.json', + 'runtime/freeze-diagnostics/report.json', 'runtime/postgres/data/physical', + 'runtime/postgres/pgsql/bin/pg_dump.exe', 'app/scanner.py', 'app/config.yaml.lock', + '.opencode/private', 'tests/test.py', + ): + self.put(name) + self.assertEqual(snapshot._inventory().files, {}) + + def test_control_never_includes_authority_or_non_reports(self): + for name in ('supervisor.instance.json', 'supervisor-report.json', 'client-token.json', + 'authority.json', 'manifest.json', 'lock.json', 'worker.pid', 'data.txt', 'report.json'): + self.put('runtime/control/' + name) + self.assertEqual(set(snapshot._inventory().files), {'windows-archive/runtime/control/report.json'}) + + def test_special_scan_errors_and_ledger_names_outside_results_survive(self): + for name in ('scan_errors.log.8', 'name.log.publication-ledger.sqlite3-wal'): + self.put('runtime/backups/' + name) + self.assertEqual(len(snapshot._inventory().files), 2) + + def test_selected_symlink_rejected(self): + link = self.put('runtime/proxy.txt') + original = Path.lstat + + def lstat(path): + if path == link: + return SimpleNamespace(st_mode=stat.S_IFLNK, st_file_attributes=0, st_nlink=1) + return original(path) + + with patch.object(Path, 'lstat', lstat), self.assertRaises(snapshot.Failure): + snapshot._inventory() + + def test_selected_hardlink_rejected(self): + target = self.put('ordinary.txt') + os.link(target, self.root / 'runtime/proxy.txt') + with self.assertRaises(snapshot.Failure): + snapshot._inventory() + + def test_reparse_directory_is_not_traversed(self): + child = self.put('runtime/results/junction/never-opened') + junction = child.parent + original = Path.lstat + + def lstat(path): + if path == junction: + return SimpleNamespace(st_mode=stat.S_IFDIR, st_file_attributes=0x400, st_nlink=1) + self.assertNotEqual(path, child) + return original(path) + + with patch.object(Path, 'lstat', lstat), patch.object(snapshot.os, 'scandir', wraps=os.scandir) as scan: + with self.assertRaises(snapshot.Failure): + snapshot._inventory() + self.assertNotIn(junction, [call.args[0] for call in scan.call_args_list]) + + def test_special_file_rejected(self): + with self.assertRaises(snapshot.Failure): + snapshot._check_type(SimpleNamespace(st_mode=stat.S_IFIFO, st_file_attributes=0, st_nlink=1)) + + def test_unsafe_tar_destinations(self): + for name in ('/absolute', '../up', 'a/../b', 'a//b', 'a/./b', 'a\\b', 'D:/file', 'a\x00b'): + with self.subTest(name=repr(name)), self.assertRaises(snapshot.Failure): + snapshot._destination(name, set(), set()) + + def test_casefold_duplicate_and_file_directory_conflicts(self): + for first, second in (('runtime-linux/X', 'runtime-linux/x'), + ('runtime-linux/x', 'runtime-linux/X/child'), + ('runtime-linux/X/child', 'runtime-linux/x'), + ('runtime-linux/strasse', 'runtime-linux/stra\u00dfe')): + used, parents = set(), set() + snapshot._destination(first, used, parents) + with self.assertRaises(snapshot.Failure): + snapshot._destination(second, used, parents) + + +class ArchiveTests(Fixture): + def test_stable_large_file_path_and_handle_fingerprints_match(self): + for index in range(20): + path = self.put('runtime/results/large-' + str(index), b'x' * (snapshot.BLOCK + 7)) + expected = snapshot._fingerprint(snapshot._file_info(path)) + with path.open('rb', buffering=0) as handle: + actual = snapshot._fingerprint(os.fstat(handle.fileno())) + self.assertEqual(expected[:6] + expected[7:], actual[:6] + actual[7:]) + handle.read(snapshot.BLOCK) + self.assertEqual(snapshot._fingerprint(os.fstat(handle.fileno())), actual) + + def test_streamed_tar_exact_hashes_sizes_and_regular_members(self): + self.put('runtime/results/segment.jsonl', b'a' * (snapshot.BLOCK + 7)) + self.put('runtime/state/file.json', b'{}') + self.output.mkdir() + inventory = snapshot._inventory() + files, archive = snapshot._write_tar(self.output, None, inventory, self.report) + data = self.output.joinpath('files.tar').read_bytes() + self.assertEqual(archive, {'bytes': len(data), 'sha256': hashlib.sha256(data).hexdigest()}) + with tarfile.open(fileobj=io.BytesIO(data)) as tar: + self.assertEqual(len(tar.getmembers()), len(files)) + for member, record in zip(tar.getmembers(), files): + content = tar.extractfile(member).read() + self.assertTrue(member.isfile()) + self.assertFalse(member.issym() or member.islnk()) + self.assertEqual(member.mode, 0o600) + self.assertEqual(record, {'path': member.name, 'size': len(content), + 'sha256': hashlib.sha256(content).hexdigest()}) + + def test_changed_file_before_copy_rejected(self): + path = self.put('runtime/results/a', b'before') + inventory = snapshot._inventory() + path.write_bytes(b'after-and-longer') + self.output.mkdir() + with self.assertRaises(snapshot.Failure): + snapshot._write_tar(self.output, None, inventory, self.report) + + def test_changed_file_during_copy_rejected_by_fstat(self): + path = self.put('runtime/results/a', b'before') + inventory = snapshot._inventory() + self.output.mkdir() + original = snapshot.HashReader.read + + def read(reader, size): + block = original(reader, size) + with path.open('ab') as handle: + handle.write(b'changed') + return block + + with patch.object(snapshot.HashReader, 'read', read), self.assertRaises(snapshot.Failure): + snapshot._write_tar(self.output, None, inventory, self.report) + + def test_changed_open_handle_fails_before_read(self): + self.put('runtime/results/a') + inventory = snapshot._inventory() + name, entry = next(iter(inventory.files.items())) + inventory.files[name] = replace(entry, fingerprint=entry.fingerprint[:-1] + (999,)) + self.output.mkdir() + with self.assertRaises(snapshot.Failure): + snapshot._write_tar(self.output, None, inventory, self.report) + + def test_handle_ctime_change_during_copy_is_still_rejected(self): + self.put('runtime/results/a') + inventory = snapshot._inventory() + self.output.mkdir() + original = os.fstat + calls = [] + + def fstat(fd): + info = original(fd) + calls.append(fd) + if len(calls) == 3: + fields = ('st_dev', 'st_ino', 'st_mode', 'st_nlink', 'st_size', 'st_mtime_ns', 'st_file_attributes') + return SimpleNamespace(**{name: getattr(info, name, 0) for name in fields}, + st_ctime_ns=info.st_ctime_ns + 1) + return info + + with patch.object(snapshot.os, 'fstat', side_effect=fstat), self.assertRaises(snapshot.Failure): + snapshot._write_tar(self.output, None, inventory, self.report) + + def test_inventory_detects_addition_deletion_and_rename(self): + path = self.put('runtime/results/a') + before = snapshot._inventory() + new = self.put('runtime/results/b') + self.assertNotEqual(snapshot._inventory(), before) + new.unlink() + path.rename(path.with_name('renamed')) + self.assertNotEqual(snapshot._inventory(), before) + + +class OutputTests(Fixture): + def test_output_requires_exact_windows_parent_and_safe_new_name(self): + with patch.object(snapshot, 'IMPORTS_ROOT', Path(r'D:\truf-docker\docker\imports')): + self.assertEqual(str(snapshot._output_path(r'D:\truf-docker\docker\imports\new-20260915')), + str(Path(r'D:\truf-docker\docker\imports\new-20260915'))) + for path in (r'D:\other\new', r'D:truf-docker\docker\imports\new', 'relative', + r'D:\truf-docker\docker\imports\old\new', + r'D:\truf-docker\docker\imports\..\imports\new', + r'D:\truf-docker\docker\imports\new:stream', + r'D:\truf-docker\docker\imports\CON', r'D:\truf-docker\docker\imports\new.'): + with self.subTest(path=path), self.assertRaises(snapshot.Failure): + snapshot._output_path(path) + + def test_existing_output_is_never_reused(self): + self.output.mkdir() + marker = self.output / 'marker' + marker.write_bytes(b'unchanged') + with self.assertRaises(FileExistsError): + snapshot._prepare_output(self.output, None) + self.assertEqual(marker.read_bytes(), b'unchanged') + self.secure.assert_not_called() + + def test_output_private_before_sensitive_writes(self): + seen = [] + + def secure(path, security, directory=False): + seen.append((path.name, directory)) + if not directory: + self.assertEqual(path.stat().st_size, 0) + + self.secure.side_effect = secure + snapshot._prepare_output(self.output, None) + with snapshot._output_file(self.output / 'database.dump', None) as handle: + self.assertEqual(seen, [('unique', True), ('database.dump', False)]) + handle.write(b'synthetic-sensitive-data') + + def test_acl_failure_prevents_writes_and_manifest(self): + self.secure.side_effect = RuntimeError(SECRET) + with self.capture_context(), self.assertRaises(RuntimeError): + snapshot.capture(self.output, self.report) + self.assertEqual(list(self.output.iterdir()), []) + self.source.pg.maintenance_start.assert_not_called() + + def test_private_acl_accepts_only_exact_user_and_system(self): + sid = 'S-1-5-21-111-222-333-1001' + security = SimpleNamespace(reject_reparse_components=Mock(), _windows_current_user_sid=lambda: sid) + good = 'O:' + sid + 'D:P(A;OICI;FA;;;' + sid + ')(A;OICI;FA;;;SY)' + for sddl in (good, good + '(A;OICI;FA;;;BA)', good.replace(';SY)', ';WD)'), good.replace('D:P', 'D:')): + security._windows_private_sddl = lambda path: sddl + if sddl == good: + snapshot._verify_private_acl(self.output, security, True) + else: + with self.assertRaises(snapshot.Failure): + snapshot._verify_private_acl(self.output, security, True) + + def test_acl_helper_is_narrowed_using_no_reparse_native_handle(self): + sid = 'S-1-5-21-111-222-333-1001' + descriptors = [] + + class Information(ctypes.Structure): + _fields_ = [('dwFileAttributes', ctypes.c_uint)] + + def convert(sddl, *args): + descriptors.append(sddl) + return 1 + + security = SimpleNamespace( + harden_private_directory=Mock(), harden_private_file=Mock(), + reject_reparse_components=Mock(), _windows_current_user_sid=lambda: sid, + _windows_private_sddl=lambda path: 'O:' + sid + descriptors[-1], + _CONVERT_SDDL=Mock(side_effect=convert), _CREATE_FILE=Mock(return_value=123), + _BY_HANDLE_FILE_INFORMATION=Information, _GET_FILE_INFORMATION=Mock(return_value=1), + _SET_KERNEL_OBJECT_SECURITY=Mock(return_value=1), _CLOSE_HANDLE=Mock(), _LOCAL_FREE=Mock()) + REAL_SECURE_PATH(self.output, security, directory=True) + security.harden_private_directory.assert_called_once_with(str(self.output)) + self.assertEqual(descriptors, ['D:P(A;OICI;FA;;;' + sid + ')(A;OICI;FA;;;SY)']) + self.assertEqual(security._CREATE_FILE.call_args.args[5], 0x2200000) + self.assertEqual(security._SET_KERNEL_OBJECT_SECURITY.call_args.args[1], 0x80000004) + security._CLOSE_HANDLE.assert_called_once_with(123) + security._LOCAL_FREE.assert_called_once() + + def test_partial_manifest_is_not_published_on_write_failure(self): + self.output.mkdir() + original = snapshot.HashWriter.write + + def write(writer, block): + original(writer, block) + raise OSError(SECRET) + + with patch.object(snapshot.HashWriter, 'write', write), self.assertRaises(OSError): + snapshot._publish_manifest(self.output, None, {'format': 'truf-windows-snapshot-v1'}) + self.assertTrue(self.output.joinpath('manifest.json.partial').exists()) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + + +class LoaderTests(Fixture): + def test_loader_uses_original_bootstrap_and_original_environment_only(self): + for name in ('child_bootstrap.py', 'postgres_runtime.py', 'runtime_security.py', 'config.yaml'): + self.put('app/' + name) + self.put('.env.postgres') + self.put('runtime/postgres/cluster_identity.json') + bootstrap = SimpleNamespace(_enable_dependency_paths=Mock()) + spec = SimpleNamespace(loader=SimpleNamespace(exec_module=Mock())) + config = {'global': {'root_dir': str(self.root), 'project_dir': str(self.root / 'app'), + 'runtime_dir': str(self.root / 'runtime'), 'postgres_data_dir': str(snapshot.POSTGRES_DATA), + 'result_bundle_dir': str(self.bundles)}} + source = self.runtime() + + def load_environment(*args): + for key in ('TRUF_POSTGRES_PASSWORD', 'PGPASSWORD', 'DATABASE_URL', 'SCANNER_DB_URL', 'SCANNER_RUNTIME_DIR'): + self.assertNotIn(key, os.environ) + return str(self.root / '.env.postgres') + + pg = SimpleNamespace( + __name__='postgres_runtime', __file__=str(self.root / 'app/postgres_runtime.py'), + _load_config=Mock(return_value=config), load_postgres_environment=Mock(side_effect=load_environment), + postgres_runtime_paths=Mock(return_value={'data_dir': str(snapshot.POSTGRES_DATA), + 'postgres_dir': str(self.root / 'runtime/postgres')}), + canonical_database_url=Mock(return_value=source.dsn)) + security = SimpleNamespace(__name__='runtime_security', __file__=str(self.root / 'app/runtime_security.py'), + preflight_lifecycle_paths=Mock()) + inherited = {key: SECRET for key in ('TRUF_POSTGRES_PASSWORD', 'PGPASSWORD', 'DATABASE_URL', + 'SCANNER_DB_URL', 'SCANNER_RUNTIME_DIR')} + with patch.object(snapshot.importlib.util, 'spec_from_file_location', return_value=spec), \ + patch.object(snapshot.importlib.util, 'module_from_spec', return_value=bootstrap), \ + patch.object(snapshot.importlib, 'import_module', side_effect=[pg, security]), \ + patch.object(sys, 'path', sys.path[:]), patch.dict(os.environ, inherited): + loaded = REAL_LOAD_SOURCE() + bootstrap._enable_dependency_paths.assert_called_once_with('postgres-runtime') + spec.loader.exec_module.assert_called_once_with(bootstrap) + self.assertIs(loaded.pg, pg) + self.assertEqual(len(loaded.inputs), 3) + snapshot.subprocess.Popen.assert_not_called() + snapshot.subprocess.run.assert_not_called() + + +class FakeProcess: + def __init__(self, output=b'', returncode=0): + self.stdin = io.BytesIO() + self.stdout = io.BytesIO(output) + self.returncode = returncode + self.killed = False + + def wait(self, timeout): + return self.returncode + + def poll(self): + return self.returncode + + def kill(self): + self.killed = True + self.returncode = -1 + + +class DatabaseTests(Fixture): + def metadata(self): + return {'version_num': 160004, 'system_identifier': self.identity['system_identifier'], + 'database_name': 'fixture', 'user_name': 'fixture', 'port': 5432, + 'data_directory': str(snapshot.POSTGRES_DATA), 'in_recovery': False, 'read_only': 'on', + 'snapshot': '00000003-0000001A-1', 'database_bytes': 100, + 'tables': ['normal', 'a"; odd', 'runtime_operations', + 'runtime_operations_control', 'runtime_audit_events'], + 'other_clients': 0, 'all_sessions_visible': True, + 'sequences': [['public', 'counter'], ['Other"Schema', 'Seq";name'], + ['public', 'runtime_audit_events_id_seq']]} + + def test_database_identity_requires_every_bound_value(self): + self.runtime() + snapshot._validate_database(self.metadata(), self.identity) + for key, value in (('version_num', 150004), ('system_identifier', 'wrong'), ('database_name', 'wrong'), + ('user_name', 'wrong'), ('port', 6543), ('data_directory', 'wrong'), + ('in_recovery', True), ('read_only', 'off'), ('other_clients', 1), + ('all_sessions_visible', False), ('other_clients', False), + ('snapshot', 'bad\ntext'), ('database_bytes', -1), ('tables', ['duplicate', 'duplicate']), + ('sequences', [['public', 'same'], ['public', 'same']]), ('sequences', ['not-a-pair']), + ('sequences', [['public', 'bad\0name']])): + with self.subTest(key=key), self.assertRaises(snapshot.Failure): + snapshot._validate_database(dict(self.metadata(), **{key: value}), self.identity) + + def test_identity_compares_original_configuration_and_dsn(self): + source = self.runtime() + snapshot._validate_identity(source, self.identity) + for key, value in (('pg_major', 15), ('database', 'other'), ('user', 'other'), ('port', 6543), + ('data_directory', 'wrong'), ('system_identifier', 'not-numeric')): + with self.subTest(key=key), self.assertRaises(snapshot.Failure): + snapshot._validate_identity(source, dict(self.identity, **{key: value})) + + def test_client_environment_is_read_only_and_contains_no_inherited_overrides(self): + source = self.runtime() + with patch.dict(os.environ, {'PGSERVICE': SECRET, 'PGHOST': 'foreign', 'PGOPTIONS': 'unsafe', + 'DATABASE_URL': SECRET, 'SCANNER_DB_URL': SECRET, 'PATH': SECRET}): + env = snapshot._client_environment(source.dsn) + self.assertEqual(env['PGPASSWORD'], SECRET) + self.assertEqual(env['PGHOST'], '127.0.0.1') + self.assertIn('default_transaction_read_only=on', env['PGOPTIONS']) + for key in ('PGSERVICE', 'DATABASE_URL', 'SCANNER_DB_URL', 'PATH'): + self.assertNotIn(key, env) + + def test_query_uses_stdin_not_arguments_and_bounded_json(self): + process = FakeProcess(b'17\n') + sql = 'SELECT count(*) FROM ONLY "public"."a""; odd";' + with patch.object(snapshot, '_deadline', wraps=snapshot._deadline) as deadline: + self.assertEqual(snapshot._query(process, sql, 99), 17) + self.assertEqual(process.stdin.getvalue(), (sql + '\n').encode()) + deadline.assert_called_once_with(process, 99) + for output in (SECRET.encode() + b'\n', b'1', b'x' * 33 + b'\n'): + with patch.object(snapshot, 'MAX_METADATA', 32), self.assertRaises(snapshot.Failure): + snapshot._query(FakeProcess(output), 'SELECT 1;') + + def test_counts_and_dump_use_shared_snapshot_and_no_secret_arguments(self): + source = self.runtime() + for name in ('psql.exe', 'pg_dump.exe'): + self.put('runtime/postgres/pgsql/bin/' + name) + self.output.mkdir() + statements, commands = [], [] + metadata = self.metadata() + psql, dump = FakeProcess(), FakeProcess(b'PGDMPsynthetic-dump') + + @contextlib.contextmanager + def client(command, env, timeout, interactive=False): + commands.append((command, env, timeout)) + yield psql if interactive else dump + + def query(process, sql, timeout=snapshot.QUERY_TIMEOUT): + statements.append((sql, timeout)) + if sql == snapshot.DATABASE_METADATA: + return metadata + if 'count(*) FROM ONLY' in sql: + return 7 + if "'last_value', last_value" in sql: + return {'last_value': 42, 'is_called': True} + return 0 + + def version(command, **kwargs): + commands.append((command, kwargs['env'], kwargs['timeout'])) + return SimpleNamespace(returncode=0, stdout=(Path(command[0]).stem + ' (PostgreSQL) 16.4\n').encode()) + + with patch.object(snapshot, '_client', client), patch.object(snapshot, '_query', query), \ + patch.object(snapshot.subprocess, 'run', side_effect=version): + result = snapshot._capture_database(source, self.identity, self.output, snapshot._inventory(), self.report) + self.assertEqual(result['table_counts'], { + 'normal': 7, + 'a"; odd': 7, + 'runtime_operations': 7, + 'runtime_operations_control': 7, + 'runtime_audit_events': 7, + }) + self.assertEqual(result['sequence_states'], { + 'Other"Schema': {'Seq";name': {'last_value': 42, 'is_called': True}}, + 'public': { + 'counter': {'last_value': 42, 'is_called': True}, + 'runtime_audit_events_id_seq': {'last_value': 42, 'is_called': True}, + }, + }) + self.assertEqual(result['sequence_count'], 3) + self.assertIn('non-MVCC', result['sequence_state_mode']) + sequence_sql = ("SELECT pg_catalog.json_build_object('last_value', last_value, 'is_called', is_called) " + 'FROM "Other""Schema"."Seq"";name";') + self.assertEqual(statements.count((sequence_sql, snapshot.QUERY_TIMEOUT)), 2) + self.assertIn(('SELECT count(*) FROM ONLY "public"."a""; odd";', snapshot.COUNT_TIMEOUT), statements) + dump_command = commands[-1][0] + for flag in ('--format=custom', '--no-owner', '--no-acl', '--no-tablespaces', '--compress=1', + '--no-password', '--snapshot=' + metadata['snapshot']): + self.assertIn(flag, dump_command) + for command, env, timeout in commands: + self.assertNotIn(SECRET, repr(command)) + self.assertEqual(env['PGPASSWORD'], SECRET) + self.assertGreater(timeout, 0) + data = self.output.joinpath('database.dump').read_bytes() + self.assertEqual(result['sha256'], hashlib.sha256(data).hexdigest()) + self.assertEqual(result['bytes'], len(data)) + + def test_client_nonzero_and_stderr_suppression(self): + process = FakeProcess(returncode=9) + with patch.object(snapshot.subprocess, 'Popen', return_value=process) as popen: + with self.assertRaises(snapshot.Failure): + with snapshot._client(['fixture.exe'], {}, 5): + pass + self.assertEqual(popen.call_args.kwargs['stderr'], subprocess.DEVNULL) + self.assertTrue(process.stdout.closed) + + def test_client_timeout_kills_process(self): + process = FakeProcess() + + class Timer: + def __init__(self, seconds, callback): + self.callback = callback + + def start(self): + self.callback() + + def cancel(self): + pass + + def join(self): + pass + + with patch.object(snapshot.threading, 'Timer', Timer), self.assertRaises(snapshot.Failure) as caught: + with snapshot._deadline(process, 1): + pass + self.assertEqual(caught.exception.code, 124) + self.assertTrue(process.killed) + + def test_timeout_is_preserved_when_killed_pipe_produces_an_error(self): + process = FakeProcess() + + class Timer: + def __init__(self, seconds, callback): + self.callback = callback + + def start(self): + self.callback() + + def cancel(self): + pass + + def join(self): + pass + + with patch.object(snapshot.threading, 'Timer', Timer), self.assertRaises(snapshot.Failure) as caught: + with snapshot._deadline(process, 1): + raise OSError(SECRET) + self.assertEqual(caught.exception.code, 124) + + def test_client_session_guard_refreshes_statistics(self): + with patch.object(snapshot, '_query', side_effect=[None, 1]) as query, self.assertRaises(snapshot.Failure): + snapshot._no_other_clients(FakeProcess()) + self.assertIn('pg_stat_clear_snapshot()', query.call_args_list[0].args[1]) + + + def test_sequence_state_accepts_uncalled_bigint_and_rejects_payloads(self): + value = {'last_value': -9223372036854775808, 'is_called': False} + with patch.object(snapshot, '_query', return_value=value): + self.assertEqual(snapshot._sequence_states(FakeProcess(), [['public', 'counter']]), + {'public': {'counter': value}}) + for value in (None, {'last_value': True, 'is_called': False}, + {'last_value': 3, 'is_called': 1}, {'last_value': 3}, + {'last_value': 3, 'is_called': False, 'payload': SECRET}): + with self.subTest(value=value), patch.object(snapshot, '_query', return_value=value), \ + self.assertRaises(snapshot.Failure): + snapshot._sequence_states(FakeProcess(), [['public', 'counter']]) + + def test_changed_sequence_state_invalidates_completed_dump(self): + source = self.runtime() + for name in ('psql.exe', 'pg_dump.exe'): + self.put('runtime/postgres/pgsql/bin/' + name) + self.output.mkdir() + metadata = self.metadata() + + @contextlib.contextmanager + def client(command, env, timeout, interactive=False): + yield FakeProcess() if interactive else FakeProcess(b'PGDMPsynthetic-dump') + + def query(process, sql, timeout=snapshot.QUERY_TIMEOUT): + return metadata if sql == snapshot.DATABASE_METADATA else 0 + + def version(command, **kwargs): + return SimpleNamespace(returncode=0, stdout=(Path(command[0]).stem + ' (PostgreSQL) 16.4\n').encode()) + + before = {'public': {'counter': {'last_value': 1, 'is_called': False}}} + for after in ({'public': {'counter': {'last_value': 2, 'is_called': False}}}, + {'public': {'counter': {'last_value': 1, 'is_called': True}}}): + with self.subTest(after=after), patch.object(snapshot, '_client', client), \ + patch.object(snapshot, '_query', query), \ + patch.object(snapshot.subprocess, 'run', side_effect=version), \ + patch.object(snapshot, '_sequence_states', side_effect=[before, after]), \ + self.assertRaises(snapshot.Failure): + snapshot._capture_database(source, self.identity, self.output, snapshot._inventory(), self.report) + self.assertTrue(self.output.joinpath('database.dump').exists()) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + self.output.joinpath('database.dump').unlink() + + +class SignalTests(Fixture): + def test_handlers_only_set_flag_and_restore_previous_handlers(self): + for number in self.handlers: + with self.subTest(number=number): + with snapshot._defer_signals() as checkpoint: + checkpoint() + self.send_signal(number) + self.send_signal(number) + with self.assertRaises(snapshot.Failure): + checkpoint() + self.assertEqual(self.handlers, self.previous_handlers) + + def test_partial_handler_installation_failure_never_loads_source(self): + if len(self.handlers) < 2: + self.skipTest('SIGBREAK is Windows-only') + register = self.signal.side_effect + last = list(self.handlers)[-1] + + def fail(number, handler): + if number == last and handler is not self.previous_handlers[last]: + raise OSError(SECRET) + return register(number, handler) + + self.signal.side_effect = fail + with self.assertRaises(OSError): + snapshot.capture(self.output, self.report) + snapshot._load_source.assert_not_called() + self.assertEqual(self.handlers, self.previous_handlers) + + def test_cancel_before_start_neither_starts_nor_stops_source(self): + def report(number, count, size): + if number == 5: + self.send_signal(snapshot.signal.SIGINT) + + with self.capture_context() as source, self.assertRaises(snapshot.Failure): + snapshot.capture(self.output, report) + source.pg.maintenance_start.assert_not_called() + source.pg.maintenance_stop.assert_not_called() + self.assertFalse(self.authority.acquired) + + def test_spawn_window_signal_cannot_unwind_unpublished_child(self): + for number in self.handlers: + for outcome in ('READY', 'OWNED_START_UNCERTAIN'): + with self.subTest(number=number, outcome=outcome), self.capture_context() as source: + self.output = self.imports / (str(int(number)) + '-' + outcome) + backend = self.backend + backend._start_requested_wall_time = None + backend._accepted_start_at_monotonic = None + backend._started_postmaster_observed = False + backend._expected_process = None + child = SimpleNamespace(alive=False, visible=False) + events, window_probes, compensation_held, released_live, closed_states = [], [], [], [], [] + + def probe(): + if child.alive and child.visible: + backend._accepted_start_at_monotonic = None + backend._started_postmaster_observed = True + backend._expected_process = child + return SimpleNamespace(kind='READY') + if backend._accepted_start_at_monotonic is not None: + return SimpleNamespace(kind='OWNED_START_UNCERTAIN') + # The real helper has no PID/listener before publication; + # without its accepted-start latch, it reports STOPPED. + return SimpleNamespace(kind='STOPPED') + + def close(): + closed_states.append((backend._accepted_start_at_monotonic, + backend._started_postmaster_observed)) + backend._expected_process = None + + def start(config, backend): + try: + backend._start_requested_wall_time = 1.0 + child.alive = True + events.append('spawned') + window_probes.extend([backend.probe().kind, backend.probe().kind]) + self.send_signal(number) # Popen -> 100ms sleep/poll gap. + events.append('signal-returned') + events.append('polled-still-running') + backend._accepted_start_at_monotonic = 10.0 + events.append('accepted') + child.visible = outcome == 'READY' + result = backend.probe() + if result.kind != 'READY': + raise RuntimeError(SECRET) + return result + finally: + # maintenance_start's finally closes the process + # handle, not the accepted-start/observed latches. + backend.close() + + def stop(config, backend): + try: + if backend.probe().kind == 'STOPPED': + return SimpleNamespace(completed=True, stopped=True) + if backend._accepted_start_at_monotonic is not None and backend._expected_process is None: + compensation_held.append(self.authority.acquired and child.alive and not child.visible) + child.visible = True + backend.probe() + events.append('identity-stop' if backend._expected_process is child else 'unsafe-stop') + child.alive = False + backend._accepted_start_at_monotonic = None + backend._start_requested_wall_time = None + backend._started_postmaster_observed = False + return SimpleNamespace(completed=True, stopped=True) + finally: + backend.close() + + release = self.authority.release.side_effect + + def release_checked(): + released_live.append(child.alive) + events.append('release') + release() + + backend.probe.side_effect = probe + backend.close.side_effect = close + source.pg.maintenance_start.side_effect = start + source.pg.maintenance_stop.side_effect = stop + self.authority.release.side_effect = release_checked + caught = None + try: + snapshot.capture(self.output, self.report) + except BaseException as exc: + caught = exc + self.assertIsInstance(caught, snapshot.Failure) + self.assertEqual(window_probes, ['STOPPED', 'STOPPED']) + self.assertEqual(events, ['spawned', 'signal-returned', 'polled-still-running', + 'accepted', 'identity-stop', 'release']) + self.assertEqual(released_live, [False]) + self.assertFalse(child.alive) + self.assertNotIn('database', self.calls) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + self.assertEqual(self.handlers, self.previous_handlers) + if outcome == 'OWNED_START_UNCERTAIN': + self.assertEqual(closed_states[0], (10.0, False)) + self.assertEqual(compensation_held, [True]) + else: + self.assertEqual(closed_states[0], (None, True)) + + def test_signal_during_stop_cannot_interrupt_confirmation_retries(self): + with self.capture_context() as source: + normal = source.pg.maintenance_stop.side_effect + + def stop(config, backend): + self.assertTrue(self.authority.acquired) + if source.pg.maintenance_stop.call_count == 1: + self.send_signal(snapshot.signal.SIGINT) + return SimpleNamespace(completed=False, stopped=False) + return normal(config, backend) + + source.pg.maintenance_stop.side_effect = stop + with patch.object(snapshot.time, 'sleep') as sleep, \ + patch.object(snapshot, '_write_tar') as tar, self.assertRaises(snapshot.Failure): + snapshot.capture(self.output, self.report) + sleep.assert_called_once_with(2) + tar.assert_not_called() + self.assertEqual(source.pg.maintenance_stop.call_count, 3) + self.assertEqual(self.backend.state, 'STOPPED') + self.assertFalse(self.authority.acquired) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + + def test_signal_during_release_is_deferred_until_release_finishes(self): + with self.capture_context(): + release = self.authority.release.side_effect + + def interrupted_release(): + self.send_signal(snapshot.signal.SIGINT) + release() + + self.authority.release.side_effect = interrupted_release + with self.assertRaises(snapshot.Failure): + snapshot.capture(self.output, self.report) + self.assertFalse(self.authority.acquired) + self.assertEqual(self.backend.state, 'STOPPED') + self.assertTrue(self.output.joinpath('files.tar').exists()) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + + +class CaptureTests(Fixture): + def test_success_stops_before_tar_holds_authority_until_manifest(self): + self.put('runtime/results/segment.jsonl', b'fixture-row\n') + original_tar, original_publish = snapshot._write_tar, snapshot._publish_manifest + + def tar(*args): + self.assertTrue(self.authority.acquired) + self.assertEqual(self.backend.state, 'STOPPED') + self.calls.append('tar') + return original_tar(*args) + + def publish(*args): + self.assertTrue(self.authority.acquired) + self.assertEqual(self.backend.state, 'STOPPED') + self.calls.append('manifest') + return original_publish(*args) + + with self.capture_context(), patch.object(snapshot, '_write_tar', side_effect=tar), \ + patch.object(snapshot, '_publish_manifest', side_effect=publish): + snapshot.capture(self.output, self.report) + manifest = json.loads(self.output.joinpath('manifest.json').read_text()) + self.assertEqual(manifest['format'], 'truf-windows-snapshot-v1') + self.assertTrue(manifest['source']['supervisor_stopped']) + self.assertTrue(manifest['source']['postgres_stopped']) + self.assertEqual(manifest['counts']['files'], 1) + self.assertEqual(self.calls, ['acquire', 'start', 'database', 'stop', 'tar', 'stop', 'manifest', 'release']) + self.assertFalse(self.authority.acquired) + + def test_supervisor_metadata_refuses_start_without_stopping_existing_source(self): + for relative in ('runtime/control/supervisor.instance.json', 'runtime/logs/supervisor.instance.json', + 'runtime/logs/supervisor.pid'): + path = self.put(relative) + self.output = self.imports / ('case-' + str(len(list(self.imports.iterdir())))) + with self.capture_context() as source, self.assertRaises(snapshot.Failure): + snapshot.capture(self.output, self.report) + source.pg.maintenance_start.assert_not_called() + source.pg.maintenance_stop.assert_not_called() + self.assertFalse(self.output.joinpath('manifest.json').exists()) + self.assertFalse(self.authority.acquired) + path.unlink() + + def test_authenticated_stopped_guard_rejects_all_other_states(self): + for state in ('READY', 'RECOVERING', 'FOREIGN_OR_CONFIG_ERROR', 'OWNED_START_UNCERTAIN'): + self.output = self.imports / state + with self.capture_context() as source, self.assertRaises(snapshot.Failure): + self.backend.state = state + snapshot.capture(self.output, self.report) + source.pg.maintenance_start.assert_not_called() + source.pg.maintenance_stop.assert_not_called() + self.assertFalse(self.output.joinpath('manifest.json').exists()) + + def test_stopped_guard_is_rechecked_after_inventory_before_start(self): + with self.capture_context() as source, self.assertRaises(snapshot.Failure): + self.backend.probe.side_effect = [SimpleNamespace(kind='STOPPED'), SimpleNamespace(kind='READY')] + snapshot.capture(self.output, self.report) + source.pg.maintenance_start.assert_not_called() + source.pg.maintenance_stop.assert_not_called() + self.assertFalse(self.authority.acquired) + + def test_partial_maintenance_start_still_gets_confirmed_stop(self): + with self.capture_context() as source: + source.pg.maintenance_start.side_effect = RuntimeError(SECRET) + with self.assertRaises(RuntimeError): + snapshot.capture(self.output, self.report) + self.assertGreaterEqual(source.pg.maintenance_stop.call_count, 1) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + self.assertFalse(self.authority.acquired) + + def test_failed_database_keeps_partial_output_without_manifest(self): + def failure(*args): + self.output.joinpath('database.dump').write_bytes(b'partial') + raise RuntimeError(SECRET) + + with self.capture_context(database=failure), self.assertRaises(RuntimeError): + snapshot.capture(self.output, self.report) + self.assertTrue(self.output.joinpath('database.dump').exists()) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + self.assertEqual(self.backend.state, 'STOPPED') + self.assertEqual(self.calls[-1], 'release') + + def test_failed_tar_keeps_dump_without_manifest(self): + with self.capture_context(), patch.object(snapshot, '_write_tar', side_effect=RuntimeError(SECRET)), \ + self.assertRaises(RuntimeError): + snapshot.capture(self.output, self.report) + self.assertTrue(self.output.joinpath('database.dump').exists()) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + self.assertEqual(self.backend.state, 'STOPPED') + + def test_changed_inventory_prevents_publication(self): + original = snapshot._write_tar + + def tar(*args): + result = original(*args) + self.put('runtime/state/new.json') + return result + + with self.capture_context(), patch.object(snapshot, '_write_tar', side_effect=tar), \ + self.assertRaises(snapshot.Failure): + snapshot.capture(self.output, self.report) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + + def test_source_change_during_final_stop_is_not_published(self): + with self.capture_context() as source: + normal = source.pg.maintenance_stop.side_effect + + def stop(*args, **kwargs): + result = normal(*args, **kwargs) + if source.pg.maintenance_stop.call_count == 2: + self.put('runtime/state/late-change') + return result + + source.pg.maintenance_stop.side_effect = stop + with self.assertRaises(snapshot.Failure): + snapshot.capture(self.output, self.report) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + + def test_post_publication_cleanup_failure_revokes_manifest(self): + with self.capture_context(), self.assertRaises(OSError): + self.backend.close.side_effect = OSError(SECRET) + snapshot.capture(self.output, self.report) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + self.assertTrue(self.output.joinpath('files.tar').exists()) + self.assertFalse(self.authority.acquired) + + def test_stop_retries_retain_authority_and_no_manifest(self): + with self.capture_context() as source: + normal_stop = source.pg.maintenance_stop.side_effect + tries = [] + + def stop(config, backend): + self.assertTrue(self.authority.acquired) + self.assertFalse(self.output.joinpath('manifest.json').exists()) + tries.append(1) + if len(tries) == 1: + raise KeyboardInterrupt(SECRET) + return normal_stop(config, backend) + + source.pg.maintenance_stop.side_effect = stop + with patch.object(snapshot.time, 'sleep') as sleep: + snapshot.capture(self.output, self.report) + sleep.assert_called_once_with(2) + self.assertTrue(self.output.joinpath('manifest.json').exists()) + self.assertFalse(self.authority.acquired) + + def test_changed_identity_input_prevents_start(self): + path = self.put('app/config.yaml') + with self.capture_context() as source, self.assertRaises(snapshot.Failure): + source.inputs[path] = snapshot._fingerprint(path.stat()) + path.write_bytes(b'changed-fixture') + snapshot.capture(self.output, self.report) + source.pg.maintenance_start.assert_not_called() + self.assertFalse(self.output.joinpath('manifest.json').exists()) + + def test_main_redacts_raw_source_failure_and_restores_environment(self): + output, error = io.StringIO(), io.StringIO() + + def failure(*args): + print(SECRET) + os.environ['SYNTHETIC_SNAPSHOT_TEST'] = SECRET + raise RuntimeError(SECRET) + + with patch.object(snapshot.os, 'name', 'nt'), patch.object(snapshot, '_output_path', return_value=self.output), \ + patch.object(snapshot, '_load_source', side_effect=failure), \ + contextlib.redirect_stdout(output), contextlib.redirect_stderr(error): + result = snapshot.main(['capture', '--output', 'fixture']) + self.assertEqual(result, 1) + self.assertNotIn(SECRET, output.getvalue() + error.getvalue()) + self.assertRegex(error.getvalue(), r'^2 1\n$') + self.assertRegex(output.getvalue(), r'^2 0 0\n$') + self.assertNotIn('SYNTHETIC_SNAPSHOT_TEST', os.environ) + + def test_non_windows_refused_before_source_loading(self): + with patch.object(snapshot.os, 'name', 'posix'), contextlib.redirect_stderr(io.StringIO()): + self.assertEqual(snapshot.main(['capture', '--output', 'fixture']), 1) + snapshot._load_source.assert_not_called() + + +if __name__ == '__main__': + unittest.main() diff --git a/docker/verify.py b/docker/verify.py new file mode 100644 index 0000000..f890db4 --- /dev/null +++ b/docker/verify.py @@ -0,0 +1,909 @@ +#!/usr/bin/env python3 +"""Run the offline worker E2E against dedicated, already-built local images. + +Run from Linux/WSL: python3 docker/verify.py [--keep] +Requires Docker with Compose v2 (directly or via sudo -n docker), Linux named +volumes, and Git. docker/test-results/latest.json must already be gitignored; +this script never changes ignore files. No image builds, pulls, host data +mounts, application edits, production Compose files, or broad cleanup. + +Checks have a 3600-second aggregate budget; failure shutdown has a separate +720-second budget. Each health wait is at most 240 seconds, and each stop uses +600 seconds of grace. A forced/nonzero/OOM exit never counts as success. +Failures retain owned Docker artifacts after a guarded stop attempt. --keep +also retains them on success. Evidence contains only counts, hashes, image +IDs, fixed statuses, and durations; command logs are never printed or saved. + +Contract limits: prepare is not repaired or rerun after recreation. A stopped +container loses its /run/truf tmpfs, so its shutdown receipt cannot be read +afterward. Receipt-write success is inferred ONLY from the healthy foreground +runtime's zero-exit contract; private PostgreSQL PID-file absence is separately +checked using the same runtime image and a read-only data-volume mount. +""" + +import argparse +import hashlib +import json +import os +from pathlib import Path +import re +import secrets +import selectors +import shutil +import signal +import stat +import subprocess +import sys +import tempfile +import time + + +PYTHON = ['/usr/local/bin/python3', '-I', '-S', '-B'] +APP = '/opt/truf/app/container_runtime.py' +DRIVER = '/opt/truf/tests/container_e2e.py' +CONFIG = '/data/config/e2e.yaml' +ENTRYPOINT = ['/usr/bin/tini', '--', '/usr/local/bin/python3', '-u', '-I', '-S', '-B', APP] +HEALTH = ['CMD', *PYTHON, APP, 'health', '--config', CONFIG] +SERVICES = {'tools', 'provision', 'prepare', 'runtime', 'stopped'} +IMAGE_REFERENCES = { + 'runtime': 'truf-worker-test:runtime', + 'test': 'truf-worker-test:test', +} +TMPFS = { + '/run/truf': 'rw,nosuid,nodev,noexec,size=64m,mode=0700,uid=10001,gid=10001', + '/tmp': 'rw,nosuid,nodev,noexec,size=128m,mode=1777', +} +PROXY_ENV = { + name: '*' if name.lower() == 'no_proxy' else '' + for stem in ('http_proxy', 'https_proxy', 'ftp_proxy', 'all_proxy', 'no_proxy') + for name in (stem, stem.upper()) +} +MOUNTS = { + 'tools': {'/opt/truf/tests': ('tools', False)}, + 'provision': {'/data': ('data', False)}, + 'prepare': {'/data': ('data', False), '/opt/truf/tests': ('tools', True)}, + 'runtime': {'/data': ('data', False), '/opt/truf/tests': ('tools', True)}, + 'stopped': {'/data': ('data', True)}, +} +LABEL = 'com.docker.compose.' +MAX_OUTPUT = 128 * 1024 +MAX_DOCKER_RESOURCES = 1024 +MAX_SNAPSHOT_BYTES = 4 * 1024 * 1024 +HEX = re.compile(r'[a-f0-9]{64}') + +CONTAINER_METADATA_FORMAT = ( + '{"id":{{json .Id}},"name":{{json .Name}},"image":{{json .Image}},' + '"status":{{json .State.Status}},"running":{{json .State.Running}},' + '"paused":{{json .State.Paused}},"restarting":{{json .State.Restarting}},' + '"dead":{{json .State.Dead}},"mounts":{{json .Mounts}}}' +) +VOLUME_METADATA_FORMAT = ( + '{"name":{{json .Name}},"driver":{{json .Driver}},"scope":{{json .Scope}},' + '"created":{{json .CreatedAt}},"mountpoint":{{json .Mountpoint}},' + '"labels":{{json .Labels}},"options":{{json .Options}}}' +) + +INSPECT_FIELDS = { + 'id': '.Id', 'name': '.Name', 'image': '.Image', 'status': '.State.Status', + 'running': '.State.Running', 'pid': '.State.Pid', + 'exit_code': '.State.ExitCode', 'oom_killed': '.State.OOMKilled', + 'restarts': '.RestartCount', 'user': '.Config.User', + 'entrypoint': '.Config.Entrypoint', 'command': '.Config.Cmd', + 'stop_timeout': '.Config.StopTimeout', + 'stop_signal': '.Config.StopSignal', 'mounts': '.Mounts', + 'readonly': '.HostConfig.ReadonlyRootfs', 'network': '.HostConfig.NetworkMode', + 'cap_drop': '.HostConfig.CapDrop', 'cap_add': '.HostConfig.CapAdd', + 'security_opt': '.HostConfig.SecurityOpt', 'init': '.HostConfig.Init', + 'privileged': '.HostConfig.Privileged', 'pid_mode': '.HostConfig.PidMode', + 'ports': '.HostConfig.PortBindings', 'tmpfs': '.HostConfig.Tmpfs', + 'cpus': '.HostConfig.NanoCpus', 'memory': '.HostConfig.Memory', + 'pids_limit': '.HostConfig.PidsLimit', 'restart_policy': '.HostConfig.RestartPolicy', + **{key: '(index .Config.Labels "' + LABEL + suffix + '")' for key, suffix in ( + ('project', 'project'), ('service', 'service'), ('oneoff', 'oneoff'), + ('config_files', 'project.config_files'), ('working_dir', 'project.working_dir'), + )}, +} +# Do not inspect .State or .Config wholesale: health logs and environments can +# contain credentials. Only these selected fields enter the host process. +INSPECT_FORMAT = '{' + ','.join( + json.dumps(key) + ':{{json ' + value + '}}' for key, value in INSPECT_FIELDS.items() +) + (',"health_test":{{with index .Config "Healthcheck"}}{{json .Test}}{{else}}null{{end}}' + ',"health":{{with index .State "Health"}}{{json .Status}}{{else}}null{{end}}}') + + +class Failure(Exception): + """Only fixed check labels, never subprocess output or exception messages.""" + + +def require(condition, label): + if not condition: + raise Failure(label) + + +class Verifier: + def __init__(self): + self.script = Path(__file__).absolute() + self.root = self.script.parent.parent.resolve(strict=True) + self.file = self.root / 'compose.e2e.yaml' + self.project = 'truf-worker-test-' + secrets.token_hex(16) + self.started = time.monotonic() + self.deadline = self.started + 3600 + self.docker = [] + self.compose = [] + self.images = {} + self.owned_containers = {} + self.owned_volumes = {} + self.foreign_baseline = None + self.mutated = False + self.evidence_ready = False + self.file_hashes = {} + self.report = { + 'status': {'result': 'running', 'cleanup': 'not_started'}, + 'counts': {'schema': 1}, + 'hashes': {'project_sha256': hashlib.sha256(self.project.encode('ascii')).hexdigest()}, + 'image_ids': self.images, 'durations': {}, + } + allowed_env = ( + 'PATH', 'HOME', 'XDG_CONFIG_HOME', 'XDG_RUNTIME_DIR', 'SSH_AUTH_SOCK', + 'DOCKER_HOST', 'DOCKER_CONTEXT', 'DOCKER_CONFIG', 'DOCKER_TLS_VERIFY', + 'DOCKER_CERT_PATH', + ) + self.env = {name: os.environ[name] for name in allowed_env if name in os.environ} + self.env['COMPOSE_DISABLE_ENV_FILE'] = '1' + + def execute(self, label, args, *, timeout=30, capture=False, check=True): + started = time.monotonic() + end = min(self.deadline, started + timeout) + require(end > started, 'aggregate_timeout') + process = None + output = bytearray() + selector = selectors.DefaultSelector() + counts = self.report['counts'].setdefault(label, {}) + counts['cli_calls'] = counts.get('cli_calls', 0) + 1 + self.report['status'][label] = 'running' + try: + process = subprocess.Popen( + args, cwd=self.root, env=self.env, stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE if capture else subprocess.DEVNULL, + stderr=subprocess.DEVNULL, start_new_session=True, + ) + if capture: + selector.register(process.stdout, selectors.EVENT_READ) + while process.poll() is None or selector.get_map(): + remaining = end - time.monotonic() + require(remaining > 0, label + '_timeout') + if selector.get_map(): + for key, _ in selector.select(min(0.1, remaining)): + chunk = os.read(key.fd, 65536) + if not chunk: + selector.unregister(key.fileobj) + else: + output.extend(chunk) + require(len(output) <= MAX_OUTPUT, label + '_output_limit') + else: + time.sleep(min(0.05, remaining)) + code = process.returncode + counts['exit_code' if code >= 0 else 'signal'] = abs(code) + self.report['status'][label] = 'passed' if code == 0 else 'failed' + if check and code != 0: + error = Failure(label + '_command_failed') + error.exit_code = code + raise error + return code, bytes(output) + except OSError: + self.report['status'][label] = 'failed' + raise Failure(label + '_unavailable') from None + except BaseException: + self.report['status'][label] = 'failed' + raise + finally: + selector.close() + if process is not None: + if process.poll() is None: + # Terminate only the local CLI process group, not containers. + # An interrupted Docker API operation can outlive its client; + # failure handling discovers and stops owned containers. + try: + os.killpg(process.pid, signal.SIGKILL) + except (ProcessLookupError, PermissionError): + pass + try: + process.wait(timeout=2) + except subprocess.TimeoutExpired: + pass + if process.stdout is not None: + process.stdout.close() + durations = self.report['durations'] + durations[label] = round(durations.get(label, 0) + time.monotonic() - started, 3) + + def json_command(self, label, args, timeout=30): + _, output = self.execute(label, args, timeout=timeout, capture=True) + try: + return json.loads(output) + except (ValueError, UnicodeError): + raise Failure(label + '_invalid_json') from None + + def words(self, label, args): + _, output = self.execute(label, [*self.docker, *args], capture=True) + try: + words = output.decode('ascii').split() + except UnicodeError: + raise Failure(label + '_invalid_inventory') from None + require(len(words) <= 16, label + '_inventory_limit') + return set(words) + + def metadata_names(self, kind): + options = (['--all', '--no-trunc', '--format', '{{.ID}}'] + if kind == 'container' else ['--format', '{{.Name}}']) + _, output = self.execute( + 'foreign_' + kind + '_list', [*self.docker, kind, 'ls', *options], capture=True, + ) + try: + values = {line for line in output.decode('ascii').splitlines() if line} + except UnicodeError: + raise Failure('foreign_' + kind + '_inventory_encoding') from None + require(len(values) <= MAX_DOCKER_RESOURCES, 'foreign_' + kind + '_inventory_bound') + return values + + def container_metadata(self, identifier): + require(HEX.fullmatch(identifier), 'foreign_container_id_guard') + value = self.json_command('foreign_container_inspect', [ + *self.docker, 'container', 'inspect', '--format', CONTAINER_METADATA_FORMAT, + identifier, + ]) + require( + value.get('id') == identifier and isinstance(value.get('name'), str) + and HEX.fullmatch(str(value.get('image') or '').removeprefix('sha256:')) + and value.get('status') in ( + 'created', 'running', 'paused', 'restarting', 'removing', 'exited', 'dead', + ) + and all(type(value.get(key)) is bool for key in ( + 'running', 'paused', 'restarting', 'dead', + )) + and isinstance(value.get('mounts'), list) and len(value['mounts']) <= 128, + 'foreign_container_metadata_guard', + ) + mounts = [{ + key: mount.get(key) for key in ( + 'Type', 'Name', 'Source', 'Destination', 'Driver', 'Mode', 'RW', 'Propagation', + ) + } for mount in value['mounts']] + return { + 'id': value['id'], 'name': value['name'], 'image': value['image'], + 'status': value['status'], 'running': value['running'], 'paused': value['paused'], + 'restarting': value['restarting'], 'dead': value['dead'], + 'mounts_sha256': hashlib.sha256(json.dumps( + mounts, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).hexdigest(), + } + + def volume_metadata(self, name): + value = self.json_command('foreign_volume_inspect', [ + *self.docker, 'volume', 'inspect', '--format', VOLUME_METADATA_FORMAT, name, + ]) + require( + value.get('name') == name and isinstance(value.get('driver'), str) + and isinstance(value.get('scope'), str) and isinstance(value.get('mountpoint'), str) + and (value.get('created') is None or isinstance(value['created'], str)) + and (value.get('labels') is None or isinstance(value['labels'], dict)) + and (value.get('options') is None or isinstance(value['options'], dict)), + 'foreign_volume_metadata_guard', + ) + return { + 'name': name, 'driver': value['driver'], 'scope': value['scope'], + 'created': value['created'], + 'mountpoint_sha256': hashlib.sha256(value['mountpoint'].encode('utf-8')).hexdigest(), + 'labels_sha256': hashlib.sha256(json.dumps( + value['labels'], ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).hexdigest(), + 'options_sha256': hashlib.sha256(json.dumps( + value['options'], ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).hexdigest(), + } + + def metadata_snapshot(self, *, exclude_owned=False): + containers = {} + for identifier in sorted(self.metadata_names('container')): + if exclude_owned and identifier in self.owned_containers: + self.inspect(identifier, 'foreign_owned_exclusion_guard') + continue + containers[identifier] = self.container_metadata(identifier) + volumes = {} + for name in sorted(self.metadata_names('volume')): + value = self.volume_metadata(name) + if exclude_owned and self.owned_volumes.get(name) == value: + continue + volumes[name] = value + snapshot = {'containers': containers, 'volumes': volumes} + require(len(json.dumps(snapshot, ensure_ascii=True, sort_keys=True)) <= MAX_SNAPSHOT_BYTES, + 'foreign_metadata_snapshot_bound') + return snapshot + + def snapshot_foreign(self): + require(self.foreign_baseline is None, 'foreign_snapshot_already_taken') + self.foreign_baseline = self.metadata_snapshot() + + def assert_foreign_unchanged(self): + require(self.foreign_baseline is not None, 'foreign_snapshot_missing') + require(self.metadata_snapshot(exclude_owned=True) == self.foreign_baseline, + 'foreign_docker_state_changed') + self.report['status']['foreign_docker_state'] = 'unchanged' + + def guard_files(self): + require(re.fullmatch(r'truf-worker-test-[a-f0-9]{32}', self.project), + 'worker_test_project_guard') + require(self.script == self.root / 'docker' / 'verify.py', 'verifier_path_guard') + require(self.file == self.root / 'compose.e2e.yaml', 'compose_path_guard') + for path in (self.script, self.file): + require(path.resolve(strict=True) == path and path.is_file() and not path.is_symlink(), + 'canonical_file_required') + require(path.stat().st_size <= 256 * 1024, 'source_file_size_bound') + digest = hashlib.sha256(path.read_bytes()).hexdigest() + require(path not in self.file_hashes or self.file_hashes[path] == digest, + 'source_changed_during_verification') + self.file_hashes[path] = digest + + def compose_command(self, label, args, **kwargs): + self.guard_files() + return self.execute(label, [*self.compose, *args], **kwargs) + + def inspect(self, container, label='inspect', timeout=30): + require(HEX.fullmatch(container), 'container_id_guard') + value = self.json_command(label, [ + *self.docker, 'container', 'inspect', '--format', INSPECT_FORMAT, container, + ], timeout=timeout) + require(value['id'] == container and value['project'] == self.project + and isinstance(value['name'], str) + and value['name'].startswith('/' + self.project + '-') + and value['service'] in SERVICES + and value['config_files'] == str(self.file) + and value['working_dir'] == str(self.root), 'container_ownership_guard') + service = value['service'] + require(value['oneoff'] == ('False' if service == 'runtime' else 'True'), + 'container_role_guard') + require(value['image'] == self.images['test' if service == 'tools' else 'runtime'], + 'container_image_guard') + require(value['status'] in ('created', 'running', 'paused', 'restarting', 'removing', 'exited', 'dead'), + 'container_state_guard') + require(all(type(value[key]) is int and value[key] >= 0 for key in ('exit_code', 'restarts', 'pid')) + and all(type(value[key]) is bool for key in ('oom_killed', 'running')), + 'container_state_types_guard') + identity = {key: value[key] for key in ( + 'id', 'name', 'image', 'project', 'service', 'config_files', 'working_dir', + )} + require(self.owned_containers.setdefault(container, identity) == identity, + 'container_identity_changed') + self.report['status']['container_' + service] = value['status'] + self.report['counts']['container_' + service] = { + 'exit_code': value['exit_code'], 'oom_killed': int(value['oom_killed']), + 'restarts': value['restarts'], 'running': int(value['running']), + } + return value + + def inventory(self, *, complete=False): + self.guard_files() + found = {} + for kind in ('container', 'volume', 'network'): + options = ['--all', '--quiet', '--no-trunc'] if kind == 'container' else ['--format', '{{.Name}}'] + named = self.words('inventory_' + kind, [ + kind, 'ls', *options, '--filter', 'name=' + self.project, + ]) + labeled = self.words('ownership_' + kind, [ + kind, 'ls', *options, '--filter', 'label=' + LABEL + 'project=' + self.project, + ]) + require(named == labeled, 'resource_ownership_guard') + found[kind] = named + require(not found['network'], 'unexpected_project_network') + expected = {self.project + '_data', self.project + '_tools'} + require(found['volume'] <= expected, 'volume_name_guard') + if complete: + require(found['volume'] == expected, 'required_volumes_missing') + for volume in sorted(found['volume']): + value = self.json_command('inspect_volume', [ + *self.docker, 'volume', 'inspect', '--format', + '{"name":{{json .Name}},"driver":{{json .Driver}},"options":{{json .Options}},' + '"scope":{{json .Scope}},"project":{{json (index .Labels "com.docker.compose.project")}},' + '"volume":{{json (index .Labels "com.docker.compose.volume")}}}', volume, + ]) + require(value['name'] == volume and value['project'] == self.project + and value['volume'] in ('data', 'tools') + and volume == self.project + '_' + value['volume'] + and value['driver'] == 'local' and not value['options'] + and value['scope'] == 'local', 'native_named_volume_guard') + metadata = self.volume_metadata(volume) + require(self.owned_volumes.setdefault(volume, metadata) == metadata, + 'volume_identity_changed') + users = self.words('volume_users', [ + 'container', 'ls', '--all', '--quiet', '--no-trunc', '--filter', 'volume=' + volume, + ]) + require(users <= found['container'], 'foreign_volume_user_guard') + self.report['counts']['owned_resources'] = { + kind + 's': len(names) for kind, names in found.items() + } + return [self.inspect(container) for container in sorted(found['container'])] + + def validate_compose(self, value): + require(value.get('name') == self.project, 'worker_test_project_guard') + require(set(value['services']) == SERVICES and set(value['volumes']) == {'data', 'tools'} + and not any(value.get(key) for key in ('networks', 'secrets', 'configs')), + 'compose_project_contract') + for name, volume in value['volumes'].items(): + require(volume.get('name') == self.project + '_' + name, + 'production_volume_forbidden') + require( + volume.get('name') not in {'truf-docker_data', 'truf-docker_tools'} + and volume.get('driver') == 'local' + and set(volume) <= {'name', 'driver'}, 'compose_volume_contract') + allowed = { + 'image', 'pull_policy', 'read_only', 'user', 'cap_drop', 'cap_add', 'security_opt', + 'network_mode', 'volumes', 'tmpfs', 'cpus', 'mem_limit', 'pids_limit', 'shm_size', + 'stop_signal', 'stop_grace_period', 'logging', 'restart', 'entrypoint', 'command', + 'healthcheck', 'environment', 'init', 'ports', + } + for name, service in value['services'].items(): + require(set(service) <= allowed, 'compose_service_options_guard') + require(service['image'] == self.images['test' if name == 'tools' else 'runtime'], + 'worker_test_image_reference_guard') + require(not service.get('ports'), 'published_port_contract') + require( + service.get('network_mode') == 'none', 'internal_network_contract') + require( + service['image'] == self.images['test' if name == 'tools' else 'runtime'] + and service.get('pull_policy') == 'never' + and service.get('read_only') is True + and service.get('environment') == PROXY_ENV + and service.get('init') is False + and service.get('user') == ('0:0' if name == 'provision' else '10001:10001') + and set(service.get('cap_drop', ())) == {'ALL'} + and set(service.get('cap_add', ())) == ( + {'CHOWN', 'DAC_OVERRIDE', 'FOWNER'} if name == 'provision' else set()) + and service.get('security_opt') == ['no-new-privileges:true'] + and service.get('restart') == 'no' + and float(service['cpus']) == 2 + and int(service['mem_limit']) == 6 * 1024 ** 3 + and int(service['pids_limit']) == 512 + and int(service['shm_size']) == 256 * 1024 ** 2 + and service['stop_signal'] == 'SIGTERM' + and service['logging'] == { + 'driver': 'json-file', 'options': {'max-size': '16m', 'max-file': '4'}, + } + and set(service['tmpfs']) == {key + ':' + val for key, val in TMPFS.items()}, + 'compose_isolation_contract') + mounts = {} + for mount in service['volumes']: + require(mount.get('type') == 'volume', 'bind_mount_forbidden') + require(set(mount) <= {'type', 'source', 'target', 'read_only', 'volume'} + and set(mount.get('volume', {})) <= {'nocopy'}, 'compose_mount_guard') + require(mount['target'] not in mounts, 'duplicate_mount_guard') + mounts[mount['target']] = (mount['source'], mount.get('read_only', False)) + if mount['source'] == 'tools': + require(mount.get('volume', {}).get('nocopy', False) is (name != 'tools'), + 'tools_copy_up_contract') + require(mounts == MOUNTS[name], 'compose_mount_contract') + runtime = value['services']['runtime'] + require(runtime.get('entrypoint') is None and runtime['command'] == ['run', '--config', CONFIG] + and runtime['healthcheck']['test'] == HEALTH, 'production_entrypoint_contract') + require(value['services']['provision'].get('entrypoint') is None + and value['services']['provision']['command'] == ['provision'] + and value['services']['prepare']['entrypoint'] == [*PYTHON, DRIVER] + and value['services']['prepare']['command'] == ['prepare', '--config', CONFIG], + 'prepare_command_contract') + + def preflight(self, env_file): + self.guard_files() + for path, digest in self.file_hashes.items(): + self.report['hashes']['compose_sha256' if path == self.file else 'verifier_sha256'] = digest + docker = shutil.which('docker') + require(docker, 'docker_cli_required') + candidates = [[docker]] + sudo = shutil.which('sudo') + if sudo: + candidates.append([sudo, '-n', docker]) + for index, candidate in enumerate(candidates): + try: + code, output = self.execute('docker_probe_' + str(index), [ + *candidate, 'info', '--format', '{{json .OSType}}', + ], timeout=15, capture=True, check=False) + except Failure: + continue + if code == 0: + require(json.loads(output) == 'linux', 'linux_docker_daemon_required') + self.docker = candidate + break + require(self.docker, 'docker_or_passwordless_sudo_required') + self.execute('compose_available', [*self.docker, 'compose', 'version', '--short']) + self.snapshot_foreign() + for name in ('runtime', 'test'): + value = self.json_command('image_' + name, [ + *self.docker, 'image', 'inspect', '--format', + '{"id":{{json .Id}},"os":{{json .Os}},"user":{{json .Config.User}},' + '"entrypoint":{{json .Config.Entrypoint}},"volumes":{{json (index .Config "Volumes")}}}', + IMAGE_REFERENCES[name], + ]) + require(re.fullmatch(r'sha256:[a-f0-9]{64}', value['id']) + and value['os'] == 'linux' and value['user'] == '10001:10001' + and not value['volumes'], 'prebuilt_image_contract') + if name == 'runtime': + require(value['entrypoint'] == ENTRYPOINT, 'runtime_image_entrypoint_contract') + self.images[name] = value['id'] + require(self.images['runtime'] != self.images['test'], 'distinct_image_targets_required') + env_file.write('TRUF_WORKER_TEST_PROJECT=' + self.project + '\n' + 'TRUF_WORKER_TEST_RUNTIME_IMAGE=' + self.images['runtime'] + '\n' + 'TRUF_WORKER_TEST_TEST_IMAGE=' + self.images['test'] + '\n') + env_file.flush() + # An explicit env file works with sudo's environment reset, too. Never + # load the checkout's .env or accept COMPOSE_FILE/PROJECT_NAME overrides. + self.compose = [ + *self.docker, 'compose', '--ansi', 'never', '--project-name', self.project, + '--project-directory', str(self.root), '--env-file', env_file.name, + '--file', str(self.file), + ] + self.validate_compose(self.json_command('compose_contract', [ + *self.compose, 'config', '--format', 'json', + ])) + git = shutil.which('git') + require(git, 'git_required_for_evidence_ignore_guard') + code, _ = self.execute('evidence_ignore_guard', [ + git, 'check-ignore', '--quiet', '--no-index', '--', 'docker/test-results/latest.json', + ], check=False) + require(code == 0, 'evidence_path_must_be_gitignored') + require(not self.inventory(), 'fresh_project_required') + require(not self.words('fresh_volumes', [ + 'volume', 'ls', '--format', '{{.Name}}', '--filter', 'name=' + self.project, + ]), 'fresh_volumes_required') + directory = self.root / 'docker' / 'test-results' + require(not directory.is_symlink(), 'evidence_directory_guard') + directory.mkdir(mode=0o700, exist_ok=True) + require(directory.resolve(strict=True) == directory and directory.is_dir(), + 'evidence_directory_guard') + self.evidence_ready = True + + def summary(self, label, args, timeout): + print(label + ': running', flush=True) + code, output = self.execute(label, args, timeout=timeout, capture=True, check=False) + try: + value = json.loads(output) + except (ValueError, UnicodeError): + raise Failure(label + '_invalid_summary') from None + require(isinstance(value, dict) and set(value) == {'counts', 'hashes'} + and all(isinstance(value[key], dict) and len(value[key]) <= 128 for key in value), + label + '_invalid_summary') + require(all(re.fullmatch(r'[a-z][a-z0-9_:]{0,120}', key) + and type(count) is int and 0 <= count <= 2 ** 63 - 1 + for key, count in value['counts'].items()) + and all(re.fullmatch(r'[a-z][a-z0-9_:]{0,120}_sha256', key) + and isinstance(digest, str) and HEX.fullmatch(digest) + for key, digest in value['hashes'].items()), label + '_invalid_summary') + self.report['counts'][label].update(value['counts']) + self.report['hashes'][label] = value['hashes'] + require(code == 0 and value['counts'].get('ok') == 1, label + '_check_failed') + return value + + def oneoff(self, service, label=None, timeout=180): + self.guard_files() + args = [*self.compose, 'run', '--rm', '--no-deps', '--pull', 'never', '-T', service] + if service == 'provision': + print('provision: running', flush=True) + self.execute('provision', args, timeout=timeout) + return None + return self.summary(label or service, args, timeout) + + def start(self, label, previous=None): + if previous: + value = self.inspect(previous, label + '_previous_ownership') + require(value['status'] == 'exited' and not value['running'], + label + '_previous_not_stopped') + self.inspect(previous, label + '_previous_remove_guard') + self.execute(label + '_remove_previous', [ + *self.docker, 'container', 'rm', previous, + ]) + self.compose_command(label, [ + 'up', '--detach', '--no-deps', '--no-build', '--pull', 'never', + 'runtime', + ], timeout=120) + containers = self.inventory(complete=True) + runtimes = [value for value in containers if value['service'] == 'runtime'] + require(len(runtimes) == 1, 'single_runtime_required') + value = runtimes[0] + container = value['id'] + require(container != previous, 'new_runtime_container_required') + require(value['entrypoint'] == ENTRYPOINT and value['command'] == ['run', '--config', CONFIG] + and value['health_test'] == HEALTH and value['user'] == '10001:10001' + and value['readonly'] is True and value['network'] == 'none' + and set(value['cap_drop'] or ()) == {'ALL'} and not value['cap_add'] + and value['security_opt'] == ['no-new-privileges:true'] + and not value['init'] and not value['privileged'] and not value['pid_mode'] + and not value['ports'] and value['tmpfs'] == TMPFS + and value['cpus'] == 2_000_000_000 and value['memory'] == 6 * 1024 ** 3 + and value['pids_limit'] == 512 and value['stop_timeout'] == 600 + and value['stop_signal'] == 'SIGTERM' + and value['restart_policy'] == {'Name': 'no', 'MaximumRetryCount': 0}, + 'actual_runtime_isolation_contract') + mounts = {} + for mount in value['mounts']: + if mount['Type'] == 'tmpfs': + require(mount['Destination'] in TMPFS, 'unexpected_tmpfs') + continue + require(mount['Type'] == 'volume' and mount['Destination'] not in mounts, + 'actual_named_mount_required') + mounts[mount['Destination']] = (mount['Name'], not mount['RW']) + require(mounts == {target: (self.project + '_' + name, ro) + for target, (name, ro) in MOUNTS['runtime'].items()}, + 'actual_runtime_mount_contract') + self.report['hashes'][label + '_container_sha256'] = hashlib.sha256(container.encode('ascii')).hexdigest() + health_started = time.monotonic() + health_end = min(self.deadline, health_started + 240) + while time.monotonic() < health_end: + value = self.inspect(container, label + '_health', timeout=min(10, health_end - time.monotonic())) + require(value['running'] and value['status'] == 'running' + and not value['oom_killed'] and value['restarts'] == 0, + label + '_runtime_exited_or_restarted') + if value['health'] == 'healthy': + self.report['durations'][label + '_ready'] = round(time.monotonic() - health_started, 3) + return container + time.sleep(min(2, max(0, health_end - time.monotonic()))) + raise Failure(label + '_health_timeout') + + def driver(self, container, mode, label, timeout=240): + self.inspect(container) + return self.summary(label, [ + *self.docker, 'exec', '--user', '10001:10001', container, *PYTHON, + DRIVER, mode, '--config', CONFIG, '--timeout', '180', + ], timeout) + + def health(self, container, label): + self.inspect(container) + value = self.json_command(label, [ + *self.docker, 'exec', '--user', '10001:10001', container, *PYTHON, + APP, 'health', '--config', CONFIG, + ], timeout=30) + require(value == { + 'healthy': True, 'activation_state': 'ACTIVE', 'postgres': 'READY', + 'workers': ['gitlab', 'janitor', 'jsonl-projector', 'result-ingester'], + }, label + '_authenticated_health_failed') + self.report['counts'][label]['authenticated_health'] = 1 + + def stop(self, container, label): + containers = self.inventory(complete=True) + require(any(value['id'] == container and value['running'] for value in containers), + label + '_runtime_not_running') + print(label + ': stopping (600-second grace)', flush=True) + self.inspect(container, label + '_ownership_guard') + self.execute(label, [ + *self.docker, 'container', 'stop', '--time', '600', container, + ], timeout=660) + value = self.inspect(container, label + '_inspect') + self.report['counts'][label].update({ + 'container_exit_code': value['exit_code'], 'oom_killed': int(value['oom_killed']), + 'restarts': value['restarts'], + }) + require(value['status'] == 'exited' and not value['running'] and value['pid'] == 0 + and value['exit_code'] == 0 and value['oom_killed'] is False + and value['restarts'] == 0, label + '_unclean_exit') + self.oneoff('stopped', label + '_data') + # Foreground supervisor.main returns zero only after receipt publication. + # Do not claim to have read that receipt from a destroyed tmpfs. + self.report['status'][label + '_receipt'] = 'exit_contract_only_tmpfs_removed' + + def cleanup(self, keep): + containers = self.inventory(complete=True) + require(len(containers) == 1 and containers[0]['service'] == 'runtime' + and containers[0]['status'] == 'exited' and containers[0]['exit_code'] == 0 + and not containers[0]['oom_killed'], 'cleanup_stopped_runtime_guard') + if keep: + self.assert_foreign_unchanged() + self.report['status']['cleanup'] = 'kept' + return + self.report['status']['cleanup'] = 'running' + container = containers[0]['id'] + self.inspect(container, 'cleanup_container_remove_guard') + self.execute('remove_owned_container', [*self.docker, 'container', 'rm', container]) + for volume in sorted(self.owned_volumes): + require(self.volume_metadata(volume) == self.owned_volumes[volume], + 'cleanup_volume_identity_changed') + self.execute('remove_owned_volume', [*self.docker, 'volume', 'rm', volume]) + require(not self.inventory(), 'cleanup_containers_remaining') + require(not self.words('cleanup_volumes', [ + 'volume', 'ls', '--format', '{{.Name}}', '--filter', 'name=' + self.project, + ]), 'cleanup_volumes_remaining') + self.assert_foreign_unchanged() + self.report['status']['cleanup'] = 'removed' + + def failure_stop(self): + self.deadline = time.monotonic() + 720 + self.report['status']['cleanup'] = 'retained_after_failure' + try: + containers = self.inventory() + active = [value for value in containers + if value['running'] or value['status'] in ('restarting', 'paused')] + if active: + print('failure_stop: stopping owned containers (600-second grace)', flush=True) + for value in active: + container = value['id'] + self.inspect(container, 'failure_stop_ownership_guard') + self.execute('failure_stop', [ + *self.docker, 'container', 'stop', '--time', '600', container, + ], timeout=660) + remaining = self.inventory() + self.report['counts']['failure_retained'] = { + 'containers': len(remaining), 'running': sum(int(value['running']) for value in remaining), + } + self.report['status']['failure_stop'] = ( + 'incomplete' if any(value['running'] for value in remaining) else 'observed_stopped' + ) + except (Exception, KeyboardInterrupt): + # Loss of the daemon, changed files, or unproven ownership must never + # lead to an unguarded down/prune/kill attempt or a false success. + self.report['status']['failure_stop'] = 'failed_or_ownership_unproven' + + def write_evidence(self): + self.report['durations']['total'] = round(time.monotonic() - self.started, 3) + directory = self.root / 'docker' / 'test-results' + require(directory.resolve(strict=True) == directory, 'evidence_directory_guard') + descriptor = os.open(directory, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + temporary = '.' + self.project + '.json' + try: + try: + details = os.stat('latest.json', dir_fd=descriptor, follow_symlinks=False) + require(stat.S_ISREG(details.st_mode), 'evidence_file_guard') + except FileNotFoundError: + pass + output = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, dir_fd=descriptor) + with os.fdopen(output, 'w', encoding='ascii') as handle: + json.dump(self.report, handle, sort_keys=True, indent=2, allow_nan=False) + handle.write('\n') + handle.flush() + os.fsync(handle.fileno()) + os.replace(temporary, 'latest.json', src_dir_fd=descriptor, dst_dir_fd=descriptor) + os.fsync(descriptor) + finally: + try: + os.unlink(temporary, dir_fd=descriptor) + except FileNotFoundError: + pass + os.close(descriptor) + + +def main(argv=None): + class Parser(argparse.ArgumentParser): + def error(self, message): + raise Failure('invalid_arguments') + + parser = Parser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter, allow_abbrev=False) + parser.add_argument('--keep', action='store_true', help='retain stopped, owned Docker artifacts on success too') + args = parser.parse_args(argv) + require(sys.platform == 'linux', 'linux_host_required_use_wsl_python3') + verifier = Verifier() + print('E2E project: ' + verifier.project, flush=True) + + def interrupted(_signum, _frame): + raise KeyboardInterrupt + + previous = signal.signal(signal.SIGTERM, interrupted) + # This private file contains only the project nonce and image IDs. + with tempfile.NamedTemporaryFile(mode='w', encoding='ascii', prefix=verifier.project + '-', suffix='.env') as env_file: + try: + verifier.preflight(env_file) + verifier.write_evidence() + verifier.mutated = True + verifier.oneoff('tools') + verifier.oneoff('provision') + verifier.oneoff('prepare') + first = verifier.start('first_start') + verifier.driver(first, 'run-local-pipeline', 'local_pipeline', timeout=240) + verifier.driver(first, 'assert-pipeline', 'pipeline') + baseline = verifier.driver(first, 'keycheck-fixture', 'first_keycheck', timeout=510) + require(baseline['counts'].pop('http_requests', None) == 1, 'first_provider_request_count') + verifier.health(first, 'first_authenticated_health') + verifier.stop(first, 'first_stop') + second = verifier.start('second_start', previous=first) + persisted = verifier.driver(second, 'assert-persisted', 'persisted') + require(persisted == baseline, 'persisted_public_summary_changed') + checked = verifier.driver(second, 'keycheck-fixture', 'second_keycheck', timeout=510) + require(checked['counts'].pop('http_requests', None) == 0, 'repeated_provider_made_http_requests') + require(checked == baseline, 'repeated_provider_changed_public_summary') + verifier.health(second, 'second_authenticated_health') + verifier.driver(second, 'assert-remote-recovery', 'remote_recovery') + verifier.driver(second, 'prepare-remote-transport', 'remote_transport') + dockerhub_canary_prepare = verifier.driver( + second, 'prepare-dockerhub-canary', + 'dockerhub_canary_prepare', timeout=510, + ) + require( + dockerhub_canary_prepare['counts'].get('search_pages') == 1 + and dockerhub_canary_prepare['counts'].get('cohort_targets') == 4 + and dockerhub_canary_prepare['counts'].get('digest_resolutions') == 4 + and dockerhub_canary_prepare['counts'].get('worker_provider_failures') == 2 + and dockerhub_canary_prepare['counts'].get('accepted_uploads') == 3 + and dockerhub_canary_prepare['counts'].get('expired_assignments') == 1 + and dockerhub_canary_prepare['counts'].get('drain_cycles') == 2 + and dockerhub_canary_prepare['counts'].get('pending_targets') == 1, + 'dockerhub_canary_prepare_evidence', + ) + remote_full_prepare = verifier.driver( + second, 'prepare-remote-full-race', 'remote_full_prepare', timeout=510, + ) + require(remote_full_prepare['counts'].get('real_claims') == 2 + and remote_full_prepare['counts'].get('native_scans') == 3 + and remote_full_prepare['counts'].get('exact_expiries') == 1 + and remote_full_prepare['counts'].get('pending_restart') == 1, + 'remote_full_prepare_evidence') + verifier.stop(second, 'second_stop') + third = verifier.start('third_start', previous=second) + remote_full_finish = verifier.driver( + third, 'finish-remote-full-race', 'remote_full_finish', timeout=510, + ) + require(remote_full_finish['counts'].get('concurrent_uploads') == 2 + and remote_full_finish['counts'].get('authoritative_receipts') == 1 + and remote_full_finish['counts'].get('authoritative_scans') == 1 + and remote_full_finish['counts'].get('expired_losers') == 1 + and remote_full_finish['counts'].get('completed_winners') == 1 + and remote_full_finish['counts'].get('cached_keychecks') == 1 + and remote_full_finish['counts'].get('provider_http_requests') == 0 + and remote_full_finish['counts'].get('projection_streams') == 3, + 'remote_full_finish_evidence') + verifier.driver( + third, 'assert-remote-transport-replay', 'remote_transport_replay', + ) + dockerhub_canary_finish = verifier.driver( + third, 'finish-dockerhub-canary', + 'dockerhub_canary_finish', timeout=510, + ) + require( + dockerhub_canary_finish['counts'].get('runtime_restarts') == 1 + and dockerhub_canary_finish['counts'].get('lost_claim_receipts') == 1 + and dockerhub_canary_finish['counts'].get('receipt_replays') == 1 + and dockerhub_canary_finish['counts'].get('conflicts_rejected') == 1 + and dockerhub_canary_finish['counts'].get('exactly_once') == 1 + and dockerhub_canary_finish['counts'].get('former_expiry_replays') == 1, + 'dockerhub_canary_finish_evidence', + ) + verifier.health(third, 'third_authenticated_health') + verifier.stop(third, 'third_stop') + verifier.report['status']['checks'] = 'passed' + # Persist the check results before deleting their Docker artifacts. + verifier.write_evidence() + verifier.cleanup(args.keep) + verifier.report['status']['result'] = 'passed' + except (Exception, KeyboardInterrupt) as exc: + verifier.report['status']['result'] = 'failed' + label = str(exc) if isinstance(exc, Failure) else ( + 'interrupted' if isinstance(exc, KeyboardInterrupt) else 'verifier_exception' + ) + failure_class = type(exc).__name__ + if not re.fullmatch(r'[a-z0-9_]{1,160}', label): + label = 'verifier_failure' + if not re.fullmatch(r'[A-Za-z][A-Za-z0-9_]{0,79}', failure_class): + failure_class = 'Exception' + exit_code = getattr(exc, 'exit_code', None) + if type(exit_code) is not int or not -(2 ** 31) <= exit_code < 2 ** 31: + exit_code = None + verifier.report['failure'] = { + 'stage': label, 'class': failure_class, + 'exit_code': exit_code, + } + print('E2E failed: ' + label, flush=True) + if verifier.mutated: + verifier.failure_stop() + finally: + signal.signal(signal.SIGTERM, previous) + if verifier.evidence_ready: + try: + verifier.write_evidence() + except (Exception, KeyboardInterrupt): + verifier.report['status']['result'] = 'failed' + print('E2E failed: evidence_write_failed', flush=True) + else: + print('Evidence: docker/test-results/latest.json', flush=True) + print('E2E ' + verifier.report['status']['result'] + '; project ' + verifier.project + + '; artifacts ' + verifier.report['status']['cleanup'], flush=True) + return 0 if verifier.report['status']['result'] == 'passed' else 1 + + +if __name__ == '__main__': + try: + raise SystemExit(main()) + except (Exception, KeyboardInterrupt) as exc: + print('E2E failed: ' + (str(exc) if isinstance(exc, Failure) else 'verifier_exception'), flush=True) + raise SystemExit(1) from None diff --git a/docker/verify_edge_e2e.py b/docker/verify_edge_e2e.py new file mode 100644 index 0000000..04918cd --- /dev/null +++ b/docker/verify_edge_e2e.py @@ -0,0 +1,1164 @@ +#!/usr/bin/env python3 +"""Run the standalone edge E2E with only random, labelled Docker resources.""" + +import argparse +import configparser +import hashlib +import ipaddress +import json +import os +from pathlib import Path +import re +import secrets +import shutil +import ssl +import stat +import subprocess +import sys +import time + + +TEST_IMAGE = 'truf-edge-e2e-runtime:test' +EDGE_IMAGE = 'truf-edge-e2e:test' +FAIL2BAN_IMAGE = 'truf-fail2ban-edge-e2e:test' +RUN_LABEL = 'com.truf.edge-e2e.run' +KIND_LABEL = 'com.truf.edge-e2e.kind' +MAX_OUTPUT = 2 * 1024 * 1024 +MAX_DOCKER_RESOURCES = 1024 +MAX_SNAPSHOT_BYTES = 8 * 1024 * 1024 +MAX_SAFE_EVIDENCE = 64 * 1024 +HEX_64 = re.compile(r'(?:sha256:)?[a-f0-9]{64}') +RESOURCE_NAME = re.compile(r'truf-edge-e2e-[a-f0-9]{16}-[a-z][a-z0-9-]{0,48}') + +CONTAINER_METADATA_FORMAT = ( + '{"id":{{json .Id}},"name":{{json .Name}},"image":{{json .Image}},' + '"status":{{json .State.Status}},"running":{{json .State.Running}},' + '"paused":{{json .State.Paused}},"restarting":{{json .State.Restarting}},' + '"dead":{{json .State.Dead}},"mounts":{{json .Mounts}}}' +) +VOLUME_METADATA_FORMAT = ( + '{"name":{{json .Name}},"driver":{{json .Driver}},"scope":{{json .Scope}},' + '"created":{{json .CreatedAt}},"mountpoint":{{json .Mountpoint}},' + '"labels":{{json .Labels}},"options":{{json .Options}}}' +) + + +def isolated_test_subnet(run_id, scope='edge'): + value = int.from_bytes( + hashlib.sha256(f'{run_id}\0{scope}'.encode('ascii')).digest()[:2], 'big', + ) & 0x1fff + return f'198.{18 + (value >> 12)}.{(value >> 4) & 0xff}.{(value & 0xf) * 16}/28' + + +class Failure(RuntimeError): + pass + + +def require(condition, label): + if not condition: + raise Failure(label) + + +def json_bytes(value, label): + require(len(value) <= MAX_OUTPUT, label + '_output_bound') + try: + result = json.loads(value.decode('utf-8', errors='strict')) + except (UnicodeError, ValueError): + raise Failure(label + '_json') from None + return result + + +def write_bytes(path, content): + path = Path(path) + path.parent.mkdir(parents=True, exist_ok=True) + with open(path, 'xb') as handle: + handle.write(content) + + +class CommandRunner: + def __init__(self, root, run_root, distro, deadline): + self.root = Path(root) + self.run_root = Path(run_root) + self.deadline = deadline + self.wsl = shutil.which('wsl.exe') or shutil.which('wsl') + require(self.wsl, 'wsl_unavailable') + self.distro = distro + self.docker_prefix = [self.wsl, '-d', distro, '--', 'sudo', '-n', 'docker'] + allowed = ( + 'PATH', 'PATHEXT', 'SystemRoot', 'SYSTEMROOT', 'WINDIR', 'COMSPEC', + 'TEMP', 'TMP', 'USERPROFILE', 'WSLENV', + ) + self.host_env = { + name: os.environ[name] for name in allowed if name in os.environ + } + + def execute(self, label, command, *, timeout=60, check=True): + remaining = self.deadline - time.monotonic() + require(remaining > 0, 'aggregate_timeout') + timeout = max(0.1, min(float(timeout), remaining)) + try: + completed = subprocess.run( + command, cwd=os.fspath(self.root), env=self.host_env, + stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, + stderr=subprocess.PIPE, timeout=timeout, check=False, + creationflags=getattr(subprocess, 'CREATE_NO_WINDOW', 0), + ) + except subprocess.TimeoutExpired: + raise Failure(label + '_timeout') from None + except OSError: + raise Failure(label + '_unavailable') from None + require( + len(completed.stdout) <= MAX_OUTPUT and len(completed.stderr) <= MAX_OUTPUT, + label + '_output_bound', + ) + if check: + if completed.returncode != 0: + error = Failure(label + '_failed') + error.exit_code = completed.returncode + raise error + return completed.returncode, completed.stdout, completed.stderr + + def docker(self, label, arguments, *, timeout=60, check=True): + return self.execute( + label, [*self.docker_prefix, *arguments], timeout=timeout, check=check, + ) + + def wsl_path(self, path): + _, stdout, _ = self.execute( + 'wsl_path', + [self.wsl, '-d', self.distro, '--', 'wslpath', '-a', '-u', + os.fspath(Path(path).resolve(strict=True)).replace('\\', '/')], + timeout=30, + ) + value = stdout.decode('utf-8', errors='strict').strip() + require(value.startswith('/') and '\x00' not in value, 'wsl_path_invalid') + return value + + +class Resources: + def __init__(self, runner, run_id): + self.runner = runner + self.run_id = run_id + self.created = {'containers': {}, 'volumes': {}, 'networks': {}} + self.foreign_baseline = None + + def labels(self, kind): + return [ + '--label', f'{RUN_LABEL}={self.run_id}', + '--label', f'{KIND_LABEL}={kind}', + ] + + @staticmethod + def singular(kind): + return {'containers': 'container', 'volumes': 'volume', 'networks': 'network'}[kind] + + def list_names(self, kind, *, name=None, owned=False): + singular = self.singular(kind) + arguments = [singular, 'ls'] + if kind == 'containers': + arguments.append('--all') + if name is not None: + pattern = '^/' + name + '$' if kind == 'containers' else '^' + name + '$' + arguments.extend(['--filter', 'name=' + pattern]) + if owned: + arguments.extend(['--filter', f'label={RUN_LABEL}={self.run_id}']) + field = 'Names' if kind == 'containers' else 'Name' + arguments.append('--format={{.' + field + '}}') + _, stdout, _ = self.runner.docker('list_' + kind, arguments) + names = {line for line in stdout.decode('ascii', errors='strict').splitlines() if line} + require(len(names) <= 16, 'resource_inventory_bound') + return names + + def guard(self, kind, name): + require(RESOURCE_NAME.fullmatch(name), 'resource_name_guard') + require(not self.list_names(kind, name=name), 'resource_name_collision') + + def inspect_owned(self, kind, name, expected=None): + singular = self.singular(kind) + if kind == 'containers': + template = ( + '{"id":{{json .Id}},"name":{{json .Name}},' + '"run":{{json (index .Config.Labels "' + RUN_LABEL + '")}},' + '"kind":{{json (index .Config.Labels "' + KIND_LABEL + '")}},' + '"running":{{json .State.Running}},"paused":{{json .State.Paused}},' + '"restarting":{{json .State.Restarting}}}' + ) + else: + template = ( + '{"id":{{json ' + ('.Id' if kind == 'networks' else '""') + '}},' + '"name":{{json .Name}},"run":{{json (index .Labels "' + RUN_LABEL + '")}},' + '"kind":{{json (index .Labels "' + KIND_LABEL + '")}}}' + ) + _, stdout, _ = self.runner.docker( + 'inspect_owned_' + singular, + [singular, 'inspect', '--format', template, name], + ) + value = json_bytes(stdout, 'inspect_owned_' + singular) + require( + value.get('name') == ('/' + name if kind == 'containers' else name) + and value.get('run') == self.run_id and isinstance(value.get('kind'), str) + and (kind == 'volumes' or HEX_64.fullmatch(str(value.get('id') or ''))), + 'owned_' + singular + '_identity_guard', + ) + if expected is not None: + require( + (expected.get('id') is None or value.get('id') == expected['id']) + and value['kind'] == expected['kind'], + 'owned_' + singular + '_identity_changed', + ) + return value + + def metadata_names(self, kind): + arguments = ([kind, 'ls', '--all', '--no-trunc', '--format={{.ID}}'] + if kind == 'container' else [kind, 'ls', '--format={{.Name}}']) + _, stdout, _ = self.runner.docker('foreign_' + kind + '_list', arguments) + try: + values = {line for line in stdout.decode('ascii').splitlines() if line} + except UnicodeError: + raise Failure('foreign_' + kind + '_inventory_encoding') from None + require(len(values) <= MAX_DOCKER_RESOURCES, 'foreign_' + kind + '_inventory_bound') + return values + + def container_metadata(self, identifier): + require(re.fullmatch(r'[a-f0-9]{64}', identifier), 'foreign_container_id_guard') + _, stdout, _ = self.runner.docker( + 'foreign_container_inspect', + ['container', 'inspect', '--format', CONTAINER_METADATA_FORMAT, identifier], + ) + value = json_bytes(stdout, 'foreign_container_inspect') + require( + value.get('id') == identifier and isinstance(value.get('name'), str) + and HEX_64.fullmatch(str(value.get('image') or '')) + and value.get('status') in ( + 'created', 'running', 'paused', 'restarting', 'removing', 'exited', 'dead', + ) + and all(type(value.get(key)) is bool for key in ( + 'running', 'paused', 'restarting', 'dead', + )) + and isinstance(value.get('mounts'), list) and len(value['mounts']) <= 128, + 'foreign_container_metadata_guard', + ) + mounts = [{ + key: mount.get(key) for key in ( + 'Type', 'Name', 'Source', 'Destination', 'Driver', 'Mode', 'RW', 'Propagation', + ) + } for mount in value['mounts']] + return { + 'id': value['id'], 'name': value['name'], 'image': value['image'], + 'status': value['status'], 'running': value['running'], 'paused': value['paused'], + 'restarting': value['restarting'], 'dead': value['dead'], + 'mounts_sha256': hashlib.sha256(json.dumps( + mounts, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')).hexdigest(), + } + + def volume_metadata(self, name): + _, stdout, _ = self.runner.docker( + 'foreign_volume_inspect', + ['volume', 'inspect', '--format', VOLUME_METADATA_FORMAT, name], + ) + value = json_bytes(stdout, 'foreign_volume_inspect') + require( + value.get('name') == name and isinstance(value.get('driver'), str) + and isinstance(value.get('scope'), str) and isinstance(value.get('mountpoint'), str) + and (value.get('created') is None or isinstance(value['created'], str)) + and (value.get('labels') is None or isinstance(value['labels'], dict)) + and (value.get('options') is None or isinstance(value['options'], dict)), + 'foreign_volume_metadata_guard', + ) + digest = lambda item: hashlib.sha256(item).hexdigest() + return { + 'name': name, 'driver': value['driver'], 'scope': value['scope'], + 'created': value['created'], + 'mountpoint_sha256': digest(value['mountpoint'].encode('utf-8')), + 'labels_sha256': digest(json.dumps( + value['labels'], ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')), + 'options_sha256': digest(json.dumps( + value['options'], ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')), + } + + def metadata_snapshot(self, *, exclude_owned=False): + containers = {} + owned_ids = { + value['id']: (name, value) + for name, value in self.created['containers'].items() if value.get('id') + } + for identifier in sorted(self.metadata_names('container')): + if exclude_owned and identifier in owned_ids: + name, expected = owned_ids[identifier] + self.inspect_owned('containers', name, expected) + continue + containers[identifier] = self.container_metadata(identifier) + volumes = {} + for name in sorted(self.metadata_names('volume')): + value = self.volume_metadata(name) + expected = self.created['volumes'].get(name) + if exclude_owned and expected is not None and expected.get('metadata') == value: + continue + volumes[name] = value + snapshot = {'containers': containers, 'volumes': volumes} + require(len(json.dumps(snapshot, ensure_ascii=True, sort_keys=True)) <= MAX_SNAPSHOT_BYTES, + 'foreign_metadata_snapshot_bound') + return snapshot + + def snapshot_foreign(self): + require(self.foreign_baseline is None, 'foreign_snapshot_already_taken') + self.foreign_baseline = self.metadata_snapshot() + + def assert_foreign_unchanged(self): + require(self.foreign_baseline is not None, 'foreign_snapshot_missing') + require(self.metadata_snapshot(exclude_owned=True) == self.foreign_baseline, + 'foreign_docker_state_changed') + + def create_volume(self, name, role): + self.guard('volumes', name) + self.created['volumes'][name] = {'id': None, 'kind': role} + _, stdout, _ = self.runner.docker( + 'create_' + role, + ['volume', 'create', *self.labels(role), name], + ) + require(stdout.decode('ascii', errors='strict').strip() == name, role + '_create') + self.inspect_owned('volumes', name, self.created['volumes'][name]) + self.created['volumes'][name]['metadata'] = self.volume_metadata(name) + + def create_network(self, name): + self.guard('networks', name) + self.created['networks'][name] = {'id': None, 'kind': 'internal-network'} + _, stdout, _ = self.runner.docker( + 'create_network', + ['network', 'create', '--driver', 'bridge', '--internal', + '--subnet', isolated_test_subnet(self.run_id), + *self.labels('internal-network'), name], + ) + identifier = stdout.decode('ascii', errors='strict').strip() + require(re.fullmatch(r'[a-f0-9]{64}', identifier), 'network_create') + self.created['networks'][name]['id'] = identifier + self.inspect_owned('networks', name, self.created['networks'][name]) + + def create_container(self, name, role, arguments): + self.guard('containers', name) + self.created['containers'][name] = {'id': None, 'kind': role} + _, stdout, _ = self.runner.docker( + 'create_' + role, + ['container', 'create', '--pull=never', '--name', name, + *self.labels(role), *arguments], + timeout=120, + ) + identifier = stdout.decode('ascii', errors='strict').strip() + require(re.fullmatch(r'[a-f0-9]{64}', identifier), role + '_container_id') + self.created['containers'][name]['id'] = identifier + self.inspect_owned('containers', name, self.created['containers'][name]) + + def inventory(self): + result = {} + for kind in self.created: + try: + result[kind] = sorted(self.list_names(kind, owned=True)) + except Exception: + result[kind] = sorted(self.created[kind]) + return result + + def guarded_stop(self): + stopped = 0 + for name, expected in reversed(self.created['containers'].items()): + try: + value = self.inspect_owned('containers', name, expected) + if expected['id'] is None: + expected['id'] = value['id'] + if value['running'] or value['paused'] or value['restarting']: + value = self.inspect_owned('containers', name, expected) + code, _, _ = self.runner.docker( + 'guarded_stop_owned_container', + ['container', 'stop', '--time', '30', value['id']], + timeout=45, check=False, + ) + stopped += int(code == 0) + except (Exception, KeyboardInterrupt): + continue + return stopped + + def cleanup(self): + inventory = { + kind: self.list_names(kind, owned=True) for kind in self.created + } + for kind, names in self.created.items(): + require(inventory[kind] == set(names), 'cleanup_ownership_guard') + for name, expected in reversed(self.created['containers'].items()): + value = self.inspect_owned('containers', name, expected) + if value['running'] or value['paused'] or value['restarting']: + value = self.inspect_owned('containers', name, expected) + self.runner.docker( + 'stop_owned_container', ['container', 'stop', '--time', '30', value['id']], + timeout=45, + ) + value = self.inspect_owned('containers', name, expected) + self.runner.docker( + 'remove_owned_container', ['container', 'rm', value['id']], timeout=45, + ) + for name, expected in reversed(self.created['volumes'].items()): + self.inspect_owned('volumes', name, expected) + self.runner.docker('remove_owned_volume', ['volume', 'rm', name]) + for name, expected in reversed(self.created['networks'].items()): + value = self.inspect_owned('networks', name, expected) + self.runner.docker('remove_owned_network', ['network', 'rm', value['id']]) + require( + all(not self.list_names(kind, owned=True) for kind in self.created), + 'cleanup_incomplete', + ) + self.assert_foreign_unchanged() + + +class Verifier: + def __init__(self, args): + self.args = args + self.script = Path(__file__).resolve(strict=True) + self.root = self.script.parent.parent.resolve(strict=True) + require(self.script == self.root / 'docker' / 'verify_edge_e2e.py', 'script_path') + require('build/' in (self.root / '.gitignore').read_text(encoding='utf-8').splitlines(), + 'build_not_gitignored') + self.run_id = secrets.token_hex(8) + self.prefix = 'truf-edge-e2e-' + self.run_id + build = self.root / 'build' + build.mkdir(exist_ok=True) + self.run_root = build / ('edge-e2e-' + self.run_id) + self.run_root.mkdir() + self.deadline = time.monotonic() + args.timeout_seconds + self.runner = CommandRunner(self.root, self.run_root, args.wsl_distro, self.deadline) + self.resources = Resources(self.runner, self.run_id) + self.names = { + 'network': self.prefix + '-network', + 'backend_data': self.prefix + '-backend-data', + 'edge_data': self.prefix + '-edge-data', + 'edge_config': self.prefix + '-edge-config', + 'auth_logs': self.prefix + '-auth-logs', + 'denylist': self.prefix + '-denylist', + 'denylist_state': self.prefix + '-denylist-state', + 'fail2ban_data': self.prefix + '-fail2ban-data', + 'hash': self.prefix + '-hash', + 'seed': self.prefix + '-seed', + 'gate_seed': self.prefix + '-gate-seed', + 'backend': self.prefix + '-backend', + 'edge': self.prefix + '-edge', + 'fail2ban': self.prefix + '-fail2ban', + 'client': self.prefix + '-client', + } + self.prefix_secret = secrets.token_hex(32) + self.edge_marker = secrets.token_hex(32) + self.admin_user = 'edge-e2e-admin' + self.admin_password = 'edge-e2e-admin-' + secrets.token_hex(12) + self.tokens = { + name: secrets.token_urlsafe(36) for name in ('good', 'wrong', 'revoked') + } + self.image_ids = {} + self._prepare_host_files() + + def _prepare_host_files(self): + self.tls_dir = self.run_root / 'tls' + self.tls_dir.mkdir() + self.edge_env = self.run_root / 'edge.env' + source_cert = self.root / 'tests' / 'fixtures' / 'worker_tls_cert.pem' + source_key = self.root / 'tests' / 'fixtures' / 'worker_tls_key.pem' + decoded = ssl._ssl._test_decode_cert(os.fspath(source_cert)) + require(('DNS', 'localhost') in decoded.get('subjectAltName', ()), 'tls_localhost_san') + require(ssl.cert_time_to_seconds(decoded['notAfter']) > time.time(), 'tls_expired') + shutil.copyfile(source_cert, self.tls_dir / source_cert.name) + shutil.copyfile(source_key, self.tls_dir / source_key.name) + write_bytes( + self.tls_dir / 'static-tls.caddy', + b'tls /etc/caddy/tls/worker_tls_cert.pem /etc/caddy/tls/worker_tls_key.pem\n', + ) + + def write_edge_environment(self, password_hash): + values = { + 'TRUF_EDGE_HOST': 'localhost', + 'TRUF_EDGE_TLS_INCLUDE': '/etc/caddy/tls/static-tls.caddy', + 'TRUF_ADMIN_PREFIX': self.prefix_secret, + 'TRUF_ADMIN_USER': self.admin_user, + 'TRUF_ADMIN_PASSWORD_HASH': password_hash, + 'TRUF_ADMIN_EDGE_MARKER': self.edge_marker, + } + content = ''.join(f'{name}={value}\n' for name, value in values.items()).encode('ascii') + write_bytes(self.edge_env, content) + + def image(self, reference, label): + _, stdout, _ = self.runner.docker('inspect_' + label, ['image', 'inspect', reference]) + value = json_bytes(stdout, label + '_image') + require(isinstance(value, list) and len(value) == 1, label + '_image_shape') + details = value[0] + require( + details.get('Os') == 'linux' and details.get('Architecture') == 'amd64' + and HEX_64.fullmatch(str(details.get('Id') or '')), + label + '_image_platform', + ) + self.image_ids[label] = details['Id'] + + def preflight(self): + require(os.name == 'nt', 'windows_host_required') + self.resources.snapshot_foreign() + self.image(TEST_IMAGE, 'test') + self.image(EDGE_IMAGE, 'edge') + self.image(FAIL2BAN_IMAGE, 'fail2ban') + + def hash_password(self): + name = self.names['hash'] + self.resources.create_container( + name, 'password-hash', + ['--network', 'none', '--read-only', '--cap-drop', 'ALL', + '--cap-add', 'NET_BIND_SERVICE', + '--entrypoint', '/usr/bin/caddy', EDGE_IMAGE, + 'hash-password', '--plaintext', self.admin_password], + ) + _, stdout, _ = self.runner.docker( + 'hash_password', ['container', 'start', '--attach', name], timeout=60, + ) + value = stdout.decode('ascii', errors='strict').strip() + require( + re.fullmatch(r'\$2[aby]\$(?:0[4-9]|[12][0-9]|3[01])\$[./A-Za-z0-9]{53}', value), + 'password_hash_shape', + ) + return value + + def create_resources(self): + self.resources.create_network(self.names['network']) + for key in ( + 'backend_data', 'edge_data', 'edge_config', 'auth_logs', 'denylist', + 'denylist_state', 'fail2ban_data', + ): + self.resources.create_volume(self.names[key], key.replace('_', '-')) + password_hash = self.hash_password() + self.write_edge_environment(password_hash) + self.resources.create_container( + self.names['seed'], 'backend-volume-seed', + ['--network', 'none', '--user', '0:0', + '--mount', f'type=volume,source={self.names["backend_data"]},target=/data', + '--entrypoint', '/usr/bin/install', TEST_IMAGE, + '-d', '-o', '10001', '-g', '10001', '-m', '0700', '/data'], + ) + self.runner.docker( + 'seed_backend_volume', ['container', 'start', '--attach', self.names['seed']], + timeout=60, + ) + self.resources.create_container( + self.names['gate_seed'], 'gate-volume-seed', + ['--network', 'none', '--user', '0:0', '--read-only', '--cap-drop', 'ALL', + '--cap-add', 'CHOWN', + '--mount', f'type=volume,source={self.names["auth_logs"]},target=/logs', + '--mount', f'type=volume,source={self.names["denylist"]},target=/denylist', + '--mount', f'type=volume,source={self.names["denylist_state"]},target=/state', + '--mount', f'type=volume,source={self.names["fail2ban_data"]},target=/fail2ban', + '--entrypoint', '/bin/sh', EDGE_IMAGE, '-c', + 'set -eu; : > /logs/admin-auth-failures.json; ' + ': > /state/.edge-e2e-owned; : > /fail2ban/.edge-e2e-owned; ' + "printf '%s\\n' '# Managed by truf-caddy-admin-denylist. Admin-route import only.' " + '> /denylist/admin-denylist.caddy; ' + 'chmod 0770 /logs; chmod 0660 /logs/admin-auth-failures.json; ' + 'chmod 0750 /denylist; chmod 0640 /denylist/admin-denylist.caddy; ' + 'chmod 0700 /state /fail2ban; ' + 'chown 10001:10001 /logs/admin-auth-failures.json ' + '/denylist/admin-denylist.caddy /logs /denylist /state /fail2ban'], + ) + self.runner.docker( + 'seed_gate_volumes', + ['container', 'start', '--attach', self.names['gate_seed']], timeout=60, + ) + + backend = [ + '--network', self.names['network'], '--user', '10001:10001', + '--read-only', '--cap-drop', 'ALL', '--security-opt', 'no-new-privileges', + '--pids-limit', '512', '--stop-timeout', '30', + '--tmpfs', '/tmp:rw,nosuid,nodev,noexec,size=128m,mode=1777', + '--mount', f'type=volume,source={self.names["backend_data"]},target=/data', + '--env', 'TRUF_ADMIN_EDGE_MARKER=' + self.edge_marker, + ] + for name, token in self.tokens.items(): + backend.extend(['--env', f'TRUF_EDGE_E2E_{name.upper()}_TOKEN={token}']) + backend.extend([ + '--entrypoint', '/usr/bin/tini', TEST_IMAGE, '--', + '/usr/local/bin/python3', '-u', '-I', '-S', '-B', + '/opt/truf/tests/edge_e2e_backend.py', + ]) + self.resources.create_container(self.names['backend'], 'backend', backend) + + tls = self.runner.wsl_path(self.tls_dir) + edge = [ + '--network', 'container:' + self.names['backend'], + '--user', '10001:10001', '--read-only', '--cap-drop', 'ALL', + '--cap-add', 'NET_BIND_SERVICE', '--security-opt', 'no-new-privileges', + '--pids-limit', '128', '--stop-timeout', '30', + '--tmpfs', '/tmp:rw,nosuid,nodev,noexec,size=16m,mode=1777', + '--tmpfs', '/run:rw,nosuid,nodev,noexec,size=4m,mode=0700,uid=10001,gid=10001', + '--mount', f'type=volume,source={self.names["edge_data"]},target=/data', + '--mount', f'type=volume,source={self.names["edge_config"]},target=/config', + '--mount', f'type=volume,source={self.names["auth_logs"]},target=/var/log/caddy', + '--mount', f'type=volume,source={self.names["denylist"]},target=/etc/caddy/denylist,readonly', + '--mount', f'type=bind,source={tls},target=/etc/caddy/tls,readonly', + '--env', 'TRUF_EDGE_HOST=localhost', + '--env', 'TRUF_EDGE_TLS_INCLUDE=/etc/caddy/tls/static-tls.caddy', + '--env', 'TRUF_ADMIN_PREFIX=' + self.prefix_secret, + '--env', 'TRUF_ADMIN_USER=' + self.admin_user, + '--env', 'TRUF_ADMIN_PASSWORD_HASH=' + password_hash.replace('$', r'\$'), + '--env', 'TRUF_ADMIN_EDGE_MARKER=' + self.edge_marker, + EDGE_IMAGE, + ] + self.resources.create_container(self.names['edge'], 'edge', edge) + + client = [ + '--network', self.names['network'], + '--user', '10001:10001', '--read-only', '--cap-drop', 'ALL', + '--security-opt', 'no-new-privileges', '--pids-limit', '128', + '--tmpfs', '/tmp:rw,nosuid,nodev,noexec,size=16m,mode=1777', + '--env', 'TRUF_ADMIN_PREFIX=' + self.prefix_secret, + '--env', 'TRUF_ADMIN_USER=' + self.admin_user, + '--env', 'TRUF_EDGE_E2E_ADMIN_PASSWORD=' + self.admin_password, + '--env', 'TRUF_EDGE_E2E_CONNECT_HOST=' + self.names['backend'], + ] + for name, token in self.tokens.items(): + client.extend(['--env', f'TRUF_EDGE_E2E_{name.upper()}_TOKEN={token}']) + client.extend(['--entrypoint', '/bin/sleep', TEST_IMAGE, '1800']) + self.resources.create_container(self.names['client'], 'same-ip-client', client) + + def create_fail2ban(self): + tls = self.runner.wsl_path(self.tls_dir) + edge_env = self.runner.wsl_path(self.edge_env) + fail2ban = [ + '--network', 'container:' + self.names['backend'], + '--pid', 'container:' + self.names['edge'], + '--user', '10001:10001', '--read-only', '--cap-drop', 'ALL', + '--security-opt', 'no-new-privileges', '--pids-limit', '128', + '--stop-timeout', '30', + '--tmpfs', '/tmp:rw,nosuid,nodev,noexec,size=16m,mode=1777', + '--tmpfs', '/run/fail2ban:rw,nosuid,nodev,noexec,size=4m,mode=0700,uid=10001,gid=10001', + '--mount', f'type=volume,source={self.names["auth_logs"]},target=/var/log/truf-edge,readonly', + '--mount', f'type=volume,source={self.names["auth_logs"]},target=/var/log/caddy', + '--mount', f'type=volume,source={self.names["denylist"]},target=/etc/truf-edge/denylist', + '--mount', f'type=volume,source={self.names["denylist"]},target=/etc/caddy/denylist,readonly', + '--mount', f'type=volume,source={self.names["denylist_state"]},target=/var/lib/truf-edge', + '--mount', f'type=volume,source={self.names["fail2ban_data"]},target=/var/lib/fail2ban', + '--mount', f'type=bind,source={tls},target=/etc/caddy/tls,readonly', + '--mount', f'type=bind,source={edge_env},target=/etc/truf-edge/edge.env,readonly', + FAIL2BAN_IMAGE, + ] + self.resources.create_container(self.names['fail2ban'], 'fail2ban-daemon', fail2ban) + + def container_running(self, name): + code, stdout, _ = self.runner.docker( + 'container_running', + ['container', 'inspect', '--format={{.State.Running}}', name], check=False, + ) + return code == 0 and stdout.strip() == b'true' + + def wait_until(self, label, function, seconds, *, alive=None): + end = min(self.deadline, time.monotonic() + seconds) + while time.monotonic() < end: + value = function() + if value is not None and value is not False: + return value + if alive is not None: + require(alive(), label + '_container_exited') + time.sleep(0.25) + raise Failure(label + '_timeout') + + def wait_backend(self): + expected = { + 'schema': 1, 'backend': 'private-network', 'postgres': 'fresh', + 'sources': ['gitlab', 'dockerhub', 'huggingface'], + } + + def ready(): + code, stdout, _ = self.runner.docker( + 'backend_ready', + ['container', 'exec', self.names['backend'], '/bin/cat', + '/data/control/ready.json'], + check=False, + ) + if code: + return None + value = json_bytes(stdout, 'backend_ready') + require(value == expected, 'backend_ready_content') + return value + + return self.wait_until( + 'backend_ready', ready, 300, + alive=lambda: self.container_running(self.names['backend']), + ) + + def client_mode(self, mode, *, check=True): + code, stdout, stderr = self.runner.docker( + 'client_' + mode.replace('-', '_'), + ['container', 'exec', self.names['client'], '/usr/local/bin/python3', + '-u', '-I', '-S', '-B', '/opt/truf/tests/edge_e2e_client.py', mode], + timeout=120, check=False, + ) + if code: + if check: + try: + failure = json_bytes(stdout, 'client_failure') + stage = failure.get('failure_stage') + except Failure: + stage = None + if not isinstance(stage, str) or not re.fullmatch( + r'[a-z][a-z0-9_]{0,39}', stage, + ): + stage = 'failed' + raise Failure('client_' + mode.replace('-', '_') + '_' + stage) + return None + require(not stderr, 'client_' + mode.replace('-', '_') + '_stderr') + value = json_bytes(stdout, 'client_' + mode.replace('-', '_')) + require(value.get('mode') == mode, 'client_' + mode.replace('-', '_') + '_mode') + require(value.get('tls') == 'validated-localhost-certificate', 'tls_validation') + return value + + def wait_edge(self, *, require_security_headers=True): + last_reason = ['client'] + + def ready(): + value = self.client_mode('probe', check=False) + if value is None: + return None + if value.get('status') == 404 and ( + not require_security_headers or value.get('ready') is True + ): + return value + status = value.get('status') + missing = value.get('missing') + if type(status) is int and status != 404: + last_reason[0] = 'status' + elif isinstance(missing, list) and missing: + name = str(missing[0]).replace('-', '_') + last_reason[0] = name if re.fullmatch(r'[a-z_]{1,40}', name) else 'headers' + else: + last_reason[0] = 'response' + return None + + try: + return self.wait_until( + 'edge_ready', ready, 45, + alive=lambda: self.container_running(self.names['edge']), + ) + except Failure as exc: + if str(exc) == 'edge_ready_timeout': + raise Failure('edge_ready_' + last_reason[0]) from None + raise + + def fail2ban_client(self, arguments, label, *, check=True): + code, stdout, stderr = self.runner.docker( + label, + ['container', 'exec', self.names['fail2ban'], '/usr/bin/fail2ban-client', + *arguments], + timeout=45, check=False, + ) + if check: + require(code == 0, label + '_failed') + if code: + return None + require(not stderr, label + '_stderr') + return stdout + + def fail2ban_bans(self): + stdout = self.fail2ban_client( + ['get', 'truf-admin-auth', 'banip'], 'fail2ban_get_bans', check=False, + ) + if stdout is None: + return None + try: + words = stdout.decode('ascii', errors='strict').split() + addresses = {ipaddress.ip_address(word).compressed.lower() for word in words} + except (UnicodeError, ValueError): + raise Failure('fail2ban_ban_inventory') from None + require(len(addresses) <= 16, 'fail2ban_ban_inventory_bound') + return addresses + + def wait_fail2ban(self, expected=None, seconds=45, label='fail2ban_ready'): + def ready(): + bans = self.fail2ban_bans() + if bans is None or (expected is not None and bans != set(expected)): + return None + return bans if bans else True + + value = self.wait_until( + label, ready, seconds, + alive=lambda: self.container_running(self.names['fail2ban']), + ) + return set() if value is True else value + + def fail2ban_set(self, arguments, label): + self.fail2ban_client(['set', 'truf-admin-auth', *arguments], label) + + def gate_file(self, path, label): + code, stdout, stderr = self.runner.docker( + label, + ['container', 'exec', self.names['fail2ban'], '/bin/cat', path], + check=False, + ) + require(code == 0 and not stderr, label + '_read') + require(len(stdout) <= MAX_OUTPUT, label + '_bound') + return stdout + + def auth_records(self): + content = self.gate_file( + '/var/log/truf-edge/admin-auth-failures.json', 'read_auth_log', + ) + records = [] + for line in content.splitlines(): + if line.strip(): + value = json_bytes(line, 'auth_log_line') + require(isinstance(value, dict), 'auth_log_object') + records.append(value) + return records + + def wait_auth_records(self, count): + return self.wait_until( + 'auth_log_records', + lambda: (records if len(records := self.auth_records()) == count else None), + 15, + ) + + def validate_auth_records(self, records, expected_count): + require(len(records) == expected_count, 'auth_log_exact_count') + parser = configparser.ConfigParser(interpolation=None) + parser.read( + self.root / 'deploy' / 'fail2ban' / 'filter.d-truf-admin-auth.conf', + encoding='ascii', + ) + failregex = parser['Definition']['failregex'] + pattern = re.compile(failregex.replace('', r'(?P[0-9A-Fa-f:.]+)')) + timestamps = [] + direct_ip = str(records[0].get('remote_ip') or '') + try: + direct_address = ipaddress.ip_address(direct_ip) + except ValueError: + raise Failure('auth_log_direct_ip') from None + require(not direct_address.is_loopback, 'auth_log_direct_ip') + for record in records: + encoded = json.dumps( + record, ensure_ascii=True, sort_keys=False, separators=(',', ':'), + ) + match = pattern.match(encoded) + require(match is not None and match.group('host') == direct_ip, 'failregex_direct_ip') + require( + record.get('status') == 401 + and record.get('event') == 'admin_auth_failure' + and record.get('remote_ip') == direct_ip, + 'auth_log_fields', + ) + require( + not re.search( + r'"(?:request|uri|headers|authorization|password|token|prefix)"\s*:', + encoded, + flags=re.IGNORECASE, + ) + and self.prefix_secret not in encoded, + 'auth_log_redaction', + ) + timestamps.append(float(record['ts'])) + require(max(timestamps) - min(timestamps) <= 600, 'fail2ban_findtime') + jail = (self.root / 'deploy' / 'fail2ban' / 'jail.d-truf-admin-auth.local').read_text( + encoding='ascii', + ) + require( + 'maxretry = 2' in jail and 'findtime = 10m' in jail + and 'bantime = 24h' in jail, + 'fail2ban_jail_semantics', + ) + return direct_ip + + def docker_logs(self, container): + _, stdout, stderr = self.runner.docker( + 'docker_logs', ['container', 'logs', container], check=False, + ) + return stdout + (b'\n' if stdout and stderr else b'') + stderr + + def assert_logs_safe(self): + forbidden = [self.admin_password, self.edge_marker, self.prefix_secret, *self.tokens.values()] + logs = [self.docker_logs(name) for name in self.names.values() if name in self.resources.created['containers']] + logs.append(self.gate_file( + '/var/log/truf-edge/admin-auth-failures.json', 'safe_auth_log', + )) + for content in logs: + require(len(content) <= MAX_OUTPUT, 'log_output_bound') + for secret in forbidden: + require(secret.encode('ascii') not in content, 'test_secret_in_log') + + def clear_run_artifacts(self): + require( + self.run_root.parent == self.root / 'build' + and re.fullmatch(r'edge-e2e-[a-f0-9]{16}', self.run_root.name) + and self.run_root.is_dir() and not self.run_root.is_symlink(), + 'run_root_guard', + ) + + def retry_writable(function, name, _error): + details = os.lstat(name) + if stat.S_ISLNK(details.st_mode): + function(name) + return + mode = stat.S_IRUSR | stat.S_IWUSR + if stat.S_ISDIR(details.st_mode): + mode |= stat.S_IXUSR + os.chmod(name, mode) + function(name) + + for path in self.run_root.iterdir(): + if path.is_symlink() or path.is_file(): + try: + path.unlink() + except PermissionError: + require(not path.is_symlink(), 'run_artifact_symlink_permission') + os.chmod(path, stat.S_IRUSR | stat.S_IWUSR) + path.unlink() + elif path.is_dir(): + shutil.rmtree(path, onerror=retry_writable) + else: + raise Failure('run_artifact_type') + require(not any(self.run_root.iterdir()), 'run_artifact_cleanup') + + def retain_failure_evidence(self, value): + content = json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + b'\n' + require(len(content) <= MAX_SAFE_EVIDENCE, 'safe_evidence_bound') + forbidden = [ + self.admin_password, self.edge_marker, self.prefix_secret, *self.tokens.values(), + ] + for secret in forbidden: + require(secret.encode('ascii') not in content, 'secret_present_in_evidence') + self.clear_run_artifacts() + write_bytes(self.run_root / 'failure.json', content) + + def run(self): + self.preflight() + self.create_resources() + self.runner.docker('start_backend', ['container', 'start', self.names['backend']]) + self.wait_backend() + self.runner.docker('start_client', ['container', 'start', self.names['client']]) + require(self.container_running(self.names['client']), 'client_not_running') + self.runner.docker('start_edge', ['container', 'start', self.names['edge']]) + self.wait_edge() + self.create_fail2ban() + self.runner.docker( + 'start_fail2ban', ['container', 'start', self.names['fail2ban']], + ) + self.wait_fail2ban(expected=set()) + self.wait_edge(require_security_headers=False) + + baseline = self.client_mode('baseline') + admin_identity = json_bytes( + self.runner.docker( + 'read_admin_identity_evidence', + ['container', 'exec', self.names['backend'], '/bin/cat', + '/data/control/last-admin-request.json'], + )[1], + 'admin_identity_evidence', + ) + require( + admin_identity == { + 'schema': 1, + 'path': '/admin-internal/operations/' + baseline['operation_id'], + 'marker_authorized': True, + 'operators': [self.admin_user], + }, + 'trusted_admin_operator_identity', + ) + worker_identity = json_bytes( + self.runner.docker( + 'read_worker_identity_evidence', + ['container', 'exec', self.names['backend'], '/bin/cat', + '/data/control/last-worker-request.json'], + )[1], + 'worker_identity_evidence', + ) + require( + worker_identity == {'schema': 1, 'admin_headers_absent': True}, + 'worker_admin_headers_absent', + ) + time.sleep(0.5) + require(self.auth_records() == [], 'missing_credentials_were_counted') + bad = self.client_mode('bad-auth') + records = self.wait_auth_records(2) + direct_ip = self.validate_auth_records(records, 2) + self.wait_fail2ban( + expected={direct_ip}, seconds=30, label='automatic_second_failure_ban', + ) + state = json_bytes( + self.gate_file('/var/lib/truf-edge/admin-denylist.json', 'read_ban_state'), + 'ban_state', + ) + expires_at = state.get('bans', {}).get(direct_ip) + now = int(time.time()) + require( + isinstance(expires_at, int) and now + 86300 <= expires_at <= now + 86500, + 'production_updater_ban_state', + ) + self.wait_edge() + self.client_mode('assert-ban') + + fail2ban = self.resources.inspect_owned( + 'containers', self.names['fail2ban'], + self.resources.created['containers'][self.names['fail2ban']], + ) + self.runner.docker( + 'kill_fail2ban_for_restart', + ['container', 'kill', '--signal', 'SIGKILL', fail2ban['id']], timeout=45, + ) + edge = self.resources.inspect_owned( + 'containers', self.names['edge'], + self.resources.created['containers'][self.names['edge']], + ) + self.runner.docker( + 'restart_edge', ['container', 'restart', '--time', '30', edge['id']], + timeout=60, + ) + self.wait_edge() + self.client_mode('assert-ban') + + fail2ban = self.resources.inspect_owned( + 'containers', self.names['fail2ban'], + self.resources.created['containers'][self.names['fail2ban']], + ) + require(not fail2ban['running'], 'fail2ban_restart_stop') + self.runner.docker( + 'restart_fail2ban', ['container', 'start', fail2ban['id']], timeout=60, + ) + self.wait_fail2ban( + expected={direct_ip}, seconds=45, label='persisted_fail2ban_ban', + ) + self.client_mode('assert-ban') + + self.fail2ban_set(['unbanip', direct_ip], 'operator_unban') + self.wait_fail2ban(expected=set(), seconds=30, label='operator_unban_applied') + state = json_bytes( + self.gate_file('/var/lib/truf-edge/admin-denylist.json', 'read_unban_state'), + 'operator_unban_state', + ) + require(state.get('bans') == {}, 'operator_unban_state') + self.wait_edge() + self.client_mode('assert-unban') + + self.fail2ban_set(['bantime', '3'], 'set_short_test_bantime') + repeated_bad = self.client_mode('bad-auth') + records = self.wait_auth_records(4) + require(self.validate_auth_records(records, 4) == direct_ip, 'repeated_direct_ip') + self.wait_fail2ban( + expected={direct_ip}, seconds=30, label='short_lived_automatic_ban', + ) + self.wait_edge() + self.client_mode('assert-ban') + self.wait_fail2ban( + expected=set(), seconds=20, label='automatic_fail2ban_expiry', + ) + state = json_bytes( + self.gate_file('/var/lib/truf-edge/admin-denylist.json', 'read_expired_state'), + 'expired_state', + ) + require(state['bans'] == {}, 'expired_state_persistence') + require( + self.gate_file( + '/etc/truf-edge/denylist/admin-denylist.caddy', 'read_expired_snippet', + ) + == b'# Managed by truf-caddy-admin-denylist. Admin-route import only.\n', + 'expired_snippet_persistence', + ) + self.wait_edge() + self.client_mode('assert-unban') + require(len(self.auth_records()) == 4, 'auth_log_count_changed_after_ban_flow') + audit = self.gate_file( + '/var/lib/truf-edge/edge-e2e-reload.audit', 'read_reload_audit', + ).decode('ascii', errors='strict').splitlines() + require( + len(audit) >= 10 and len(audit) <= 32 + and set(audit) == {'validate:0', 'reload:0'} + and audit.count('validate:0') == audit.count('reload:0'), + 'constrained_reload_audit', + ) + self.assert_logs_safe() + + summary = { + 'schema': 1, + 'status': 'passed', + 'run_id': self.run_id, + 'image_ids': self.image_ids, + 'baseline_responses': baseline['checked_responses'], + 'bad_basic_attempts': bad['attempts'] + repeated_bad['attempts'], + 'direct_ip_log_records': len(records), + 'direct_ip': direct_ip, + 'fail2ban_daemon': 'production-filter-jail-action-applied', + 'caddy_restart': 'admin-ban-persisted', + 'fail2ban_restart': 'jail-ban-restored', + 'operator_unban': 'fail2ban-actionunban-restored', + 'expiry': 'fail2ban-actionunban-restored-and-persisted', + 'worker_same_ip': 'usable-throughout', + 'admin_actor': 'authenticated-basic-user', + 'constrained_reload_pairs': audit.count('reload:0'), + 'foreign_docker_state': 'unchanged', + 'cleanup': 'complete', + } + self.resources.cleanup() + shutil.rmtree(self.run_root) + return summary + + +def parse_args(argv=None): + parser = argparse.ArgumentParser( + description='Verify the real standalone Caddy edge against a private test backend.', + ) + parser.add_argument( + '--wsl-distro', default='Ubuntu-24.04', + help='WSL distribution that owns the Docker socket (default: %(default)s)', + ) + parser.add_argument( + '--timeout-seconds', type=int, default=1200, + help='aggregate timeout (default: %(default)s)', + ) + args = parser.parse_args(argv) + if not re.fullmatch(r'[A-Za-z0-9][A-Za-z0-9._+-]{0,127}', args.wsl_distro): + parser.error('--wsl-distro contains unsupported characters') + if not 300 <= args.timeout_seconds <= 3600: + parser.error('--timeout-seconds must be between 300 and 3600') + return args + + +def main(argv=None): + verifier = None + try: + verifier = Verifier(parse_args(argv)) + summary = verifier.run() + except KeyboardInterrupt as exc: + label = 'interrupted' + failure_class = type(exc).__name__ + exit_code = None + except Failure as exc: + label = str(exc) + failure_class = type(exc).__name__ + exit_code = getattr(exc, 'exit_code', None) + except Exception as exc: + label = 'unexpected_exception' + failure_class = type(exc).__name__ + exit_code = None + else: + print(json.dumps(summary, ensure_ascii=True, sort_keys=True)) + return 0 + if verifier is not None: + verifier.runner.deadline = time.monotonic() + 600 + verifier.resources.guarded_stop() + if not re.fullmatch(r'[a-z0-9_]{1,160}', label): + label = 'verifier_failure' + if not re.fullmatch(r'[A-Za-z][A-Za-z0-9_]{0,79}', failure_class): + failure_class = 'Exception' + if type(exit_code) is not int or not -(2 ** 31) <= exit_code < 2 ** 31: + exit_code = None + failure = { + 'stage': label, 'class': failure_class, 'exit_code': exit_code, + } + if verifier is not None: + try: + verifier.retain_failure_evidence(failure) + except Exception: + pass + print(json.dumps(failure, ensure_ascii=True, sort_keys=True), file=sys.stderr) + return 1 + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/docker/verify_packaged_workers.py b/docker/verify_packaged_workers.py new file mode 100644 index 0000000..ba63e42 --- /dev/null +++ b/docker/verify_packaged_workers.py @@ -0,0 +1,2072 @@ +#!/usr/bin/env python3 +"""Exercise the real packaged Windows and Linux remote workers end to end. + +Docker is intentionally reached only through the configured WSL distribution. +The verifier creates individually named and labelled resources, never invokes +Compose or a broad cleanup command, and retains all owned artifacts on failure. +""" + +import argparse +import hashlib +import json +import os +from pathlib import Path +import re +import secrets +import shutil +import socket +import ssl +import stat +import subprocess +import sys +import threading +import time + + +LINUX_IMAGE = 'truf-remote-worker:linux-x86_64' +TEST_IMAGE = 'truf-worker-test:test' +WINDOWS_ARTIFACT = 'dist/truf-worker-windows-x86_64' +TARGETS = ( + 'https://gitlab.com/truf-e2e/a.git', + 'https://gitlab.com/truf-e2e/b.git', +) +DOCKER_TARGET = 'docker.io/truf-e2e/synthetic@sha256:' + ('7' * 64) +HUGGINGFACE_TARGET = 'truf-e2e/synthetic' +DIRECT_TARGETS = { + 'dockerhub': DOCKER_TARGET, + 'huggingface': HUGGINGFACE_TARGET, +} +ALL_TARGETS = (*TARGETS, DOCKER_TARGET, HUGGINGFACE_TARGET) +PARALLELISM = 2 +MAX_OUTPUT = 2 * 1024 * 1024 +MAX_DOCKER_RESOURCES = 1024 +MAX_SNAPSHOT_BYTES = 8 * 1024 * 1024 +MAX_SAFE_EVIDENCE = 64 * 1024 +RUN_LABEL = 'com.truf.packaged-worker-e2e.run' +PHASE_LABEL = 'com.truf.packaged-worker-e2e.phase' +KIND_LABEL = 'com.truf.packaged-worker-e2e.kind' +RESOURCE_NAME = re.compile(r'truf-packaged-worker-e2e-[a-f0-9]{16}-(windows|linux)-[a-z-]+') +HEX_40 = re.compile(r'[a-f0-9]{40}') +HEX_64 = re.compile(r'[a-f0-9]{64}') + +CONTAINER_METADATA_FORMAT = ( + '{"id":{{json .Id}},"name":{{json .Name}},"image":{{json .Image}},' + '"status":{{json .State.Status}},"running":{{json .State.Running}},' + '"paused":{{json .State.Paused}},"restarting":{{json .State.Restarting}},' + '"dead":{{json .State.Dead}},"mounts":{{json .Mounts}}}' +) +VOLUME_METADATA_FORMAT = ( + '{"name":{{json .Name}},"driver":{{json .Driver}},"scope":{{json .Scope}},' + '"created":{{json .CreatedAt}},"mountpoint":{{json .Mountpoint}},' + '"labels":{{json .Labels}},"options":{{json .Options}}}' +) + + +def isolated_test_subnet(run_id, scope): + value = int.from_bytes( + hashlib.sha256(f'{run_id}\0{scope}'.encode('ascii')).digest()[:2], 'big', + ) & 0x1fff + return f'198.{18 + (value >> 12)}.{(value >> 4) & 0xff}.{(value & 0xf) * 16}/28' + + +LOOPBACK_PROXY_SCRIPT = r'''import socket +import sys +import threading + +listen_port = int(sys.argv[1]) +upstream_host = sys.argv[2] +upstream_port = int(sys.argv[3]) + + +def pump(source, destination): + try: + while True: + block = source.recv(64 * 1024) + if not block: + break + destination.sendall(block) + except OSError: + pass + try: + destination.shutdown(socket.SHUT_WR) + except OSError: + pass + + +def bridge(client): + upstream = None + try: + upstream = socket.create_connection((upstream_host, upstream_port), timeout=5) + print('proxy upstream connected', flush=True) + client.settimeout(None) + upstream.settimeout(None) + outbound = threading.Thread(target=pump, args=(client, upstream), daemon=True) + outbound.start() + pump(upstream, client) + outbound.join(5) + except OSError as exc: + print('proxy upstream failed: ' + type(exc).__name__, flush=True) + finally: + client.close() + if upstream is not None: + upstream.close() + + +listener = socket.socket(socket.AF_INET, socket.SOCK_STREAM) +listener.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) +listener.bind(('127.0.0.1', listen_port)) +listener.listen(32) +print('proxy ready', flush=True) +while True: + connection, _ = listener.accept() + threading.Thread(target=bridge, args=(connection,), daemon=True).start() +''' + + +SEED_SCRIPT = r'''import os +import stat + +root = '/seed' +for current, directories, files in os.walk(root, topdown=False, followlinks=False): + for name in directories + files: + path = os.path.join(current, name) + details = os.lstat(path) + if stat.S_ISLNK(details.st_mode) or not ( + stat.S_ISDIR(details.st_mode) or stat.S_ISREG(details.st_mode) + ): + raise SystemExit('unsupported seed entry') + os.chown(path, 10001, 10001, follow_symlinks=False) + os.chmod(path, 0o700 if stat.S_ISDIR(details.st_mode) else 0o600) +os.chown(root, 10001, 10001, follow_symlinks=False) +os.chmod(root, 0o700) +os.unlink('/seed/_seed.py') +''' + + +INSPECT_WORKER_SCRIPT = r'''import hashlib +import json +import os +from pathlib import Path +import stat + + +def digest(path): + value = hashlib.sha256() + with open(path, 'rb', buffering=0) as handle: + for block in iter(lambda: handle.read(1024 * 1024), b''): + value.update(block) + return value.hexdigest() + + +def regular(path): + details = os.stat(path, follow_symlinks=False) + if not stat.S_ISREG(details.st_mode) or os.path.islink(path): + raise RuntimeError('non-regular worker artifact') + return details + + +def tree_identity(root): + identity = hashlib.sha256() + count = 0 + active_entries = 0 + root = Path(root) + if not root.exists(): + return {'count': 0, 'active_entries': 0, 'sha256': identity.hexdigest()} + for current, directories, files in os.walk(root, followlinks=False): + directories.sort() + files.sort() + for name in directories: + path = Path(current) / name + if path.is_symlink() or not path.is_dir(): + raise RuntimeError('unsupported worker directory') + if path.relative_to(root).parts[0] != 'abandoned': + active_entries += 1 + for name in files: + path = Path(current) / name + details = regular(path) + if path.relative_to(root).parts[0] != 'abandoned': + active_entries += 1 + relative = path.relative_to(root).as_posix().encode('utf-8') + identity.update(relative + b'\0') + identity.update(str(details.st_size).encode('ascii') + b'\0') + identity.update(str(details.st_mtime_ns).encode('ascii') + b'\0') + identity.update(digest(path).encode('ascii') + b'\n') + count += 1 + return { + 'count': count, 'active_entries': active_entries, + 'sha256': identity.hexdigest(), + } + + +states = {} +state_root = Path('/data/state-base/truf/remote-worker') +if state_root.exists(): + for path in sorted(state_root.glob('slot-*.json')): + details = regular(path) + with open(path, 'rb') as handle: + value = json.load(handle) + reservation = dict((value.get('assignment') or {}).get('reservation') or {}) + assignment = dict(value.get('assignment') or {}) + snapshot = dict(assignment.get('execution_snapshot') or {}) + planning = dict(snapshot.get('planning') or {}) + states[path.name] = { + 'phase': value.get('phase'), + 'reservation_id': reservation.get('reservation_id'), + 'bundle_id': reservation.get('bundle_id'), + 'scan_event_id': reservation.get('scan_event_id'), + 'source': reservation.get('source'), + 'platform': reservation.get('platform'), + 'target': reservation.get('target'), + 'planning_kind': planning.get('kind'), + 'sha256': digest(path), + 'size': details.st_size, + 'mtime_ns': details.st_mtime_ns, + } + +bundles = {} +bundle_root = Path('/data/client/truf/remote-worker/bundles') +if bundle_root.exists(): + for path in sorted(bundle_root.rglob('*.trb')): + details = regular(path) + bundles[path.relative_to(bundle_root).as_posix()] = { + 'sha256': digest(path), + 'size': details.st_size, + 'mtime_ns': details.st_mtime_ns, + } + +shutdown_receipt = None +receipt_path = state_root / 'control' / 'worker.exit.json' +if receipt_path.exists(): + regular(receipt_path) + with open(receipt_path, 'rb') as handle: + shutdown_receipt = json.load(handle) + +print(json.dumps({ + 'states': states, + 'bundles': bundles, + 'work': tree_identity('/data/client/truf/remote-worker/work'), + 'shutdown_receipt': shutdown_receipt, +}, ensure_ascii=True, sort_keys=True, separators=(',', ':'))) +''' + +PUBLISH_DIRECT_SCRIPT = r'''import json +import os +from datetime import datetime, timezone +from pathlib import Path +import stat +import sys + + +def fail(message): + raise RuntimeError('direct bundle publisher: ' + message) + + +def require(condition, message): + if not condition: + fail(message) + + +require(len(sys.argv) == 4, 'arguments') +package_root = Path(sys.argv[1]).resolve(strict=True) +state_root = Path(sys.argv[2]).resolve(strict=True) +bundle_root = Path(sys.argv[3]).resolve(strict=True) +manifest_path = package_root / 'worker-package.json' +with open(manifest_path, 'rb') as handle: + bootstrap_manifest = json.load(handle) +app_root = package_root / str(bootstrap_manifest.get('app_root') or '') +require(app_root.is_dir(), 'application root') +sys.path.insert(0, os.fspath(app_root / 'dependencies')) +sys.path.insert(0, os.fspath(app_root)) + +from scan_execution import ( # noqa: E402 + QueueDispositionPolicy, + stage_scan_result_in_scope, + validate_protocol2_remote_assignment, +) +from worker_package import ( # noqa: E402 + load_worker_package_manifest, + worker_package_build_compatibility, +) + +manifest = load_worker_package_manifest(manifest_path) +local_build = worker_package_build_compatibility(manifest) +capabilities = tuple( + (item['source'], item['platform'], item['planning_kind']) + for item in manifest['capabilities'] +) +expected = { + ('dockerhub', 'docker', 'docker_direct_v1'), + ('huggingface', 'huggingface', 'huggingface_space_v1'), +} +state_paths = sorted(state_root.glob('slot-*.json')) +require({path.name for path in state_paths} == {'slot-0.json', 'slot-1.json'}, 'slot states') +require(not any(bundle_root.rglob('*.trb')), 'bundle storage is not empty') +seen = set() +for path in state_paths: + details = path.stat(follow_symlinks=False) + require(stat.S_ISREG(details.st_mode) and not path.is_symlink(), 'regular slot state') + require(0 < details.st_size <= 64 * 1024 * 1024, 'slot state bound') + with open(path, 'rb') as handle: + state = json.load(handle) + require(state.get('phase') == 'assigned', 'slot phase') + assignment = dict(state.get('assignment') or {}) + validated = validate_protocol2_remote_assignment( + assignment, local_build, capabilities, + ) + reservation = validated['reservation'] + identity = ( + reservation.source, reservation.platform, validated['planning_kind'], + ) + require(identity in expected and identity not in seen, 'direct capability') + require( + validated['snapshot']['credential_ref'] == { + 'source': reservation.source, 'auth_entry': '', + }, + 'direct credential reference', + ) + now = datetime.now(timezone.utc).isoformat() + result = { + 'target': reservation.target, + 'scan_type': reservation.platform, + 'scan_event_id': reservation.scan_event_id, + 'findings': [], + 'errors': [], + 'warnings': [], + 'scan_started_at': now, + 'timestamp': now, + 'duration_sec': 0.0, + 'scan_meta': {'synthetic_claim_to_ingestion': True}, + } + limits = dict(assignment['limits']) + staged = stage_scan_result_in_scope( + result, + reservation, + os.fspath(bundle_root), + assignment['event_scan_options'], + QueueDispositionPolicy(**dict(assignment['queue_policy'])), + attempts=int(assignment['reservation'].get('attempts') or 1), + candidate_max_items=int(limits.get('candidate_max_items') or 2000), + candidate_max_bytes=int(limits.get('candidate_max_bytes') or 2 * 1024 * 1024), + require_s_drive=False, + ) + require( + staged.queue_status == 'done' + and staged.finding_count == staged.error_count == staged.candidate_count == 0, + 'clean staged bundle', + ) + seen.add(identity) +require(seen == expected, 'direct source coverage') +require(len(tuple(bundle_root.rglob('*.trb'))) == 2, 'published bundle count') +print('published direct bundles: 2') +''' + + +class Failure(RuntimeError): + """A fixed diagnostic label that cannot contain credentials or findings.""" + + +def require(condition, label): + if not condition: + raise Failure(label) + + +def sha256_bytes(value): + return hashlib.sha256(value).hexdigest() + + +def sha256_file(path): + digest = hashlib.sha256() + with open(path, 'rb', buffering=0) as handle: + for block in iter(lambda: handle.read(1024 * 1024), b''): + digest.update(block) + return digest.hexdigest() + + +def write_bytes(path, value): + path = Path(path) + path.parent.mkdir(parents=True, exist_ok=True) + with open(path, 'xb') as handle: + handle.write(value) + + +def write_json(path, value): + write_bytes( + path, + json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + b'\n', + ) + + +def load_json_bytes(value, label): + require(len(value) <= MAX_OUTPUT, label + '_size') + try: + result = json.loads(value.decode('utf-8', errors='strict')) + except (UnicodeError, ValueError): + raise Failure(label + '_json') from None + return result + + +def regular_file(path, label, maximum=MAX_OUTPUT): + path = Path(path) + try: + details = path.stat(follow_symlinks=False) + except OSError: + raise Failure(label + '_missing') from None + require( + stat.S_ISREG(details.st_mode) and not path.is_symlink() + and 0 < details.st_size <= maximum, + label + '_regular', + ) + return path + + +def synthetic_openai_token(label): + """Return a fixed, deliberately fake, detector-shaped high-entropy token.""" + alphabet = 'abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789' + + def material(part): + seed = hashlib.sha512( + ('truf-packaged-worker-e2e:' + label + ':' + part).encode('ascii') + ).hexdigest() + return ''.join( + alphabet[int(seed[index:index + 2], 16) % len(alphabet)] + for index in range(0, len(seed), 2) + )[:24] + + token = 'sk-proj-' + material('prefix') + 'T3BlbkFJ' + material('suffix') + require( + re.fullmatch(r'sk-proj-[A-Za-z0-9]{24}T3BlbkFJ[A-Za-z0-9]{24}', token), + 'synthetic_token_shape', + ) + return token + + +class CommandRunner: + def __init__(self, root, run_root, distro, deadline): + self.root = Path(root) + self.run_root = Path(run_root) + self.deadline = deadline + self.wsl = shutil.which('wsl.exe') or shutil.which('wsl') + require(self.wsl, 'wsl_unavailable') + self.docker_prefix = [ + self.wsl, '-d', distro, '--', 'sudo', '-n', 'docker', + ] + allowed = ( + 'PATH', 'PATHEXT', 'SystemRoot', 'SYSTEMROOT', 'WINDIR', 'COMSPEC', + 'TEMP', 'TMP', 'USERPROFILE', 'WSLENV', + ) + self.host_env = { + name: os.environ[name] for name in allowed if name in os.environ + } + + def execute(self, label, command, *, timeout=60, check=True, env=None, cwd=None): + remaining = self.deadline - time.monotonic() + require(remaining > 0, 'aggregate_timeout') + timeout = max(0.1, min(float(timeout), remaining)) + process = None + try: + process = subprocess.Popen( + command, + cwd=os.fspath(cwd or self.root), + env=env or self.host_env, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + creationflags=getattr(subprocess, 'CREATE_NO_WINDOW', 0), + ) + try: + stdout, stderr = process.communicate(timeout=timeout) + except subprocess.TimeoutExpired: + process.kill() + process.communicate() + raise Failure(label + '_timeout') from None + except Failure: + raise + except OSError: + raise Failure(label + '_unavailable') from None + require( + len(stdout) <= MAX_OUTPUT and len(stderr) <= MAX_OUTPUT, + label + '_output_bound', + ) + if check: + if process.returncode != 0: + error = Failure(label + '_failed') + error.exit_code = process.returncode + raise error + return process.returncode, stdout, stderr + + def docker(self, label, arguments, *, timeout=60, check=True): + return self.execute( + label, [*self.docker_prefix, *arguments], timeout=timeout, check=check, + ) + + def docker_json(self, label, arguments, *, timeout=60): + _, stdout, _ = self.docker(label, arguments, timeout=timeout) + return load_json_bytes(stdout, label) + + def wsl_path(self, path): + _, stdout, _ = self.execute( + 'wsl_path', + [self.wsl, '-d', self.docker_prefix[2], '--', 'wslpath', '-a', '-u', + os.fspath(path).replace('\\', '/')], + timeout=30, + ) + try: + value = stdout.decode('utf-8', errors='strict').strip() + except UnicodeError: + raise Failure('wsl_path_encoding') from None + require(value.startswith('/') and '\x00' not in value, 'wsl_path_invalid') + return value + + +class ResourceTracker: + def __init__(self, runner, run_id): + self.runner = runner + self.run_id = run_id + self.resources = {'containers': {}, 'volumes': {}, 'networks': {}} + self.foreign_baseline = None + + def labels(self, phase, kind): + return [ + '--label', f'{RUN_LABEL}={self.run_id}', + '--label', f'{PHASE_LABEL}={phase}', + '--label', f'{KIND_LABEL}={kind}', + ] + + @staticmethod + def singular(kind): + return {'containers': 'container', 'volumes': 'volume', 'networks': 'network'}[kind] + + def list_names(self, kind, *, name=None, owned=False): + singular = self.singular(kind) + arguments = [singular, 'ls'] + if kind == 'containers': + arguments.append('--all') + if name is not None: + pattern = '^/' + name + '$' if kind == 'containers' else '^' + name + '$' + arguments.extend(['--filter', 'name=' + pattern]) + if owned: + arguments.extend(['--filter', f'label={RUN_LABEL}={self.run_id}']) + field = 'Names' if kind == 'containers' else 'Name' + arguments.append('--format={{.' + field + '}}') + _, stdout, _ = self.runner.docker('list_' + kind, arguments) + try: + names = {line for line in stdout.decode('ascii').splitlines() if line} + except UnicodeError: + raise Failure('resource_name_encoding') from None + require(len(names) <= 32, 'resource_inventory_bound') + return names + + def guard_new(self, kind, name): + require(RESOURCE_NAME.fullmatch(name), 'resource_name_guard') + require(not self.list_names(kind, name=name), 'resource_name_collision') + + def inspect_owned(self, kind, name, expected=None): + singular = self.singular(kind) + if kind == 'containers': + template = ( + '{"id":{{json .Id}},"name":{{json .Name}},' + '"run":{{json (index .Config.Labels "' + RUN_LABEL + '")}},' + '"phase":{{json (index .Config.Labels "' + PHASE_LABEL + '")}},' + '"kind":{{json (index .Config.Labels "' + KIND_LABEL + '")}},' + '"running":{{json .State.Running}},"paused":{{json .State.Paused}},' + '"restarting":{{json .State.Restarting}}}' + ) + else: + template = ( + '{"id":{{json ' + ('.Id' if kind == 'networks' else '""') + '}},' + '"name":{{json .Name}},"run":{{json (index .Labels "' + RUN_LABEL + '")}},' + '"phase":{{json (index .Labels "' + PHASE_LABEL + '")}},' + '"kind":{{json (index .Labels "' + KIND_LABEL + '")}}}' + ) + value = self.runner.docker_json( + 'inspect_owned_' + singular, [singular, 'inspect', '--format', template, name], + ) + require( + value.get('name') == ('/' + name if kind == 'containers' else name) + and value.get('run') == self.run_id + and isinstance(value.get('phase'), str) and isinstance(value.get('kind'), str) + and (kind == 'volumes' or HEX_64.fullmatch(str(value.get('id') or ''))), + 'owned_' + singular + '_identity_guard', + ) + if expected is not None: + require( + (expected.get('id') is None or value.get('id') == expected['id']) + and value['phase'] == expected['phase'] and value['kind'] == expected['kind'], + 'owned_' + singular + '_identity_changed', + ) + return value + + def metadata_names(self, kind): + arguments = ([kind, 'ls', '--all', '--no-trunc', '--format={{.ID}}'] + if kind == 'container' else [kind, 'ls', '--format={{.Name}}']) + _, stdout, _ = self.runner.docker('foreign_' + kind + '_list', arguments) + try: + values = {line for line in stdout.decode('ascii').splitlines() if line} + except UnicodeError: + raise Failure('foreign_' + kind + '_inventory_encoding') from None + require(len(values) <= MAX_DOCKER_RESOURCES, 'foreign_' + kind + '_inventory_bound') + return values + + def container_metadata(self, identifier): + require(HEX_64.fullmatch(identifier), 'foreign_container_id_guard') + value = self.runner.docker_json( + 'foreign_container_inspect', + ['container', 'inspect', '--format', CONTAINER_METADATA_FORMAT, identifier], + ) + require( + value.get('id') == identifier and isinstance(value.get('name'), str) + and HEX_64.fullmatch(str(value.get('image') or '').removeprefix('sha256:')) + and value.get('status') in ( + 'created', 'running', 'paused', 'restarting', 'removing', 'exited', 'dead', + ) + and all(type(value.get(key)) is bool for key in ( + 'running', 'paused', 'restarting', 'dead', + )) + and isinstance(value.get('mounts'), list) and len(value['mounts']) <= 128, + 'foreign_container_metadata_guard', + ) + mounts = [{ + key: mount.get(key) for key in ( + 'Type', 'Name', 'Source', 'Destination', 'Driver', 'Mode', 'RW', 'Propagation', + ) + } for mount in value['mounts']] + return { + 'id': value['id'], 'name': value['name'], 'image': value['image'], + 'status': value['status'], 'running': value['running'], 'paused': value['paused'], + 'restarting': value['restarting'], 'dead': value['dead'], + 'mounts_sha256': sha256_bytes(json.dumps( + mounts, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')), + } + + def volume_metadata(self, name): + value = self.runner.docker_json( + 'foreign_volume_inspect', + ['volume', 'inspect', '--format', VOLUME_METADATA_FORMAT, name], + ) + require( + value.get('name') == name and isinstance(value.get('driver'), str) + and isinstance(value.get('scope'), str) and isinstance(value.get('mountpoint'), str) + and (value.get('created') is None or isinstance(value['created'], str)) + and (value.get('labels') is None or isinstance(value['labels'], dict)) + and (value.get('options') is None or isinstance(value['options'], dict)), + 'foreign_volume_metadata_guard', + ) + return { + 'name': name, 'driver': value['driver'], 'scope': value['scope'], + 'created': value['created'], + 'mountpoint_sha256': sha256_bytes(value['mountpoint'].encode('utf-8')), + 'labels_sha256': sha256_bytes(json.dumps( + value['labels'], ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')), + 'options_sha256': sha256_bytes(json.dumps( + value['options'], ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii')), + } + + def metadata_snapshot(self, *, exclude_owned=False): + containers = {} + owned_ids = { + value['id']: (name, value) + for name, value in self.resources['containers'].items() if value.get('id') + } + for identifier in sorted(self.metadata_names('container')): + if exclude_owned and identifier in owned_ids: + name, expected = owned_ids[identifier] + self.inspect_owned('containers', name, expected) + continue + containers[identifier] = self.container_metadata(identifier) + volumes = {} + for name in sorted(self.metadata_names('volume')): + value = self.volume_metadata(name) + expected = self.resources['volumes'].get(name) + if exclude_owned and expected is not None and expected.get('metadata') == value: + continue + volumes[name] = value + snapshot = {'containers': containers, 'volumes': volumes} + require(len(json.dumps(snapshot, ensure_ascii=True, sort_keys=True)) <= MAX_SNAPSHOT_BYTES, + 'foreign_metadata_snapshot_bound') + return snapshot + + def snapshot_foreign(self): + require(self.foreign_baseline is None, 'foreign_snapshot_already_taken') + self.foreign_baseline = self.metadata_snapshot() + + def assert_foreign_unchanged(self): + require(self.foreign_baseline is not None, 'foreign_snapshot_missing') + require(self.metadata_snapshot(exclude_owned=True) == self.foreign_baseline, + 'foreign_docker_state_changed') + + def create_volume(self, name, phase, role): + self.guard_new('volumes', name) + self.resources['volumes'][name] = {'id': None, 'phase': phase, 'kind': role} + _, stdout, _ = self.runner.docker( + 'create_volume', + ['volume', 'create', *self.labels(phase, role), name], + ) + require(stdout.decode('ascii', errors='ignore').strip() == name, 'volume_create_result') + self.inspect_owned('volumes', name, self.resources['volumes'][name]) + self.resources['volumes'][name]['metadata'] = self.volume_metadata(name) + + def create_network(self, name, phase): + self.guard_new('networks', name) + self.resources['networks'][name] = { + 'id': None, 'phase': phase, 'kind': 'network', + } + _, stdout, _ = self.runner.docker( + 'create_network', + ['network', 'create', '--driver', 'bridge', '--internal', + '--subnet', isolated_test_subnet(self.run_id, phase), + *self.labels(phase, 'network'), name], + ) + identifier = stdout.decode('ascii', errors='strict').strip() + require(HEX_64.fullmatch(identifier), 'network_create_result') + self.resources['networks'][name]['id'] = identifier + self.inspect_owned('networks', name, self.resources['networks'][name]) + + def planned_container(self, name, phase, role): + self.resources['containers'][name] = {'id': None, 'phase': phase, 'kind': role} + + def created_container(self, name, identifier): + self.resources['containers'][name]['id'] = identifier + self.inspect_owned('containers', name, self.resources['containers'][name]) + + def inventory(self): + inventory = {} + for kind in self.resources: + try: + inventory[kind] = sorted(self.list_names(kind, owned=True)) + except Exception: + inventory[kind] = sorted(self.resources[kind]) + return inventory + + def guarded_stop(self): + stopped = 0 + for name, expected in reversed(self.resources['containers'].items()): + try: + value = self.inspect_owned('containers', name, expected) + if expected['id'] is None: + expected['id'] = value['id'] + if value['running'] or value['paused'] or value['restarting']: + value = self.inspect_owned('containers', name, expected) + code, _, _ = self.runner.docker( + 'guarded_stop_owned_container', + ['container', 'stop', '--time', '30', value['id']], + timeout=45, check=False, + ) + stopped += int(code == 0) + except (Exception, KeyboardInterrupt): + continue + return stopped + + def cleanup(self): + inventory = { + kind: self.list_names(kind, owned=True) for kind in self.resources + } + for kind, expected in self.resources.items(): + require(inventory[kind] == set(expected), 'cleanup_ownership_guard') + for name, expected in reversed(self.resources['containers'].items()): + value = self.inspect_owned('containers', name, expected) + if value['running'] or value['paused'] or value['restarting']: + value = self.inspect_owned('containers', name, expected) + self.runner.docker( + 'stop_owned_container', ['container', 'stop', '--time', '30', value['id']], + timeout=45, + ) + value = self.inspect_owned('containers', name, expected) + self.runner.docker( + 'remove_owned_container', ['container', 'rm', value['id']], + timeout=45, + ) + for name, expected in reversed(self.resources['volumes'].items()): + self.inspect_owned('volumes', name, expected) + self.runner.docker('remove_owned_volume', ['volume', 'rm', name]) + for name, expected in reversed(self.resources['networks'].items()): + value = self.inspect_owned('networks', name, expected) + self.runner.docker('remove_owned_network', ['network', 'rm', value['id']]) + require( + all(not self.list_names(kind, owned=True) for kind in self.resources), + 'cleanup_incomplete', + ) + self.assert_foreign_unchanged() + + +class WindowsClient: + def __init__(self, command, environment, log_path): + self.command = command + self.environment = environment + self.log_path = Path(log_path) + self.process = None + self.reader = None + self.output = bytearray() + self.output_lock = threading.Lock() + self.output_exceeded = False + + def _read_output(self, process): + for block in iter(lambda: process.stdout.read(64 * 1024), b''): + with self.output_lock: + if len(self.output) + len(block) <= MAX_OUTPUT: + self.output.extend(block) + else: + self.output_exceeded = True + + def read_log(self): + with self.output_lock: + require(not self.output_exceeded, 'windows_log_bound') + return bytes(self.output) + + def start(self): + require(self.process is None, 'windows_client_already_started') + try: + self.process = subprocess.Popen( + self.command, + cwd=os.fspath(self.log_path.parent), + env=self.environment, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + creationflags=( + getattr(subprocess, 'CREATE_NEW_PROCESS_GROUP', 0) + | getattr(subprocess, 'CREATE_NO_WINDOW', 0) + ), + ) + except OSError: + raise Failure('windows_client_unavailable') from None + self.reader = threading.Thread(target=self._read_output, args=(self.process,), daemon=True) + self.reader.start() + + def alive(self): + return self.process is not None and self.process.poll() is None + + def stop(self): + if self.process is None: + return + if self.process.poll() is None: + taskkill = shutil.which('taskkill.exe') or 'taskkill.exe' + subprocess.run( + [taskkill, '/PID', str(self.process.pid), '/T', '/F'], + stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, check=False, + creationflags=getattr(subprocess, 'CREATE_NO_WINDOW', 0), + ) + try: + self.process.wait(timeout=15) + except subprocess.TimeoutExpired: + self.process.kill() + self.process.wait(timeout=5) + if self.reader is not None: + self.reader.join(timeout=5) + if self.process.stdout is not None: + self.process.stdout.close() + if self.reader is not None: + self.reader.join(timeout=1) + self.process = None + self.reader = None + + +class LoopbackProxy: + def __init__(self, runner, upstream, log_path): + self.runner = runner + self.upstream = upstream + self.log_path = Path(log_path) + self.process = None + self.port = None + self.output = None + + def start(self): + require(self.process is None, 'loopback_proxy_already_started') + with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as reservation: + reservation.bind(('127.0.0.1', 0)) + self.port = int(reservation.getsockname()[1]) + try: + self.output = self.log_path.open('wb') + self.process = subprocess.Popen( + [self.runner.wsl, '-d', self.runner.docker_prefix[2], '--', + 'python3', '-u', '-c', LOOPBACK_PROXY_SCRIPT, + str(self.port), self.upstream, '8443'], + cwd=os.fspath(self.runner.run_root), env=self.runner.host_env, + stdin=subprocess.DEVNULL, stdout=self.output, + stderr=subprocess.STDOUT, + creationflags=getattr(subprocess, 'CREATE_NO_WINDOW', 0), + ) + except OSError: + if self.output is not None: + self.output.close() + self.output = None + raise Failure('loopback_proxy_unavailable') from None + + end = min(self.runner.deadline, time.monotonic() + 60) + context = ssl.create_default_context( + cafile=os.fspath(self.runner.root / 'tests' / 'fixtures' / 'worker_tls_cert.pem'), + ) + while time.monotonic() < end: + require(self.process.poll() is None, 'loopback_proxy_exited') + try: + with socket.create_connection( + ('127.0.0.1', self.port), timeout=0.5, + ) as connection, context.wrap_socket( + connection, server_hostname='localhost', + ): + return self.port + except OSError: + time.sleep(0.1) + try: + detail = self.log_path.read_text(encoding='utf-8', errors='replace') + except OSError: + detail = '' + if 'ConnectionRefusedError' in detail: + raise Failure('loopback_proxy_upstream_refused') + if 'TimeoutError' in detail: + raise Failure('loopback_proxy_upstream_timeout') + if 'proxy upstream connected' in detail: + raise Failure('loopback_proxy_tls_timeout') + raise Failure('loopback_proxy_timeout') + + def stop(self): + if self.process is not None and self.process.poll() is None: + self.process.terminate() + try: + self.process.wait(timeout=10) + except subprocess.TimeoutExpired: + self.process.kill() + self.process.wait(timeout=5) + self.process = None + if self.output is not None: + self.output.close() + self.output = None + + +class Verifier: + def __init__(self, args): + self.args = args + self.script = Path(__file__).resolve(strict=True) + self.root = self.script.parent.parent.resolve(strict=True) + require(self.script == self.root / 'docker' / 'verify_packaged_workers.py', 'script_path') + require('build/' in (self.root / '.gitignore').read_text(encoding='utf-8').splitlines(), + 'build_not_gitignored') + self.run_id = secrets.token_hex(8) + self.prefix = 'truf-packaged-worker-e2e-' + self.run_id + build = self.root / 'build' + build.mkdir(exist_ok=True) + self.run_root = build / ('pwe-' + self.run_id) + self.run_root.mkdir() + self.deadline = time.monotonic() + args.timeout_seconds + self.runner = CommandRunner(self.root, self.run_root, args.wsl_distro, self.deadline) + self.tracker = ResourceTracker(self.runner, self.run_id) + self.windows_artifact = Path(args.windows_artifact) + if not self.windows_artifact.is_absolute(): + self.windows_artifact = self.root / self.windows_artifact + self.windows_artifact = self.windows_artifact.resolve(strict=True) + require(self.windows_artifact.is_relative_to(self.root), 'windows_artifact_outside_root') + self.tokens = { + TARGETS[0]: synthetic_openai_token('a'), + TARGETS[1]: synthetic_openai_token('b'), + } + require(len(set(self.tokens.values())) == 2, 'synthetic_tokens_distinct') + self.device_tokens = {} + self.windows_client = None + self.loopback_proxy = None + self.image_details = {} + self.repository_heads = {} + self.phase_results = {} + + def image(self, reference, label, *, worker=False): + value = self.runner.docker_json(label, ['image', 'inspect', reference]) + require(isinstance(value, list) and len(value) == 1, label + '_shape') + details = value[0] + require( + details.get('Os') == 'linux' and details.get('Architecture') == 'amd64' + and HEX_64.fullmatch(str(details.get('Id') or '').removeprefix('sha256:')), + label + '_platform', + ) + if worker: + expected = [ + '/usr/bin/tini', '--', '/usr/local/bin/python3', '-u', '-I', '-S', '-B', + '/opt/truf-worker/app/remote_worker_bootstrap.py', '--', + ] + require(details.get('Config', {}).get('Entrypoint') == expected, + 'linux_worker_entrypoint') + require(details.get('Config', {}).get('Cmd') == ['run'], + 'linux_worker_default_command') + require(details.get('Config', {}).get('User') == '10001:10001', + 'linux_worker_user') + self.image_details[label] = details['Id'] + + def prepare_repositories(self): + git = shutil.which('git.exe') or shutil.which('git') + require(git, 'git_unavailable') + repositories = self.run_root / 'repositories' + repositories.mkdir() + base_env = dict(self.runner.host_env) + base_env.update({ + 'GIT_CONFIG_NOSYSTEM': '1', 'GIT_CONFIG_GLOBAL': 'NUL', + 'GIT_TERMINAL_PROMPT': '0', 'TZ': 'UTC', + 'GIT_AUTHOR_NAME': 'TRUF Packaged Worker Fixture', + 'GIT_AUTHOR_EMAIL': 'packaged-worker-fixture@invalid.example', + 'GIT_COMMITTER_NAME': 'TRUF Packaged Worker Fixture', + 'GIT_COMMITTER_EMAIL': 'packaged-worker-fixture@invalid.example', + 'GIT_AUTHOR_DATE': '2001-01-01T00:00:00+00:00', + 'GIT_COMMITTER_DATE': '2001-01-01T00:00:00+00:00', + }) + for index, target in enumerate(TARGETS): + name = chr(ord('a') + index) + path = repositories / name + path.mkdir() + write_bytes(path / 'synthetic.env', ('OPENAI_API_KEY=' + self.tokens[target] + '\n').encode('ascii')) + self.runner.execute( + 'git_init_' + name, + [git, 'init', '--quiet', '--initial-branch=main', '--template=', os.fspath(path)], + env=base_env, + ) + self.runner.execute( + 'git_config_' + name, + [git, '-C', os.fspath(path), 'config', 'core.autocrlf', 'false'], + env=base_env, + ) + self.runner.execute( + 'git_add_' + name, [git, '-C', os.fspath(path), 'add', '--', 'synthetic.env'], + env=base_env, + ) + self.runner.execute( + 'git_commit_' + name, + [git, '-C', os.fspath(path), 'commit', '--quiet', '--no-gpg-sign', + '--message', 'Add deterministic synthetic detector fixture'], + env=base_env, + ) + _, stdout, _ = self.runner.execute( + 'git_head_' + name, [git, '-C', os.fspath(path), 'rev-parse', 'HEAD'], + env=base_env, + ) + head = stdout.decode('ascii', errors='strict').strip() + require(HEX_40.fullmatch(head), 'fixture_commit_hash') + _, status_output, _ = self.runner.execute( + 'git_status_' + name, + [git, '-C', os.fspath(path), 'status', '--porcelain=v1'], env=base_env, + ) + require(not status_output, 'fixture_repository_dirty') + self.repository_heads[target] = head + require(len(set(self.repository_heads.values())) == 2, 'fixture_commits_distinct') + + def prepare_fixtures(self): + manifest_path = regular_file( + self.windows_artifact / 'worker-package.json', 'windows_manifest', + ) + for relative, label in ( + ('truf-worker.cmd', 'windows_cli_entrypoint'), + ('run-worker.cmd', 'windows_entrypoint'), + ('runtime/python/python.exe', 'windows_python'), + ('bin/trufflehog.exe', 'windows_scanner'), + ): + regular_file(self.windows_artifact / Path(relative), label, maximum=256 * 1024 * 1024) + require( + b'remote_worker_bootstrap.py" -- %*' in + (self.windows_artifact / 'truf-worker.cmd').read_bytes(), + 'windows_cli_entrypoint_contents', + ) + require( + b'remote_worker_bootstrap.py" -- run %*' in + (self.windows_artifact / 'run-worker.cmd').read_bytes(), + 'windows_compat_entrypoint_contents', + ) + windows_manifest = manifest_path.read_bytes() + windows_value = load_json_bytes(windows_manifest, 'windows_manifest') + expected_capabilities = { + ('gitlab', 'gitlab', 'exact_git_v1'), + ('dockerhub', 'docker', 'docker_direct_v1'), + ('huggingface', 'huggingface', 'huggingface_space_v1'), + } + windows_capabilities = { + ( + item.get('source'), item.get('platform'), + item.get('planning_kind'), + ) + for item in windows_value.get('capabilities', []) + if isinstance(item, dict) + } + require( + windows_value.get('schema') == 3 + and windows_value.get('protocol_version') == 2 + and windows_value.get('platform_tag') == 'windows-x86_64', + 'windows_manifest_identity', + ) + require( + windows_capabilities == expected_capabilities, + 'windows_manifest_capabilities', + ) + helper_name = self.prefix + '-linux-manifest' + self.create_container( + helper_name, 'linux', 'manifest-reader', + ['--network', 'none', '--entrypoint', '/bin/cat', self.args.linux_image, + '/opt/truf-worker/worker-package.json'], + ) + _, linux_manifest, _ = self.runner.docker( + 'linux_manifest', + ['container', 'start', '--attach', helper_name], + ) + _, exit_code, _ = self.runner.docker( + 'linux_manifest_exit', + ['container', 'inspect', '--format={{.State.ExitCode}}', helper_name], + ) + require(exit_code.strip() == b'0', 'linux_manifest_reader_exit') + linux_value = load_json_bytes(linux_manifest, 'linux_manifest') + linux_capabilities = { + ( + item.get('source'), item.get('platform'), + item.get('planning_kind'), + ) + for item in linux_value.get('capabilities', []) + if isinstance(item, dict) + } + require( + linux_value.get('schema') == 3 + and linux_value.get('protocol_version') == 2 + and linux_value.get('platform_tag') == 'linux-x86_64', + 'linux_manifest_identity', + ) + require( + linux_capabilities == expected_capabilities + and linux_capabilities == windows_capabilities, + 'linux_manifest_capabilities', + ) + + fixture = self.run_root / 'fixture' + fixture.mkdir() + write_json(fixture / 'windows-manifest.json', windows_value) + write_json(fixture / 'linux-manifest.json', linux_value) + write_json(fixture / 'fixture.json', { + 'schema': 1, + 'targets': list(TARGETS), + 'direct_targets': DIRECT_TARGETS, + 'repositories': { + target: {'head_sha': self.repository_heads[target]} for target in TARGETS + }, + }) + certificate = regular_file( + self.root / 'tests' / 'fixtures' / 'worker_tls_cert.pem', 'tls_certificate', + ) + private_key = regular_file( + self.root / 'tests' / 'fixtures' / 'worker_tls_key.pem', 'tls_private_key', + ) + decoded = ssl._ssl._test_decode_cert(os.fspath(certificate)) + require(('DNS', 'localhost') in decoded.get('subjectAltName', ()), 'tls_localhost_san') + require(ssl.cert_time_to_seconds(decoded['notAfter']) > time.time(), 'tls_certificate_expired') + write_bytes(fixture / 'worker_tls_cert.pem', certificate.read_bytes()) + write_bytes(fixture / 'worker_tls_key.pem', private_key.read_bytes()) + self.fixture = fixture + self.manifest_hashes = { + 'windows': sha256_bytes(windows_manifest), + 'linux': sha256_bytes(linux_manifest), + } + + def preflight(self): + require(os.name == 'nt', 'windows_host_required') + self.tracker.snapshot_foreign() + self.image(self.args.linux_image, 'linux_image', worker=True) + self.image(self.args.test_image, 'test_image') + self.prepare_repositories() + self.prepare_fixtures() + + def phase_names(self, phase): + base = self.prefix + '-' + phase + return { + 'network': base + '-network', + 'server_data': base + '-server-data', + 'server_seed': base + '-server-seed', + 'fixture': base + '-fixture-data', + 'worker_data': base + '-worker-data', + 'fixture_seed': base + '-fixture-seed', + 'worker_seed': base + '-worker-seed', + 'server': base + '-server', + 'worker': base + '-worker', + } + + def create_container(self, name, phase, role, arguments): + self.tracker.guard_new('containers', name) + self.tracker.planned_container(name, phase, role) + _, stdout, _ = self.runner.docker( + 'create_' + role, + ['container', 'create', '--pull=never', '--name', name, + *self.tracker.labels(phase, role), *arguments], + ) + identifier = stdout.decode('ascii', errors='ignore').strip() + require(HEX_64.fullmatch(identifier), role + '_container_id') + self.tracker.created_container(name, identifier) + + def seed_volume(self, phase, role, name, volume, source): + self.create_container( + name, phase, role, + ['--network', 'none', '--user', '0:0', + '--mount', f'type=volume,source={volume},target=/seed', + '--entrypoint', '/usr/local/bin/python3', self.args.test_image, + '-u', '-I', '-S', '-B', '/seed/_seed.py'], + ) + source_wsl = self.runner.wsl_path(Path(source).resolve(strict=True)) + self.runner.docker( + 'copy_' + role, ['container', 'cp', source_wsl + '/.', name + ':/seed/'], + timeout=120, + ) + self.runner.docker('start_' + role, ['container', 'start', '--attach', name], timeout=120) + _, stdout, _ = self.runner.docker( + 'wait_' + role, + ['container', 'inspect', '--format={{.State.ExitCode}}', name], + ) + require(stdout.strip() == b'0', role + '_exit') + + def prepare_phase_resources(self, phase): + names = self.phase_names(phase) + phase_root = self.run_root / phase + phase_root.mkdir() + if phase == 'windows': + write_bytes( + phase_root / 'publish-direct.py', PUBLISH_DIRECT_SCRIPT.encode('ascii'), + ) + fixture_seed = phase_root / 'fixture-seed' + shutil.copytree(self.fixture, fixture_seed) + write_bytes(fixture_seed / '_seed.py', SEED_SCRIPT.encode('ascii')) + server_seed = phase_root / 'server-seed' + server_seed.mkdir() + write_bytes(server_seed / '_seed.py', SEED_SCRIPT.encode('ascii')) + self.tracker.create_network(names['network'], phase) + self.tracker.create_volume(names['server_data'], phase, 'server-data') + self.seed_volume( + phase, 'server-seed', names['server_seed'], names['server_data'], server_seed, + ) + self.tracker.create_volume(names['fixture'], phase, 'fixture-data') + self.seed_volume( + phase, 'fixture-seed', names['fixture_seed'], names['fixture'], fixture_seed, + ) + if phase == 'linux': + worker_seed = phase_root / 'worker-seed' + worker_seed.mkdir() + shutil.copytree(self.run_root / 'repositories', worker_seed / 'repos') + shutil.copyfile( + self.fixture / 'worker_tls_cert.pem', worker_seed / 'worker_tls_cert.pem', + ) + write_bytes(worker_seed / '_seed.py', SEED_SCRIPT.encode('ascii')) + write_bytes(worker_seed / '_inspect.py', INSPECT_WORKER_SCRIPT.encode('ascii')) + write_bytes( + worker_seed / '_publish_direct.py', PUBLISH_DIRECT_SCRIPT.encode('ascii'), + ) + self.tracker.create_volume(names['worker_data'], phase, 'worker-data') + self.seed_volume( + phase, 'worker-seed', names['worker_seed'], names['worker_data'], worker_seed, + ) + self.device_tokens[phase] = secrets.token_hex(32) + self.start_server(phase, names) + self.wait_control(phase, names['server'], 'prepared.json', { + 'schema': 1, 'phase': phase, 'target_count': 2, + }, 180) + return names, phase_root + + def start_server(self, phase, names): + self.create_container( + names['server'], phase, 'server', + ['--network', names['network'], + '--read-only', '--cap-drop', 'ALL', '--security-opt', 'no-new-privileges', + '--pids-limit', '512', '--stop-timeout', '30', + '--tmpfs', '/tmp:rw,nosuid,nodev,noexec,size=128m,mode=1777', + '--mount', f'type=volume,source={names["server_data"]},target=/data', + '--mount', f'type=volume,source={names["fixture"]},target=/fixture,readonly', + '--env', 'TRUF_WORKER_E2E_TOKEN=' + self.device_tokens[phase], + '--env', 'TRUF_WORKER_E2E_PHASE=' + phase, + '--entrypoint', '/usr/bin/tini', self.args.test_image, + '--', '/usr/local/bin/python3', '-u', '-I', '-S', '-B', + '/opt/truf/tests/packaged_worker_e2e_server.py'], + ) + self.runner.docker('start_server', ['container', 'start', names['server']]) + require(self.container_running(names['server']), 'server_not_running') + + def container_running(self, name): + code, stdout, _ = self.runner.docker( + 'container_running', + ['container', 'inspect', '--format={{.State.Running}}', name], + check=False, + ) + return code == 0 and stdout.strip() == b'true' + + def read_control(self, server, name): + code, stdout, _ = self.runner.docker( + 'read_control', ['container', 'exec', server, '/bin/cat', '/data/control/' + name], + check=False, + ) + if code: + return None + value = load_json_bytes(stdout, 'control_' + name.replace('.', '_')) + require(isinstance(value, dict), 'control_object') + return value + + def wait_until(self, label, function, seconds, *, alive=None): + end = min(self.deadline, time.monotonic() + seconds) + while time.monotonic() < end: + value = function() + if value is not None and value is not False: + return value + if alive is not None: + require(alive(), label + '_process_exited') + time.sleep(0.25) + raise Failure(label + '_timeout') + + def wait_control(self, phase, server, name, expected, seconds, *, alive=None): + def marker(): + value = self.read_control(server, name) + if value is None: + return None + require(value == expected, phase + '_' + name.replace('.', '_') + '_content') + return value + + return self.wait_until( + phase + '_' + name.replace('.', '_'), marker, seconds, + alive=lambda: self.container_running(server) and (alive is None or alive()), + ) + + @staticmethod + def git_environment(local_repositories, certificate, *, linux=False): + environment = { + 'SSL_CERT_FILE': os.fspath(certificate), + 'GIT_ALLOW_PROTOCOL': 'file', + 'GIT_TERMINAL_PROMPT': '0', + 'GIT_LFS_SKIP_SMUDGE': '1', + 'GIT_CONFIG_NOSYSTEM': '1', + 'GIT_CONFIG_GLOBAL': '/dev/null' if linux else 'NUL', + 'GIT_CONFIG_COUNT': '2', + } + for index, target in enumerate(TARGETS): + local = local_repositories[index] + uri = local if linux else Path(local).resolve(strict=True).as_uri() + environment[f'GIT_CONFIG_KEY_{index}'] = f'url.{uri}.insteadOf' + environment[f'GIT_CONFIG_VALUE_{index}'] = target + return environment + + def windows_snapshot(self, phase_root): + state_root = self.run_root / 'w' / 'TRUF' / 'RemoteWorker' + bundle_root = state_root / 'bundles' + work_root = state_root / 'work' + states = {} + if state_root.exists(): + for path in sorted(state_root.glob('slot-*.json')): + regular_file(path, 'windows_state', maximum=64 * 1024 * 1024) + value = load_json_bytes(path.read_bytes(), 'windows_state') + assignment = dict(value.get('assignment') or {}) + reservation = dict(assignment.get('reservation') or {}) + snapshot = dict(assignment.get('execution_snapshot') or {}) + planning = dict(snapshot.get('planning') or {}) + details = path.stat(follow_symlinks=False) + states[path.name] = { + 'phase': value.get('phase'), + 'reservation_id': reservation.get('reservation_id'), + 'bundle_id': reservation.get('bundle_id'), + 'scan_event_id': reservation.get('scan_event_id'), + 'source': reservation.get('source'), + 'platform': reservation.get('platform'), + 'target': reservation.get('target'), + 'planning_kind': planning.get('kind'), + 'sha256': sha256_file(path), 'size': details.st_size, + 'mtime_ns': details.st_mtime_ns, + } + bundles = {} + if bundle_root.exists(): + for path in sorted(bundle_root.rglob('*.trb')): + regular_file(path, 'windows_bundle', maximum=64 * 1024 * 1024) + details = path.stat(follow_symlinks=False) + bundles[path.relative_to(bundle_root).as_posix()] = { + 'sha256': sha256_file(path), 'size': details.st_size, + 'mtime_ns': details.st_mtime_ns, + } + snapshot = {'states': states, 'bundles': bundles} + if not self.durable(snapshot): + return snapshot + snapshot['work'] = self.host_tree_identity(work_root) + return snapshot + + @staticmethod + def host_tree_identity(root): + digest = hashlib.sha256() + count = 0 + active_entries = 0 + root = Path(root) + if not root.exists(): + return {'count': 0, 'active_entries': 0, 'sha256': digest.hexdigest()} + for current, directories, files in os.walk(root, followlinks=False): + directories.sort() + files.sort() + for name in directories: + path = Path(current) / name + require(not path.is_symlink() and path.is_dir(), 'windows_work_directory') + if path.relative_to(root).parts[0] != 'abandoned': + active_entries += 1 + for name in files: + path = regular_file( + Path(current) / name, 'windows_work_file', maximum=256 * 1024 * 1024, + ) + if path.relative_to(root).parts[0] != 'abandoned': + active_entries += 1 + details = path.stat(follow_symlinks=False) + digest.update(path.relative_to(root).as_posix().encode('utf-8') + b'\0') + digest.update(str(details.st_size).encode('ascii') + b'\0') + digest.update(str(details.st_mtime_ns).encode('ascii') + b'\0') + digest.update(sha256_file(path).encode('ascii') + b'\n') + count += 1 + return { + 'count': count, 'active_entries': active_entries, + 'sha256': digest.hexdigest(), + } + + def linux_snapshot(self, worker): + code, stdout, _ = self.runner.docker( + 'linux_worker_snapshot', + ['container', 'exec', worker, '/usr/local/bin/python3', + '-u', '-I', '-S', '-B', '/data/_inspect.py'], + check=False, + ) + if code: + return None + value = load_json_bytes(stdout, 'linux_worker_snapshot') + require(isinstance(value, dict), 'linux_snapshot_object') + return value + + @staticmethod + def durable(snapshot): + if not isinstance(snapshot, dict): + return False + states = snapshot.get('states') + bundles = snapshot.get('bundles') + if set(states or {}) != {'slot-0.json', 'slot-1.json'} or len(bundles or {}) != 2: + return False + bundle_ids = set() + for state in states.values(): + if ( + state.get('phase') != 'bundle_ready' + or not HEX_64.fullmatch(str(state.get('sha256') or '')) + or not re.fullmatch(r'[a-f0-9]{32,64}', str(state.get('bundle_id') or '')) + or int(state.get('reservation_id') or 0) <= 0 + ): + return False + bundle_ids.add(state['bundle_id']) + if len(bundle_ids) != 2: + return False + for path, entry in bundles.items(): + if ( + Path(path).name.removesuffix('.trb') not in bundle_ids + or not HEX_64.fullmatch(str(entry.get('sha256') or '')) + or int(entry.get('size') or 0) <= 0 + ): + return False + return True + + @staticmethod + def direct_assigned(snapshot, bundle_count): + if not isinstance(snapshot, dict): + return False + states = snapshot.get('states') + bundles = snapshot.get('bundles') + if ( + set(states or {}) != {'slot-0.json', 'slot-1.json'} + or len(bundles or {}) != bundle_count + ): + return False + expected = { + ('dockerhub', 'docker', 'docker_direct_v1', DOCKER_TARGET), + ('huggingface', 'huggingface', 'huggingface_space_v1', HUGGINGFACE_TARGET), + } + identities = set() + bundle_ids = set() + for state in states.values(): + if ( + state.get('phase') != 'assigned' + or not HEX_64.fullmatch(str(state.get('sha256') or '')) + or not re.fullmatch(r'[a-f0-9]{32,64}', str(state.get('bundle_id') or '')) + or int(state.get('reservation_id') or 0) <= 0 + or not re.fullmatch(r'[a-f0-9]{32,64}', str(state.get('scan_event_id') or '')) + ): + return False + identities.add(( + state.get('source'), state.get('platform'), + state.get('planning_kind'), state.get('target'), + )) + bundle_ids.add(state['bundle_id']) + if identities != expected or len(bundle_ids) != 2: + return False + for path, entry in bundles.items(): + if ( + Path(path).name.removesuffix('.trb') not in bundle_ids + or not HEX_64.fullmatch(str(entry.get('sha256') or '')) + or int(entry.get('size') or 0) <= 0 + ): + return False + work = snapshot.get('work') + if work is not None and int(work.get('active_entries') or 0) != 0: + return False + return True + + @staticmethod + def resolved_client_storage(snapshot): + if not isinstance(snapshot, dict) or snapshot.get('bundles'): + return False + work = snapshot.get('work') or {} + if int(work.get('active_entries') or 0) != 0: + return False + for state in (snapshot.get('states') or {}).values(): + if ( + state.get('phase') != 'claiming' + or state.get('reservation_id') is not None + or state.get('bundle_id') is not None + or state.get('scan_event_id') is not None + ): + return False + return True + + def assert_logs_safe(self, logs): + forbidden = [*self.tokens.values(), *self.device_tokens.values()] + for value in logs: + require(len(value) <= MAX_OUTPUT, 'log_output_bound') + for secret in forbidden: + require(secret.encode('ascii') not in value, 'secret_present_in_log') + + def docker_logs(self, container): + _, stdout, stderr = self.runner.docker( + 'docker_logs', ['container', 'logs', container], check=False, + ) + return stdout + (b'\n' if stdout and stderr else b'') + stderr + + def assert_restart_stable(self, phase, snapshot, expected, alive): + started = time.monotonic() + end = min(self.deadline, time.monotonic() + 3) + # The expected snapshot was captured before the real process restart. + checks = 1 + current = None + while time.monotonic() < end: + require(alive(), phase + '_restart_process_exited') + current = snapshot() + require( + isinstance(current, dict) + and current.get('states') == expected.get('states') + and current.get('bundles') == expected.get('bundles') + and int((current.get('work') or {}).get('count') or 0) + <= int((expected.get('work') or {}).get('count') or 0), + phase + '_restart_rescanned_or_rewrote_bundle', + ) + require(alive(), phase + '_restart_process_exited') + checks += 1 + time.sleep(0.25) + require( + checks >= 2 and time.monotonic() - started >= 2.5, + phase + '_restart_stability_window', + ) + return current + + def restore_and_wait_direct(self, phase, names, snapshot, alive): + self.runner.docker( + phase + '_restore', + ['container', 'exec', names['server'], '/usr/bin/touch', '/data/control/restore'], + ) + return self.wait_until( + phase + '_direct_claims', + lambda: ( + value if self.direct_assigned(value := snapshot(), 0) else None + ), + 240, + alive=lambda: self.container_running(names['server']) and alive(), + ) + + def publish_windows_direct(self, phase_root): + state_root = self.run_root / 'w' / 'TRUF' / 'RemoteWorker' + environment = dict(self.runner.host_env) + environment.update({ + 'LOCALAPPDATA': os.fspath(phase_root / 'localappdata'), + 'TEMP': os.fspath(phase_root / 'temp'), + 'TMP': os.fspath(phase_root / 'temp'), + }) + _, stdout, stderr = self.runner.execute( + 'windows_publish_direct', + [ + os.fspath(self.windows_artifact / 'runtime' / 'python' / 'python.exe'), + '-I', '-S', '-B', os.fspath(phase_root / 'publish-direct.py'), + os.fspath(self.windows_artifact), os.fspath(state_root), + os.fspath(state_root / 'bundles'), + ], + timeout=120, + env=environment, + ) + require( + stdout.strip() == b'published direct bundles: 2' and not stderr, + 'windows_direct_publish_output', + ) + + def publish_linux_direct(self, worker): + _, stdout, stderr = self.runner.docker( + 'linux_publish_direct', + [ + 'container', 'exec', worker, '/usr/local/bin/python3', + '-u', '-I', '-S', '-B', '/data/_publish_direct.py', + '/opt/truf-worker', '/data/state-base/truf/remote-worker', + '/data/client/truf/remote-worker/bundles', + ], + timeout=120, + ) + require( + stdout.strip() == b'published direct bundles: 2' and not stderr, + 'linux_direct_publish_output', + ) + + def start_windows_client(self, phase_root, port): + localappdata = self.run_root / 'w' + temporary = phase_root / 'temp' + localappdata.mkdir() + temporary.mkdir() + system_keys = ( + 'PATH', 'PATHEXT', 'SystemRoot', 'SYSTEMROOT', 'WINDIR', 'COMSPEC', + 'NUMBER_OF_PROCESSORS', 'PROCESSOR_ARCHITECTURE', + ) + environment = { + key: os.environ[key] for key in system_keys if key in os.environ + } + environment.update({ + 'LOCALAPPDATA': os.fspath(localappdata), + 'TEMP': os.fspath(temporary), 'TMP': os.fspath(temporary), + **self.git_environment( + [self.run_root / 'repositories' / 'a', self.run_root / 'repositories' / 'b'], + self.fixture / 'worker_tls_cert.pem', + ), + }) + command = [ + environment.get('COMSPEC') or 'cmd.exe', '/d', '/c', + os.fspath(self.windows_artifact / 'run-worker.cmd'), + '--server', f'https://localhost:{port}', + '--token', self.device_tokens['windows'], + '--parallelism', str(PARALLELISM), + ] + self.windows_client = WindowsClient(command, environment, phase_root / 'client.log') + self.windows_client.start() + + def windows_phase(self): + names, phase_root = self.prepare_phase_resources('windows') + _, stdout, _ = self.runner.docker( + 'windows_server_address', + ['container', 'inspect', + '--format={{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}', + names['server']], + ) + address = stdout.decode('ascii', errors='strict').strip() + require(re.fullmatch(r'(?:[0-9]{1,3}\.){3}[0-9]{1,3}', address), + 'windows_server_address') + self.loopback_proxy = LoopbackProxy( + self.runner, address, phase_root / 'loopback-proxy.log', + ) + port = self.loopback_proxy.start() + self.start_windows_client(phase_root, port) + try: + self.wait_control('windows', names['server'], 'outage.json', { + 'schema': 1, 'claim_count': 2, 'phase': 'windows', + }, 240, alive=self.windows_client.alive) + before = self.wait_until( + 'windows_durable_bundle', + lambda: (snapshot if self.durable(snapshot := self.windows_snapshot(phase_root)) else None), + 240, alive=self.windows_client.alive, + ) + self.windows_client.stop() + self.windows_client.start() + self.assert_restart_stable( + 'windows', lambda: self.windows_snapshot(phase_root), before, + self.windows_client.alive, + ) + snapshot = lambda: self.windows_snapshot(phase_root) + self.restore_and_wait_direct( + 'windows', names, snapshot, self.windows_client.alive, + ) + self.publish_windows_direct(phase_root) + direct = self.wait_until( + 'windows_direct_bundles', + lambda: ( + value if self.direct_assigned(value := snapshot(), 2) else None + ), + 120, + alive=self.windows_client.alive, + ) + self.windows_client.stop() + self.windows_client.start() + self.assert_restart_stable( + 'windows', snapshot, direct, self.windows_client.alive, + ) + self.complete_after_direct_ready('windows', names, snapshot) + return self.finish_phase('windows', names, phase_root, (before, direct)) + finally: + if self.windows_client is not None: + self.windows_client.stop() + if self.loopback_proxy is not None: + self.loopback_proxy.stop() + + def create_linux_worker(self, names): + environment = self.git_environment( + ['file:///data/repos/a', 'file:///data/repos/b'], + '/data/worker_tls_cert.pem', linux=True, + ) + arguments = [ + '--network', 'container:' + names['server'], + '--read-only', '--cap-drop', 'ALL', '--security-opt', 'no-new-privileges', + '--pids-limit', '256', '--stop-timeout', '45', + '--tmpfs', '/tmp:rw,nosuid,nodev,noexec,size=128m,mode=1777', + '--mount', f'type=volume,source={names["worker_data"]},target=/data', + ] + for key, value in environment.items(): + arguments.extend(['--env', key + '=' + value]) + arguments.extend([ + '--env', 'XDG_DATA_HOME=/data/client', '--env', 'XDG_STATE_HOME=/data/state-base', + self.args.linux_image, + '--server', 'https://localhost:8443', '--token', self.device_tokens['linux'], + '--parallelism', str(PARALLELISM), + ]) + self.create_container(names['worker'], 'linux', 'worker', arguments) + self.runner.docker('start_linux_worker', ['container', 'start', names['worker']]) + require(self.container_running(names['worker']), 'linux_worker_not_running') + + def linux_phase(self): + names, phase_root = self.prepare_phase_resources('linux') + self.create_linux_worker(names) + self.wait_control('linux', names['server'], 'outage.json', { + 'schema': 1, 'claim_count': 2, 'phase': 'linux', + }, 240, alive=lambda: self.container_running(names['worker'])) + before = self.wait_until( + 'linux_durable_bundle', + lambda: (snapshot if self.durable(snapshot := self.linux_snapshot(names['worker'])) else None), + 240, alive=lambda: self.container_running(names['worker']), + ) + worker = self.tracker.inspect_owned( + 'containers', names['worker'], self.tracker.resources['containers'][names['worker']], + ) + self.runner.docker( + 'stop_linux_worker', ['container', 'stop', '--time', '45', worker['id']], + timeout=60, + ) + require(not self.container_running(names['worker']), 'linux_worker_stop') + self.runner.docker('restart_linux_worker', ['container', 'start', names['worker']]) + self.assert_restart_stable( + 'linux', lambda: self.linux_snapshot(names['worker']), before, + lambda: self.container_running(names['worker']), + ) + snapshot = lambda: self.linux_snapshot(names['worker']) + alive = lambda: self.container_running(names['worker']) + self.restore_and_wait_direct('linux', names, snapshot, alive) + self.publish_linux_direct(names['worker']) + direct = self.wait_until( + 'linux_direct_bundles', + lambda: ( + value if self.direct_assigned(value := snapshot(), 2) else None + ), + 120, + alive=alive, + ) + worker = self.tracker.inspect_owned( + 'containers', names['worker'], self.tracker.resources['containers'][names['worker']], + ) + self.runner.docker( + 'stop_linux_direct_worker', + ['container', 'stop', '--time', '45', worker['id']], + timeout=60, + ) + require(not self.container_running(names['worker']), 'linux_direct_worker_stop') + self.runner.docker('restart_linux_direct_worker', ['container', 'start', names['worker']]) + self.assert_restart_stable('linux_direct', snapshot, direct, alive) + self.complete_after_direct_ready('linux', names, snapshot) + return self.finish_phase('linux', names, phase_root, (before, direct)) + + def complete_after_direct_ready(self, phase, names, snapshot): + self.runner.docker( + phase + '_direct_ready', + ['container', 'exec', names['server'], '/usr/bin/touch', + '/data/control/direct-ready'], + ) + + def completed(): + return self.read_control(names['server'], 'completed.json') + + try: + evidence = self.wait_until( + phase + '_completed', completed, 240, + alive=lambda: self.container_running(names['server']), + ) + except Failure: + waiting = self.read_control(names['server'], 'completion-wait.json') + if waiting is not None: + require( + set(waiting) == {'schema', 'phase', 'reason'} + and waiting['schema'] == 1 and waiting['phase'] == phase + and isinstance(waiting['reason'], str) + and re.fullmatch(r'[A-Za-z0-9 _-]{1,80}', waiting['reason']), + phase + '_completion_wait_shape', + ) + print(phase + ' completion wait: ' + waiting['reason'], flush=True) + raise + + def clean(): + value = snapshot() + if self.resolved_client_storage(value): + return value + return None + + self.wait_until(phase + '_client_cleanup', clean, 120) + self.phase_results[phase] = evidence + + def finish_phase(self, phase, names, phase_root, restart_snapshots): + if phase == 'linux': + worker = self.tracker.inspect_owned( + 'containers', names['worker'], self.tracker.resources['containers'][names['worker']], + ) + self.runner.docker( + 'final_stop_linux_worker', + ['container', 'stop', '--time', '45', worker['id']], timeout=60, + ) + final, stopped_reader = self.linux_snapshot_stopped( + names['worker'], names['worker_data'], phase, + ) + else: + self.windows_client.stop() + final = self.windows_snapshot(phase_root) + require(self.resolved_client_storage(final), phase + '_final_cleanup') + if phase == 'linux': + receipt = final.get('shutdown_receipt') + require( + isinstance(receipt, dict) + and set(receipt) == { + 'schema', 'instance_id', 'completed_at', 'exit_code', + 'drained', 'last_sequence', + } + and receipt.get('schema') == 1 + and receipt.get('drained') is True + and receipt.get('exit_code') == 0, + 'linux_clean_shutdown_receipt', + ) + logs = [] + for role in ('server_seed', 'fixture_seed', 'server'): + logs.append(self.docker_logs(names[role])) + if phase == 'linux': + logs.extend([ + self.docker_logs(names['worker_seed']), self.docker_logs(names['worker']), + self.docker_logs(stopped_reader), + ]) + else: + logs.append(self.windows_client.read_log()) + self.assert_logs_safe(logs) + return { + 'bundle_sha256': sorted( + entry['sha256'] + for snapshot in restart_snapshots + for entry in snapshot['bundles'].values() + ), + 'evidence': self.normalize_evidence(phase, self.phase_results[phase]), + } + + def linux_snapshot_stopped(self, worker, volume, phase): + reader = self.prefix + '-' + phase + '-stopped-reader' + self.create_container( + reader, phase, 'stopped-reader', + ['--network', 'none', '--read-only', + '--mount', f'type=volume,source={volume},target=/data', + '--entrypoint', '/usr/local/bin/python3', self.args.linux_image, + '-u', '-I', '-S', '-B', '/data/_inspect.py'], + ) + _, stdout, _ = self.runner.docker( + 'run_stopped_reader', ['container', 'start', '--attach', reader], timeout=60, + ) + value = load_json_bytes(stdout, 'stopped_reader') + require(isinstance(value, dict), 'stopped_reader_object') + return value, reader + + def normalize_evidence(self, phase, value): + expected_keys = { + 'schema', 'phase', 'counts', 'claim_count', 'detectors', + 'candidate_services', 'secret_hashes', 'commit_hashes', + 'receipt_count', 'source_counts', 'planning_counts', 'capacity', + } + require(isinstance(value, dict) and set(value) == expected_keys, phase + '_evidence_shape') + require( + value['schema'] == 1 and value['phase'] == phase + and value['claim_count'] == value['receipt_count'] == 4 + and value['detectors'] == ['OpenAI'] + and value['candidate_services'] == ['openai'] + and value['source_counts'] == { + 'gitlab': 2, 'dockerhub': 1, 'huggingface': 1, + } + and value['planning_counts'] == { + 'exact_git_v1': 2, + 'docker_direct_v1': 1, + 'huggingface_space_v1': 1, + }, + phase + '_evidence_identity', + ) + expected_secret_hashes = sorted( + sha256_bytes(token.encode('ascii')) for token in self.tokens.values() + ) + require(value['secret_hashes'] == expected_secret_hashes, phase + '_secret_hashes') + require( + value['commit_hashes'] == sorted(self.repository_heads.values()), + phase + '_commit_hashes', + ) + normalized = dict(value) + normalized.pop('phase') + return normalized + + def clear_run_artifacts(self): + require( + self.run_root.parent == self.root / 'build' + and re.fullmatch(r'pwe-[a-f0-9]{16}', self.run_root.name) + and self.run_root.is_dir() and not self.run_root.is_symlink(), + 'run_root_guard', + ) + + def retry_writable(function, name, _error): + details = os.lstat(name) + if stat.S_ISLNK(details.st_mode): + function(name) + return + mode = stat.S_IRUSR | stat.S_IWUSR + if stat.S_ISDIR(details.st_mode): + mode |= stat.S_IXUSR + os.chmod(name, mode) + function(name) + + for path in self.run_root.iterdir(): + if path.is_symlink() or path.is_file(): + try: + path.unlink() + except PermissionError: + require(not path.is_symlink(), 'run_artifact_symlink_permission') + os.chmod(path, stat.S_IRUSR | stat.S_IWUSR) + path.unlink() + elif path.is_dir(): + shutil.rmtree(path, onerror=retry_writable) + else: + raise Failure('run_artifact_type') + require(not any(self.run_root.iterdir()), 'run_artifact_cleanup') + + def safe_evidence(self, value): + content = json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + b'\n' + require(len(content) <= MAX_SAFE_EVIDENCE, 'safe_evidence_bound') + for secret in (*self.tokens.values(), *self.device_tokens.values()): + require(secret.encode('ascii') not in content, 'secret_present_in_evidence') + return content + + def retain_safe_evidence(self, name, value): + content = self.safe_evidence(value) + self.clear_run_artifacts() + write_bytes(self.run_root / name, content) + + def write_summary(self, windows, linux, cleanup): + summary = { + 'schema': 1, 'status': 'passed', 'run_id': self.run_id, + 'parallelism': PARALLELISM, 'targets': list(ALL_TARGETS), + 'repository_heads': self.repository_heads, + 'manifest_sha256': self.manifest_hashes, + 'image_ids': self.image_details, + 'bundle_sha256': { + 'windows': windows['bundle_sha256'], 'linux': linux['bundle_sha256'], + }, + 'normalized_evidence_sha256': sha256_bytes(json.dumps( + windows['evidence'], ensure_ascii=True, sort_keys=True, + separators=(',', ':'), + ).encode('ascii')), + 'foreign_docker_state': 'unchanged', + 'cleanup': cleanup, + } + self.retain_safe_evidence('summary.json', summary) + return summary + + def run(self): + self.preflight() + windows = self.windows_phase() + linux = self.linux_phase() + require(windows['evidence'] == linux['evidence'], 'cross_platform_evidence_mismatch') + cleanup = 'retained' if self.args.keep else 'complete' + if not self.args.keep: + self.tracker.cleanup() + else: + self.tracker.assert_foreign_unchanged() + return self.write_summary(windows, linux, cleanup) + + +def parse_args(argv=None): + parser = argparse.ArgumentParser( + description='Verify real packaged Windows and Linux workers across an API outage.', + ) + parser.add_argument('--windows-artifact', default=WINDOWS_ARTIFACT, + help='portable Windows worker directory (default: %(default)s)') + parser.add_argument('--linux-image', default=LINUX_IMAGE, + help='already-built Linux worker image (default: %(default)s)') + parser.add_argument('--test-image', default=TEST_IMAGE, + help='already-built server/test image (default: %(default)s)') + parser.add_argument('--wsl-distro', default='Ubuntu-24.04', + help='WSL distribution that owns the Docker socket (default: %(default)s)') + parser.add_argument('--timeout-seconds', type=int, default=1800, + help='aggregate subprocess and test timeout (default: %(default)s)') + parser.add_argument('--keep', action='store_true', + help='retain labelled Docker resources after a successful run') + args = parser.parse_args(argv) + if not 300 <= args.timeout_seconds <= 7200: + parser.error('--timeout-seconds must be between 300 and 7200') + for name in ('linux_image', 'test_image', 'wsl_distro'): + if not re.fullmatch(r'[A-Za-z0-9][A-Za-z0-9._:/@+-]{0,255}', getattr(args, name)): + parser.error('--' + name.replace('_', '-') + ' contains unsupported characters') + return args + + +def main(argv=None): + verifier = None + try: + verifier = Verifier(parse_args(argv)) + summary = verifier.run() + except KeyboardInterrupt as exc: + label = 'interrupted' + failure_class = type(exc).__name__ + exit_code = None + except Failure as exc: + label = str(exc) + failure_class = type(exc).__name__ + exit_code = getattr(exc, 'exit_code', None) + except Exception as exc: + label = 'unexpected_exception' + failure_class = type(exc).__name__ + exit_code = None + else: + print(json.dumps({ + 'status': summary['status'], + 'artifacts': os.fspath(verifier.run_root), + 'cleanup': summary['cleanup'], + }, ensure_ascii=True, sort_keys=True)) + return 0 + finally: + if verifier is not None and verifier.windows_client is not None: + verifier.windows_client.stop() + if verifier is not None and verifier.loopback_proxy is not None: + verifier.loopback_proxy.stop() + if verifier is not None: + verifier.runner.deadline = time.monotonic() + 720 + verifier.tracker.guarded_stop() + if not re.fullmatch(r'[a-z0-9_-]{1,160}', label): + label = 'verifier_failure' + if not re.fullmatch(r'[A-Za-z][A-Za-z0-9_]{0,79}', failure_class): + failure_class = 'Exception' + if type(exit_code) is not int or not -(2 ** 31) <= exit_code < 2 ** 31: + exit_code = None + failure = { + 'stage': label, 'class': failure_class, 'exit_code': exit_code, + } + if verifier is not None: + try: + if verifier.args.keep: + write_bytes( + verifier.run_root / 'failure.json', + verifier.safe_evidence(failure), + ) + else: + verifier.retain_safe_evidence('failure.json', failure) + except Exception: + pass + print(json.dumps(failure, ensure_ascii=True, sort_keys=True), file=sys.stderr) + return 1 + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/docker/windows_snapshot.py b/docker/windows_snapshot.py new file mode 100644 index 0000000..2d572b3 --- /dev/null +++ b/docker/windows_snapshot.py @@ -0,0 +1,880 @@ +"""Offline Windows export, never a supervisor launcher or a source migrator. + +Run only after the canonical Windows supervisor stop: + python -I -S -B docker/windows_snapshot.py capture --output D:\\truf-docker\\docker\\imports\\NAME + +Output is numeric: phase, count, bytes; failures are phase, exit code. Phases: +1 arguments, 2 source loading, 3 private output, 4 offline inventory, 5 maintenance +start, 6 database, 7 confirmed stop (count = retries), 8 tar, 9 revalidation, +10 publication. Exit 1 is a guarded failure; 124 is a client timeout. + +An unconfirmed maintenance stop deliberately RETAINS authority and retries. Do +not terminate this process to bypass that guard. There is no valid manifest +until cleanup confirms STOPPED. Killing Windows/processes can defeat any lock. +SIGINT/SIGBREAK only request cancellation while capture owns the source. They +cannot unwind the original backend's spawned-but-not-yet-bookkept launch window. + +Assumes intact original helper APIs/layout and existing credentials permitted to +dump all data, read pg_control_system(), and observe all client sessions. +""" + +import argparse +import contextlib +import ctypes +from dataclasses import dataclass +from datetime import datetime, timezone +import fnmatch +import hashlib +import importlib +import importlib.util +import json +import ntpath +import os +from pathlib import Path, PureWindowsPath +import re +import shutil +import signal +import stat +import subprocess +import sys +import tarfile +import threading +import time +from types import SimpleNamespace +from urllib.parse import unquote, urlsplit + + +SOURCE_ROOT = Path(r'D:\truf') +POSTGRES_DATA = Path(r'S:\postgres-data') +BUNDLE_ROOT = Path(r'S:\scanner-result-bundles') +IMPORTS_ROOT = Path(r'D:\truf-docker\docker\imports') +KEYCHECK_COPY = 'keychecks \u2014 \u043a\u043e\u043f\u0438\u044f' +BLOCK = 1024 * 1024 +QUERY_TIMEOUT = 30 +COUNT_TIMEOUT = 1800 +DUMP_TIMEOUT = 6 * 3600 +SESSION_TIMEOUT = 12 * 3600 +MAX_METADATA = 4 * BLOCK +ACTIVE_DIRS = ('queues', 'state', 'keychecks', 'results', 'postman_cache', 'result_spool') +EXCLUSION_POLICY = [ + 'Only explicitly reviewed source roots and mappings are selected.', + 'No physical PGDATA/WAL, PostgreSQL binaries/logs, or active control authority.', + 'No runtime/downloads, git, traces, freeze-diagnostics, or S: scanner-work.', + 'No gharchive_cache, .git, .opencode, tests, code caches, *.lock*, or *.pid.', + 'Ordinary *.log* excluded outside result/keycheck projections; scan_errors.log* retained.', + 'Windows scratch databases, temporary state and janitor cursor are archival only.', +] + + +class Failure(Exception): + def __init__(self, code=1): + self.code = code + super().__init__(code) + + +@contextlib.contextmanager +def _defer_signals(): + pending = False + previous = {} + + def request(_number, _frame): + nonlocal pending + pending = True + + def checkpoint(): + if pending: + raise Failure() + + try: + for number in (signal.SIGINT, getattr(signal, 'SIGBREAK', None)): + if number is not None: + previous[number] = signal.signal(number, request) + yield checkpoint + finally: + for number, handler in previous.items(): + signal.signal(number, handler) + + +@dataclass(frozen=True) +class File: + source: Path + size: int + fingerprint: tuple + + +@dataclass +class Inventory: + files: dict + directories: dict + exclusions: dict + + +def _fingerprint(info): + return (info.st_dev, info.st_ino, info.st_mode, info.st_nlink, info.st_size, + info.st_mtime_ns, info.st_ctime_ns, getattr(info, 'st_file_attributes', 0)) + + +def _check_type(info, directory=False): + if (getattr(info, 'st_file_attributes', 0) & 0x400 + or stat.S_ISLNK(info.st_mode) + or not (stat.S_ISDIR(info.st_mode) if directory else stat.S_ISREG(info.st_mode)) + or (not directory and info.st_nlink != 1)): + raise Failure() + + +def _check_chain(path): + # Inspect ancestors first: even lstat(child) otherwise traverses a junction. + path = Path(path).absolute() + for part in (*reversed(path.parents), path): + info = part.lstat() + if stat.S_ISLNK(info.st_mode) or getattr(info, 'st_file_attributes', 0) & 0x400: + raise Failure() + + +def _file_info(path): + _check_chain(path) + info = path.lstat() + _check_type(info) + return info + + +def _same_windows_path(left, right): + return ntpath.normcase(ntpath.normpath(str(left))) == ntpath.normcase(ntpath.normpath(str(right))) + + +def _output_path(value): + path = PureWindowsPath(value) + name = path.name + if (not path.is_absolute() or not _same_windows_path(path.parent, IMPORTS_ROOT) + or any(part in ('.', '..') for part in re.split(r'[\\/]', value)) + or not re.fullmatch(r'[A-Za-z0-9][A-Za-z0-9_.-]{0,95}', name) + or name.endswith('.') + or re.fullmatch(r'CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9]', name.split('.')[0], re.I)): + raise Failure() + return Path(str(path)) + + +def _verify_private_acl(path, security, directory): + security.reject_reparse_components(str(path)) + sddl = security._windows_private_sddl(str(path)).upper() + alias = lambda sid: 'SY' if sid == 'S-1-5-18' else sid + sid = alias(security._windows_current_user_sid().upper()) + owner = re.search(r'O:([^:()]+?)(?=[GDS]:|$)', sddl) + aces = [ace.split(';') for ace in re.findall(r'\(([^()]*)\)', sddl)] + expected = {sid, 'SY'} + if (not owner or alias(owner.group(1)) != sid or 'D:P' not in sddl + or len(aces) != len(expected)): + raise Failure() + trustees = set() + for ace in aces: + if (len(ace) != 6 or ace[:5] != ['A', 'OICI' if directory else '', 'FA', '', '']): + raise Failure() + trustees.add(alias(ace[5])) + if trustees != expected: + raise Failure() + + +def _secure_path(path, security, directory=False): + # The original public helper includes BA. Narrow its verified, empty path + # using the same no-reparse Win32 primitives, before any sensitive write. + harden = security.harden_private_directory if directory else security.harden_private_file + harden(str(path)) + sid = security._windows_current_user_sid() + trustees = ('SY',) if sid.upper() == 'S-1-5-18' else (sid, 'SY') + flags = 'OICI' if directory else '' + sddl = 'D:P' + ''.join(f'(A;{flags};FA;;;{trustee})' for trustee in trustees) + descriptor = ctypes.c_void_p() + if not security._CONVERT_SDDL(sddl, 1, ctypes.byref(descriptor), None): + raise Failure() + handle = None + try: + handle = security._CREATE_FILE(str(path), 0x60000, 7, None, 3, 0x2200000, None) + if handle == ctypes.c_void_p(-1).value: + raise Failure() + info = security._BY_HANDLE_FILE_INFORMATION() + if (not security._GET_FILE_INFORMATION(handle, ctypes.byref(info)) + or info.dwFileAttributes & 0x400 + or not security._SET_KERNEL_OBJECT_SECURITY(handle, 0x80000004, descriptor)): + raise Failure() + finally: + if handle is not None and handle != ctypes.c_void_p(-1).value: + security._CLOSE_HANDLE(handle) + security._LOCAL_FREE(descriptor) + _verify_private_acl(path, security, directory) + + +def _prepare_output(output, security): + if output.parent != IMPORTS_ROOT: + raise Failure() + _check_chain(IMPORTS_ROOT.parent) + _check_type(IMPORTS_ROOT.parent.lstat(), directory=True) + try: + IMPORTS_ROOT.mkdir() + except FileExistsError: + _check_chain(IMPORTS_ROOT) + _check_type(IMPORTS_ROOT.lstat(), directory=True) + else: + _secure_path(IMPORTS_ROOT, security, directory=True) + output.mkdir() # Never reuse, overwrite or repair an existing snapshot. + _secure_path(output, security, directory=True) + + +@contextlib.contextmanager +def _output_file(path, security): + _check_chain(path.parent) + with path.open('xb', buffering=0) as handle: + _secure_path(path, security) + info = _file_info(path) + opened = os.fstat(handle.fileno()) + if (info.st_dev, info.st_ino) != (opened.st_dev, opened.st_ino): + raise Failure() + yield handle + handle.flush() + os.fsync(handle.fileno()) + + +def _destination(path, used, directories): + parts = path.split('/') + if (not parts or any(not p or p in ('.', '..') for p in parts) + or any(ord(c) < 32 or ord(c) == 127 for c in path) + or '\\' in path or ':' in path): + raise Failure() + folded = path.casefold() + parents = {'/'.join(parts[:i]).casefold() for i in range(1, len(parts))} + if folded in used or folded in directories or parents.intersection(used): + raise Failure() + used.add(folded) + directories.update(parents) + + +def _excluded(relative, control=False): + parts = relative.casefold().split('/') + name = parts[-1] + if any(p in {'.git', '.opencode', 'tests', '__pycache__', '.pytest_cache', + '.mypy_cache', '.ruff_cache', 'node_modules'} for p in parts): + return 'code_cache_or_unreviewed_code' + if 'gharchive_cache' in parts: + return 'downloaded_archives' + if any(fnmatch.fnmatchcase(p, '*.lock*') or p.endswith('.pid') for p in parts): + return 'locks_or_pids' + if name.endswith(('.pyc', '.pyo')): + return 'code_cache_or_unreviewed_code' + if control and any(re.search( + r'supervisor|instance|token|authority|manifest|handshake|capability|(?:^|[._-])(?:pid|lock)(?:[._-]|$)', + p) for p in parts[2:]): + return 'control_authority' + projection = any(p in ('results', 'keychecks', KEYCHECK_COPY.casefold()) + or p.startswith(('found_secrets.jsonl', 'scan_results.jsonl', 'scan_errors.log')) for p in parts) + config_backup = len(parts) == 2 and parts[0] == 'app' and name.startswith(('config.yaml.', 'secrets.yaml.')) + if (fnmatch.fnmatchcase(name, '*.log*') and not projection + and not config_backup + and not fnmatch.fnmatchcase(name, 'scan_errors.log*') + and 'publication-ledger.sqlite3' not in name): + return 'ordinary_logs' + return None + + +def _inventory(): + files, watched, excluded, used, parents = {}, {}, {}, set(), set() + + def visit(path, relative, target, control=False, expect_directory=None): + reason = _excluded(relative, control) + if reason: + excluded[reason] = excluded.get(reason, 0) + 1 + return + try: + _check_chain(path) + info = path.lstat() + except FileNotFoundError: + watched[str(path)] = None + return + directory = stat.S_ISDIR(info.st_mode) + _check_type(info, directory=directory) + if expect_directory is not None and directory != expect_directory: + raise Failure() + if directory: + watched[str(path)] = _fingerprint(info) + with os.scandir(path) as entries: + names = sorted(entry.name for entry in entries) + for name in names: + visit(path / name, relative + '/' + name, target + '/' + name, control) + if _fingerprint(path.lstat()) != watched[str(path)]: + raise Failure() + return + if control and not path.name.casefold().endswith('.json'): + excluded['non_report_control'] = excluded.get('non_report_control', 0) + 1 + return + if relative.casefold().startswith('runtime/state/'): + scratch = relative.casefold().split('/')[2:] + if any(fnmatch.fnmatchcase(p, 'scan_limiter*.db*') + or fnmatch.fnmatchcase(p, '*.tmp*') or p == 'janitor.cursor.json' for p in scratch): + target = 'windows-archive/' + relative + _destination(target, used, parents) + files[target] = File(path, info.st_size, _fingerprint(info)) + + for name in ACTIVE_DIRS: + visit(SOURCE_ROOT / 'runtime' / name, 'runtime/' + name, 'runtime-linux/' + name, + expect_directory=True) + visit(SOURCE_ROOT / 'runtime/proxy.txt', 'runtime/proxy.txt', 'runtime-linux/proxy.txt', + expect_directory=False) + for name in ('secrets.yaml', 'trufflehog-custom-detectors.yaml'): + visit(SOURCE_ROOT / 'app' / name, 'app/' + name, 'config/' + name, expect_directory=False) + for relative in ('state', 'runtime/backups', 'runtime/imports', 'runtime/' + KEYCHECK_COPY): + visit(SOURCE_ROOT / relative, relative, 'windows-archive/' + relative, expect_directory=True) + for relative in ( + 'app/config.yaml', 'app/.streamlit/config.toml', '.env.postgres', 'docker-compose.postgres.yml', + 'runner_state.json', + 'runtime/keychecks.7z', 'runtime/orkey.txt', 'runtime/check-openrouter-keys.ps1', + ): + visit(SOURCE_ROOT / relative, relative, 'windows-archive/' + relative, expect_directory=False) + for folder, patterns in ( + ('', ('checked_*.txt', 'todo_*.txt', 'scanner.db*', 'found_secrets.jsonl*', + 'scan_results.jsonl*', 'scan_errors.log*', '*.publication-ledger.sqlite3*')), + ('app', ('config.yaml.*', 'secrets.yaml.*', 'scanner.db*')), + ('runtime', ('*.md',)), + ): + path = SOURCE_ROOT / folder + _check_chain(path) + info = path.lstat() + _check_type(info, directory=True) + watched[str(path)] = _fingerprint(info) + with os.scandir(path) as entries: + names = sorted(entry.name for entry in entries + if any(fnmatch.fnmatchcase(entry.name.casefold(), p) for p in patterns)) + for name in names: + relative = folder + '/' + name if folder else name + projection_family = not folder and name.casefold().startswith( + ('found_secrets.jsonl', 'scan_results.jsonl', 'scan_errors.log')) + visit(path / name, relative, 'windows-archive/' + relative, + expect_directory=None if projection_family else False) + visit(SOURCE_ROOT / 'runtime/control', 'runtime/control', 'windows-archive/runtime/control', + control=True, expect_directory=True) + _check_chain(BUNDLE_ROOT) + visit(BUNDLE_ROOT, 'scanner-result-bundles', 'scanner-result-bundles', expect_directory=True) + return Inventory(files, watched, excluded) + + +class HashWriter: + def __init__(self, handle): + self.handle = handle + self.digest = hashlib.sha256() + self.size = 0 + + def write(self, block): + if self.handle.write(block) != len(block): + raise Failure() + self.digest.update(block) + self.size += len(block) + return len(block) + + def metadata(self): + return {'bytes': self.size, 'sha256': self.digest.hexdigest()} + + +class HashReader: + def __init__(self, handle): + self.handle = handle + self.digest = hashlib.sha256() + self.size = 0 + + def read(self, size): + block = self.handle.read(size) + self.digest.update(block) + self.size += len(block) + return block + + +def _write_tar(output, security, inventory, report): + manifest_files = [] + with _output_file(output / 'files.tar', security) as handle: + writer = HashWriter(handle) + with tarfile.open(fileobj=writer, mode='w|', format=tarfile.PAX_FORMAT, copybufsize=BLOCK) as archive: + for name, entry in sorted(inventory.files.items()): + if _fingerprint(_file_info(entry.source)) != entry.fingerprint: + raise Failure() + with entry.source.open('rb', buffering=0) as source: + before = _fingerprint(os.fstat(source.fileno())) + # Windows Python 3.12 stat/fstat use different ctime bases. + # Compare ctime within each API, not across the two APIs. + if before[:6] + before[7:] != entry.fingerprint[:6] + entry.fingerprint[7:]: + raise Failure() + reader = HashReader(source) + info = tarfile.TarInfo(name) + info.size, info.mode, info.mtime = entry.size, 0o600, 0 + archive.addfile(info, reader) + if (reader.size != entry.size + or _fingerprint(os.fstat(source.fileno())) != before): + raise Failure() + if _fingerprint(_file_info(entry.source)) != entry.fingerprint: + raise Failure() + manifest_files.append({'path': name, 'size': entry.size, 'sha256': reader.digest.hexdigest()}) + report(8, len(manifest_files), writer.size) + metadata = writer.metadata() + return manifest_files, metadata + + +@contextlib.contextmanager +def _silence(): + with open(os.devnull, 'w', encoding='utf-8') as sink: + with contextlib.redirect_stdout(sink), contextlib.redirect_stderr(sink): + yield + + +def _load_source(): + app = SOURCE_ROOT / 'app' + for name in ('child_bootstrap.py', 'postgres_runtime.py', 'runtime_security.py'): + _file_info(app / name) + spec = importlib.util.spec_from_file_location('_snapshot_child_bootstrap', app / 'child_bootstrap.py') + bootstrap = importlib.util.module_from_spec(spec) + spec.loader.exec_module(bootstrap) + bootstrap._enable_dependency_paths('postgres-runtime') + sys.path.insert(0, str(app)) + pg = importlib.import_module('postgres_runtime') + security = importlib.import_module('runtime_security') + for module in (pg, security): + if not _same_windows_path(module.__file__, app / (module.__name__ + '.py')): + raise Failure() + # The original loader does not overwrite inherited credentials. Remove all + # connection/path overrides first, so only the original .env can choose them. + for key in list(os.environ): + if key.upper().startswith(('PG', 'TRUF_', 'SCANNER_', 'TRUFFLEHOG_')) or key.upper() == 'DATABASE_URL': + os.environ.pop(key, None) + config_path, env_path = app / 'config.yaml', SOURCE_ROOT / '.env.postgres' + inputs = {p: _fingerprint(_file_info(p)) for p in (config_path, env_path)} + config = pg._load_config(str(config_path)) + expected = {'root_dir': SOURCE_ROOT, 'project_dir': app, 'runtime_dir': SOURCE_ROOT / 'runtime', + 'postgres_data_dir': POSTGRES_DATA, 'result_bundle_dir': BUNDLE_ROOT} + for key, path in expected.items(): + if not _same_windows_path(config.get('global', {}).get(key, ''), path): + raise Failure() + for key, name in (('queue_dir', 'queues'), ('state_dir', 'state'), ('keycheck_dir', 'keychecks'), + ('results_dir', 'results'), ('postman_cache_dir', 'postman_cache'), + ('result_spool_dir', 'result_spool'), ('control_dir', 'control'), ('log_dir', 'logs')): + value = config.get('global', {}).get(key) + if value and not _same_windows_path(value, SOURCE_ROOT / 'runtime' / name): + raise Failure() + paths = pg.postgres_runtime_paths(config) + if (not _same_windows_path(paths['data_dir'], POSTGRES_DATA) + or not _same_windows_path(paths['postgres_dir'], SOURCE_ROOT / 'runtime/postgres')): + raise Failure() + security.preflight_lifecycle_paths(str(config_path), config) + loaded = pg.load_postgres_environment(str(config_path), config) + if not _same_windows_path(loaded or '', env_path): + raise Failure() + dsn = pg.canonical_database_url() + if not dsn: + raise Failure() + identity_path = SOURCE_ROOT / 'runtime/postgres/cluster_identity.json' + inputs[identity_path] = _fingerprint(_file_info(identity_path)) + return SimpleNamespace(pg=pg, security=security, config=config, dsn=dsn, inputs=inputs) + + +def _supervisor_absent(source): + supervisor = source.config.get('supervisor', {}) + paths = {SOURCE_ROOT / 'runtime/control/supervisor.instance.json', + SOURCE_ROOT / 'runtime/logs/supervisor.instance.json', + SOURCE_ROOT / 'runtime/logs/supervisor.pid'} + for key in ('instance_file',): + if supervisor.get(key): + paths.add(Path(supervisor[key])) + for key in ('control_dir', 'log_dir'): + if supervisor.get(key): + paths.add(Path(supervisor[key]) / 'supervisor.instance.json') + if any(os.path.lexists(path) for path in paths): + raise Failure() + + +def _validate_identity(source, identity): + values = source.pg.configured_cluster_values() + parsed = urlsplit(source.dsn) + if (identity['pg_major'] != 16 or not str(identity['system_identifier']).isdigit() + or not _same_windows_path(identity['data_directory'], POSTGRES_DATA) + or any(identity[k] != values[k] for k in ('database', 'user', 'port')) + or parsed.scheme not in ('postgresql', 'postgres') or parsed.hostname != '127.0.0.1' + or parsed.port != identity['port'] or unquote(parsed.username or '') != identity['user'] + or unquote(parsed.path[1:]) != identity['database'] or parsed.password is None + or parsed.query or parsed.fragment): + raise Failure() + + +def _client_environment(dsn): + parsed = urlsplit(dsn) + env = {key: value for key, value in os.environ.items() + if key.upper() in {'SYSTEMROOT', 'WINDIR', 'SYSTEMDRIVE', 'TEMP', 'TMP'}} + env.update(PGHOST='127.0.0.1', PGHOSTADDR='127.0.0.1', PGPORT=str(parsed.port), + PGDATABASE=unquote(parsed.path[1:]), PGUSER=unquote(parsed.username or ''), + PGPASSWORD=unquote(parsed.password or ''), PGPASSFILE=os.devnull, + PGSERVICEFILE=os.devnull, PGSSLMODE='disable', PGGSSENCMODE='disable', + PGCONNECT_TIMEOUT='5', PGCLIENTENCODING='UTF8', PGAPPNAME='truf-windows-snapshot', + PGOPTIONS=f'-c default_transaction_read_only=on -c statement_timeout={COUNT_TIMEOUT * 1000} ' + '-c lock_timeout=10000 -c idle_in_transaction_session_timeout=0 ' + '-c search_path=pg_catalog -c row_security=off', + LC_ALL='C', LANG='C') + return env + + +@contextlib.contextmanager +def _deadline(process, seconds): + expired = threading.Event() + + def expire(): + expired.set() + try: + process.kill() + except OSError: + pass + + timer = threading.Timer(seconds, expire) + timer.daemon = True + timer.start() + try: + yield + except BaseException: + if expired.is_set(): + raise Failure(124) from None + raise + else: + if expired.is_set(): + raise Failure(124) + finally: + timer.cancel() + timer.join() + + +@contextlib.contextmanager +def _client(command, env, timeout, interactive=False): + process = subprocess.Popen(command, stdin=subprocess.PIPE if interactive else subprocess.DEVNULL, + stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, env=env, + creationflags=0x08000000, close_fds=True, bufsize=0) + try: + with _deadline(process, timeout): + yield process + if process.stdin is not None: + process.stdin.close() + if process.wait(timeout=10) != 0: + raise Failure() + finally: + if process.poll() is None: + process.kill() + process.wait(timeout=10) + if process.stdin is not None: + process.stdin.close() + process.stdout.close() + + +def _query(process, sql, timeout=QUERY_TIMEOUT): + with _deadline(process, timeout): + payload = (sql + '\n').encode('utf-8') + if process.stdin.write(payload) != len(payload): + raise Failure() + process.stdin.flush() + line = process.stdout.readline(MAX_METADATA + 1) + if not line.endswith(b'\n') or len(line) > MAX_METADATA: + raise Failure() + try: + return json.loads(line) + except (ValueError, UnicodeError): + raise Failure() from None + + +OTHER_CLIENTS = """(SELECT count(*) FROM pg_catalog.pg_stat_activity + WHERE backend_type = 'client backend' AND pid <> pg_catalog.pg_backend_pid())""" +DATABASE_METADATA = """BEGIN ISOLATION LEVEL REPEATABLE READ READ ONLY; +SELECT pg_catalog.json_build_object( + 'version_num', current_setting('server_version_num')::integer, + 'system_identifier', (SELECT system_identifier::text FROM pg_catalog.pg_control_system()), + 'database_name', current_database(), 'user_name', current_user, + 'port', current_setting('port')::integer, 'data_directory', current_setting('data_directory'), + 'in_recovery', pg_is_in_recovery(), 'read_only', current_setting('transaction_read_only'), + 'all_sessions_visible', (SELECT rolsuper FROM pg_catalog.pg_roles WHERE rolname = current_user) + OR pg_has_role(current_user, 'pg_read_all_stats', 'MEMBER'), + 'snapshot', pg_export_snapshot(), 'database_bytes', pg_database_size(current_database()), + 'tables', (SELECT COALESCE(json_agg(c.relname ORDER BY c.relname), '[]'::json) + FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = 'public' AND c.relkind = 'r'), + 'sequences', (SELECT COALESCE(jsonb_agg(jsonb_build_array(n.nspname, c.relname) + ORDER BY n.nspname, c.relname), '[]'::jsonb) + FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind = 'S' AND n.nspname <> 'information_schema' AND n.nspname !~ '^pg_'), + 'other_clients', """ + OTHER_CLIENTS + ');' + + +def _validate_database(metadata, identity): + if (type(metadata.get('version_num')) is not int or metadata['version_num'] // 10000 != 16 + or metadata.get('system_identifier') != identity['system_identifier'] + or metadata.get('database_name') != identity['database'] + or metadata.get('user_name') != identity['user'] or type(metadata.get('port')) is not int + or metadata['port'] != identity['port'] + or not _same_windows_path(metadata.get('data_directory', ''), POSTGRES_DATA) + or metadata.get('in_recovery') is not False or metadata.get('read_only') != 'on' + or metadata.get('all_sessions_visible') is not True + or type(metadata.get('other_clients')) is not int or metadata['other_clients'] != 0 + or not re.fullmatch(r'[0-9A-Fa-f]+-[0-9A-Fa-f]+-[0-9]+', metadata.get('snapshot', '')) + or type(metadata.get('database_bytes')) is not int or metadata['database_bytes'] < 0): + raise Failure() + tables = metadata.get('tables') + if (not isinstance(tables, list) or any(not isinstance(t, str) or not t or '\0' in t for t in tables) + or len(tables) != len(set(tables))): + raise Failure() + sequences = metadata.get('sequences') + if (not isinstance(sequences, list) + or any(not isinstance(pair, list) or len(pair) != 2 + or any(not isinstance(name, str) or not name or '\0' in name for name in pair) + for pair in sequences) + or len(sequences) != len({tuple(pair) for pair in sequences})): + raise Failure() + + +def _sequence_states(process, sequences): + states = {} + for schema, name in sequences: + quoted = '.'.join('"' + part.replace('"', '""') + '"' for part in (schema, name)) + value = _query(process, "SELECT pg_catalog.json_build_object('last_value', last_value, " + "'is_called', is_called) FROM " + quoted + ';') + if (not isinstance(value, dict) or set(value) != {'last_value', 'is_called'} + or type(value['last_value']) is not int or type(value['is_called']) is not bool): + raise Failure() + states.setdefault(schema, {})[name] = value + return states + + +def _no_other_clients(process): + # Activity views cache within a transaction; explicitly refresh before checking. + _query(process, "SELECT json_build_object('cleared', pg_stat_clear_snapshot() IS NULL);") + count = _query(process, 'SELECT ' + OTHER_CLIENTS + ';') + if type(count) is not int or count != 0: + raise Failure() + + +def _capture_database(source, identity, output, inventory, report): + report(6, 0, 0) + env = _client_environment(source.dsn) + binaries = SOURCE_ROOT / 'runtime/postgres/pgsql/bin' + for name in ('psql', 'pg_dump'): + path = binaries / (name + '.exe') + before = _fingerprint(_file_info(path)) + result = subprocess.run([str(path), '--version'], stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, timeout=15, env=env, creationflags=0x08000000, + close_fds=True) + if (result.returncode != 0 or not re.fullmatch( + rb'(?:psql|pg_dump) \(PostgreSQL\) 16(?:\.[0-9]+)*(?: \([^\r\n]*\))?\s*', result.stdout) + or _fingerprint(_file_info(path)) != before): + raise Failure() + source.inputs[path] = before + command = [str(binaries / 'psql.exe'), '-X', '-q', '-A', '-t', '-w', '-v', 'ON_ERROR_STOP=1', '-f', '-'] + with _client(command, env, SESSION_TIMEOUT, interactive=True) as process: + metadata = _query(process, DATABASE_METADATA) + _validate_database(metadata, identity) + file_bytes = sum(entry.size for entry in inventory.files.values()) + if shutil.disk_usage(output).free < file_bytes + metadata['database_bytes'] + 512 * BLOCK: + raise Failure() + counts = {} + for name in metadata['tables']: + quoted = '"' + name.replace('"', '""') + '"' + count = _query(process, 'SELECT count(*) FROM ONLY "public".' + quoted + ';', COUNT_TIMEOUT) + if type(count) is not int or count < 0: + raise Failure() + counts[name] = count + report(6, len(counts), 0) + _no_other_clients(process) + # Sequences are not MVCC-isolated, even in this exported-snapshot session. + # With the source stopped, require their values to stay fixed across dump. + sequences = _sequence_states(process, metadata['sequences']) + dump = [str(binaries / 'pg_dump.exe'), '--format=custom', '--no-owner', '--no-acl', + '--no-tablespaces', '--compress=1', '--no-password', '--lock-wait-timeout=10s', + '--snapshot=' + metadata['snapshot']] + dump_env = dict(env, PGOPTIONS=env['PGOPTIONS'].replace( + f'statement_timeout={COUNT_TIMEOUT * 1000}', f'statement_timeout={DUMP_TIMEOUT * 1000}')) + with _output_file(output / 'database.dump', source.security) as handle: + writer = HashWriter(handle) + prefix = b'' + with _client(dump, dump_env, DUMP_TIMEOUT) as dumping: + while True: + block = dumping.stdout.read(BLOCK) + if not block: + break + if len(prefix) < 5: + prefix = (prefix + block)[:5] + writer.write(block) + if prefix != b'PGDMP' or writer.size <= 5: + raise Failure() + dump_metadata = writer.metadata() + _no_other_clients(process) + if _sequence_states(process, metadata['sequences']) != sequences: + raise Failure() + report(6, len(counts), dump_metadata['bytes']) + return {key: metadata[key] for key in ('version_num', 'system_identifier', 'database_name', + 'user_name', 'port', 'data_directory', 'database_bytes')} | { + 'table_counts': counts, 'table_count_mode': 'ONLY public ordinary tables; shared exported snapshot', + 'sequence_states': sequences, 'sequence_count': len(metadata['sequences']), + 'sequence_state_mode': 'Non-system schemas; non-MVCC values checked unchanged across dump in export session', + 'schema_migration_counts': {name: counts[name] for name in ('runtime_schema_migrations', 'schema_migrations') + if name in counts}, **dump_metadata} + + +def _stop_confirmed(source, backend, report): + retries = 0 + while True: + try: + with _silence(): + result = source.pg.maintenance_stop(source.config, backend=backend) + if not result.completed or not result.stopped or backend.probe().kind != source.pg.ProbeKind.STOPPED: + raise Failure() + return + except BaseException: + # Even Ctrl-C must not release authority over a possibly live source. + retries += 1 + try: + report(7, retries, 0) + time.sleep(2) + except BaseException: + pass + + +def _unchanged_inputs(source): + for path, fingerprint in source.inputs.items(): + if _fingerprint(_file_info(path)) != fingerprint: + raise Failure() + + +def _publish_manifest(output, security, manifest): + temporary = output / 'manifest.json.partial' + with _output_file(temporary, security) as handle: + writer = HashWriter(handle) + for chunk in json.JSONEncoder(ensure_ascii=True, sort_keys=True, indent=2).iterencode(manifest): + writer.write(chunk.encode('utf-8')) + writer.write(b'\n') + # Windows rename refuses an existing destination. A partial JSON is not valid + # snapshot authority, even when all preceding large files were completed. + os.rename(temporary, output / 'manifest.json') + + +def capture(output, report): + with _defer_signals() as checkpoint: + def progress(number, count, size): + if number != 7: + checkpoint() + report(number, count, size) + + _capture(output, progress, checkpoint) + + +def _capture(output, report, checkpoint): + report(2, 0, 0) + with _silence(): + source = _load_source() + report(3, 0, 0) + with _silence(): + _prepare_output(output, source.security) + authority = source.security.ClusterAuthorityLock(source.config, endpoint_dsn=source.dsn) + authority.acquire() + backend, attempted, manifest, published = None, False, None, False + try: + report(4, 0, 0) + with _silence(): + _supervisor_absent(source) + identity = source.pg.verify_cluster_identity(source.config) + _validate_identity(source, identity) + backend = source.pg.PostgresBackend(source.config) + if backend.probe().kind != source.pg.ProbeKind.STOPPED: + raise Failure() + inventory = _inventory() + _unchanged_inputs(source) + report(4, len(inventory.files), sum(entry.size for entry in inventory.files.values())) + try: + report(5, 0, 0) + with _silence(): + _supervisor_absent(source) + if backend.probe().kind != source.pg.ProbeKind.STOPPED: + raise Failure() + checkpoint() + attempted = True + # Never check cancellation inside the original start helper: + # Popen precedes _accepted_start_at_monotonic. Its non-raising + # signal fence lets that bookkeeping and close() finish first. + if source.pg.maintenance_start(source.config, backend=backend).kind != source.pg.ProbeKind.READY: + raise Failure() + checkpoint() + database = _capture_database(source, identity, output, inventory, report) + finally: + if attempted: + report(7, 0, 0) + _stop_confirmed(source, backend, report) + checkpoint() + files, archive = _write_tar(output, source.security, inventory, report) + report(9, len(files), archive['bytes']) + manifest = { + 'format': 'truf-windows-snapshot-v1', 'created_at': datetime.now(timezone.utc).isoformat(), + 'database': database, 'files': files, 'archive': archive, + 'source': {'root': str(SOURCE_ROOT), 'postgres_data_dir': str(POSTGRES_DATA), + 'supervisor_stopped': True, 'postgres_stopped': True}, + 'exclusions': {'policy': EXCLUSION_POLICY, 'observed_entries': inventory.exclusions}, + 'counts': {'files': len(files), 'file_bytes': sum(entry['size'] for entry in files), + 'active_files': sum(not entry['path'].startswith('windows-archive/') for entry in files), + 'archival_files': sum(entry['path'].startswith('windows-archive/') for entry in files), + 'public_tables': len(database['table_counts']), 'sequences': database['sequence_count']}, + } + finally: + # Do not use the lock's __exit__: an unconfirmed stop must retain it. + if attempted: + _stop_confirmed(source, backend, report) + try: + try: + if manifest is not None: + checkpoint() + with _silence(): + _supervisor_absent(source) + _unchanged_inputs(source) + _verify_private_acl(output, source.security, directory=True) + if _inventory() != inventory: + raise Failure() + report(10, len(manifest['files']), manifest['archive']['bytes']) + _publish_manifest(output, source.security, manifest) + published = True + finally: + with _silence(): + try: + if backend is not None: + backend.close() + finally: + authority.release() + checkpoint() + except BaseException: + if published: + (output / 'manifest.json').unlink() + raise + + +def main(argv=None): + phase = 1 + + def report(number, count=0, size=0): + nonlocal phase + phase = number + print(number, count, size, flush=True) + + environment, paths = os.environ.copy(), sys.path[:] + try: + if os.name != 'nt' or not (sys.flags.isolated and sys.flags.no_site and sys.dont_write_bytecode): + raise Failure() + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('action', choices=('capture',)) + parser.add_argument('--output', required=True) + with _silence(): + args = parser.parse_args(argv) + output = _output_path(args.output) + capture(output, report) + return 0 + except BaseException as exc: + timed_out = isinstance(exc, subprocess.TimeoutExpired) or isinstance(exc, Failure) and exc.code == 124 + code = 124 if timed_out else 1 + print(phase, code, file=sys.stderr, flush=True) + return code + finally: + os.environ.clear() + os.environ.update(environment) + sys.path[:] = paths + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/docker/worker-package-pins.json b/docker/worker-package-pins.json new file mode 100644 index 0000000..f72ad8f --- /dev/null +++ b/docker/worker-package-pins.json @@ -0,0 +1,43 @@ +{ + "schema": 1, + "linux": { + "aarch64": { + "python_image_manifest": "sha256:d04f49f5882f49a3b91f874e75e19f0c265f7222da8659741a9d7eab148f22a9", + "trufflehog_archive_sha256": "7e65e771d2a247964056aa5edba0f8ae3945895e5dce867fe0ffbc7b0128239a" + }, + "ca_certificates": "20250419~deb12u1", + "debian_snapshot": "20260914T000000Z", + "git": "1:2.39.5-0+deb12u3", + "tini": "0.19.0-1+b3", + "trufflehog_version": "3.97.4", + "x86_64": { + "python_image_manifest": "sha256:9c47360a2a0355e2da18516d0b1c2126ec22c195d2185e97347c9d98398c5bef", + "trufflehog_archive_bytes": 34970205, + "trufflehog_archive_sha256": "dc24007c2f233bd61c05beabeb44aa27ea9b43288166279209abe0458c5ce76b" + } + }, + "python_image": "python:3.12-slim-bookworm@sha256:782412e85d0f0984994c290652577d4018aff08145c85b262bb63dc0c7522254", + "python_version": "3.12.14", + "windows": { + "x86_64": { + "git": { + "archive_bytes": 47241394, + "archive_sha256": "50b04b55425b5c465d076cdb184f63a0cd0f86f6ec8bb4d5860114a713d2c29a", + "url": "https://github.com/git-for-windows/git/releases/download/v2.47.1.windows.1/MinGit-2.47.1-64-bit.zip", + "version": "2.47.1.windows.1" + }, + "python": { + "archive_bytes": 11133606, + "archive_sha256": "4acbed6dd1c744b0376e3b1cf57ce906f9dc9e95e68824584c8099a63025a3c3", + "url": "https://www.python.org/ftp/python/3.12.10/python-3.12.10-embed-amd64.zip", + "version": "3.12.10" + }, + "trufflehog": { + "archive_bytes": 73316636, + "archive_sha256": "6ce9a957ac62bfb19463048333d9e8481327dbbf5bdc0c43f5ab5327b9631fb9", + "url": "https://github.com/trufflesecurity/trufflehog/releases/download/v3.97.4/trufflehog_3.97.4_windows_amd64.tar.gz", + "version": "3.97.4" + } + } + } +} diff --git a/docs/defect-dockerhub-discovery-retry-null-type-2026-09-25.md b/docs/defect-dockerhub-discovery-retry-null-type-2026-09-25.md new file mode 100644 index 0000000..9b208ef --- /dev/null +++ b/docs/defect-dockerhub-discovery-retry-null-type-2026-09-25.md @@ -0,0 +1,78 @@ +# DockerHub discovery retry coalescing fails for an unbounded retry time + +## Status + +Open in the accepted live server runtime. Reproduced from the complete DockerHub +producer state and PostgreSQL authority on 2026-09-25. A minimal source fix and +PostgreSQL regression coverage have been added locally but have not been deployed +to the live runtime during the worker artifact validation. + +## Impact + +When DockerHub discovery cannot acquire a search page and delegates work without +an `available_after` timestamp, an already pending retry row cannot be coalesced. +The producer reports a generic `DockerHub discovery retry delegation failed`, the +source cycle fails, and the configured query does not advance. This can keep the +managed DockerHub producer in a restart loop while other sources remain healthy. + +The complete producer history contained 265 failed DockerHub cycles with this +masked message. A protected diagnostic cycle using the captured zero-available-auth +state reproduced the original database exception exactly. + +## Raw evidence + +- Producer log: `build/live-trace-20260925/dockerhub-discovery.log` +- Captured runner state: `build/live-trace-20260925/runner_state_dockerhub.json` +- Complete retry/pass/cycle/lock capture: + `build/live-trace-20260925/dockerhub-discovery-db-raw.json` +- Unmasked exception: + `build/live-trace-20260925/dockerhub-unmasked-cycle-failure.json` + +The unmasked exception is PostgreSQL `IndeterminateDatatype`, SQLSTATE `42P18`: +`could not determine data type of parameter $1`. It originates in +`ScannerDB.enqueue_discovery_retry()` while coalescing an existing `pending` row. + +## Root cause + +`app/scanner_db.py` used an untyped nullable placeholder in the PostgreSQL +expression: + +```sql +WHEN available_after IS NULL OR ? IS NULL THEN NULL +``` + +When `available_after=None`, PostgreSQL has no typed expression from which it can +infer the placeholder type. SQLite accepts the same query, so existing SQLite +coalescing coverage did not expose the problem. Existing PostgreSQL integration +coverage inserted a retry but did not coalesce the same row with a null retry time. + +The outer discovery helper intentionally replaced the original database exception +with a generic delegation error, which obscured the SQLSTATE in normal managed logs. + +## Local correction + +The placeholder is now explicitly typed as the schema's text timestamp form: + +```sql +WHEN available_after IS NULL OR CAST(? AS TEXT) IS NULL THEN NULL +``` + +`tests/test_pipeline_postgres_integration.py` now coalesces the same retry with no +`available_after` value and checks that the existing row is reused. + +The corrected statement was also executed against the live PostgreSQL schema in a +transaction and rolled back. It matched one pending row and completed without an +exception. Local focused results: + +- DockerHub incremental discovery tests: 27 passed. +- SQLite retry lifecycle regression: 1 passed. +- PostgreSQL integration test: skipped locally because no disposable PostgreSQL DSN + was configured; the corrected SQL shape was verified transactionally against the + live schema without persisting a change. + +## Operational note + +The configured DockerHub credential pool was independently exhausted at capture +time: ten entries were invalid and the remaining entry was rate-limited. Correcting +retry coalescing preserves the failed work and stops this database error, but it does +not make an unavailable credential pool healthy. diff --git a/docs/defect-terminal-status-scan-deadline-readback-2026-09-25.md b/docs/defect-terminal-status-scan-deadline-readback-2026-09-25.md new file mode 100644 index 0000000..32225f1 --- /dev/null +++ b/docs/defect-terminal-status-scan-deadline-readback-2026-09-25.md @@ -0,0 +1,50 @@ +# Terminal assignment status loses the durable scan deadline + +Status: open and reproducible on the live `sec` validation cohort. + +## Impact + +`GET /api/v1/worker/assignments/{reservation_id}` can return +`deadlines.scan_deadline_at: null` for a resolved assignment even though the +durable terminal receipt contains the concrete scan deadline. The same response +labels the deadline set as immutable, so replay no longer faithfully exposes the +receipt that was committed at bundle acceptance. + +Receipt, payload, scan-event, diagnostic, and reservation identities remain +correct. The loss is limited to terminal status readback of the scan deadline. + +## Live evidence + +The accepted Linux assignment used for this check had: + +- a durable `bundle_accepted` receipt with a concrete `scan_deadline_at`; +- a later persisted `awaiting_receipt` progress event with + `scan_deadline_at: null`; +- a successful authenticated status response whose terminal receipt fields all + matched PostgreSQL, except that `scan_deadline_at` had become null. + +The complete API response is retained locally as +`D:\truf\worker-linux-terminal-status-raw.json` with SHA-256 +`0954e40b4ccb6921bbd572abb3d4898ee56e2955345de21ed7df5f85177536d1`. +The durable receipt is retained in +`build/live-trace-20260925/raw-evidence-expanded.json`. + +## Root cause + +`ScannerDB.remote_assignment_status()` loads the durable receipt and then calls +`result.update(observability)` (`app/scanner_db.py` around lines 16707-16710). +The observability object reconstructs its complete `deadlines` object from the +latest progress row. Its `scan_deadline_at` comes only from +`event.get('scan_deadline_at')` (`app/scanner_db.py` around lines 16588-16605). + +Consequently, a later progress event with a null scan deadline replaces the +entire deadline object stored in the terminal receipt. This is a readback merge +problem; the persisted receipt itself is unchanged and correct. + +## Expected correction + +For resolved assignments, preserve the durable receipt's immutable deadline +object while overlaying only live observability fields such as latest progress, +known reason, and diagnostic authority. Add PostgreSQL regression coverage where +the accepted receipt has a concrete scan deadline and a later progress event has +a null deadline. diff --git a/docs/defect-windows-scan-timestamps-utc-2026-09-25.md b/docs/defect-windows-scan-timestamps-utc-2026-09-25.md new file mode 100644 index 0000000..af34ac5 --- /dev/null +++ b/docs/defect-windows-scan-timestamps-utc-2026-09-25.md @@ -0,0 +1,72 @@ +# Defect: Windows scan timestamps lose their UTC offset + +## Status + +Open and reproducible in the accepted Windows worker package validated on +2026-09-25. The live validation did not modify product code so that the tested +package remained identical to the accepted artifact. + +## Symptom + +Windows `scan.result_error` diagnostics can be stored with `occurred_at` about +the machine's local UTC offset in the future. In the validation environment the +offset was approximately three hours. The same naive timestamps also populate +Windows `target_scans.started_at` and `target_scans.ended_at`. + +This can misorder diagnostics, distort time-window filters, and make the admin +panel show a scan event later than the server receipt that contains it. + +## Live evidence + +The accepted Windows package ran with a fresh state directory and one slot. In +the expanded live cohort, all 45 Windows `scan.result_error` diagnostics had an +`occurred_at` to server `received_at` delta between approximately 10,816 and +10,819 seconds. All 135 normal Windows scan-result rows used naive local start +and end timestamps with the same approximately three-hour displacement when +treated as UTC. The four Windows timeout-path rows and all 102 Linux rows had +normal small timing deltas. + +The retained raw scanner material contained an explicit `+03:00` timestamp. +The persisted diagnostic retained the same wall-clock digits but labeled them +as UTC with `Z`. Monotonic scan durations and worker progress transport +timestamps remained correct. + +Sensitive raw targets, scanner output, and credentials are retained only in the +restricted live evidence file and are intentionally not reproduced here. + +## Root cause + +`app/scanner.py` creates scan timestamps with naive local datetimes: + +- `scan_target_result()` uses `datetime.now().isoformat()` for + `scan_started_at` and `timestamp` near lines 16339 and 16416. +- fallback result construction in `scan_single_target()` does the same near + lines 16844, 16863, and 16877. + +Diagnostic construction parses those values and, when no timezone is present, +uses `occurred.replace(tzinfo=timezone.utc)` near line 16581. That operation +relabels local wall-clock time as UTC instead of converting it. The error is +visible on non-UTC hosts and is hidden on UTC Linux hosts. + +## Expected behavior + +All persisted protocol timestamps must identify a real UTC instant. Scanner +result timestamps should be emitted as timezone-aware UTC values, and legacy +naive values must not be silently reinterpreted as known UTC instants. + +## Suggested correction and regression coverage + +Emit `datetime.now(timezone.utc).isoformat()` at every result-construction site +and preserve the offset through serialization. Add a non-UTC-host regression +test that verifies: + +- scanner start/end timestamps identify the actual UTC instant; +- diagnostic `occurred_at` precedes or closely tracks server `received_at`; +- Windows and Linux admin time-window filters return the same logical events; +- monotonic duration fields remain unchanged. + +## Validation artifact + +The expanded unrestricted evidence is retained at +`build/live-trace-20260925/raw-evidence-expanded.json`. It contains sensitive +raw internals and must not be published as a general operator report. diff --git a/docs/defect-worker-network-oserror-mislabeled-local-io-2026-09-25.md b/docs/defect-worker-network-oserror-mislabeled-local-io-2026-09-25.md new file mode 100644 index 0000000..d9135aa --- /dev/null +++ b/docs/defect-worker-network-oserror-mislabeled-local-io-2026-09-25.md @@ -0,0 +1,50 @@ +# Worker network OSError is logged as local I/O + +Status: open, reproducible from the error-classification path. + +## Observed behavior + +The accepted Linux worker logged the following safe summary during the live +operator-experience validation: + +```text +2026-09-25T13:10:05.909Z worker slot 0: local I/O operation failed +``` + +The complete worker journal shows that reservation 1689 had already entered +`assigned` at `13:09:46.641Z`. No runner phase had started. After the configured +error delay, the worker retried the same durable assignment, entered +`preparing` at `13:10:22.215Z`, completed it, and received one +`bundle_accepted` receipt at `13:10:28Z`. + +The only operation between the successful `assigned` event and runner startup +is the authenticated assignment-status request in +`WorkerSlot.step()` (`app/remote_worker_client.py`, around lines 2141-2153). +The HTTPS client lets socket and transport `OSError` exceptions propagate. +`safe_worker_error_summary()` (`app/remote_worker_client.py`, around lines +128-135) maps every `OSError` to `local I/O operation failed`, including network +socket failures. + +## Impact + +- No assignment, result, or progress data was lost in the observed incident. +- The durable retry behavior worked and did not rescan the target. +- The operator-facing message misclassifies a transient network failure as a + local storage/filesystem problem, which can send diagnosis in the wrong + direction. +- The original exception type is not retained in the safe worker log, so the + transport subtype cannot be recovered after the fact. + +## Evidence + +- `build/live-trace-20260925/linux-worker-state-interim.tar.gz` +- Worker event sequences 1292-1301 and the matching history row for reservation + 1689 +- `app/remote_worker_client.py` status-read and safe-summary paths + +## Expected correction + +Classify network/socket failures before the broad `OSError` branch and emit a +safe transport-specific summary. Keep filesystem/storage `OSError` failures as +local I/O. Add a regression test covering an `OSError` raised by +`WorkerHTTPClient.status()` and verify that retry behavior remains unchanged. diff --git a/docs/end-to-end-scanner-validation-2026-09-22.md b/docs/end-to-end-scanner-validation-2026-09-22.md new file mode 100644 index 0000000..3f342a3 --- /dev/null +++ b/docs/end-to-end-scanner-validation-2026-09-22.md @@ -0,0 +1,316 @@ +# Scanner End-to-End Validation Evidence: 2026-09-22 + +## Verdict + +The bounded production validation passed on the approved `sec` deployment. +It exercised the real PostgreSQL queue, protocol-2 remote assignment, existing +Windows/WSL worker, TruffleHog execution, bundle upload, durable receipt, +transactional ingestion, normalized findings/errors, and JSONL compatibility +projection paths. + +The evidence consists of: + +- 36 ordinary public-target scans under realistic production backlog; +- one separately managed non-live synthetic GitLab fixture scan proving the + positive finding path; +- exact append-region validation for `scan_results.jsonl` and + `found_secrets.jsonl` against PostgreSQL reconstruction; +- byte-identical restoration of the original production config; +- cleanup of private validation target files; and +- audited reopening of discovery and dispatch. + +This is strong bounded production evidence, not a claim that every source, +failure mode, platform, detector, scale, or deployment environment is proven. + +## Safety Envelope + +- Only `sec` was used. `prod` was never touched. +- Raw targets, raw findings, credentials, device tokens, runtime YAML, worker + argv, the protected admin prefix, and edge markers were not printed. +- Configuration changes used managed Preview -> Save candidate -> Apply. +- Discovery and dispatch were paused and the runtime was drained before every + apply. +- Long operations and monitors ran detached and were observed with bounded + status polls. +- Host Caddy and X-UI remained outside the managed lifecycle. +- PostgreSQL remained the sole authority; JSONL was treated as a rebuildable + compatibility projection. + +## Original Baseline + +- Original active config SHA-256: + `f055a9f2506ab4fffa6953a95c2f6c07b1202e6558f4ce52bbf1b463ed6b1781`. +- Drained baseline high-water IDs: + - target queue: 1,534,069; + - result reservations: 880; + - target scans: 879; + - findings: 7; + - errors: 1,913. +- Runtime controls were revision 26, paused/paused, `drained`, blockers zero. +- One active worker device had contacted the server recently. +- All baseline orphan and referential invariants were zero. + +Root-only baseline evidence: + +- `/opt/truf-remote-server/staging/scanner-validation-pre.json`; +- `/opt/truf-remote-server/staging/scanner-validation-drained.json`; +- `/opt/truf-remote-server/staging/scanner-validation-pre-discovery-v4.json`; +- `/opt/truf-remote-server/staging/scanner-validation-pre-dispatch-newest-v5.json`. + +## Defects Found and Corrected + +The validation exposed defects that synthetic tests had not modeled precisely. +Each failure was contained by pause/drain, rollback, or failed-hold behavior +before dispatch was opened. + +### Protected Config Parent + +Discovery required the parent of a private file to be runtime-owned mode 0700, +while the deployed contract intentionally uses root-owned mode 0755 +`/data/config` with runtime-owned mode 0600 documents. Writable runtime +directories still require private runtime ownership. Sensitive file parents now +also accept a non-link root-owned directory with no group/other write bits and +effective-user search access. The private file itself remains strictly checked. + +### Supervisor Startup Locking + +PostgreSQL readiness previously launched core children and discovery producers +while the supervisor held `control_lock`, then could perform another PostgreSQL +query under that lock. Child bootstrap/entrypoint authentication needed the same +lock and had bounded deadlines. Pipeline status refresh now happens before the +lock, structured snapshots use cached-only status, core children start before +source admission, and discovery producers use the source dependency gate. + +### Strict Discovery Health + +Ordinary Docker health intentionally tolerates periodic producer waits. Managed +lifecycle health now additionally uses explicit +`--require-discovery-producers` and rejects enabled producers that are absent, +blocked, never run, runtime-blocked, or waiting after a nonzero exit. Waiting +after a successful exit remains valid. + +### Transient Strict-Health Probe + +The first fixture apply encountered one bounded HuggingFace PostgreSQL +connection timeout after every core worker had started successfully. The host +lifecycle formerly performed only one strict probe after Docker health became +healthy. It now retries only health-category strict failures inside the existing +240-second runtime-health deadline. Identity and metadata errors remain +immediate failures, and persistent health failure still rolls back. + +### Managed Claim Order + +The remote assignment path already supported `oldest`, `newest`, and `balanced` +PostgreSQL admission, but the exact managed template omitted this field for the +three core sources. The optional field is now represented and semantically +validated. Temporary `newest` ordering allowed recent bounded discoveries to be +tested against the real 1.5-million-row queue without direct SQL mutation or +mass-hiding historical backlog. The restored original config omits the optional +field and therefore uses the normal `oldest` default. + +### Candidate Base Authority + +Candidate preparation originally used editor text that could represent an old +candidate rather than active config. This inherited an earlier intentionally +disabled Worker API setting. Candidate tools now read and hash-bind active +config bytes explicitly before deriving changes. + +## Runtime Deployment Evidence + +The corrected runtime was built as small derived immutable images rather than +modifying a running container. The final validation image ID was: + +`sha256:5b9c86f68719d8c1f2358e0c4565dd2795ed608482968747feb961cf14908b5a` + +Prior images remain under rollback tags. Image Entrypoint, Cmd, User, source +hashes, and in-image compilation were checked. An official lifecycle restart on +the final image completed `succeeded/succeeded`, reconciled, without a safe +category or failed hold. + +Relevant local regression evidence accumulated during the run: + +- runtime-document and worker-assignment tests: 45 passed; +- host lifecycle after transient-health retry: 34 passed, 4 platform skips; +- combined ACL/supervisor/health-focused suite: 246 passed, 9 platform skips; +- authenticated supervisor control class: 20 passed; +- focused compiles and `git diff --check`: passed. + +## Bounded Discovery + +The temporary candidate enabled one-page/one-result search settings for GitLab +and DockerHub and a four-item private custom file for HuggingFace. Dispatch +remained paused. Five successful cycles for each source completed before the +monitor's conservative time limit; no source cycle failed. + +Because source cycles do not map directly to queue rows and uniqueness conflicts +consume sequence values, queue high-water deltas were not treated as exact +cohort membership. Eight new queue rows were observed: five GitLab pending and +three DockerHub deferred. No direct queue updates were made. + +## Realistic 36-Scan Cohort + +The worker processed exactly 36 new remote reservations, IDs 881 through 916, +while discovery remained paused. A fail-closed monitor paused dispatch at the +target and started drain. Final source mix: + +| Source | Scans | +|---|---:| +| DockerHub | 11 | +| GitLab | 14 | +| HuggingFace | 11 | +| Total | 36 | + +All 36 reservations were remote, resolved, acknowledged, and +`bundle_accepted`. They had 36 distinct queue IDs, bundle IDs, and scan event +IDs, and every reservation had a receipt, payload hash, and execution-snapshot +hash. + +### Results + +| Source | Result summary | +|---|---| +| DockerHub | 8 clean, 3 degraded | +| GitLab | 12 clean, 1 retryable API error, 1 permanent not-found | +| HuggingFace | 11 clean | + +- Queue completion: 34 done, one deferred, one failed; no row remained fenced. +- Findings: zero, a valid outcome for random public targets. +- Errors: exactly two GitLab errors with queue dispositions matching their + retryable/permanent categories. +- Quarantine: zero new rows. +- Bundle/projection capacity after completion: zero items and zero bytes. +- Existing unrelated keycheck capacity was unchanged. + +### Bundle and Projection Invariants + +- 36 acknowledged bundles contained 110 frames. +- All bundle identities and counts matched their reservations and scans. +- Acknowledged physical `.trb` files were absent only after both pipeline + artifact records reached durable `deleted` state, as designed. +- 36 scans used `raw_result_storage=normalized_v2`. +- 36 compatibility rows used expected bounded reconstruction. +- Exactly 36 projection jobs completed, one per scan, without duplicates or + errors; all projection capacity was released. +- Physical append evidence covered 36 `scan_results` records and two + `scan_errors` records. +- Every registered append generation/offset/length existed and matched its + payload SHA-256, record count, and required JSON structure. +- `scan_results.jsonl` grew by exactly 83,752 bytes. +- `found_secrets.jsonl` did not change, matching zero random-target findings. +- All global queue/reservation and orphan invariants remained zero. + +Root-only evidence: + +- post snapshot: + `/opt/truf-remote-server/staging/scanner-validation-post-dispatch-newest-v5.json`, + SHA-256 + `45f9db81e88dbcbe4d6a04094dd1d892df06dd3f7cdd05eb9280c19da54aed94`; +- aggregate report: + `/opt/truf-remote-server/staging/scanner-validation-cohort-report-v5.json`, + SHA-256 + `e0826587e4ba667de10a994d5842aae5fae8fa11dc071210c202b38b0e643bf4`. + +## Controlled Positive Fixture + +Random public targets produced no finding, so a separate one-target run used a +public GitLab project whose README declares that its secret examples are +generated and non-live. No detector or verification behavior was weakened. + +- Fixture queue ID: 1,534,100. +- Reservation ID: 917. +- The immutable Git plan bound the approved exact head commit + `2a09bd6767d39b95cf39ce4b5fd210721275d503`. +- The reservation became acknowledged with `bundle_accepted` and a durable + receipt. +- Queue completion was `done` with no remaining reservation fence. +- Target scan status was `found` with 116 findings and zero errors. +- All 116 findings used the existing OpenAI detector. +- Verified count was zero, consistent with the unchanged no-verification policy. +- All findings had distinct finding UIDs, nonempty identities, private raw + material, redaction different from raw material, correct secret hashes, and + complete non-omitted compatibility payloads. +- No raw finding value was emitted by validation tooling. +- Bundle retirement and both pipeline artifact tombstones were correct. +- The single projection job completed and released capacity. +- The registered `scan_results` region contained one record with exactly 116 + findings and zero errors. +- The registered `found_secrets` region contained exactly 116 records. +- Both physical append regions matched the database payload SHA-256 and were + byte-identical to fresh PostgreSQL compatibility reconstruction. +- No new quarantine row was created. + +Root-only fixture report: + +`/opt/truf-remote-server/staging/scanner-validation-fixture-report-v2.json` + +SHA-256: + +`243a16b24bf9ae898bfdeb8f857c56ef1cf78e12e637730ff5b0674a250a4984` + +## Restoration and Final State + +The original 36,354 config bytes were passed through managed Preview, saved as +a candidate, and applied through the host agent. Preview preserved the exact +original SHA-256 and reported 169 semantic reversions. + +- Restore Save operation: + `af2a3cc4-90ae-5831-b932-bbe78ceb2cab`. +- Restore Apply operation: + `b12a2fc4-202a-578e-8dcf-b0a88cb028ef`. +- Apply terminal state: `succeeded/succeeded`, reconciled, category `None`. +- Active and candidate config SHA-256 both equal the original + `f055a9f2506ab4fffa6953a95c2f6c07b1202e6558f4ce52bbf1b463ed6b1781`. +- Lifecycle preflight and strict Worker API/discovery health passed. +- Runtime and edge were healthy; no failed hold existed. +- All private validation target/evidence files were removed. +- The root-only original backup was retained for audit. + +The drained post-restore snapshot is root-only at +`/opt/truf-remote-server/staging/scanner-validation-post-restore-drained-v1.json`, +SHA-256 +`de4bd98d728dc551b10712a4afb6be2db0ac9111428d46fa5dbb16f8d2d611ca`. + +Final audited control transitions advanced revision 46 to 49 in this order: + +1. cancel drain; +2. resume discovery; +3. resume dispatch. + +Final state was discovery open, dispatch open, drain `normal`. The existing +worker contacted the server within five minutes and immediately received normal +production work. A live assignment after reopening is expected and is not a +drain blocker because drain is no longer requested. + +The final post-resume snapshot had zero orphan/referential invariants and +preserved the original config SHA-256: + +`/opt/truf-remote-server/staging/scanner-validation-post-resume-final-v1.json` + +SHA-256: + +`42e19edae09550693d563b74631430cb1d2c1b807d636ccff20e545cebec3c2d` + +External route checks through existing host Caddy returned: + +- invalid Worker API authentication: 401; +- unauthenticated protected admin route: 401; +- unrelated path: 404. + +Host-agent, Caddy, and X-UI services remained active. Caddy and X-UI were not +lifecycle targets. + +## Residual Limits + +This validation does not prove: + +- long-duration soak or high-concurrency behavior; +- every detector and verification provider; +- every source mode, browser, OS, architecture, or network failure; +- every secrets/config mutation and rotation case; +- HA or multi-server operation; +- resistance to an independent penetration test; or +- correctness of arbitrary unsupported Compose, ingress, or proxy layouts. + +Within its declared scope, the real queue, worker, scanner, ingestion, +findings, error, compatibility, restoration, and resumed-production paths all +produced internally consistent durable evidence. diff --git a/docs/end-to-end-scanner-validation.md b/docs/end-to-end-scanner-validation.md new file mode 100644 index 0000000..d174878 --- /dev/null +++ b/docs/end-to-end-scanner-validation.md @@ -0,0 +1,339 @@ +# End-to-End Scanner Validation + +This runbook validates the real discovery, remote-worker, result-ingestion, and +compatibility-projection path with a bounded cohort of 30-40 targets. It is an +evidence procedure, not a claim that every repository feature and environment +has been proven correct. + +The completed 2026-09-22 production evidence is recorded in +[`end-to-end-scanner-validation-2026-09-22.md`](end-to-end-scanner-validation-2026-09-22.md). + +## Validation Questions + +The run must answer all of the following: + +1. Does bounded discovery create the expected immutable queue identities? +2. Do remote workers receive each authoritative assignment with the correct + source, target identity, execution snapshot, and lease fencing? +3. Does each worker run the intended TruffleHog scan and upload a canonical + protocol-2 result bundle? +4. Does the server accept a result exactly once and make receipt replay + idempotent? +5. Does the ingester transactionally connect the reservation, queue row, + target scan, findings, errors, and bundle record? +6. Does the JSONL projector reproduce the Windows-compatible + `scan_results.jsonl` and `found_secrets.jsonl` structures without becoming + a second source of truth? +7. Are naturally found or controlled-canary findings stored with safe identity, + location, detector, verification, redaction, and provenance fields? +8. After the test, are queues settled, projections caught up, no pipeline item + quarantined, and the normal production configuration restored? + +## Authorities and Expected Data Flow + +The authoritative sequence is: + +```text +discovery cycle + -> target_queue + -> result_reservation / immutable remote assignment + -> worker TruffleHog execution + -> canonical .trb upload + -> durable accepted receipt + -> result ingester transaction + -> target_scans + findings + errors + queue completion + -> projection_jobs + -> scan_results.jsonl + found_secrets.jsonl +``` + +PostgreSQL is authoritative. Result bundles are durable pipeline artifacts. +JSONL files are rebuildable compatibility projections and may lag briefly. +`accepted` proves durable server receipt; `ingested` proves the database +transaction completed. These states must not be treated as synonyms. + +## Safety Rules + +- Use only the approved test server and approved worker devices. +- Never print device tokens, source credentials, raw secrets, active runtime + YAML, worker argv, or complete unredacted findings into a terminal/log. +- Query finding structure using IDs, hashes, redacted values, lengths, booleans, + detector names, and location metadata. Review any raw secret only through the + already protected admin workflow if explicitly required. +- Apply and restore configuration through Preview -> Save candidate -> Apply. + Do not edit the active runtime document in place. +- Pause discovery and dispatch and drain before each runtime-document apply. +- Record the original config SHA-256 and require byte-identical restoration at + the end. +- Use a unique test-run label and database high-water marks. Never infer the + cohort from wall-clock time alone. +- Do not delete queue, bundle, finding, projection, or receipt evidence to make + a failed test look clean. + +## Bounded Test Configuration + +Use a temporary candidate derived from the active document. Preserve all +secrets and unrelated settings. The exact candidate must be reviewed before it +is applied. + +### DockerHub + +- `mode: search` +- `pages: 1` +- `per_page: 1` +- `docker_images_per_repository: 1` +- Use a reviewed finite query list for the test window. + +One query is consumed per source cycle. An already-known or unsuitable search +result can produce no new queue row, so the number of cycles is not the cohort +size. + +### GitLab + +- `mode: search` +- `pages: 1` +- `per_page: 1` +- Use a reviewed finite query list for the test window. +- Keep current age, commit-boundary, exact-ref, visibility, and history-depth + safety controls unless the test explicitly records a different expectation. + +### HuggingFace + +HuggingFace recent discovery does not support a real `per_page: 1` keyword +test. Its API runner fetches newest-modified Spaces and the current API page +size is fixed at 100; the configured query is only a rotation placeholder. + +For a bounded cohort, use `mode: custom` with a reviewed private `target_file` +containing a small list of Space IDs. Do not claim that changing `per_page` to +1 bounded this source when it did not. + +### Recommended Cohort + +Target 36 authoritative terminal scans: + +- 16 DockerHub immutable digest targets; +- 16 GitLab exact-ref/commit-planned targets; +- 4 HuggingFace custom Space targets. + +The exact split may vary between 30 and 40 when discovery deduplicates known +targets or a target becomes permanently inaccessible. Continue only until the +recorded cohort reaches the agreed bound. Do not inflate discovery simply to +hit an exact aesthetic number. + +## Positive-Finding Requirement + +A random public cohort may correctly produce zero findings. Zero findings +cannot validate the finding-storage and `found_secrets.jsonl` path. + +Include at least one separately identified, non-live controlled fixture that is +expected to trigger an already approved detector. The fixture must contain no +usable credential. Record its expected detector and identity before scanning. +Do not weaken verification, introduce a new detector, or publish a real secret +merely to force a positive result. + +If no approved positive fixture is available, report the finding path as +unverified by this run even if all zero-finding scans succeed. + +## Phase 1: Baseline + +With runtime healthy, record a secret-safe baseline: + +- active config SHA-256 and semantic config SHA-256; +- runtime-control revision and open/paused/drain state; +- enabled source set and active worker package manifests; +- remote worker/device count, recent contact, and package capability match; +- high-water IDs for `target_queue`, `result_reservations`, `target_scans`, + `findings`, `errors`, `result_bundles`, and `projection_jobs`; +- queue counts by source and status; +- active reservation count and oldest age; +- pipeline worker readiness and capacity counters; +- pending/leased/quarantined bundle, projection, and keycheck counts; +- current projection stream/cursor identity; +- byte size and final complete-line identity of active JSONL files. + +The baseline collector must print aggregates and hashes only. It must not emit +targets, assignment payloads, tokens, raw findings, or runtime documents. + +## Phase 2: Apply the Test Candidate + +1. Pause discovery. +2. Pause dispatch. +3. Start drain and wait for blocker count zero and `drained`. +4. Preview the bounded candidate and review the semantic diff. +5. Save and apply the candidate through the host-agent lifecycle. +6. Require reconciled `succeeded`, no failed hold, strict runtime health, edge + health, admin health, and Worker API health. +7. Cancel drain, then resume discovery and dispatch in that order. + +Do not continue if the lifecycle operation rolls back or enters failed hold. + +## Phase 3: Build and Freeze the Cohort + +Record the baseline `target_queue.id` high-water mark. Let the bounded sources +cycle until 30-40 new eligible queue rows have been created after that mark. + +Then: + +1. Pause discovery so the cohort cannot grow. +2. Leave dispatch open until the selected queue rows settle. +3. Record cohort queue IDs and only their safe identities: source, normalized + target hash, query hash, immutable planning kind, and creation order. +4. Separate deduplicated, permanently inaccessible, deferred, retried, and + actually assigned items. Do not count an API result as a scan. + +The authoritative cohort is a fixed set of queue IDs, not "whatever completed +during the same hour." + +## Phase 4: Observe Remote Execution + +For every cohort queue ID, verify: + +- no more than one current authoritative reservation; +- assignment package/platform capability matches the registered worker; +- lease token and execution snapshot are bound but never printed; +- Docker targets are immutable `repo@sha256` identities; +- GitLab targets have the intended exact planning/ref identity; +- HuggingFace targets use the direct Space execution kind; +- terminal report classification is success, permanent target failure, or + retryable provider failure as designed; +- retries preserve queue identity and increment attempts without creating a + second authoritative acceptance; +- accepted receipt replay returns the same durable result. + +Physical work can repeat after a lease expiry or network partition. Correctness +means fencing permits one authoritative acceptance and one queue completion, +not that duplicate physical execution is impossible. + +## Phase 5: Validate Bundles and PostgreSQL Structure + +For each accepted result, validate without dumping body content: + +- bundle exists at the registered private relative path; +- bundle byte count and SHA-256 match database metadata; +- bundle schema/version, event ID/hash, reservation ID, queue ID, source, + normalized target identity, execution snapshot identity, and scan policy are + internally consistent; +- result is ingested exactly once; +- `target_queue.target_scan_id` references the corresponding `target_scans.id`; +- queue completion is applied once with a terminal disposition; +- `target_scans.queue_id` and claim lease identity refer back to the cohort row; +- `target_scans.findings_count` and `error_count` equal actual child-row counts; +- every finding/error references the same target scan, source, cycle, and run; +- no legacy `raw_result_json`, publication outbox row, or orphan relation is + introduced; +- ingested bundle credit and pipeline-capacity counters are released exactly + according to the durable state machine; +- no cohort item enters `pipeline_quarantine`. + +Aggregate checks must cover the entire cohort. Additionally inspect a small +redacted structural sample from every source and every terminal disposition. + +## Phase 6: Validate Findings + +For every finding in the cohort, inspect structure only: + +- stable `finding_uid` and finding fingerprint; +- detector name/type and verification flag; +- source, target hash, file path, line/commit/source timestamp where applicable; +- redacted secret and secret/detector hashes; +- provider and credential-kind enrichment; +- required-context and raw-payload-omitted flags; +- bounded source metadata and enrichment JSON decode successfully; +- no unexpected raw-secret exposure in logs, queue rows, assignment metadata, + admin list views, or compatibility scan summaries. + +For the controlled positive fixture, require the expected finding to exist in +PostgreSQL and to project once to `found_secrets.jsonl`. + +## Phase 7: Validate Windows-Compatible Files + +Use the active configured `global.results_dir`. The relevant compatibility +outputs are: + +- `scan_results.jsonl` for one sanitized scan event per projected scan; +- `found_secrets.jsonl` for projected finding events; +- their publication ledgers, active stream metadata, and rotated segments; +- per-service keycheck result files only if keychecks run for the finding. + +For the cohort, verify: + +- every required `projection_job` reaches `completed`; +- projector cursor and append ledger advance monotonically; +- each cohort scan event appears exactly once by `scan_event_id`; +- each cohort finding appears exactly once by `finding_uid`; +- JSON lines parse and match the current compatibility schema; +- scan summaries match PostgreSQL counts and terminal status; +- finding projections are redacted as designed and preserve safe provenance; +- active and rotated segments together contain the events; checking only the + active file is insufficient when rotation occurs; +- no torn-tail quarantine, duplicate append, skipped cursor, or unpublished + completed job exists. + +These files should have the same logical structure as the Windows deployment, +but path separators and host/container root paths are platform-specific. + +## Phase 8: Queue and Pipeline Closure + +After all cohort rows settle, require: + +- 30-40 cohort queue rows accounted for by terminal, deferred, or explicitly + classified retry state; +- no expired active reservation remains unreaped; +- no queue row has multiple authoritative accepted results; +- no accepted result remains un-ingested beyond the bounded pipeline window; +- no completed scan remains unprojected beyond the bounded projector window; +- no stale pipeline lease or capacity leak; +- no unexpected quarantine; +- runtime, Worker API, edge, host-agent, host Caddy, and unrelated host service + health remain good. + +The final report must show counts for discovered, deduplicated, assigned, +retried, accepted, ingested, projected, succeeded, skipped/permanent, +retryable/deferred, findings, errors, and quarantines. + +## Phase 9: Restore Production Configuration + +1. Pause discovery and dispatch. +2. Drain to zero blockers. +3. Apply the exact original runtime document through the normal lifecycle. +4. Require byte-identical original config SHA-256, reconciled lifecycle success, + strict health, and no failed hold. +5. Cancel drain, resume discovery, then resume dispatch according to the + original control state. +6. Confirm worker contact and normal post-test assignment flow. + +Do not restore by manually editing YAML or replacing files behind the +host-agent. + +## Pass Criteria + +The run passes only when: + +- at least 30 and at most 40 fixed-cohort rows are fully accounted for; +- all accepted cohort results ingest exactly once; +- all required cohort projections complete exactly once; +- queue/reservation/scan/finding/error/bundle relationships are consistent; +- the controlled positive finding reaches PostgreSQL and + `found_secrets.jsonl`, or the report explicitly marks positive-finding + validation incomplete because no approved fixture existed; +- no unexplained retry, orphan, duplicate acceptance, capacity leak, + quarantine, failed hold, or projection gap remains; +- the original production config is restored exactly and services are healthy. + +Any failure must retain its operation IDs, queue IDs, reservation IDs, hashes, +safe categories, and aggregate evidence for diagnosis. A partial pass must not +be reported as "100% scanner correctness." + +## Evidence Report + +Append or link a dated report containing: + +- environment and worker package identities; +- original/test/restored config hashes; +- cohort definition and aggregate source split; +- lifecycle operation IDs for test apply and restore; +- queue and pipeline baseline/final aggregates; +- per-stage reconciliation counts; +- redacted structural examples for a scan, an error/skip, and a finding; +- JSONL/ledger reconciliation counts; +- deviations, retries, quarantines, and unresolved questions; +- final verdict with explicit tested and untested boundaries. diff --git a/docs/extended-live-validation-2026-09-26.md b/docs/extended-live-validation-2026-09-26.md new file mode 100644 index 0000000..bcd08ff --- /dev/null +++ b/docs/extended-live-validation-2026-09-26.md @@ -0,0 +1,105 @@ +# Extended Live Validation + +Date: 2026-09-26 +Run: `e6ac8aec-ec20-4ba4-a924-abe5ee95d82c` +Conclusion: `pass_with_documented_deviations` + +## Scope + +The production validation ran one native Windows worker slot and one WSL/Docker +worker slot through discovery, assignment, execution, progress, diagnostics, +bundle ingestion, projection, capacity release, and scheduled keycheck. The run +started at `2026-09-26T00:29:30.500768Z`; terminal server evidence was captured +at `2026-09-26T03:54:04.573328Z`. + +This report contains derived counts, classifications, and cryptographic +identities only. Raw provider values and worker payloads remain in private +evidence storage. + +## Outcome + +- 314 reservations reached terminal outcomes: 309 acknowledged and 5 refunded. +- All 309 accepted bundles were ingested in one attempt, projected, settled, and + released; no unresolved reservation or duplicate receipt remained. +- The accepted cohort covered DockerHub (105), GitLab (102), and Hugging Face + (107), split across Windows (167) and WSL (147). +- 3,042 progress events covered all 309 accepted reservations with no duplicate + reservation sequence. +- All 348 projection jobs completed and released in one attempt with no error. +- The sealed terminal cut at controller revision 143 had no live assignment, + pre-commit bundle, capacity use, publication outbox item, quarantine row, + waiting lock, or blocker. + +The 309 scan outcomes were 174 clean, 75 degraded, 56 error, and 4 found. Five +findings were recorded and none were verified. The run recorded 100 classified +scan errors, led by 71 download failures; source distribution was GitLab 92, +DockerHub 7, and Hugging Face 1. + +## Diagnostics And Keycheck + +The validation captured 61 worker diagnostics: 52 result errors, 4 stage +timeouts, 4 runner protocol failures, and 1 client process failure. Fifty-one +were retryable and ten were nonretryable. The records remain available in the +private raw bundle for exact-body investigation. + +The captured keycheck cohort contained 39 candidates. All completed in one +attempt, released capacity, linked to a result, and projected. Results were 35 +invalid or revoked and 4 no-balance; 37 came from API execution and 2 from +cache. At the terminal cut the global keycheck queue had 184 completed and no +pending, leased, or deferred candidate. + +Nine pending DockerHub discovery retry rows with zero attempts remained as +expected durable discovery backlog, not as leaked worker-pipeline work. + +## Evidence Integrity + +- Server chain: 202 valid records, sequence `-1..199`, with all 575 referenced + objects (1,245,586,992 bytes) verified. +- Server NDJSON SHA-256: + `a6beb1864c540fc5f22b2b647730a39e45dfddb10cf5ceff5e0181eb1a8f0bd8`. +- Server run SHA-256: + `5ae58df5f67a8d2a8d4e84d73262a6e7009e03dcbcc9ea176537b384d4e693c3`. +- Worker chain: 158 valid records, sequence `0..157`, with all 2,136 referenced + objects (7,733,582,955 bytes) verified. +- Worker NDJSON SHA-256: + `03a04a0da5a2db6bd02f2aaed41c191554b135ec287fbc1e2d9998d77da59898`. +- Worker final payload SHA-256: + `f652e5f98fb23f9aa2fb4e44e10168a931ea973be616fb09dbdb72001ce961ad`. + +The machine-readable verification manifest is at +`build/extended-live-validation/runs/e6ac8aec-ec20-4ba4-a924-abe5ee95d82c/verification-manifest.json`. +The complete server evidence root is retained in private production storage; +the local run root contains compact server artifacts and complete worker +evidence. + +## Observed Deviations + +The authenticated `recheck all` operation was accidentally used where only two +pending candidates should have been rechecked. This caused 35 broad provider +checks contrary to `KEYCHECK-001`. The command processed all 35, skipped none, +returned success, produced no failed operation, and all resulting effects +settled. This is an operator-scope deviation, not an approved workflow change. + +Six monitor samples encountered transient database statement timeouts. Every +monitor error recovered, and the evidence chains remained valid. The runtime +and edge each recorded zero restart during the validation. + +## Remaining Defects + +- Windows GitLab filename-too-long checkout recovery exists locally but is not + deployed. +- Invalid API-key classification has a local fix that is not deployed. +- A roughly 20-second WSL clock-domain monotonic failure remains open. +- Generic `WorkerContractError` diagnostics can lose structured field detail. +- Monitor aggregate queries can exceed their statement timeout under load. + +Focused regression coverage for Git checkout recovery, Hugging Face long paths, +and direct remote credentials passed: 145 tests in 155.26 seconds. + +## Restore + +Production was restored with compare-and-swap transitions from revision 143 to +146: cancel the validation drain, resume discovery, then resume +dispatch. Final state was dispatch open, discovery open, drain normal, healthy +core services and producers, and exactly one live assignment on each worker. +No further validation probe is required. diff --git a/docs/remote-worker-cheatsheet-docker-ru.md b/docs/remote-worker-cheatsheet-docker-ru.md new file mode 100644 index 0000000..fb5934b --- /dev/null +++ b/docs/remote-worker-cheatsheet-docker-ru.md @@ -0,0 +1,77 @@ +# TRUF worker: шпаргалка Docker + +Откройте Bash или PowerShell в корне bundle с `compose.yaml`, worker image archive +и helper scripts. Нужен Docker Engine с Compose v2 либо Docker Desktop. + +## Один раз + +Создайте локальный `worker-install.yaml` через текстовый редактор: + +```yaml +server: https://pregnant.horsecock.store +token: PASTE_DEVICE_TOKEN_HERE +parallelism: 1 +``` + +Bash: + +```sh +chmod 600 ./worker-install.yaml +chmod 700 ./workerctl.sh +docker load -i truf-worker-linux-x86_64.tar.gz +docker compose run --rm -T worker install --config - < ./worker-install.yaml +rm -- ./worker-install.yaml +docker compose run --rm worker doctor --json +``` + +PowerShell: + +```powershell +docker load -i .\truf-worker-linux-x86_64.tar.gz +Get-Content -Raw .\worker-install.yaml | + docker compose run --rm -T worker install --config - +Remove-Item -LiteralPath .\worker-install.yaml +docker compose run --rm worker doctor --json +``` + +## Каждый день + +Bash: + +```sh +docker compose up -d +./workerctl.sh status +./workerctl.sh attach --follow-seconds 300 +./workerctl.sh watch --follow-seconds 300 +./workerctl.sh stop --timeout 120 --json +docker compose down +``` + +PowerShell: + +```powershell +docker compose up -d +.\workerctl.ps1 status +.\workerctl.ps1 attach --follow-seconds 300 +.\workerctl.ps1 watch --follow-seconds 300 +.\workerctl.ps1 stop --timeout 120 --json +docker compose down +``` + +`q` или `Ctrl-C` отсоединяет `attach`/`watch`, но не останавливает worker. +`docker compose down` без `--volumes` сохраняет identity, token, незавершённую +работу и history в `truf-worker-data`. + +## Диагностика + +```sh +./workerctl.sh status --json +./workerctl.sh logs --tail 200 +./workerctl.sh logs --follow --follow-seconds 300 +./workerctl.sh history --limit 50 +./workerctl.sh doctor --json +``` + +Не используйте `docker compose down --volumes`. Clean stop должен вернуть +`state: stopped`, `drained: true`, `exit_code: 0`; только после этого выполняйте +`docker compose down`. diff --git a/docs/remote-worker-cheatsheet-linux-ru.md b/docs/remote-worker-cheatsheet-linux-ru.md new file mode 100644 index 0000000..3576d0e --- /dev/null +++ b/docs/remote-worker-cheatsheet-linux-ru.md @@ -0,0 +1,70 @@ +# TRUF worker: шпаргалка Linux без Docker + +Нужны системный Python 3.12 и package-local Git, TruffleHog и Python dependencies +из проверенного artifact. Native executable authority должна находиться под +root-owned путём, поэтому один раз перенесите распакованный package в `/opt`: + +## Один раз + +```sh +sudo mv ./truf-worker-linux-x86_64 /opt/truf-worker +cd /opt/truf-worker +``` + +Подготовьте доверенные ownership и permissions package. Скрипт оставляет +application code приватным для текущего user, а native Git и TruffleHog - +неизменяемыми для него: + +```sh +sudo ./prepare-worker.sh +``` + +Затем создайте `~/.config/truf/worker-install.yaml` через локальный текстовый +редактор: + +```sh +install -d -m 700 ~/.config/truf +nano ~/.config/truf/worker-install.yaml +``` + +```yaml +server: https://pregnant.horsecock.store +token: PASTE_DEVICE_TOKEN_HERE +parallelism: 1 +``` + +Затем выполните: + +```sh +chmod 600 ~/.config/truf/worker-install.yaml +./truf-worker install --config ~/.config/truf/worker-install.yaml +rm -- ~/.config/truf/worker-install.yaml +./truf-worker doctor --json +``` + +## Каждый день + +```sh +./truf-worker start --startup-timeout 30 +./truf-worker status +./truf-worker attach --follow-seconds 300 +./truf-worker watch --follow-seconds 300 +./truf-worker stop --timeout 120 --json +``` + +`q` или `Ctrl-C` отсоединяет `attach`/`watch`, но не останавливает worker. +Portable package не устанавливает systemd unit: после перезагрузки выполните +`start` из того же package под тем же OS user. + +## Диагностика + +```sh +./truf-worker status --json +./truf-worker logs --tail 200 +./truf-worker logs --follow --follow-seconds 300 +./truf-worker history --limit 50 +./truf-worker doctor --json +``` + +Clean stop должен вернуть `state: stopped`, `drained: true`, `exit_code: 0`. +Иначе сохраните state и диагностику; не удаляйте work или bundles. diff --git a/docs/remote-worker-cheatsheet-windows-ru.md b/docs/remote-worker-cheatsheet-windows-ru.md new file mode 100644 index 0000000..afb71ef --- /dev/null +++ b/docs/remote-worker-cheatsheet-windows-ru.md @@ -0,0 +1,55 @@ +# TRUF worker: шпаргалка Windows + +Откройте PowerShell в корне распакованного Windows package. Обычный запуск не +требует прав администратора. + +## Один раз + +```powershell +Set-ExecutionPolicy -Scope Process Bypass +notepad .\worker-install.yaml +``` + +Заполните открытый файл: + +```yaml +server: https://pregnant.horsecock.store +token: PASTE_DEVICE_TOKEN_HERE +parallelism: 1 +``` + +Затем выполните: + +```powershell +.\prepare-worker.ps1 +.\truf-worker.cmd install --config .\worker-install.yaml +Remove-Item -LiteralPath .\worker-install.yaml +.\truf-worker.cmd doctor --json +``` + +## Каждый день + +```powershell +.\truf-worker.cmd start --startup-timeout 30 +.\truf-worker.cmd status +.\truf-worker.cmd attach --follow-seconds 300 +.\truf-worker.cmd watch --follow-seconds 300 +.\truf-worker.cmd stop --timeout 120 --json +``` + +`q` или `Ctrl-C` отсоединяет `attach`/`watch`, но не останавливает worker. +После перезагрузки снова выполните только `start`; token уже находится в +приватном installed config. + +## Диагностика + +```powershell +.\truf-worker.cmd status --json +.\truf-worker.cmd logs --tail 200 +.\truf-worker.cmd logs --follow --follow-seconds 300 +.\truf-worker.cmd history --limit 50 +.\truf-worker.cmd doctor --json +``` + +Clean stop должен вернуть `state: stopped`, `drained: true`, `exit_code: 0`. +Иначе сохраните state и диагностику; не удаляйте work или bundles. diff --git a/docs/remote-worker-operations.md b/docs/remote-worker-operations.md new file mode 100644 index 0000000..a3f5b02 --- /dev/null +++ b/docs/remote-worker-operations.md @@ -0,0 +1,454 @@ +# Remote Scan Worker Operations + +This runbook covers the opt-in remote worker boundary. Production defaults remain +disabled: `supervisor.worker_api.enabled` and +`supervisor.worker_api.admin.enabled` are both `false` in +`app/config.linux.yaml`. Enabling either one, applying schema changes, or starting +the production edge requires a separate reviewed rollout. + +Remote workers are trusted clients. Their only operator-authored runtime settings +are the HTTPS server origin, one opaque device token, and a positive slot count +`N`. Target/source settings, immutable plans, scanner policy, limits, and only the +credentials needed for an assignment come from the server. Protocol-2 package +schema 3 manifests advertise exact `(source, platform, planning_kind)` +capabilities. The distributed core package contains GitLab `exact_git_v1`, +DockerHub `docker_direct_v1`, and HuggingFace `huggingface_space_v1`; GitHub is a +legacy optional capability and is not in the core profile. Detailed keycheck +remains server-only after bundle ingestion; clients must not receive keycheck +configuration or run keycheckers. + +## Isolated verification + +Run each verifier independently from the root of the isolated +`D:\truf-workers` checkout. Do not run plain `docker compose`, production +Compose files, import overrides, unrestricted pytest, or broad Docker cleanup. +Do not run these verifiers concurrently. They create random, ownership-labelled +resources and never use production data, credentials, volumes, or image tags. + +### Main Docker E2E + +From Linux or WSL, with the already-built local images +`truf-worker-test:runtime` and `truf-worker-test:test` and an already ignored +`docker/test-results/latest.json`: + +```sh +python3 -I -S -B docker/verify.py +``` + +The verifier invokes only `compose.e2e.yaml`, does no build or pull, and normally +removes only its ownership-verified resources. It retains artifacts after a +failure. Use `--keep` only when a reviewed investigation needs stopped artifacts; +record the printed project name and never substitute prune, broad `down`, or +`down --volumes` commands. + +### Packaged Windows/Linux client E2E + +From Windows, with Windows Python, `wsl.exe`, passwordless `sudo -n docker` in the +selected WSL distribution, the built portable directory, and the already-built +Linux worker and test images: + +```powershell +python -I -S -B docker/verify_packaged_workers.py ` + --windows-artifact dist/truf-worker-windows-x86_64 ` + --linux-image truf-remote-worker:linux-x86_64 ` + --test-image truf-worker-test:test ` + --wsl-distro Ubuntu-24.04 +``` + +This single gate runs real packaged Windows and Linux scanners at `N=2`, tests a +server outage and restart recovery, and compares normalized cross-platform +evidence. It does not use Compose. Successful resources are removed unless +`--keep` is supplied; failures retain the labelled resources and the reported +`build/pwe-*` evidence directory. + +### Edge E2E + +From Windows with the same WSL Docker access and already-built +`truf-worker-test:test` and `truf-edge-e2e:test` images: + +```powershell +python -I -S -B docker/verify_edge_e2e.py --wsl-distro Ubuntu-24.04 +``` + +This gate uses only random labelled resources and a private synthetic backend. It +checks authenticated routes, two-failure admin bans, restart persistence, +automatic expiry, SSH-equivalent unban behavior, forwarded-header handling, and +worker availability from the banned admin IP. Success removes its resources; +failure reports the retained owned inventory and evidence path. + +## Client bootstrap + +Use a separately issued token for every device. The server stores only its +SHA-256 digest. Do not place a real token in documentation, source control, +shell transcripts, support output, or process diagnostics. The client validates +the server certificate and accepts only an HTTPS origin without credentials, +path, query, or fragment. `N` must be between 1 and 128; it bounds local occupied +slots but never overrides the user's server cap across devices. + +Generated state, pending bundles, and work directories are recovery data, not +additional source/provider configuration. Preserve them across restarts until +the server authoritatively resolves the corresponding slots. + +### Artifact source and distribution + +Worker tools come from a reviewed release checkout; they are not installed +piecemeal on each client. The Windows builder and Linux worker package/image +targets assemble and verify the complete worker authority, Python runtime where +applicable, Git helper, TruffleHog binary, detector policy, CA roots, and +hash-locked dependencies. They deliberately exclude PostgreSQL tools, +server/runtime authority, provider implementations, detailed keycheck code, +server credentials, and database credentials. + +There is no public worker download, image registry, installer, or automatic +updater. Build each release once in a controlled release environment, retain its +generated manifest/release metadata, and distribute the exact ZIP or image digest +through a trusted artifact channel. Do not rebuild independently on every worker. +Install the corresponding trusted package manifest beneath +`/etc/truf/worker-packages` on the server before allowing that artifact to claim +protocol-2 work. + +Onboard a new worker in this order: + +1. Create or select its server-side user and start with active-assignment cap `1`. +2. Create a distinct device identity and issue its token. The plaintext token is + shown once; never reuse it for another device. +3. Deliver the exact reviewed Windows ZIP, native Linux package, or Linux image, + and compare its package identity with the registered server manifest. +4. Create a private three-field YAML document with the HTTPS origin, token, and + conservative parallelism such as `1`, then run `install --config`. Delete the + input YAML after installation. This writes the existing private local + configuration; lifecycle commands never need the token on their command line. +5. Run `doctor`, start the supervisor, inspect `status` and `attach`, and confirm + server `last_contact_at`. +6. Reconcile one real assignment through accepted receipt, ingestion, settlement, + and projection before increasing either the user cap or local parallelism. +7. Preserve the private state tree across restart or outage. Drain work before + token rotation, revocation, artifact replacement, or state removal. + +For a routine new device, artifact delivery and token issuance are the only +installation work. Building the artifact and registering its trusted manifest are +release-management operations and should not be delegated to the device operator. + +The verified copy-paste command sheets are: + +- `docs/remote-worker-cheatsheet-windows-ru.md` +- `docs/remote-worker-cheatsheet-linux-ru.md` +- `docs/remote-worker-cheatsheet-docker-ru.md` + +### Windows portable client + +Build the pinned amd64 package in the isolated checkout when producing a release: + +```powershell +New-Item -ItemType Directory -Path dist -Force | Out-Null +python -B app/worker_package_builder.py windows ` + --project-root . ` + --output dist/truf-worker-windows-x86_64 ` + --archive dist/truf-worker-windows-x86_64.zip ` + --cache build/worker-cache +``` + +Verify the ZIP and adjacent release JSON through the release process. On the +client, extract to a private local directory and run `prepare-worker.ps1` once to +replace inherited ACLs. Create `worker-install.yaml` in that private directory and +put only `server`, `token`, and `parallelism` in it: + +```powershell +.\prepare-worker.ps1 +.\truf-worker.cmd install --config .\worker-install.yaml +Remove-Item -LiteralPath .\worker-install.yaml +.\truf-worker.cmd doctor +.\truf-worker.cmd start --startup-timeout 30 +.\truf-worker.cmd status +``` + +By default, state and data are below the current user's `LOCALAPPDATA`. The +package verifies its manifest and application files before launch and bundles +Python, Git, TruffleHog, detector policy, and locked dependencies. There is no +automatic updater. Use the same Windows account for installation and operation; +the supervisor instance and private state belong to that account. + +### Native Linux portable client + +Install the reviewed package beneath a root-owned path such as +`/opt/truf-worker`, then run `sudo ./prepare-worker.sh` from that directory as the +worker OS user. The preparation keeps native Git and TruffleHog immutable and +root-owned while making application code exact-private to the worker user. It +does not install a systemd unit. Use the private YAML installation and lifecycle +commands in `docs/remote-worker-cheatsheet-linux-ru.md` from that package root. + +### Linux client image + +Build the worker-only image for the target architecture through the reviewed +`worker` target. For x86-64: + +```sh +docker build --target worker -t truf-remote-worker:linux-x86_64 . +``` + +Install one device into a private persistent volume. Put the server, token, and +parallelism in a mode-0600 `worker-install.yaml`, pass it on standard input to the +short-lived install container, and delete it after success: + +```sh +docker volume create truf-worker-device-a-data +chmod 600 worker-install.yaml +docker run --rm -i \ + --mount type=volume,source=truf-worker-device-a-data,target=/data \ + --env XDG_DATA_HOME=/data/client --env XDG_STATE_HOME=/data/state-base \ + truf-remote-worker:linux-x86_64 install \ + --config - < worker-install.yaml +rm -- worker-install.yaml + +docker run --rm \ + --mount type=volume,source=truf-worker-device-a-data,target=/data \ + --env XDG_DATA_HOME=/data/client --env XDG_STATE_HOME=/data/state-base \ + truf-remote-worker:linux-x86_64 doctor --json +``` + +Run the installed supervisor in the foreground under Tini while Docker supplies +detachment and restart policy: + +```sh +docker run --detach --name truf-worker-device-a --restart unless-stopped \ + --read-only --cap-drop ALL --security-opt no-new-privileges --pids-limit 256 \ + --tmpfs /tmp:rw,nosuid,nodev,noexec,size=128m,mode=1777 \ + --mount type=volume,source=truf-worker-device-a-data,target=/data \ + --env XDG_DATA_HOME=/data/client --env XDG_STATE_HOME=/data/state-base \ + truf-remote-worker:linux-x86_64 run +``` + +The image entrypoint supplies `python -u -I -S -B` and the integrity-checking +bootstrap. It contains no PostgreSQL client, server runtime, provider +implementations, or detailed keycheck code. + +Ordinary private OS storage is supported on both platforms. Application-layer +encryption of the local workspace is not required. Operators may still use host +full-disk encryption according to their own endpoint policy; this is not a TRUF +protocol requirement. + +## Daily worker operation + +The commands below use `truf-worker.cmd` on Windows. On Linux without Docker, use +the generated `truf-worker` launcher. For a running Docker worker, define a local +helper that executes the same package bootstrap inside its container: + +```sh +workerctl() { + docker exec truf-worker-device-a /usr/local/bin/python3 -u -I -S -B \ + /opt/truf-worker/app/remote_worker_bootstrap.py -- "$@" +} +``` + +Use these commands for normal operation: + +| Command | Purpose | +| --- | --- | +| `truf-worker status` | One current human-readable supervisor and slot snapshot. | +| `truf-worker status --json` | Versioned snapshot for automation. | +| `truf-worker attach --follow-seconds 300` | Follow the verified running instance for a bounded interval; detaching does not stop it. | +| `truf-worker watch --follow-seconds 300` | Human live-status alias over the same verified attach stream. | +| `truf-worker attach --ndjson --follow-seconds 300` | Machine-readable bounded event stream. | +| `truf-worker logs --tail 200` | Read bounded rotating supervisor logs. | +| `truf-worker logs --follow --follow-seconds 300` | Follow logs for a bounded interval. | +| `truf-worker history --limit 50` | Show terminal assignment outcomes and durations. | +| `truf-worker history --reservation --json` | Retrieve one assignment's terminal local record. | +| `truf-worker doctor --json` | Validate package identity, directories, configuration, retention, instance state, and runtime prerequisites. | + +Run `status` first when investigating. It reports configured/occupied slots, +current phase, phase age, scan deadline, assignment time remaining, child state, +last progress age, backoff/idle reason, pending retention data, and progress-outbox +cursor. It does not invent percentage completion. + +### Phases and deadlines + +| Phase | Operator interpretation | +| --- | --- | +| `idle`, `claiming` | Slot is available or asking the server for work. | +| `assigned` | Immutable assignment identity and deadlines are persisted locally. | +| `waiting_permit` | Assignment owns a slot but is waiting for the shared scanner permit. This time counts against the scan-stage deadline. | +| `preparing`, `resolving`, `downloading`, `cloning` | Source-specific preparation before scanning. | +| `scanning` | Scanner process tree is active. | +| `filtering`, `cleaning`, `bundling` | Findings are converted, work is cleaned, and deterministic result bytes are staged. These phases remain inside the hard scan-stage deadline. | +| `uploading`, `awaiting_receipt` | Staged result is being transferred or waiting for authoritative server acknowledgement. | +| `backoff` | A bounded retry delay is active; inspect the reason and next-claim time. | +| `draining`, `stopped` | No new local work is starting; existing work is resolving or shutdown completed. | + +The scan-stage deadline starts before `waiting_permit` and covers preparation, +provider access, scanning, filtering, cleanup, and bundle staging. Crossing it +terminates the contained runner tree and produces a normal phase-specific timeout +result while the assignment upload window remains available. The server's +assignment deadline is fixed at issue time and is never renewed by progress, +polling, restart, or upload retries. `status` therefore presents both deadlines +separately. + +### Diagnostics and retained evidence + +`history` identifies the terminal receipt, prebundle, timeout, stale, or recovery +outcome. `logs` shows supervisor operation; diagnostic records carry the stable +phase/category/code and optional body/log material. The private state tree stores: + +```text +events/worker-events.jsonl +history/worker-history.jsonl +diagnostics/YYYY-MM-DD//.json +diagnostics/YYYY-MM-DD//.body +diagnostics/YYYY-MM-DD//.log +logs/worker.log +``` + +Diagnostic JSON records original/stored sizes, hash, encoding, and truncation +state. A missing or truncated body must not be described as complete. The server +admin detail view separately shows the ordered progress/receipt/ingestion/ +settlement/projection timeline and canonical diagnostic snapshot. Assignment +transport outcome, scan outcome, and diagnostics are independent fields. + +### Graceful stop and drain + +Request local stop before maintenance. The supervisor closes new claims for this +worker, leaves authentication and the Worker API available while existing work and +pending uploads resolve, and requires a clean drain receipt: + +```powershell +.\truf-worker.cmd stop --timeout 120 --json +``` + +For Docker, run `workerctl stop --timeout 120 --json`; the foreground supervisor +then exits and Tini returns the shutdown code to Docker. A successful shutdown +receipt reports `drained: true` and `exit_code: 0`. Do not interpret `docker stop`, +process termination, or loss of contact as assignment cancellation. + +### Recovery + +After an OS restart, network outage, server outage, or unclean process exit: + +1. Preserve the entire private state/volume; do not remove work, bundle, event, + history, or control files. +2. Run `doctor --json` and `status --json` from the exact installed package. +3. Start the same package with `start` on Windows/Linux, or restart the same Docker + container/volume. The supervisor replays its journal and recovers assigned, + staged, upload-retry, and awaiting-receipt slots. +4. Use `attach --ndjson --follow-seconds 300` and server assignment detail to + distinguish active recovery from backoff. +5. Keep the same device identity and state until recovered slots and + accepted-but-not-ingested bundles are reconciled. Escalate only if the instance + is unverifiable, a fixed assignment deadline has passed without server recovery, + or repeated startup validation fails. + +Re-uploading identical accepted bytes returns the original receipt. Conflicting +bytes or stale ownership are rejected; never delete local bytes merely to silence +that signal. + +### Update and rollback + +1. Request local graceful stop, reconcile unresolved/precommit work, and obtain a + successful clean drain receipt. +2. Retain the complete state tree and previous exact artifact/digest. +3. Verify the new ZIP/image and its registered server manifest. On Windows run + `prepare-worker.ps1`, then `doctor`; for Docker recreate only the container and + mount the same volume. +4. Start at parallelism `1`, confirm package identity/contact and one complete + accepted-ingested-projected assignment, then restore the intended cap. +5. If validation fails, stop and return to the previous exact artifact with the + same state. Additive server progress/diagnostic records need no rollback. + +Do not replace binaries beneath a running supervisor or switch packages while an +assignment runner is active. + +### Device removal + +1. Stop every device that uses the identity locally; leave authentication valid + while all unresolved assignments and precommit bundles reach zero. +2. Stop gracefully and retain the shutdown receipt and terminal history. +3. Revoke the device and disable its user only if that user is not shared by an + active device. +4. Confirm the old token no longer authenticates and no authoritative recovery + remains. +5. Remove the container/package. Remove its private volume/state only after the + server reconciliation evidence is retained and no rollback requires it. + +## Server tuning + +Make tuning changes in the reviewed private runtime configuration, not on the +client. Keep the raw worker service on loopback/private addressing and expose it +only through certificate-validating Caddy HTTPS. + +| Control | Meaning | +| --- | --- | +| Client `--parallelism N` | Maximum locally occupied slots, one claim per free slot. | +| Typed admin assignment cap | Atomic positive active-assignment cap for one user across all devices; lowering it does not cancel existing assignments. | +| `assignment_ttl_seconds` | Fixed server-clock lifetime covering download, scan, and upload; default `86400`. API contact and restart do not renew it. | +| `bundle_body_timeout_seconds` | Upload body deadline; default `1800`. The assignment lifetime must exceed the largest configured source scan timeout plus this value plus 60 seconds. | +| `json_body_timeout_seconds` / `body_idle_timeout_seconds` | Request and idle transport bounds; defaults `60` and `30`. They do not renew ownership. | +| `reaper_interval_seconds` / `reaper_batch_size` | Expired-assignment recovery cadence and bounded batch; defaults `60` and `1000`. | +| `limit_concurrency` | Worker API request concurrency bound, not a replacement for user quotas; default `64`. | + +Shortening the fixed lifetime can reject a valid long scan or upload and allow a +second physical execution after recovery. Lengthening it holds user quota, +target ownership, and dependent leases longer after a lost client. Tune it from +observed end-to-end duration plus upload headroom, not from HTTP polling cadence. + +## Typed administration + +Use only the authenticated random-prefix admin page configured by the production +edge. It provides CSRF/Origin-checked typed operations to create/enable/disable +users, set assignment caps, issue/rotate/revoke/unrevoke device tokens, and +requeue selected deferred queue IDs. It deliberately provides no shell or generic +supervisor command. + +An issued or rotated token is shown once. Rotation replaces the stored digest, so +the old token stops authenticating; update that device without copying the token +to other devices. Rotation does not clear an existing revoked state. Revocation +blocks further API authentication but does not invent cancellation for assigned +work; plan for outstanding work to be completed before revocation or recovered at +its fixed expiry. Use local graceful stop to close claims before planned device +maintenance or removal while preserving authentication for pending work. + +For an admin IP ban, use SSH and fail2ban first so fail2ban and Caddy agree: + +```sh +sudo fail2ban-client set truf-admin-auth unbanip 203.0.113.10 +sudo /usr/local/sbin/truf-caddy-admin-denylist status +sudo /usr/local/sbin/truf-caddy-admin-denylist expire +``` + +If fail2ban is unavailable, use the explicit admin-only updater: + +```sh +sudo /usr/local/sbin/truf-caddy-admin-denylist unban 203.0.113.10 +``` + +These commands change only the admin-route matcher. Never replace them with a +global port 443 firewall unban/ban. The full damaged-snippet recovery procedure +is in `deploy/edge/README.md`. + +## Accepted custody and expiry + +`accepted` means the server validated the canonical v2 `.trb`, durably published +it, and persisted ready/recovery state before returning a receipt. It does not +mean ingestion, projection, candidate handling, or detailed keycheck has +finished. Track `accepted` and `ingested` separately in the typed admin view. + +The client keeps assignment identity and pending bytes until authoritative +acknowledgement. Retrying identical accepted bytes returns the original receipt, +including after ingestion, spool cleanup, restart, or the former deadline; +conflicting bytes are rejected. An interrupted, invalid, expired-before-first- +acceptance, or stale upload is not a successful scan. + +An unfinished assignment expires at the original server-set deadline, normally +24 hours after issue. The periodic recovery pass reconciles quota, credits, +target ownership, and dependent plan/blob leases through existing retry policy. +Already accepted ready bundles are not requeued as unfinished. A crashed client +can therefore delay work for about a day, while a partitioned client can continue +physical scanning after the server has expired and reissued the target. Ownership +fencing guarantees one authoritative acceptance, not exactly-once physical work. + +## Staged rollout and rollback + +1. Obtain separate review for production schema/deployment changes. Drain protocol-1 work before replacing packages. Keep worker API and admin disabled while configuring exact protocol-2 capability profiles, task-specific auth entries, trusted package manifests, edge origin/marker, and per-user caps. +2. Pass the main Docker, packaged Windows/Linux, and edge gates with their pinned artifacts. Do not infer readiness from unit mocks or one platform. +3. Enable a small canary: one user, one device, cap `1`, and client `N=1`. Confirm assignment, acceptance, ingestion, projection, detailed server keycheck, expiry, and safe logs before increasing either cap or device count. +4. Expand caps and clients in stages while comparing unfinished/completed/failed/expired counts, last authenticated contact, accepted-versus-ingested state, durations, capacity, and stale/duplicate events. Contact age alone is not a liveness failure while a client holds long-running work and has not yet returned to claim polling. +5. To drain, request local graceful stop on every affected device. Leave authentication and the Worker API available so pending uploads and terminal reports can resolve. Wait for unfinished assignments to complete or pass through fixed-expiry recovery, and separately reconcile accepted bundles awaiting ingestion. +6. After the drain is authoritative, disable remote admission and return scheduling to local-only execution. Then revoke unused device tokens if required. Retain reservation metadata, accepted receipts, bundles/recovery state, and schema until reviewed reconciliation is complete. +7. Do not drop worker metadata, clear spool state, rotate/revoke tokens before a drain, switch server binaries underneath unfinished work, or treat a service stop as cancellation. The local execution path remains available and must not depend on a remote worker. diff --git a/docs/remote-worker-quickstart-ru.md b/docs/remote-worker-quickstart-ru.md new file mode 100644 index 0000000..9c875d1 --- /dev/null +++ b/docs/remote-worker-quickstart-ru.md @@ -0,0 +1,98 @@ +# TRUF worker: установка и работа + +Worker получает задания от сервера, скачивает публичные targets и отправляет +только результат сканирования. Для каждого компьютера или Docker volume нужен +отдельный device token. + +Сервер: `https://pregnant.horsecock.store` + +## Что получить у администратора + +1. Проверенный artifact для своей платформы и соседний файл с SHA-256. +2. Одноразово показанный device token. Не отправляйте его в чат, лог или снимок + экрана. +3. Подтверждение, что server-side User и Device включены и artifact зарегистрирован. + +Администратор создаёт отдельные User и Device на странице `Workers / Dispatch`, +выдаёт token и назначает положительный assignment cap. Один token нельзя +использовать на нескольких устройствах. + +## Приватный install YAML + +Token не нужно передавать в аргументах процесса. Создайте локальный +`worker-install.yaml` в приватной папке через текстовый редактор: + +```yaml +server: https://pregnant.horsecock.store +token: PASTE_DEVICE_TOKEN_HERE +parallelism: 1 +``` + +`parallelism` задаёт число локальных занятых slots и должен быть от 1 до 128. +После `install` worker сохраняет настройки в своём приватном +`worker.config.json`; исходный YAML нужно удалить. При обычных `start`, `stop`, +`status`, `attach` и `watch` token больше не вводится. + +## Шпаргалки + +- [Windows](remote-worker-cheatsheet-windows-ru.md) +- [Linux без Docker](remote-worker-cheatsheet-linux-ru.md) +- [Docker Engine / Docker Desktop](remote-worker-cheatsheet-docker-ru.md) + +Каждая шпаргалка начинается из корня распакованного artifact или Compose bundle +и содержит проверенные команды установки и lifecycle. + +## Что делают lifecycle-команды + +| Команда | Результат | +| --- | --- | +| `start` | Запускает установленный worker в фоне; повторный запуск не создаёт второй instance. | +| `stop --timeout 120` | Локально закрывает новые claims, завершает текущую работу и требует clean drain receipt. | +| `status` | Показывает instance, slots, текущие phases, deadlines и retained state. | +| `attach` | Подключает live status/event view; `q` или `Ctrl-C` только отсоединяет. | +| `watch --follow-seconds 300` | Запускает bounded live status view и затем отсоединяется. | +| `logs --follow --follow-seconds 300` | Показывает bounded live event/log stream. | +| `doctor --json` | Проверяет package, config, пути, native tools, TLS endpoint и singleton state. | + +`stop` управляет только выбранным локальным worker. Менять server assignment cap +для обычного stop, restart или обновления не требуется. Не завершайте процесс и +не удаляйте state, пока clean drain receipt не подтверждён. + +## Capacity и backpressure + +- Один result bundle имеет hard limit 64 MiB. +- При выдаче remote assignment server резервирует baseline 2 MiB для bundle и + 2 MiB для projection; это не новый hard limit. +- Валидный результат больше baseline атомарно расширяет reservation по фактическому + размеру. При временной нехватке capacity worker повторяет upload позже. +- Server допускает не более 50 unresolved remote assignments глобально и + одновременно применяет положительный per-user cap. Фактический предел равен + меньшему из доступной capacity, global limit и user cap. +- Client `parallelism` ограничивает только локальные slots и не повышает server cap. + +## Частые состояния + +- `idle` или `claiming`: slot свободен или запрашивает задание. +- `downloading`, `cloning`, `scanning`: выполняется задание. +- `uploading`, `awaiting_receipt`: результат отправляется или ждёт подтверждения. +- `backoff`: временная ошибка; причину и следующую попытку показывает `status`. +- `draining`: новые локальные claims закрыты, текущая работа завершается. +- `stopped`: clean shutdown завершён. + +При проблеме сохраните вывод `doctor --json`, `status --json` и +`logs --tail 200`. Никогда не прикладывайте device token, install YAML или +приватный `worker.config.json`. + +## Безопасное обновление + +1. Выполните локальный `stop --timeout 120 --json` и получите `drained: true`, + `exit_code: 0`. +2. Сохраните предыдущий точный artifact/image и весь state/volume. +3. Проверьте SHA-256 и package identity новой версии. +4. Windows/Linux: распакуйте новую версию в отдельную папку. Docker: загрузите + новый image и пересоздайте только container с прежним volume. +5. Выполните `doctor`, `start`, `status` и bounded `watch`. + +Не удаляйте локальные `state`, `work`, `bundles`, `events`, `history` или Docker +volume при ошибке и не используйте `docker compose down --volumes`. Они нужны +для безопасного продолжения и authoritative receipt recovery. diff --git a/docs/session-handoff/CURRENT_STATE.md b/docs/session-handoff/CURRENT_STATE.md new file mode 100644 index 0000000..9c87559 --- /dev/null +++ b/docs/session-handoff/CURRENT_STATE.md @@ -0,0 +1,78 @@ +# Current State + +Updated: 2026-09-26 +Workspace: `D:\truf-workers` + +## STATUS: Primary Objective Complete + +Extended production validation with exactly one native Windows worker slot and +one WSL/Docker worker slot is complete. Run +`e6ac8aec-ec20-4ba4-a924-abe5ee95d82c` exercised discovery, assignment, worker +execution, progress, diagnostics, bundle upload and ingestion, projection, +capacity release, and scheduled keycheck. + +The authoritative public-safe result is +`docs/extended-live-validation-2026-09-26.md`. The machine-readable aggregate +manifest is +`build/extended-live-validation/runs/e6ac8aec-ec20-4ba4-a924-abe5ee95d82c/verification-manifest.json`. + +Conclusion: `pass_with_documented_deviations`. + +## STATUS: Terminal Validation Cut + +- Controller revision 143 was sealed with dispatch and discovery paused and the + validation drain complete. +- All 314 reservations were terminal: 309 acknowledged and 5 refunded. +- All 309 accepted bundles were ingested, projected, settled, and released. +- No unresolved reservation, live assignment, pre-commit bundle, capacity use, + publication outbox item, quarantine row, waiting lock, or blocker remained. +- All 348 projection jobs completed and released in one attempt without error. +- The captured 39-candidate keycheck cohort completed, linked, projected, and + released; the global queue had no pending, leased, or deferred candidate. +- Server and worker hash chains and all referenced objects verified. + +## STATUS: Production Restored + +Production restore completed through compare-and-swap revisions 143 to 146: + +1. Cancel the validation drain: 143 to 144. +2. Resume discovery: 144 to 145. +3. Resume dispatch: 145 to 146. + +Last verified state: + +- Controller revision 146, actor `validation-final-restore`. +- Dispatch open, discovery open, drain normal. +- Exactly one live assignment on Windows and one on WSL/Docker. +- Runtime healthy with zero validation-run restarts. +- Edge running with zero validation-run restarts. +- Core services and discovery producers healthy. + +This is a recorded final snapshot, not a substitute for a fresh query in a +future operational session. + +## STATUS: Documented Deviation + +The authenticated `recheck all` operation was accidentally used instead of +rechecking only the two pending candidates, causing 35 broad provider checks +contrary to `KEYCHECK-001`. All 35 were processed, none was skipped, the command +returned success, and all effects settled. Do not repeat this probe. + +## STATUS: Remaining Defects + +- Windows GitLab filename-too-long checkout recovery is local and not deployed. +- Invalid API-key classification is fixed locally and not deployed. +- A roughly 20-second WSL clock-domain monotonic failure remains open. +- Generic `WorkerContractError` diagnostics lose structured field detail. +- Monitor aggregate queries produced six recovered statement timeouts under + load. + +Focused regression tests for the local checkout, long-path, and credential +changes passed: 145 tests in 155.26 seconds. + +## NEXT: No Validation Work Pending + +Do not repeat the completed live validation or broad keycheck. Future work, if +requested, is limited to reviewing and deploying the documented local fixes, +investigating the remaining defects, and archiving the completed OpenSpec +change. Do not archive OpenSpec without explicit user instruction. diff --git a/docs/session-handoff/DECISIONS.md b/docs/session-handoff/DECISIONS.md new file mode 100644 index 0000000..773c712 --- /dev/null +++ b/docs/session-handoff/DECISIONS.md @@ -0,0 +1,94 @@ +# Durable Decisions + +Updated: 2026-09-25 + +## DECISION: RAW-001 - Raw Analysis Is Mandatory + +For analysis and live testing, it is forbidden to design or add token hiders, +maskers, redactors, secret-scrubbers, post-hoc sanitizers, lossy summaries, or +similar mechanisms that conceal the values being validated. + +- Inspect and retain the original raw payloads, targets, findings, credentials, + provider responses, diagnostics, logs, database rows, and worker records needed + to prove correctness. +- The monitor must not replace, hash-only, truncate, partially reveal, or rewrite + a value before the analysis has consumed and preserved the raw record. +- If a UI already has a reveal control, use the revealed raw value for the test; + do not build another concealment layer. +- Raw evidence must remain in private local/protected storage and must not be + published to a public report, issue, commit, or chat excerpt. This storage rule + is not permission to hide data from the analysis itself. +- A later public/operator report may reference counts and hashes, but it must be + derived only after raw correctness has been checked. + +This decision supersedes any old handoff wording that instructed the testing +session to analyze only sanitized aggregates. Historical sanitized reports remain +valid as reports; they are not sufficient evidence for the new run. + +## DECISION: ENV-001 - Production Target + +- Use the configured SSH server named `sec` only. +- Never call, connect to, or mutate the configured server named `prod`. +- Workspace is `D:\truf-workers`. +- Do not run the inherited native runtime launchers in this source-only workspace. +- Do not mount or mutate unrelated `D:\truf` runtime data. + +## DECISION: RUN-001 - Dual Worker Bounds + +- Native Windows: exactly one worker slot/thread. +- Docker under WSL: exactly one worker slot/thread. +- Expected maximum combined worker concurrency: two. +- Do not increase caps or parallelism to accelerate the observation window. +- Waits of up to ten minutes are allowed; several hours of observation are + explicitly authorized. + +## DECISION: KEYCHECK-001 - Scheduled Validation + +- Keycheck may be enabled for the run every 30 minutes (`1800` seconds). +- Verify candidate leases, provider execution, append-only results, current-state + selection, projection jobs/appends, and capacity release from raw records. +- Provider probes may have real external effects or cost; do not silently widen + service args or recheck policy beyond the active configuration. + +## DECISION: CONFIG-001 - Stale Candidate Must Not Be Applied + +Do not apply the stale config candidate with SHA-256 +`c0966cac4f7f0610a813fa8732e91953f2e3e838ad880f91fd1a9437096925c7`. +It was based on an older active hash and would reduce +`global.keycheck_queue_max_items` from the retained `8192` to `4096`. + +Always fetch the current active config identity and use the authenticated +fresh-hash/CAS workflow for any 1800-second keycheck edit. + +## DECISION: ARCH-001 - Worker Authority + +- The worker is final authority for real provider access. +- Server planning may bind immutable Git/Docker identity but must not add + per-target preflight/provider-access proof machinery. +- Do not add credential sandboxes, environment rewriting, durable access proofs, + or security-specific infrastructure without a separate explicit user decision + and OpenSpec requirement. +- Prefer bounded direct error classification. Authentication/access/not-found is + permanent when target-scoped; rate limits, network failures, and provider 5xx + are retryable. + +The complete engineering decision is in `AGENTS.md`. + +## DECISION: CHANGE-001 - Repository and OpenSpec + +- This repository has no baseline commit; the full tree appears untracked. + Never use Git to revert or clean files and never treat `git diff` as complete. +- Preserve unrelated files and evidence directories. +- `add-worker-operator-experience` is complete but must not be archived without + an explicit request. +- Do not repeat the already completed 295-assignment production validation unless + a fresh verification proves its retained evidence invalid. + +## DECISION: CONTEXT-001 - Session Continuity + +- Use these files for continuation instead of recursive DCP summaries. +- Do not proactively invoke conversation compression in the new session. +- Batch searches and process large evidence in tools; avoid injecting raw + multi-megabyte files into the conversation context. +- The prohibition on injecting large evidence into chat does not permit masking + or omitting it from the private analysis artifact. diff --git a/docs/session-handoff/EVIDENCE_INDEX.md b/docs/session-handoff/EVIDENCE_INDEX.md new file mode 100644 index 0000000..40fd314 --- /dev/null +++ b/docs/session-handoff/EVIDENCE_INDEX.md @@ -0,0 +1,140 @@ +# Evidence Index + +Updated: 2026-09-26 + +This file maps facts to their existing source. Do not duplicate the underlying +evidence in handoff prose. + +## STATUS: Authoritative Reports + +- `docs/extended-live-validation-2026-09-26.md` + - Final public-safe report for the two-slot extended live run, including + settlement, keycheck, evidence integrity, deviation, restore, and open + defects. +- `build/extended-live-validation/runs/e6ac8aec-ec20-4ba4-a924-abe5ee95d82c/verification-manifest.json` + - Machine-readable derived aggregates, classifications, evidence hashes, + terminal state, post-restore state, and test result. +- `docs/worker-operator-experience-live-trace-2026-09-25.md` + - 295-assignment Windows/Linux cohort, aggregate outcomes, recovery, admin UI, + pipeline state, tests, and evidence hashes. +- `docs/worker-operator-experience-validation-2026-09-24.md` + - Earlier bounded production/package acceptance and operator validation. +- `docs/remote-worker-operations.md` + - Canonical worker install, lifecycle, diagnostics, drain, update, and removal. +- `WORKER_OPERATOR_EXPERIENCE_HANDOFF.md` + - Historical implementation and reproducible artifact detail. Its stop point + predates the final live trace and defect-fix rollout. +- `openspec/changes/add-worker-operator-experience/tasks.md` + - Current completion authority: all 27 tasks checked. + +## RAW: Private Live Evidence + +- `build/extended-live-validation/runs/e6ac8aec-ec20-4ba4-a924-abe5ee95d82c` + - Local run root with compact server artifacts and complete worker evidence. + - Server NDJSON SHA-256: + `a6beb1864c540fc5f22b2b647730a39e45dfddb10cf5ceff5e0181eb1a8f0bd8`. + - Worker NDJSON SHA-256: + `03a04a0da5a2db6bd02f2aaed41c191554b135ec287fbc1e2d9998d77da59898`. + - Server run SHA-256: + `5ae58df5f67a8d2a8d4e84d73262a6e7009e03dcbcc9ea176537b384d4e693c3`. +- Private production server evidence root: + `/var/lib/docker/volumes/truf-remote-server-data/_data/extended-live-validation/e6ac8aec-ec20-4ba4-a924-abe5ee95d82c` + - Complete server evidence, approximately 1.2 GiB. Analyze in place and do not + copy raw values into public documents. +- `build/live-trace-20260925/raw-evidence-final.json` + - Complete server cohort evidence. Historical recorded SHA-256: + `4071e38a1dec540663bc6febd9538ff54bedafa1627250933045beb8f7d09ec5`. +- `build/live-trace-20260925/raw-evidence-expanded.json` + - Expanded unrestricted evidence used for defect diagnosis. +- `build/live-trace-20260925/monitor-final.ndjson` + - Time-series server monitor. Historical SHA-256: + `e07507b0b8f0f86d1a1c7aade7186297c372b1ff177067bea19dfd219f5027a9`. +- `build/live-trace-20260925/windows-localappdata/TRUF/RemoteWorker` + - Final retained Windows worker state, events, history, logs, and inactive + abandoned roots. +- `build/live-trace-20260925/linux-worker-state-final.tar.gz` + - Final retained Linux worker state. Historical SHA-256: + `b64e81c90b232f46b400a63ed08f5660f46e34fedb1f67b64afba079d8d36364`. +- `build/live-trace-20260925/dockerhub-discovery-db-raw.json` +- `build/live-trace-20260925/dockerhub-unmasked-cycle-failure.json` +- `build/live-trace-20260925/dockerhub-discovery.log` + - Raw DockerHub retry defect evidence. + +These files may contain sensitive raw values. Analyze them in place; do not copy +their contents into a public document. + +## STATUS: Defects and Current Treatment + +- Extended-run open items: + - Windows GitLab filename-too-long checkout recovery is local and not + deployed. + - Invalid API-key classification is fixed locally and not deployed. + - A roughly 20-second WSL clock-domain monotonic failure remains open. + - Generic `WorkerContractError` diagnostics lose structured field detail. + - Six monitor aggregate-query statement timeouts recovered during the run. +- `KEYCHECK-001` deviation: + - An unintended authenticated broad recheck processed 35 provider checks + instead of only the two pending candidates. All effects settled; do not + repeat this probe. +- `docs/defect-windows-scan-timestamps-utc-2026-09-25.md` + - Still open. Local `app/scanner.py` continues to create naive timestamps. +- `docs/defect-dockerhub-discovery-retry-null-type-2026-09-25.md` + - Original report says local-only. Current source has the PostgreSQL + `CAST(? AS TEXT)` correction and the defect-fix rollout included it. Reverify + live retry coalescing during the extended run. +- `docs/defect-terminal-status-scan-deadline-readback-2026-09-25.md` + - Original report says open. Current source preserves durable deadlines with + `result.setdefault('deadlines', observability['deadlines'])`; included in the + defect-fix rollout. Reverify terminal readback. +- `docs/defect-worker-network-oserror-mislabeled-local-io-2026-09-25.md` + - Original report says open. Current source introduces `WorkerNetworkError` + before broad `OSError` classification; included in the defect-fix worker + artifacts. Reverify under a real transport failure. + +## STATUS: Defect-Fix Rollout Artifacts + +- `build/runtime-defect-fixes-v1/deploy-runtime.sh` + - Atomic runtime/manifests cutover and rollback logic. +- `build/runtime-defect-fixes-v1/windows-b.zip` +- `build/runtime-defect-fixes-v1/windows-b.zip.json` +- `build/runtime-defect-fixes-v1/linux-worker-package.json` +- `build/runtime-defect-fixes-v1/verify-production-worker.py` +- `build/runtime-defect-fixes-v1/restore-production-worker.py` + +Recorded identities: + +- Deployed runtime image: + `sha256:7d84fdf57a1cb9e6d38a571fbd3566b7549f1cda04ae02c4864cac70f74f2aaa`. +- Current WSL worker image: + `sha256:491b3a2343571072209a7f92e83399fe206006dcccf247a0c551c50fc9f35e30`. +- Rollback tag: `truf-local:runtime-pre-defect-fixes-v1`. +- New registered manifest file hashes from the deploy script: + - Linux: `b9d3594e4846a21ca12de5fc6973c04d9eea6f61fd0ecda83875426aa48c7b4b`. + - Windows: `4cc97d17c34f7d89150927719f131f426f1068ab99af7f3f2f156d8642ae539e`. + +## STATUS: Historical Accepted Artifacts + +The pre-defect-fix live trace used: + +- Windows package manifest: + `78a962b2bd3fa411413c79e9a8ffb021608a08ff020b1ad851f4505ea634b2b6`. +- Linux package identity: + `45588f2cf406b41b239cfa3b8a9dc83fe84b587229bc997b2729016e1f0dde42`. +- Linux image: + `sha256:3a088f5743121d823aae132234a29730a84339cecbfda5fc601e8e942f9948c3`. + +These remain valid historical evidence but are not the preferred artifacts for +the new defect-fix observation run. + +## VERIFY: Fast Orientation Commands + +Run from `D:\truf-workers`: + +```powershell +openspec list --json +git status --short --branch +python -B -m pytest tests/test_worker_api.py tests/test_worker_api_runtime.py tests/test_worker_assignment.py tests/test_worker_assignment_runner.py tests/test_worker_cli.py tests/test_worker_contracts.py tests/test_worker_local_state.py tests/test_worker_observability_db.py tests/test_worker_package.py tests/test_worker_runner_handoff_linux.py tests/test_worker_supervisor.py tests/test_remote_worker_db.py tests/test_scan_execution.py tests/test_admin_api.py -q +``` + +Do not use an unrestricted repository-wide pytest run as the release gate. Do +not use Git clean/reset/checkout in this uncommitted snapshot. diff --git a/docs/session-handoff/README.md b/docs/session-handoff/README.md new file mode 100644 index 0000000..12bc2ae --- /dev/null +++ b/docs/session-handoff/README.md @@ -0,0 +1,35 @@ +# Session Handoff Index + +Updated: 2026-09-26 + +This directory is the authoritative entry point for a new session working on +the live remote-worker validation. Read only these files first: + +1. `CURRENT_STATE.md` - where work stopped and the exact next actions. +2. `DECISIONS.md` - binding user decisions and operational constraints. +3. `EVIDENCE_INDEX.md` - existing reports, raw captures, artifacts, and hashes. + +The older root `WORKER_OPERATOR_EXPERIENCE_HANDOFF.md` is historical background. +It remains useful for implementation detail and artifact provenance, but its +"Immediate next actions" section is obsolete. + +## Knowledge Layout + +- A current fact has exactly one owner: `CURRENT_STATE.md`. +- A durable rule has exactly one owner: `DECISIONS.md`. +- Evidence is not copied into handoff prose; `EVIDENCE_INDEX.md` points to it. +- Dated reports are immutable history. Record later corrections here instead of + rewriting the original report. +- Replace stale current-state statements rather than appending contradictory + status paragraphs. + +Useful grep tags are `STATUS:`, `NEXT:`, `BLOCKER:`, `DECISION:`, `VERIFY:`, and +`RAW:`. + +## New Session Start + +Use this prompt: + +> Read `docs/session-handoff/README.md` and its three linked files. Continue the +> `NEXT:` work in `CURRENT_STATE.md` autonomously. Do not repeat completed live +> validation. Follow every `DECISION:` literally, especially RAW-001 and ENV-001. diff --git a/docs/worker-operator-experience-live-trace-2026-09-25.md b/docs/worker-operator-experience-live-trace-2026-09-25.md new file mode 100644 index 0000000..fc55be8 --- /dev/null +++ b/docs/worker-operator-experience-live-trace-2026-09-25.md @@ -0,0 +1,336 @@ +# Worker Operator Experience Live Trace - 2026-09-25 + +## Result + +The accepted Windows package and Linux image completed a fresh, concurrent live +cohort against the `sec` runtime. All 295 assignments were acknowledged, +ingested, projected, and settled. There were no unresolved assignments, +pre-commit bundles, expiries, pre-bundle failures, quarantine rows, append +failures, or publication-outbox rows at the final cut. + +The worker transport and server pipeline acceptance result is **pass**. Four +product defects and two operational warnings were found. The defects did not +invalidate the one-authoritative-acceptance, ingestion, projection, recovery, or +shutdown guarantees demonstrated by this cohort, but they remain release inputs +and are listed below. + +This report is sanitized. It intentionally excludes authentication values, +private routes, worker command lines, raw targets, raw findings, secrets, and +runtime configuration bodies. The protected evidence files referenced below +contain sensitive material and must not be published. + +## Validated artifacts + +The run used the previously accepted reproducible artifacts without modifying +their product code: + +| Platform | Accepted identity | +| --- | --- | +| Windows x86-64 | package manifest `78a962b2bd3fa411413c79e9a8ffb021608a08ff020b1ad851f4505ea634b2b6` | +| Linux x86-64 | package identity `45588f2cf406b41b239cfa3b8a9dc83fe84b587229bc997b2729016e1f0dde42` | +| Linux image | `sha256:3a088f5743121d823aae132234a29730a84339cecbfda5fc601e8e942f9948c3` | + +Both trusted manifests remained registered on the server. Each validation worker +ran one slot with local parallelism `1`. The measured combined concurrency +reached exactly `2`, proving concurrent accepted Windows and Linux execution +without increasing either worker's local parallelism. + +## Final cohort + +### Platform and source distribution + +| Worker | DockerHub | GitLab | Hugging Face | Total | +| --- | ---: | ---: | ---: | ---: | +| Windows | 58 | 57 | 51 | 166 | +| Linux | 50 | 30 | 49 | 129 | +| Total | 108 | 87 | 100 | 295 | + +Every history row ended with `bundle_accepted` and mapped to exactly one server +reservation and receipt. The Windows and Linux reservation sets were disjoint and +their union exactly matched the 295-row server cohort. + +### Scan and queue outcomes + +| Scan outcome | Count | +| --- | ---: | +| clean | 143 | +| degraded | 73 | +| error | 75 | +| found | 4 | + +| Queue disposition | Count | +| --- | ---: | +| done | 220 | +| deferred | 74 | +| failed | 1 | + +These are scanner/queue outcomes, not transport failures. All corresponding +result bundles were accepted and projected. In particular, timeout and scanner +error results remained normal uploadable terminal results. + +### Persisted detail + +The stable capture contains: + +- 295 admission intents, reservations, bundles, target scans, compatibility rows, + and completed scan projection jobs; +- 2,905 ordered progress events; +- 75 structured diagnostics; +- 129 normalized scan errors; +- 4 normalized findings, 4 stable UID mappings, and 4 bounded compatibility + payloads; +- 4 pending keycheck candidates linked to 2 normalized credentials; +- 374 appended projection records across 3 streams; +- 956 pipeline artifacts, all deleted by the final snapshot; and +- 19 successful typed runtime operations with 38 chained audit events. + +The diagnostic aggregate was: + +| Worker | Scanner result errors | Stage timeouts | Total diagnostics | +| --- | ---: | ---: | ---: | +| Windows | 55 | 8 | 63 | +| Linux | 1 | 11 | 12 | + +Of the Windows scanner-result errors, 54 were retryable and one was +non-retryable. The Linux scanner-result error was retryable. Complete safe +diagnostic envelopes, exception identities, bounded process-log representations, +occurrence times, receipt authority, and scan links were retained and checked. + +## End-to-end integrity checks + +`build/operator-experience-validation/analyze-live-trace-final-20260925.py` +executed 65,675 checks with zero failures. It verified, row by row: + +- admission, reservation, queue, bundle, scan, compatibility, and projection + foreign-key relationships; +- receipt, payload, bundle, event, device, and deadline identities; +- canonical SHA-256 values for execution snapshots, progress events, + diagnostics, compatibility metadata, plans, projection events, and finding + payloads; +- exact bounded reconstruction of the four compatibility findings, including + their original numeric detector identity and explicit null mapped fields; +- every normalized error, finding, keycheck candidate, credential reference, + projection append, capacity release, and deleted artifact; +- all 19 operation-to-audit pairs; and +- the complete 38-event audit parent/hash chain. + +`build/operator-experience-validation/analyze-worker-states-final-20260925.py` +executed a further 28,686 checks with zero failures. It parsed every retained +worker JSON/JSONL record and verified: + +- 166 Windows and 129 Linux history rows against the server receipts; +- 2,338 contiguous Windows events and 1,672 contiguous Linux events; +- receipt payload, bundle, event, acceptance-time, and reservation identities; +- clean local shutdown with `drained=true`, `exit_code=0`, and empty progress + outboxes; and +- no active work root in either final worker snapshot. + +The two analyzers therefore executed 94,361 deterministic checks without a +failure. + +## Recovery and shutdown + +The validation exercised durable recovery rather than only clean executions: + +- transient server `502` responses were retained in both local worker logs and + recovered without duplicate authoritative acceptance; +- one Linux assignment survived a worker stop in the persistent volume, resumed + after restart, produced one accepted result, and released all capacity; +- a transient assignment-status network failure retried the same durable + assignment without rescanning or data loss; and +- both workers then drained and stopped cleanly. + +The final local states were: + +| Worker | State | Drained | Exit | Pending outbox | +| --- | --- | --- | ---: | ---: | +| Windows | stopped | true | 0 | 0 | +| Linux container | exited | true | 0 | 0 | + +Retained `work/abandoned` roots are inactive evidence governed by normal worker +retention. They are not active assignments. + +## Worker API validation + +Authenticated worker API validation covered: + +- device identity, package-manifest trust, and assignment-cap enforcement; +- claim, reservation replay, assignment status, and immutable execution snapshot; +- monotonic progress submission and latest-progress readback; +- bounded diagnostic body and process-log payloads; +- durable bundle upload, idempotent receipt replay, and accepted resolution; +- local restart recovery and terminal history; +- server ingestion, normalized scan authority, compatibility reconstruction, and + projection completion; and +- terminal capacity release and zero unresolved work. + +The API preserved receipt, payload, scan-event, diagnostic, and reservation +identities throughout the cohort. A separate terminal readback defect affecting +only `scan_deadline_at` is documented below. + +## Admin UI and operator workflow + +The authenticated admin UI was exercised through a browser across the complete +operator surface: + +- Workers / Dispatch: control state, users, devices, assignment caps, enable, + disable, revoke, unrevoke, and bounded worker detail; +- Overview: runtime health, producer lifecycle, pipeline workers, leases, queue + counts, capacity, controls, recent operations, and duration groups; +- Search: bounded assignment, scan, finding, error, diagnostic, and progress + lookup with safe empty and populated states; +- Supervisor: source start, restart, stop, lifecycle, and safe error rendering; +- Logs: bounded source/component/level/time filters and empty-result handling; +- Config and Secrets: active identities, stale-candidate warning, redacted + projections, validation, and non-secret operation results; +- Files: bounded listing, file identity/hash verification, and safe download + behavior; +- Operations: accepted/running/succeeded projections and filters; and +- Audit: accepted/succeeded event pairs, pagination, actor/action filters, and + parent/hash continuity. + +The final UI snapshot showed: + +- Supervisor `ACTIVE` and PostgreSQL `READY`; +- result ingester, JSONL projector, janitor, and worker API all running; +- ingester and projector leases ready; +- control revision `126`, discovery and dispatch open, drain state normal; +- zero active assignments and zero pre-commit bundles; and +- zero quarantined queue rows. + +The current DockerHub producer safe state remained `runtime_error`; it is the +known retry-coalescing defect plus unavailable credential pool described below, +not an unclassified new failure. + +## Server and pipeline final state + +The final server snapshot at 2026-09-25 14:59 UTC confirmed: + +- 295 issued, 295 accepted, 295 ingested, 295 projected, and 295 settled; +- 0 unresolved, expired, pre-bundle-failed, pre-commit, quarantine, and drain + blockers; +- bundle, projection, and quarantine capacity at zero; +- keycheck capacity at 137 items / 27,262,866 bytes, representing real pending + work rather than leaked assignment or projection capacity; +- runtime container healthy with zero restarts and no OOM; +- edge container running with zero restarts and no OOM; and +- dashboard health endpoint returning `200 ok`. + +## Defects found + +### 1. Windows scanner timestamps lose their UTC offset + +Status: open in the accepted Windows package. + +Naive local scanner timestamps are relabeled as UTC, producing approximately a +three-hour future displacement on the validation host. Progress transport and +monotonic durations remain correct, but diagnostic ordering, time filters, and +scan start/end instants are wrong. + +Detailed evidence and correction: +`docs/defect-windows-scan-timestamps-utc-2026-09-25.md`. + +### 2. DockerHub retry coalescing fails with a null retry time + +Status: open in the live runtime; fixed locally but not deployed. + +PostgreSQL cannot infer the type of a nullable retry placeholder while +coalescing an existing row, raising SQLSTATE `42P18`. The managed producer masks +that exception as a generic delegation failure. A minimal local correction casts +the placeholder to text, and regression coverage now exercises same-row null-time +coalescing. + +Detailed evidence and local fix: +`docs/defect-dockerhub-discovery-retry-null-type-2026-09-25.md`. + +### 3. Terminal status can lose the durable scan deadline + +Status: open in the live runtime. + +A later progress event with a null scan deadline can replace the durable +receipt's concrete immutable deadline in authenticated terminal status readback. +The persisted receipt and all other identities remain correct. + +Detailed evidence: +`docs/defect-terminal-status-scan-deadline-readback-2026-09-25.md`. + +### 4. Network `OSError` is mislabeled as local I/O + +Status: open in the accepted workers. + +The broad safe-summary branch classifies socket/transport `OSError` as +`local I/O operation failed`. Recovery worked and no data was lost, but the +operator message incorrectly points toward local storage. + +Detailed evidence: +`docs/defect-worker-network-oserror-mislabeled-local-io-2026-09-25.md`. + +## Operational warnings + +- The server root filesystem was approximately 90% used with roughly 1 GiB free. + Containers remained healthy, but capacity should be reclaimed or expanded. +- The DockerHub credential pool was independently unavailable: ten credentials + were invalid and the remaining entry was rate-limited. Deploying the SQL fix + preserves retries correctly but cannot make an unavailable credential pool + healthy. + +## Tests and specification gates + +The retained gates are: + +- worker/operator focused matrix: `355 passed, 3 skipped`; +- admin/runtime focused matrix: `124 passed, 2 warnings`; +- DockerHub incremental discovery matrix: `27 passed`; +- SQLite retry lifecycle regression: `1 passed`; +- corrected PostgreSQL statement verified transactionally against the live schema + and rolled back; and +- OpenSpec strict validation passed with all 27 implementation tasks complete. + +The disposable PostgreSQL integration test remains skipped locally because no +disposable DSN was configured. The live transactional SQL verification did not +persist a change. + +## Evidence manifest + +| Artifact | Bytes | SHA-256 | +| --- | ---: | --- | +| `build/live-trace-20260925/raw-evidence-final.json` | 10,030,750 | `4071e38a1dec540663bc6febd9538ff54bedafa1627250933045beb8f7d09ec5` | +| `build/live-trace-20260925/monitor-final.ndjson` | 1,260,087 | `e07507b0b8f0f86d1a1c7aade7186297c372b1ff177067bea19dfd219f5027a9` | +| `build/live-trace-20260925/summary-final.json` | 3,669 | `19e5ec40cf354c6095a478b68a69f8e51fb3b8ea1c6f278c1d9c90bdaab9ebe2` | +| `build/live-trace-20260925/final-analysis-summary.json` | 946 | `2d17605ba7b730ad78a48bcc70e40e78f90637e3fd59004ebf5eb4a08241ba42` | +| `build/live-trace-20260925/final-worker-state-analysis-summary.json` | 1,376 | `b0ab428eb08587f13f59a1763f835125216f769713e259d85989ea8b7f8af95c` | +| `build/live-trace-20260925/linux-worker-state-final.tar.gz` | 134,380 | `b64e81c90b232f46b400a63ed08f5660f46e34fedb1f67b64afba079d8d36364` | + +The final Windows state is retained under +`build/live-trace-20260925/windows-localappdata/TRUF/RemoteWorker`. + +## Deliberately retained live state + +The user requested that validation state not be restored. The following changes +therefore remain deliberate: + +- accepted Windows and Linux trusted manifests remain installed; +- the keycheck queue maximum remains increased from 4,096 to 8,192 items; +- the Linux validation user remains enabled at assignment cap `0`; +- the Windows validation user remains enabled at assignment cap `1`; +- both validation workers themselves are stopped; +- normal global controls remain open at revision `126`; and +- the dashboard remains running. + +The config editor still contains an intentionally stale candidate based on the +pre-validation active hash. It must not be applied without first rebasing it onto +the current active configuration. + +## Release conclusion + +The accepted worker artifacts passed live cross-platform execution, concurrent +dispatch, bounded progress/diagnostics, durable recovery, one-authoritative +receipt handling, normalized ingestion, projection, audit, local shutdown, and +operator UI validation. The complete cohort settled without leaked worker, +bundle, projection, or quarantine capacity. + +Before broad rollout, deploy and revalidate the DockerHub SQL correction, decide +release treatment for the Windows timestamp and terminal-deadline defects, fix +network error classification, and address server disk pressure and DockerHub +credential health. The OpenSpec change is complete but remains unarchived until +explicitly requested. diff --git a/docs/worker-operator-experience-validation-2026-09-24.md b/docs/worker-operator-experience-validation-2026-09-24.md new file mode 100644 index 0000000..5e9edf0 --- /dev/null +++ b/docs/worker-operator-experience-validation-2026-09-24.md @@ -0,0 +1,269 @@ +# Worker Operator Experience Validation - 2026-09-24 + +## Scope and acceptance + +This report closes the release and production-proof work for OpenSpec change +`add-worker-operator-experience`. Validation covered the shared worker event +contract, local supervisor and contained runner, progress and diagnostics APIs, +admin projections, reproducible Windows and Linux packages, packaged +cross-platform operation, bounded production behavior, and final restoration. + +Acceptance required: + +- the focused unit, integration, protocol, and package matrix to pass; +- independently reproducible Windows and Linux artifacts with documented and + registered package manifests; +- packaged Windows/Linux evidence for multi-slot operation, outage and restart + recovery, durable bundles and receipts, shutdown, and local cleanup; +- bounded production evidence for progress, a full scan-stage timeout, + diagnostics, reconciliation, and restoration; and +- a from-zero operator runbook covering acquisition through removal. + +No production assignment was repeated to prepare this report. All production +facts below are derived from the retained validation snapshots. Raw targets, +findings, credentials, private routes, runtime configuration, and authenticated +worker command lines are intentionally excluded. + +## Focused test matrix + +The final 14-file worker-operator matrix completed on 2026-09-25: + +```text +355 passed, 3 skipped in 49.54s +``` + +The matrix includes worker API/runtime, assignment and contained-runner, +CLI/contracts/local state, observability persistence, package, Linux handoff, +supervisor, remote database, scan execution, and admin API coverage. The three +independent-watchdog timing regressions also passed after their test setup bounds +were stabilized. The timing change did not alter product deadlines or watchdog +behavior. + +An unrestricted repository-wide test run is not a release gate for this change: +the checkout has unrelated missing private/generated assets and platform +assumptions. The focused matrix, packaged E2E, production evidence, and strict +OpenSpec validation are the scoped gates. + +## Reproducible artifacts + +### Windows portable package + +Final independently built archives: + +- `build/operator-experience-validation/windows-i.zip` +- `build/operator-experience-validation/windows-j.zip` + +Both archives have the following identical identities: + +| Identity | Value | +| --- | --- | +| Archive bytes | `134850988` | +| Archive SHA-256 | `6ea9290736a059f1e17d8e89d9cf83506fa4abe2ba2f3731a7422a7b0f386e97` | +| Package manifest identity | `78a962b2bd3fa411413c79e9a8ffb021608a08ff020b1ad851f4505ea634b2b6` | +| Build-input identity | `6991ebbce6ae758c2bdd19a6ae934335aa585a50f86b18ccde8d88bca40ce436` | +| Raw `worker-package.json` SHA-256 | `e0b17d70fcb868fe39fac45ab6e05a17c6d40852e6034010fb63b6cab31f8a3c` | + +Acceptance used the fresh extraction at `build/pwe-final-i-extracted`. Its own +`prepare-worker.ps1` established protected explicit ACLs before direct package +verification. Older G/H Windows archives are excluded because their preparation +script could leave packaged executables inaccessible. + +### Linux worker image + +Final independently built local tags: + +- `truf-worker-test:operator-experience-final-3g` +- `truf-worker-test:operator-experience-final-3h` + +Both provenance-disabled builds have the following identical identities: + +| Identity | Value | +| --- | --- | +| Worker package identity | `45588f2cf406b41b239cfa3b8a9dc83fe84b587229bc997b2729016e1f0dde42` | +| Image manifest / accepted image ID | `sha256:3a088f5743121d823aae132234a29730a84339cecbfda5fc601e8e942f9948c3` | +| Config SHA-256 | `sha256:687a1c4c51c1b962c7fa7ea0cc4b04d159e7ba4f94ef347940c9fb225f7cb87d` | +| Raw `worker-package.json` SHA-256 | `ee926cce3c19e9e6094753f51fa902415bd7364c24fa649cd0c1b659c0aa4d60` | + +The retained manifest snapshot is +`build/operator-experience-validation/linux-worker-package-g.json`. + +### Trusted manifest registration + +The accepted Linux and Windows manifests were registered after packaged E2E +acceptance. Their remote SHA-256 values match the raw manifest hashes above. +Both files are owned by `root:root` with mode `0644`; pre-change backups remain +intact and upload temporary files were removed. Registration required no runtime +restart or configuration mutation, and canonical health remained successful. + +## Packaged Windows/Linux E2E + +Run `35f3f52e232067c1` passed with the freshly extracted/prepared Windows I +package and Linux G image. The safe summary is +`build/pwe-35f3f52e232067c1/summary.json`. + +The gate confirmed: + +- real packaged Windows and Linux operation at two slots; +- server outage handling and restart recovery; +- durable and direct assignment bundle paths; +- authoritative receipt handling; +- graceful shutdown receipts; +- no active local work after completion while intentionally retained abandoned + roots remained inactive; +- matching normalized cross-platform evidence; and +- complete cleanup of owned resources with foreign Docker state unchanged. + +## Production evidence + +### Reconciliation + +The retained snapshot records 34 issued assignments: 33 accepted and one +intentional expected expiry. All 33 accepted bundles were ingested, settled, and +projected. Final unresolved, precommit, quarantine, and drain-blocker counts were +zero. + +Evidence sources: + +- `build/operator-experience-validation/final-evidence.json` +- `build/operator-experience-validation/progress-v3-evidence.json` +- `build/operator-experience-validation/timeout-evidence.json` +- `build/operator-experience-validation/server-baseline.json` + +### Duration percentiles + +The table reports every retained end-to-end metric group. Values are seconds. +`Sufficient` means the server-side minimum sample count of five was met. Rows +below that minimum are retained observations, not statistically sufficient +percentile estimates. + +| Platform | Source | Outcome | Samples | p50 | p95 | p99 | Sufficient | +| --- | --- | --- | ---: | ---: | ---: | ---: | --- | +| Linux | DockerHub | degraded | 1 | 305 | 305 | 305 | no | +| Linux | DockerHub | error | 1 | 19 | 19 | 19 | no | +| Linux | GitLab | error | 1 | 5371 | 5371 | 5371 | no | +| Linux | HuggingFace | expired | 1 | 7219 | 7219 | 7219 | no | +| Windows | DockerHub | clean | 2 | 31 | 31.9 | 31.98 | no | +| Windows | DockerHub | degraded | 4 | 275.5 | 443.45 | 462.29 | no | +| Windows | DockerHub | error | 7 | 619 | 1679.5 | 1731.1 | yes | +| Windows | GitLab | clean | 3 | 27 | 873 | 948.2 | no | +| Windows | GitLab | error | 6 | 342.5 | 1723.75 | 1916.75 | yes | +| Windows | HuggingFace | clean | 1 | 1019 | 1019 | 1019 | no | +| Windows | HuggingFace | error | 7 | 971 | 3092.6 | 3452.12 | yes | + +The snapshot contains 11 Linux and 21 Windows phase/outcome metric groups in +total. Three Windows end-to-end error groups met the minimum; the other 29 +phase/outcome groups did not. Rollout decisions must therefore preserve the +sample-count qualification rather than treating all reported percentiles as +stable capacity estimates. + +### Progress and watchdog evidence + +Reservation `1455` is the retained complete-stage progress reference. It was +acknowledged with an accepted bundle and persisted ten monotonic events spanning +`assigned`, `preparing`, `waiting_permit`, `scanning`, `filtering`, `cleaning`, +`bundling`, `uploading`, and `awaiting_receipt`. This demonstrates one coherent +server-visible sequence across the complete local execution and upload boundary. + +Independent watchdog fault-injection coverage passed for blocked state +persistence, startup-gate persistence, and event draining. The contained runner +tests verify bounded process-tree termination rather than relying on scanner +cooperation. Packaged E2E additionally passed its watchdog, restart, durable +bundle, and cleanup gates. + +### Natural full-stage timeout + +Reservation `1453` is the retained natural timeout reference. The scan ended +with one `timeout / scan.stage_timeout / scanning` diagnostic after 603.367 +seconds. The diagnostic was current and available, with no body or process-log +payload fabricated for the exception. The end-to-end assignment-resolution +duration was 619 seconds. + +The scan outcome was `error`, while the transport outcome was independently +accepted: the bundle was acknowledged, the projection completed, and the +diagnostic was attached to the authoritative result. This confirms that a hard +scan-stage timeout remains a normal, uploadable terminal result and does not +collapse scan, transport, and projection outcomes into one status. + +### Diagnostic and admin snapshots + +The final diagnostic aggregate contains five grouped rows and 23 occurrences: + +| Category | Code | Phase | Occurrences | +| --- | --- | --- | ---: | +| scanner | `scan.result_error` | `scanning` | 20 | +| timeout | `scan.stage_timeout` | `scanning` | 1 | +| assignment expiry | `assignment.deadline_expired` | `assigned` | 1 | +| network | `scan.result_error` | `scanning` | 1 | + +The retained snapshots also confirm separate assignment and scan outcomes, +ordered progress, current diagnostic availability, accepted receipt state, +ingestion/settlement/projection completion, duration metrics, and zero unresolved +or precommit work. This is the durable machine-readable substitute for copying +private admin pages or unbounded diagnostic bodies into the report. + +## Sanitized operator transcript + +The release and validation sequence was: + +1. Build the Windows package twice from the same reviewed inputs and compare the + archive, package-manifest, build-input, and raw-manifest identities. +2. Extract Windows I into a fresh directory, run its packaged + `prepare-worker.ps1`, and run direct package verification. +3. Build Linux G and H independently with provenance disabled and compare image, + config, package, and raw-manifest identities. +4. Run `docker/verify_packaged_workers.py` with the accepted Windows extraction, + Linux image, and isolated test image; retain only its safe summary and owned + evidence directory. +5. Register the two accepted trusted manifests through the reviewed deployment + path and recheck canonical runtime health. +6. Use typed operations to bound production dispatch, start at assignment cap + `1`, observe status/attach/history and server progress, exercise normal, + timeout, outage, and restart paths, and reconcile accepted, ingested, settled, + and projected counts. +7. Restore standard identities and normal/open controls, disable/revoke temporary + validation identities, and recheck runtime and edge health. +8. Run the exact focused pytest matrix recorded above. + +Authentication values, worker argv, raw targets/findings, private route names, +and runtime configuration are omitted by design. + +## Known limits + +- There is no public ZIP download, image registry, installer, or automatic + updater. Release artifacts must move through a trusted channel and match a + server-registered manifest. +- Completed runner roots move under top-level `work/abandoned` and are retained + for at least 60 seconds; normal retention maintenance runs every 300 seconds. + They are inactive evidence, not live work. +- Most retained percentile groups have fewer than five samples. Their values are + useful validation observations but not stable performance baselines. +- Ownership fencing guarantees one authoritative acceptance, not exactly-once + physical execution across a long partition and server-side expiry/reissue. +- Local state remains recovery authority until the server resolves the slot. + Operators must not remove pending bundles or work trees to clear an alert. + +## Rollout, rollback, and restoration + +Rollout uses the exact accepted package identities, begins with one user/device at +server cap `1` and local parallelism `1`, and requires one accepted, ingested, +settled, and projected assignment before expansion. Caps and client count should +increase in stages while unresolved/precommit counts, diagnostic availability, +duration sample counts, and authenticated contact remain observable. + +Rollback first sets the affected cap to `0`, allows pending uploads to resolve, +and obtains a graceful shutdown receipt. The operator then returns to the +previous exact artifact while preserving the same private state tree or volume. +Additive server progress and diagnostic records do not require schema rollback. + +Final restored production state: + +- operations controls normal/open at revision `126`; +- standard WSL production worker user enabled at assignment cap `1`; +- standard production device enabled and not revoked; +- temporary validation identities disabled/revoked; +- unresolved, precommit, quarantine, and drain blockers at zero; +- canonical runtime healthy; and +- edge service remained available. + +Strict OpenSpec validation passed. The operationally validated change is ready +for archival. diff --git a/docs/worker-parallelism-validation-2026-09-23.md b/docs/worker-parallelism-validation-2026-09-23.md new file mode 100644 index 0000000..1f6b98e --- /dev/null +++ b/docs/worker-parallelism-validation-2026-09-23.md @@ -0,0 +1,473 @@ +# Worker Parallelism Validation, 2026-09-23 + +## Scope + +This report records the production validation performed on `sec` using the local +Windows computer: + +- bounded discovery for GitLab, DockerHub, and HuggingFace; +- a native Windows protocol-2 worker with client and server parallelism 3; +- a dual-worker run with native Windows parallelism 1 and the existing trusted + WSL production worker at cap 1; +- authoritative reconciliation from discovery through queue, reservation, + receipt, bundle ingestion, scan settlement, projection, and JSONL append; +- restoration of production source/capacity settings and normal controls. + +No provider credential, device token, raw target, raw finding, runtime YAML, or +worker command line is included in this report. + +## Result + +The core validation passed. + +- Native Windows reached exact unfinished concurrency 3 and never exceeded 3. +- The dedicated Windows cohort accounted for 180 accepted assignments across all + three sources. +- Windows and WSL independently reached concurrency 1 at the same time, for exact + combined concurrency 2, and never exceeded their individual caps. +- All accepted results were ingested, queue-settled, projected, and appended as + required. +- Final cohort and global lineage checks reported zero violations. +- No pipeline quarantine or failed hold was created. +- Production controls and the WSL worker were restored. + +The originally requested 10-hour observation was not completed continuously. +After successful active parallelism-3 work, 14,105 seconds (about 3 hours 55 +minutes) of detached idle stability observation completed before the user changed +the objective to the dual-worker test. The active concurrency evidence itself is +complete; the residual limit is soak duration, not functional coverage. + +## Baseline + +The fresh pre-test snapshot was stored root-only on `sec`: + +- Path: `/opt/truf-remote-server/staging/windows-p3-before2.json` +- SHA-256: `469bcd69888cacce55523fef17508ee48191c4c879a6d39f2251abad4aa0efdb` +- Active config SHA-256: + `e48797689ab0ee7328d21cf5c23d0ffedb979dba492bf49d1da3afe4b075cf5b` +- Config bytes: 37,233 +- Controls: revision 56, discovery open, dispatch open, drain normal +- Existing blocker: one stale production assignment, allowed to settle naturally +- Runtime image: + `sha256:3b4d6e19e29e85a32b75d64265d75e71100d3f7f00221d21de544256dadd25cf` +- Failed hold: absent +- All recorded orphan/mismatch invariants: zero + +Baseline high-water values included reservation 929, scan 926, projection job +926, projection append 984, error 1919, finding 127, and zero quarantine rows. + +The existing WSL assignment was never canceled or directly mutated. Dispatch was +paused and the lease was allowed to resolve before changing worker caps. + +## Windows Package + +The trusted run artifact was built from pinned local caches: + +- Package directory: + `build/worker-parallelism3-20260922/package-fixed1` +- Archive: + `build/worker-parallelism3-20260922/truf-worker-windows-x86_64-fixed1.zip` +- Archive SHA-256: + `6ca460a47d31606c430e8c6f081443530892332cfc183fc0c6edfe7568a454cc` +- Manifest file SHA-256: + `cd55820e0a04a04ded9ba340b9c6e5c6259422aab80989180a6ba28ff3b77103` +- Canonical code-manifest SHA-256: + `e981da19283aad5971fc2865268bb8151b3540d71fbd057ae56d9d2379993374` +- Platform: `windows-x86_64` +- Protocol and bundle schema: v2 +- Capabilities: GitLab `exact_git_v1`, DockerHub `docker_direct_v1`, + HuggingFace `huggingface_space_v1` + +The registered Windows v2 manifest was stale relative to the current production +authority files. It was not overwritten. The fixed package was registered as the +new root-owned `windows-worker-package-v3.json` trust authority. + +### Reproducibility defect + +Two initial package builds differed only in four timestamp bytes in each +pip-generated Windows launcher executable and the corresponding RECORD hashes. +The package builder now normalizes the embedded launcher ZIP timestamps to the +DOS epoch and recomputes the RECORD rows. + +Changed local source: + +- `app/worker_package_builder.py` +- `tests/test_worker_package.py` + +Verification: + +- Two post-fix archives were byte-identical. +- `python -m pytest tests/test_worker_package.py -q`: 16 passed. + +## Discovery + +The temporary discovery candidate was derived from the exact active config. + +GitLab used 16 reviewed keyword queries, one API page, one result per page, and +one target maximum per query. DockerHub used 16 reviewed keyword queries, one +repository result per query, and one image per repository. + +HuggingFace was deliberately reported differently: the configured implementation +does not perform keyword search. It requests the newest-modified Spaces with a +fixed API limit of 100. `pages: 1` therefore means one real recent page, not one +keyword result. Repeated polls occurred while GitLab and DockerHub rotated their +queries; deduplication bounded the admitted rows. + +Final bounded discovery evidence: + +- GitLab: 16/16 distinct successful query rotations, 7 new queue rows +- DockerHub: 16/16 distinct successful query rotations, 10 new queue rows +- HuggingFace: 17 successful recent-page cycles, 31 new queue rows +- Discovery failures: 0 +- Hard limits: GitLab 20, DockerHub 20, HuggingFace 120; none exceeded + +## Native Windows Parallelism 3 + +An isolated temporary server user and device were created with typed, audited +`AdminService` operations. The one-time token was transferred privately, the +server transfer file was deleted after successful launch, and the native worker +used a dedicated protected `LOCALAPPDATA` state tree. + +The first foreground launch exposed a local automation issue: the tool runner +retained the descendant process and did not return. The worker itself survived. +Subsequent starts used `Win32_Process.Create` so the worker was genuinely +detached. No process command line was inspected. + +### Capacity findings + +Client `--parallelism 3` and server user cap 3 were not by themselves enough to +permit three physical scans. + +The first run reached only one active reservation because the production config +had: + +- `global.max_active_scans = 1` +- 64 MiB maximum event size +- 256 MiB projection backlog maximum +- 128 MiB projection headroom + +The capacity model reserves twice the event maximum per assignment, plus +headroom. Three slots therefore require 512 MiB. The temporary test candidate +changed only: + +- `max_active_scans: 1 -> 3` +- `projection_backlog_max_bytes: 256 MiB -> 512 MiB` + +A second run still reached only one assignment. Admission evidence showed +`pipeline_capacity_closed` on the keycheck axis. The stable keycheck backlog was +127 items and each assignment reserved up to 2,000 items. The production limit +of 4,096 allowed one reservation but not three. The temporary candidate changed: + +- `keycheck_queue_max_items: 4096 -> 8192` + +No unrelated limit was increased. + +### Successful run + +After both capacity corrections: + +- Exact max unfinished concurrency: 3 +- Never exceeded: 3 +- Issued: 180 +- Accepted: 180 +- Prebundle failures: 0 +- Expired: 0 +- Unfinished after settlement: 0 +- Accepted-not-ingested: 0 +- Queue-not-settled: 0 +- Projection-not-completed: 0 + +Source totals: + +| Source | Accepted | +|---|---:| +| DockerHub | 63 | +| GitLab | 65 | +| HuggingFace | 52 | + +Scan outcomes were transport-successful but not necessarily target-clean: + +| Source | Outcome | Count | +|---|---|---:| +| DockerHub | clean | 44 | +| DockerHub | degraded | 18 | +| DockerHub | error | 1 | +| GitLab | clean | 57 | +| GitLab | error | 8 | +| HuggingFace | clean | 43 | +| HuggingFace | error | 9 | + +These scan statuses are provider/target results. They are not lost transport, +ingestion, or projection records. All 180 authoritative results settled. + +## Stability Observation + +After the 180-assignment cap closed dispatch for the test identity, the worker +remained online at cap 0 under detached monitoring. + +- Successful checks: 43 +- Elapsed observation: 14,105 seconds +- Runtime restarts: 0 +- Edge restarts: 0 +- Native worker OOM/restart: none +- Pipeline debt: 0 +- Quarantine: 0 +- Global invariants: 0 + +A single strict-health probe can observe a discovery producer during its normal +short startup window. Monitors were corrected to require failure across three +attempts with 20-second delays rather than treating one transient sample as a +production defect. + +## Dual Worker Validation + +The user then requested a second topology: + +- native Windows worker: real client parallelism 1 and server cap 1; +- existing trusted production WSL worker: cap 1; +- desired combined concurrency: 2. + +The native worker was cleanly stopped at cap 0, relaunched from the same trusted +package and protected state with parallelism 1, and verified by PID/resource +evidence. The production WSL token and command line were never inspected. + +### Monitor corrections + +Several fail-closed attempts improved the monitoring policy without losing or +canceling work: + +- A scanner subprocess can remain active without an HTTP contact until it + reports. Contact age alone must not fail an identity with unfinished work. +- A worker may honor a prior cap-0 `Retry-After` after caps change. Arming uses a + separate 600-second threshold; the 180-second idle threshold applies only + after dispatch has opened. +- Operation IDs are one-use and identity-bound. Every retry used a fresh audited + operation namespace. +- A transient PostgreSQL connection timeout while reopening a read-only monitor + connection terminated observability after caps and gates were already safely + closed. The monitor now retries database opens three times with 20-second + delays. A separate read-only settlement monitor completed the active run. + +### Attempt 1 evidence + +The first active dual run already proved max concurrency Windows 1, WSL 1, +combined 2. It fail-closed because of the old contact policy while WSL was doing +a long DockerHub scan. + +- Windows: 19 accepted assignments +- WSL: 2 accepted HuggingFace assignments +- WSL: 1 DockerHub reservation naturally expired after its two-hour lease +- No assignment was canceled + +### Final attempt 4 + +The final bounded run used baseline reservation ID 1131. + +- Issued: 30 +- Windows issued/accepted: 29/29 +- WSL issued: 1 +- WSL accepted: 0 +- WSL expired/refunded: 1 long DockerHub assignment after its two-hour lease +- Max Windows concurrency: 1 +- Max WSL concurrency: 1 +- Max combined concurrency: 2 +- Accepted-not-ingested: 0 +- Queue-not-settled: 0 +- Projection-not-completed: 0 +- Cohort integrity violations: 0 +- Global invariant violations: 0 +- Failed holds in the cohort window: 0 +- Quarantine bytes/items: 0/0 + +Windows source/outcome totals in the final run: + +| Source | Outcome | Count | +|---|---|---:| +| DockerHub | clean | 12 | +| GitLab | clean | 7 | +| HuggingFace | clean | 9 | +| HuggingFace | error | 1 | + +The WSL expiry does not invalidate the concurrency result: both independent +clients simultaneously held one real reservation and never exceeded cap. Earlier +in attempt 1, the same WSL path also completed two accepted HuggingFace results. + +Authoritative dual evidence: + +- `/opt/truf-remote-server/staging/windows-dual-worker-report.json` +- SHA-256: + `8fd03bf79960393406ef1a0f788d90db3ed3521d32ffa83f802c9d1c2c3509bf` + +## Data Reconciliation + +The validation distinguished assignment acceptance, bundle ingestion, queue +settlement, and projection completion. + +Checks covered: + +- accepted reservation has a durable receipt; +- accepted reservation has exactly one matching bundle; +- reservation/bundle/scan event IDs and hashes agree; +- target scan references the same queue row; +- queue completion is `applied`; +- projection job exists, matches the event identity, and is completed; +- required `scan_results` append exists; +- required `found_secrets` append exists when requested by the mask; +- projection append event identity matches the job; +- no orphan bundle, scan, finding, error, or projection job exists; +- JSONL files are present, newline-terminated, and cursor offsets match. + +All cohort and global mismatch counters were zero at the final report points. + +## Code Defects Fixed + +### Integer cap zero + +`AdminService._validate_cap()` used `str(value or '')`, converting integer zero +to an empty string even though cap 0 is valid. It now distinguishes `None` from +zero. + +Changed local source: + +- `app/admin_api.py` +- `tests/test_admin_api.py` + +Verification: + +- `python -m pytest tests/test_admin_api.py -q`: 62 passed. + +The currently deployed runtime image predates this source fix. Run helpers used +canonical string `"0"` through the same typed/audited API; no direct database +mutation was used. + +### Operational helpers + +Run-specific helpers were added under `build/sec-deploy/` for config lifecycle, +controls, typed worker administration, monitoring, observation, local launch, +safe stop, settlement, and reporting. Python helpers passed `py_compile` and +Ruff; PowerShell helpers passed parser validation before use. + +## Configuration Restoration + +Temporary active config identities were: + +| Stage | SHA-256 | +|---|---| +| Bounded discovery + v3 profile | `ec2ae20f9a23ee9ec423c28f7da0b45db3eaff17abb3f484e1a1f0c9227fada4` | +| Parallel scan/projection capacity | `4fee05f3a9ae001767f2de0388dcd5db75183dd954be8e024e98f47d4b64b6b1` | +| Parallel keycheck capacity | `b3509aacc53ab0bfa1e6b13dd789f889d9fea34daf0f680579d2a54697393bf9` | +| Restored production semantics + v3 trust | `c0966cac4f7f0610a813fa8732e91953f2e3e838ad880f91fd1a9437096925c7` | + +The final config restores the original production discovery/source settings, +global scan capacity, projection capacity, keycheck capacity, and supervisor +interval. The only intentional permanent semantic difference from the initial +config is the Windows compatibility profile path changing from stale v2 to the +reproducible/current v3 manifest. Therefore the final config hash intentionally +differs from the initial hash while retaining the same 37,233-byte size. + +All four managed config apply operations succeeded and reconciled. The final +apply operation was: + +- `a9943967-372e-522a-8386-4f9465cc03f9` + +The complete root-only operation export contains 69 operations for the run +actor, all with status `succeeded`: + +- `/opt/truf-remote-server/staging/windows-p3-operations.json` +- SHA-256: + `318aaa56cc2ba348e905b72fd63d7795e24cbd1aa9e84d0a75fa341744ef0acf` + +## Final Production State + +The temporary native process was stopped only after its local bundle/work trees +and authoritative pipeline were empty. + +Typed cleanup completed: + +- temporary Windows cap: 0 +- temporary device: revoked +- temporary user: disabled +- server plaintext token transfer: absent +- production WSL cap: 1 + +Controls after restoration: + +- revision: 98 +- discovery: open +- dispatch: open +- drain: normal +- effective gates: open + +The production WSL container was idle but still honoring a previous cap-0 retry +delay. With zero unfinished work it was safely restarted using the same container, +token, and state. It immediately made a fresh server contact and received a new +production assignment, proving restored admission. + +At the final snapshot: + +- production WSL contact was fresh and one real production assignment was active; +- blocker count 1 was therefore expected, not a drain defect; +- WSL container running, OOM false, restart count 0; +- runtime healthy, restart count 0; +- edge running, restart count 0; +- host-agent, host Caddy, and X-UI active; +- strict runtime health: ACTIVE/READY with DockerHub, GitLab, HuggingFace, + janitor, JSONL projector, result ingester, and Worker API; +- failed hold absent. + +Runtime image remained: + +`sha256:3b4d6e19e29e85a32b75d64265d75e71100d3f7f00221d21de544256dadd25cf` + +## Final Snapshot + +- Path: `/opt/truf-remote-server/staging/windows-p3-final.json` +- SHA-256: `1e3ab6bdb50c5be1416974bf56235c220166a6b731e76a87d8abebb947aae074` +- Bytes: 6,883 +- Config SHA-256: + `c0966cac4f7f0610a813fa8732e91953f2e3e838ad880f91fd1a9437096925c7` +- Reservation high-water: 1165 +- Scan high-water: 1159 +- Projection job high-water: 1159 +- Projection append high-water: 1239 +- Error high-water: 1945 +- Finding high-water: 127 +- Quarantine rows: 0 +- All six recorded global orphan/mismatch invariants: 0 +- `scan_results.jsonl`: 4,592,879 bytes, cursor exact, newline-terminated +- `found_secrets.jsonl`: 146,366 bytes, cursor exact, + newline-terminated + +Final strict-health evidence: + +- `/opt/truf-remote-server/staging/windows-p3-final-health.json` +- SHA-256: + `68c2b0dcd16591b2407fefcb5c3921aa7f58443011539ff912e2cb19b9fdef13` + +## Retained Local Evidence + +Per the user's explicit instruction on 2026-09-23, no additional local files were +deleted during final reporting. + +The following remain intentionally retained: + +- protected local one-time token file; +- dedicated native worker state directory; +- worker stdout/stderr and PID evidence; +- fixed package and both reproducibility builds; +- local run-specific helpers and reports. + +The token is no longer usable because the server device is revoked and user is +disabled, but the local protected file remains until the user authorizes deletion. + +## Residual Limits + +- The continuous 10-hour soak was shortened by the later dual-worker request. +- HuggingFace configured discovery is recent-page discovery, not keyword search. +- Long DockerHub work twice reached the fixed two-hour WSL lease and expired; + expiry/refund behavior was correct, but those targets did not produce bundles. +- The local source fixes for package reproducibility and integer cap zero are + tested in this checkout but were not deployed as a new production runtime + image during this validation. +- Local sensitive/state evidence is deliberately retained and requires a later + explicit cleanup instruction. diff --git a/monitor_runtime_lag.ps1 b/monitor_runtime_lag.ps1 new file mode 100644 index 0000000..5f05492 --- /dev/null +++ b/monitor_runtime_lag.ps1 @@ -0,0 +1,262 @@ +param( + [string]$OutputPath = 'D:\truf\runtime\freeze-diagnostics\runtime-lag.csv', + [int]$IntervalSeconds = 30, + [switch]$Once +) + +$ErrorActionPreference = 'Stop' +throw 'Docker development copy: runtime control is disabled until container isolation is ready. See DOCKER_MIGRATION.md.' +$created = $false +$mutex = New-Object System.Threading.Mutex($true, 'Local\TrufRuntimeLagMonitor', [ref]$created) +if (-not $created) { + exit 0 +} + +try { + $counterTable = (Get-ItemProperty -LiteralPath 'HKLM:\SOFTWARE\Microsoft\Windows NT\CurrentVersion\Perflib\019').Counter + $counterNames = @{} + for ($index = 0; $index -lt $counterTable.Count; $index += 2) { + $counterNames[[int]$counterTable[$index]] = $counterTable[$index + 1] + } + $counterPaths = @( + "\$($counterNames[238])(_Total)\$($counterNames[6])", + "\$($counterNames[4])\$($counterNames[822])", + "\$($counterNames[234])(*)\$($counterNames[198])", + "\$($counterNames[234])(*)\$($counterNames[206])", + "\$($counterNames[234])(*)\$($counterNames[220])", + "\$($counterNames[234])(*)\$($counterNames[222])" + ) + $gpuCounterPath = '\GPU Engine(*)\Utilization Percentage' + try { + Get-Counter -Counter $gpuCounterPath -MaxSamples 1 -ErrorAction Stop | Out-Null + $counterPaths += $gpuCounterPath + } catch { + $gpuCounterPath = $null + } + $outputDirectory = Split-Path -Parent $OutputPath + if (-not (Test-Path -LiteralPath $outputDirectory)) { + New-Item -ItemType Directory -Path $outputDirectory | Out-Null + } + if (-not (Test-Path -LiteralPath $OutputPath)) { + [IO.File]::AppendAllText( + $OutputPath, + "timestamp,cpu_percent,free_ram_mb,commit_used_mb,pagefile_used_mb,pages_input_sec,c_queue,c_latency_ms,d_queue,d_latency_ms,s_queue,s_latency_ms,h_queue,h_latency_ms,h_read_bytes_sec,h_write_bytes_sec,supervisor_pid,supervisor_private_mb,supervisor_handles,supervisor_threads,python_private_mb,python_handles,python_threads,largest_pid,largest_name,largest_private_mb,status_age_sec,scan_active,scan_limit,bundle_items,bundle_bytes,projection_items,projection_bytes,keycheck_items,keycheck_bytes,quarantine_items,quarantine_bytes,top_cpu_pid,top_cpu_name,top_cpu_percent,top_handles_pid,top_handles_name,top_handles_count,dpc_percent,interrupt_percent,interrupts_sec,gpu_engine_sum_percent,gpu_top_pid,gpu_top_name,gpu_top_engine_sum_percent`r`n", + [Text.UTF8Encoding]::new($false) + ) + } + + # PhysicalDisk instances can omit drive letters, so resolve H: to its disk number once. + $hDiskInstancePattern = '*h:*' + try { + foreach ($association in @(Get-CimInstance Win32_LogicalDiskToPartition -ErrorAction Stop)) { + if ($association.Dependent.ToString().Contains('DeviceID = "H:"')) { + $diskMatch = [regex]::Match($association.Antecedent.ToString(), 'Disk #(\d+)') + if ($diskMatch.Success) { + $hDiskInstancePattern = "$($diskMatch.Groups[1].Value)*" + break + } + } + } + } catch { + $hDiskInstancePattern = '*h:*' + } + + $previousCpuByPid = @{} + $previousProcessSampleAt = $null + $logicalProcessorCount = [math]::Max(1, [Environment]::ProcessorCount) + + while ($true) { + try { + $sampleSets = @(Get-Counter -Counter $counterPaths -SampleInterval 1 -MaxSamples 2) + $samples = $sampleSets[-1].CounterSamples + $cpu = @($samples | Where-Object { + $_.InstanceName -eq '_total' -and $_.Path.EndsWith("\$($counterNames[6])") + })[0].CookedValue + $pagesInput = @($samples | Where-Object { + -not $_.InstanceName -and $_.Path.EndsWith("\$($counterNames[822])") + })[0].CookedValue + $cQueue = @($samples | Where-Object { + $_.InstanceName -like '*c:*' -and $_.Path.EndsWith("\$($counterNames[198])") + })[0].CookedValue + $cLatency = @($samples | Where-Object { + $_.InstanceName -like '*c:*' -and $_.Path.EndsWith("\$($counterNames[206])") + })[0].CookedValue * 1000 + $dQueue = @($samples | Where-Object { + $_.InstanceName -like '*d:*' -and $_.Path.EndsWith("\$($counterNames[198])") + })[0].CookedValue + $dLatency = @($samples | Where-Object { + $_.InstanceName -like '*d:*' -and $_.Path.EndsWith("\$($counterNames[206])") + })[0].CookedValue * 1000 + $sQueue = @($samples | Where-Object { + $_.InstanceName -like '*s:*' -and $_.Path.EndsWith("\$($counterNames[198])") + })[0].CookedValue + $sLatency = @($samples | Where-Object { + $_.InstanceName -like '*s:*' -and $_.Path.EndsWith("\$($counterNames[206])") + })[0].CookedValue * 1000 + $hDiskSamples = @($samples | Where-Object { $_.InstanceName -like $hDiskInstancePattern }) + $hQueueSample = @($hDiskSamples | Where-Object { + $_.CounterType.ToString() -eq 'NumberOfItems32' + }) + $hLatencySample = @($hDiskSamples | Where-Object { + $_.CounterType.ToString() -eq 'AverageTimer32' + }) + $hThroughputSamples = @($hDiskSamples | Where-Object { + $_.CounterType.ToString() -eq 'RateOfCountsPerSecond64' + }) + $hReadSample = @($hThroughputSamples | Select-Object -First 1) + $hWriteSample = @($hThroughputSamples | Select-Object -Skip 1 -First 1) + $hQueue = if ($hQueueSample.Count) { $hQueueSample[0].CookedValue } else { -1 } + $hLatency = if ($hLatencySample.Count) { $hLatencySample[0].CookedValue * 1000 } else { -1 } + $hReadBytes = if ($hReadSample.Count) { $hReadSample[0].CookedValue } else { -1 } + $hWriteBytes = if ($hWriteSample.Count) { $hWriteSample[0].CookedValue } else { -1 } + $os = Get-CimInstance Win32_OperatingSystem + $pagefile = Get-CimInstance Win32_PageFileUsage | Where-Object { $_.Name -like 'D:*' } + $python = @(Get-Process -Name python -ErrorAction SilentlyContinue) + $allProcesses = @(Get-Process -ErrorAction SilentlyContinue) + $largest = $allProcesses | Sort-Object PrivateMemorySize64 -Descending | Select-Object -First 1 + $mostHandles = $allProcesses | Sort-Object HandleCount -Descending | Select-Object -First 1 + + $processSampleAt = Get-Date + $elapsedMilliseconds = if ($previousProcessSampleAt) { + ($processSampleAt - $previousProcessSampleAt).TotalMilliseconds + } else { 0 } + $currentCpuByPid = @{} + $topCpu = $null + $topCpuPercent = 0.0 + foreach ($process in $allProcesses) { + try { + $processId = [int]$process.Id + $totalProcessorMilliseconds = $process.TotalProcessorTime.TotalMilliseconds + $currentCpuByPid[$processId] = $totalProcessorMilliseconds + if ($elapsedMilliseconds -gt 0 -and $previousCpuByPid.ContainsKey($processId)) { + $deltaMilliseconds = [math]::Max( + 0, + $totalProcessorMilliseconds - [double]$previousCpuByPid[$processId] + ) + $processCpuPercent = 100 * $deltaMilliseconds / ($elapsedMilliseconds * $logicalProcessorCount) + if ($processCpuPercent -gt $topCpuPercent) { + $topCpu = $process + $topCpuPercent = $processCpuPercent + } + } + } catch { + continue + } + } + $previousCpuByPid = $currentCpuByPid + $previousProcessSampleAt = $processSampleAt + + $processorPerf = Get-CimInstance ` + Win32_PerfFormattedData_PerfOS_Processor ` + -Filter "Name='_Total'" ` + -ErrorAction SilentlyContinue + $gpuSamples = if ($gpuCounterPath) { + @($samples | Where-Object { $_.Path -like '*\gpu engine(*)\utilization percentage' }) + } else { @() } + $gpuEngineSum = [double](($gpuSamples | Measure-Object CookedValue -Sum).Sum) + $gpuByPid = @{} + foreach ($sample in $gpuSamples) { + if ($sample.InstanceName -match '^pid_(\d+)_') { + $gpuProcessId = [int]$matches[1] + if (-not $gpuByPid.ContainsKey($gpuProcessId)) { + $gpuByPid[$gpuProcessId] = 0.0 + } + $gpuByPid[$gpuProcessId] += [double]$sample.CookedValue + } + } + $gpuTop = $gpuByPid.GetEnumerator() | Sort-Object Value -Descending | Select-Object -First 1 + $gpuTopProcess = if ($gpuTop) { + $allProcesses | Where-Object { $_.Id -eq [int]$gpuTop.Key } | Select-Object -First 1 + } else { $null } + $supervisor = $null + $instancePath = 'D:\truf\runtime\control\supervisor.instance.json' + if (Test-Path -LiteralPath $instancePath) { + $metadata = [IO.File]::ReadAllText($instancePath) | ConvertFrom-Json + $supervisor = Get-Process -Id ([int]$metadata.pid) -ErrorAction SilentlyContinue + } + $statusPath = 'D:\truf\runtime\logs\supervisor.status.txt' + $statusAge = if (Test-Path -LiteralPath $statusPath) { + [math]::Max(0, ((Get-Date) - (Get-Item -LiteralPath $statusPath).LastWriteTime).TotalSeconds) + } else { -1 } + $statusText = if (Test-Path -LiteralPath $statusPath) { [IO.File]::ReadAllText($statusPath) } else { '' } + $scanMatch = [regex]::Match($statusText, 'Scan workers: active=(\d+)/(\d+)') + $pipelineMatch = [regex]::Match( + $statusText, + 'bundles=(\d+)/(\d+)B\s+projection=(\d+)/(\d+)B\s+candidates=(\d+)/(\d+)B\s+quarantine=(\d+)/(\d+)B' + ) + $values = @( + (Get-Date).ToString('o'), + [math]::Round($cpu, 2), + [math]::Round($os.FreePhysicalMemory / 1KB, 1), + [math]::Round(($os.TotalVirtualMemorySize - $os.FreeVirtualMemory) / 1KB, 1), + [math]::Round((($pagefile | Measure-Object CurrentUsage -Sum).Sum), 1), + [math]::Round($pagesInput, 2), + [math]::Round($cQueue, 2), + [math]::Round($cLatency, 2), + [math]::Round($dQueue, 2), + [math]::Round($dLatency, 2), + [math]::Round($sQueue, 2), + [math]::Round($sLatency, 2), + [math]::Round($hQueue, 2), + [math]::Round($hLatency, 2), + [math]::Round($hReadBytes, 2), + [math]::Round($hWriteBytes, 2), + $(if ($supervisor) { $supervisor.Id } else { 0 }), + $(if ($supervisor) { [math]::Round($supervisor.PrivateMemorySize64 / 1MB, 1) } else { 0 }), + $(if ($supervisor) { $supervisor.HandleCount } else { 0 }), + $(if ($supervisor) { $supervisor.Threads.Count } else { 0 }), + [math]::Round((($python | Measure-Object PrivateMemorySize64 -Sum).Sum / 1MB), 1), + [int](($python | Measure-Object HandleCount -Sum).Sum), + [int](($python | ForEach-Object { $_.Threads.Count } | Measure-Object -Sum).Sum), + $(if ($largest) { $largest.Id } else { 0 }), + $(if ($largest) { ($largest.ProcessName -replace ',', '_') } else { '' }), + $(if ($largest) { [math]::Round($largest.PrivateMemorySize64 / 1MB, 1) } else { 0 }), + [math]::Round($statusAge, 1), + $(if ($scanMatch.Success) { [int64]$scanMatch.Groups[1].Value } else { 0 }), + $(if ($scanMatch.Success) { [int64]$scanMatch.Groups[2].Value } else { 0 }), + $(if ($pipelineMatch.Success) { [int64]$pipelineMatch.Groups[1].Value } else { 0 }), + $(if ($pipelineMatch.Success) { [int64]$pipelineMatch.Groups[2].Value } else { 0 }), + $(if ($pipelineMatch.Success) { [int64]$pipelineMatch.Groups[3].Value } else { 0 }), + $(if ($pipelineMatch.Success) { [int64]$pipelineMatch.Groups[4].Value } else { 0 }), + $(if ($pipelineMatch.Success) { [int64]$pipelineMatch.Groups[5].Value } else { 0 }), + $(if ($pipelineMatch.Success) { [int64]$pipelineMatch.Groups[6].Value } else { 0 }), + $(if ($pipelineMatch.Success) { [int64]$pipelineMatch.Groups[7].Value } else { 0 }), + $(if ($pipelineMatch.Success) { [int64]$pipelineMatch.Groups[8].Value } else { 0 }), + $(if ($topCpu) { $topCpu.Id } else { 0 }), + $(if ($topCpu) { ($topCpu.ProcessName -replace ',', '_') } else { '' }), + [math]::Round($topCpuPercent, 2), + $(if ($mostHandles) { $mostHandles.Id } else { 0 }), + $(if ($mostHandles) { ($mostHandles.ProcessName -replace ',', '_') } else { '' }), + $(if ($mostHandles) { $mostHandles.HandleCount } else { 0 }), + $(if ($processorPerf) { [math]::Round($processorPerf.PercentDPCTime, 2) } else { 0 }), + $(if ($processorPerf) { [math]::Round($processorPerf.PercentInterruptTime, 2) } else { 0 }), + $(if ($processorPerf) { [math]::Round($processorPerf.InterruptsPersec, 2) } else { 0 }), + [math]::Round($gpuEngineSum, 2), + $(if ($gpuTop) { [int]$gpuTop.Key } else { 0 }), + $(if ($gpuTopProcess) { ($gpuTopProcess.ProcessName -replace ',', '_') } else { '' }), + $(if ($gpuTop) { [math]::Round([double]$gpuTop.Value, 2) } else { 0 }) + ) + $serializedValues = foreach ($value in $values) { + if ($value -is [IFormattable]) { + $value.ToString($null, [Globalization.CultureInfo]::InvariantCulture) + } else { + [string]$value + } + } + [IO.File]::AppendAllText($OutputPath, (($serializedValues -join ',') + "`r`n"), [Text.UTF8Encoding]::new($false)) + } catch { + [IO.File]::AppendAllText( + "$OutputPath.errors.log", + "$(Get-Date -Format o) $($_.Exception.Message)`r`n", + [Text.UTF8Encoding]::new($false) + ) + } + if ($Once) { + break + } + Start-Sleep -Seconds ([math]::Max(5, $IntervalSeconds)) + } +} finally { + $mutex.ReleaseMutex() + $mutex.Dispose() +} diff --git a/openspec/changes/add-adaptive-docker-payload-scanning/.openspec.yaml b/openspec/changes/add-adaptive-docker-payload-scanning/.openspec.yaml new file mode 100644 index 0000000..032461f --- /dev/null +++ b/openspec/changes/add-adaptive-docker-payload-scanning/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-02 diff --git a/openspec/changes/add-adaptive-docker-payload-scanning/design.md b/openspec/changes/add-adaptive-docker-payload-scanning/design.md new file mode 100644 index 0000000..001192b --- /dev/null +++ b/openspec/changes/add-adaptive-docker-payload-scanning/design.md @@ -0,0 +1,352 @@ +## Context + +Docker scanning currently has two materially different paths. The full-image TruffleHog source is +authoritative for new and normally completing images, but it treats an immutable image as one +indivisible operation. The bounded layer path can resume and globally reuse content digests, but its +fixed highest-eight-layer selector was admitted only for prior full-image timeouts. In the completed +control cohort it retained 21.1% of routed identities and 5.8% of detector identities, so enabling +that selector broadly would trade too much useful coverage for speed. + +The layer path already provides the expensive safety primitives this change needs: exact manifest +resolution, authenticated bounded Registry transfer, digest verification, private artifacts, +contained TruffleHog filesystem execution, canonical reservation-bound plans, policy-scoped global +blob leases, fenced result ingestion, and explicit per-image coverage. The missing pieces are a +content-aware selector, a reusable execution-policy identity that is independent of selector +budgets, and trustworthy evidence for deciding whether the adaptive path is safe to broaden. + +Docker/OCI configuration contains an ordered `history` list that can help identify `COPY`, `ADD`, +application setup, package installation, generic `RUN`, and likely bulk-data layers. That history is +untrusted and may itself contain secret material. It is therefore only a bounded selection hint; it +must never become an authority for content identity or successful coverage and its raw commands +must never be persisted or logged. + +The production database already contains durable version-one layer plans and covered blob rows. +Changing their interpretation in place would invalidate audit and lease fences. The migration must +be additive, retain version-one validation, and allow only explicitly proven compatible successful +coverage to enter the new execution namespace. + +## Goals / Non-Goals + +**Goals:** + +- Scan all supported unique content for images that fit conservative bounds. +- Prioritize likely application, configuration, and source-bearing layers when a large image cannot + fit those bounds. +- Reuse successful immutable blob coverage across images and selector revisions when execution + semantics are unchanged. +- Freeze the first deterministic selection for an image and selector policy across every checkpoint, + retry, and execution-policy transition. +- Reduce per-image checkpoint overhead with bounded multi-blob leases without weakening per-blob + execution and ingestion fences. +- Preserve exact selected, reused, skipped, failed, and partial coverage semantics. +- Produce private, aggregate, non-authoritative shadow evidence against 50-100 completed full-image + controls before any broad adaptive rollout. +- Retain full-image scanning and the existing timeout-only layer canary as immediate rollback paths. + +**Non-Goals:** + +- Infer arbitrary file contents without downloading a compressed layer. +- Claim complete coverage for an image with unsupported, oversized, failed, or budget-excluded + descriptors. +- Persist raw Docker config history, shadow findings, provider keys, target names, or Registry bearer + tokens in rollout evidence. +- Change detector classification, keycheck routing, Docker account ownership, repository resolver + scheduling, guaranteed scan-slot capacity, worker count, or non-Docker scanners. +- Reconstruct a merged container filesystem or remove historical whiteout content from individual + layer evidence. +- Destructively rewrite or delete version-one plans and coverage rows. + +## Decisions + +### 1. Introduce a version-two immutable content plan + +Adaptive execution uses a canonical version-two plan. It retains the version-one image, repository, +manifest, platform, media, limits, scan-policy, descriptor, reservation, and plan-hash fences and +adds: + +- a versioned selector algorithm and `selection_policy_sha256`; +- an `execution_policy_sha256` for reusable successful blob evidence; +- one bounded classifier value per descriptor; +- exact selection and omission reasons; and +- bounded checkpoint lease limits that do not alter the frozen selection set. + +Version-one plan and execution validators remain exact because their JSON is already durable. +Version-two validation dispatches by the exact integer version and rejects unknown fields, unknown +classes, malformed hashes, descriptor reordering, and oversized canonical JSON. A reservation can +bind one canonical plan only; idempotent replay must be byte-identical. + +After exact manifest resolution and before plan binding, the scanner fetches the configuration blob +through the existing bounded, authenticated, digest-verifying Registry path. It parses the JSON in +bounded private memory/storage and maps non-`empty_layer` history entries from base to top onto the +ordered manifest layers. The optional `rootfs.diff_ids` count and the non-empty history count must +agree with the layer count. A missing field, malformed value, excess entry count, or alignment +mismatch classifies every layer as `unknown`; it does not change descriptor identity or fail a +valid immutable manifest. + +The classifier uses a small versioned allow-list and emits only these bounded classes: +`config`, `copy_add`, `app_config_run`, `package_run`, `other_run`, `bulk_data`, and `unknown`. +Raw `created_by` values are discarded before plan construction and are excluded from logs, result +metadata, errors, and database rows. + +Alternatives rejected: + +- Persisting normalized command text would retain unnecessary secret-bearing input. +- Treating history as authoritative would let malformed or adversarial metadata hide content. +- Mutating version-one plans would break exact replay and auditability. + +### 2. Separate scan, execution, and selection identities + +Three hashes have distinct responsibilities: + +- `scan_policy_sha256` remains the installed scanner/detector/config fingerprint. +- `execution_policy_sha256` hashes the scan policy plus versioned content validation and all archive + semantics that can change which bytes a successful command examines. It excludes image/layer + selection budgets, classifier weights, retry counts, lease duration, checkpoint size, and delay. +- `selection_policy_sha256` hashes the selector version, class ordering, deterministic tie-breaks, + supported descriptor classes, and all limits that determine the initial selected set. It excludes + mutable global coverage and execution scheduling. + +For version-two rows, the existing `docker_content_blobs.coverage_policy_sha256` key stores the +execution-policy hash. Selector changes therefore do not force an identical successfully scanned +digest through TruffleHog again, while archive or detector semantic changes still create a separate +coverage namespace. + +`docker_image_blob_coverage` receives a non-null `selection_policy_sha256`. Existing rows are +backfilled from the exact bound plan on their linked reservation and indexed by +`(queue_id, manifest_digest, selection_policy_sha256, position, reservation_id)`. The earliest full +position map under that key is the immutable selection baseline. Coverage-policy changes may require +new execution but cannot expand or contract that baseline silently. + +Successfully covered version-one evidence may be copied lazily into the version-two execution +namespace only in the same transaction that validates all of the following: + +- the old row is durably `covered`, not pending, leased, submitted, failed, or ambiguous; +- a linked covered image row and reservation contain an exact valid version-one plan; +- the descriptor digest, kind, declared bytes, and semantic media class match; +- the old coverage key is exactly the legacy hash derived from that plan; and +- the scan fingerprint and every scan-affecting archive semantic equal the requested version-two + execution policy. + +The alias keeps the original successful reservation, plan, byte count, and completion provenance. +Legacy rows are never rekeyed or deleted. If any compatibility proof is absent, the new namespace +starts uncovered and normal fenced execution is required. + +Alternatives rejected: + +- Keeping selector limits in the coverage hash defeats global reuse whenever budgets are tuned. +- Reusing every covered digest across scanner versions can suppress required rescans. +- Bulk-rekeying legacy rows destroys provenance and races active leases. + +### 3. Select all-fit images and rank large-image payload deterministically + +Configuration remains independently eligible under its hard configuration-byte bound. Supported +unique layer digests are evaluated under the hard per-layer bound. Covered digests in the matching +execution namespace and duplicate positions in the same image are selected at zero new transfer +bytes and zero new execution count. + +If every supported unique descriptor fits the configured aggregate bytes and unique-layer count, +the selector selects all of them regardless of history class. This is the complete bounded path for +small images and avoids reducing their coverage merely because history hints are absent. + +When the complete set does not fit, new unique layer candidates are sorted by: + +1. class priority: `copy_add`, `app_config_run`, `package_run`, `unknown`, `other_run`, `bulk_data`; +2. highest manifest position first; +3. smallest compressed descriptor first; and +4. lexical digest as the final stable tie-break. + +The selector greedily admits candidates while both aggregate compressed-byte and unique-layer-count +limits permit them. Hard per-descriptor limits are never exceeded. Each descriptor records one exact +reason, including selected class, `already_covered`, `duplicate_digest`, `unsupported_media_type`, +`config_too_large`, `layer_too_large`, `image_budget_exhausted`, or `layer_limit_exhausted`. +Changing any class order, classifier rule, supported-media rule, or selection bound changes the +selector hash. + +The selection algorithm receives a transactionally consistent coverage snapshot, but mutable +coverage is not part of its identity. The first complete descriptor-position map is written before +any new lease and reused exactly on later checkpoints. Consequently a layer skipped by the original +budget never becomes newly selected merely because an earlier selected layer became globally +covered. + +Alternatives rejected: + +- Fixed highest-first selection has already failed the completed-control recall gate. +- A whole-image byte cutoff loses small application layers above giant data layers. +- Selecting only recognized commands lets missing or unusual history hide useful payload. +- Selecting globally covered content only when it still fits the current budget wastes verified + immutable evidence. + +### 4. Lease bounded multi-blob checkpoints + +The current executor and ingestion format already support more than one leased descriptor, but the +binder leases one new digest and then defers the parent for 60 seconds. Adaptive execution leases a +deterministic bounded batch from the frozen selected set under configurable maximum blob count and +compressed bytes. A first eligible blob larger than the checkpoint-byte target but within its hard +per-layer bound may be leased alone so it cannot starve indefinitely. + +Every digest still has its own advisory lock, lease token, attempt count, execution record, digest +verification, and final state. The executor processes the batch sequentially inside the same owned +slot and bundle. Ingestion may cover successful earlier blobs while returning a later retryable blob +to pending. A crash before durable handoff covers none of the un-ingested batch and normal exact +lease expiry/recovery applies. + +Checkpoint count, byte target, retry count, lease duration, and continuation delay are scheduling +controls. They do not enter execution or selector hashes because they cannot turn an incomplete blob +into successful coverage or change the frozen selected set. + +### 5. Keep image coverage explicit and policy-specific + +An image is complete only when its configuration and every manifest layer position are successfully +covered under the requested execution policy. Reused and duplicate digests count as covered only +after exact policy-compatible evidence exists. Any unsupported, oversized, budget-excluded, failed, +or otherwise unselected descriptor makes the image bounded partial coverage. + +Selected retryable work keeps the parent deferred. Shared active work does not consume another blob +attempt. Exhausted selected work produces terminal incomplete disposition. Findings from completed +selected blobs retain image, digest, kind, class, and position provenance and use the existing +authoritative ingestion, projection, and keycheck paths. + +Config history classification affects only selection order. It never changes detector output, +finding authority, digest identity, or completion criteria. + +### 6. Add adaptive modes without changing legacy rollout semantics + +Existing `full`, timeout-only `canary`, and legacy `layer` meanings remain available for durable +version-one work and rollback. Two explicit version-two modes are added: + +- `adaptive-canary` assigns a configured basis-point cohort across all immutable Docker manifests by + a stable versioned hash; cohort members use adaptive plans and non-members use full-image scanning. +- `adaptive` uses adaptive plans for every eligible immutable Docker claim. + +Neither mode depends on a previous full-image timeout. Retry and checkpoint attempts for the same +manifest and selector retain the same assignment. Invalid mode, policy, migration, or gate state +fails closed to full-image execution before any adaptive plan is bound. Returning configuration to +`full` changes only new claims and leaves adaptive plans and audit rows intact. + +The existing timeout-only canary remains independent and may continue while adaptive shadow evidence +is gathered. The broad legacy `layer` mode remains operationally disabled because its selector did +not pass recall gates. + +### 7. Gate rollout with non-authoritative aggregate shadow evidence + +An operator-invoked bounded shadow evaluator selects 50-100 exact immutable images whose authoritative +full-image scans completed successfully under one scan fingerprint. It executes the candidate +adaptive policy using the same downloader, validators, process containment, deadlines, and scanner +fingerprint, but under a shadow authority that cannot call normal result ingestion or mutate target +status, result reservations, global blob coverage, findings, keycheck candidates, projections, or +source counters. + +For each paired control, full routed identities are read from the protected database as +`(service, provider_key_hash)` and adaptive routed identities are derived in private memory through +the same candidate normalization. Detector identities use `detector_secret_hash`. Identity sets, +raw findings, commands, provider material, image names, and bearer tokens are discarded after +intersection counts are computed. + +The durable report contains only policy hashes, cohort and completion counts, aggregate full, +adaptive, and intersection counts, aggregate slot milliseconds, bounded failure counts, threshold +results, and timestamps. It records no per-image row or identity. Slot timing uses the same outer +monotonic boundary from admitted work through durable shadow sink completion for both paths; scan +subprocess duration remains a diagnostic, not the gate denominator. + +A report passes only when: + +- 50-100 controls completed both paths without integrity, containment, or fence failure; +- aggregate routed-identity recall, `intersection / full`, is at least 85%; +- aggregate adaptive/full slot-time ratio is at most 40%; +- every adaptive omission is represented in coverage counts; and +- no credential persistence, quarantine, projection, source-failure, or resource-bound regression + is observed. + +Reports are bound to exact selector, execution, and scan policy hashes. Stale or incomplete reports +cannot authorize another policy. Adaptive canary remains fail-closed until a matching report passes. +Broad `adaptive` enablement additionally requires a stable low-percentage production canary over at +least one repository-refresh interval. Operators change rollout configuration explicitly; shadow +evidence never changes execution mode by itself. + +Alternatives rejected: + +- Routing shadow candidates through keycheck would make the experiment authoritative and consume + external capacity. +- Persisting per-image shadow identities creates unnecessary sensitive correlation data. +- Comparing only detector counts does not measure the routed identities the scanner is intended to + produce. +- Automatically enabling adaptive mode from a report removes the operational rollback checkpoint. + +### 8. Preserve incomplete warning semantics without redundant retries + +The first completed 50-control production shadow report failed closed. Routed recall was 12 of 18 +identities (66.7%), adaptive/full slot time was 46.6%, and the report recorded 43 aggregate +failures. Its selection evidence showed 230 descriptors omitted by the eight-layer limit and 14 +oversized descriptors, so selector recall remains the primary rollout blocker. + +The same evidence exposed a separate execution defect. TruffleHog diagnostics such as +`chunk_processing` and `detector_timeout` are explicitly deterministic, non-retryable warnings. +Their findings must be retained, but the affected blob cannot establish complete coverage. The +diagnostic adapter previously discarded the non-retryable bit, causing the layer executor to +download and scan the same incomplete blob up to three times before reaching the same terminal +state. The adapter now preserves aggregate warning retryability and the layer executor terminates +that blob after the first deterministic warning. It does not mark the blob covered or remove the +report failure. + +The historical report schema retained only a total failure count, so its 43 failures cannot be +decomposed exactly after the fact. Future shadow runs keep a fixed allow-list of aggregate-only +failure categories in protected memory and print their totals in the final operator summary without +changing report authority or persisting target-level evidence. + +Shadow execution also suppresses target labels in finding-filter logs. Production scans retain their +existing target logging, while both private full and private layer paths emit only aggregate filter +counts. This closes a privacy gap found in the first report log without weakening normal operational +diagnostics. + +## Risks / Trade-offs + +- [History is malformed, misleading, or secret-bearing] -> Bound and validate it, persist only an + enum, fall back to `unknown`, and keep identity/coverage independent of classification. +- [Application secrets exist in a low-priority or giant layer] -> Scan all-fit images, keep unknown + ahead of generic/bulk classes, record partial scope, retain full controls, and enforce the 85% + routed-recall gate. +- [Unsafe policy reuse suppresses a required rescan] -> Separate hashes and permit legacy aliasing + only from exact successful compatible evidence in one fenced transaction. +- [Selector changes expand coverage during retry] -> Freeze the earliest full position map under the + selector hash before leases are issued. +- [Multi-blob checkpoints increase work lost on crash] -> Bound count/bytes and retain independent + per-blob leases and ingestion records; no pre-handoff result becomes covered. +- [Config prefetch adds Registry traffic] -> Reuse the already bounded authenticated downloader and + avoid a second fetch when the config is leased in the same plan. +- [Individual layer scans expose whiteouted historical files] -> Preserve position provenance and + describe evidence as image-content coverage, not merged-root state. +- [Shadow evaluation consumes slots] -> Keep it operator-invoked, bounded, deterministic, and + subject to the existing slot/resource controls. +- [Aggregate reports hide individual anomalies] -> Fail the whole report on incomplete paired work + and retain bounded failure counts without persisting target identity. +- [Deterministic scanner warnings consume repeated transfer and slot time] -> Preserve their + non-retryable policy, keep findings and incomplete coverage, and terminate the blob on its first + bounded attempt. + +## Migration Plan + +1. Ship version-one and version-two validators, new modes, and migration code while production stays + on its existing timeout-only canary. +2. Stop authoritative runtime and verify no active result, queue, or blob leases remain. +3. Add the selector-policy coverage column, aggregate shadow-report state, required timing/fence + columns, indexes, and a new migration marker. +4. Backfill every historical image-coverage row from its exact valid reservation plan. Abort and + roll back the migration transaction on an orphan, malformed plan, or invalid hash; then make the + column non-null and run exact schema validation. +5. Restart with unchanged mode and verify legacy claims, projection, keycheck, resolver, quarantine, + and source health before creating version-two work. +6. Run the private shadow evaluator for 50-100 completed controls. Keep adaptive modes fail-closed if + the matching report misses recall, timing, safety, or completion gates. +7. Enable a low deterministic `adaptive-canary`, monitor at least one repository-refresh interval, + and compare source failures, coverage reasons, routed yield, slot time, and quarantine. +8. Increase canary basis points and finally enable `adaptive` only after every gate remains satisfied. +9. Roll back immediately by setting mode to `full` or the existing timeout-only `canary`. Keep all + version-two plans, reports, and coverage rows for audit and exact future resume. + +## Open Questions + +- Which initial bounded checkpoint count and byte target provide the best reduction in continuation + delay without increasing crash rework materially? +- Which classifier allow-list revisions improve routed recall in the first 50-100 controls? Every + revision will receive a new selector-policy hash rather than changing an existing policy. +- What adaptive-canary basis-point sequence should operators use after the shadow gate passes? diff --git a/openspec/changes/add-adaptive-docker-payload-scanning/proposal.md b/openspec/changes/add-adaptive-docker-payload-scanning/proposal.md new file mode 100644 index 0000000..ddd2d5f --- /dev/null +++ b/openspec/changes/add-adaptive-docker-payload-scanning/proposal.md @@ -0,0 +1,29 @@ +## Why + +The timeout-only Docker layer fallback cuts heavy-image work to roughly 5-6% of full-image time, but its fixed highest-eight-layer policy retained only 21.1% of routed identities on completed controls. Broad Docker throughput therefore remains limited by indivisible full-image scans, while enabling the existing bounded layer selector globally would lose too much useful coverage. + +## What Changes + +- Add a deterministic adaptive Docker payload policy that scans all eligible content for small images and prioritizes application-bearing content for large images using bounded, untrusted config-history hints. +- Reuse successfully covered immutable blobs across images under a scan-execution policy independent of selector budgets, without weakening digest, reservation, lease, plan, or ingestion fences. +- Give every adaptive plan a versioned selector identity and freeze its first selection across checkpoints and retries. +- Record explicit reasons and bytes for every selected, reused, unsupported, oversized, and budget-excluded descriptor; adaptive completion remains partial whenever any content is omitted. +- Add non-authoritative shadow evaluation and aggregate privacy-preserving evidence for completed full-image controls. +- Keep full-image scanning as the default and rollback path; permit broad adaptive rollout only after a 50-100 image control cohort retains at least 85% routed-identity recall while using at most 40% of full-image slot time. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `docker-layer-content-scanning`: Replace fixed highest-first broad selection with versioned adaptive payload selection, separate selector identity from reusable execution evidence, and require non-authoritative shadow gates before broad rollout. + +## Impact + +- Affects Docker Registry config access, post-claim mode assignment, layer-plan binding, policy hashing, image-to-blob coverage state, aggregate rollout evidence, Docker configuration, and related PostgreSQL migration/runtime validation. +- Reuses the existing immutable digest identities, account pool, bounded Registry downloader, global blob leases, scan slots, Windows Job containment, result bundles, findings projection, and keycheck pipeline. +- Does not change detector classification, credential persistence, Git or Hugging Face scanning, guaranteed scan-slot capacity, Docker worker count, or repository resolver scheduling. +- Requires an additive stopped-runtime migration before adaptive execution can be enabled; rollback leaves durable adaptive audit state intact. diff --git a/openspec/changes/add-adaptive-docker-payload-scanning/specs/docker-layer-content-scanning/spec.md b/openspec/changes/add-adaptive-docker-payload-scanning/specs/docker-layer-content-scanning/spec.md new file mode 100644 index 0000000..52920d7 --- /dev/null +++ b/openspec/changes/add-adaptive-docker-payload-scanning/specs/docker-layer-content-scanning/spec.md @@ -0,0 +1,218 @@ +## MODIFIED Requirements + +### Requirement: Immutable image content plans are bound after claim +The system SHALL resolve a claimed Docker image's exact platform manifest into a canonical bounded plan containing its configuration and ordered layer descriptors, and SHALL bind that plan to the active result reservation before content execution. Adaptive plans SHALL use an exactly validated version-two schema that binds versioned selector and execution-policy identities plus one bounded payload class per descriptor while retaining exact validation of durable version-one plans. + +#### Scenario: Valid immutable manifest +- **WHEN** the claimed `repository@sha256:` resolves to a valid requested-platform manifest +- **THEN** the bound plan identifies the same manifest digest and contains only bounded valid SHA-256 content descriptors, sizes, media types, order, policy hashes, classes, and selection reasons + +#### Scenario: Changed replay +- **WHEN** the same reservation attempts to bind a different content plan +- **THEN** the system rejects the replay as a fenced conflict and executes neither plan + +#### Scenario: Invalid or oversized manifest +- **WHEN** the Registry manifest is malformed, exceeds descriptor bounds, or disagrees with the immutable target +- **THEN** the system fails closed without claiming complete content coverage + +#### Scenario: Valid aligned configuration history +- **WHEN** a bounded digest-verified configuration has non-empty history entries aligned base-to-top with every manifest layer +- **THEN** the selector classifies each layer through the versioned bounded classifier and persists only its class enum + +#### Scenario: Untrusted or misaligned configuration history +- **WHEN** configuration history is absent, malformed, oversized, or inconsistent with layer or rootfs counts +- **THEN** every affected layer is deterministically classified `unknown` and no raw history command is persisted or logged + +#### Scenario: Durable version-one replay +- **WHEN** recovery loads a previously bound valid version-one plan +- **THEN** the system validates and executes it under its original exact schema without rewriting it as version two + +### Requirement: Content selection is bounded and application-first +The system SHALL select bounded image configuration and SHALL deterministically select supported unique layers under configured per-layer, aggregate compressed-byte, and unique-layer-count limits. It SHALL select every eligible unique descriptor when the complete set fits, and for larger images SHALL prioritize versioned payload classes before manifest position and stable tie-breaks. + +#### Scenario: Complete bounded image fits +- **WHEN** configuration and every supported unique layer fit all configured descriptor, aggregate-byte, and unique-layer-count bounds +- **THEN** the system selects every descriptor regardless of history class + +#### Scenario: Large image requires prioritization +- **WHEN** all new unique layers cannot fit the configured aggregate bounds +- **THEN** the system greedily considers `copy_add`, `app_config_run`, `package_run`, `unknown`, `other_run`, and `bulk_data` in that order, then highest position, smallest compressed size, and lexical digest + +#### Scenario: Giant base or model layer +- **WHEN** a layer exceeds the configured per-layer limit +- **THEN** the layer is not downloaded by the normal layer scanner and coverage records `layer_too_large` + +#### Scenario: Image byte budget is exhausted +- **WHEN** another unscanned layer would exceed the remaining per-image budget +- **THEN** the layer remains unselected and coverage records `image_budget_exhausted` + +#### Scenario: Unique layer count is exhausted +- **WHEN** another unscanned unique layer would exceed the configured selected-layer count +- **THEN** the layer remains unselected and coverage records `layer_limit_exhausted` + +#### Scenario: Shared layer is already covered +- **WHEN** a layer digest has successful coverage under the matching execution policy +- **THEN** the image reuses that coverage without consuming its transfer-byte or new-execution-count budget + +#### Scenario: Duplicate positions share one digest +- **WHEN** multiple positions in one manifest reference the same eligible immutable digest +- **THEN** the system selects all matching positions but budgets and executes that digest at most once + +#### Scenario: Selector input changes +- **WHEN** a classifier rule, class order, supported-media rule, or selection bound changes +- **THEN** the system derives a different versioned `selection_policy_sha256` + +### Requirement: Layer coverage is globally deduplicated and fenced +The system SHALL maintain one authoritative content-scan state per immutable digest and scan-execution policy and SHALL change successful coverage only through matching reservation, lease, plan, and ingestion fences. Selection budgets, classifier weights, retries, lease timing, and checkpoint scheduling MUST NOT partition otherwise identical successful execution evidence. + +#### Scenario: Concurrent images share a layer +- **WHEN** two image plans reference the same unscanned digest under the same execution policy concurrently +- **THEN** at most one reservation owns its active scan and the other image records shared pending work without duplicate execution + +#### Scenario: Selector policy changes only +- **WHEN** an immutable digest was covered successfully and a later plan changes only selector or scheduling policy +- **THEN** the later plan reuses the existing execution-policy coverage without launching another scan + +#### Scenario: Execution semantics change +- **WHEN** the scanner fingerprint, content validator, or scan-affecting archive semantics differ +- **THEN** the system uses a distinct execution-policy namespace and does not reuse incompatible successful coverage + +#### Scenario: Compatible legacy coverage exists +- **WHEN** an exact valid version-one plan proves a linked digest is durably covered with execution semantics identical to the requested version-two policy +- **THEN** the binding transaction may create a covered version-two alias that retains the original successful reservation, plan, byte count, and completion provenance + +#### Scenario: Legacy evidence is incomplete or ambiguous +- **WHEN** legacy evidence is pending, leased, submitted, failed, orphaned, malformed, or not exactly execution-compatible +- **THEN** the system does not alias it and requires normal fenced execution under the new policy + +#### Scenario: Successful matching ingestion +- **WHEN** a result bundle contains a successful execution for a blob leased by its exact bound plan +- **THEN** ingestion marks that digest globally covered for the matching execution policy in the same durable transaction + +#### Scenario: Stale completion +- **WHEN** a bundle or worker presents an expired, refunded, or mismatched blob lease +- **THEN** it cannot mark the digest covered or advance image coverage + +#### Scenario: Reservation is refunded +- **WHEN** an image reservation is durably refunded before handoff +- **THEN** only blob leases owned by that reservation are released for bounded reclamation + +### Requirement: Image coverage is explicit and honest +The system SHALL persist and expose selected, covered, shared-pending, failed, and intentionally skipped content for each immutable image plan, including bounded payload class, exact reason, and execution and selector policy identities. Configuration history classification SHALL influence priority only and SHALL NOT establish content identity or successful coverage. + +#### Scenario: Every descriptor is covered +- **WHEN** configuration and all image layer positions have successful compatible execution-policy coverage +- **THEN** the image records complete content coverage + +#### Scenario: Bounds skip content +- **WHEN** one or more descriptors are excluded by configured size, budget, count, or format bounds +- **THEN** the image may finish as bounded partial coverage but SHALL NOT report complete content coverage + +#### Scenario: Adaptive priority skips content +- **WHEN** a large-image descriptor loses deterministic selection to a higher-priority candidate +- **THEN** its class, declared bytes, omission reason, and partial image scope remain durable without persisting raw history + +#### Scenario: Selected content remains retryable +- **WHEN** at least one selected blob failed retryably or is actively covered by another reservation +- **THEN** the image remains deferred without claiming complete coverage + +#### Scenario: Selected content exhausts retries +- **WHEN** required selected content reaches its terminal attempt limit +- **THEN** the image receives terminal incomplete disposition with durable coverage detail + +#### Scenario: Covered duplicate appears at multiple positions +- **WHEN** one successfully covered digest backs multiple positions in an image +- **THEN** every matching position records compatible covered scope without duplicate execution + +### Requirement: Layer work resumes without repeating completed content +The system SHALL resume an incomplete image from its earliest complete descriptor-position selection map for the same manifest and selector policy, SHALL NOT expand or contract that selection because mutable coverage changed, and SHALL NOT relaunch content with compatible successful execution-policy coverage. It MAY lease a bounded deterministic batch while retaining independent per-blob fences. + +#### Scenario: Parent image retries +- **WHEN** an image retry follows partial layer completion +- **THEN** the new plan reuses the exact frozen selection, reuses compatible covered digests, and leases only remaining eligible content + +#### Scenario: Covered content changes the available budget +- **WHEN** a selected digest becomes globally covered after the first plan was bound +- **THEN** a descriptor skipped by the original byte or count budget remains skipped on every later checkpoint under that selector policy + +#### Scenario: Execution policy changes +- **WHEN** the same frozen selector baseline runs under a new incompatible execution policy +- **THEN** its selected positions remain unchanged while only content lacking compatible coverage becomes executable + +#### Scenario: Bounded multi-blob checkpoint +- **WHEN** multiple remaining selected digests fit the configured checkpoint count and byte targets +- **THEN** one reservation may lease that deterministic batch while each digest retains an independent lease token, attempt, execution record, and ingestion transition + +#### Scenario: Process crashes after one layer +- **WHEN** one layer was durably ingested before a later checkpoint or parent process failed +- **THEN** recovery preserves the completed layer and reclaims only unfinished leased content + +#### Scenario: Batch fails before durable handoff +- **WHEN** a process scans one or more leased blobs but crashes before their result bundle is durably accepted +- **THEN** none of those un-ingested blobs becomes covered and exact lease recovery remains required + +### Requirement: Full-image compatibility and deterministic canary are retained +The system SHALL retain the existing full-image scanner, timeout-only canary, and legacy layer path behind configuration. It SHALL additionally provide deterministic `adaptive-canary` and `adaptive` version-two modes, fail closed to full execution without a matching passed rollout gate, and preserve full-image rollback without deleting durable layer state. + +#### Scenario: Full mode +- **WHEN** Docker layer mode is disabled or set to `full` +- **THEN** the existing immutable full-image execution path remains authoritative + +#### Scenario: Legacy canary retry +- **WHEN** an image with a durable previous full-image timeout is retried under the existing canary +- **THEN** its immutable manifest digest selects the same legacy scanner mode as its previous attempt + +#### Scenario: Non-timeout image during legacy canary rollout +- **WHEN** an image has no durable previous full-image command timeout +- **THEN** existing canary configuration keeps that image on the full-image execution path + +#### Scenario: Adaptive canary assignment +- **WHEN** a matching passed shadow gate permits `adaptive-canary` +- **THEN** a versioned stable hash of every immutable manifest identity selects the same configured adaptive cohort across retries while non-members remain full-image controls + +#### Scenario: Adaptive gate is absent or stale +- **WHEN** adaptive configuration lacks a completed passing report for its exact scan, execution, and selector policy hashes +- **THEN** no adaptive plan is bound and the claim uses full-image execution + +#### Scenario: Broad adaptive mode +- **WHEN** matching shadow evidence passes and the low-percentage production canary remains within safety gates for at least one repository-refresh interval +- **THEN** operators may explicitly configure `adaptive` for all eligible immutable Docker claims + +#### Scenario: Layer mode rollback +- **WHEN** operators return configuration from `layer`, `canary`, `adaptive-canary`, or `adaptive` to `full` +- **THEN** new claims use full-image execution without deleting durable layer plans, coverage, or rollout evidence + +### Requirement: Controlled evidence gates production rollout +The system SHALL compare adaptive scanning with completed full-image controls through a bounded non-authoritative evaluator and SHALL keep adaptive production modes fail-closed until a policy-exact aggregate report passes security, coverage, completion, and throughput gates. Shadow execution MUST NOT mutate authoritative queue, reservation, coverage, finding, candidate, keycheck, projection, or source-counter state. + +#### Scenario: Controlled adaptive shadow cohort +- **WHEN** the evaluator runs against 50-100 exact immutable images with completed full-image controls under one scan fingerprint +- **THEN** it executes the candidate adaptive policy under the same bounded scanner semantics and records only aggregate policy hashes, counts, slot milliseconds, failures, thresholds, and timestamps + +#### Scenario: Shadow identity comparison +- **WHEN** routed and detector recall are calculated +- **THEN** `(service, provider_key_hash)` and `detector_secret_hash` sets exist only in protected memory long enough to calculate aggregate full, adaptive, and intersection counts + +#### Scenario: Shadow privacy +- **WHEN** a shadow run completes, fails, or logs diagnostics +- **THEN** no target name, raw config command, finding, secret, provider material, identity set, or Registry bearer is persisted in rollout evidence or ordinary logs + +#### Scenario: Shadow non-authority +- **WHEN** shadow execution emits candidate material or completes a blob +- **THEN** it cannot create authoritative findings or candidates, route keychecks, mark global blob coverage, alter image disposition, or change production mode + +#### Scenario: Routed recall or slot-time gate fails +- **WHEN** paired routed-identity recall is below 85% or aggregate adaptive slot time exceeds 40% of aggregate full slot time +- **THEN** the report does not authorize adaptive production execution + +#### Scenario: Safety or completion gate fails +- **WHEN** fewer than 50 paired controls complete, any digest/fence/containment requirement fails, omissions are unaccounted, or resource, quarantine, projection, credential, or source health regresses +- **THEN** the report does not authorize adaptive production execution + +#### Scenario: Policy changes after a passing report +- **WHEN** scan, execution, or selector policy identity changes +- **THEN** the earlier report is stale and adaptive execution remains fail-closed until the new exact policy passes another shadow cohort + +#### Scenario: Acceptance criteria pass +- **WHEN** 50-100 paired controls retain at least 85% routed identities at no more than 40% slot time with every safety gate satisfied +- **THEN** operators may start a low deterministic adaptive canary but SHALL NOT enable broad adaptive mode until that canary remains stable for at least one repository-refresh interval diff --git a/openspec/changes/add-adaptive-docker-payload-scanning/tasks.md b/openspec/changes/add-adaptive-docker-payload-scanning/tasks.md new file mode 100644 index 0000000..1fe17f8 --- /dev/null +++ b/openspec/changes/add-adaptive-docker-payload-scanning/tasks.md @@ -0,0 +1,46 @@ +## 1. Add Versioned Adaptive Policy Inputs + +- [x] 1.1 Add fail-closed `adaptive-canary` and `adaptive` configuration/CLI parsing, bounded selector and checkpoint settings, and stable immutable-manifest assignment +- [x] 1.2 Separate scanner, execution, and selector policy hashes so selection and scheduling changes do not partition compatible successful blob coverage +- [x] 1.3 Fetch and verify bounded image configuration, map aligned non-empty history to ordered layers, and emit only versioned bounded payload classes with deterministic `unknown` fallback +- [x] 1.4 Add exact version-two plan validation and canonical hashing while retaining unchanged version-one plan and execution validation + +## 2. Migrate Durable Policy And Evidence State + +- [x] 2.1 Add and exactly validate the image-coverage selector-policy column, policy-specific baseline index, aggregate shadow-report state, timing/fence fields, and migration marker +- [x] 2.2 Backfill selector identities transactionally from exact linked version-one plans and fail the migration on orphaned, malformed, or ambiguous coverage rows +- [x] 2.3 Implement transactionally fenced legacy-to-execution-policy coverage aliasing only for exact successful semantically compatible evidence +- [x] 2.4 Add stopped-runtime migration preconditions and prove rollback leaves all legacy plans and coverage rows intact + +## 3. Implement Adaptive Selection And Resume + +- [x] 3.1 Select all supported unique descriptors for complete bounded images and implement deterministic class, position, size, and digest ordering for larger images +- [x] 3.2 Persist exact classes and selected, reused, duplicate, unsupported, oversized, byte-budget, and count-budget reasons with honest partial coverage +- [x] 3.3 Freeze the earliest complete descriptor-position map by queue, manifest, and selector policy across checkpoints, retries, and execution-policy changes +- [x] 3.4 Lease deterministic bounded multi-blob checkpoints while retaining independent digest locks, lease tokens, attempts, execution records, and reclaim behavior +- [x] 3.5 Execute and ingest version-two plans with mixed per-blob outcomes, immutable class/provenance metadata, and no repeat work for compatible covered digests + +## 4. Add Non-Authoritative Shadow Gates + +- [x] 4.1 Implement a bounded operator-invoked paired evaluator for 50-100 completed full-image controls using the exact candidate scan, execution, and selector policies +- [x] 4.2 Derive routed and detector identity intersections only in protected memory and persist only aggregate counts, slot timing, failures, thresholds, policy hashes, and timestamps +- [x] 4.3 Fence shadow execution from queue disposition, reservations, global coverage, findings, candidates, keychecks, projections, source counters, and automatic mode changes +- [x] 4.4 Require a matching completed report with routed recall at least 85% and adaptive/full slot ratio at most 40% before adaptive canary assignment +- [x] 4.5 Expose aggregate adaptive selection, reuse, omission, checkpoint, completion, slot-time, and gate metrics without target or secret material + +## 5. Verify Safety And Behavior + +- [x] 5.1 Add unit tests for bounded history parsing, secret-free class persistence, all-fit selection, large-image ranking, stable hashes, exact reasons, and malformed-plan rejection +- [x] 5.2 Add PostgreSQL tests for atomic migration/backfill, selector-frozen resume, compatible coverage aliasing, incompatible policy partitioning, concurrent deduplication, and stale fences +- [x] 5.3 Add checkpoint tests proving bounded multi-blob mixed outcomes, crash recovery, independent attempts, and no post-coverage selection expansion +- [x] 5.4 Add rollout tests for stable adaptive canary assignment, stale/missing gate fallback, shadow non-authority/privacy, aggregate recall, and claim-through-handoff slot timing +- [x] 5.5 Run targeted unit, runtime-safety, migration, query-shape, and real PostgreSQL integration suites plus strict OpenSpec validation +- [x] 5.6 Preserve deterministic warning retryability so incomplete findings remain visible without repeated blob downloads or false successful coverage +- [x] 5.7 Suppress target labels in private shadow filter logs and expose only fixed aggregate failure categories for future evidence + +## 6. Migrate And Roll Out Conservatively + +- [x] 6.1 Stop runtime, verify lease quiescence, apply the additive migration, restart in unchanged timeout-only canary mode, and verify source/keycheck/projection/quarantine health +- [ ] 6.2 Run the aggregate-only shadow evaluator on 50-100 completed controls and keep adaptive execution disabled unless every recall, timing, completion, privacy, and safety gate passes +- [ ] 6.3 Enable a low deterministic adaptive canary only after a matching passing report and monitor it for at least one repository-refresh interval +- [ ] 6.4 Expand canary or enable broad adaptive mode only if runtime gates remain satisfied; otherwise return new claims to full or timeout-only canary without deleting audit state diff --git a/openspec/changes/add-layer-aware-docker-scanning/.openspec.yaml b/openspec/changes/add-layer-aware-docker-scanning/.openspec.yaml new file mode 100644 index 0000000..b4b3ece --- /dev/null +++ b/openspec/changes/add-layer-aware-docker-scanning/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-01 diff --git a/openspec/changes/add-layer-aware-docker-scanning/design.md b/openspec/changes/add-layer-aware-docker-scanning/design.md new file mode 100644 index 0000000..45329c9 --- /dev/null +++ b/openspec/changes/add-layer-aware-docker-scanning/design.md @@ -0,0 +1,254 @@ +## Context + +DockerHub currently queues immutable `repository@sha256:` targets and gives each target to TruffleHog's Docker source as one indivisible operation. The process must fetch and inspect every layer before the external 600-second deadline. Large model images contain tens of compressed GiB, commonly dominated by one binary/model-weight layer, so two such targets can occupy both Docker workers while still producing only partial findings. + +Measured over 48 hours, 196 deadline terminations consumed about 32.7 Docker worker-hours. Every timeout emitted findings, but all timeout findings routed to only 14 provider keys and none was usable. The queue currently resets target attempts after a timeout and the timeout disposition bypasses the maximum-attempt check, so the same immutable target can restart from byte zero indefinitely. + +Registry manifest resolution already obtains the exact platform child manifest and ordered layer digests. The missing information is each descriptor's compressed size, media type, image configuration descriptor, durable per-layer coverage, and a scanner capable of processing one bounded content blob independently. + +The runtime must retain its authenticated Docker pool, immutable digest identities, slot-first PostgreSQL admission, result-bundle fencing, Windows Job containment, output bounds, private temporary storage, and fail-closed behavior. Bearer tokens and provider keys must remain memory-only or in their existing protected stores. + +## Goals / Non-Goals + +**Goals:** + +- End unbounded timeout retries while preserving findings emitted before a deadline. +- Scan image configuration and useful application layers without downloading giant model/data layers. +- Resume image coverage after failure at layer granularity rather than restarting completed work. +- Scan each immutable content digest once globally and reuse its durable coverage across images. +- Keep download, disk, archive expansion, command execution, and database state bounded and fenced. +- Report exact selected, covered, skipped, failed, and shared-pending scope for every image. +- Prove the strategy against full-image control results before broad production enablement. + +**Non-Goals:** + +- Reconstruct a runnable merged container root filesystem. +- Claim complete image coverage when configured size or format bounds skip content. +- Download giant layers in a separate long-running production lane in the initial change. +- Implement arbitrary Registry hosts, private registries, cross-host credential forwarding, or resumable CDN downloads. +- Change global guaranteed scan slots, Docker worker count, keycheck classification, or non-Docker scanners. +- Delete the existing full-image implementation or schema on rollback. + +## Decisions + +### 1. Keep the image target, bind a fenced layer plan after claim + +The existing immutable image target remains the queue and reporting identity. After slot-first admission claims an image, a dedicated PostgreSQL connection resolves its exact manifest and binds a canonical `docker_layer_plan` to that result reservation under the queue lease/event fences, analogous to exact Git plan binding. + +The plan contains a version, repository, platform manifest digest, configuration descriptor, ordered layer descriptors, configured byte limits, selection reason for each descriptor, and a canonical SHA-256. The reservation stores the canonical JSON and hash. Replay is idempotent; changed replay or manifest mismatch is a conflict. + +This avoids multiplying normal target-queue rows and keeps one authoritative parent disposition while still permitting durable child coverage. + +Alternatives rejected: + +- A separate full-image heavy lane still redownloads all content and cannot resume. +- One target-queue row per layer complicates parent completion, finding attribution, and alternate repository fetch sources. +- Encoding mutable coverage metadata into queue identity would break immutable image deduplication. + +### 2. Add durable content and image-coverage tables + +Add PostgreSQL-compatible tables through the additive runtime-safety migration: + +- `docker_content_blobs`: one row per valid SHA-256 content digest, descriptor kind (`config` or `layer`), declared compressed bytes, media type, state, bounded attempts, active reservation/lease fence, completion metadata, and last bounded error. +- `docker_image_blob_coverage`: one row per platform manifest and content digest with ordered position, selected/skipped reason, plan hash, and covered timestamp. + +Successful content state is global because the digest authenticates the bytes. The current image repository is retained only as the bounded fetch source in the reservation plan. If one reservation owns a blob, another image records it as shared-pending and defers without duplicating the download. Expired/refunded reservation leases are reclaimable. + +Blob completion changes only during ingestion of a matching reservation plan and matching per-blob execution record. Refund/recovery releases matching blob leases. A stale token cannot mark coverage or insert findings. + +### 3. Select content by configurable byte budget, top layers first + +The image configuration is always selected within a small independent hard bound because `Env`, labels and history are high-value and cheap. + +Layer selection walks the manifest from highest layer to base layer. Already-covered digests require no byte budget. New layers are selected while both the per-layer compressed-byte cap and remaining per-image compressed-byte budget permit them. Non-selected descriptors are recorded with explicit reasons such as `layer_too_large`, `image_budget_exhausted`, or `unsupported_media_type`. + +Initial numeric defaults are chosen only after the controlled spike. Configuration always has hard upper bounds; invalid values fail closed to full-image mode while the production gate is disabled. + +This is a deliberate partial-coverage policy. It targets application/config layers and globally amortizes common base layers without pretending that skipped model weights were scanned. + +Alternatives rejected: + +- A whole-image size cutoff can discard a small valuable application layer sitting above giant weights. +- Selecting oldest/base layers first spends budget on widely shared dependencies before application content. +- Inferring binary content without downloading a compressed tar stream is not reliable. + +### 4. Stream, verify, scan, and delete one blob at a time + +For each blob leased by the reservation: + +1. Obtain an account-scoped Registry bearer through the existing Docker account manager. +2. Request the exact Registry blob with a bounded custom redirect policy. Redirects must remain HTTPS, must not contain userinfo, must reject local/private destinations, and must never receive the Registry Authorization header on another host. +3. Stream into a private bounded work file while computing SHA-256. Reject excess bytes, short bodies, digest mismatch, unsupported media types, disk-floor violations, and response deadline exhaustion. +4. Scan configuration JSON directly. Scan supported layer archives with TruffleHog `filesystem` under the existing OwnedProcess Job, timeout, output cap, configured detector policy, archive size/depth/time bounds, and normal process priority. +5. Delete the blob work file before releasing the scan slot. No layer bearer, provider key, or raw result is written outside existing protected result artifacts. + +Each layer runs as its own bounded TruffleHog command. This adds small startup overhead but gives exact completion, global deduplication, and restart from the first unfinished layer. Findings are enriched with immutable image, blob digest, kind, and layer position before normal bundle staging. + +The controlled spike must prove that the installed TruffleHog build correctly scans supported real layer archive media types. Unsupported formats remain explicit uncovered scope rather than being silently accepted. + +### 5. Parent disposition derives from explicit child execution + +The result bundle carries the exact bound plan plus one execution record per claimed blob. Ingestion verifies plan hash and lease ownership before updating blob state. + +- Successful blob command and digest verification marks that blob covered globally. +- Incomplete blob execution preserves its findings but does not mark it covered. +- Retryable blob failures release it to bounded retry; terminal failures remain explicit uncovered scope. +- A blob active under another valid reservation causes a short parent deferral without charging a content attempt. +- An image whose selected blobs are covered and whose remaining blobs are intentionally skipped completes with `coverage_complete=false` and detailed reasons. +- An image with retryable selected work remains deferred; exhausted selected work becomes terminal failed/degraded according to the bound policy. + +The existing full-image path remains available as a rollback/control path. + +### 6. Fix timeout accounting before layer rollout + +Both production-v2 and legacy completion paths stop resetting attempts for target-scoped timeouts. Timeout disposition checks the configured maximum before returning deferred. Source-wide infrastructure failures may retain their existing attempt-refund semantics. + +A stopped-runtime repair reconciles only unfenced Docker queue rows whose latest durable result is a command timeout. It derives prior immutable-target attempt count from durable scans, sets the queue attempt count up to the configured maximum, and terminally closes already exhausted rows. It does not requeue or mutate active leases, reservations, findings, or successful targets. + +### 7. Roll out through deterministic modes and evidence gates + +Configuration exposes `full`, `canary`, and `layer` modes. Canary eligibility requires a durable previous full-image command timeout, and membership within that eligible set is a stable hash of immutable manifest digest. New images and normally completed controls therefore remain on the full path during canary rollout. Defaults remain `full` until migration and spike criteria pass. + +The offline spike compares 20-50 timeout-heavy images and completed controls without printing findings or keys. It records bytes transferred, wall/slot time, peak resource use, distinct detector identities, and routed key identity recall. Production canary additionally tracks coverage reasons, blob reuse, timeout rate, keycheck candidate yield, and strict usable yield. + +Broad enablement requires: + +- no credential/token persistence regression; +- no stale-fence or duplicate-blob completion; +- exact digest verification for every covered blob; +- material byte and slot-hour reduction on heavy images; +- all routed key identities from completed control images retained, unless an explicitly reviewed coverage bound explains the difference; +- no quarantine growth, projection regression, or source failure increase. + +## Risks / Trade-offs + +- [Secrets can exist in skipped giant layers] -> Record explicit incomplete scope, keep configurable budgets, compare controls, and retain full mode for targeted replay. +- [Scanning individual layers can report files deleted by later whiteouts] -> Preserve layer provenance and treat this as historical image-content evidence rather than merged-root truth. +- [Archive support differs by media type] -> Verify installed binary in the spike and mark unsupported formats uncovered. +- [Global deduplication can be poisoned by stale completion] -> Require streamed digest verification plus reservation/lease/plan fences in the same ingestion transaction. +- [Registry blob redirects introduce SSRF or credential-forwarding risk] -> Use a bounded validated redirect implementation and strip authorization across hosts. +- [Per-layer process startup adds overhead for tiny layers] -> Batch measurement first; skip already-covered layers and permit a bounded future batching optimization only if needed. +- [A crash can strand blob leases] -> Tie leases to result reservations, release on refund, and permit exact expiry reclamation. +- [Database state grows with layer relationships] -> Enforce descriptor count bounds, compact metadata, indexed identities, and retention metrics. +- [Layer mode can reduce broad secret coverage] -> Report coverage honestly and retain deterministic full-mode controls and rollback. + +## Migration Plan + +1. Ship and test bounded timeout accounting independently; repair exhausted historical timeout rows with sources stopped. +2. Add nullable reservation plan columns, content/coverage tables, indexes, runtime validation, and additive migration marker. Keep mode `full`. +3. Run a local controlled spike against retained immutable targets and select conservative byte/archive defaults from evidence. +4. Deploy code with `full` mode, migrate offline, restart, and verify no behavior change. +5. Enable deterministic low-percentage canary only for DockerHub images whose latest durable full-image result timed out. Monitor at least one full repository-refresh interval and sufficient heavy-image samples. +6. Increase the timeout-fallback canary only after acceptance gates pass. Keep broad `layer` mode disabled until completed-control routed recall becomes adequate under revised bounds. +7. Roll back by returning mode to `full`. Durable layer tables and nullable columns remain for audit and future resume; no destructive migration is required. + +## Capability Spike Evidence + +The installed `C:\Tools\trufflehog.exe` development build was exercised through 39 bounded +`filesystem` invocations over synthetic direct tar, gzip-tar, zstd-tar, Docker outer-tar, and OCI +outer-tar fixtures. All invocations exited successfully and all 24 expected synthetic findings were +preserved. Extensionless gzip and zstd blobs were content-sniffed successfully. + +The minimum archive depth was two for a direct tar, three for direct compressed layers, and four +for an OCI/Docker archive containing a compressed layer. Shallower bounds produced an explicit +non-fatal `max archive depth reached` diagnostic. `--archive-max-size` was proven to be a per-member +bound rather than a cumulative compressed-image bound, so application-side per-blob and aggregate +byte limits remain mandatory. + +Initial conservative implementation bounds are therefore archive depth four, archive member size +256 MiB, archive timeout 30 seconds, one-GiB aggregate selected compressed bytes per image, 256 MiB +per selected layer, and filesystem concurrency two. These are implementation starting points, not +broad-rollout acceptance evidence. The required aggregate timeout-heavy/completed-control image +comparison remains an explicit gate before canary expansion. + +The controlled aggregate comparison then ran against ten repeatedly timed-out immutable images and +ten longest completed controls. No target names, findings, keys, or credentials were printed or +persisted. Exact manifest resolution succeeded for all 20 images. The timeout-heavy manifests +contained 341.68 GB of compressed descriptors; the bounded layer policy selected 944.02 MB (0.28%) +and processed 84 blobs in 313.03 worker-seconds versus 6,015.31 historical full-image seconds +(5.2%). All selected timeout-heavy blobs completed. The bounded path produced three routed +credential identities, none overlapping the two identities in the historical incomplete full +results, so it added useful scope while avoiding another byte-zero full-image retry. + +The completed controls contained 98.05 GB; the policy selected 1.61 GB (1.64%) and processed 77 +blobs in 357.83 worker-seconds versus 5,956.38 historical seconds (6.0%). Five blobs reported +bounded incomplete chunk processing. Only four of 19 historical routed identities were retained +(21.1% recall), and distinct detector-identity recall was 19 of 326 (5.8%). System sampling across +the run observed average CPU 20.27%, peak CPU 44.48%, minimum available physical memory 17.38 GB, +and peak committed memory 30.00 GB; no resource-limit failure occurred. + +This evidence accepts the existing 1 MiB config, 256 MiB per-layer, 1 GiB aggregate, eight-layer, +archive-depth-four, 256 MiB archive-member, 30-second archive, 600-second blob, and filesystem +concurrency-two bounds only for a deterministic timeout-fallback canary. It rejects broad random +canary or broad layer mode because completed-control routed recall failed the acceptance gate. Full +mode remains authoritative for new and normally completed images. + +The post-refinement regression gate passed 329 related tests, including bounded transfer and content +validation, parent/blob timeout accounting, PostgreSQL migration atomicity, policy-scoped global +deduplication, stale fences, reclaim/quarantine, multi-checkpoint resume, and the durable full-timeout +canary-eligibility transition. Strict OpenSpec validation passed, and no application bytecode was +present after the run. + +Rejected approaches from the capability spike are `--force-skip-archives` (it suppresses expected +archive findings), relying on media-type labels without content verification, and relying on +TruffleHog's `--archive-max-size` as an outer download or aggregate expansion bound. The exact +development binary must remain fingerprinted by a capability contract because its reported version +does not identify a stable release. + +## Initial Production Canary Evidence + +The additive migration and guarded historical timeout repair were applied offline before the +timeout-only canary. Runtime first restarted in full mode with no layer rows, and the bounded canary +was then enabled at 2,500 basis points only for images whose latest durable full-image result was a +command timeout. New and normally completed images remained on the full scanner. + +The first selected production image completed nine unique content checkpoints plus one final +no-work completion checkpoint. The nine bounded blobs transferred 327,512 bytes and completed in +33.176 seconds of aggregate parent duration, including 9.358 seconds of transfer and 23.496 seconds +of contained filesystem scanning. All nine blobs reached policy-scoped global coverage, the parent +finished `done`, no selected continuation remained, and quarantine stayed unchanged at 192 items / +337,349,428 bytes. The prior full-image path for this eligibility class reached the 600-second +deadline. + +The initial run exposed and then verified a selection-continuity invariant: per-image byte/layer +bounds must apply to the first immutable selection set, not be recomputed after each covered blob. +The binder now reuses the earliest exact `(queue, manifest, coverage policy, position)` selection map +on every later reservation. A PostgreSQL regression proves that layers skipped by the original +count budget remain skipped after selected layers become globally covered. Production replay then +completed without leasing content outside the original config-plus-eight-layer selection. + +This single successful image proves the end-to-end checkpoint, resume, bounded selection and final +completion paths, but is not enough evidence to increase the 25% timeout-only canary. Expansion +still requires a longer observation window and more naturally eligible timeout samples. + +The following overnight window added a second timeout-only image before a host reboot. Across both +images, twelve unique blobs reached policy-scoped coverage with no retryable or terminal blob +failure. The layer path used 53.192 seconds and transferred 79,614,382 bytes, compared with 1,201.330 +seconds consumed by the immediately preceding full-image timeout attempts. One parent completed; +the second retained six exact selected checkpoints for durable resume. No layer findings or routed +candidate identities were produced in this small sample, and quarantine remained unchanged. + +Because this installation is an experimental rather than production service, the operator approved +expanding the stable timeout-only cohort from 2,500 to 10,000 basis points. This does not enable broad +layer mode: every new or normally completing image still uses the full scanner, and only an image +with a durable prior full-image timeout may enter the bounded layer fallback. Broad layer mode +remains rejected by the completed-control recall result. Task 7.5 remains open until the expanded +cohort produces additional completed parents and a stable runtime observation window. + +The expanded cohort exposed one additional metadata-normalization defect: identical content digests +and sizes can be referenced through equivalent Docker and OCI media-type labels. Treating the label +text itself as immutable metadata caused the Docker source to stop fail-closed before handoff. The +binder now compares the validated semantic content class (`config-json`, `layer-tar`, `layer-gzip`, +or `layer-zstd`) while still rejecting kind, byte-size, and compression-class conflicts. A real +PostgreSQL concurrency regression and the related layer/runtime suites passed 126 tests. After the +restart, 116 layer reservations were acknowledged in the first ten minutes, global covered blobs +grew from 9 to 113, no blob entered a failed state, the source remained running, and quarantine was +unchanged. The immediate digest queue remained intentionally thin (six due and eighteen delayed), +while 12,679 repository anchors remained available to refill it after claimable digest work drains. + +## Open Questions + +- Which revised per-layer and per-image bounds can improve completed-control routed recall beyond the measured 21.1% without losing the measured slot-hour advantage? +- Which OCI/Docker layer compression media types does the installed TruffleHog filesystem source handle reliably? +- Is one TruffleHog process per selected layer sufficiently efficient, or is a later bounded multi-layer archive batch warranted? +- What short deferral is appropriate when all remaining selected blobs are actively leased by other reservations? diff --git a/openspec/changes/add-layer-aware-docker-scanning/proposal.md b/openspec/changes/add-layer-aware-docker-scanning/proposal.md new file mode 100644 index 0000000..8debc17 --- /dev/null +++ b/openspec/changes/add-layer-aware-docker-scanning/proposal.md @@ -0,0 +1,31 @@ +## Why + +Large Docker images currently monopolize both Docker scan workers until the 600-second deadline and can retry indefinitely because timeout completion resets the target attempt counter. Over the measured 48-hour window, hard timeouts consumed about 32.7 worker-hours while repeated partial scans produced no usable LLM access, so full-image retries are reducing useful throughput without providing proportional coverage. + +## What Changes + +- Enforce the existing bounded target-attempt policy for Docker timeouts while preserving findings emitted before termination. +- Resolve immutable image manifests into image configuration and ordered content-addressed layers with bounded size metadata. +- Scan image configuration and selected layer content under an explicit per-image byte budget instead of treating every image as an indivisible download. +- Deduplicate successful layer scans globally by immutable layer digest so shared base layers are not downloaded and scanned repeatedly. +- Prioritize upper application layers and small layers; record oversized or out-of-budget layers as explicit uncovered scope rather than silently claiming complete image coverage. +- Preserve the existing full-image path behind a rollout gate for controlled comparison and rollback. +- Repair currently deferred Docker targets whose timeout attempts were incorrectly reset. + +## Capabilities + +### New Capabilities + +- `docker-layer-content-scanning`: Bounded, content-addressed Docker config and layer scanning with global deduplication, explicit coverage, safe retry limits, and controlled rollout against the existing full-image scanner. + +### Modified Capabilities + +None. + +## Impact + +- Affects Docker Registry manifest/blob access, immutable Docker target planning, scan queue state, result metadata, and Docker source configuration. +- Adds durable PostgreSQL state for layer identities, leases, coverage, attempts, and image-to-layer plans. +- Reuses the existing authenticated Docker account pool, scan-slot limiter, Windows Job containment, bundle ingestion, findings projection, and keycheck pipeline. +- Requires an offline additive runtime-safety migration before enabling production layer scanning. +- Does not change Git, Hugging Face, keycheck classification, global guaranteed scan-slot capacity, or secret persistence boundaries. diff --git a/openspec/changes/add-layer-aware-docker-scanning/specs/docker-layer-content-scanning/spec.md b/openspec/changes/add-layer-aware-docker-scanning/specs/docker-layer-content-scanning/spec.md new file mode 100644 index 0000000..33c0c3a --- /dev/null +++ b/openspec/changes/add-layer-aware-docker-scanning/specs/docker-layer-content-scanning/spec.md @@ -0,0 +1,190 @@ +## ADDED Requirements + +### Requirement: Docker timeout retries are bounded +The system SHALL count target-scoped Docker command timeouts against the configured target-attempt maximum and SHALL preserve partial findings without creating an unbounded retry loop. + +#### Scenario: Timeout before attempt limit +- **WHEN** a Docker command times out before the configured maximum attempt +- **THEN** its emitted findings remain durable and the immutable target is deferred using the configured timeout delay without resetting its attempt count + +#### Scenario: Timeout reaches attempt limit +- **WHEN** a Docker command times out at the configured maximum attempt +- **THEN** its emitted findings remain durable and the queue records a terminal target disposition + +#### Scenario: Source-wide outage +- **WHEN** Docker execution is prevented by a source-wide infrastructure failure rather than target-scoped work +- **THEN** the existing fenced source-failure recovery policy remains applicable + +### Requirement: Immutable image content plans are bound after claim +The system SHALL resolve a claimed Docker image's exact platform manifest into a canonical bounded plan containing its configuration and ordered layer descriptors, and SHALL bind that plan to the active result reservation before content execution. + +#### Scenario: Valid immutable manifest +- **WHEN** the claimed `repository@sha256:` resolves to a valid requested-platform manifest +- **THEN** the bound plan identifies the same manifest digest and contains only bounded valid SHA-256 content descriptors, sizes, media types, and order + +#### Scenario: Changed replay +- **WHEN** the same reservation attempts to bind a different content plan +- **THEN** the system rejects the replay as a fenced conflict and executes neither plan + +#### Scenario: Invalid or oversized manifest +- **WHEN** the Registry manifest is malformed, exceeds descriptor bounds, or disagrees with the immutable target +- **THEN** the system fails closed without claiming complete content coverage + +### Requirement: Content selection is bounded and application-first +The system SHALL always select bounded image configuration and SHALL select new layers from highest to lowest under configured per-layer and per-image compressed-byte limits. + +#### Scenario: Giant base or model layer +- **WHEN** a layer exceeds the configured per-layer limit +- **THEN** the layer is not downloaded by the normal layer scanner and coverage records `layer_too_large` + +#### Scenario: Image byte budget is exhausted +- **WHEN** another unscanned layer would exceed the remaining per-image budget +- **THEN** the layer remains unselected and coverage records `image_budget_exhausted` + +#### Scenario: Shared layer is already covered +- **WHEN** a layer digest has successful global coverage +- **THEN** the image reuses that coverage without consuming its transfer budget or launching another scan + +#### Scenario: Upper and base layers both fit +- **WHEN** multiple unscanned layers fit within all configured bounds +- **THEN** the system selects them in highest-to-lowest manifest order + +### Requirement: Layer coverage is globally deduplicated and fenced +The system SHALL maintain one authoritative content-scan state per immutable digest and SHALL change successful coverage only through matching reservation, lease, plan, and ingestion fences. + +#### Scenario: Concurrent images share a layer +- **WHEN** two image plans reference the same unscanned digest concurrently +- **THEN** at most one reservation owns its active scan and the other image records shared pending work without duplicate execution + +#### Scenario: Successful matching ingestion +- **WHEN** a result bundle contains a successful execution for a blob leased by its exact bound plan +- **THEN** ingestion marks that digest globally covered in the same durable transaction + +#### Scenario: Stale completion +- **WHEN** a bundle or worker presents an expired, refunded, or mismatched blob lease +- **THEN** it cannot mark the digest covered or advance image coverage + +#### Scenario: Reservation is refunded +- **WHEN** an image reservation is durably refunded before handoff +- **THEN** only blob leases owned by that reservation are released for bounded reclamation + +### Requirement: Registry blob transfer is authenticated, bounded, and verified +The system SHALL fetch selected content from the trusted Docker Registry using the existing account pool, bounded streaming, private storage, safe redirect handling, and exact digest verification. + +#### Scenario: Valid content download +- **WHEN** the Registry returns exactly the declared bounded blob bytes whose SHA-256 matches the descriptor +- **THEN** the private work artifact becomes eligible for scanning + +#### Scenario: Cross-host redirect +- **WHEN** the trusted Registry redirects a blob request to an allowed public HTTPS content host +- **THEN** the system follows only the bounded validated redirect and does not forward Registry authorization to the other host + +#### Scenario: Unsafe redirect +- **WHEN** a blob redirect uses HTTP, userinfo, a local/private destination, or exceeds redirect bounds +- **THEN** the transfer fails closed without exposing authentication material + +#### Scenario: Size or digest mismatch +- **WHEN** streamed bytes exceed bounds, end short, or do not match the expected digest +- **THEN** the system deletes the work artifact and does not record successful coverage + +#### Scenario: Insufficient private storage +- **WHEN** the configured private work volume cannot retain its required free-space floor +- **THEN** no blob download begins and the failure receives bounded retry disposition + +### Requirement: Configuration and layers are scanned independently +The system SHALL scan bounded image configuration and each newly leased supported layer as independent contained commands while preserving image and layer provenance on findings. + +#### Scenario: Configuration contains candidate material +- **WHEN** bounded configuration JSON contains detector-matching data +- **THEN** findings identify the image and configuration digest and enter the normal result and keycheck pipeline + +#### Scenario: Supported layer completes +- **WHEN** TruffleHog filesystem scanning of a verified layer archive completes successfully +- **THEN** its findings retain image, layer digest, kind, and position provenance and the layer becomes globally covered after fenced ingestion + +#### Scenario: Layer scan is incomplete +- **WHEN** a layer command times out or exits without confirmed completion +- **THEN** emitted findings remain durable but that digest does not become covered + +#### Scenario: Unsupported media type +- **WHEN** a layer compression or media type is not supported by the validated scanner path +- **THEN** no unsafe fallback executes and image coverage records `unsupported_media_type` + +### Requirement: Image coverage is explicit and honest +The system SHALL persist and expose selected, covered, shared-pending, failed, and intentionally skipped content for each immutable image plan. + +#### Scenario: Every descriptor is covered +- **WHEN** configuration and all image layers have successful global coverage +- **THEN** the image records complete content coverage + +#### Scenario: Bounds skip content +- **WHEN** one or more descriptors are excluded by configured size, budget, or format bounds +- **THEN** the image may finish as bounded partial coverage but SHALL NOT report complete content coverage + +#### Scenario: Selected content remains retryable +- **WHEN** at least one selected blob failed retryably or is actively covered by another reservation +- **THEN** the image remains deferred without claiming complete coverage + +#### Scenario: Selected content exhausts retries +- **WHEN** required selected content reaches its terminal attempt limit +- **THEN** the image receives terminal incomplete disposition with durable coverage detail + +### Requirement: Layer work resumes without repeating completed content +The system SHALL resume an incomplete image from selected content that lacks successful global coverage and SHALL NOT relaunch completed content digests. + +#### Scenario: Parent image retries +- **WHEN** an image retry follows partial layer completion +- **THEN** the new plan reuses covered digests and leases only remaining eligible content + +#### Scenario: Process crashes after one layer +- **WHEN** one layer was durably ingested before a later layer or parent process failed +- **THEN** recovery preserves the completed layer and reclaims only unfinished leased content + +### Requirement: Full-image compatibility and deterministic canary are retained +The system SHALL retain the existing full-image scanner behind configuration. Canary layer execution SHALL require a durable previous full-image command timeout and SHALL be selected deterministically from immutable manifest identity within that eligible set. + +#### Scenario: Full mode +- **WHEN** Docker layer mode is disabled or set to `full` +- **THEN** the existing immutable full-image execution path remains authoritative + +#### Scenario: Canary retry +- **WHEN** a canary image is retried +- **THEN** its immutable manifest digest selects the same scanner mode as its previous attempt + +#### Scenario: Non-timeout image during canary rollout +- **WHEN** an image has no durable previous full-image command timeout +- **THEN** canary configuration keeps that image on the full-image execution path + +#### Scenario: Layer mode rollback +- **WHEN** operators return configuration from `layer` or `canary` to `full` +- **THEN** new claims use full-image execution without deleting durable layer audit state + +### Requirement: Controlled evidence gates production rollout +The system SHALL compare layer scanning with completed full-image controls and SHALL keep broad production layer mode disabled until security, coverage, and throughput gates pass. + +#### Scenario: Controlled spike +- **WHEN** the offline spike runs against bounded heavy and completed-control samples +- **THEN** it records aggregate bytes, wall time, slot time, coverage, distinct detector identities, and routed key recall without printing findings or keys + +#### Scenario: Acceptance criteria fail +- **WHEN** digest integrity, fence safety, routed-key recall, resource bounds, or throughput criteria fail +- **THEN** production remains in full mode + +#### Scenario: Completed-control recall fails but timeout fallback passes +- **WHEN** bounded layer scanning materially reduces timeout-heavy work but does not retain completed-control routed-key recall +- **THEN** operators may canary only prior full-image timeout retries and SHALL NOT enable broad layer mode + +#### Scenario: Acceptance criteria pass +- **WHEN** the controlled spike and deterministic production canary satisfy all defined gates +- **THEN** operators may increase canary coverage or enable layer mode through configuration + +### Requirement: Historical timeout state is repaired safely +The system SHALL reconcile incorrectly reset Docker timeout attempts only while sources are stopped and only for unfenced immutable targets backed by durable timeout results. + +#### Scenario: Exhausted historical timeout target +- **WHEN** an unfenced deferred Docker target has durable timeout executions at or above the configured maximum +- **THEN** repair marks it terminal without deleting its existing findings + +#### Scenario: Active or ambiguous target +- **WHEN** a Docker target has an active lease, reservation, event fence, or ambiguous latest result +- **THEN** repair leaves it unchanged diff --git a/openspec/changes/add-layer-aware-docker-scanning/tasks.md b/openspec/changes/add-layer-aware-docker-scanning/tasks.md new file mode 100644 index 0000000..b877f26 --- /dev/null +++ b/openspec/changes/add-layer-aware-docker-scanning/tasks.md @@ -0,0 +1,45 @@ +## 1. Restore Bounded Timeout Semantics + +- [x] 1.1 Make Docker timeout disposition terminal at the configured target-attempt maximum in production-v2 and legacy paths without resetting attempts +- [x] 1.2 Add regression tests proving partial findings survive and timeout attempts stop at the configured limit +- [x] 1.3 Add a stopped-runtime guarded repair for unfenced historical Docker targets whose durable timeout attempts were reset + +## 2. Validate Layer Scanning + +- [x] 2.1 Prove the installed TruffleHog filesystem source safely scans bounded Docker gzip and supported OCI layer archives with preserved findings +- [x] 2.2 Run an aggregate-only spike across timeout-heavy and completed-control images and select conservative config, layer, image, archive, and deadline bounds +- [x] 2.3 Record spike acceptance evidence and rejected formats/approaches in the design + +## 3. Add Durable Layer State + +- [x] 3.1 Add reservation plan columns plus Docker content-blob and image-coverage tables, indexes, migration marker, and exact runtime validation +- [x] 3.2 Implement bounded canonical Docker layer-plan validation, hashing, idempotent fenced binding, and deterministic canary selection +- [x] 3.3 Implement content lease claim, expiry, refund, retry, terminal failure, and globally successful coverage transitions +- [x] 3.4 Wire matching blob execution and image coverage updates into the authoritative fenced result-ingestion transaction + +## 4. Implement Bounded Registry Content Access + +- [x] 4.1 Extend exact platform manifest resolution with bounded configuration and ordered layer size/media descriptors +- [x] 4.2 Implement authenticated Registry blob streaming with safe redirects, byte/disk/deadline bounds, private artifacts, and SHA-256 verification +- [x] 4.3 Implement contained configuration and layer archive scans with immutable image/blob provenance and deterministic cleanup + +## 5. Integrate Layer-Aware Execution + +- [x] 5.1 Select configuration and highest-first layers under per-layer and per-image byte budgets while reusing globally covered digests +- [x] 5.2 Resolve and bind the Docker layer plan after the fenced parent claim using a dedicated database connection +- [x] 5.3 Execute only leased blobs, preserve partial findings, and emit exact plan/execution/coverage metadata through result bundles +- [x] 5.4 Resume deferred images without rerunning covered blobs and defer shared active content without charging duplicate attempts + +## 6. Add Controlled Rollout and Observability + +- [x] 6.1 Add bounded `full`, deterministic `canary`, and `layer` configuration with full-image rollback +- [x] 6.2 Expose aggregate selected/covered/skipped/shared/failed bytes, blob reuse, transfer duration, timeout, and image coverage metrics without secret material +- [x] 6.3 Keep existing scan-slot, Windows Job, output, keycheck, and credential-isolation invariants unchanged + +## 7. Verify and Deploy + +- [x] 7.1 Add unit tests for plan bounds, selection order, downloader security, digest verification, provenance, timeout policy, and canary stability +- [x] 7.2 Add PostgreSQL integration tests for concurrent global deduplication, stale fences, refund/reclaim, partial ingestion, resume, and image coverage +- [x] 7.3 Run related regression suites, strict OpenSpec validation, and aggregate control comparison; document evidence +- [x] 7.4 Apply the additive migration offline, repair historical attempts, restart in full mode, and verify no behavior regression +- [ ] 7.5 Enable a bounded deterministic canary, monitor throughput/coverage/keycheck/quarantine gates, and expand only if acceptance criteria pass diff --git a/openspec/changes/add-minimal-remote-scan-workers/.openspec.yaml b/openspec/changes/add-minimal-remote-scan-workers/.openspec.yaml new file mode 100644 index 0000000..d28e909 --- /dev/null +++ b/openspec/changes/add-minimal-remote-scan-workers/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-17 diff --git a/openspec/changes/add-minimal-remote-scan-workers/baseline-evidence.md b/openspec/changes/add-minimal-remote-scan-workers/baseline-evidence.md new file mode 100644 index 0000000..ae56aeb --- /dev/null +++ b/openspec/changes/add-minimal-remote-scan-workers/baseline-evidence.md @@ -0,0 +1,128 @@ +# Task 1.3 Baseline Evidence + +## Status + +This file records the best available historical baseline for task 1.3 and its +reproducibility limits. It does not claim that a pre-change image was rerun during +this change, and it does not convert current tests into pre-change evidence. + +An immutable rerun of the exact pre-change source and image is unavailable. On +2026-09-18, the reviewer explicitly accepted this documentary baseline and waived +that rerun requirement. Task 1.3 therefore relies on the historical record below; +current regression results remain separately identified as post-change evidence. + +## Historical pre-change record + +The repository's [`DOCKER_MIGRATION.md`](../../../DOCKER_MIGRATION.md), under +"Current Verified Evidence," records a final run on 2026-09-15. It reports: + +- The selected container regression suite completed with **578 passed** and + **7 expected Windows-only tests skipped on Linux**, with no failures recorded. +- The selected suite used test image + `sha256:4ce11325728ba3e58e6643c1c8e800f317179d5c7c50e7e80568b58f62dbdfd0`. +- The separate six-test `docker/test_verify.py` suite passed on Windows and WSL + Linux; those six tests were not included in the 578 count. +- The recorded offline E2E passed its result/check gates and removed its owned + resources. This is contextual historical evidence, not a rerun for task 1.3. + +The `add-minimal-remote-scan-workers` OpenSpec change was created on 2026-09-17, +after that recorded run. The preserved records contain no pre-existing failure +for the cited 578-test selection to list separately. "No recorded failure" means +only that no failure record was found; it is not proof that no unrecorded attempt +failed. + +## Reproducibility limits + +- The old test image is unavailable and was not retrieved, rebuilt, or executed. + Current Docker runs use new, dedicated test image identities. +- This workspace has no commit history. On 2026-09-18, `git status` reported + `No commits yet on master` and every repository path as untracked; `git log` + failed because the branch has no commits. +- `WORKSPACE.md` names a source-side provenance commit, but also states that this + source-only snapshot includes modified and selected untracked files and did not + copy source Git history. It is not an immutable tree for either 2026-09-15 or + the moment immediately before the 2026-09-17 OpenSpec change. +- The image digest and prose result are therefore useful historical evidence but + cannot independently reconstruct or rerun the claimed chronology from this + repository. + +## Current expanded selection + +The current reviewed allowlist is the `SELECTION` mapping in +`tests/container_unit.py`. On 2026-09-18, its stdlib-only selection check reported +**33 modules and 649 test definitions**, with no runner skips at declaration time; +pytest parametrization may expand the executed count. + +The current mapping selects: + +```text +test_docker_foundation.py +test_container_runtime.py +test_owned_process.py +test_owned_process_linux.py +test_owned_process_boundary.py +test_runtime_bootstrap_authority.py +test_supervisor_foreground_shutdown.py +test_supervisor_startup_rollback.py +test_supervisor_managed_postgres_gate.py +test_observer_only_coordinated_shutdown.py +test_supervisor_safety.py +test_postgres_runtime.py +test_container_security.py +test_runtime_security.py +test_postgres_empty_initialization.py +test_container_migration_paths.py +test_container_provider_portability.py +test_container_e2e_helpers.py +test_container_import.py +test_container_import_config.py +test_container_projection_recovery.py +test_result_bundle_v2.py +test_pipeline_cutover_invariants.py +test_custom_provider_detector_compatibility.py +test_scan_execution.py +test_synthetic_llm_pipeline.py +test_worker_api.py +test_worker_api_runtime.py +test_worker_assignment.py +test_worker_package.py +test_remote_worker_db.py +test_admin_api.py +test_edge_deployment.py +``` + +The remote-worker, package, administration, edge, synthetic-pipeline, and bundle +modules in this current selection postdate the historical baseline. Their +presence demonstrates current review scope, not pre-change execution. + +## Current post-change evidence + +The expanded selection and isolated end-to-end gates were rerun on 2026-09-19. +They validate the completed implementation but do not replace the historical +pre-change baseline: + +```text +python -I -S -B tests/container_unit.py --check-selection +container-unit: AST OK; 33 modules, 650 test definitions, no runner skips (parametrizations expand in pytest) + +isolated Linux container selection +867 passed, 7 expected platform skips + +wsl.exe -d Ubuntu-24.04 -- python3 docker/verify.py +E2E passed; project truf-worker-test-99b193a64f46a0002e34da9c5ff8029a; artifacts removed + +python -B docker/verify_packaged_workers.py ... +Windows/Linux packaged-worker E2E passed; run 348d24046fb7b8d4; cleanup complete; foreign Docker state unchanged +Windows package manifest sha256 f0bbbbf79d9f5f3f17440a60561b307fa7bafd312ab6110b14098ff90755f8e6 +Linux worker image manifest list sha256 888bffc5519fd3a27cb52d20eaea1143f89ac9779bb2e4b0d998e450583c57a0 + +python -B docker/verify_edge_e2e.py --timeout-seconds 1200 +Edge/fail2ban E2E passed; run 597b3ff7826055e1; cleanup complete; foreign Docker state unchanged + +openspec validate add-minimal-remote-scan-workers --strict --no-interactive +Change 'add-minimal-remote-scan-workers' is valid +``` + +The current runs used only random or dedicated test-owned identities and verified +that foreign container and volume metadata remained unchanged. No current result +is represented as historical or pre-change evidence. diff --git a/openspec/changes/add-minimal-remote-scan-workers/design.md b/openspec/changes/add-minimal-remote-scan-workers/design.md new file mode 100644 index 0000000..da4a5d6 --- /dev/null +++ b/openspec/changes/add-minimal-remote-scan-workers/design.md @@ -0,0 +1,147 @@ +## Context + +The current PostgreSQL pipeline already has single-target admission, capacity reservation, immutable Git/Docker plans, scan/error classification, canonical v2 `.trb` staging, transactional ingestion, projection, and detailed keychecks. The execution seam is the claim-to-`stage_claim` path in `app/console_runner.py`, not a replacement scheduler. + +`scan_target_result()` in `app/scanner.py` dispatches existing source implementations. Bundle staging extracts candidates, including structured Postman evidence; ingestion persists findings/candidates and separately schedules projection and keycheck. JSONL projection is not a prerequisite for keycheck, and a TruffleHog `Verified` field is not a completed detailed keycheck. + +Current execution is not automatically portable to a DB-free client: process launch authority, exact Git/Docker plans, reservation accounting, and producer recovery depend on the local runtime. Recovery can inspect local PID/executable identity and local files, which cannot establish remote worker liveness. The design adapts these boundaries while keeping scanner/provider/error behavior intact. + +Development takes place in `D:\truf-workers`, a source-only snapshot of the current `D:\truf-docker` working tree, including uncommitted fixes. Production data and Git history were not copied. Existing unarchived specifications are references; `openspec/specs/` has no canonical baseline. Old three-global-permit/`S:` handoff assumptions apply to their historical local deployment, not this remote boundary. PostgreSQL remains authoritative despite older status-file accounting descriptions. + +## Goals / Non-Goals + +**Goals:** +- Move expensive download/scan work to trusted Windows/Linux clients with minimum new code and persistent state. +- Preserve scanner results, attribution, exact-plan coverage, error classification, retry decisions, candidate routing, and detailed server-side keychecks. +- Keep source/provider/scanner configuration centralized; support client-selected N slots capped by the server across a user's devices. +- Recover through a fixed 24-hour default assignment deadline, durable retries, and existing identity fencing, without heartbeat traffic. +- Secure the small public surface and test locally against empty storage and synthetic inputs. + +**Non-Goals:** +- Client detailed keychecks, a second broker/queue/result format, generic job infrastructure, batch claims, worker affinity, or worker blacklists. +- New scan retry counts, a three-attempt/dead-letter rule, exactly-once physical execution, or heartbeat/lease renewal protocols. +- Client disk encryption, mTLS, public/untrusted workers, auto-update, complex roles, multiple server replicas, or PostgreSQL container extraction. +- Production database import, production deployment, changes to the active runtime, or archiving unrelated OpenSpec changes. + +## Decisions + +### 1. Keep one runtime and add a thin remote boundary + +```text +Internet -> Caddy :443 + |-- /api/v1/worker/* [device token] -> Worker API + `-- //* [login/password] -> micro-admin + +Server runtime: PostgreSQL + existing queue/reservations + discovery/scheduling -> admission + receiver -> existing ingester -> projection + `-> detailed keycheck + +Client slot: claim -> existing download/scan -> canonical .trb -> upload/ack +``` + +Run Worker API and micro-admin as runtime-managed processes, not independent database-owning services. Keep the current dashboard backend private; any dashboard information in the admin area shares its authentication boundary. PostgreSQL, supervisor control, Caddy's control API, and raw backend ports have no public host bindings. + +Reuse `reserve_and_claim_target`, ambiguous-admission reconciliation, `scan_target_result`, bundle staging/validation, `mark_result_bundle_ready`, and ingester/projector/keycheck paths. Extract only the code needed to run one already-planned scan and build its complete bundle without PostgreSQL. Adapt DB-bound plan inputs and node-local process authority rather than giving a client a DSN or a supervisor credential. Preserve `OwnedProcess` containment and cleanup; the client must launch only the known scanner tools, not arbitrary server-supplied commands. + +Alternative rejected: copying `run_cycle_v2` unchanged or implementing a second scanner/queue. Both either retain server authority on clients or create divergent behavior. + +### 2. Central configuration, minimal device identity, compatible jobs + +Client-authored operational configuration has only server URL, opaque device token, and desired positive slot count N. Generated local pending-work state and workspace paths are not independent source configuration. The server resolves source/scanner settings, immutable plan, limits, and only the credentials required for this task. It must not send the whole config, discovery credential pool, database credentials, admin secrets, or control authority. + +Bind each device token to its owning user and a server-issued/stable worker identity; store token hashes and support revocation. Keep only the identity/quota metadata required for administration, bound to current reservations rather than creating `remote_jobs` or another scheduler. Server configuration supplies default limits; admin changes the per-user cap across all devices. + +Send protocol/build/policy compatibility metadata with normal claim traffic, not a background registration/liveness service. Validate the pinned scanner/custom detector policy and supported source execution on the client's OS/architecture before consuming a target. Pass a config snapshot or stable effective-config identity so the result stays attributable even if the server config changes mid-task. A client rejects an incompatible job instead of silently using its own settings. + +Alternative rejected: worker-owned provider settings/tokens or another config management system. Trusted clients receive the task-specific secrets they need over HTTPS; this is not a sandbox against a compromised client. + +### 3. One claim per free slot with existing backpressure + +Each free client slot requests one assignment. Effective admission is bounded by N locally, the atomic per-user server cap across devices, and existing source/global/output-capacity rules. A client advertising N=8 does not override an admin cap of 3. Concurrent claims from different devices must not overshoot a shared cap. Reducing a cap stops new admissions until usage falls below it; it does not invent cancellation semantics for already-issued work. + +Keep the current node-local scan-permit mechanics. A client work slot holds its assignment through durable result acknowledgement, which bounds pending uploads when the server is unavailable. Release the native scan permit at its existing handoff point; do not conflate that permit with remote user quota. A durably accepted bundle, acknowledged pre-bundle terminal disposition, or completed expiry recovery releases remote assignment quota exactly once; server spool credits remain governed by existing ingestion accounting. + +Persist the existing admission/request identity before the first claim request. Retry ambiguous requests with the same identity and reconcile the same reservation; do not allocate another task because a claim response was lost. Scope recovery to the authenticated device. Empty queues or denied capacity return a bounded polling delay, not a scan error or a new queue item. + +Alternative rejected: batches and client backlogs. Per-free-slot claims reuse current scheduling and avoid another recovery structure. + +### 4. Fixed expiry, no remote heartbeat + +Set `expires_at` from the server clock at assignment commit, with a configurable duration defaulting to 24 hours. The deadline includes download, scan, and upload. Ordinary API contact, retries, and client restarts do not renew it. Internal scanner timeouts and server discovery leases keep their existing meanings; local slot bookkeeping is not a worker heartbeat. + +Use one periodic runtime recovery pass for expired remote assignments, initially once per minute and configurable. Extend existing recovery ownership/state checks instead of inventing a remote liveness monitor. Reconcile reservations, target claims, output credits, and dependent exact-plan/blob leases together. Local producer-PID recovery must not release remote claims. No associated lease may silently expire earlier and allow a second owner while the remote assignment remains valid. + +Requeue unfinished expired work using the existing pre-handoff infrastructure-loss/refund path where applicable, not a fabricated scanner/provider failure or a new retry limit. The same worker may claim the task again. New issuance has a distinct current reservation/attempt identity. Both periodic recovery and result acceptance use the same atomic ownership/deadline boundary. + +Reject an unaccepted old result once its assignment expires or is superseded; it must not finish newer work, update plan coverage, or return somebody else's credit. A retry of an already accepted identical result still receives its original acknowledgement even after the deadline. Ready/accepted bundles awaiting ingestion are not unfinished worker assignments and must not be expired back into the queue. + +Trade-off accepted: a crashed client can delay work for about a day, and a partitioned client can keep physically scanning after reissue. Fencing gives one authoritative acceptance, not exactly-once physical execution. + +### 5. Reuse the full bundle and durable handoff + +The API needs only claim/reconcile, upload/acknowledge, and the existing pre-bundle failure/release outcome where necessary. Exact URL naming is implementation detail under `/api/v1/worker/`; do not add a heartbeat endpoint, remote shell, separate keycheck jobs, or a general command API. + +Pre-bundle terminal reports use the same issuance fencing and replay rules as results. Persist/retry their identity until acknowledged; a lost reply resolves to the original disposition without charging retries, updating counters, or releasing quota/credits again. A delayed report from an expired/superseded issuance cannot mutate a new issuance even when the same device owns both. Authoritative acknowledgement resolves that client slot exactly once. + +Send canonical v2 `.trb` bytes containing the existing findings, detector identities, source/origin/context, errors, metadata, candidate evidence, and exact-plan identity. Preserve structured Postman candidate extraction before ingestion. Do not send only provider names or reconstruct a reduced result JSON. Keep detailed keychecks entirely server-side after ingestion through current candidate/dedup/cache/routing logic. Retained `runtime/keychecks` outputs keep their existing location and retention rules. + +Upload a binary stream, not base64 JSON, into a bounded server-owned partial file. Enforce the existing bundle limit/capacity reservation (currently 192 MiB where configured), time limits, identity, version, and hash. Validate all paths/archive members with the existing codec; clients cannot choose arbitrary server paths. Publish durably using the existing same-filesystem atomic handoff and mark the exact reservation ready before acknowledging server custody. + +Acknowledgement means the validated bundle and its ready/recovery state survive a server restart; it does not mean projection or detailed keycheck has finished. An identical retry returns the same receipt without double ingestion/accounting. A conflicting body for an accepted identity is rejected. Interrupted uploads never count as successful scans; full-bundle retry is sufficient for the MVP, with no resumable-upload protocol. + +Keep the accepted identity, digest, and reconciliation outcome in existing authoritative records independently of the spool file. Ordinary ingestion and bundle cleanup must not erase the information needed to recover a receipt: accept-with-lost-reply, ingest, clean up, restart, and retry after the deadline must still return the original acceptance for identical bytes and reject conflicting bytes. No second receipt queue or result store is required. + +Keep the pending bundle and its assignment identity locally until durable acceptance is confirmed. After acknowledgement, existing cleanup may delete local task artifacts. A definitive fenced rejection transitions local work to an explicit stale/discard outcome with bounded cleanup, not an infinite retry or a false success. Scan errors still use the existing disposition path; HTTP retry/backoff does not consume scanner/provider retry budgets. + +Crash recovery must cover publication-before-ready and ready-before-response windows using exact current ownership and existing spool reconciliation. Never infer success from the mere presence of an unvalidated file or expire already accepted work because a producer PID is absent. + +### 6. Restricted edge and separate admin login bans + +Use direct DNS to Caddy for the initial deployment and HTTPS for all worker/admin transport. The public admin prefix contains at least 128 bits of randomness, but login/password remains mandatory for every administrative route and asset. Caddy password authentication with a supported password hash is the minimal starting point; no plaintext password configuration or public signup. Protect typed mutating actions against CSRF with validated Origin/CSRF handling. No generic supervisor-command passthrough. + +Only authenticated admin pages show operational summaries or typed user/device-token/quota and existing queue controls. Keep the standalone `/dashboard` route closed. Unknown paths return 404. Add no-store/same-origin-referrer, a restrictive compatible CSP, and production HSTS; avoid third-party assets and secrets/raw findings in diagnostics. + +Run fail2ban on the host, reading redacted Caddy authentication-failure events. Two actual bad-credential submissions to the admin boundary within ten minutes produce a 24-hour IP ban. Do not count an ordinary initial Basic-auth challenge without credentials, unrelated 404s, or Worker API authentication failures. With shared HTTPS ingress, the ban action must update an ADMIN-ONLY Caddy IP deny matcher, with validated configuration reload, not globally drop that address on port 443. Persist ban state and provide SSH unban/recovery. + +Use the direct connection address as client IP. Ignore arbitrary forwarded IP headers; a future trusted proxy requires explicit trust configuration and new tests. If network-level fail2ban actions are ever chosen instead, prove Docker forwarding-chain enforcement and preserve worker availability; a blanket host INPUT rule does not satisfy the contract. + +Alternative rejected: a secret URL as sole protection, separate dashboard exposure, or a shared admin/worker IP jail. No extra auth service, mTLS, Cloudflare dependency, or client-disk encryption is required. + +### 7. Observability from existing state + +Correlate source/target, user/worker, reservation/issuance, issue/finish time, duration, outcome, and safe error category. Expose per-worker unfinished, completed, failed, and expired counts and last authenticated API contact. Distinguish bundle acceptance from later ingestion/keycheck completion and define counters from existing authoritative transitions so duplicate uploads do not inflate them. + +Record issuance, acceptance, expiry/requeue, duplicate upload, and stale-result rejection events using existing logs/records. Never label a worker online/offline merely from silence: there is no heartbeat. Do not introduce a telemetry database or monitoring stack. Raw findings belong in the existing protected result storage, not transport/auth/debug logs. + +### 8. Empty and isolated local validation first + +Keep production checkout, processes, Docker containers, and volumes untouched. Reuse the existing stdlib checks, reviewed unit harness, and synthetic E2E driver. Before any runtime test, make test project names, image tags, volumes, ports, and config independent of inherited deployment defaults. The copied `compose.yaml`/legacy import overrides are not safe test launch instructions. + +Initialize PostgreSQL empty in new test-owned storage; never restore a production dump or mount production PGDATA. Scrub inherited DSNs, provider secrets, runtime overrides, and proxies before application imports. Disable live discovery/keycheck autostart; explicitly feed synthetic tasks and mock detailed provider transports. No production credentials, paid calls, discovered public targets, or uncontrolled TruffleHog verification. + +Exercise existing local and new remote scan execution on the same fixtures and compare normalized bundle/DB results while ignoring only transport identity/timing. Include Git/Docker immutable-plan and Postman-candidate parity, both custom detector directions, and all current source error dispositions. Test Windows and Linux clients, N>1 and multi-device caps, fixed-clock expiry/reissue races, lost claim/result replies, restarts, malformed/conflicting bundles, and auth isolation. Use controlled clocks instead of waiting a real day. + +Use an isolated local TLS/internal network for Caddy/API tests and deny external egress. Validate admin bans through the actual edge, including a worker sharing the banned admin IP. Package a pinned Windows portable client and Linux client/container using current tool dependencies; no installer service or updater is necessary. Do not claim cross-platform readiness from mocked scans alone. + +## Risks / Trade-offs + +- [Long assignment lifetime] Slow recovery and duplicate physical work are deliberate trade-offs; fencing and idempotent acceptance protect authoritative state. +- [DB-bound execution seams] Git/Docker plans and process authority require adaptation; parity tests are a release gate, not permission to rewrite provider logic. +- [Trusted client access] Clients can see task credentials and findings; issue least-needed credentials, redact logs, revoke device tokens, and use HTTPS. +- [Shared NAT and aggressive admin bans] Two mistakes can lock out an operator; scope bans to admin and retain tested SSH recovery. +- [Existing active OpenSpec deltas] Avoid mixing unrelated changes or reviving superseded local-only assumptions; archive/sync is a separate request. +- [Unsafe inherited launch defaults] Static checks can run now; runtime/E2E execution waits for explicit test isolation and image/config provenance checks. + +## Migration Plan + +1. Prepare the separate source-only Git workspace and complete this plan; do not start runtime services. +2. Establish neutral isolated tests, then implement the narrow execution/admission/transport boundary against empty PostgreSQL with synthetic fixtures. +3. Run local parity, failure/recovery, cross-platform, and edge-auth gates before enabling any real remote claims. +4. Keep remote admission disabled by default until explicitly configured. Preserve the local execution path using shared scanner logic; do not require a remote worker for existing local operation. +5. Production deployment and any additive identity/reservation schema migration need a separate reviewed rollout. No production data migration or rebuild is performed during this planning task. +6. To roll back a later rollout, stop new remote claims, retain/reconcile accepted bundles, and drain or expire outstanding assignments through current recovery before returning to local-only execution. Do not blindly drop reservation metadata or switch binaries underneath unfinished remote work. + +## Open Questions + +No blocking product decisions remain. Implementation must verify the smallest DB-free Git/Docker execution seam, the exact existing reservation fields to extend, and the host-specific fail2ban/Caddy reload mechanism through tests before release. Deployment-specific hostname, generated admin prefix/password, device tokens, and quota values are supplied during provisioning, not embedded in this plan. diff --git a/openspec/changes/add-minimal-remote-scan-workers/proposal.md b/openspec/changes/add-minimal-remote-scan-workers/proposal.md new file mode 100644 index 0000000..fc7f1ca --- /dev/null +++ b/openspec/changes/add-minimal-remote-scan-workers/proposal.md @@ -0,0 +1,29 @@ +## Why + +Truf needs to use trusted Windows and Linux machines for download/scan work without rewriting its already-debugged parser, queue, error policy, ingestion, or detailed keychecks. A separate source-only workspace and an empty, isolated local test environment let us develop this boundary without copying the large production database or touching the active runtime. + +## What Changes + +- Add a thin authenticated HTTPS adapter around existing target reservation and canonical `.trb` result handoff, not a second task system. +- Keep PostgreSQL, discovery, scheduling, ingestion, projection, and detailed keycheck in one server Docker runtime; use a separate Caddy edge. +- Run existing download/scan logic on trusted clients. The server supplies the task, immutable plan, and required source/scanner settings; the client config contains server URL, device token, and desired execution slots. +- Claim one task per free client slot, subject to an atomic server-side per-user cap across devices and existing admission/backpressure rules. +- Use a configurable fixed assignment lifetime, default 24 hours, with periodic expiry recovery and no worker heartbeat. Preserve existing retry/error decisions, allow the same worker to reclaim work, fence stale assignments, and acknowledge result retries idempotently. +- Expose only the authenticated Worker API and a long-random-path authenticated admin area. Keep the standalone dashboard and backend/control/database ports private; ban admin IPs for 24 hours after two actual failed login attempts within ten minutes without banning Worker API traffic. +- Reuse existing records/logs for worker counts, durations, assignment outcomes, and last API contact. +- Validate with empty local storage and synthetic fixtures, including crash/retry/expiry scenarios, without production data or real provider requests. + +## Capabilities + +### New Capabilities + +- `distributed-scan-workers`: Minimal remote scan execution, centralized settings, admission, fixed expiry, durable result handoff, observability, and isolated local verification. +- `restricted-public-access`: Private backend/dashboard topology, authenticated worker/admin access, admin-only login bans, and safe diagnostics. + +### Modified Capabilities + +None. `openspec/specs/` is empty in this snapshot. Existing unarchived deltas are design references, not canonical specifications to modify or archive as part of this work. + +## Impact + +The change touches the execution boundary in `app/console_runner.py` and `app/scanner.py`, reservation/recovery in `app/scanner_db.py`, the existing bundle/ingester pipeline, runtime lifecycle wiring, packaging, Docker/Caddy deployment, and regression tests. New persistent state is limited to necessary worker identity/token/quota bindings and metadata attached to existing reservations; PostgreSQL remains the only server queue authority. Client detailed keychecks, another broker, a parallel result format, worker heartbeats, automatic updates, and production migration are out of scope. diff --git a/openspec/changes/add-minimal-remote-scan-workers/specs/distributed-scan-workers/spec.md b/openspec/changes/add-minimal-remote-scan-workers/specs/distributed-scan-workers/spec.md new file mode 100644 index 0000000..67948c3 --- /dev/null +++ b/openspec/changes/add-minimal-remote-scan-workers/specs/distributed-scan-workers/spec.md @@ -0,0 +1,145 @@ +## ADDED Requirements + +### Requirement: Existing pipeline semantics remain authoritative +The system SHALL execute existing download/scan logic on trusted Windows/Linux clients and SHALL retain discovery, queue/reservation authority, ingestion, projection, candidate routing, and detailed keycheck on the server. It MUST NOT create a second queue, reduced result format, client detailed-keycheck flow, or new scanner retry/dead-letter policy. + +#### Scenario: Remote scan matches current local behavior +- **WHEN** local and remote execution process the same synthetic target, immutable plan, and effective scanner settings +- **THEN** normalized findings, detector/provider identities, origin/context, errors, candidate evidence, coverage decisions, and queue dispositions match apart from transport identities and timing +- **AND** detailed keychecks run through the existing server candidate pipeline after ingestion, independently of JSONL projection completion + +### Requirement: Centralized task settings and bounded client authority +The server SHALL supply the assigned target, immutable plan, effective source/scanner configuration, compatibility identity, and only task-required credentials. Client-authored operational settings SHALL be server URL, device token, and desired slot count. Clients MUST NOT require PostgreSQL, server supervisor authority, independent provider configuration, or arbitrary remote-command execution. + +#### Scenario: A client has no provider configuration +- **WHEN** an authorized compatible client with only its bootstrap settings claims work +- **THEN** it receives enough task-specific input to use the existing scanner without database access or worker-maintained provider settings +- **AND** it receives no database/admin secrets or unrelated discovery credential pool + +#### Scenario: Incompatible client cannot silently change scan policy +- **WHEN** a client's scanner build, detector policy, or source/OS support is incompatible +- **THEN** admission refuses incompatible work without consuming a target or falling back to different scan settings + +### Requirement: Per-slot claims respect atomic shared quotas +Each free client work slot SHALL claim at most one task. The server SHALL atomically enforce the owner's active-assignment cap across devices together with existing admission/capacity restrictions. Client slots SHALL bound pending unacknowledged work, and quota accounting MUST NOT be confused with node-local scan permits or server spool credits. + +#### Scenario: Multiple devices race for the last slots +- **WHEN** a user capped at three active assignments runs two clients configured for eight slots each and they claim concurrently +- **THEN** at most three assignments are admitted across both clients and no unused batch/backlog is handed out + +#### Scenario: An administrator lowers a running user's cap +- **WHEN** the new cap is below the user's current active-assignment count +- **THEN** new claims wait until usage permits admission without inventing cancellation or error outcomes for current work + +### Requirement: Ambiguous claim delivery is recoverable +The client SHALL persist a stable admission/request identity before sending a claim, and retries SHALL reconcile the same existing reservation scoped to the authenticated device rather than allocating another target. + +#### Scenario: A claim commits but its reply is lost +- **WHEN** a client retries the original claim identity after a network failure or restart +- **THEN** the server returns that assignment or its authoritative terminal state without creating an additional reservation or consuming extra quota + +### Requirement: Assignment expiry is fixed and server-owned +Remote assignments SHALL expire after a configurable fixed interval, default 24 hours from server-side issuance, including upload. API contact SHALL NOT renew this deadline. No worker heartbeat or liveness probe SHALL be required. A periodic server recovery pass SHALL process expired unfinished assignments using existing infrastructure-loss recovery/accounting, preserving existing scan retry policy. + +#### Scenario: Worker disappears without reporting a scan outcome +- **WHEN** its assignment deadline passes and the periodic recovery pass runs +- **THEN** the unfinished target becomes claimable again with correctly reconciled credits/quota and dependent plan leases +- **AND** the loss does not become a fabricated scanner/provider error, arbitrary retry limit, or worker blacklist + +#### Scenario: The same worker returns after expiry +- **WHEN** the previous worker requests available work after its expired assignment is recovered +- **THEN** it is eligible to claim that target again under a new issuance identity + +#### Scenario: Client contact does not extend work +- **WHEN** the client retries an upload or another API request before the deadline +- **THEN** the original server expiry remains unchanged + +### Requirement: Remote ownership fences stale results +Result acceptance and expiry/reissue SHALL serialize on current reservation ownership and deadline. Local producer PID checks MUST NOT reclaim remote assignments. Dependent plan/blob/target leases SHALL remain consistent with the remote ownership interval. Only the current unexpired assignment can first publish an authoritative result. + +#### Scenario: Old worker uploads after another issuance +- **WHEN** an old worker uploads a previously unaccepted result after expiry or reissue +- **THEN** the server rejects it as stale without completing the newer assignment, advancing plan coverage, or releasing its credits + +#### Scenario: Result races with periodic recovery +- **WHEN** upload acceptance and expiry recovery race for the same assignment +- **THEN** exactly one authoritative transition wins and neither duplicate ingestion nor double capacity release occurs + +#### Scenario: Server restarts while a remote client is still working +- **WHEN** runtime recovery cannot find a local producer PID for a valid remote assignment +- **THEN** it retains remote ownership until the fixed deadline rather than treating the absent local process as worker death + +### Requirement: Full canonical results have durable idempotent acceptance +Clients SHALL send the existing canonical v2 `.trb` as a bounded binary stream, preserving findings, errors, metadata, attribution, exact-plan identity, and candidate evidence. The server SHALL validate identity, format, size, paths, and hash and durably publish the bundle plus ready/recovery state before acknowledging custody. Existing transactional ingestion SHALL remain authoritative. Accepted identity/digest/receipt information SHALL survive ordinary ingestion and spool cleanup in authoritative records independently of the bundle file. + +#### Scenario: Acceptance reply is lost +- **WHEN** the server accepts a bundle but its acknowledgement is lost and the client uploads the identical bundle again +- **THEN** the server returns the original acceptance without duplicate ingestion, candidates, quota release, or counters +- **AND** this acknowledgement remains recoverable after the assignment deadline because acceptance already occurred + +#### Scenario: Accepted identity receives a different body +- **WHEN** another bundle with conflicting content is submitted for an accepted identity +- **THEN** the server rejects the conflict without replacing the accepted result + +#### Scenario: Client retries after ingestion and normal cleanup +- **WHEN** an accepted bundle's reply is lost, ingestion and ordinary cleanup finish, the server restarts, and the client retries after the original deadline +- **THEN** identical bytes recover the original receipt despite the absence of the spool file +- **AND** conflicting bytes are rejected without duplicate results, candidates, accounting, or counters + +#### Scenario: Upload is truncated or invalid +- **WHEN** an upload exceeds bounds, fails validation, disconnects, or crosses the deadline before first acceptance +- **THEN** it produces no successful acknowledgement or authoritative findings and cannot mutate another assignment +- **AND** bounded partial-file cleanup and existing transport/recovery handling apply + +#### Scenario: Server crashes around bundle publication +- **WHEN** the receiver crashes after durable publication or ready-state recording but before replying +- **THEN** restart reconciliation and a same-identity client retry recover a single consistent acceptance or authoritative rejection without adopting an unvalidated/stale file + +#### Scenario: Ingestion is delayed past the worker deadline +- **WHEN** an accepted ready bundle awaits server ingestion after its former assignment deadline +- **THEN** worker expiry recovery does not requeue it as unfinished remote work + +### Requirement: Client recovery separates transport from scan outcomes +The client SHALL persist pending bundle and assignment identity until durable acknowledgement and SHALL retry transport using that identity without consuming scanner/provider retry budgets. Existing scan errors SHALL retain their existing dispositions. Definitive stale rejection SHALL be recorded as a local stale/discard outcome with bounded cleanup, not success or infinite upload retry. + +#### Scenario: Client restarts with a pending result +- **WHEN** a client restarts before confirming server acceptance +- **THEN** it recovers the pending result and retries/reconciles it before claiming replacement work for that occupied slot + +#### Scenario: Scanner reports an existing deferred or terminal error +- **WHEN** the existing scanner returns an error disposition +- **THEN** the client/server handoff preserves that disposition and the server applies the existing queue policy rather than a transport-specific retry rule + +### Requirement: Pre-bundle terminal reports are replay-safe and fenced +Pre-bundle failure/release reports SHALL use the current issuance identity and existing outcome/accounting rules. Clients SHALL retain and retry the report identity until its authoritative outcome is acknowledged. Duplicate accepted reports SHALL return the original outcome without duplicate retry charges, counters, or quota/credit release; expired/superseded unaccepted reports SHALL NOT alter newer work. + +#### Scenario: Failure or release reply is lost +- **WHEN** the server commits a pre-bundle terminal disposition but its reply is lost and the client repeats the report +- **THEN** the original disposition is acknowledged, remote quota/credits are reconciled once, and the client slot resolves once without an extra scanner retry charge + +#### Scenario: Same worker reports failure for an old issuance +- **WHEN** a worker reclaims a target under a new issuance and a delayed unaccepted terminal report for its expired issuance arrives +- **THEN** the server rejects the stale report without changing the new issuance, its quota, or the target's current outcome + +### Requirement: Worker observability derives from authoritative events +The system SHALL expose safe per-worker unfinished/completed/failed/expired counts, last authenticated API contact, correlated target/source/reservation identities, issue/finish times, duration, and outcome/error category using existing logs and records. It SHALL distinguish result acceptance from later processing and MUST NOT infer online/offline status without a heartbeat. + +#### Scenario: Duplicate or stale result arrives +- **WHEN** a result is duplicated or rejected as stale +- **THEN** a correlated safe event is recorded without inflating completed counts or exposing credentials/raw findings + +#### Scenario: A valid worker is silent during a long scan +- **WHEN** the worker makes no API request before its assignment deadline +- **THEN** the admin view shows last contact and outstanding assignment state without declaring the worker dead solely from silence + +### Requirement: Local validation is isolated and synthetic +Implementation SHALL be developed and verified in the independent source-only workspace, with fresh test-owned storage and empty PostgreSQL initialized only for synthetic fixtures. Tests MUST NOT copy/restore/mount production data, reuse active runtime volumes/image tags, inherit production credentials/DSNs, or perform live discovery/provider verification. Windows/Linux real scanner parity and controlled-clock recovery tests SHALL precede release. + +#### Scenario: Empty local end-to-end execution +- **WHEN** a test runtime, Caddy, and clients are started for the worker scenario +- **THEN** they use isolated project/image/volume/port/config identities, synthetic targets, mocked provider transports, and restricted external egress +- **AND** existing production checkouts, containers, PGDATA, logs, and results are neither read as runtime inputs nor modified + +#### Scenario: Unsafe inherited deployment defaults are detected +- **WHEN** test setup would reuse the inherited production project, shared image tag, existing data volume, runtime bind mount, or live source configuration +- **THEN** validation refuses to start runtime services until explicit isolation is established diff --git a/openspec/changes/add-minimal-remote-scan-workers/specs/restricted-public-access/spec.md b/openspec/changes/add-minimal-remote-scan-workers/specs/restricted-public-access/spec.md new file mode 100644 index 0000000..64837df --- /dev/null +++ b/openspec/changes/add-minimal-remote-scan-workers/specs/restricted-public-access/spec.md @@ -0,0 +1,75 @@ +## ADDED Requirements + +### Requirement: Only authenticated edge routes are public +The deployment SHALL expose HTTPS through Caddy only for `/api/v1/worker/*` and a configured random administrative prefix. The standalone dashboard, PostgreSQL, supervisor control, Caddy control API, and raw backend ports SHALL remain private. Unknown application paths SHALL return 404; an administrative URL MUST NOT replace authentication. + +#### Scenario: Internet visitor requests backend or dashboard access +- **WHEN** an unauthenticated visitor requests `/dashboard`, a normal `/admin` path, an unknown route, or a backend/control/database port +- **THEN** no dashboard data or backend/control access is available + +#### Scenario: Dashboard information appears inside admin +- **WHEN** operational dashboard information is rendered in the administrative area +- **THEN** every page, asset, and data request is protected by the same admin authentication boundary without publishing the standalone backend + +### Requirement: Worker access has narrow device-scoped authority +Every Worker API operation SHALL require a valid revocable opaque device token over certificate-validated HTTPS, scoped to its user/device assignments and quotas. The server SHALL store token hashes, not reusable plaintext tokens. Worker tokens MUST NOT authorize admin actions, another device's result mutation, database access, or generic commands. + +#### Scenario: Invalid or revoked token submits work +- **WHEN** an invalid/revoked device token claims a task or uploads a result +- **THEN** the request is refused before task/result mutation and does not count toward the admin login jail + +#### Scenario: Authorized device tries to finish somebody else's task +- **WHEN** a device submits another device's reservation identity +- **THEN** the request is refused without changing that reservation or disclosing task credentials + +### Requirement: Admin routes require credentials and safe mutations +The administrative prefix SHALL contain at least 128 bits of randomness and all administrative routes SHALL require login/password authentication using a supported password hash. Typed mutations SHALL enforce appropriate Origin/CSRF protection. The interface MUST NOT expose a shell or generic supervisor-command passthrough. + +#### Scenario: Visitor knows the full administrative URL +- **WHEN** a visitor accesses the correct administrative prefix without valid credentials +- **THEN** no administrative page, data, asset, or mutation becomes available + +#### Scenario: Cross-site request attempts a queue or token mutation +- **WHEN** a browser submits an administrative mutation without valid same-origin/CSRF authorization +- **THEN** the mutation is rejected even if browser-level password authentication is present + +### Requirement: Two failed admin logins trigger an admin-only day ban +Two actual invalid administrative credential submissions from one IP within ten minutes SHALL ban that IP from administrative routes for 24 hours. The ban SHALL persist across service restart, expire automatically, and support SSH/operator removal. Initial unauthenticated authentication challenges, Worker API failures, and unrelated requests MUST NOT count. Enforcement SHALL leave Worker API traffic from the same IP unaffected. + +#### Scenario: Repeated invalid admin credentials +- **WHEN** an IP submits a second invalid admin login within the ten-minute window +- **THEN** subsequent admin requests from that IP are denied for 24 hours, including otherwise valid credentials, until expiry or operator unban + +#### Scenario: Admin and worker share one NAT address +- **WHEN** admin login failures trigger a ban for an IP also used by an authorized worker +- **THEN** the worker can still claim/upload over HTTPS and administrative requests remain blocked + +#### Scenario: Normal first visit receives an authentication challenge +- **WHEN** a browser initially requests the protected area without sending credentials +- **THEN** the normal challenge does not consume either of the two failed-login attempts + +#### Scenario: Ban expires or operator recovers access +- **WHEN** 24 hours elapse or an operator removes the ban through the documented SSH procedure +- **THEN** credential-authenticated administrative access is restored without restarting or weakening Worker API authorization + +### Requirement: IP identification and ban enforcement match the deployed edge +The initial direct-DNS deployment SHALL derive client IP from the direct connection, ignoring untrusted forwarded headers. Any future proxy SHALL require an explicit trusted-proxy configuration and verification before enabling IP bans. The fail2ban action SHALL enforce the admin route boundary at Caddy rather than indiscriminately blocking shared port 443. + +#### Scenario: Attacker supplies another address in a forwarded header +- **WHEN** a direct client sends arbitrary `X-Forwarded-For` or equivalent headers with failed admin credentials +- **THEN** another user's address is not selected for banning and the caller cannot evade its own ban + +#### Scenario: Deployed ban is verified through Docker ingress +- **WHEN** the admin jail applies its ban against the actual Caddy/Docker topology +- **THEN** external administrative requests are denied and worker requests remain available, not merely a host firewall rule being present + +### Requirement: Diagnostics and transport protect sensitive data +All worker/admin transport SHALL use HTTPS without disabling certificate verification. Tokens, passwords, secret admin prefixes in routine access logs, task credentials, and raw findings SHALL be redacted or omitted from diagnostics. Admin responses SHALL disable caching and referrer disclosure and use a compatible restrictive CSP; production HTTPS SHALL use HSTS. Client disk/workspace encryption SHALL NOT be required. + +#### Scenario: Auth or upload error is logged +- **WHEN** authentication, payload validation, or upload handling fails +- **THEN** logs contain only safe correlation/status/error metadata and enough redacted authentication outcome for the admin jail, not request credentials or raw result content + +#### Scenario: Trusted client stores pending work locally +- **WHEN** a worker downloads or persists a bundle before acknowledgement +- **THEN** ordinary local storage is supported without an application encryption layer while HTTPS and post-acknowledgement cleanup remain enforced diff --git a/openspec/changes/add-minimal-remote-scan-workers/tasks.md b/openspec/changes/add-minimal-remote-scan-workers/tasks.md new file mode 100644 index 0000000..8f4d9ae --- /dev/null +++ b/openspec/changes/add-minimal-remote-scan-workers/tasks.md @@ -0,0 +1,49 @@ +## 1. Establish Safe Empty Test Isolation + +- [x] 1.1 Replace unsafe inherited test launch defaults with explicit test-owned project/image/volume/port identities and refusal checks for production binds/volumes; do not run the copied default Compose or import overrides. +- [x] 1.2 Reuse existing test harnesses to initialize empty PostgreSQL, scrub inherited secrets/DSNs/proxies before imports, disable live source/keycheck autostart, and provide synthetic targets/provider transports with external egress blocked. +- [x] 1.3 Establish baseline synthetic scan/bundle/queue/keycheck fixtures and run the reviewed isolated unit selection before changing execution semantics; record relevant pre-existing failures separately. + +## 2. Extract The Existing Scan Execution Boundary + +- [x] 2.1 Trace `stage_claim`, `scan_target_result`, process authority, bundle candidate extraction, and Git/Docker exact-plan inputs; define the smallest DB-free job input using existing types/identities rather than a parallel task model. +- [x] 2.2 Adapt one planned download/scan/bundle path to run on Windows/Linux without PostgreSQL or supervisor credentials while preserving `OwnedProcess`, native slot handling, source errors/dispositions, and structured Postman candidates. +- [x] 2.3 Add centralized effective-config/plan delivery and protocol/scanner/detector-policy compatibility validation, supplying only task-needed credentials and forbidding arbitrary commands or client provider overrides. +- [x] 2.4 Compare local and DB-free execution on existing source fixtures, including Git/Docker coverage, Postman evidence, and Xai/ZAI custom-detector direction regressions; preserve detailed keycheck exclusively on the server. + +## 3. Extend Existing Admission And Recovery + +- [x] 3.1 Add only necessary user/device token-hash/quota bindings and remote metadata on existing reservations, with revocation and no second queue or independent remote-job state machine. +- [x] 3.2 Implement authenticated one-task claims through existing admission with atomic per-user caps across devices, existing capacity limits, stable request identity, ambiguous-claim reconciliation, and bounded empty/capacity polling. +- [x] 3.3 Apply configurable server-clock fixed expiry (default 24 hours) to remote ownership and dependent target/plan/blob leases, separating it from local PID recovery and scanner timeouts without heartbeat or renewal endpoints. +- [x] 3.4 Wire periodic expired-assignment recovery into runtime maintenance using existing refund/requeue accounting; permit same-worker reclaim, avoid new error/retry/blacklist policies, and release quota exactly once. +- [x] 3.5 Test concurrent quota admission, lower-cap behavior, lost claim replies, restart recovery, fixed-clock expiry, dependent plan leases, and stale ownership fencing against isolated PostgreSQL. + +## 4. Add Durable Canonical Bundle Transport + +- [x] 4.1 Receive `.trb` binary streams into bounded server-owned partial files, enforcing existing capacity/size limits, upload timeouts, ownership, expiry, codec/path/identity validation, and content hash. +- [x] 4.2 Reuse durable atomic publication and `mark_result_bundle_ready` before acknowledgement; reconcile upload/recovery races and reject stale/conflicting bodies without affecting current plan coverage or credits. +- [x] 4.3 Preserve accepted identity/digest/receipt in existing records independently of spool cleanup and return the same receipt after ingestion, cleanup, restart, and expiry; reuse existing ingester/projector/candidate/keycheck behavior and exclude ready bundles from worker expiry. +- [x] 4.4 Preserve current scan-failure dispositions and pre-bundle infrastructure release paths with issuance-fenced idempotent terminal-report replay; test lost/repeated/stale reports, single slot/quota/credit resolution, interrupted/invalid/conflicting uploads, and publication/ready/commit crashes. + +## 5. Build The Minimal Client Loop + +- [x] 5.1 Implement server-URL/token/N bootstrap and one claim per free slot, with node-local execution containment and no independent source/provider configuration, client keycheck, heartbeat, batching, or updater. +- [x] 5.2 Persist assignment identity and pending bundles, recover them on restart, retry transport without scan retry charges, free work slots only after authoritative resolution, and handle definitive stale rejection with explicit bounded cleanup. +- [x] 5.3 Package pinned scanner/config assets for a Windows portable client and Linux client/container; verify certificate-validating HTTPS and avoid secret/raw-finding logs without requiring local encryption. +- [x] 5.4 Exercise real synthetic scans and complete bundle round trips on both Windows and Linux with N greater than one, restart during pending upload, server outage, and subsequent recovery; do not substitute mocks for cross-platform scanner verification. + +## 6. Restrict Edge Access And Add Minimal Administration + +- [x] 6.1 Wire API/admin into the single runtime and separate Caddy edge, publishing only authenticated worker routes and a random admin prefix; keep dashboard/backend/database/control ports private and remote admission disabled until configured. +- [x] 6.2 Add device-scoped token authorization and hash-based admin password authentication with protected assets, typed CSRF/Origin-checked mutations, no-store/same-origin-referrer/CSP/production-HSTS headers, and redacted logs. +- [x] 6.3 Add a host fail2ban admin jail for two actual bad logins in ten minutes and a 24-hour persisted ban; implement validated admin-only Caddy denylist updates, direct-IP handling, automatic expiry, and documented SSH unban without blocking worker traffic. +- [x] 6.4 Build only necessary admin user/device-token/quota and existing queue controls plus read-only worker counts, durations, outcomes, and last-contact summaries from existing records; do not add online/offline guesses or a telemetry store. +- [x] 6.5 Verify unknown/private routes, revoked/wrong-device tokens, cross-site mutations, absent-credentials challenges, spoofed forwarded headers, actual two-failure bans/restart/unban, and an authorized worker sharing the banned admin IP through the isolated edge. + +## 7. Complete Regression Gates And Handoff + +- [x] 7.1 Run the synthetic empty-database end-to-end flow through claim, real scan, complete bundle, ingestion, projection, and mocked detailed server keycheck; compare normalized output/dispositions with the existing local path. +- [x] 7.2 Run combined disconnect/restart/expiry/reissue/upload races with controlled clocks; verify single authoritative acceptance, intact plan coverage, no credit leaks, and no duplicate worker statistics. +- [x] 7.3 Verify test manifests, build context, logs, artifacts, and cleanup exclude production state/credentials and that active production containers/volumes were untouched; retain only test-owned failure evidence. +- [x] 7.4 Document isolated run commands, client bootstrap, quota/deadline tuning, token revocation, admin unban, accepted-custody semantics, known 24-hour recovery trade-offs, and later rollout/drain rollback; validate OpenSpec and leave unrelated changes unarchived. diff --git a/openspec/changes/add-postman-source/.openspec.yaml b/openspec/changes/add-postman-source/.openspec.yaml new file mode 100644 index 0000000..db47328 --- /dev/null +++ b/openspec/changes/add-postman-source/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-06-02 diff --git a/openspec/changes/add-postman-source/design.md b/openspec/changes/add-postman-source/design.md new file mode 100644 index 0000000..b0f210e --- /dev/null +++ b/openspec/changes/add-postman-source/design.md @@ -0,0 +1,80 @@ +## Context + +The scanner is already organized around independent sources managed by `console_runner.py` and `supervisor.py`. Each source discovers targets, writes them to per-source queue files, scans targets through TruffleHog, records results in JSONL and SQLite, and can stop pagination when consecutive pages contain only known targets. Existing package sources already download and extract npm/PyPI artifacts before running TruffleHog filesystem scans. + +Postman artifacts fit this model as filesystem scan targets, but they need separate discovery and enrichment. Public Postman web search is comparatively fragile, while GitHub code search exposes many real `*.postman_collection.json` and `*.postman_environment.json` files. npm and PyPI can also expose Postman artifacts during the package extraction windows that already exist. + +## Goals / Non-Goals + +**Goals:** + +- Add a first-class `postman` source with standard queue, checked, supervisor, dashboard, and database behavior. +- Discover Postman collection/environment JSON from GitHub code search using the existing GitHub auth pool. +- Support initial backfill over up to the GitHub Search API result cap per query and ongoing tail scans over recently indexed pages. +- Filter discovered GitHub code artifacts by last file commit age so stale files can be skipped during backfill. +- Rotate across multiple GitHub tokens and pause only when all usable tokens are rate-limited. +- Cache discovered Postman artifacts durably so package-derived artifacts survive temp directory cleanup. +- Harvest Postman artifacts from npm and PyPI extraction flows without disrupting existing package scans. +- Enrich findings with Postman-specific context derived from auth configuration, headers, query params, request bodies, environment variables, and endpoint hosts. + +**Non-Goals:** + +- Scraping Postman's own web application in the first implementation. +- Replacing TruffleHog detectors with custom regex-only detection. +- Adding new keychecker services as part of this change. +- Scanning private Postman workspaces through the Postman API. + +## Decisions + +1. Use GitHub code search as the primary discovery channel. + +GitHub code search has authenticated API support, predictable pagination, and high-quality results for `filename:postman_collection.json ` and `filename:postman_environment.json `. Public Postman web pages and `postman.com` links are lower-yield and more likely to change without notice. Direct Postman URL discovery can be added later as an additional provider without changing the scanner contract. + +2. Treat Postman artifacts as durable filesystem scan targets. + +The source will download or copy each discovered artifact into a runtime Postman cache and scan a temporary directory containing the cached JSON. This reuses the existing TruffleHog filesystem path and avoids keeping npm/PyPI extraction directories alive. + +3. Identify GitHub-discovered targets by `repo:path:sha`. + +The same file at the same SHA must not be rescanned, while a new SHA for the same path must be queued again. This matches the existing queue/checked model and makes `stop_on_seen_pages` useful for tail scans. + +4. Identify package-harvested targets by content hash. + +npm/PyPI packages can contain duplicate Postman artifacts across versions or package names. A SHA-256 content hash provides stable dedupe and allows different origins to point to the same cached artifact without rescanning identical content. + +5. Filter GitHub code artifacts by path commit age after discovery. + +The code search response does not include reliable file modification dates. The implementation will query the latest commit for each `repo:path` and skip artifacts older than `max_file_age_days`. This costs extra core API requests but keeps the code search query simple and reliable. + +6. Use a source-local GitHub token pool for discovery. + +Postman discovery may make many GitHub requests in one cycle. A per-request token pool can rotate across all configured GitHub auth entries, cool down only the token that failed, and sleep when no token remains available. This is more efficient than one token per source cycle. + +7. Keep Postman enrichment separate from detection. + +TruffleHog remains responsible for finding candidate secrets. Postman enrichment will add context and confidence by correlating findings with request auth, headers, variables, endpoints, and placeholder detection. This avoids increasing false positives from regex-only scans. + +## Risks / Trade-offs + +- GitHub code search is rate-limited to roughly 10 requests per minute per token -> throttle to a configurable safe RPM and rotate across the auth pool. +- Commit-age filtering adds extra core API requests -> make `max_file_age_days` configurable and cache commit metadata per `repo:path:sha` within a cycle. +- Broad queries such as `ai` can hit the 1000-result search cap and include noisy results -> use a Postman-specific query list and allow per-source query tuning. +- Package harvesting adds small overhead during npm/PyPI scans -> limit file walking to reasonable extensions, max file size, and known Postman filename patterns. +- Postman variables often contain placeholders rather than live secrets -> classify placeholders separately and keep TruffleHog verification/keycheckers as the authority for live/dead status. +- All tokens may become unavailable -> sleep until the earliest known reset time, or a configured fallback such as 30 minutes when no reset is known. + +## Migration Plan + +1. Add the `postman` source disabled by default in `config.yaml`. +2. Add queue/dashboard/database support for `postman` without changing existing source behavior. +3. Run a small verification cycle with `pages: 1`, `per_page: 10`, and `max_targets` set. +4. Run the one-time backfill with `pages: 10`, `per_page: 100`, `stop_on_seen_pages: false`, and `max_file_age_days: 365`. +5. Switch the source to tail mode with `pages: 1-3` and `stop_on_seen_pages: true`. +6. Enable npm/PyPI harvesting after the base Postman source is verified. + +Rollback is to disable `sources.postman.enabled`, leave its queues/cache intact, and continue running existing sources unchanged. + +## Open Questions + +- Whether the default backfill query list should include broad terms like `ai`, or keep only higher-intent terms such as `openai`, `anthropic`, `gemini`, `llm`, `rag`, and `agent`. +- Whether package-harvested artifacts should always be enqueued, or only when they include auth/secret-related markers. diff --git a/openspec/changes/add-postman-source/proposal.md b/openspec/changes/add-postman-source/proposal.md new file mode 100644 index 0000000..96ef947 --- /dev/null +++ b/openspec/changes/add-postman-source/proposal.md @@ -0,0 +1,32 @@ +## Why + +Public Postman collections and environments are a high-signal source for leaked API credentials because they often preserve request auth settings, headers, variables, and example payloads close to real API usage. The existing scanner already supports multi-source discovery, queues, TruffleHog filesystem scans, token rotation, and observability, so adding Postman can reuse the current architecture while expanding coverage beyond repositories, packages, containers, and HuggingFace Spaces. + +## What Changes + +- Add a new `postman` source that discovers, queues, scans, and records Postman collection/environment artifacts. +- Seed Postman targets from GitHub code search using public `*.postman_collection.json` and `*.postman_environment.json` files. +- Support a one-time backfill mode that scans up to the GitHub Search API result limit per query while filtering out artifacts older than a configured age window. +- Support a daily tail mode that fetches recently indexed pages and stops early when all targets on consecutive pages are already known. +- Use the configured GitHub auth pool for Postman discovery, rotating across tokens and sleeping when all tokens are rate-limited. +- Add durable Postman artifact caching so targets discovered from GitHub, npm, and PyPI can be scanned after temporary extraction directories are removed. +- Harvest Postman artifacts from npm and PyPI packages during existing package extraction flows and enqueue them into the shared Postman queue. +- Add Postman-aware result enrichment that classifies credentials using TruffleHog findings plus Postman auth/header/query/body/environment context. + +## Capabilities + +### New Capabilities + +- `postman-source`: Discovery, queueing, scanning, caching, and enrichment for Postman collection and environment artifacts. + +### Modified Capabilities + +- None. + +## Impact + +- Affected scanner paths: `app/scanner.py`, `app/console_runner.py`, `app/scanner_db.py`, `app/dashboard.py`, and `app/config.yaml`. +- Adds runtime files under `runtime/queues/` for `todo_postman.txt` and `checked_postman.txt`. +- Adds durable artifact storage under a runtime Postman cache directory. +- Uses existing GitHub auth pools from `secrets.yaml`; no new secret format is required for GitHub discovery. +- Uses existing TruffleHog filesystem scanning and keychecker follow-up flows; no breaking changes to current sources are expected. diff --git a/openspec/changes/add-postman-source/specs/postman-source/spec.md b/openspec/changes/add-postman-source/specs/postman-source/spec.md new file mode 100644 index 0000000..fb6ae22 --- /dev/null +++ b/openspec/changes/add-postman-source/specs/postman-source/spec.md @@ -0,0 +1,158 @@ +## ADDED Requirements + +### Requirement: Postman source registration +The system SHALL provide a first-class `postman` source that can be configured, selected, supervised, queued, scanned, and displayed consistently with existing scanner sources. + +#### Scenario: Configured Postman source is selectable +- **WHEN** a config contains `sources.postman.enabled: true` and the runner is invoked with `--source postman` +- **THEN** the runner SHALL execute only the Postman source cycle using Postman source settings + +#### Scenario: Postman queue files are used +- **WHEN** the Postman source prepares targets +- **THEN** the system SHALL use `todo_postman.txt` and `checked_postman.txt` under the configured queue directory + +### Requirement: GitHub code search discovery +The Postman source SHALL discover public Postman artifacts from GitHub code search using configured queries and artifact kinds. + +#### Scenario: Collection files are discovered +- **WHEN** `search_kinds` includes `collection` and the query is `openai` +- **THEN** discovery SHALL search GitHub code for `filename:postman_collection.json openai` + +#### Scenario: Environment files are discovered +- **WHEN** `search_kinds` includes `environment` and the query is `openai` +- **THEN** discovery SHALL search GitHub code for `filename:postman_environment.json openai` + +#### Scenario: Search pagination is bounded +- **WHEN** `pages` is `10` and `per_page` is `100` +- **THEN** discovery SHALL request no more than 1000 search results per query and kind + +### Requirement: GitHub auth pool rotation +The Postman GitHub discovery flow SHALL use all available tokens from the configured GitHub auth pool before sleeping for rate limits. + +#### Scenario: Token rotates per GitHub request +- **WHEN** multiple GitHub auth entries are available +- **THEN** GitHub code search, commit lookup, and content download requests SHALL rotate across available tokens + +#### Scenario: One token is rate-limited +- **WHEN** a GitHub request returns a primary or secondary rate limit for the current token +- **THEN** the system SHALL mark only that token unavailable until its reset time or configured cooldown and continue with another available token + +#### Scenario: All tokens are unavailable +- **WHEN** every configured GitHub token is rate-limited or temporarily unavailable +- **THEN** the source SHALL sleep until the earliest known reset time, or for the configured fallback cooldown when no reset time is known + +#### Scenario: Token is invalid +- **WHEN** a GitHub request returns an authentication-invalid response for a token +- **THEN** the system SHALL exclude that token from the current cycle and report the authentication failure without marking other tokens invalid + +### Requirement: Backfill freshness filtering +The Postman source SHALL support a backfill mode that can scan deep code search pages while skipping GitHub artifacts older than a configured file age. + +#### Scenario: Recent file is queued +- **WHEN** a GitHub code search result has a latest path commit within `max_file_age_days` +- **THEN** the target SHALL be eligible for queueing + +#### Scenario: Old file is skipped +- **WHEN** a GitHub code search result has a latest path commit older than `max_file_age_days` +- **THEN** the target SHALL not be queued and SHALL be counted as skipped by freshness filtering + +#### Scenario: Freshness filtering is disabled +- **WHEN** `max_file_age_days` is `0` +- **THEN** the source SHALL not perform path commit age filtering + +### Requirement: Tail mode early stop +The Postman source SHALL support ongoing tail scans that stop pagination after consecutive known pages. + +#### Scenario: Known page increments stop counter +- **WHEN** `stop_on_seen_pages` is enabled and every normalized target on a fetched page already exists in `todo_postman.txt` or `checked_postman.txt` +- **THEN** the source SHALL count that page as known + +#### Scenario: Tail pagination stops +- **WHEN** the known page count reaches `seen_page_threshold` after `min_pages_before_stop` +- **THEN** the source SHALL stop fetching additional pages for that query and artifact kind + +### Requirement: Postman target identity and deduplication +The system SHALL normalize Postman targets so identical artifacts are not rescanned while changed artifacts are scanned again. + +#### Scenario: GitHub target identity includes SHA +- **WHEN** a Postman target is discovered from GitHub code search +- **THEN** its normalized target SHALL include source, repository, path, and file SHA + +#### Scenario: GitHub file changes +- **WHEN** the same GitHub repository and path is discovered with a new SHA +- **THEN** the system SHALL treat it as a new Postman target + +#### Scenario: Package target identity uses content hash +- **WHEN** a Postman artifact is harvested from npm or PyPI +- **THEN** its normalized target SHALL include the artifact content SHA-256 hash + +### Requirement: Durable Postman artifact cache +The system SHALL store discovered Postman artifact content in durable runtime cache before scanning. + +#### Scenario: GitHub content is cached +- **WHEN** a GitHub code search target is queued for scanning +- **THEN** the system SHALL download the artifact content and store it under the configured Postman cache directory + +#### Scenario: Package content is cached before cleanup +- **WHEN** npm or PyPI extraction finds a Postman artifact +- **THEN** the system SHALL copy the artifact into the durable Postman cache before the extraction directory is removed + +#### Scenario: Cache size is constrained +- **WHEN** an artifact exceeds the configured maximum Postman artifact size +- **THEN** the system SHALL skip the artifact and record a bounded error or skip reason + +### Requirement: Postman artifact scanning +The Postman source SHALL scan cached Postman collection and environment artifacts with TruffleHog filesystem scanning. + +#### Scenario: Cached artifact is scanned +- **WHEN** a Postman target points to a cached JSON artifact +- **THEN** the scanner SHALL run TruffleHog against a temporary filesystem directory containing that artifact + +#### Scenario: Findings are persisted +- **WHEN** TruffleHog reports findings for a Postman target +- **THEN** the system SHALL persist findings to existing JSONL outputs and scanner database tables with source `postman` + +#### Scenario: Scan finishes +- **WHEN** a Postman target scan completes with findings, errors, skipped status, or clean status +- **THEN** the target SHALL be moved from `todo_postman.txt` to `checked_postman.txt` + +### Requirement: npm and PyPI Postman harvesting +The npm and PyPI source flows SHALL harvest Postman artifacts discovered during existing package extraction and enqueue them for the Postman source. + +#### Scenario: npm package contains collection +- **WHEN** an extracted npm package contains a file matching `*.postman_collection.json` +- **THEN** the system SHALL cache the file and enqueue a Postman target with npm package origin metadata + +#### Scenario: PyPI package contains environment +- **WHEN** an extracted PyPI artifact contains a file matching `*.postman_environment.json` +- **THEN** the system SHALL cache the file and enqueue a Postman target with PyPI package origin metadata + +#### Scenario: Existing package scan continues +- **WHEN** Postman harvesting fails for one package artifact +- **THEN** the original npm or PyPI scan SHALL still complete and record the harvesting failure without failing unrelated package scanning + +### Requirement: Postman-aware enrichment +The system SHALL enrich Postman findings with contextual classification derived from Postman structure without replacing TruffleHog detection. + +#### Scenario: Header credential is classified +- **WHEN** a finding appears in a Postman request header such as `Authorization` or `x-api-key` +- **THEN** enrichment SHALL record the context location and infer credential kind from header type, value shape, and endpoint host when possible + +#### Scenario: Environment variable is classified +- **WHEN** a finding appears in a Postman environment variable value +- **THEN** enrichment SHALL record the variable name and classify provider or credential kind when supported by value shape or associated request endpoints + +#### Scenario: Placeholder is detected +- **WHEN** a Postman value is a placeholder such as `{{API_KEY}}`, ``, `YOUR_API_KEY`, `example`, or `changeme` +- **THEN** enrichment SHALL classify it as placeholder or low confidence rather than a live secret + +### Requirement: Observability for Postman source +The system SHALL expose Postman source activity through existing logs, queue counts, source cycle metrics, target scan records, findings, errors, and dashboard views. + +#### Scenario: Source cycle is recorded +- **WHEN** a Postman source cycle runs +- **THEN** the scanner database SHALL record source cycle metrics including fetched, queued, scanned, found, error, skipped, and queue counts + +#### Scenario: Dashboard shows Postman queues +- **WHEN** Postman queue files exist +- **THEN** the dashboard SHALL include Postman queue counts in the current queues view diff --git a/openspec/changes/add-postman-source/tasks.md b/openspec/changes/add-postman-source/tasks.md new file mode 100644 index 0000000..0a0b5bf --- /dev/null +++ b/openspec/changes/add-postman-source/tasks.md @@ -0,0 +1,69 @@ +## 1. Source Wiring + +- [x] 1.1 Add `postman` to CLI `--source` and `--platform` choices and source-to-platform mapping. +- [x] 1.2 Add `postman` queue file support through the existing `queue_files_for_args`, prepare, and mark-checked flow. +- [x] 1.3 Add `postman` to supervisor source configuration and dashboard source lists. +- [x] 1.4 Add disabled-by-default `sources.postman` configuration with GitHub auth pool, search kinds, backfill/tail controls, cache path, and rate-limit settings. + +## 2. Target Model And Cache + +- [x] 2.1 Define Postman target JSON formats for GitHub code search, npm package, PyPI package, local cache, and future URL targets. +- [x] 2.2 Implement Postman target parsing and normalization in `console_runner.py` and `scanner_db.py`. +- [x] 2.3 Implement durable Postman cache path resolution under the configured runtime directory. +- [x] 2.4 Implement safe cache writes with SHA-256 content hashing, max artifact size checks, and origin metadata preservation. + +## 3. GitHub Code Search Discovery + +- [x] 3.1 Implement GitHub code search queries for `filename:postman_collection.json ` and `filename:postman_environment.json ` based on `search_kinds`. +- [x] 3.2 Implement bounded pagination using configured `pages` and `per_page`, respecting the GitHub 1000-result search cap. +- [x] 3.3 Convert GitHub code search items into Postman target JSON containing repository, path, SHA, kind, API URL, and HTML URL. +- [x] 3.4 Implement latest path commit lookup for `max_file_age_days` filtering. +- [x] 3.5 Integrate existing known-page early stop behavior for Postman tail scans. + +## 4. GitHub Token Pool And Rate Limits + +- [x] 4.1 Build a source-local GitHub token pool from configured `auth_pool` entries and fallback token settings. +- [x] 4.2 Rotate tokens per GitHub code search, commit lookup, and content download request. +- [x] 4.3 Mark only the failing token unavailable on primary rate limit, secondary rate limit, auth invalid, or auth forbidden responses. +- [x] 4.4 Sleep until earliest known reset time, or configured fallback cooldown, when all GitHub tokens are unavailable. +- [x] 4.5 Record token cooldown status in the source runtime state without exposing token values in logs or database snapshots. + +## 5. Postman Artifact Scanning + +- [x] 5.1 Implement Postman content download from GitHub Contents API and cache it before scanning. +- [x] 5.2 Implement `scan_postman_target()` to stage cached JSON in a temporary directory and run TruffleHog filesystem scanning. +- [x] 5.3 Add Postman branch to `scan_targets_batch()` and pass timeout, detectors, excluded detectors, and verification flags. +- [x] 5.4 Preserve nearby file context and apply existing noisy finding filters to Postman scan results. +- [x] 5.5 Ensure Postman findings, errors, skipped reasons, and clean scans are persisted through existing JSONL and scanner database writes. + +## 6. npm And PyPI Harvesting + +- [x] 6.1 Add a Postman artifact finder for extracted package directories that matches collection and environment filename patterns. +- [x] 6.2 Cache npm package Postman artifacts before package temp directory cleanup and attach npm origin metadata. +- [x] 6.3 Cache PyPI package Postman artifacts before package temp directory cleanup and attach PyPI origin metadata. +- [x] 6.4 Enqueue harvested package artifacts into `todo_postman.txt` after package scan batches without failing the original package scan. +- [x] 6.5 Deduplicate harvested package artifacts by content hash before enqueueing. + +## 7. Postman-Aware Enrichment + +- [x] 7.1 Parse Postman collection and environment JSON into request, auth, header, query, body, and variable context maps. +- [x] 7.2 Correlate TruffleHog finding locations or nearby context with Postman context maps. +- [x] 7.3 Classify credential kind and provider using DetectorName, value shape, auth/header type, variable name, and endpoint host. +- [x] 7.4 Detect common placeholders and assign placeholder or low-confidence classification. +- [x] 7.5 Persist enrichment fields using existing finding enrichment/database columns where possible. + +## 8. Observability And Configuration + +- [x] 8.1 Add Postman source cycle metrics, queue snapshots, target scan records, findings, and errors to existing database flows. +- [x] 8.2 Add Postman queue counts to dashboard current queues and source health views. +- [x] 8.3 Add redaction coverage for Postman/GitHub auth pool settings in config snapshots and logs. +- [x] 8.4 Add backfill-friendly and tail-friendly config examples in `config.yaml` comments. + +## 9. Verification + +- [x] 9.1 Run a small Postman GitHub discovery cycle with `pages: 1`, `per_page: 10`, and `max_targets` set. +- [x] 9.2 Re-run the same cycle and verify duplicate targets are skipped through `todo_postman.txt` and `checked_postman.txt`. +- [x] 9.3 Verify all-token rate-limit fallback with a simulated or controlled token-unavailable state. +- [x] 9.4 Verify npm and PyPI harvesting using a package fixture containing collection and environment JSON files. +- [x] 9.5 Verify database and dashboard visibility for Postman source cycles, queues, target scans, findings, and errors. +- [x] 9.6 Run `openspec status --change add-postman-source` and ensure all implementation tasks are complete before archive. diff --git a/openspec/changes/add-web-operations-control-plane/.openspec.yaml b/openspec/changes/add-web-operations-control-plane/.openspec.yaml new file mode 100644 index 0000000..eaa6b1c --- /dev/null +++ b/openspec/changes/add-web-operations-control-plane/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-19 diff --git a/openspec/changes/add-web-operations-control-plane/HANDOFF.md b/openspec/changes/add-web-operations-control-plane/HANDOFF.md new file mode 100644 index 0000000..e8a3364 --- /dev/null +++ b/openspec/changes/add-web-operations-control-plane/HANDOFF.md @@ -0,0 +1,1549 @@ +# Implementation Handoff + +Updated: 2026-09-21 + +This file preserves implementation context for the active OpenSpec change +`add-web-operations-control-plane`. It is a working handoff, not a normative +specification. The OpenSpec artifacts and `tasks.md` remain authoritative. + +## User Direction + +- Continue implementation autonomously and pragmatically. +- Avoid overengineering. Ask through the question tool only when a real product + or architecture choice is unclear. +- Keep changes minimal and scoped to the current OpenSpec task. +- Do not modify or remove unrelated worktree changes. +- Do not touch the production server unless explicitly requested. The only + previously approved remote host was `sec`; `prod` was explicitly excluded. + +## Non-Negotiable Provider Architecture + +These decisions are also recorded in the repository `AGENTS.md`. + +- The server validates assignment shape and canonical immutable identity. +- Git planning may bind an exact commit. +- Docker planning may resolve a mutable tag to an immutable digest. +- These planning operations are not provider-access proofs. +- The worker is the final authority for real provider access. +- Discovery credentials do not enter direct assignment fields. +- Do not add server-side provider probes, durable public-access proof state, + proof TTL/freshness migrations, broad worker environment scrubbing, + credential sandboxes, or post-hoc redaction pipelines without explicit user + approval and a new OpenSpec requirement/task. +- Worker ambient HOME/XDG/Git/Docker/provider environment belongs to the worker + operator. +- Permanent provider outcomes include target-scoped auth/access/not-found. +- Retryable provider outcomes include network/rate-limit/provider 5xx failures. + +## OpenSpec Progress + +- Change: `add-web-operations-control-plane` +- Schema: `spec-driven` +- Completed through task 9.4. +- Current count: 50/58 complete. +- Current task: 9.5 (fixed host installation and production wiring). + +## Major Completed Work + +### Operations authority and distributed sources + +- Added PostgreSQL/SQLite operation-control, operation, and append-only audit + authority in migration 29. +- Discovery and dispatch pause gates and drain reconciliation are transactional. +- Server producers are discovery-only for GitLab, DockerHub, and HuggingFace. +- Protocol 2 and package schema 3 support exact GitLab, Docker direct, and + HuggingFace Space assignment capabilities. +- New assignment claims are protocol 2; protocol 1 remains completion-only. + +### Removed rejected infrastructure + +- Docker anonymous-access proof/preflight machinery was removed. +- Proof columns, claim gates, resolver proof handling, and migration 30 were + removed; migration count remains 29. +- Broad Docker/HuggingFace child-environment sanitization was removed. +- Docker tag-to-digest immutable planning remains. + +### Worker/package and canary verification + +- Windows/Linux packaged-worker E2E covers real GitLab scan/recovery plus + synthetic DockerHub/HuggingFace protocol-2 claim-to-ingestion flows. +- Bounded test-only DockerHub canary covers discovery, enqueue, claim, + worker-classified permanent/retryable failures, upload, ingestion, + projection, replay, expiry, and drain. +- Synthetic test transport is test-only. The live public DockerHub canary is + task 10.5. + +### Runtime documents + +- Strict bounded UTF-8 YAML loader rejects duplicate keys, including effective + merge collisions, and exposes content-free errors. +- Combined config/secrets validator enforces exact types, bounds, core profile, + auth pools/references, package capability evidence, and deployment paths. +- Startup, Supervisor launch, preview, and secrets import use the shared + validator without replacing existing lifecycle/path security checks. +- Candidate files use fixed paths: + - `/data/runtime-document-candidates/config.yaml` + - `/data/runtime-document-candidates/secrets.yaml` + - `/data/runtime-document-candidates/candidate.lock` +- Candidate storage uses private descriptor-based reads, raw SHA-256 CAS, + durable atomic replacement, bounded config diffs, and aggregate-only secrets + diffs. +- Candidate concurrency/error-redaction tests cover stale active/candidate + revisions, hardlinks/symlinks, interrupted writes, rollback, lock contention, + cancellation locals, and orphan temporary cleanup. + +### Supervisor and operation services + +- Control protocol stays schema 1 for compatibility. +- Legacy textual snapshot/command remains for CLI and health compatibility. +- Web-facing paths use only exact typed actions; they never call the generic + parser or shell. +- Structured runtime/source snapshot covers every configured managed child. +- Typed lifecycle/settings support all practical Supervisor-managed children, + including dashboard through its separate exact action. +- Bounded log tail accepts only an exact managed-source ID and line count. +- Durable operation methods support apply/restart, managed-source actions, + worker-admin mutations, and content-free hash-chained audits. + +### Web console through task 7.5 + +- Caddy strips inbound private admin headers and injects authenticated Basic + username plus private edge marker. +- Backend trusts operator identity only with the private marker and stores it on + request-local state. +- Shared navigation includes Workers / Dispatch, Overview, Search, Supervisor, + Logs, Config, and Secrets as pages are implemented. +- Overview uses independently degradable bounded health components. +- Search provides persistent discovery pause/resume and exact producer controls. +- Workers / Dispatch provides dispatch pause/resume, drain controls/progress, + package compatibility, users, devices, assignments, and deferred requeue. +- Supervisor page exposes structured state and exact typed controls for all + managed children. Logs page exposes bounded allowlisted tails only. +- Every mutation implemented so far receives trusted actor attribution and a + durable accepted/terminal audit operation. + +## Important Web Decisions + +- Admin pages are server-rendered and contain no application JavaScript. +- Every response remains `no-store` with restrictive security headers. +- Paths are relative so the random public Caddy admin prefix is preserved. +- Actor is the raw validated Basic-auth username; no actor form field exists. +- Exact form shapes reject missing, extra, and duplicate fields. +- POST mutations use canonical operation UUIDs embedded by the server. +- Plaintext device tokens appear only once in their direct POST response. +- Config/secrets content must never enter generic file-service scope. +- Worker status/upload/report/receipt endpoints remain independent of admin + page health. + +## Completed Task 7.6 + +Task text: + +> Add config and plaintext secrets preview/save/apply pages with no-store +> rendering, revision/hash conflicts, and no value leakage outside the editor. + +### Implemented UI and service behavior + +- GET `/admin-internal/config` +- GET `/admin-internal/secrets` +- POST `/{config|secrets}/preview` +- POST `/{config|secrets}/save` +- POST `/{config|secrets}/apply` +- POST `/runtime/apply-both` +- Editors load the candidate when present, otherwise the active document. +- Config editor never receives plaintext secrets. +- Secrets plaintext appears only inside the escaped no-store textarea. +- Preview uses the shared validator and candidate pairing rules. +- Config diff now redacts every changed string value. +- Secrets diff is aggregate-only and contains no pool, entry, username, token, + hash fragment, or secret length. +- Save stages only a candidate and never activates an active file. +- Production currently has no host-agent apply provider. Apply returns 503 + before operation creation. Host apply belongs to tasks 9.x. +- Save operations are durable and content-free: + - `runtime.config.save` + - `runtime.secrets.save` +- Accepted/terminal audit records contain only document name, hashes, byte + counts, written flag, trusted actor, and fixed outcome/category. + +### Active files + +- `app/admin_api.py` +- `app/runtime_document_io.py` +- `app/runtime_document.py` +- `app/scanner_db.py` +- `app/worker_api.py` +- `tests/test_admin_api.py` +- `tests/test_runtime_document_io.py` +- `tests/test_operations_service.py` + +### Completed replay and cancellation-safety fixes + +#### 1. Save replay after the candidate changed length + +Resolved problem: + +- `_save_runtime_document_candidate_bytes()` previews the current candidate and + recomputes `candidate_before_bytes` before asking the database to replay the + existing operation. +- After a successful candidate write, current byte length may differ from the + originally accepted before length. +- Reusing the same operation UUID then conflicts even though it is the exact + request replay. + +Implemented fix: + +- Look up `runtime_operation(operation_id)` before creating a new document + operation. +- If it exists, validate actor, action, target kind/ref, submitted active + hashes, selected candidate-before hash, proposed candidate-after raw hash and + byte count against persisted expected identity. +- Persist and validate the counterpart candidate hash as part of the document + operation expected identity. The accepted operation must bind all four CAS + hashes, not only the selected candidate. +- For a running replay: + - active config/secrets and counterpart candidate must still match accepted + hashes; + - if selected candidate still has the accepted before hash, retry the save; + - if selected candidate already has the accepted after hash, complete the + operation without writing again; + - any other state is a 409 revision conflict. +- Do not recompute accepted before-byte count from post-save state. +- Extend the real ScannerDB operation test and make the admin fake compare the + complete immutable identity, so changed-length replay is covered. + +#### 2. Preserve recovery identity on save errors + +Resolved problem: + +- A 503 `runtime document completion is pending` page renders pre-save state and + generates a fresh operation UUID. +- Pressing Save from that page starts another operation instead of replaying the + pending one. + +Implemented fix: + +- `_render_runtime_document_page()` should accept an optional operation UUID. +- Preview/error responses must keep the submitted operation UUID. +- After save failure/pending completion, reload editor state before rendering so + hashes represent the physical candidate now on disk. +- The textarea may retain the submitted document only inside the editor. + +#### 3. Apply replay after host-side file changes + +Resolved problem: + +- `request_runtime_apply()` verifies current candidate/active filesystem state + before checking for an already persisted operation. +- If the host agent changed active files and its response was lost, an exact + POST retry can fail revision verification instead of recognizing the durable + operation. + +Implemented fix: + +- Check `runtime_operation(operation_id)` first. +- Validate actor, action, target kind/ref and persisted expected hash identity + against the submitted request. +- Terminal success returns without filesystem verification. +- Terminal failure returns 409. +- Requested/running exact replay redispatches the same operation ID/action + without re-verifying post-apply filesystem state. +- A fresh operation still performs full candidate verification before durable + operation creation and dispatch. +- Provider dispatch is expected to be idempotent by operation ID; task 9.x host + agent will enforce persisted-operation/hash checks. + +#### 4. Cancellation traceback plaintext cleanup + +Resolved problem: + +- `asyncio.CancelledError` is a `BaseException` path and can leave raw + URL-encoded body, decoded fields, or `document_text` in traceback locals. + +Implemented fix: + +- Add `try/finally` cleanup in `_form_fields()` for body, decoded text, parsed + pairs, field mapping, and current chunk locals. +- Wrap the document branch of `_dispatch()` in `try/finally` and clear fields, + hashes, editor, preview, operation ID, and document text locals. +- Add a cancellation test that injects cancellation after decoding and asserts + the sentinel is absent from all `admin_api` traceback frame locals. + +## Task 7.6 Verification + +- PyCompile passed for all changed production and test modules. +- Focused runtime-document/admin/operation suites: `96 passed, 1 skipped`. +- Broader admin/runtime/operation/worker/edge/query suites: `179 passed, 1 skipped`. +- Container unit selection: 37 modules, 810 test definitions, no runner skips. +- Desktop/mobile Chrome checks passed for Config and Secrets: no body overflow, + bounded scrollable textarea, prefix-safe relative navigation, no script, + local storage or service worker use, and no console errors. +- The secrets sentinel occurred exactly once in serialized HTML and nowhere + outside the textarea. +- Strict OpenSpec validation passed. +- `git diff --check` passed. The checkout still has no useful tracked baseline. +- Live PostgreSQL was unavailable locally; SQLite service tests and SQL-shape + checks remain the current database coverage. + +## Completed Task 7.7 + +Task 7.7 added durable operation-status and bounded paginated audit pages that +remain backed by database state across runtime/admin restarts. + +### Implementation + +- `ScannerDB.runtime_audit_events(before_event_id=None, limit=50)` returns a + deterministic newest-first keyset page using event ID as the cursor. +- Cursor and page limits are strictly bounded. The query fetches at most one + extra row to determine whether an older page exists. +- The audit read uses a SQLite transaction or PostgreSQL repeatable-read, + read-only transaction and loads referenced operations in one bounded batch. +- Only operation-bound audit rows are exposed. Nullable unlinked storage events + are excluded because they have no typed operation identity safe for the web + console. +- Every returned event is revalidated against its normalized durable operation: + actor, action, target, result/category, canonical expected/resulting identity, + byte counts, parent hash metadata, timestamp, and event hash. +- Added explicit no-store server-rendered routes: + - `/admin-internal/operations` + - `/admin-internal/operations/{canonical-operation-uuid}` + - `/admin-internal/audit` + - `/admin-internal/audit?before={positive-event-id}` +- Unknown or noncanonical operation IDs return 404 without filesystem or agent + probing. Duplicate, extra, or malformed query fields fail closed with 400. +- Operation status renders only normalized persisted safe fields and canonical + expected/resulting identities. +- Audit renders only safe operation-bound event fields and hash identities. +- Apply POST redirects to the durable operation URL after successful dispatch. +- Relative URL roots preserve the random external admin prefix: list/audit pages + use `.`, operation detail uses `..`, and apply redirects use + `../operations/{operation_id}`. +- Operations and Audit were added to shared navigation. + +### Task 7.7 tests and review + +- Admin tests cover operation list/detail, audit pagination, random-prefix URL + resolution, strict query shapes, unsupported methods, security headers, + escaping, and absence of editor/Supervisor fixture secrets. +- SQLite service tests cover pagination without overlap, cursor exhaustion, + operation completion identities, and close/reopen persistence. +- PostgreSQL integration coverage was added for pagination across ScannerDB + reopen and rejection of audit UPDATE, DELETE, and TRUNCATE. All 59 bundled + PostgreSQL tests were discovered but skipped locally because PostgreSQL is + unavailable. +- An independent review found and then verified fixes for relative URL escape, + nullable unlinked audit rows, and missing PostgreSQL coverage. Its final + result reported no concrete findings. +- Focused admin/operation suites passed: 54 tests. +- Broader selected admin, runtime-document, operation/control/schema, worker API, + edge deployment, and production-query checks passed; the unittest modules + reported 183 tests with one expected Windows skip. +- Container unit selection passed: 37 modules, 814 test definitions, no runner + skips. +- Desktop and 390px mobile browser checks passed for operation list, operation + detail, and audit pages. The document had no horizontal overflow; wide tables + scroll inside their bounded container. All links remained under the admin + prefix. No scripts, local storage, service-worker controller, fixture-secret + leakage, console warnings, or console errors were present. +- Strict OpenSpec validation and `git diff --check` passed. + +## Completed Task 7.8 + +Task 7.8 adds admin API and browser coverage for routes, methods, exact form +shapes, actor spoofing, Origin/CSRF, stale forms, random-prefix navigation, CSP, +and secret non-retention. + +### Production fixes already implemented + +- `_dispatch()` rejects query fields before any state read or mutation. Only + `GET /admin-internal/audit` accepts the exact optional `before` cursor; every + other route and method requires an empty query. +- `_AdminRoute(Route)` upgrades Starlette method-only partial matches and calls + the admin endpoint directly, so arbitrary methods such as `PROPFIND` receive + the application's secured 405 response instead of an unprotected framework + response. +- Supervisor controls were compacted without changing the typed backend: + Dashboard and every managed source use collapsed `
    ` panels. Exact + action routes, CSRF, operation UUIDs, allowlists, and no-JavaScript behavior + remain unchanged. No shell, generic command/action field, or parser was added. + +### API and edge test work implemented + +- `tests/edge_e2e_client.py` now submits canonical operation UUIDs for user + creation, expects the current 303 redirect, and verifies PostgreSQL operation + plus accepted/succeeded audit rows attribute the authenticated Basic user, + not spoofed inbound operator headers. +- A table-driven contract test covers all 45 exact POST routes. Each route + rejects the wrong method, incomplete fields, and attacker Origin with secure + headers and no side effects. +- All six stale control forms return secured 409 without a transition. +- Config/secrets preview, save, individual apply, and apply-both now have direct + valid-route coverage. +- Eighteen stale runtime-document hash cases cover every active, selected + candidate, and counterpart candidate hash before mutation/operation creation. +- All rendered Search and Supervisor exact actions/settings have successful + dispatch coverage; semantically invalid keychecks loop mode remains excluded. +- CSRF/Origin tests now cover missing/duplicate CSRF and duplicate/wrong Origin, + proving only the valid request mutates. +- Static random-prefix crawl covers root, overview, Search, Supervisor, Logs, + Config, Secrets, Operations, operation detail, and Audit. Every link/form + remains under the simulated private prefix; every response has strict headers + and no script/browser-storage API references. +- The fixture secret occurs exactly once inside the Secrets textarea and nowhere + outside it or on other pages. Form and textarea autocomplete are off. +- `tests/test_admin_browser.py` starts a real local ASGI fixture behind a + simulated random external prefix and drives Chromium through pinned Playwright + (`tests/requirements-browser.txt`). Its dedicated lane fails rather than + silently skipping when Playwright is absent. +- The browser test visits every navigation page, validates exact response + security headers and inline-script CSP enforcement, rejects unexpected console + errors, checks desktop/mobile body overflow, and proves every link, stylesheet, + and form remains under the random external prefix. +- The browser test also proves the plaintext fixture secret exists only in the + Secrets textarea, no local/session/Cache/IndexedDB/service-worker persistence + exists, and an unsaved sentinel is absent after a network reload. Native + browser back-forward memory is not treated as application persistence. + +### Final verification state + +- PyCompile passed for `app/admin_api.py`, `tests/test_admin_api.py`, and + `tests/test_admin_browser.py` plus `tests/edge_e2e_client.py`. +- `tests/test_admin_api.py` passed: 45 tests. +- `tests/test_admin_browser.py` passed in real local Chrome: 1 test. +- Broad selected admin/runtime-document/operation/control/schema/worker/edge/ + query suites passed: 190 unittest tests, one expected Windows symlink skip. +- Container selection passed: 37 modules, 821 test definitions, no runner skips. +- Full local edge E2E passed with current rebuilt images. It proved durable + operation/audit attribution to the authenticated Basic user, spoofed operator + header stripping, Origin/CSRF behavior, worker-route independence, fail2ban + behavior, cleanup, and unchanged foreign Docker state. +- Desktop and 390px mobile real-browser checks passed with no body overflow, + scripts, unexpected console errors, or secret leakage. Wide Audit content + scrolls only inside its bounded table container. +- An independent read-only review found no remaining concrete task 7.8 defect. +- Strict OpenSpec validation and `git diff --check` passed. + +## Completed Task 8.1 + +Task 8.1 defines the trusted logical-root policy and exact per-root permissions +and limits without opening files or prematurely implementing traversal, file I/O, +HTTP Files pages, or mutation audit behavior from tasks 8.2–8.4. + +### Policy and configuration + +- Added `app/managed_files.py` with immutable typed root, permission, limit, and + registry records plus the exact operation enum: list, read, create-replace, + and delete. +- Root IDs are bounded lowercase logical names; absolute host/container paths + exist only in trusted server configuration and are hidden from root repr. +- Every root must explicitly provide all four permissions and all six limits: + relative-path bytes, component bytes, path depth, listing entries, listing + bytes, and file bytes. Per-root values cannot exceed fixed hard caps. +- The predefined existing roots are `runtime-logs`, `runtime-keychecks`, and + `runtime-results` at their fixed runtime directories. They are forced to + list/read only and cannot be renamed or granted mutation permissions. +- `runtime-results` permits files up to 256 MiB so active and rotated + `scan_results.jsonl` and `found_secrets.jsonl` generations remain + downloadable. Only active names and exact six-digit generation names are + visible; locks, databases, ledgers, scan errors, temporary/quarantine + directories, malformed generations, and recovery artifacts are excluded. + Keycheck and log roots retain the narrower 64 MiB per-file bound, and custom + roots cannot opt into the larger result bound. +- Future writable roots are permitted only as immediate children of the isolated + `/data/managed-files` directory. No writable root is enabled by default. +- This narrow allowlist excludes config/secrets, candidates, PostgreSQL, + application code, sockets/control state, host-agent metadata, raw result + bundles/spool, runtime state/queues/caches/work, and imported archives in both + normal and parent-container directions. Projected results and keycheck output + are available only through their fixed read-only roots. +- Root mappings are deterministically ordered, duplicate paths are rejected, + nested fields are exact, booleans cannot pass integer limits, and errors do + not echo submitted IDs or paths. +- Missing roots mean an empty registry. Existing explicit `admin: null` + compatibility remains an empty disabled admin configuration, while a present + `managed_file_roots: null` or any other malformed section fails closed. + +### Runtime wiring + +- `app/config.linux.yaml` defines the read-only `runtime-logs`, + `runtime-keychecks`, and `runtime-results` roots with bounded deployment + defaults. +- `app/runtime_document.py` treats root IDs as a strict dynamic mapping, validates + the policy even while admin is disabled, preserves omission compatibility, and + includes the new dynamic shape in the pinned template schema hash. +- `app/worker_api.py` validates roots before DSN/secrets/assignment-builder work + and passes the immutable registry into `AdminService`. +- `AdminService` accepts only a `ManagedFileRootRegistry` and otherwise defaults + to an empty registry. No `/files` navigation or route exists yet; that belongs + to task 8.4. +- Task 8.1 performs no `open`, stat, path resolution, directory creation, or + network work. Descriptor opening and target traversal remain task 8.2. + +### Verification + +- Added `tests/test_managed_files.py`; 8 policy/configuration tests passed. +- Runtime-document validation passed: 22 tests. +- Worker runtime wiring passed: 12 tests. +- Admin API passed: 46 tests. +- Runtime-document I/O regressions passed: 25 tests with one expected Windows + symlink skip. +- Container selection passed: 38 modules, 832 test definitions, no runner skips. +- Strict OpenSpec validation and `git diff --check` passed. +- Independent read-only review found no remaining concrete correctness or + security findings. + +## Completed Task 8.2 + +Task 8.2 adds the Linux descriptor-containment layer. It deliberately does not +yet enumerate directory entries, return file bytes, mutate files, add HTTP +routes, or write audit events; those belong to tasks 8.3 and 8.4. + +### Canonical path and descriptor model + +- `parse_managed_relative_path()` requires an exact UTF-8 string and returns + unchanged components only after enforcing the configured byte, component, and + depth limits. +- Empty, absolute, drive-qualified (including nested drive components), + backslash, NUL, repeated-separator, trailing-separator, dot, and dot-dot paths + fail before target traversal. +- `ManagedFileTraversal` is Linux-only for nonempty registries and fails closed + unless `O_PATH`, `O_DIRECTORY`, `O_NOFOLLOW`, `O_CLOEXEC`, descriptor-relative + `os.open`, and the required nonblocking flags are available. +- Each configured absolute root is opened from `/` one component at a time with + `dir_fd`, `O_PATH | O_DIRECTORY | O_NOFOLLOW`, and retained for the traversal + lifetime. Symlinked root or parent components cannot be followed. +- Each operation duplicates the retained root under a lock before walking. This + prevents concurrent `close()` and descriptor-number reuse from redirecting a + request; already-started operations remain anchored after close. +- Intermediate client components use the same `O_PATH | O_DIRECTORY | + O_NOFOLLOW` traversal. A renamed/replaced configured pathname does not change + the retained root object. +- Listing opens only a verified directory descriptor. `None` is the internal + root-list sentinel; an empty client path remains invalid. +- Read targets are first opened with `O_PATH | O_NOFOLLOW`, checked as a + single-link regular inode, then reopened through their own `/proc/self/fd` + descriptor and revalidated by type, link count, device, and inode before a + readable descriptor is returned. Devices/FIFOs/sockets therefore are not + opened for reading before type rejection. +- Opened targets are context-managed; partial root walks, operation walks, + ordinary exceptions, and `BaseException` paths close every descriptor they + unambiguously own. `close()` is lock-safe and idempotent; use-after-close + fails closed. +- Errors expose only bounded categories such as `invalid_path`, `unknown_root`, + `operation_not_allowed`, `not_found`, `unsafe_target`, `root_unavailable`, + `filesystem_unavailable`, and `closed`. They never echo client paths, root + paths, errno text, or filenames. + +### Adversarial coverage and packaging + +- Traversal tests cover canonical ASCII/Unicode parsing, UTF-8 byte accounting, + exact no-follow/dir-fd flags, partial-constructor and operation cancellation, + close/open races, permissions, unknown roots, non-Linux behavior, and procfs + failure classification. +- Real Linux tests cover nested list/read descriptor opens, root rename/path + replacement anchoring, root/intermediate/final symlinks, hardlinks, + directory-as-file, FIFO, Unix socket, component swap after intermediate open, + idempotent close, and use-after-close. +- `.dockerignore` now admits only the new managed-file module/test explicitly, + and `container_unit.py` selects the policy, mock traversal, and Linux kernel + traversal classes. + +### Verification + +- Native managed-file suite: 17 tests, 13 passed and 4 expected Linux-only + skips. +- WSL Linux managed-file suite: 17/17 passed. +- Rebuilt read-only Linux test image: 17/17 managed-file tests passed under UID + 10001 and the container audit fences. +- Related runtime-document, worker-runtime, admin, and runtime-document-I/O + suites passed: 105 tests with one expected Windows symlink skip. +- Container selection passed: 38 modules, 841 test definitions, no runner skips. +- Strict OpenSpec validation and `git diff --check` passed. +- Final independent security review found no concrete task 8.2 findings. + +## Completed Task 8.3 + +Task 8.3 adds the bounded listing/download and durable file-mutation service on +top of task 8.2 descriptor containment. It deliberately adds no HTTP route, +operator form, durable operation row, or audit event; those belong to task 8.4. + +### Listing and download + +- `list_directory()` lazily scans an already verified directory descriptor, + counts every encountered entry against the configured work limit, bounds + returned UTF-8 name bytes, and sorts accepted logical names deterministically. +- Listings expose only directories and bounded single-link regular files. + Symlinks, hardlinks, special files, over-limit files, invalid names, and + internal temporary names are omitted without revealing host paths. +- `download_file()` reuses the `O_PATH`/`/proc/self/fd` inode-safe open, rejects + oversized files before reading, reads at most the configured bound, computes + SHA-256 while reading, and verifies stable inode/type/link/size/mtime/ctime + metadata before returning content. +- Public result records expose logical names, file/directory kind, bounded byte + counts, hashes, and content only on the direct download result. Download + content and host descriptors remain hidden from repr. + +### Durable create/replace/delete + +- Client relative components beginning with the reserved internal temporary + prefix are rejected; the prefix is also hidden from listings. +- Create/replace accepts exact `bytes` up to the root file limit. `None` expected + hash means create-only; a canonical lowercase SHA-256 means replace-only. +- Temporary files are created with random same-directory names using + `O_RDWR | O_CREAT | O_EXCL | O_NOFOLLOW | O_CLOEXEC`, start private, are + hardened to exact mode `0600`, and must be regular, single-link, and owned by + the effective runtime UID. +- Writes handle short writes/interruption, verify exact size and SHA-256, fsync + the temporary inode, and retain its descriptor through publication. +- Create publishes atomically with Linux `renameat2(RENAME_NOREPLACE)`, so an + existing name is never overwritten. Replace revalidates the expected inode + and hash before descriptor-relative `os.replace`; readers observe complete old + or complete new files. +- Every mutation takes both an in-process lock and an advisory exclusive flock + on the opened parent-directory inode. Separate traversal instances/processes + using this service therefore implement one cooperative hash-CAS authority. + Writable roots are dedicated immediate children of `/data/managed-files`; + arbitrary SSH/root writers that ignore the lock are outside this contract. +- Namespace publication and unlink run inside a `finally`-protected parent + directory fsync. A failure after visible mutation is reported only as bounded + `durability_uncertain`/`concurrent_change`; the service never attempts an + unsafe rollback over a later writer. +- Final published files are re-read and verified for expected hash/bytes, + original staged inode, effective UID, exact `0600`, regular type, and one + link. The parent descriptor is then revalidated. +- Replace with identical expected/proposed identity is an idempotent no-write. +- Delete requires the exact current SHA-256, revalidates the named inode, + unlinks descriptor-relatively, and fsyncs the parent directory. +- Prepublication failure and `BaseException` paths unlink and directory-fsync + the known temporary name before potentially interrupted descriptor close. + Explicit `LOCK_UN` prevents an interrupted parent close from retaining the + mutation lock. Cancellation is never hidden by an ordinary conflict/error. +- Errors remain content/path/errno-free categories such as `invalid_hash`, + `invalid_content`, `limit_exceeded`, `hash_conflict`, `concurrent_change`, + `durability_uncertain`, and the task-8.2 access categories. + +### Verification + +- Native Windows managed-file suite: 28 tests, 13 passed and 15 expected + Linux-only skips. +- WSL Linux managed-file suite: 28/28 passed, including real `renameat2`, flock, + inode/mode/owner checks, fsync, cleanup, and cross-instance concurrency. +- Rebuilt read-only Linux test image: 28/28 managed-file tests passed under UID + 10001 with network disabled; 1048 unrelated tests were deselected. +- Cross-instance and independent-process replace/replace and replace/delete + probes produced exactly one winner. +- Related runtime-document, worker-runtime, admin, runtime-document-I/O, and + operation-service suites passed: 121 tests with one expected Windows symlink + skip. +- Container selection passed: 38 modules, 852 test definitions, no runner skips. +- Final independent security review found no concrete task 8.3 findings. + +## Completed Task 8.4 + +Task 8.4 exposes the descriptor-safe managed-file service through exact +server-rendered admin routes and binds every accepted mutation to a durable, +content-free operation and audit lifecycle. + +### Files pages and forms + +- Shared admin navigation now includes `Files`. +- `GET /admin-internal/files` renders either the logical-root index or a bounded + root/nested listing selected by exact `root_id` and optional canonical + `relative_path` query fields. +- `GET /admin-internal/files/download` requires exact logical root/path fields + and returns `application/octet-stream` with no-store/security headers, exact + length, SHA-256 ETag, and a logical-leaf attachment filename. Ordinary files + retain the bounded in-memory response. An allowed result projection is first + copied and hashed in 64 KiB chunks into an anonymous stable snapshot on the + same data volume, then streamed from that snapshot with bounded memory. + Source revision/size changes retry once and then fail closed; only one result + snapshot may exist at a time, and response completion/failure releases it. + No host path is passed to a response object. +- Async request cancellation retains cleanup ownership until a background + snapshot build finishes, while the streaming response closes the snapshot in + a response-level `finally`; disconnects and ASGI send failures therefore + cannot strand the sole result-download permit. +- Exact POST routes are `/files/create`, `/files/replace`, and `/files/delete`. + They accept only CSRF, operation UUID, logical root ID, canonical relative + path, canonical expected hash where required, and canonical URL-safe Base64 + content for create/replace. +- The existing bounded URL-encoded admin body is the explicit web-upload cap; + the page displays that bound. No multipart parser, JavaScript, command field, + actor field, absolute root, or generic action field was added. +- Root pages expose only logical IDs, permissions, configured limits, safe + relative names/kinds/byte counts, and permission-appropriate forms. Private + configured absolute roots and arbitrary file bytes never enter HTML. +- Query and form shapes reject duplicate, missing, extra, partial, empty, or + noncanonical values before filesystem or database mutation. +- Managed-file errors map only to bounded HTTP outcomes (400/403/404/409/413/ + 503) and never expose paths, errno text, content, or raw exceptions. + +### Durable mutation authority + +- ScannerDB recognizes exact actions `files.create`, `files.replace`, and + `files.delete` with target kind `managed-file` and logical root ID as the + bounded target reference. +- Canonical expected identity contains only root ID, relative path, expected + SHA-256, proposed SHA-256, and proposed byte count. Successful resulting + identity contains only before/after hashes and counts, outcome, and written + flag. No migration was required; migration count remains 29. +- `create_runtime_managed_file_operation()` atomically commits a running + operation and accepted audit before physical mutation. +- `complete_runtime_managed_file_operation()` commits one terminal audit and + safe resulting identity, or the fixed content-free failure category + `managed_file_mutation_failed`. +- Uploaded bytes, Base64, CSRF/Origin/auth data, credentials, absolute roots, + exception text, and rendered content cannot be accepted by the operation or + audit APIs. +- List and download remain read-only and create no operation/audit rows. + +### Replay and execution serialization + +- Each mutation holds a per-operation execution claim on a dedicated DB + connection across terminal lookup, preflight, acceptance, physical CAS, and + terminal completion. +- PostgreSQL uses a session advisory lock keyed by the canonical operation UUID. + SQLite uses deterministic process-global lock stripes for local/test + connections. Different operation IDs targeting one path remain serialized by + task 8.3 parent-directory flock and hash CAS. +- Lock order is operation advisory claim, AdminService local lock, then traversal + parent-directory lock. Release and connection close run through unconditional + nested cleanup even under `BaseException`; cancellation remains primary and + uploaded payload references are cleared. +- Exact terminal success validates immutable operation identity and returns + without requiring current traversal/root availability. Terminal failure + requires a new operation UUID. +- A fresh request validates current root permission/path and expected namespace + state before accepted audit creation. Create requires absence; replace/delete + require the exact expected hash. A fresh CAS loss always fails even if another + writer produced matching bytes. +- A running exact replay may retry only from its accepted before state. It may + complete without a second mutation only after observing the accepted desired + after-state (or stable deletion), fsyncing the parent directory, and + revalidating the same named inode/revision or absence after fsync. +- Matching replayed create/replace after-state must also be a private regular + single-link file owned by the runtime UID with exact mode `0600`. +- Content-bearing parser, dispatcher, service, and traversal frames clear or + release URL-encoded, Base64, decoded payload, and memory-view locals on normal, + error, and cancellation paths. + +### Traversal lifecycle + +- Worker API lifespan constructs exactly one retained `ManagedFileTraversal` + when admin is enabled and reuses it for all Files requests. +- Root-open failure safely degrades Files to 503 without taking down worker + claim/status/upload/report routes and logs only a bounded category/type. +- Nested shutdown cleanup clears app state and closes the traversal exactly once + even when assignment-reaper shutdown fails. Admin-disabled apps never open + managed roots. + +### Verification + +- Focused host suites passed: operation service 19/19, admin API 54/54, Worker + API 35/35, and real-browser admin coverage 1/1. +- Managed-file tests passed 30/30 on WSL Linux; native Windows ran the same 30 + with 17 expected Linux-only skips. +- Related runtime-document, runtime-document-I/O, worker-runtime, operation + control/schema, edge-deployment, and production-query suites passed; the only + skip was the expected Windows symlink case. +- Real Chromium covered Files navigation under the random prefix, security/CSP, + no scripts/storage, mobile layout, and console cleanliness. +- Rebuilt read-only/no-network Linux test image passed all 46 selected + managed-file/admin/operation/lifecycle tests under UID 10001; 1046 unrelated + tests were deselected. +- Container selection passed: 38 modules, 868 test definitions, no runner skips. +- SQLite two-connection execution serialization and cancellation cleanup have + dedicated regressions. A PostgreSQL two-session advisory-block/unlock/ + session-close integration test is committed and discovered, but skipped + locally because bundled PostgreSQL is unavailable. +- Strict OpenSpec validation and `git diff --check` passed. +- Final independent security review found no concrete task 8.4 findings. + +## Completed Task 8.5 + +Task 8.5 closes the managed-file service with adversarial traversal, link, +special-file, limit, concurrency, durability, transport-encoding, and +forbidden-root coverage. + +### Production hardening + +- Nested listings validate the full logical child path, not only the leaf name, + before exposing an entry. Children beyond the configured path-byte or depth + bound remain hidden. +- Temporary-file cleanup fsyncs the parent directory even when the allocated + temporary name is already absent, preserving durable cleanup evidence. +- Admin dispatch requires byte-valued ASGI `raw_path` to be the exact canonical + ASCII encoding of the decoded fixed route. Missing or percent-encoded route + aliases fail with a secured 404; query and form values still decode exactly + once through their normal parsers. +- Download stability again requires an unchanged regular single-link inode, + including device, inode, size, link count, mtime and ctime. On a concurrent + atomic replacement, download performs at most one canonical reopen and strict + reread, so callers receive one complete version rather than mixed bytes. + +### Adversarial coverage + +- Policy tests explicitly exclude active secrets, candidate documents, + PostgreSQL, host-agent metadata, Docker/runtime sockets, application code, + raw bundles, spool/results/state/queues/cache/work, and imported archives. +- Portable tests prove FIFO, socket, character-device, block-device and + directory modes are rejected before a readable `/proc/self/fd` reopen. +- Real Linux tests cover root and final symlink swaps, retained-descriptor + anchoring, hardlinks, directories, FIFOs, hidden unsafe listing entries, and + rejection by download/replace/delete. +- Exact and one-over tests cover total path bytes, component bytes, depth, + listing entry/name-byte limits, nested logical paths, and file download/upload + limits. +- Two traversal instances prove one winner for create/create and delete/delete; + prior replace/replace and replace/delete tests remain. Reader replacement + tests prove complete-version behavior, including a hostile in-place write, + restored mtime, and atomic pathname replacement. +- Durability fault tests cover temporary inode fsync, already-absent temporary + cleanup, cleanup unlink/fsync failures, and replace/delete parent-fsync + uncertainty while asserting the resulting visible namespace. +- HTTP tests cover encoded and mixed-case traversal, backslash, NUL, drive + prefixes, exact one-pass double decoding, encoded fixed-route aliases, + forbidden logical roots, generic action/command/path/service field injection, + and exact/over/streaming request-body limits before decode, traversal, or + operation/audit creation. + +### Writer authority boundary + +- Writable roots are dedicated managed-service roots. Cooperating app instances + and processes serialize mutations with the parent-directory advisory flock + and hash CAS. +- Arbitrary SSH/root writers that deliberately ignore that flock are outside + this authority contract; POSIX has no atomic primitive for + `replace/unlink iff current content hash equals X` against such writers. +- The service still detects bounded concurrent drift wherever possible and + never follows a swapped symlink target. + +### Verification + +- Managed-file tests passed 37/37 on WSL Linux; native Windows ran the same 37 + with 23 expected Linux-only skips. +- Native scoped review reported 71 passed plus the expected Linux skips, and the + admin API adversarial suite passed. +- Broad runtime-document, operation/control/schema, Worker API/runtime, + production-query, edge-deployment, and real-browser suites passed; the only + broad skip was the expected Windows symlink case. +- Container selection passed: 38 modules, 878 test definitions, no runner + skips. +- Rebuilt `truf-worker-test:task-8-5`; a read-only, no-network Linux run under + UID 10001 passed all 55 selected managed-file/admin/operation/lifecycle tests, + including all 37 Linux managed-file tests; 1047 unrelated tests were + deselected. +- Independent final security review found no concrete in-scope correctness or + security findings. +- Strict OpenSpec validation and `git diff --check` passed before artifact + closure. + +## Completed Task 9.4 + +Task 9.4 adds one fixed automatic rollback attempt, durable terminal evidence, +and a global failed-hold fence without adding request-selected lifecycle or file +surfaces. + +### Byte-identical rollback + +- Stopped-runtime proofs are operation-bound, purpose-bound (`forward` or + `rollback`), uniquely issued, and single-use. Forward proof cannot authorize + restoration and rollback proof cannot authorize candidate publication. +- `HostApplySession` retains authoritative original snapshots separately from + observed active/candidate state and classifies publication as `original`, + `partial`, or `candidate`. +- Exact mixed old/candidate replay is recovery-only. It never reapplies the + candidate deployment. +- `restore_backups()` accepts no paths or payloads. It rereads fixed root-owned + operation backups, accepts active bytes only when exactly original or exact + candidate, stages byte-identical originals, rechecks CAS, atomically restores, + adopts fixed runtime ownership/mode, fsyncs, and verifies the complete original + pair. Backups and candidates remain as evidence. +- Exact root-owned originals left by interruption between rollback rename and + ownership adoption are recognized and safely adopted on replay. Any unknown + active bytes fail closed without overwrite. +- Publication state becomes `partial` before the first restoration mutation, so + a failed second restore cannot be misreported as a complete candidate state. + +### One-attempt recovery and failed hold + +- Lifecycle execution retains a private mutable attempt before the first stop, + including the original attested image/config identities and every observed + stop/remove/recreate milestone. +- Any failure after lifecycle mutation quiesces only the exact fixed edge and + runtime containers, restores selected backups, and recreates/verifies the + original attested runtime and edge exactly once with the same strict aggregate + health gates. +- Successful recovery emits `rolled_back` with both original active hashes and + the bounded forward failure category. It never reports apply success. +- Recovery failure writes the global failed-hold marker before one containment + stop, preserves operation/backups/candidates, emits `failed_hold`, and performs + no third deployment attempt, restoration loop, or forced authority release. +- A global marker fences every different operation. The same operation may only + finish a missing `failed_hold` phase/result; it skips document preparation and + cannot resume forward or rollback lifecycle work. +- Pre-mutation validation/identity failure records `failed` without downtime or + rollback. + +### Durable phase and result evidence + +- New `app/host_agent_state.py` uses only fixed operation/result/hold paths under + `/var/lib/truf/host-agent`, canonical bounded JSON, no-follow stable reads, + root metadata checks, same-directory atomic replacement, file fsync, directory + fsync, and exact-byte idempotent replay. +- Closed phases are `prepared`, `forward_started`, `rollback_started`, + `succeeded`, `failed`, `rolled_back`, and `failed_hold`; transitions bind the + exact operation UUID, action, four request hashes, publication state, bounded + category/detail, and containment evidence. +- The terminal phase is persisted before the immutable seven-field ScannerDB- + compatible result. Missing results are deterministically reconstructed from a + terminal phase without lifecycle mutation. +- A write that renamed phase evidence but could not confirm parent durability is + classified `uncertain`. Terminal uncertainty never rolls back or contains an + already healthy deployment; nonterminal rollback uncertainty durably fences + and contains instead of retrying. +- `KeyboardInterrupt` and `SystemExit` propagate directly before mutation. After + mutation, the one rollback/failed-hold safety path runs first and the original + cancellation is still propagated, including result-write and phase-rename + uncertainty windows. + +### Verification + +- Lifecycle tests passed 32 selected cases in the final Linux container and + native Windows passed with four expected POSIX-only skips. +- Host apply passed 23/23 under WSL root; native passed 18 with five expected + root/POSIX skips. Durable state passed 6/6, protocol passed 15/15. +- Broad operations, container-runtime, runtime-document/I/O, admin, Worker API, + worker-runtime, and edge-deployment regressions passed. +- Container selection passed: 43 modules, 968 test definitions, no runner skips. +- Rebuilt `truf-worker-test:task-9-4`; a read-only, no-network container run as + UID 10001 passed 76 selected host-agent tests with two expected distinct-root- + ownership skips; 1119 unrelated tests were deselected. +- Independent security/correctness review completed repeated crash/cancellation + review rounds and returned no findings after terminal ordering, early fencing, + replay, uncertainty, and cancellation fixes. +- Strict OpenSpec validation and `git diff --check` passed before artifact + closure. + +## Completed Task 9.1 + +Task 9.1 establishes the authenticated, closed privileged-agent request +boundary without implementing file replacement, restart, health, or rollback. + +### Protocol and client + +- `app/host_agent_protocol.py` defines a strict canonical UTF-8 JSON protocol + framed by a four-byte big-endian payload length. +- Requests are bounded to 1024 payload bytes and contain exactly operation UUID, + one of `apply-config`, `apply-secrets`, `apply-both`, or `restart`, and the + four active/candidate hash fields used by ScannerDB expected identity. +- Canonical nonzero UUID, lowercase SHA-256 values, and the action-specific + candidate-null/hash matrix are enforced. Duplicate, missing, extra, reordered, + whitespace-altered, non-finite, or incorrectly typed fields fail closed. +- No path, service, unit, command, argv, environment, Docker/Compose argument, + timeout, actor, content, or generic options field exists. +- Each connection carries exactly one request and one response. The client + half-closes its write side; the server requires EOF, rejecting trailing or + pipelined bytes. Absolute monotonic deadlines prevent trickle extension. +- `app/host_agent_client.py` has no configurable socket path and uses only + `/run/truf/host-agent.sock`. It requires a root-owned socket pathname and + kernel `SO_PEERCRED` proving the connected server UID is root; there is no + retry loop. + +### Root agent boundary + +- `app/host_agent_server.py` checks kernel peer credentials before reading any + body and permits only the fixed runtime UID 10001. GID is intentionally not + part of the identity contract. +- The server validates an inherited systemd socket as exact `AF_UNIX`, exact + `SOCK_STREAM`, listening, fixed-path, socket-typed, and root-owned. +- `deploy/host-agent/truf_host_agent.py` requires root, no command-line + arguments, and exactly one systemd-activated listener. +- The production task-9.1 handler always returns `unavailable`. It never claims + `accepted` before task 9.2 has verified the persisted operation and durably + assumed ownership. +- Systemd units, host path installation and socket permissions remain task 9.5; + the agent executable is ready for that fixed activation contract but no unit + was added early. + +### Runtime integration boundary + +- Admin apply provider calls now pass the persisted expected identity fields: + active config/secrets hashes and candidate config/secrets hashes, in addition + to operation ID and action. +- Production Worker API still installs no apply provider. An absent agent + therefore preserves the existing 503 before operation creation. Wiring is + deferred until persisted verification/execution exists; no request can + currently trigger privileged lifecycle work. + +### Verification + +- Portable protocol/client/server tests passed 15/15, including exact schemas, + candidate matrix, framing bounds, duplicate/extra fields, absolute trickle + deadlines, response pipelining, cancellation cleanup, unauthorized-before- + read ordering, listener type/listening checks, and inherited-FD cleanup. +- Real Linux tests passed 2/2 under fixed UID 10001 using kernel `SO_PEERCRED`. +- Rebuilt `truf-worker-test:task-9-1`; a read-only, no-network container run as + UID 10001 passed all 17 selected host-agent tests, including the real Linux + peer path; 1102 unrelated tests were deselected. +- Broad admin 57, operations 19, Worker API 35, worker-runtime 12, edge-static, + and provider-hash-binding regressions passed. +- Container selection passed: 40 modules, 895 test definitions, no runner + skips. +- Independent security review found no concrete task-9.1 correctness or + security findings after exact `SO_TYPE`/`SO_ACCEPTCONN` hardening. +- Strict OpenSpec validation and `git diff --check` passed before artifact + closure. + +## Completed Task 9.2 + +Task 9.2 adds a fail-closed execution core for one persisted runtime apply under +fixed host paths. It remains deliberately unwired until task 9.3 can stop the +runtime before publication and task 9.5 installs protected DB/path authority. + +### Persisted operation claim + +- `ScannerDB.claim_runtime_operation_execution(...)` atomically binds the + canonical operation UUID, action, runtime-deployment target, and exact four + active/candidate hashes to the requested-to-running transition. +- The method uses the existing SQLite write transaction or PostgreSQL control + and operation row locks. The host session separately requires PostgreSQL; + SQLite support exists only for deterministic database tests. +- Requested/pending operations can be claimed once. An exact running/running + operation replays; unknown, mismatched, terminal, or incoherent rows fail. +- Claim requires exactly one canonical accepted audit for the operation and + verifies its actor/action/target/identity/time, global predecessor link, the + predecessor's own payload hash, and the accepted event hash before transition. +- Schema-valid unlinked audit predecessors remain supported. No schema migration + or new authority table was added. + +### Fixed host apply session + +- `app/host_agent_apply.py` defines only fixed constants: active config/secrets, + candidate config/secrets, root apply lock, operation-ID backup directory, and + the trusted packaged config template. Socket requests cannot provide paths, + commands, services, environment, or options. +- `HostApplySession` takes the root-private nonblocking singleton filesystem lock + before claim and keeps it through validation, backup, publication, and later + task-9.3 lifecycle insertion points. The lock survives PostgreSQL shutdown. +- The session requires PostgreSQL authority, claims the exact persisted request, + stable-reads regular single-link active/candidate/template files with nofollow, + enforces fixed owner/group/mode and raw hash CAS, selects the action-specific + effective pair, and calls the shared strict runtime-document validator. +- Package capability evidence is host-supplied trusted input, never a socket + field. Production remains unwired until task 9.5 provides fixed protected + package/DB/path mappings. + +### Backup, publication, and replay + +- Selected active bytes are backed up byte-for-byte under the canonical + operation UUID using exclusive creation, root-only metadata, file fsync, + reread/hash verification, and parent-directory fsync. Existing backups are + accepted only when their bytes and identity exactly match. +- All selected replacements are staged and fsynced before publication. Stages + remain root-owned until same-directory atomic rename, then receive fixed + runtime ownership/mode and are fsynced again; the runtime UID cannot mutate a + predictable stage before publication. +- Active and candidate snapshots are rechecked immediately before publication. + On POSIX, the displaced active inode is retained and reread after rename, so a + late in-place writer becomes a conservative partial result rather than silent + data loss. +- Candidates and backups remain as evidence. A stale fixed stage from the same + operation is durably removed before retry. +- Exact running replay distinguishes not-published, completely-published, and + partially-published state. Complete replay requires exact backups, candidates, + and active bytes, recovers a root-owned post-rename/pre-adoption file, and + revalidates again at `replace()`. Partial replay requires valid backups and + fails closed for task 9.4 recovery. +- Each config/secrets rename is atomic; the pair is not a filesystem transaction. + A failure after any publication is categorized `partial`; task 9.2 does not + invent rollback or failed-hold behavior ahead of task 9.4. +- `restart` still claims and validates the active pair but creates no backup and + publishes no file. + +### Production boundary + +- `deploy/host-agent/truf_host_agent.py` was not wired to this core and still + returns `unavailable`. Current deployment has no protected host PostgreSQL + endpoint, fixed mounts/permissions, or safe stop/recreate sequence. +- Task 9.3 must guarantee stopped-runtime publication and perform fixed + recreation/health checks. Task 9.4 owns rollback. Task 9.5 owns installation, + host DB/path authority, systemd units, and permissions. Task 9.6 owns the full + crash/fault matrix. + +### Verification + +- Host apply tests passed 17/17 under WSL root with distinct root/runtime + ownership; native Windows passed with four expected POSIX-only skips. +- Operation tests passed 25/25, including exact claim/replay, mismatch, + accepted/predecessor corruption, valid unlinked predecessor, missing, and + terminal cases. +- Broad protocol 15, runtime-document 22, runtime-document I/O 25, admin 57, + Worker API 35, worker-runtime 12, and edge deployment suites passed. +- Container selection passed: 41 modules, 918 test definitions, no runner skips. +- Rebuilt `truf-worker-test:task-9-2`; a read-only, no-network container run as + UID 10001 passed 33 selected host-agent tests with one expected root-only + ownership-recovery skip; 1108 unrelated tests were deselected. +- Independent security/correctness review found no findings after root-owned + staging, displaced-inode detection, durable replay, and predecessor-hash fixes. +- Strict OpenSpec validation and `git diff --check` passed before artifact + closure. + +## Completed Task 9.3 + +Task 9.3 adds a fixed, fail-closed runtime/edge lifecycle and bounded aggregate +health verification. It remains deliberately unwired until rollback and +installation authority are complete. + +### Fixed lifecycle authority + +- `app/host_agent_lifecycle.py` accepts no production paths, services, commands, + environment, or timeouts. It uses only `/usr/bin/docker`, project + `truf-docker`, the two fixed Compose files, fixed edge environment, and the + `runtime` and `edge` services. +- Preflight validates the fixed Compose model, captures immutable runtime/edge + image IDs, resolves exactly one container per service, and inspects only a + bounded selected projection. It never reads container environment, full state, + health logs, or host mount source paths. +- Attestation binds IDs, images, users, entrypoint/command, runtime healthcheck, + Compose labels/config hash, capabilities/security options, read-only and + privileged state, namespaces, restart policy, resource limits, stop policy, + log driver, destination-only mounts, ports/tmpfs, and absence of device access. +- Edge stops first with a fixed 30-second grace; runtime then receives the fixed + 600-second coordinated stop. Each captured container must prove exited, PID 0, + exit 0, non-OOM, non-restarting/non-dead, and unchanged restart count. +- Immediately before publication, lifecycle re-resolves both service IDs, + revalidates their stopped state and immutable image tags, and issues a private + operation-bound stopped proof. `HostApplySession.replace()` rejects absent, + forged, or wrong-operation proof. +- The executor removes only the stopped edge and runtime containers, recreates + runtime with `--no-deps --no-build --pull never --force-recreate`, requires a + new exact identity, waits under one bounded deadline, and recreates edge only + after strict runtime health. Edge likewise requires a new exact identity, + fixed network namespace, successful fixed Caddy validation, and bounded stable + running observation. +- No `down`, kill, build/pull, volume/orphan, provision, systemd, fail2ban, + request-selected action, automatic recreation retry, rollback, or terminal + result reconciliation was added. + +### Document and execution ordering + +- `execute_fixed_forward()` keeps the task-9.2 singleton lock across backup, + deployment preflight, pre-stop CAS, stop proof, publication, recreation, and + health verification. +- `HostApplySession.revalidate_for_stop()` rechecks active/candidate identities + after backup/preflight and before downtime; `replace()` repeats validation + after clean stop and immediately before publication. +- Restart follows the same fixed stop/recreate/health sequence but does not back + up or publish documents. Complete publication replay recreates and verifies + once rather than treating file presence as runtime health evidence. + +### Bounded aggregate health + +- `container_runtime.py health --require-worker-api` preserves all existing + authenticated Supervisor ACTIVE, PostgreSQL READY/schema/cutover/PG16/storage, + and instance-bound ingester/projector lease checks, while requiring Worker API + to be explicitly enabled and running. +- Strict Worker API health sends a fixed non-secret invalid Bearer credential to + loopback `/api/v1/worker/claim` and requires the exact bounded 401 Bearer + response. Authentication performs the normal ScannerDB lookup before rejection, + proving the API-to-PostgreSQL path without a real credential or mutation. +- Ordinary Compose health remains unchanged; the strict flag is valid only for + health/status. Production cannot pass strict mode until task 9.5 installs the + intended enabled Worker API configuration. + +### Command and deadline hardening + +- Docker subprocesses use a fixed minimal environment, no shell, closed stdin, + suppressed stderr, a new process session, bounded 16 KiB stdout, and monotonic + phase deadlines. +- Timeout, output overflow, reader failure, cancellation, and a descendant that + retains stdout terminate only the owned process group with bounded waits. + Successful commands never signal a stale process-group ID, and the raw output + descriptor has single-owner close semantics. +- Stop/recreate/publication commands run at most once. Only bounded state/health + observation polls. + +### Verification + +- Lifecycle tests passed 18/18 on WSL, including real timeout, output-overflow, + retained-descendant, and successful-process cleanup behavior; native Windows + had four expected POSIX-only skips. +- Host apply tests passed 18/18 on WSL and protocol tests passed 15/15. The full + container-runtime suite passed with strict Worker API positive and hostile + response coverage. +- Broad operations 25, admin 57, Worker API 35, worker-runtime 12, + runtime-document 22, runtime-document I/O 25, and edge deployment suites + passed. +- Container selection passed: 42 modules, 943 test definitions, no runner + skips. +- Rebuilt `truf-worker-test:task-9-3`; a read-only, no-network container run as + UID 10001 passed 52 selected protocol/apply/lifecycle tests with one expected + root-ownership skip; 1119 unrelated tests were deselected. +- The selected real Docker inspect format was exercised locally and parsed all + 49 expected keys; a legacy container with an extra bind/import command was + correctly rejected. +- Independent security/correctness review found no findings after stopped-proof, + pre-stop CAS, attestation, Worker API DB-path, deadline, process-group, and FD + ownership fixes. +- Strict OpenSpec validation and `git diff --check` passed before artifact + closure. + +## Known Residual Boundaries + +- Apply cannot execute in production until crash verification from task 9.6 is + implemented. Returning 503 before operation creation remains correct when the + fixed host-agent socket or result authority is unavailable. +- A process crash exactly after an external side effect and before terminal DB + persistence is a distributed boundary. Stable operation IDs and idempotent + replay are the intended mitigation; do not add a general coordinator here. +- Worker API self-stop/restart can terminate the admin request before terminal + operation completion. Full external lifecycle reconciliation belongs to the + host operation architecture, not task 7.5/7.6. +- Managed log pages display bounded child output as written. Do not add a + post-hoc redaction pipeline without user approval. +- Candidate and active file handoff across privileged apply will be revalidated + by the host agent; task 7.6 must not implement active replacement itself. + +## Known Environment/Test Issues Unrelated to Current Work + +- Several broad Supervisor safety tests fail in this checkout because external + runtime authority fixture files such as `runtime/check-openrouter-keys.ps1` + are absent or legacy fixture config is incomplete. Focused clean runs exclude + only the documented affected classes; do not weaken production code for them. +- Bundled PostgreSQL integration tests skip when `initdb`/test DSN is not + available. +- Windows symlink tests may skip when the process lacks symlink privilege. +- The worktree has no useful commit baseline and contains many pre-existing + untracked/dirty files. Never perform broad resets or cleanup. +- Caddy 2.10.2 does not implement SIGUSR1 reload while production fail2ban + documentation assumes it. Edge E2E uses an admin-enabled test reload path. + This predates task 7.1 and needs a separate Caddy-upgrade/reload decision. +- Post-fail2ban-reload generic 404/401 responses in edge E2E do not reliably + retain global security headers. Authenticated admin pages and mutations still + enforce strict headers; the harness isolates the unrelated generic response + behavior. + +## Important Successful End-to-End Evidence + +- Packaged Windows/Linux protocol-2 E2E passed with GitLab plus synthetic + DockerHub/HuggingFace claim-to-ingestion. +- Full container E2E passed after strict runtime-document integration. +- Real edge E2E passed twice after rebuilding edge, fail2ban and runtime images; + final safe result proved authenticated Basic username attribution, private + marker acceptance, worker header stripping, worker availability, and fail2ban + behavior. +- Tasks 7.2–7.8 received desktop/mobile browser checks with real Starlette + routes and trusted headers; task 7.8 also has committed real-Chromium coverage. + +## Task 9.5 Completion + +- Added the fixed systemd socket/service, tmpfiles layout, root-only installer, + installed `/usr/lib/truf-host-agent` entrypoints, graceful host-agent shutdown, + PostgreSQL peer authority, async operation dispatch, and durable result + reconciliation. +- Runtime Compose now uses exact long-form mounts with host-path creation + disabled. Host config and secrets use the read-only `/etc/truf/runtime` to + `/data/config` authority, while immutable package manifests use the separate + root-owned `/etc/truf/worker-packages` to `/data/worker-packages` authority. + Candidates, result handoff, host-agent socket, and PostgreSQL socket mounts + retain their fixed access. Runtime-generated initialization state, lock, and + PostgreSQL password remain in the private writable `/data` volume. +- Lifecycle inspection validates exact bind sources and the named data volume. + The installer parses bounded rendered Compose JSON and independently requires + the same source/target/read-only/create-host-path projection. +- Package manifests are stable-read as root-owned `0644` files beneath a fixed + descriptor-walked root. Every descendant directory is root-owned and not + group/other writable; host and container readers consume byte-identical files. +- The installer validates the full root-owned application tree, installed bytes, + executables, Docker and agent sockets, units, tmpfiles, directories, and file + modes. Repeat install stops active socket/service units before replacement, + reloads systemd, enables the socket, and explicitly restarts it. Existing + active config and secrets are never overwritten. + +### Task 9.5 Verification + +- Native focused runtime-document/security/host deployment suites passed 67 + tests with seven platform skips; the wider host deployment/lifecycle focused + suites passed independently. +- WSL root stdlib suites passed 109/109. `systemd-analyze verify` accepted both + units; warnings were limited to Windows-mounted source-file permissions. +- Canonical combined base+edge Compose rendered JSON passed the exact installer + projection validator. Base Compose validation also passed under WSL Docker. +- The deny-by-default test image rebuilt successfully. Its UID 10001, read-only, + no-network run passed all task-9.5 host-agent/deployment selections. Four of + five initially failed broad tests passed alone after packaging `docker/verify.py`; + the remaining deep-JSON hostile-input assertion belongs to task 9.6. +- Container selection is AST-clean with 47 modules and 994 test definitions. +- Three independent review passes found no remaining task-9.5 blocker after + application-tree, mount-source, cleanup, manifest-root, Compose-projection, + and repeat-install fixes. + +## Task 9.6 Completion + +- Host-result JSON rejects nesting deeper than 64 before parsing, using an + iterative string-aware scanner so hostile depth is deterministic across Python + versions without recursive traversal. +- Candidate validation after a durable database claim now terminalizes as an + exact `failed/validation_failed` result before acceptance. A failed result + publication is replayable, and non-original prepared state is never rewritten. +- Crash-boundary coverage now proves interrupted combined-backup replay, retained + rollback evidence, restart-specific rollback, and durable rollback failure. +- Protocol coverage rejects deeply nested unknown request fields at the real + server boundary without invoking the privileged handler. + +### Task 9.6 Verification + +- Native task-matrix suites passed 106 tests with nine platform skips; the WSL + root stdlib matrix also exited successfully. +- Targeted UID 10001/Python 3.12 container tests passed 102 tests with two + expected root-identity skips. +- The full hardened, read-only, no-network container suite passed 1219 tests + with ten platform skips and one deprecation warning. +- Container selection is AST-clean with 47 modules and 1000 test definitions. +- Strict OpenSpec validation and diff checks passed. A final independent audit + found no remaining task-9.6 blocker. + +## Production Completion Status + +- The authorized production host now uses the exact root-selected + `shared-host-edge-v1` profile. Existing host Caddy remains the sole owner of + ports 80/443, the managed Truf edge binds only `127.0.0.1:18766`, and the + host agent has no lifecycle authority over host Caddy, X-UI, or another + unrelated service. +- Task 10.4 production execution is complete: protocol-2 package, runtime, + admin, edge, and agent checks passed; one reconciled restart succeeded; one + health-triggered automatic rollback restored the original documents and + reconciled as `rolled_back`; only then were production controls reopened. + +## Task 10.1 Completion + +- The production server image and lifecycle authority are remote-only and + Git-only; TruffleHog remains confined to worker and test images. +- Explicit code authority covers the runtime-document, managed-file, and + host-agent modules plus the core-profile launcher. +- Existing marked volumes gain only the fixed private `/data/managed-files` + directory; all other incomplete or unsafe layouts still fail closed. +- Production runbooks now require the fixed host-agent installer, protected + edge environment, active documents, package manifests, socket, and combined + runtime/edge Compose deployment in the correct order. +- Container selection includes the operations, discovery-only, core-profile, + multisource, and direct-source fixtures. The edge fixture advertises the + exact GitLab, DockerHub, and HuggingFace profile and real-Caddy coverage + traverses every new admin route plus operation detail. + +## Task 10.1 Verification + +- Focused native coverage passed 401 tests with eleven platform skips; the + final edge deployment suite passed 32 tests with one skip. +- Container selection is AST-clean with 54 modules and 1058 definitions. +- A hardened container run passed 1282 tests before fixture-only corrections; + all five corrected image/static/real-CLI cases then passed in the rebuilt + image. +- A built production runtime image was verified to contain executable Git and + no `/usr/local/bin/trufflehog`. +- Two independent reviews found no remaining task-10.1 blocker. Strict + OpenSpec validation and diff checks passed. + +## Task 10.2 Completion + +- The offline rollback test removes the protocol-2 worker and operations + authorities, applies the real stopped-runtime migration, and confirms all + additive tables are restored. +- A live remote reservation and pre-commit bundle prevent drain completion; + after both resolve, the authoritative state advances to `drained` with no + blockers. +- A raw previous-image enqueue contract then opens the same database and writes + using only legacy columns. The new control state, operation/audit evidence, + and protocol-2 tables remain intact. +- The complete operations-control suite passed 13 tests on Windows and 13 under + WSL. Container selection is AST-clean with 54 modules and 1059 definitions; + strict OpenSpec validation and diff checks passed. + +## Task 10.3 Completion + +- Focused unit, browser, packaged-worker, edge, PostgreSQL, and formal + verification gates completed without a remaining failure. +- The PostgreSQL fixture now supports an explicitly selected POSIX PostgreSQL + binary directory while preserving its existing bundled-Windows default. +- Real PostgreSQL execution exposed and closed one exact local-admission + recovery identity omission plus stale protocol-2 fixture limits and bundle + path evidence. + +## Task 10.3 Verification + +- The real PostgreSQL 16 integration suite passed all 60 tests as UID 10001 in + a read-only, no-network, capability-free container. Native Windows collected + the same 60 tests and skipped them because the optional bundled PostgreSQL is + absent. +- Packaged Windows and Linux worker verification passed after rebuilding the + protocol-2 artifacts. The real edge E2E passed with 28 baseline responses, + the exact core source profile, authenticated Caddy routes, and fail2ban + restart, unban, and expiry evidence. +- Browser coverage passed. The focused task-10.1 matrix passed 401 tests with + eleven platform skips, and the operations-control suite passed on Windows + and WSL. +- Container selection is AST-clean with 54 modules and 1059 definitions. + Strict OpenSpec validation and diff checks passed. + +## Task 10.4 Completion + +- The production cutover used `shared-host-edge-v1` while both explicit gates + were paused and drain was authoritative with zero blockers. Root-owned profile, + package, Compose projection, loopback edge, runtime, admin, agent, Caddy, and + X-UI service checks passed without granting the Truf lifecycle authority any + control over shared-host services. +- Reconciled restart operation `08dc7e8b-551f-5c64-b803-de905f040420` + completed `succeeded/succeeded` with no safe failure category, healthy runtime + and edge, and no failed hold. +- Reconciled rollback operation `6d09ea58-2c48-5366-b5ee-951c341fcd6f` + deliberately applied a valid candidate with Worker API disabled. Strict + forward health classified the failure as `health_check_failed`; the agent + restored the byte-identical original config and secrets exactly once and + completed `rolled_back/rolled_back` with no failed hold. +- Audited, revision-checked operations then canceled drain, resumed discovery, + and resumed dispatch in that order. Controls advanced from revision 19 to 22 + and are now discovery open, dispatch open, and drain `normal`. +- After reopening, the active production worker contacted the server and + received new DockerHub work. Lifecycle preflight, strict runtime health, and + host-agent/Caddy/X-UI service checks remained healthy. + +## Production Architecture Corrections + +- Real candidate validation found that private writable runtime configuration + could not also be the root-owned immutable package authority. Worker package + manifests now use the separate read-only `/data/worker-packages` authority, + while initialization, lock, and PostgreSQL password state use private files + directly beneath `/data`. +- PostgreSQL maintenance and offline migration use the discovery-only server + authority profile and therefore do not require the worker-only TruffleHog + executable. The offline cutover reconciled two stale leases, migrated the + production database, and established audited discovery and dispatch pauses. +- Production validation also corrected Worker API HTTP concurrency from one to + eight while retaining the real assignment cap of one. +- DockerHub discovery now configures account metadata without creating scanner + Docker directories or requiring scanner runtime initialization. Scanner + execution keeps the original full-runtime guard and Docker config behavior. + +## Task 10.4 Production Evidence + +- The active thin-v6 runtime image is + `sha256:dd55500ee2f947c0088a57061d1689025653fc0b08ed7fb22baea6816fb93270` + under tag `truf-local:runtime`; the managed edge image is + `sha256:6ec35cae4f4cf4bcd2bb1fe427b9cd2f16f8b5c2577442eae0941d806d362529` + under tag `truf-local:edge`. +- Canonical health reports Supervisor `ACTIVE`, PostgreSQL `READY`, and healthy + DockerHub, GitLab, HuggingFace, janitor, projector, ingester, and Worker API + processes. Lifecycle attestation confirms host-network runtime with no + published ports and a loopback-only edge. +- Root-owned protocol-2 Windows and Linux manifests advertise only GitLab, + DockerHub, and HuggingFace. Candidate initialization passed against the real + production volume before cutover. +- Public checks after rollback returned 401 for an invalid Worker API token, + 401 for the unauthenticated protected admin route, and 404 for an unrelated + Truf path. Host Caddy, X-UI, and the Truf host agent remained active. +- Shared-host restart validation exposed a real Docker Compose identity nuance: + `network_mode: service:runtime` changes the edge Compose config hash whenever + runtime is recreated. Lifecycle now attests all fixed edge metadata before + observing and pinning each forward or rollback edge hash; it never weakens the + fixed topology checks. +- Strict runtime-health command failures are normalized to lifecycle health + failures. This made the intentional rollback drill reconcile with the exact + `health_check_failed` safe category rather than generic `apply_failed`. +- Root-only evidence for two diagnostic failed holds remains preserved under + `/opt/truf-remote-server/staging/failed-hold-recovery-20260922` and + `/opt/truf-remote-server/staging/failed-hold-recovery-v2-20260922`. Both holds + were cleared only after strict health, identity, and evidence checks. +- The exact pre-v4 unit, previous runtime image and configuration, verified + custom-format PostgreSQL backup, failed-hold evidence, and reconciled operation + results remain available for rollback and audit. No X-UI or unrelated host + service was modified. + +### Task 10.4 Verification + +- The final focused host-agent matrix passed 107 tests with fourteen platform + skips. Edge deployment and container-runtime pytest coverage passed 221 tests + with one platform skip. +- Runtime-document coverage passed 22 tests and worker-package coverage passed + 15 tests. Changed lifecycle and drill scripts compile successfully. +- Both standalone base+edge and shared-host base+edge Compose projections passed + `config --quiet` with non-secret synthetic validation values; required edge + variables still fail closed when omitted. +- Strict OpenSpec validation and diff checks passed after production evidence + was reconciled and controls were reopened. + +## Task 10.5 Production Evidence + +- A single immutable DockerHub digest was processed independently by the + packaged Windows worker and the hardened WSL worker with assignment caps and + parallelism fixed at one. Both reservations produced accepted protocol-2 + bundles and acknowledged ingestion/projection records. +- A completed assignment replay returned the original accepted receipt without + rescanning. A separate claim-only assignment expired at its real wall-clock + deadline, refunded queue capacity and attempts, and replayed the same durable + expiry receipt over the public Worker API. +- An audited drain advanced from `draining` to `drained` with zero live remote + assignments and zero pre-commit bundles. Audited cancel restored `normal` + while both explicit gates remained paused. +- Temporary canary devices were revoked, their token file was overwritten and + removed, the normal 7200-second assignment TTL was restored, and no unresolved + canary assignment remained. + +## Task 10.6 Production Evidence + +- Only after the DockerHub canary gates passed, audited operations enabled + production discovery and dispatch and created one production WSL worker user + and device with an active assignment cap of one. +- Real production cycles completed successfully for GitLab with `gl_1`, public + HuggingFace with `hf_1`, and DockerHub with its real account pool. The final + DockerHub cycle reported eleven available accounts and exited with code zero. +- The active server and both protocol-2 compatibility profiles contain exactly + `gitlab`, `dockerhub`, and `huggingface`; GitHub remains excluded. +- Current controls are revision 22 with discovery and dispatch open and drain + state `normal`. The production WSL worker is running detached with + parallelism one; its audited user is enabled, its device is not revoked, and + it contacted the server after the task-10.4 rollback gates reopened. +- Rollback evidence retains the previous image, exact units, previous active + configuration, and verified database dump. The v4 unit and image identities + were checked before and after activation. + +## Managed Files Production Extension (2026-09-22) + +- Production Files now exposes exactly three configured logical roots: + `runtime-logs`, `runtime-keychecks`, and `runtime-results`. All are list/read + only. Logs and keychecks retain the 64 MiB per-file limit; only the fixed + result root permits 256 MiB files. +- The result root exposes only active `scan_results.jsonl` and + `found_secrets.jsonl` projections and their exact six-digit generations. + Internal locks, databases, ledgers, scan errors, malformed generations, + recovery files, and temporary/quarantine directories remain unavailable. +- The deployment paused discovery and dispatch and allowed the existing remote + DockerHub assignment to resolve naturally. Drain reached `drained` with zero + blockers before any runtime or configuration activation; no assignment was + cancelled. +- Runtime image `sha256:3b4d6e19e29e85a32b75d64265d75e71100d3f7f00221d21de544256dadd25cf` + was activated with a fail-safe Compose transition. The preceding image is + retained as `truf-local:runtime-pre-managed-files-v1`. +- Official restart operation `1978f770-c4a5-537d-b474-c3c9428cb79c` + completed `succeeded/succeeded`, reconciled without a safe category or failed + hold, and proved the full fixed lifecycle on the new immutable image before + the root configuration changed. +- Official Preview/Save operation `87b78ff6-d8eb-5fb4-a897-83b52c3697e5` + changed only the managed-root mapping. Apply operation + `7a572971-08d2-5e51-be78-4fb7039a93d4` completed + `succeeded/succeeded`, reconciled without rollback or failed hold. The active + configuration identity is + `e48797689ab0ee7328d21cf5c23d0ffedb979dba492bf49d1da3afe4b075cf5b`. +- Live runtime traversal verified all three exact policies, an empty keycheck + listing, two allowlisted active result projections, stable snapshot + length/hash, one-snapshot serialization, permit reuse, safe rejection of an + internal result artifact, and denied result mutation. +- Live Worker API HTML rendered all three root IDs and the read-only keycheck + and result pages. A permitted 146366-byte projection download matched its + exact source SHA-256 and length through the streaming route; an internal + result name returned 404. No file contents were emitted during validation. +- Final audited controls are revision 56 with discovery and dispatch open and + drain `normal`. Canonical Worker API plus discovery-producer strict health and + lifecycle preflight pass; failed hold is absent; host agent, host Caddy, and + X-UI are active. Public checks remain 401 for invalid Worker authentication, + 401 for unauthenticated admin access, and 404 for an unrelated path. diff --git a/openspec/changes/add-web-operations-control-plane/design.md b/openspec/changes/add-web-operations-control-plane/design.md new file mode 100644 index 0000000..bc39712 --- /dev/null +++ b/openspec/changes/add-web-operations-control-plane/design.md @@ -0,0 +1,155 @@ +## Context + +The current runtime combines provider discovery, queue admission, PostgreSQL claiming, and local scanner execution in the same `console_runner.py` source cycle. Supervisor can pause a child in memory, but that state is lost on restart and does not fence concurrent Worker API claims. The remote protocol supports GitHub and GitLab exact-Git assignments, while DockerHub and HuggingFace exist only as local scan paths. The typed admin UI can mutate worker users, devices, and queue rows, but routine runtime control and file changes still require SSH. + +The deployment is deliberately split across trust boundaries. Worker API/admin runs unprivileged in the read-only runtime container; either the standalone Truf edge or an existing root-owned host Caddy plus a loopback Truf edge owns public routing and injects a private edge marker; PostgreSQL is the durable queue authority; Supervisor exposes an authenticated loopback control protocol; systemd/Docker lifecycle control remains on the host. The exact ingress profile is root-installed policy, not runtime or request input. The design must retain those boundaries, keep uploads available during operational pauses, avoid a local server scanner, and never expose a shell or unrestricted host path. + +The default distributed core profile changes to exactly `gitlab`, `dockerhub`, and `huggingface`. Existing protocol-1 Git assignments may still be in flight when the new server is deployed, so result compatibility and migration ordering matter. + +## Goals / Non-Goals + +**Goals:** + +- Run provider discovery as server-side producer processes that only search, normalize, and enqueue. +- Persist and atomically enforce independent discovery pause, dispatch pause, and drain controls. +- Complete remote claim, scan, upload, ingestion, and projection for GitLab, DockerHub, and HuggingFace. +- Prove the first new-source path with a bounded public DockerHub immutable-digest canary. +- Provide typed, server-rendered admin operations for runtime state, controls, Supervisor, logs, configuration, plaintext secrets, managed files, and audit history. +- Apply configuration and secrets through validated candidates, backups, coordinated restart, health checks, and automatic rollback. +- Keep operator identity, operation status, and audit records durable across admin/runtime restarts. +- Support exact standalone and shared-host ingress profiles while preserving route confinement, loopback-only shared ingress, and the closed host-agent request schema. + +**Non-Goals:** + +- Local scanning or TruffleHog execution on the server. +- Arbitrary shell commands, arbitrary Supervisor command strings, Docker socket access, or browsing host root. +- PostgreSQL data-file access through the file page. +- Private Docker registry credential delivery or Docker layer-plan transport in the first rollout. +- Private HuggingFace credential delivery in the first canary; server-side discovery credentials remain supported. +- A client-side single-page application or storage of configuration/secrets in browser persistence. +- Replacing PostgreSQL queue, reservation, bundle, or projection authority. +- Managing, restarting, or reconfiguring unrelated host Caddy sites, X-UI, or other shared-host services. +- Runtime-selected topology, arbitrary ingress ports/upstreams, or wildcard/private-interface shared-edge binding. + +## Decisions + +### 1. Separate discovery into an explicit process role + +Supervisor will launch a `discovery-producer` role for each enabled core source instead of launching the existing combined source cycle. The role is part of the authenticated runtime bootstrap identity and is visible in structured Supervisor state. + +`console_runner.py` will expose a discovery-only cycle that performs provider requests, normalization, DockerHub retry/tag resolution where applicable, and idempotent queue admission. It will finish the source-cycle record with no scan requests and cannot call scan option preparation, scan-slot acquisition, target claiming, bundle staging, or scanner execution. Discovery runs according to its interval even when pending queue backlog exists. + +This is preferred over a mutable `enqueue_only` flag on the existing combined cycle because a distinct bootstrap role and call graph make accidental local scanner entry testable and fail-closed. The old combined path may remain for non-server/local workflows, but the server profile will not launch it. + +### 2. Make PostgreSQL the durable control authority + +An additive singleton control row will hold: + +- monotonically increasing `revision`; +- explicit `discovery_paused` and `dispatch_paused` flags; +- `drain_state` (`normal`, `draining`, or `drained`); +- actor, operation ID, and update timestamps. + +Mutations use compare-and-swap on the expected revision and append an audit event in the same transaction. Explicit pause flags remain independent; entering drain overlays both effective gates, and cancelling drain does not clear pauses that the operator set explicitly. + +Discovery producers check the effective discovery gate before provider I/O and enforce it again in the same transaction as discovery queue admission. Source retry and Docker tag-resolution claims are also gated. Generic enqueue functions used by ingestion and maintenance are not globally disabled. + +`reserve_and_claim_target()` enforces the effective dispatch gate inside its existing claim transaction. This closes the race between an API pre-check and target reservation. Authentication, status, terminal reports, assignment expiry, result upload, receipt replay, `mark_result_bundle_ready()`, and ingestion remain available while dispatch is paused or draining. + +Drain is complete when there are no live remote assignments and no accepted result bundles that have not reached database commit. Pending target backlog, projection work, and keycheck work do not prevent `drained`; those workers can safely resume after restart. A reconciler advances `draining` to `drained` from database state. + +This is preferred over stopping Worker API or Supervisor-only pause state because in-flight workers must retain their upload path and all enforcement must survive process restart. + +### 3. Generalize assignments through source adapters and protocol 2 + +The Git-specific assignment builder will become an adapter registry. Each adapter declares the canonical queue source, worker scan platform, planning kind, package capability, execution-snapshot validator, and claim/reconciliation behavior: + +| Queue source | Worker platform | Planning kind | Initial execution | +| --- | --- | --- | --- | +| `gitlab` | `gitlab` | `exact_git_v1` | Existing exact commit/snapshot path | +| `dockerhub` | `docker` | `docker_direct_v1` | Public immutable digest reference | +| `huggingface` | `huggingface` | `huggingface_space_v1` | Public Space identifier | + +The assignment API paths remain stable, but worker protocol/package compatibility advances to version 2 and manifests advertise explicit source/planning capabilities. Protocol-1 packages receive no new claims after cutover. Status, terminal report, upload, receipt replay, and immutable snapshot reconciliation remain available for already-issued protocol-1 assignments through their fixed expiry. + +Source selection will try other eligible configured sources when one queue has no claimable target instead of permanently choosing one source by request-ID modulo. Every successful claim still binds one fenced reservation, fixed expiry, device identity, immutable execution snapshot, and result-bundle identity. + +The first DockerHub canary uses an image resolved to an immutable digest and direct worker execution. Digest resolution is assignment planning, not proof that the worker can access the registry. The worker is the final access check and reports an inaccessible image using the bounded provider-failure result contract. The assignment does not serialize `DockerRegistryAuth`, process-local monotonic deadlines, server blob leases, or a Docker layer plan. The first HuggingFace canary follows the same worker-authoritative access model. Discovery drops Spaces that its existing provider response explicitly marks private, protected, gated, or disabled; the worker reports an inaccessible repository as non-retryable. Existing GitLab credential behavior remains, but provider discovery credentials are not assignment fields. + +Source adapters SHALL remain minimal. The server validates canonical target and assignment shape, performs only planning needed to identify the target, and leaves real provider access to the worker. A new per-target server access probe, durable public-access proof, proof freshness schema, broad child-environment credential scrubbing, credential sandbox, or post-hoc redaction pipeline is not implied by the credential non-transfer rule. Any such mechanism requires separate operator approval and an explicit OpenSpec requirement and task before implementation. Existing defensive code is not precedent for adding the same machinery to another source. + +### 4. Keep the admin interface typed and server-rendered + +`admin_api.py` will add explicit GET and POST routes for overview, search, dispatch/workers, Supervisor, logs, config, secrets, files, audit, and operation status. Forms retain exact field sets, bounded URL-encoded bodies, exact HTTPS Origin checks, CSRF, escaped output, CSP/HSTS/no-store headers, and POST/redirect/GET behavior. Unknown methods, route shapes, action names, source IDs, and file-root IDs fail closed. + +Caddy will strip any inbound operator header and inject the authenticated Basic-auth username alongside the existing trusted edge marker. The backend accepts the actor only with that marker and records it in control/audit rows. Plaintext secrets are rendered only in the dedicated no-store page; no JavaScript, local storage, or audit payload receives their values. + +The root-owned deployment profile is exactly `standalone-edge-v1` or `shared-host-edge-v1`. Standalone remains the default and owns host port 443. In shared-host mode, the existing host Caddy remains the sole owner of ports 80/443 and imports a fixed route-only snippet for only Worker API and the exact random admin prefix. It strips private/transit headers, injects an independent ingress marker, and proxies to the Truf edge at fixed loopback `127.0.0.1:18766`. The Truf edge rejects a missing marker before trusting the forwarded client address, binds only loopback, and retains Basic authentication, operator attribution, private backend marker, denylist, redacted logging, and security headers. It has no catch-all route for unrelated host applications. + +Admin will call new exact Supervisor actions for structured snapshot, one managed-source lifecycle action, and bounded log tail. Web input will never be forwarded to Supervisor's generic command parser. Long-running apply/restart operations return an operation ID and status page because the process serving the POST may be restarted. + +### 5. Share one strict configuration/secrets validator + +A side-effect-free validator will be used by preview, runtime startup, and the host operations agent. It will enforce bounded UTF-8 YAML, duplicate-key rejection, mapping roots, strict scalar types and bounds, known keys, the exact core profile, credential-pool entry schemas and unique names, reference integrity, package capabilities, and managed deployment paths. Validation errors identify fields but never echo secret values. + +Edits are candidate revisions, not direct active-file writes. Preview shows a structural/text diff with secret values redacted in audit and operation records. Candidate save and apply use expected SHA-256 hashes as compare-and-swap guards against stale forms or concurrent SSH changes. + +Configuration and secrets remain separate logical resources and are not exposed through the generic file browser. + +### 6. Use a narrow host operations agent for privileged lifecycle work + +A root-owned systemd socket/service will accept local requests from the runtime UID over a Unix socket. Its request schema contains only an operation UUID, one action enum (`apply-config`, `apply-secrets`, `apply-both`, or `restart`), and expected active/candidate hashes. It accepts no command, service name, path, environment, Compose argument, or shell text. + +Candidates live under a fixed host-managed bind directory shared read-only/read-write as required; active config/secrets will migrate from Docker-volume-only storage to fixed host-managed files before the agent is enabled. Immutable worker package manifests use a separate root-owned `/etc/truf/worker-packages` authority mapped read-only at `/data/worker-packages`; they never share the runtime-writable active-document trust root. Runtime-generated initialization state, lock, and PostgreSQL password remain in the private `/data` volume rather than the read-only active-document bind. The agent reads one root-owned exact profile and uses its fixed Compose files, network, ports, volume, capabilities, and Caddyfile. It never accepts that profile through its six-field request. The agent uses a singleton lock, revalidates candidate bytes, verifies all hashes, takes byte-identical backups, stops the fixed Truf runtime and edge, atomically replaces fixed files, recreates and attests that exact profile, and waits for Supervisor ACTIVE, PostgreSQL READY, Worker API, ingester, projector, and edge health. It never performs lifecycle actions on host Caddy, X-UI, or unrelated services. Failure restores backups and verifies the previous runtime. If both forward start and rollback fail, it enters a failed hold without deleting evidence or retrying indefinitely. + +The admin request and operation row are committed before the agent begins. The agent writes a bounded result envelope that the runtime reconciles into PostgreSQL after restart. The Docker socket is never mounted into the runtime container. + +### 7. Restrict managed files by logical root and descriptor-safe traversal + +The file page exposes configured logical roots for logs, backups, and selected result/export directories. The client submits a root ID plus canonical relative path, never an absolute root. Config, secrets, PostgreSQL storage, application code, sockets, host-agent metadata, and raw result bundles are excluded. + +Paths reject empty/absolute/drive-qualified/backslash/NUL/dot components and enforce byte, depth, listing, and file-size bounds. Linux traversal retains a root directory descriptor and uses component-wise `openat`/`dir_fd` operations with `O_NOFOLLOW`. Only single-link regular files are readable or replaceable; symlinks, hardlinks, reparse points, devices, FIFOs, and sockets are rejected. Writes use an exclusive same-directory temporary file, fsync, atomic replacement, directory fsync, and final owner/type/mode verification. + +Typed operations are limited to list, view/download, create/replace, and delete within roots that explicitly allow each action. Every mutation records actor, logical root/path, before/after hashes, byte counts, operation result, and timestamp, never file content. + +### 8. Persist operations and append-only audit records + +Additive PostgreSQL tables will store operation lifecycle and audit events. Operation rows contain typed action/target, requested/started/completed timestamps, safe status/category/detail, expected and resulting revisions/hashes, and host-agent reconciliation state. Audit events are append-only and include actor, operation ID, action, logical target, before/after identities, result, and a previous-event/hash-chain identity. + +Control mutation and its audit event commit atomically. File/config operation requests are audited when accepted and again when completed. Secret values, authorization headers, device tokens, provider tokens, CSRF values, and uploaded file bytes are forbidden from both schemas and logs. + +## Risks / Trade-offs + +- [A discovery code path accidentally reaches local scanning] -> Use a separate bootstrap role and call graph, omit TruffleHog from the server image/profile, and test that scan/claim functions are never invoked. +- [Pause races enqueue or claim] -> Enforce gates transactionally at queue admission and reservation, not only in UI or process state. +- [A runtime restart interrupts uploads] -> Drain to database-committed bundles before planned apply; preserve upload/status routes during pause; rely on fixed-expiry replay for network failures. +- [Protocol-2 rollout strands old work] -> Stop protocol-1 issuance first, retain its reconciliation/upload readers until no unresolved assignments remain, and only then remove compatibility in a later change. +- [Direct Docker scanning is less efficient than layer reuse] -> Accept the bandwidth cost for the first bounded canary; add layer-plan transport only after the simpler authority path is proven. +- [A source-specific access check grows into preventive server or worker security infrastructure] -> Keep provider access worker-authoritative and require separate operator approval plus an explicit requirement/task before adding probes, durable proofs, broad environment scrubbing, sandboxes, or redaction pipelines. +- [Plaintext secret editing exposes values to an operator browser] -> Require the existing protected admin boundary, no-store responses, no client persistence/scripts, bounded rendering, and value-free audit/log records. +- [The host agent becomes a root command proxy] -> Use a closed action enum and fixed paths/units, peer-credential checks, hash CAS, no shell, and adversarial request-schema tests. +- [Filesystem containment has TOCTOU or link attacks] -> Use retained directory descriptors and no-follow operations for every component; reject multi-link and non-regular files. +- [Rollback binary cannot read an additive schema] -> Keep migrations additive, preserve old markers, avoid incompatible constraint rewrites, and test old-image rollback before production cutover. +- [Shared-host ingress exposes a private listener] -> Require host networking only in the exact shared profile, no Docker-published ports, explicit loopback bind, an independent ingress marker, and metadata attestation before lifecycle work. +- [A Truf route captures or disrupts another host application] -> Install only a fixed route-only host-Caddy snippet with no listener, catch-all, global policy, or unrelated lifecycle authority. +- [Shared profile drift changes the trust boundary] -> Read one stable root-owned mode-0444 profile, reject unknown values and metadata, and attest exact Compose labels, mounts, network, ports, capabilities, and Caddyfile. +- [Server-rendered pages are less dynamic] -> Prefer explicit refresh/status pages over JavaScript to preserve the current CSP and reduce secret-retention surface. + +## Migration Plan + +1. Add and test the control/audit/operation schema, discovery role, source adapters, protocol-2 package support, typed admin routes, validator, file service, and host agent while all new controls remain disabled. +2. Build new server and Windows/Linux worker artifacts and verify code-authority/package manifests. +3. Set dispatch paused, stop discovery, and allow or expire all protocol-1 assignments while continuing to accept their uploads. +4. Stop runtime through the authenticated deployment path, create database and file backups, and run the additive migration under existing offline migration guards. +5. Migrate active config/secrets to the fixed host-managed bind directory; install the exact root-owned ingress profile and socket-activated agent; start the new runtime and Truf edge with discovery and dispatch paused. In shared-host mode, install and validate the fixed route-only snippet in the existing host Caddy without granting the agent authority over that service. +6. Verify health, admin actor attribution, control CAS/audit, Supervisor typed actions, managed-file containment, and rollback using a non-secret candidate. +7. Publish protocol-2 worker packages. Enable a low-cap DockerHub public immutable-digest canary and verify search, enqueue, claim, scan, upload, ingestion, projection, expiry/replay, and drain. +8. Enable GitLab and then public HuggingFace after the canary gates pass. Switch the default core profile exactly once and keep GitHub disabled. + +Rollback restores byte-identical config/secrets and the previous runtime/edge images while retaining additive database tables and audit evidence. Shared-host rollback does not modify or restart host Caddy or unrelated services. Rollback must not begin while protocol-2 DockerHub/HuggingFace assignments or pre-commit bundles are unresolved. If rollback health also fails, the agent leaves the Truf deployment stopped/held with backups intact for SSH recovery. + +## Open Questions + +- The retention duration and byte caps for browsable logs/backups/results need deployment defaults, but remain configurable within validator bounds. +- Private DockerHub and HuggingFace worker credential delivery is deferred. It must not be designed or implemented without separate operator approval and a dedicated change defining only the agreed delivery and failure semantics. +- Multi-operator authorization roles are deferred. This change records the Caddy Basic-auth username as actor but grants the existing admin policy uniformly. diff --git a/openspec/changes/add-web-operations-control-plane/proposal.md b/openspec/changes/add-web-operations-control-plane/proposal.md new file mode 100644 index 0000000..26a3056 --- /dev/null +++ b/openspec/changes/add-web-operations-control-plane/proposal.md @@ -0,0 +1,32 @@ +## Why + +The remote deployment can accept GitHub and GitLab worker assignments, but it cannot continuously discover targets without also entering the local scan path, and routine operation still requires SSH and direct file edits. The server needs a web-operated control plane that keeps discovery, dispatch, remote scanning, configuration, and runtime supervision separate and makes the intended distributed pipeline usable end to end. + +## What Changes + +- Add discovery-only server producers for the core source set `gitlab`, `dockerhub`, and `huggingface`; they may search and enqueue targets but must never claim or scan them locally. +- Add persistent controls for pausing discovery, pausing new assignment dispatch, and draining the system while continuing to accept uploads for existing assignments. +- Extend remote assignments, worker packages, scan execution, and result acceptance to support full GitLab, DockerHub, and HuggingFace claim-to-ingestion cycles, with DockerHub as the first deployed end-to-end canary. +- Add authenticated admin pages for overview/search controls, workers and dispatch, Supervisor status/commands/logs, runtime configuration, plaintext `secrets.yaml`, managed files, and an operation audit trail. +- Apply configuration and secret changes as validated, backed-up operations with coordinated runtime restart, health verification, and automatic rollback rather than in-place live mutation. +- Support two exact root-selected production ingress profiles: the standalone Truf edge and a shared-host edge behind an existing root-owned host Caddy, without adding topology, port, path, service, or command fields to the host-agent request. +- Expose only explicitly managed project directories through the file page; host root, PostgreSQL data, Docker control sockets, and arbitrary shell execution remain outside the web interface. +- **BREAKING** Replace GitHub in the default core source set with HuggingFace; the new default core set is exactly GitLab, DockerHub, and HuggingFace. +- **BREAKING** Advance worker compatibility so packages that support only the current GitHub/GitLab assignment contract are not eligible for the new core-source profile and must be rebuilt. + +## Capabilities + +### New Capabilities + +- `distributed-core-source-processing`: Discovery-only production, persistent discovery/dispatch/drain controls, and remote-only processing for the configured core sources. +- `multisource-worker-assignments`: Compatible worker packaging and fenced assignment/result lifecycles for GitLab, DockerHub, and HuggingFace. +- `web-operations-console`: Authenticated web views and mutations for runtime overview, search, dispatch, workers, Supervisor, logs, and audited operations. +- `managed-runtime-editing`: Validated editing and coordinated application of runtime configuration, plaintext secrets, and allowlisted project files with backup and rollback. + +### Modified Capabilities + +None. + +## Impact + +The change affects source-cycle separation in `app/console_runner.py` and `app/supervisor.py`; queue and control state in `app/scanner_db.py`; worker assignment, package, client, scan execution, and API modules; the typed admin API and its HTML/CSS; standalone and shared-host Caddy/Compose profiles; profile-specific host lifecycle, installer, and denylist validation; runtime configuration and worker package manifests; and focused unit, integration, browser, and deployment tests. Existing queue and result authority remains PostgreSQL-backed, existing uploads remain accepted during drain, and no host filesystem or generic shell API is introduced. diff --git a/openspec/changes/add-web-operations-control-plane/specs/distributed-core-source-processing/spec.md b/openspec/changes/add-web-operations-control-plane/specs/distributed-core-source-processing/spec.md new file mode 100644 index 0000000..c0afa13 --- /dev/null +++ b/openspec/changes/add-web-operations-control-plane/specs/distributed-core-source-processing/spec.md @@ -0,0 +1,102 @@ +## ADDED Requirements + +### Requirement: Exact distributed core source set +The server core profile SHALL contain exactly `gitlab`, `dockerhub`, and `huggingface`, and SHALL NOT start GitHub or any other discovery source as part of that profile. + +#### Scenario: Core profile starts +- **WHEN** Supervisor starts the distributed core profile +- **THEN** it starts one discovery producer for GitLab, DockerHub, and HuggingFace and no GitHub producer + +### Requirement: Discovery-only producer isolation +Each core discovery producer SHALL search its provider, normalize targets, and admit them to PostgreSQL without claiming targets, acquiring scan slots, invoking a scanner, or staging result bundles locally. + +#### Scenario: GitLab discovery finds targets +- **WHEN** the GitLab producer completes a provider search +- **THEN** it enqueues the normalized targets and records zero local scan requests + +#### Scenario: DockerHub discovery resolves targets +- **WHEN** the DockerHub producer processes pages, retries, tags, or digests +- **THEN** it may persist discovery progress and immutable targets but never invokes TruffleHog or reserves those targets locally + +#### Scenario: HuggingFace discovery finds Spaces +- **WHEN** the HuggingFace producer returns Space identifiers +- **THEN** it drops records explicitly marked private, protected, gated, or disabled and enqueues the remaining identifiers without entering a local scan path or making a second per-Space verification request + +#### Scenario: Pending backlog exists +- **WHEN** a scheduled discovery interval arrives while pending targets already exist +- **THEN** the producer still performs the configured discovery cycle unless discovery is paused + +### Requirement: Durable operations control state +The system SHALL persist discovery pause, dispatch pause, drain state, revision, actor, operation identity, and update timestamps in PostgreSQL so that control state survives process and host restarts. + +#### Scenario: Runtime restarts while paused +- **WHEN** discovery and dispatch are paused and the runtime restarts +- **THEN** both effective gates remain paused after startup + +#### Scenario: Stale control form is submitted +- **WHEN** a mutation supplies a revision older than the current control revision +- **THEN** the system rejects it without changing control state or writing a success audit event + +#### Scenario: Explicit pause coexists with drain +- **WHEN** an operator explicitly pauses discovery, enters drain, and later cancels drain +- **THEN** the explicit discovery pause remains set + +### Requirement: Transactional discovery admission gate +The system SHALL enforce the effective discovery gate in the same transaction that admits discovered targets or claims discovery-specific retry work. + +#### Scenario: Pause races target admission +- **WHEN** discovery pause commits before a producer admission transaction commits +- **THEN** no newly discovered target is admitted by that transaction + +#### Scenario: Discovery is paused before provider request +- **WHEN** a producer begins a cycle while discovery is effectively paused +- **THEN** it performs no provider request and records a paused cycle outcome + +#### Scenario: Upload-derived work arrives during pause +- **WHEN** result ingestion creates projection or keycheck work while discovery is paused +- **THEN** that work remains admissible because it is not provider discovery + +### Requirement: Transactional dispatch gate +The system SHALL enforce the effective dispatch gate within the reservation transaction so that no new remote assignment can be issued after dispatch pause or drain commits. + +#### Scenario: Dispatch pause races claim +- **WHEN** dispatch pause commits before a worker claim transaction commits +- **THEN** the claim returns a paused or no-work response and creates no reservation + +#### Scenario: Existing worker uploads while paused +- **WHEN** dispatch is paused and a worker with an existing assignment reports status or uploads its result +- **THEN** the server accepts the valid request under the existing assignment authority + +#### Scenario: Assignment expires while paused +- **WHEN** an existing assignment expires during dispatch pause +- **THEN** the reaper processes it normally without issuing replacement work + +### Requirement: Drain lifecycle +Entering drain SHALL effectively pause discovery and dispatch while preserving status, terminal report, upload, receipt replay, ingestion, projection, and maintenance paths needed to finish accepted work. + +#### Scenario: Drain begins with active assignments +- **WHEN** drain is requested while remote assignments are active +- **THEN** the state becomes `draining`, no new targets or assignments are admitted, and existing workers retain their result path + +#### Scenario: Drain reaches completion +- **WHEN** no live remote assignments remain and every accepted result bundle has reached database commit +- **THEN** the reconciler advances the state to `drained` + +#### Scenario: Pending targets remain +- **WHEN** pending queue targets remain but all issued assignments and pre-commit bundles are resolved +- **THEN** drain may still become `drained` + +#### Scenario: Projection work remains +- **WHEN** projection or keycheck work remains after its result bundle is database-committed +- **THEN** that work does not prevent the control state from becoming `drained` + +### Requirement: Discovery process observability +Supervisor and the operations console SHALL expose each discovery producer's source, role, lifecycle state, last cycle result, last successful discovery time, next scheduled run, and bounded safe error category. + +#### Scenario: Provider rejects credentials +- **WHEN** a discovery producer receives a provider authorization error +- **THEN** operations state reports the source and safe authorization category without exposing the credential or provider response body containing secrets + +#### Scenario: Producer is stopped +- **WHEN** an operator stops a managed producer through a typed Supervisor action +- **THEN** structured state identifies it as stopped without changing the persisted discovery pause flag diff --git a/openspec/changes/add-web-operations-control-plane/specs/managed-runtime-editing/spec.md b/openspec/changes/add-web-operations-control-plane/specs/managed-runtime-editing/spec.md new file mode 100644 index 0000000..a946cdf --- /dev/null +++ b/openspec/changes/add-web-operations-control-plane/specs/managed-runtime-editing/spec.md @@ -0,0 +1,182 @@ +## ADDED Requirements + +### Requirement: Shared strict document validation +Configuration and secrets preview, startup, and privileged apply SHALL use the same side-effect-free validator for bounded UTF-8 YAML, duplicate keys, mapping roots, strict scalar types and bounds, known keys, exact core profile, auth-pool schemas, unique entry names, reference integrity, package capabilities, and deployment paths. + +#### Scenario: Valid configuration is previewed +- **WHEN** an operator submits a candidate satisfying the complete schema +- **THEN** preview returns normalized validation success and a bounded diff without activating the candidate + +#### Scenario: Duplicate or unknown key is submitted +- **WHEN** a candidate contains a duplicate mapping key or unsupported field +- **THEN** validation fails before staging or restart and identifies the field without echoing secret values + +#### Scenario: Secret reference is invalid +- **WHEN** configuration selects an auth-pool entry that does not exist in the candidate secrets +- **THEN** combined validation fails and neither document becomes active + +### Requirement: Separate immutable package authority +Worker package manifests SHALL resolve only beneath a separate root-owned, non-writable package authority, while runtime-generated initialization state, locks, and PostgreSQL credentials SHALL remain in the private writable data volume rather than the active config/secrets bind. + +#### Scenario: Package manifest uses the active-document directory +- **WHEN** configuration references a package manifest beneath the config/secrets authority or outside the fixed package root +- **THEN** validation fails before startup or privileged apply + +#### Scenario: Runtime initializes with active documents mounted read-only +- **WHEN** the runtime initializes or restarts with the active config/secrets directory mounted read-only +- **THEN** its initialization marker, singleton lock, and generated PostgreSQL password remain writable only through fixed private data paths + +### Requirement: Candidate revisions and compare-and-swap +The system SHALL stage validated candidate revisions separately from active files and SHALL require expected active and candidate SHA-256 identities when saving or applying them. + +#### Scenario: Candidate is saved +- **WHEN** an operator saves valid bytes against the current active hash +- **THEN** the candidate is durably staged with a new hash and the active file is unchanged + +#### Scenario: Active file changed through SSH +- **WHEN** the active hash differs from the expected hash submitted by a stale page +- **THEN** save or apply fails without replacing either active file + +#### Scenario: Candidate changed concurrently +- **WHEN** the supplied candidate hash is no longer current +- **THEN** apply is rejected before host lifecycle changes begin + +### Requirement: Plaintext secrets editing without persistence leakage +The protected secrets page SHALL allow authorized operators to view and edit the complete plaintext YAML document while responses remain no-store and secret values remain absent from browser persistence, application logs, diffs outside that page, operation status, and audit records. + +#### Scenario: Operator opens secrets page +- **WHEN** an authenticated operator requests the dedicated secrets editor +- **THEN** the current document is rendered in a server-side form over the protected no-store response + +#### Scenario: Secrets candidate is validated +- **WHEN** an operator previews or saves changed secrets +- **THEN** validation results name safe field paths and hashes but do not repeat credential values + +#### Scenario: Operator leaves the page +- **WHEN** the browser navigates to another admin page +- **THEN** the application has written no secret value to local storage, session storage, service workers, or client-side application state + +### Requirement: Closed host-agent protocol +The privileged host operations agent SHALL accept only a canonical operation UUID, one fixed action enum, expected active hashes, and expected candidate hashes from the authorized runtime peer over a local Unix socket. Deployment profile is root-installed host policy and SHALL NOT be added to that request. + +#### Scenario: Valid apply request arrives +- **WHEN** the authorized runtime UID submits an exact valid request +- **THEN** the agent verifies the persisted operation and hashes before acquiring the singleton apply lock + +#### Scenario: Request includes path or command data +- **WHEN** a request includes a service name, path, shell text, environment, Docker argument, or unknown field +- **THEN** the agent rejects it before any privileged action + +#### Scenario: Request attempts topology selection +- **WHEN** a request includes a profile, ingress port, upstream, Caddy path, unit, service, command, environment, Compose file, or Compose argument +- **THEN** the agent rejects it before reading candidate documents or stopping the deployment + +#### Scenario: Unauthorized local peer connects +- **WHEN** a process with an unapproved peer identity uses the socket +- **THEN** the agent rejects the request regardless of its JSON body + +### Requirement: Coordinated apply with health verification +For config, secrets, or combined apply, the host agent SHALL revalidate fixed candidate files, verify compare-and-swap hashes, create byte-identical backups, stop the fixed Truf deployment, atomically replace active files, recreate the exact root-selected runtime and edge profile, attest its fixed topology, and wait for required health before reporting success. In shared-host mode it SHALL NOT stop, restart, reload, reconfigure, or remove host Caddy, X-UI, or another unrelated service. + +#### Scenario: Combined apply succeeds +- **WHEN** both candidates validate and the restarted deployment reaches Supervisor ACTIVE, PostgreSQL READY, Worker API, ingester, and projector health +- **THEN** the operation completes successfully with resulting hashes and retained rollback evidence + +#### Scenario: Validation changes between preview and apply +- **WHEN** host-side revalidation or hash verification differs from the accepted candidate operation +- **THEN** the agent aborts before stopping the healthy runtime + +#### Scenario: Apply is requested while another is active +- **WHEN** the singleton operation lock is held +- **THEN** the second request is rejected or remains queued without overlapping lifecycle mutations + +#### Scenario: Shared-host profile is healthy +- **WHEN** shared-host apply recreates a host-network runtime with no published ports and an edge bound only to fixed loopback, and all runtime and edge health checks pass +- **THEN** the operation succeeds without a host-Caddy or X-UI lifecycle action + +#### Scenario: Shared-host topology drifts +- **WHEN** runtime publishes a port, edge binds a non-loopback address, profile metadata changes, or Compose labels, mounts, network, capabilities, or Caddyfile differ from the fixed profile +- **THEN** the agent rejects the deployment before candidate replacement or reports failed health without claiming success + +#### Scenario: Shared-host rollback succeeds +- **WHEN** candidate health fails and the previous Truf runtime and edge are restored +- **THEN** rollback completes exactly once without changing host Caddy, X-UI, or another unrelated service + +### Requirement: Automatic rollback and failed hold +If the new deployment fails its bounded health check, the agent SHALL restore byte-identical backups and verify the previous deployment; if rollback also fails, it SHALL stop retrying and retain a failed-hold state and all evidence for SSH recovery. + +#### Scenario: New configuration fails startup +- **WHEN** the restarted runtime cannot reach required health within the deadline +- **THEN** the agent restores the prior active files and restarts the previous deployment + +#### Scenario: Rollback succeeds +- **WHEN** the restored deployment reaches required health +- **THEN** the operation records rolled-back status and safe failure category without claiming apply success + +#### Scenario: Rollback fails +- **WHEN** neither the candidate nor restored deployment becomes healthy +- **THEN** the agent enters failed hold, performs no replacement loop or forced authority release, and preserves backups and diagnostics + +### Requirement: Logical managed roots +The generic file page SHALL address only configured logical roots with explicit read, create/replace, and delete permissions, and SHALL never accept an absolute root from a client. + +#### Scenario: Operator lists a managed runtime root +- **WHEN** a valid logical root ID for logs, keycheck projections, or result projections and a canonical relative directory are requested +- **THEN** the service returns a bounded read-only listing of permitted regular files and directories under that root + +#### Scenario: Operator downloads rotated result projections +- **WHEN** the operator requests a permitted active or rotated result projection within its configured file-size bound +- **THEN** the service returns that regular single-link file without granting mutation access or exposing the backing runtime path + +#### Scenario: Result projection names are allowlisted +- **WHEN** the result-projection root is listed or read +- **THEN** only active `scan_results.jsonl` and `found_secrets.jsonl` files and their exact six-digit generation names are visible, while locks, databases, ledgers, scan errors, temporary/quarantine directories, malformed generations, and recovery artifacts remain unavailable + +#### Scenario: Large result projection is downloaded +- **WHEN** an allowed result projection is within the larger result-root byte bound +- **THEN** the service copies and hashes an unchanged source revision into an anonymous same-volume snapshot using bounded chunks, permits only one such snapshot at a time, streams the snapshot with bounded memory, and releases the snapshot and concurrency slot after response completion or failure + +#### Scenario: Operator requests excluded storage +- **WHEN** a request targets application code, config/secrets through the generic page, PostgreSQL storage, sockets, host-agent metadata, raw result bundles, or an unknown root +- **THEN** the service rejects it without revealing host paths or existence details + +### Requirement: Descriptor-safe path containment +Managed-file traversal and mutation SHALL use a retained root directory descriptor, component-wise no-follow operations, canonical relative components, and regular single-link file checks. + +#### Scenario: Relative traversal is attempted +- **WHEN** a path contains an empty, dot, dot-dot, absolute, drive-qualified, backslash, NUL, over-depth, or over-length component +- **THEN** the request is rejected before filesystem access outside the retained root + +#### Scenario: Symlink is swapped during access +- **WHEN** a path component becomes a symlink between validation and open +- **THEN** no-follow descriptor traversal fails without accessing the link target + +#### Scenario: Hardlink or special file is targeted +- **WHEN** the final object is multi-linked or is not a regular file +- **THEN** view, download, replace, and delete are rejected + +### Requirement: Durable bounded file mutation +An allowed managed-file create or replace SHALL use an exclusive same-directory temporary regular file, bounded bytes, fsync, atomic descriptor-relative replacement, directory fsync, and final ownership/type/mode verification. + +#### Scenario: File replacement succeeds +- **WHEN** an authorized bounded replacement is submitted against the current file hash +- **THEN** readers observe either the complete old file or complete new file and audit records the safe before/after hashes + +#### Scenario: File is concurrently changed +- **WHEN** the current file hash differs from the expected hash +- **THEN** replacement fails without overwriting the concurrent change + +#### Scenario: Upload exceeds the root limit +- **WHEN** submitted bytes exceed the configured bounded file size +- **THEN** the service rejects and removes temporary data without changing the target + +### Requirement: Content-free operational audit +Configuration, secrets, and managed-file operations SHALL record actor, typed action, logical target, timestamps, result, safe category, byte counts where applicable, and before/after hashes, but SHALL NOT record file contents or credentials. + +#### Scenario: Managed file is deleted +- **WHEN** an allowed delete succeeds against the expected hash +- **THEN** audit records the logical root/path and previous hash without retaining deleted content + +#### Scenario: Secrets apply fails +- **WHEN** a secrets operation fails validation, startup, or rollback +- **THEN** status and audit expose only the bounded failure category and document hashes diff --git a/openspec/changes/add-web-operations-control-plane/specs/multisource-worker-assignments/spec.md b/openspec/changes/add-web-operations-control-plane/specs/multisource-worker-assignments/spec.md new file mode 100644 index 0000000..3498d32 --- /dev/null +++ b/openspec/changes/add-web-operations-control-plane/specs/multisource-worker-assignments/spec.md @@ -0,0 +1,120 @@ +## ADDED Requirements + +### Requirement: Protocol 2 source capabilities +Protocol-2 worker packages SHALL advertise explicit source, worker-platform, and planning-kind capabilities for GitLab, DockerHub, and HuggingFace, and the server SHALL issue work only when the selected package supports the complete assignment capability. + +#### Scenario: Compatible package requests work +- **WHEN** a protocol-2 package advertising the required capability requests a supported target +- **THEN** the server may create an assignment using that capability + +#### Scenario: Package lacks planning capability +- **WHEN** a package advertises the source but not the required planning kind +- **THEN** the server rejects the claim without reserving a target + +#### Scenario: Package advertises unknown capability +- **WHEN** a package manifest contains an unknown source, platform, or planning kind +- **THEN** package validation fails closed + +### Requirement: Canonical multisource execution plans +The server SHALL create and validate immutable execution snapshots using `exact_git_v1` for GitLab, `docker_direct_v1` for DockerHub, and `huggingface_space_v1` for HuggingFace. + +#### Scenario: GitLab assignment is issued +- **WHEN** a GitLab target is claimed +- **THEN** the assignment binds the existing exact commit and Git scan plan under `exact_git_v1` + +#### Scenario: DockerHub assignment is issued +- **WHEN** a public DockerHub target is claimed +- **THEN** the assignment binds an immutable digest reference under `docker_direct_v1` and does not depend on a mutable tag + +#### Scenario: HuggingFace assignment is issued +- **WHEN** a public HuggingFace Space is claimed +- **THEN** the assignment binds its canonical Space identifier under `huggingface_space_v1` + +#### Scenario: Snapshot shape does not match source +- **WHEN** an execution snapshot's source, worker platform, or planning kind combination is invalid +- **THEN** the server and worker reject it before scanner execution + +### Requirement: Fenced assignment authority for every source +Every supported source assignment SHALL bind one user, device, target, fixed expiry, immutable execution snapshot, result reservation, and result bundle identity using the existing PostgreSQL authority model. + +#### Scenario: Lost claim response is retried +- **WHEN** the server committed an assignment but the worker did not receive the response +- **THEN** retrying the same admission request returns the same assignment and immutable execution snapshot + +#### Scenario: Stale worker uploads +- **WHEN** a worker uploads with an expired, replaced, or mismatched reservation token +- **THEN** the server rejects the upload without changing queue or bundle authority + +#### Scenario: Valid result commits +- **WHEN** a valid assigned worker uploads and finalizes its bundle +- **THEN** ingestion commits the queue result and downstream projection work exactly once + +### Requirement: Eligible source fallback +The assignment service SHALL try other eligible configured sources when one supported source has no claimable target, while still issuing at most one assignment for an admission request. + +#### Scenario: Initially selected source is empty +- **WHEN** the first eligible source has no claimable target and another eligible source does +- **THEN** the same claim request may receive one assignment from the other source + +#### Scenario: All eligible sources are empty +- **WHEN** no compatible source has a claimable target +- **THEN** the claim returns no work and creates no reservation + +### Requirement: Legacy protocol-1 completion compatibility +After protocol-2 cutover, the server SHALL stop issuing new claims to protocol-1 packages but SHALL continue status, terminal report, upload, receipt replay, and immutable snapshot reconciliation for already-issued protocol-1 assignments until they resolve or expire. + +#### Scenario: Protocol-1 package requests a new claim +- **WHEN** a legacy GitHub/GitLab-only package requests new work after cutover +- **THEN** the server returns an incompatibility response and creates no assignment + +#### Scenario: Existing protocol-1 assignment uploads +- **WHEN** a legacy worker uploads a valid result for an assignment issued before cutover +- **THEN** the server accepts and ingests it under its original immutable authority + +#### Scenario: Legacy snapshot is reconciled +- **WHEN** the server reconstructs a lost response for an existing protocol-1 assignment +- **THEN** it reads the original snapshot without rewriting it into protocol 2 + +### Requirement: DockerHub end-to-end canary +The rollout SHALL prove a bounded DockerHub `search -> enqueue -> claim -> scan -> upload -> ingestion -> projection` cycle using a public image resolved to an immutable digest before broader new-source enablement. + +#### Scenario: DockerHub canary succeeds +- **WHEN** a canary producer discovers the configured public image and a compatible worker processes it +- **THEN** the target reaches database-committed ingestion and projection under one fenced assignment + +#### Scenario: Mutable tag changes during canary +- **WHEN** the discovered tag changes after queue admission +- **THEN** the worker still scans the immutable digest bound in its assignment + +#### Scenario: Registry credentials would be required +- **WHEN** the DockerHub canary target cannot be scanned without private registry credentials +- **THEN** the worker returns a bounded inaccessible-provider result, the server applies its declared retryability, and no discovery or registry credential is transferred in the assignment + +### Requirement: HuggingFace remote processing +The system SHALL support the same fenced claim-to-ingestion lifecycle for public HuggingFace Spaces without invoking the scanner on the server. + +#### Scenario: Public Space completes +- **WHEN** a compatible worker claims and scans a public HuggingFace Space +- **THEN** its result is uploaded, ingested, and projected under the bound assignment + +#### Scenario: Space is inaccessible without worker credentials +- **WHEN** a tokenless worker cannot read a claimed HuggingFace Space because its repository is private, protected, removed, or otherwise unavailable +- **THEN** it returns a non-retryable inaccessible result, the server does not retry that target, and the server discovery token is never exposed + +### Requirement: Worker-authoritative provider access +The server SHALL validate canonical target and assignment authority but SHALL treat worker execution as the final provider-access check. A source SHALL NOT require a per-target server access probe, durable public-access proof, proof-freshness state, broad child-environment credential scrubbing, credential sandbox, or post-hoc redaction pipeline unless the operator separately approves an explicit OpenSpec requirement and implementation task. + +#### Scenario: Provider accessibility changes after discovery +- **WHEN** a canonical target becomes inaccessible before worker execution +- **THEN** the worker returns the source's bounded permanent or retryable provider-failure result and the server settles or retries it according to that result + +#### Scenario: Another source adapter is proposed +- **WHEN** implementation would add preventive access proof or source-specific security infrastructure beyond the assignment's declared fields +- **THEN** implementation pauses until the operator approves a dedicated requirement and task + +### Requirement: Credential and result secrecy +Provider credentials, worker device tokens, authorization headers, and result contents SHALL NOT appear in operation status, audit records, Supervisor snapshots, or routine assignment logs. + +#### Scenario: Assignment logging occurs +- **WHEN** any supported source assignment is created, retried, rejected, or completed +- **THEN** logs identify bounded source and authority metadata without credential values or result payload bytes diff --git a/openspec/changes/add-web-operations-control-plane/specs/web-operations-console/spec.md b/openspec/changes/add-web-operations-control-plane/specs/web-operations-console/spec.md new file mode 100644 index 0000000..36eb720 --- /dev/null +++ b/openspec/changes/add-web-operations-control-plane/specs/web-operations-console/spec.md @@ -0,0 +1,142 @@ +## ADDED Requirements + +### Requirement: Protected typed admin routes +The operations console SHALL expose explicit server-rendered routes and exact mutation forms behind the existing random admin path, Caddy Basic authentication, trusted edge marker, exact same-origin check, CSRF validation, no-store responses, and restrictive security headers. + +#### Scenario: Authorized operator opens a page +- **WHEN** Caddy authenticates the request and injects the trusted marker and operator identity +- **THEN** the requested operations page renders escaped server-side HTML with no client-side secret persistence + +#### Scenario: Direct backend request lacks marker +- **WHEN** a request reaches an admin route without the trusted edge marker +- **THEN** the backend rejects it regardless of supplied operator headers + +#### Scenario: Mutation has stale or invalid CSRF +- **WHEN** a POST has a missing, duplicate, or invalid CSRF value or wrong Origin +- **THEN** the backend rejects the mutation without side effects + +#### Scenario: Unknown route or form action is submitted +- **WHEN** a request contains an unsupported method, route shape, action, field, or duplicate field +- **THEN** it fails closed without invoking Supervisor, database mutations, or host operations + +### Requirement: Trusted operator attribution +Caddy SHALL strip any inbound operator identity header and inject the authenticated Basic-auth username, and the backend SHALL trust that identity only with the private edge marker. + +#### Scenario: Client spoofs operator header +- **WHEN** a public request supplies its own operator identity header +- **THEN** Caddy removes it and the audit actor is the authenticated Basic-auth user + +#### Scenario: Mutation is accepted +- **WHEN** an authenticated operator performs a valid mutation +- **THEN** the control or operation record and its audit event identify that operator + +### Requirement: Exact production ingress profiles +The production deployment SHALL use exactly one root-installed profile: `standalone-edge-v1` or `shared-host-edge-v1`. The selected profile SHALL NOT be supplied by an admin request, host-agent request, runtime document, or other unprivileged input. + +#### Scenario: Standalone edge is selected +- **WHEN** `standalone-edge-v1` is installed +- **THEN** the managed Truf edge remains the sole Truf listener on host port 443 and retains the exact runtime-network-namespace contract + +#### Scenario: Shared-host edge is selected +- **WHEN** `shared-host-edge-v1` is installed +- **THEN** the existing root-owned host Caddy remains the sole owner of ports 80/443 and proxies only fixed Truf routes to a managed edge bound at `127.0.0.1:18766` + +#### Scenario: A request attempts to select topology +- **WHEN** a request supplies a deployment mode, upstream, port, Caddy path, unit, service, command, or Compose argument +- **THEN** it is rejected before lifecycle work + +### Requirement: Shared-host route confinement +The shared-host profile SHALL install a fixed root-owned route-only host-Caddy snippet. It SHALL claim only `/api/v1/worker/*`, the exact random admin-prefix root, and that prefix's subtree. It SHALL strip inbound private and transit headers, inject an independent ingress marker and canonical client address, and preserve the managed edge's authentication, operator attribution, denylist, redacted logging, and security-header behavior without adding a listener, TLS policy, global error handler, trusted-proxy policy, catch-all, or unrelated route. + +#### Scenario: An unrelated host route is requested +- **WHEN** a request does not match a Truf worker or admin path +- **THEN** the Truf snippet does not handle or alter the request + +#### Scenario: The loopback Truf edge is unavailable +- **WHEN** a matching route cannot reach `127.0.0.1:18766` +- **THEN** host Caddy fails that Truf request without forwarding it to X-UI or another fallback upstream + +#### Scenario: A client supplies transit headers +- **WHEN** a public request supplies an ingress marker, forwarded address, private edge marker, or operator identity +- **THEN** host Caddy strips those values and injects only its reviewed ingress marker and observed client address + +### Requirement: Runtime overview +The overview page SHALL report bounded structured health for Supervisor, PostgreSQL, required pipeline workers, discovery producers, queue status, active remote assignments, result bundles, operation controls, and recent operation outcomes. + +#### Scenario: Runtime is healthy +- **WHEN** all required components hold valid authority and health +- **THEN** the overview reports the runtime active and identifies each required component without exposing secrets + +#### Scenario: Component is unavailable +- **WHEN** a health source times out or returns malformed state +- **THEN** the overview reports that component unavailable without blocking the rest of the page + +### Requirement: Search controls +The search page SHALL expose each core producer's structured state and typed start, stop, restart, pause, resume, and interval controls while clearly separating process lifecycle from the persistent discovery gate. + +#### Scenario: Operator pauses search +- **WHEN** an operator submits pause with the current control revision +- **THEN** the persistent discovery gate changes atomically and every producer stops admitting new discovered targets + +#### Scenario: Operator restarts one producer +- **WHEN** an operator selects restart for an allowed producer ID +- **THEN** only that managed discovery process restarts and the persistent pause state is unchanged + +### Requirement: Dispatch and drain controls +The workers/dispatch page SHALL expose persistent dispatch pause/resume, drain start/cancel, drain progress, compatible package state, worker users/devices, assignment counts, and upload availability. + +#### Scenario: Operator pauses dispatch +- **WHEN** the current revision is submitted to the pause action +- **THEN** no new worker assignment can commit while valid existing uploads remain accepted + +#### Scenario: Operator starts drain +- **WHEN** drain is started +- **THEN** the page reports draining progress from authoritative assignment and bundle counts until the state becomes drained + +#### Scenario: Stale page attempts resume +- **WHEN** another operator has changed the control revision before resume is submitted +- **THEN** the console reports a revision conflict and does not overwrite the newer state + +### Requirement: Typed Supervisor operations +The console SHALL use a closed Supervisor protocol for structured snapshot, allowlisted managed-source lifecycle actions, and bounded log tail, and SHALL NOT forward generic command strings. + +#### Scenario: Operator requests source status +- **WHEN** the Supervisor page loads +- **THEN** it displays structured source IDs, roles, phases, process state, restart state, and safe errors without parsing a text dashboard + +#### Scenario: Operator tails logs +- **WHEN** an allowed managed source and bounded line count are submitted +- **THEN** Supervisor returns only that source's bounded log tail + +#### Scenario: Input resembles a shell command +- **WHEN** an operator submits command text, a path, or an unrecognized source ID +- **THEN** the request is rejected and no generic Supervisor command or operating-system shell is called + +### Requirement: Durable asynchronous operation status +Long-running restart and apply actions SHALL create a PostgreSQL operation record before execution and SHALL remain queryable by operation ID across runtime/admin restarts. + +#### Scenario: Apply restarts the admin process +- **WHEN** the process that accepted an apply request terminates during the coordinated restart +- **THEN** the operator can reopen the operation URL and observe reconciled success, rollback, or failure state + +#### Scenario: Unknown operation is requested +- **WHEN** an operator requests an operation ID that does not exist or is not canonical +- **THEN** the console returns not found without searching filesystem paths or host-agent state by user input + +### Requirement: Append-only audit view +The audit page SHALL show bounded append-only events for accepted and completed controls, Supervisor actions, configuration/secrets operations, and managed-file mutations, including actor, action, logical target, time, result, and safe before/after identity. + +#### Scenario: Secret apply is audited +- **WHEN** a secrets candidate is accepted and later applied or rolled back +- **THEN** audit events record hashes and outcomes but no secret value, candidate bytes, authorization data, or CSRF value + +#### Scenario: Audit pagination is requested +- **WHEN** an operator navigates audit history +- **THEN** the backend returns a bounded deterministic page without unbounded database or browser output + +### Requirement: Existing worker API availability +Adding the operations console SHALL NOT weaken or couple public worker endpoints to admin page availability. + +#### Scenario: Admin feature is disabled or unhealthy +- **WHEN** the admin console is disabled or a Supervisor/host-agent status dependency is unavailable +- **THEN** authenticated worker status, upload, terminal report, and receipt paths continue under their existing authority diff --git a/openspec/changes/add-web-operations-control-plane/tasks.md b/openspec/changes/add-web-operations-control-plane/tasks.md new file mode 100644 index 0000000..cccddba --- /dev/null +++ b/openspec/changes/add-web-operations-control-plane/tasks.md @@ -0,0 +1,92 @@ +## 1. PostgreSQL Operations Authority + +- [x] 1.1 Add an additive runtime-safety migration for the singleton operations control row, durable operation records, and append-only audit events +- [x] 1.2 Add schema invariants, final-cutover checks, import/export handling, and migration-count fixtures for the new tables +- [x] 1.3 Implement ScannerDB control-state reads and revision-checked discovery, dispatch, and drain mutations with atomic audit insertion +- [x] 1.4 Enforce the discovery gate transactionally in provider admission, discovery retry, and Docker tag-resolution paths without blocking ingestion-derived work +- [x] 1.5 Enforce the dispatch gate transactionally in remote reservation admission while preserving status, terminal report, upload, replay, expiry, and ingestion +- [x] 1.6 Implement drain progress queries and reconciliation from live assignments and pre-commit result bundles +- [x] 1.7 Add concurrent PostgreSQL tests for pause/admission races, stale revisions, restart persistence, drain completion, and uninterrupted uploads + +## 2. Discovery-Only Server Producers + +- [x] 2.1 Extract a discovery-only cycle for GitLab, DockerHub, and HuggingFace that performs provider work and enqueueing without scan preparation or claiming +- [x] 2.2 Preserve DockerHub page cursors, retry-lane processing, tag resolution, and immutable target admission in the discovery role +- [x] 2.3 Add an authenticated `discovery-producer` runtime bootstrap role and Supervisor managed-process type with structured state +- [x] 2.4 Change the default distributed core profile to exactly GitLab, DockerHub, and HuggingFace and remove GitHub from that profile +- [x] 2.5 Add tests proving each producer runs with backlog, respects persistent pause, reports safe state, and never enters local scanner/claim/bundle code + +## 3. Protocol-2 Multisource Assignments + +- [x] 3.1 Replace the Git-only assignment switch with source adapters that declare queue source, worker platform, planning kind, snapshot validation, and package capability +- [x] 3.2 Implement GitLab `exact_git_v1`, DockerHub `docker_direct_v1`, and HuggingFace `huggingface_space_v1` execution-snapshot models and canonical validation +- [x] 3.3 Generalize ScannerDB claim recovery and result-ready validation for the three planning kinds while retaining fixed reservation and device fences +- [x] 3.4 Update source selection to try other compatible eligible queues while issuing at most one assignment per admission request +- [x] 3.5 Advance worker/package manifests to protocol 2 with explicit source/platform/planning capabilities and fail-closed manifest validation +- [x] 3.6 Update Windows and Linux worker clients to dispatch the bound Docker and HuggingFace scan platforms and validate protocol-2 snapshots before execution +- [x] 3.7 Retain protocol-1 status, upload, terminal-report, receipt, and immutable reconciliation for existing assignments while refusing new protocol-1 claims +- [x] 3.8 Add unit and PostgreSQL integration tests for capability matching, fallback, replay, expiry, stale upload rejection, source aliases, and legacy completion + +## 4. New-Source End-to-End Canaries + +Implementation guardrail: provider accessibility is finalized by the worker. New per-target server preflight/proof state, broad worker-environment credential scrubbing, or other source-specific defensive infrastructure requires separate operator approval and an explicit OpenSpec requirement/task before implementation. + +- [x] 4.1 Implement public DockerHub immutable-digest assignment execution without server registry-credential or layer-plan transport +- [x] 4.2 Implement tokenless HuggingFace Space assignment execution with explicit discovery visibility filtering and non-retryable inaccessible results, without leaking server discovery credentials +- [x] 4.3 Extend packaged-worker verification for Windows and Linux with synthetic DockerHub and HuggingFace claim-to-ingestion flows +- [x] 4.4 Add bounded canary configuration and assertions for search, enqueue, claim, worker-classified provider failures, upload, ingestion, projection, replay, expiry, and drain without adding per-target server access proofs + +## 5. Shared Runtime Document Validation + +- [x] 5.1 Add a side-effect-free bounded YAML loader with duplicate-key rejection and secret-safe errors +- [x] 5.2 Define strict configuration, core-profile, auth-pool, reference-integrity, package-capability, and deployment-path validation +- [x] 5.3 Use the shared validator in preview, runtime startup, and secrets import without weakening existing runtime security checks +- [x] 5.4 Implement fixed config/secrets candidate storage with private durable writes, SHA-256 compare-and-swap, and bounded redacted diffs +- [x] 5.5 Add validation and concurrency tests for unknown keys, duplicate keys, invalid references, stale active hashes, stale candidates, and error redaction + +## 6. Typed Supervisor and Operations Services + +- [x] 6.1 Extend Supervisor control protocol with structured runtime/source snapshots and exact managed-source lifecycle actions +- [x] 6.2 Add bounded log-tail actions keyed only by allowlisted managed source IDs and reject generic web command forwarding +- [x] 6.3 Implement operation creation, lifecycle transition, bounded result reconciliation, and append-only audit service methods +- [x] 6.4 Add Supervisor and operation-service tests for malformed actions, unknown sources, log bounds, restart races, and content-free audit records + +## 7. Web Operations Console + +- [x] 7.1 Add trusted Caddy operator-header stripping/injection and backend actor validation tied to the private edge marker +- [x] 7.2 Add shared admin navigation and overview page with bounded runtime, queue, assignment, bundle, control, and operation health +- [x] 7.3 Add Search pages and exact forms for persistent discovery controls plus typed producer lifecycle and interval actions +- [x] 7.4 Add Workers/Dispatch pages and exact forms for pause/resume, drain start/cancel/progress, package compatibility, users, devices, and assignments +- [x] 7.5 Add Supervisor status/action and bounded log pages without shell, path, or generic command inputs +- [x] 7.6 Add config and plaintext secrets preview/save/apply pages with no-store rendering, revision/hash conflicts, and no value leakage outside the editor +- [x] 7.7 Add durable operation-status and bounded paginated audit pages that survive runtime restart +- [x] 7.8 Add admin API and browser tests for routes, methods, exact form shapes, actor spoofing, Origin/CSRF, stale forms, navigation, CSP, and secret non-retention + +## 8. Managed File Service + +- [x] 8.1 Define logical managed roots and per-root list/read/create-replace/delete permissions with bounded path, listing, and byte limits +- [x] 8.2 Implement descriptor-relative Linux traversal with no-follow component opens and rejection of absolute, dot, drive, backslash, symlink, hardlink, and special-file targets +- [x] 8.3 Implement bounded download, durable compare-and-swap create/replace, and expected-hash delete with private temporary files and directory fsync +- [x] 8.4 Add typed Files pages/forms and content-free mutation audit events while keeping config, secrets, database, sockets, agent metadata, and raw bundles excluded +- [x] 8.5 Add adversarial traversal, encoded traversal, symlink-swap, hardlink, special-file, limit, concurrent-replacement, and forbidden-root tests + +## 9. Privileged Host Operations Agent + +- [x] 9.1 Implement a root-owned Unix-socket agent with peer-credential checks and a closed request schema containing only operation ID, action enum, and expected hashes +- [x] 9.2 Implement singleton locking, persisted-operation verification, host-side revalidation, fixed-path backups, and atomic config/secrets replacement +- [x] 9.3 Implement fixed runtime/edge stop and recreation plus bounded health verification for Supervisor, PostgreSQL, Worker API, ingester, and projector +- [x] 9.4 Implement byte-identical automatic rollback and failed-hold behavior without arbitrary services, paths, commands, or retry loops +- [x] 9.5 Add systemd socket/service units, fixed host-managed config/candidate directories, permissions, and deployment installer validation +- [x] 9.6 Add crash-boundary and hostile-request tests for validation, backup, replacement, restart, health failure, rollback success, rollback failure, and request-field injection + +## 10. Deployment Migration and Verification + +- [x] 10.1 Update container, code-authority, package, Caddy, systemd, and deployment fixtures for all new modules, profiles, routes, headers, sockets, and fixed managed paths +- [x] 10.2 Add an offline migration/rollback test proving the previous image can coexist with additive tables after protocol-2 work is drained +- [x] 10.3 Run focused unit and PostgreSQL integration suites, packaged worker verification, edge E2E, browser coverage, strict OpenSpec validation, and diff checks +- [x] 10.4 On the approved production host, deploy one exact supported ingress profile with discovery and dispatch paused, verify protocol-2 packages plus runtime/admin/edge/agent health, complete one reconciled lifecycle operation and one successful automatic rollback drill, and only then issue or resume new work +- [x] 10.5 Run the bounded DockerHub end-to-end canary and retain evidence for queue authority, ingestion, projection, expiry/replay, and drain +- [x] 10.6 Enable GitLab and public HuggingFace only after canary gates pass, confirm GitHub remains excluded, and document rollback evidence +- [x] 10.7 Add exact root-owned `standalone-edge-v1` and `shared-host-edge-v1` deployment profiles without changing the host-agent request schema +- [x] 10.8 Add the shared-host Compose/Caddy route confinement and profile-specific lifecycle, installer, and denylist validation while preserving standalone behavior +- [x] 10.9 Add dual-profile tests, deployment documentation, strict validation, and real hardened Caddy/Compose checks diff --git a/openspec/changes/add-worker-operator-experience/.openspec.yaml b/openspec/changes/add-worker-operator-experience/.openspec.yaml new file mode 100644 index 0000000..265da3d --- /dev/null +++ b/openspec/changes/add-worker-operator-experience/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-23 diff --git a/openspec/changes/add-worker-operator-experience/design.md b/openspec/changes/add-worker-operator-experience/design.md new file mode 100644 index 0000000..26dc2f5 --- /dev/null +++ b/openspec/changes/add-worker-operator-experience/design.md @@ -0,0 +1,206 @@ +## Context + +The protocol-2 remote worker has proven its authoritative data path under real production load, including native Windows concurrency 3 and simultaneous Windows/WSL execution. Its operator experience has not reached the same level: the package exposes only `--server`, `--token`, and `--parallelism`; successful work is mostly silent; scanner output is captured out of view; slot state remains `assigned` during synchronous execution; and the server sees no phase progress between claim and result. + +Two production DockerHub assignments remained unresolved until the fixed 7,200-second assignment deadline even though the configured Docker target timeout was 600 seconds. The current watchdog terminates the owned TruffleHog process tree, but permit acquisition and surrounding Python work such as cleanup, filtering, result serialization, bundle staging, and handoff are not one hard-preemptible unit. Current evidence cannot identify which phase stalled. + +The administration page compounds the problem by labeling `result_reservations.last_error_code` as `Error category`. That field describes assignment transport failures and expiry, while accepted scan errors live in `target_scans`, `errors`, and result metadata and are not shown as assignment diagnostics. + +Legacy `supervisor.py --attach` demonstrates a useful interaction model: a verified background instance, an authenticated local control channel, a live status table, bounded logs, and detach without shutdown. The remote-worker package does not contain that supervisor and needs a smaller worker-specific implementation. + +## Goals / Non-Goals + +**Goals:** + +- Deliver one coherent operator product rather than temporary UI over incomplete fields. +- Give owners a cross-platform lifecycle CLI with detached operation, attach, status, logs, history, diagnostics, and doctor commands. +- Define one phase/event model used locally, over the Worker API, in PostgreSQL, and in the admin UI. +- Apply a hard local scan-stage deadline to the complete assignment execution unit, not only its scanner child process. +- Preserve exact diagnostic material when available and explicitly describe size truncation or other transformations. +- Separate assignment transport outcome, scan outcome, and diagnostics in storage and UI. +- Make global and per-source assignment deadline policy visible and measurable with phase duration percentiles. +- Ship a from-zero operator guide and fault-injection coverage as part of the same change. + +**Non-Goals:** + +- Replacing PostgreSQL assignment authority, immutable assignment expiry, durable receipts, or bundle ingestion. +- Renewing assignment ownership from progress events. +- Reporting fabricated percentage completion when the scanner has no reliable denominator. +- Rebuilding the full server runtime supervisor inside the worker package. +- Adding a temporary compatibility-only error page that will be removed after diagnostics land. +- Adding a new diagnostic masking or redaction subsystem. + +## Decisions + +### 1. One worker supervisor owns lifecycle and authoritative local projections + +The package will expose a `truf-worker` command with `install`, `run`, `start`, `stop`, `status`, `attach`, `logs`, `history`, and `doctor` subcommands. `run` executes the supervisor in the foreground; `start` launches the same supervisor detached and waits for a startup handshake. Docker continues to run the supervisor in the foreground under Tini while Docker supplies detachment. + +The supervisor owns the singleton lock, slot controllers, local control endpoint, rotating human log, event JSONL, current status snapshot, and terminal local history. A verified instance record binds PID, process creation identity, executable, package identity, local control endpoint, and lifecycle state. `attach` reads status/events through the local control channel; it does not attach to arbitrary stdout and detaching never stops the worker. + +The current three-flag invocation remains a foreground-run migration alias for already shipped package launch definitions, but all new documentation and generated launchers use explicit subcommands. + +Alternative considered: add more prints to `remote_worker_client.py`. Rejected because prints cannot provide detached lifecycle control, reliable concurrent-slot rendering, machine output, history, or process identity. + +### 2. A single append-only event model drives every projection + +Each state transition emits a versioned event with a monotonic local sequence: + +```json +{ + "schema": 1, + "sequence": 42, + "timestamp": "2026-09-23T12:00:00Z", + "instance_id": "...", + "slot_id": 0, + "reservation_id": 123, + "source": "dockerhub", + "type": "slot.phase", + "phase": "scanning", + "phase_started_at": "...", + "scan_deadline_at": "...", + "assignment_deadline_at": "...", + "progress": {} +} +``` + +Canonical phases are `idle`, `claiming`, `assigned`, `waiting_permit`, `preparing`, `resolving`, `downloading`, `cloning`, `scanning`, `filtering`, `cleaning`, `bundling`, `uploading`, `awaiting_receipt`, `backoff`, `draining`, and `stopped`. + +The local status JSON is a rebuildable projection of the event stream, not a second independently written state machine. Server progress uses the same event schema with a server-assigned receive timestamp. Terminal history records authoritative receipt/prebundle/stale outcomes plus durations. + +Alternative considered: define separate local, API, and admin status shapes. Rejected because they would drift and recreate the current ambiguity. + +### 3. Execute each scan stage in a supervised assignment runner process + +Threads in the controller continue to own claim/recovery/upload state, but scanner execution moves into a package-local assignment runner subprocess. The controller writes one validated input record and starts the runner in a contained process tree with a dedicated work/output directory. + +The hard scan-stage deadline starts before permit acquisition and covers: + +- permit wait; +- assignment preparation and source resolution performed locally; +- downloads/clones; +- scanner subprocess execution; +- filtering and result conversion; +- cleanup; +- result and bundle staging. + +The runner emits phase events over a local pipe/file protocol. At the deadline the controller terminates the runner process tree, atomically detaches its work directory for later janitor handling, and creates a normal timeout result bundle with the phase and diagnostic envelope. The controller then uploads that result while the immutable assignment deadline still permits it. + +Normal cleanup remains cooperative. Cleanup that exceeds the deadline cannot keep the slot occupied; abandoned work is moved into a janitor-owned tree and reported in status. + +Alternative considered: add deadline checks around existing Python calls. Rejected because blocking Python/filesystem/library calls cannot be hard-preempted reliably in the controller process. + +### 4. Progress events are informational and never renew authority + +The Worker API gains an authenticated progress endpoint accepting ordered phase events for the worker's current reservation. It validates reservation/device ownership and monotonically advances the latest accepted event sequence. Duplicate events are idempotent. + +Progress updates do not change `remote_expires_at`, queue leases, bundle capacity, or receipt authority. Failure to send progress does not abort a healthy local scan; events remain in the local journal and retry with bounded backoff. The server can therefore display `last phase` and `last progress` without turning progress into a lease heartbeat. + +Alternative considered: renewable leases. Rejected because a stuck client could retain work indefinitely and because the fixed-expiry fencing model is already proven. + +### 5. Server-owned deadlines support global fallback and per-source policy + +`supervisor.worker_api.assignment_ttl_seconds` remains the global fallback. A managed per-source override map adds GitLab, DockerHub, and HuggingFace assignment TTL values. The server selects and commits the immutable deadline when issuing an assignment and includes the effective scan and assignment deadlines in the response. + +The editor labels these separately as `Target scan timeout`, `Assignment deadline (end-to-end)`, and `Result upload body deadline`. Validation retains absolute bounds and checks that each effective assignment deadline covers its source scan timeout, upload deadline, and handoff margin. + +The server aggregates phase and end-to-end durations by source and outcome as p50, p95, and p99. Configuration remains explicit; metrics inform changes but do not silently rewrite policy. + +Alternative considered: let each worker select or renew its TTL. Rejected because workload policy belongs to the assigning server and must be consistent for queue fencing. + +### 6. One diagnostic envelope spans scan and prebundle failures + +Diagnostics use one versioned envelope with indexed dimensions and exact optional payloads: + +- diagnostic ID and schema; +- reservation, scan event, slot, attempt, and source; +- phase, kind, category, stable code, summary, and retryable disposition; +- provider operation and HTTP status/content type/request ID; +- process name, exit code, signal, and timeout state; +- exception type/message/fingerprint; +- raw body material; +- log stream head/tail material; +- occurred/captured/received timestamps; +- original byte count, stored byte count, content hash, and truncation state. + +Categories are broad query dimensions such as `authorization`, `rate_limit`, `not_found`, `network`, `timeout`, `scanner`, `storage`, `protocol`, and `assignment_expired`. Stable codes express the concrete cause, such as `docker.manifest_http_403` or `trufflehog.exit_nonzero`. Phase is independent of category. + +Text/bytes are preserved as captured. Non-text bodies use an explicit encoding field. Storage bounds are deterministic: body excerpt 16 KiB, combined log head/tail 32 KiB, one transmitted envelope 64 KiB, at most 32 diagnostics and 256 KiB per assignment. Metadata records every truncation; no truncation is presented as a complete body. + +Accepted result bundles gain a diagnostic frame. Prebundle reports carry the same envelope under the existing JSON body limit. The old E-frame error string remains ingestible during rollout but is projected into the new model exactly once. + +Alternative considered: keep scanner error strings, reservation error codes, and source metadata as separate taxonomies. Rejected because operators cannot correlate or filter them consistently. + +### 7. Local diagnostics retain complete operator evidence when available + +The supervisor writes: + +```text +control/worker.instance.json +control/worker.status.json +events/worker-events.jsonl +history/worker-history.jsonl +diagnostics/YYYY-MM-DD//.json +diagnostics/YYYY-MM-DD//.body +diagnostics/YYYY-MM-DD//.log +logs/worker.log +``` + +The JSON envelope points to optional body/log files and records their hashes and sizes. Local retention is configurable by age and total bytes and is reported by `status` and `doctor`. Rotation never mutates terminal history entries; it changes attached-artifact availability explicitly. + +Alternative considered: store every raw artifact directly in one JSONL. Rejected because large multiline/process output makes append recovery and bounded tailing expensive. + +### 8. PostgreSQL stores diagnostics as first-class records + +Add `worker_diagnostics` with an idempotent diagnostic UID, reservation FK, optional target-scan FK, indexed phase/category/code/kind/retryable columns, summary, canonical envelope JSON, received timestamp, and optional bounded body/log payload columns. Bundle ingestion writes diagnostics in the same transaction as the target scan and error projection. Prebundle diagnostics attach to the reservation before a target scan exists. + +Existing `errors` rows remain the compatibility scan-error projection. Existing `last_error_code` is retained as assignment failure code but is no longer labeled as the complete error category. + +Alternative considered: put all envelopes only into `metadata_json`. Rejected because filtering, detail lookup, idempotency, and prebundle diagnostics require first-class rows. + +### 9. Admin views assignment outcome, scan outcome, progress, and diagnostics separately + +The worker assignment table presents: + +- assignment outcome: unfinished, accepted, prebundle failed, expired; +- scan outcome: clean, found, degraded, error, skipped, or not available; +- diagnostic count and highest-priority category/code; +- current/latest phase, phase age, assignment deadline, and last progress age; +- worker/device/package identity and slot where available. + +Each row links to a detail page with an event timeline, duration breakdown, transport/receipt data, scan summary, diagnostics, raw body/log tabs, canonical JSON copy/download, and explicit truncation metadata. Filters operate independently on assignment outcome, scan outcome, source, phase, category, code, retryability, and time. + +Alternative considered: make the existing `Error category` cell open a modal. Rejected because the list model itself conflates transport and scan semantics. + +### 10. Human output and machine output are equal product contracts + +Human status/attach uses a stable table and event stream. It displays elapsed time, configured scan deadline, assignment time remaining, last progress age, child state, and trustworthy counters. It never fabricates completion percentages. + +`--json` commands emit one versioned JSON document. Follow modes emit NDJSON with one event per line and no decorative output. Tests treat both output forms as contracts. + +## Risks / Trade-offs + +- [Runner process split touches scanner lifecycle and recovery paths] -> Introduce it behind the same `execute_protocol2_remote_claim` contract, preserve deterministic bundle validation, and fault-test every boundary before replacing in-process execution. +- [Progress traffic increases database writes] -> Persist only monotonic phase transitions and coarse progress changes, deduplicate by reservation/sequence, and keep high-frequency local samples local. +- [Diagnostic payloads increase bundle and database volume] -> Enforce deterministic per-item/per-assignment byte and count limits, expose truncation metadata, and track storage usage. +- [One broad change can take too long] -> Implement as large vertical chunks that each finish a final architecture slice; do not ship throwaway status/error models. +- [Per-source policy adds configuration complexity] -> Keep one global fallback, explicit source overrides, editor-derived effective values, and validation based on existing scan/upload settings. +- [Local full-stage termination can leave work trees] -> Atomically detach them to janitor ownership and surface retained bytes/counts in status and doctor. +- [Old packages do not emit progress/diagnostics] -> Admin renders legacy records from existing fields and marks phase/diagnostic availability explicitly until packages are upgraded. + +## Migration Plan + +1. Add schema/event/diagnostic libraries, PostgreSQL tables, and read paths without changing current assignment execution. +2. Add the worker supervisor CLI and local projections while the current foreground invocation remains a migration alias. +3. Add assignment runner execution, full-stage watchdog, local phase events, and fault-injection tests. +4. Add Worker API progress and diagnostic transport, then enable server persistence and detail queries. +5. Replace the worker admin list/detail presentation and add percentile/deadline editor views. +6. Rebuild Windows/Linux packages, run protocol compatibility tests, then run bounded Windows and WSL production validation. +7. Update generated launchers and the from-zero operator guide; migrate the production worker launch definition to explicit `run`. +8. Remove the migration alias only in a separately declared breaking change after all known deployments use subcommands. + +Rollback disables progress ingestion and runner selection while retaining additive event/diagnostic tables. Existing immutable assignment, bundle, receipt, and legacy E-frame paths remain authoritative throughout rollout. + +## Open Questions + +No blocking product questions remain. Exact local retention defaults and percentile windows can be selected from implementation benchmarks without changing the external contracts above. diff --git a/openspec/changes/add-worker-operator-experience/proposal.md b/openspec/changes/add-worker-operator-experience/proposal.md new file mode 100644 index 0000000..1ff423a --- /dev/null +++ b/openspec/changes/add-worker-operator-experience/proposal.md @@ -0,0 +1,38 @@ +## Why + +Remote workers execute and settle production work correctly, but they behave as opaque background processes: an owner cannot tell whether a slot is downloading, scanning, cleaning up, bundling, uploading, stalled, or close to its deadline, while the admin console conflates assignment failures with scan errors. The next change should turn the proven worker engine into an understandable operator-facing product without building temporary UI around the current incomplete status and error fields. + +## What Changes + +- Add a worker-specific supervisor and CLI for install, foreground run, detached start, graceful stop, status, attach, live logs, local history, and diagnostics. +- Add one versioned assignment phase/event model shared by local status, attach, server progress, history, diagnostics, and admin rendering. +- Apply a hard local scan-stage deadline across permit acquisition, scanner execution, cleanup, filtering, and result staging instead of bounding only the scanner subprocess path. +- Make assignment deadlines observable and configurable as server-owned global and per-source policy, with explicit scan, upload, and assignment deadline semantics. +- Add a versioned diagnostic envelope for provider responses, process failures, exceptions, timeout state, structured categories/codes, and bounded raw body/log material. +- Persist local worker events and diagnostics as JSON/JSONL and transmit the same diagnostic model through terminal reports and accepted result bundles. +- Replace the ambiguous assignment-table `Error category` presentation with distinct assignment outcome, scan outcome, diagnostic count, and a clickable detail/timeline view. +- Add machine-readable `--json`/NDJSON output and human-readable status/attach views without inventing progress percentages. +- Add a from-zero operator guide covering installation, first start, attach/status, interpreting phases and errors, graceful drain/stop, recovery, update, and troubleshooting. +- Do not add a separate masking/redaction feature or silently generalize captured diagnostics. Storage bounds and any unavoidable transformation must be explicit in diagnostic metadata. + +## Capabilities + +### New Capabilities +- `worker-operator-supervisor`: Lifecycle CLI, detached supervision, attach/status/log/history/doctor commands, local projections, and operator documentation. +- `worker-progress-deadlines`: Canonical assignment phases, local and server progress events, full scan-stage watchdog behavior, deadline policy, and duration percentile observability. +- `worker-diagnostics`: Unified diagnostic envelope, local diagnostic archive, terminal/bundle transport, taxonomy, retention bounds, and exact transformation metadata. +- `worker-admin-experience`: Assignment and scan outcome separation, diagnostic persistence/querying, clickable timeline/detail UI, and human/machine-readable diagnostic views. + +### Modified Capabilities + +None. There is no synchronized main-spec directory in this workspace; the new capabilities define the operator-facing contract over the existing distributed worker implementation. + +## Impact + +- Worker client/bootstrap/package entrypoints and generated Windows/Linux launchers. +- Scanner subprocess lifecycle, scan-slot ownership, cleanup, bundle staging, and local runtime state. +- Worker API progress, prebundle report, result-bundle, and assignment-status contracts. +- PostgreSQL reservation, scan, error, diagnostic, and observability queries/schema. +- Server-rendered worker administration pages and runtime deadline editor fields. +- Windows process supervision and Linux/WSL/Docker foreground/detached operation. +- Worker package tests, protocol tests, fault-injection tests, admin UI tests, and operator documentation. diff --git a/openspec/changes/add-worker-operator-experience/research.md b/openspec/changes/add-worker-operator-experience/research.md new file mode 100644 index 0000000..fb09934 --- /dev/null +++ b/openspec/changes/add-worker-operator-experience/research.md @@ -0,0 +1,207 @@ +# Remote Worker Operator Experience Findings + +Recorded: 2026-09-23. + +This file preserves the investigation that led to `add-worker-operator-experience` so implementation does not have to rediscover current behavior and terminology. + +## Production Validation Facts + +- Native Windows protocol-2 worker reached exact concurrency 3 and never exceeded it. +- Its bounded production cohort accepted 180 assignments: DockerHub 63, GitLab 65, HuggingFace 52. +- All 180 accepted results ingested, queue-settled, projected, and reconciled without global lineage violations or quarantine. +- A later dual-worker run proved Windows max 1, WSL max 1, combined max 2. +- Two WSL DockerHub assignments in separate runs stayed unresolved until fixed two-hour lease expiry and produced no accepted bundle. +- The server expired/refunded those reservations correctly; the missing information is where the worker spent the time before expiry. + +Durable validation report: + +`docs/worker-parallelism-validation-2026-09-23.md` + +## Timeout and Lease Map + +### Assignment deadline + +Production explicitly used: + +```yaml +supervisor: + worker_api: + assignment_ttl_seconds: 7200 +``` + +- Production value: 7,200 seconds. +- Code/template fallback: 86,400 seconds. +- Managed validation bounds: 60 through 604,800 seconds. +- Validation requires the effective TTL to cover the maximum configured source scan timeout, bundle upload deadline, and 60 seconds of handoff margin. +- The server commits one immutable expiry at assignment issuance. +- Ordinary worker contacts and progress do not renew it. +- Relevant code: `app/runtime_document.py`, `app/worker_api.py`, `app/worker_assignment.py`, `app/scanner_db.py`. + +### Docker target timeout + +Canonical managed full-runtime and production value: + +```yaml +sources: + dockerhub: + timeout: 600 +``` + +- Managed legacy Windows/full runtime: 600 seconds. +- Current production profile: 600 seconds. +- Direct CLI/function fallback without managed config: 1,800 seconds. +- Legacy Windows Job containment terminates the owned TruffleHog process tree at the subprocess deadline. +- Historical evidence: `tests/test_scanner_queue_high_fixes.py`, `tests/test_validated_high_scanner_fixes.py`, and `openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/`. + +### Other relevant timers + +- Worker API client socket timeout: 120 seconds. +- Result upload absolute body deadline: 1,800 seconds. +- Result upload idle timeout: 30 seconds. +- Remote assignment expiry reaper interval: 60 seconds. +- Empty-claim `Retry-After`: normally 5 seconds. +- Result-ingester/projector leases: 300 seconds after upload, unrelated to pre-upload scanning. + +### Why ten minutes became two hours + +The 600-second budget hard-preempts the TruffleHog subprocess path, but it is not one hard preemptive boundary around every surrounding Python operation. Permit acquisition, process startup/termination recovery, cleanup, filtering, serialization, bundle staging/fsync, and handoff can outlive the subprocess deadline. The remote client checks assignment expiry before synchronous execution and then has no phase heartbeat or cancellation loop. + +The evidence does not prove which phase stalled. Increasing the assignment TTL would only make the unknown stall retain ownership longer. The implementation needs phase instrumentation and a complete supervised execution-unit deadline. + +## Current Worker Experience + +Current supported CLI flags in `app/remote_worker_client.py`: + +```text +--server +--token +--parallelism +``` + +There are no worker `start`, `stop`, `status`, `attach`, `logs`, `history`, `doctor`, or JSON-output commands. + +Current behavior: + +- one daemon thread per slot; +- slot recovery state in `slot-N.json`; +- ready bundles retained until an authoritative receipt; +- scanner call is synchronous from the slot controller; +- state remains broadly `assigned` until bundle readiness; +- client loop failures print generic lines without a structured timeline; +- successful claim/scan/upload/receipt is mostly silent; +- scanner stdout/stderr is captured in temporary files and returned only after process completion; +- exact Git execution can emit no useful start line; +- concurrent messages can interleave; +- server `last_contact_at` does not prove or disprove active scanner progress. + +Default paths: + +- Windows: `%LOCALAPPDATA%/TRUF/RemoteWorker`. +- Linux: `$XDG_STATE_HOME/truf/remote-worker` plus `$XDG_DATA_HOME/truf/remote-worker`. +- Docker production convention: persistent `/data` volume with separate state/data roots. + +## Legacy Attach Pattern + +The old full-runtime supervisor has an `--attach` implementation in `app/supervisor.py` and a wrapper `attach_runtime.ps1`. The wrapper is currently disabled and the supervisor is not in the remote-worker package. + +Useful semantics to reuse: + +- exact background-instance identity; +- startup and loopback control handshake; +- initial status table; +- interactive `attach>` prompt; +- alternate-screen `watch` table; +- bounded log tail; +- `q`/EOF/Ctrl-C detach without worker shutdown; +- explicit coordinated shutdown command. + +The worker needs a smaller implementation over its own event/status model, not a copy of the complete server supervisor. + +## Current Error Model + +The admin `Error category` column is rendered in `app/admin_api.py` from: + +```sql +result_reservations.last_error_code AS error_category +``` + +It therefore describes assignment/transport errors such as remote prebundle failure or assignment expiry. It is not `errors.category`, scanner `error_class`, or `source_failure_category`. + +Consequences: + +- an accepted bundle is shown as assignment `completed` even when scan status is `error`; +- accepted scan errors usually leave assignment `Error category` blank; +- process versus storage prebundle failure survives in resolution JSON but is not shown; +- permanent provider skips can exist in metadata/warnings without an `errors` row; +- provider response bodies are inconsistently reduced or discarded. + +Useful data already persisted but not presented together: + +- `result_reservations`: resolution kind/JSON, receipt, issue/expiry/resolve timestamps, last error code/detail; +- `target_scans`: status, error count, skipped reason, first error summary, start/end/duration; +- `errors`: category, summary, raw selected error line; +- `scan_result_compat.metadata_json`: error class, retryability, source failure category, warnings, degraded/skipped flags, process return/timeout/output metadata; +- `keycheck_results`: provider status group/message/metadata. + +## Diagnostic Direction + +Use separate dimensions rather than one overloaded category: + +```text +assignment outcome: accepted | prebundle_failed | expired | unfinished +scan outcome: clean | found | degraded | error | skipped | unavailable +phase: scanning | cleaning | bundling | uploading | ... +kind: provider_http | scanner_process | exception | storage | protocol | ... +category: authorization | rate_limit | timeout | network | scanner | ... +code: stable concrete identifier +retryable: true | false +``` + +Candidate transmitted limits from the investigation: + +- provider body material: 16 KiB; +- process log head/tail: 32 KiB combined; +- one diagnostic envelope: 64 KiB; +- at most 32 diagnostics and 256 KiB total per assignment; +- prebundle envelope profile sized to fit the existing Worker API JSON limit. + +The local worker archive can retain larger/full artifacts under configurable age and byte rotation. Every transport transformation must state original size, stored size, hash, encoding, and truncation state. + +## Progress and Duration Direction + +Minimum phase transitions needed to explain the two-hour event: + +```text +assignment_received +scan_permit_acquired +runner_started +source_prepare_started/completed +scanner_started/exited +filtering_started/completed +cleanup_started/completed +bundle_started/ready +upload_started/acknowledged +``` + +Progress is evidence only and does not renew the immutable lease. + +Duration percentiles: + +- p50: median duration; +- p95: 95 percent of observations finish at or below this duration; +- p99: 99 percent finish at or below it. + +Compute them separately by source, phase, outcome, and time window, with sample counts. They describe observed behavior and inform policy; they do not silently set policy. + +## Product Decisions + +- Build one final operator architecture rather than a temporary admin patch. +- Implement it in large vertical chunks that each remain part of the final system. +- Use a worker-specific supervisor and local event stream. +- Split scanner execution into a supervised per-assignment runner process so the complete scan stage can be hard-preempted. +- Keep immutable server assignment ownership and make progress non-renewing. +- Add global fallback plus per-source assignment TTL policy. +- Use one diagnostic envelope across local files, terminal reports, bundles, PostgreSQL, API, and UI. +- Preserve diagnostic fidelity and expose all explicit transformations. +- Separate assignment outcome, scan outcome, and diagnostics in the admin UI. +- Include from-zero operator documentation and fault injection in the same change. diff --git a/openspec/changes/add-worker-operator-experience/specs/worker-admin-experience/spec.md b/openspec/changes/add-worker-operator-experience/specs/worker-admin-experience/spec.md new file mode 100644 index 0000000..1cd13d2 --- /dev/null +++ b/openspec/changes/add-worker-operator-experience/specs/worker-admin-experience/spec.md @@ -0,0 +1,71 @@ +## ADDED Requirements + +### Requirement: Separate assignment and scan outcomes +The worker administration list SHALL display assignment transport outcome, scan outcome, and diagnostic count as separate fields and SHALL not label `last_error_code` as the complete scan error category. + +#### Scenario: Accepted scan has an error outcome +- **WHEN** a reservation has a durable accepted bundle whose target scan status is `error` +- **THEN** the list SHALL show assignment `accepted`, scan `error`, and the diagnostic count/categories + +#### Scenario: Assignment expires before scan ingestion +- **WHEN** a reservation expires without an accepted bundle +- **THEN** the list SHALL show assignment `expired`, scan outcome unavailable, and the expiry diagnostic + +### Requirement: Assignment detail timeline +Each worker assignment row SHALL link to a detail page that reconstructs issued, phase-progress, bundle/terminal report, receipt, ingestion, queue settlement, and projection timestamps that exist for that assignment. + +#### Scenario: Administrator opens an active assignment +- **WHEN** progress events exist for an unresolved assignment +- **THEN** the page SHALL show current phase, phase age, last progress age, scan deadline, assignment deadline, and ordered prior phases + +#### Scenario: Administrator opens a settled assignment +- **WHEN** the assignment has been accepted and projected +- **THEN** the timeline SHALL distinguish acceptance, ingestion, queue settlement, and projection completion rather than collapsing them into one completion time + +### Requirement: Clickable diagnostic detail +The assignment detail page SHALL list diagnostics and SHALL provide human summary, canonical envelope JSON, raw body view, process log view, transformation metadata, and copy/download actions for each diagnostic. + +#### Scenario: Diagnostic body is complete +- **WHEN** an HTTP diagnostic contains an untruncated body +- **THEN** the raw-body view SHALL identify it as complete and display the captured content and metadata + +#### Scenario: Diagnostic material is truncated +- **WHEN** body or log material was size-truncated +- **THEN** the view SHALL prominently display original/stored sizes, hash, and truncation state + +#### Scenario: Legacy scan error has no diagnostic envelope +- **WHEN** an older scan has only existing `errors.raw_error` or result metadata +- **THEN** the detail page SHALL display those fields as legacy evidence and SHALL not invent a new envelope + +### Requirement: Diagnostic filtering and grouping +The admin UI SHALL filter independently by source, time, worker/device, assignment outcome, scan outcome, phase, category, stable code, and retryability and SHALL group repeated diagnostic fingerprints without hiding individual occurrences. + +#### Scenario: Administrator filters rate-limit errors +- **WHEN** category `rate_limit` and a time window are selected +- **THEN** results SHALL include matching diagnostics regardless of whether their assignments were accepted or prebundle-failed + +#### Scenario: Repeated diagnostics are grouped +- **WHEN** multiple diagnostics share a fingerprint +- **THEN** the UI SHALL show aggregate count and affected assignments while retaining links to each occurrence + +### Requirement: Worker fleet status +The admin UI SHALL display each worker's configured cap, active slots, current phases, package identity, latest contact/progress ages, pending local-recovery indication when reported, and known idle/backoff reason. + +#### Scenario: Worker is scanning without recent API contact +- **WHEN** a worker has an active assignment and recent progress events but its authentication contact timestamp is old +- **THEN** fleet status SHALL show active progress rather than classifying the worker as idle solely from contact age + +#### Scenario: Worker cannot claim due to capacity +- **WHEN** the server rejects claims because a pipeline capacity axis is closed +- **THEN** fleet status SHALL show the capacity reason instead of a generic offline/idle state + +### Requirement: Deadline and duration administration +The runtime editor and worker observability pages SHALL explain effective scan, upload, and assignment deadlines and SHALL show p50/p95/p99 duration metrics with sample counts by source and phase. + +#### Scenario: Administrator edits a source assignment deadline +- **WHEN** a per-source TTL candidate is previewed +- **THEN** the editor SHALL show the effective policy, validation relationship to scan/upload bounds, and that only future assignments are affected + +#### Scenario: Administrator compares policy to observations +- **WHEN** sufficient phase-duration samples exist +- **THEN** the page SHALL show the configured deadline alongside source-specific percentile values without automatically changing configuration diff --git a/openspec/changes/add-worker-operator-experience/specs/worker-diagnostics/spec.md b/openspec/changes/add-worker-operator-experience/specs/worker-diagnostics/spec.md new file mode 100644 index 0000000..e8d41da --- /dev/null +++ b/openspec/changes/add-worker-operator-experience/specs/worker-diagnostics/spec.md @@ -0,0 +1,86 @@ +## ADDED Requirements + +### Requirement: Unified versioned diagnostic envelope +Worker scan errors, provider failures, process failures, local exceptions, timeouts, prebundle failures, and assignment expiry context SHALL use one versioned diagnostic envelope with independent phase, kind, category, stable code, summary, retryability, attempt, timestamps, and optional HTTP/process/exception material. + +#### Scenario: Provider returns an HTTP error +- **WHEN** a provider operation receives an unsuccessful HTTP response +- **THEN** the diagnostic SHALL identify the phase, provider operation, HTTP status/content type, stable category/code, retryability, and captured response material + +#### Scenario: Scanner process fails +- **WHEN** a scanner process exits unsuccessfully or is terminated at its deadline +- **THEN** the diagnostic SHALL identify its process result, timeout/signal state, phase, stable category/code, and captured log material + +#### Scenario: Assignment expires without a worker result +- **WHEN** the server expires an unresolved assignment +- **THEN** it SHALL create or expose a diagnostic describing assignment expiry and the last accepted progress phase without claiming a scanner error occurred + +### Requirement: Diagnostic fidelity and explicit transformation +Captured body and log bytes SHALL be preserved without silent semantic rewriting. Every size limit, encoding conversion, or truncation SHALL record original bytes, stored bytes, content hash, encoding, and truncation state. + +#### Scenario: Text body fits the bound +- **WHEN** a captured provider response body fits the configured diagnostic body bound +- **THEN** the transmitted diagnostic SHALL contain the complete captured text and SHALL mark it untruncated + +#### Scenario: Body exceeds the bound +- **WHEN** captured body bytes exceed the transmitted bound +- **THEN** the diagnostic SHALL carry the bounded material plus original/stored sizes, full captured-content hash when available, and `truncated=true` + +#### Scenario: Body is not text +- **WHEN** captured diagnostic body bytes are not valid text in the declared encoding +- **THEN** the envelope SHALL use an explicit binary encoding representation and SHALL preserve the same transformation metadata + +### Requirement: Bounded diagnostic transport +The protocol SHALL enforce deterministic per-body, per-log, per-envelope, diagnostic-count, and aggregate diagnostic bounds while rejecting envelopes whose declared and actual sizes disagree. + +#### Scenario: Accepted bundle contains diagnostics +- **WHEN** a worker uploads a result bundle with diagnostic frames within all bounds +- **THEN** bundle acceptance and ingestion SHALL validate and persist each diagnostic idempotently with the scan + +#### Scenario: Diagnostic aggregate exceeds its limit +- **WHEN** a bundle or terminal report exceeds a diagnostic count or byte limit +- **THEN** the API SHALL reject it with a stable protocol error and SHALL NOT partially persist diagnostics + +### Requirement: Prebundle and accepted-result parity +The same diagnostic envelope SHALL be usable in prebundle terminal reports and accepted scan-result bundles, with only transport-size profiles differing. + +#### Scenario: Worker storage fails before bundle creation +- **WHEN** the worker cannot create a result bundle +- **THEN** its terminal report SHALL include a diagnostic envelope rather than replacing the exception with one generic fixed detail string + +#### Scenario: Scan returns errors in a valid bundle +- **WHEN** scanning completes with structured errors and a valid bundle +- **THEN** those errors SHALL be represented as diagnostics attached to the ingested target scan and SHALL remain distinct from assignment transport outcome + +### Requirement: Deterministic diagnostic identity +Each diagnostic SHALL have a deterministic UID derived from its canonical identity and content so retries and replay cannot create duplicates. + +#### Scenario: Accepted upload is replayed +- **WHEN** an identical accepted result bundle is uploaded again +- **THEN** the server SHALL return the durable receipt and SHALL NOT insert duplicate diagnostic rows + +#### Scenario: Same code occurs twice in one assignment +- **WHEN** two distinct occurrences share category and code but differ in occurrence identity or content +- **THEN** both SHALL be retained as distinct diagnostics with stable UIDs + +### Requirement: Local diagnostic archive +The worker SHALL retain a queryable local JSON diagnostic envelope and optional body/log artifacts per assignment, with configurable age/byte rotation and explicit artifact-availability state in history. + +#### Scenario: Operator opens a local diagnostic +- **WHEN** `truf-worker history` or `logs` selects a retained diagnostic +- **THEN** the worker SHALL present the canonical envelope and exact paths/availability of its body and log artifacts + +#### Scenario: Artifact rotates out +- **WHEN** a body or log artifact is removed by configured local rotation +- **THEN** terminal history SHALL remain and SHALL state that the artifact is no longer locally retained + +### Requirement: Orthogonal error taxonomy +The diagnostic model SHALL keep phase, kind, broad category, stable code, retryability, assignment outcome, and scan outcome as separate dimensions. + +#### Scenario: Accepted scan has provider errors +- **WHEN** a result bundle is durably accepted but the scan outcome is `error` +- **THEN** the assignment outcome SHALL remain `accepted`, scan outcome SHALL be `error`, and provider diagnostics SHALL retain their own categories/codes + +#### Scenario: Assignment expires +- **WHEN** an assignment expires before bundle acceptance +- **THEN** assignment outcome SHALL be `expired`, scan outcome SHALL be unavailable, and the expiry diagnostic SHALL not be categorized as a provider scan failure diff --git a/openspec/changes/add-worker-operator-experience/specs/worker-operator-supervisor/spec.md b/openspec/changes/add-worker-operator-experience/specs/worker-operator-supervisor/spec.md new file mode 100644 index 0000000..8f81642 --- /dev/null +++ b/openspec/changes/add-worker-operator-experience/specs/worker-operator-supervisor/spec.md @@ -0,0 +1,74 @@ +## ADDED Requirements + +### Requirement: Unified worker lifecycle CLI +The worker package SHALL provide `install`, `run`, `start`, `stop`, `status`, `attach`, `logs`, `history`, and `doctor` commands with equivalent lifecycle semantics on supported Windows and Linux/WSL platforms. + +#### Scenario: Operator starts a detached worker +- **WHEN** an installed operator invokes `truf-worker start` +- **THEN** the command SHALL launch the worker supervisor, wait for its startup handshake, and return the verified instance identity and current state + +#### Scenario: Operator runs in the foreground +- **WHEN** an operator invokes `truf-worker run` +- **THEN** the same supervisor implementation SHALL run in the foreground and SHALL begin graceful drain on the first interrupt + +### Requirement: Verified detached supervisor lifecycle +The supervisor SHALL publish a versioned instance record, status projection, local control endpoint, startup result, and shutdown receipt tied to the exact running process identity. + +#### Scenario: Status finds a stale instance record +- **WHEN** the recorded process no longer matches the recorded executable, creation identity, or live control handshake +- **THEN** `status` SHALL report the instance as stale and SHALL NOT represent it as a running worker + +#### Scenario: Graceful stop has active slots +- **WHEN** `stop` is requested while one or more slots own assignments +- **THEN** the supervisor SHALL stop new claims, display the draining slots, and wait for terminal local reconciliation up to the requested stop deadline + +### Requirement: Attachable live operator view +The supervisor SHALL provide an `attach` session that renders current worker and per-slot state and follows new events without making attachment own the worker lifetime. + +#### Scenario: Operator detaches +- **WHEN** the operator presses `q`, sends EOF, or interrupts the attach client +- **THEN** only the attach session SHALL end and the worker supervisor SHALL continue running + +#### Scenario: Concurrent slots update +- **WHEN** multiple slots emit interleaved phase events +- **THEN** attach SHALL render one coherent row per slot and SHALL preserve event ordering by local sequence + +### Requirement: Honest per-slot status +Status and attach SHALL display source, phase, phase elapsed time, scan deadline, assignment time remaining, last progress age, and only counters measured by the execution path. They SHALL NOT synthesize percentage completion without a reliable denominator. + +#### Scenario: Long scanner execution +- **WHEN** a slot remains in `scanning` with a live runner process +- **THEN** status SHALL continue updating elapsed time, deadline remaining, and last-progress age rather than appearing frozen + +#### Scenario: Worker has no assignment +- **WHEN** a slot is idle because of server backoff, cap, paused dispatch, capacity, or an empty queue +- **THEN** status SHALL report the known idle/backoff reason and next claim time when supplied by the server + +### Requirement: Human and machine output contracts +Every non-interactive inspection command SHALL support versioned JSON output, and every follow command SHALL support versioned NDJSON output containing no human decoration. + +#### Scenario: Automation requests status +- **WHEN** `truf-worker status --json` is invoked +- **THEN** stdout SHALL contain exactly one parseable versioned status object representing the same state as the human view + +#### Scenario: Automation follows events +- **WHEN** `truf-worker logs --follow --json` is invoked +- **THEN** stdout SHALL contain one complete versioned event object per line in sequence order + +### Requirement: Local history and operational diagnosis +The supervisor SHALL retain terminal assignment history, rotating worker logs, event history, diagnostic references, and local storage usage, and SHALL expose them through `history`, `logs`, and `doctor`. + +#### Scenario: Operator investigates a completed assignment +- **WHEN** the operator requests history for a terminal reservation +- **THEN** the worker SHALL show its terminal local/receipt outcome, durations, phase timeline, and available diagnostic artifact references + +#### Scenario: Operator runs doctor +- **WHEN** `truf-worker doctor` is invoked +- **THEN** it SHALL inspect package identity, singleton/process state, local state readability, disk usage, server reachability, and retained work without claiming an assignment + +### Requirement: From-zero operator documentation +The release SHALL include one canonical guide from package acquisition through installation, first start, attach/status interpretation, graceful stop, recovery, update, diagnostics, and removal. + +#### Scenario: New operator follows the guide +- **WHEN** an operator starts with a supported worker package and issued server enrollment data +- **THEN** the documented commands SHALL lead to a running verified worker and explain every state visible before the first assignment diff --git a/openspec/changes/add-worker-operator-experience/specs/worker-progress-deadlines/spec.md b/openspec/changes/add-worker-operator-experience/specs/worker-progress-deadlines/spec.md new file mode 100644 index 0000000..80558bb --- /dev/null +++ b/openspec/changes/add-worker-operator-experience/specs/worker-progress-deadlines/spec.md @@ -0,0 +1,79 @@ +## ADDED Requirements + +### Requirement: Canonical assignment phase model +The worker SHALL represent assignment execution with one versioned phase/event model shared by local status, local history, server progress, and administrative views. + +#### Scenario: Assignment completes normally +- **WHEN** a slot claims, executes, stages, uploads, and receives acceptance for an assignment +- **THEN** it SHALL emit monotonic phase events sufficient to reconstruct the time spent from `assigned` through `awaiting_receipt` + +#### Scenario: Process restarts during an assignment +- **WHEN** a worker restarts with persisted slot state +- **THEN** recovered events SHALL continue from the persisted sequence and SHALL record recovery without rewriting the prior timeline + +### Requirement: Complete scan-stage watchdog +The worker SHALL enforce one hard scan-stage deadline across permit acquisition, local preparation/resolution, data acquisition, scanner execution, filtering, cleanup, and result staging by supervising the complete execution unit outside the controller process. + +#### Scenario: Scanner child exceeds the deadline +- **WHEN** the assignment runner remains active at the scan-stage deadline +- **THEN** the controller SHALL terminate its complete process tree, release/detach local resources, and produce a timeout result identifying the final phase + +#### Scenario: Cleanup blocks after scanner exit +- **WHEN** scanner execution has ended but cleanup or staging remains blocked at the deadline +- **THEN** the same hard deadline SHALL terminate the runner and SHALL prevent the slot from remaining occupied until assignment expiry + +#### Scenario: Permit acquisition consumes the budget +- **WHEN** no scan permit is acquired before the scan-stage deadline +- **THEN** the worker SHALL produce a phase-specific timeout result without starting the scanner + +### Requirement: Non-renewing server progress +The Worker API SHALL accept idempotent monotonic progress events for the current reservation while preserving the original immutable assignment and queue deadlines. + +#### Scenario: Progress is accepted +- **WHEN** the assigned device submits the next valid event sequence for its unresolved reservation +- **THEN** the server SHALL persist the event/latest phase and SHALL NOT alter assignment expiry or ownership + +#### Scenario: Duplicate progress is retried +- **WHEN** an already accepted event sequence is submitted again +- **THEN** the server SHALL return the prior acceptance without creating a duplicate timeline event + +#### Scenario: Progress cannot reach the server +- **WHEN** local phase transitions occur during a temporary connection failure +- **THEN** execution SHALL continue under the fixed deadline and events SHALL remain available locally for ordered retry + +### Requirement: Observable deadline semantics +Assignments SHALL carry distinct effective target-scan, result-upload, and end-to-end assignment deadlines, and every operator/admin view SHALL label them by those meanings. + +#### Scenario: Operator inspects active work +- **WHEN** status or admin renders an active reservation +- **THEN** it SHALL show the effective scan deadline, assignment deadline, time remaining, and current phase without conflating them + +#### Scenario: Assignment expires +- **WHEN** the immutable assignment deadline passes without an accepted terminal result +- **THEN** expiry evidence SHALL include the last accepted phase and last-progress timestamp when available + +### Requirement: Global and per-source assignment policy +The managed runtime configuration SHALL provide a global assignment TTL fallback and optional explicit overrides for GitLab, DockerHub, and HuggingFace, selected by the server at issuance. + +#### Scenario: Source override exists +- **WHEN** a DockerHub assignment is issued and a DockerHub assignment TTL override is configured +- **THEN** its immutable expiry SHALL use the override and the assignment SHALL report that effective policy + +#### Scenario: Source override is absent +- **WHEN** an assignment is issued for a source without an override +- **THEN** the global assignment TTL SHALL be used + +#### Scenario: Invalid deadline policy is previewed +- **WHEN** an effective assignment deadline cannot cover its source scan timeout, upload deadline, and required handoff margin +- **THEN** managed configuration preview SHALL reject the candidate with a field-specific explanation + +### Requirement: Phase duration percentiles +The server SHALL expose p50, p95, and p99 durations by source, phase, outcome, and selected time window, based only on completed observations appropriate to that metric. + +#### Scenario: Administrator reviews DockerHub latency +- **WHEN** duration metrics are requested for DockerHub +- **THEN** the result SHALL separate end-to-end, scanning, cleanup, bundling, and upload percentiles and SHALL report sample counts + +#### Scenario: Insufficient samples exist +- **WHEN** a percentile does not have the configured minimum sample count +- **THEN** the UI/API SHALL label it insufficient rather than presenting it as a stable policy recommendation diff --git a/openspec/changes/add-worker-operator-experience/tasks.md b/openspec/changes/add-worker-operator-experience/tasks.md new file mode 100644 index 0000000..c5cedfe --- /dev/null +++ b/openspec/changes/add-worker-operator-experience/tasks.md @@ -0,0 +1,44 @@ +## 1. Canonical Contracts and Persistence + +- [x] 1.1 Implement the versioned worker phase/event model, canonical phase transitions, monotonic sequence validation, JSON/NDJSON serialization, and contract tests shared by worker, API, and admin code. +- [x] 1.2 Implement the unified diagnostic envelope, orthogonal taxonomy, deterministic diagnostic UID, exact body/log representation, explicit truncation metadata, aggregate limits, and serialization/validation tests. +- [x] 1.3 Add PostgreSQL progress-event and worker-diagnostic persistence, indexes, idempotent writes, reservation/scan joins, migration coverage, and authoritative query methods. +- [x] 1.4 Add managed global/per-source assignment deadline policy, effective-value validation against scan/upload bounds, assignment serialization, and runtime-document/editor tests. + +## 2. Complete Worker Supervisor + +- [x] 2.1 Build the `truf-worker` command surface (`install`, `run`, `start`, `stop`, `status`, `attach`, `logs`, `history`, `doctor`) over one supervisor implementation, including the existing foreground invocation migration alias. +- [x] 2.2 Implement verified Windows and Linux/WSL supervisor instance lifecycle, startup handshake, local control endpoint, graceful drain/stop, shutdown receipt, stale-instance handling, and lifecycle tests. +- [x] 2.3 Implement append-only local events, rebuildable status projection, terminal history, rotating logs, per-assignment diagnostic/body/log files, retention accounting, and crash/restart recovery tests. +- [x] 2.4 Implement human status/attach views and versioned JSON/NDJSON modes with coherent concurrent-slot rendering, honest phase/deadline/progress fields, bounded follow/tail behavior, and command-level tests. +- [x] 2.5 Update Windows portable and Linux/Docker package entrypoints, manifests, launchers, and package self-tests so the supervisor is the supported runtime on every platform. + +## 3. Assignment Runner and Full-Stage Watchdog + +- [x] 3.1 Introduce the contained per-assignment runner process and controller protocol while preserving existing claim state, deterministic bundle authority, source capabilities, and receipt/recovery behavior. +- [x] 3.2 Instrument permit wait, preparation, source resolution, download/clone, scanner execution, filtering, cleanup, bundle staging, upload, and receipt transitions with the canonical phase events and measured durations. +- [x] 3.3 Enforce one hard scan-stage deadline across the runner process tree, produce a normal phase-specific timeout result, detach abandoned work to janitor ownership, and return the slot without waiting for assignment expiry. +- [x] 3.4 Implement controller and runner crash recovery for persisted assignments, incomplete runner outputs, ready bundles, stale results, lowered parallelism, and supervisor restart. +- [x] 3.5 Add deterministic fault-injection tests for blocking/failure in every phase, including permit starvation, child non-exit, cleanup stall, staging/fsync failure, upload retry, deadline crossing, and process restart. + +## 4. Worker API and Result Pipeline + +- [x] 4.1 Add the authenticated non-renewing progress endpoint with ownership checks, monotonic/idempotent sequencing, latest-phase projection, bounded retry behavior, and API/database tests. +- [x] 4.2 Add diagnostic frames to protocol-2 bundles and the same diagnostic envelope to prebundle terminal reports, including exact size accounting, deterministic replay, and protocol compatibility tests. +- [x] 4.3 Ingest diagnostics transactionally with target scans/errors and attach prebundle diagnostics to reservations, while preserving durable receipt, queue settlement, projection, and replay invariants. +- [x] 4.4 Extend assignment/status responses with effective deadlines, latest phase/progress, known idle/backoff reason, and diagnostic availability, and cover old-package records explicitly in compatibility tests. + +## 5. Final Administration Experience + +- [x] 5.1 Replace the worker-list query/view model with separate assignment outcome, scan outcome, diagnostic summary, active phase, phase/progress age, effective deadlines, slot/cap, and package fields. +- [x] 5.2 Build the assignment detail page with ordered phase/receipt/ingestion/settlement/projection timeline, duration breakdown, scan summary, diagnostic list, exact body/log views, canonical JSON copy/download, and explicit transformation metadata. +- [x] 5.3 Add independent filters and repeated-diagnostic grouping for source, worker/device, assignment outcome, scan outcome, phase, category, stable code, retryability, and time window without hiding individual occurrences. +- [x] 5.4 Add source/phase/outcome p50, p95, and p99 duration queries and admin views with sample counts, and present them beside effective scan/upload/assignment deadline policy in the runtime editor. +- [x] 5.5 Add end-to-end admin tests for active progress, accepted scan errors, prebundle failures, assignment expiry, legacy records, complete/truncated bodies, repeated fingerprints, and machine-readable detail output. + +## 6. Operator Release and Production Proof + +- [x] 6.1 Write and validate the canonical from-zero operator guide covering package acquisition, install, first run, start/status/attach/logs/history, phase/deadline interpretation, diagnostics, graceful stop/drain, recovery, update, and removal. +- [x] 6.2 Run the complete unit/integration/protocol/package test matrix and build reproducible Windows and Linux worker artifacts with registered manifests and documented identities. +- [x] 6.3 Perform bounded production validation on native Windows and WSL/Docker covering multi-slot progress, attach while active, a forced full-stage timeout, diagnostic body/log inspection, restart recovery, accepted/ingested/projected reconciliation, and final production restoration. +- [x] 6.4 Record final duration percentiles, watchdog evidence, diagnostic/admin screenshots or snapshots, operator command transcript, known limits, and rollout/rollback results in a durable dated report. diff --git a/openspec/changes/cold-policy-stale-backlog/.openspec.yaml b/openspec/changes/cold-policy-stale-backlog/.openspec.yaml new file mode 100644 index 0000000..1a62d62 --- /dev/null +++ b/openspec/changes/cold-policy-stale-backlog/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-06 diff --git a/openspec/changes/cold-policy-stale-backlog/design.md b/openspec/changes/cold-policy-stale-backlog/design.md new file mode 100644 index 0000000..ddb19f8 --- /dev/null +++ b/openspec/changes/cold-policy-stale-backlog/design.md @@ -0,0 +1,80 @@ +## Context + +The completed keyword-pruning change removed 38 globally zero-yield terms from discovery configuration, but `target_queue` admission still considers previously queued `pending` and due `deferred` rows without consulting current query policy. Since deployment, Docker work attributed to retired terms consumed about 20.6 scanner-hours and produced no strict-usable credential. GitHub Actions independently consumed one core worker while both fresh and retained work produced no strict-usable credential and its active backlog grew. + +Queue rows are durable authority referenced by scan history, result reservations, Git and Docker coverage, deduplication, and retry state. Physical deletion or overloading quarantine would destroy or misrepresent that authority. Runtime configuration and PostgreSQL schema are lifecycle-protected and require coordinated offline changes. + +## Goals / Non-Goals + +**Goals:** +- Make policy-retired pending and deferred work explicitly unclaimable while preserving every durable record and field needed for audit or reversal. +- Apply and reverse policy holds only through exact, reviewed, idempotent manifests under stopped-source lifecycle authority. +- Prevent stale Docker resolver anchors from generating new digest children after the initial cold transition. +- Pause GitHub Actions without deleting or rewriting its backlog. +- Expose cold rows separately from active backlog in operational counts. + +**Non-Goals:** +- Deleting target queue rows or changing scan, finding, credential, result, deduplication, or coverage history. +- Age-based retirement of mutable GitLab or HuggingFace targets. +- Automatically colding every disabled source or all dormant historical backlogs in the first deployment. +- Changing scanner concurrency, Docker scan policy, discovery keywords, or keycheck behavior. +- Treating query retirement as evidence that a target can never become useful. + +## Decisions + +### Add a first-class `cold` queue status + +`target_queue.status='cold'` is the policy hold. Existing scan, legacy, and Docker resolver claim paths explicitly allow only `pending` and due `deferred`, so cold rows remain fail-closed even if a caller does not understand policy metadata. Cold is not a worker result disposition and ordinary scans cannot produce it. + +Alternative: add a nullable metadata flag. Rejected because every current and older admission path would need to remember an additional predicate, making accidental claims likely. `deferred` is also unsuitable because it is a timed retry, and `quarantined` is reserved for pipeline-integrity failures with capacity accounting and review semantics. + +### Use exact reviewed manifests and append-only audit events + +Add an append-only `target_queue_policy_events` table recording each cold/reactivate transition, prior and next status, exact source/platform/query attribution, configuration and policy hashes, manifest hash, prior update fence, reason code, and reversal linkage. A private dry-run manifest lists only queue IDs and non-sensitive policy attribution; it never contains target values. + +Apply locks rows in deterministic ID order and atomically validates the manifest fence, inserts one audit event, and changes only `status` and `updated_at`. Eligible cold transitions are limited to unfenced `pending` or `deferred` rows with no active queue/result/resolver lease or reservation. Reactivation restores the exact audited prior status and is explicit rather than an enqueue side effect. + +Alternative: issue a one-off SQL update and infer reversal from `available_after`. Rejected because it is not reviewable, cannot prove the selected cohort, and loses the exact prior lifecycle state. + +### Derive stale policy from canonical exact queries + +Policy uses case-sensitive exact `(source, platform, query)` triples from canonical source configuration. Null/blank queries, unknown source/platform pairs, non-rotation provenance, and sole operational sentinel queries are not inferred as stale. Disabled sources retain their configured query policy; source disablement is not itself a retirement action. + +The initial reviewed transition is scoped to DockerHub platform `docker`, covering every eligible pending/deferred resolver anchor and digest row whose own exact query is no longer configured. Dormant source families remain preserved and unclaimable by virtue of having no worker; they can be reviewed separately later rather than mutating a six-figure backlog in this deployment. + +### Filter periodic Docker resolver claims by current query policy + +Docker digest children inherit the resolver anchor query, and completed anchors can be periodically reclaimed. The runtime therefore passes the exact configured DockerHub query allowlist to resolver admission and excludes non-null anchors whose query is not allowed. This prevents completed stale anchors, which are outside the pending/deferred cold migration, from recreating policy-retired work. + +The filter is exact and does not rewrite query attribution. Null legacy provenance remains excluded from automatic policy retirement and requires separate review. + +### Preserve cold state until explicit reactivation + +Ordinary queue synchronization, enqueue/upsert, rediscovery, retry, and completion logic must not turn a cold row back into pending/deferred. If a target becomes relevant through a retained query, an operator can review its audit lineage and explicitly reactivate it; implicit reactivation would make the policy hold non-deterministic and unaudited. + +### Pause GitHub Actions at all configuration layers + +Remove `github_actions` from `supervisor.enabled_sources` and set both supervisor-source and source-level enablement false. Keep its query and all queue/history rows unchanged. No cold transition is required for GHA because no worker remains able to claim its source/platform rows. + +## Risks / Trade-offs + +- [Earliest query attribution can cold a target rediscovered by a retained term] -> Preserve full audit lineage and require explicit reactivation; do not delete the row or overwrite its original query. +- [A live or fenced row could be transitioned] -> Require coordinated source shutdown, exact row-update fences, reservation/lease checks, deterministic locks, and atomic all-or-nothing apply. +- [Completed Docker anchors could bypass the migration] -> Filter retry and periodic resolver admission by the current exact query allowlist. +- [Cold rows could inflate active backlog displays] -> Count `cold` separately and exclude it from pending/deferred operational backlog totals. +- [A policy/config change between review and apply could invalidate the cohort] -> Fence the manifest with canonical configuration, query-policy, selection, and manifest hashes. +- [Pausing GHA may miss future useful credentials] -> Preserve its entire queue and configuration for a coordinated future re-enable after explicit review. + +## Migration Plan + +1. Add schema support, policy transition APIs, exact manifest planning/apply commands, Docker resolver query filtering, cold-preserving upsert behavior, and focused tests. +2. Stop the authenticated runtime coordinately and verify sources are stopped and no active target/result/resolver fences block the selected Docker cohort. +3. Start maintenance PostgreSQL, apply the additive schema migration, and generate a bounded private DockerHub stale-query cold manifest. +4. Review aggregate counts and hashes, then apply the exact manifest atomically. Retain only the append-only database audit; remove the temporary private manifest after verification. +5. Deploy the GHA-disabled configuration and restart through the canonical runtime lifecycle. +6. Verify no cold Docker row is claimable, no stale completed anchor is resolver-claimable, cold and active counts reconcile, GHA has no child process, and the remaining pipeline is healthy. +7. Roll back by stopping sources, generating an exact reactivation manifest from unreversed cold events, applying it atomically, restoring GHA configuration if desired, and restarting canonically. + +## Open Questions + +None. diff --git a/openspec/changes/cold-policy-stale-backlog/proposal.md b/openspec/changes/cold-policy-stale-backlog/proposal.md new file mode 100644 index 0000000..60d697f --- /dev/null +++ b/openspec/changes/cold-policy-stale-backlog/proposal.md @@ -0,0 +1,23 @@ +## Why + +Removing zero-yield discovery terms stopped new discovery but left their pending and deferred targets claimable, so retired policy continued consuming scanner time. GitHub Actions also produced no strict-usable credential from either fresh work or its large retained backlog while that backlog kept growing, so it should no longer occupy a core worker. + +## What Changes + +- Add an explicit, auditable, reversible `cold` lifecycle state for policy-retired target queue rows without deleting queue history, scans, deduplication, reservations, or coverage records. +- Cold only unfenced `pending` and `deferred` rows whose exact source query is absent from the canonical configured query policy, using a reviewed offline manifest and coordinated lifecycle authority. +- Prevent completed Docker resolver anchors attributed to retired queries from being periodically reclaimed and creating new stale digest children. +- Preserve cold rows across ordinary enqueue, rediscovery, retry, and completion paths; require an explicit audited action to reactivate them. +- Pause GitHub Actions by removing it from the active core and disabling both supervisor and source configuration, while preserving its complete queue and history. + +## Capabilities + +### New Capabilities +- `target-queue-policy-holds`: Auditable cold and reactivation transitions for policy-stale target queue work, including claim exclusion and Docker resolver filtering. + +### Modified Capabilities +- `discovery-keyword-pruning`: Retired queries stop both future discovery and claimable pending/deferred work while preserving every historical authority record. + +## Impact + +The change affects PostgreSQL target queue schema and migration authority in `app/scanner_db.py` and `app/migrate_runtime_safety.py`, Docker resolver admission and query plumbing in `app/console_runner.py`, core source selection in `app/config.yaml`, focused lifecycle/configuration tests, and dashboard/status aggregation where queue states are enumerated. Deployment requires a coordinated runtime stop, schema migration, reviewed cold manifest application, and canonical restart. No target, scan, finding, credential, result, deduplication, or coverage row is deleted. diff --git a/openspec/changes/cold-policy-stale-backlog/specs/discovery-keyword-pruning/spec.md b/openspec/changes/cold-policy-stale-backlog/specs/discovery-keyword-pruning/spec.md new file mode 100644 index 0000000..1c3484a --- /dev/null +++ b/openspec/changes/cold-policy-stale-backlog/specs/discovery-keyword-pruning/spec.md @@ -0,0 +1,20 @@ +## MODIFIED Requirements + +### Requirement: Operational and historical authority is preserved +Keyword retirement SHALL stop future discovery and SHALL permit existing unfenced pending/deferred targets attributed to retired exact queries to enter an audited, reversible cold state without deleting or rewriting historical authority. + +#### Scenario: Dedicated source sentinels remain +- **WHEN** archive and gist source rotations are loaded +- **THEN** `gharchive`, `gharchive-files`, and `gists` SHALL remain as their sole configured query tokens + +#### Scenario: Persisted rotation index remains valid +- **WHEN** an existing query index exceeds a shortened query list +- **THEN** normal modulo-based rotation SHALL select a valid configured query without a state-file edit + +#### Scenario: Existing backlog is preserved but held +- **WHEN** a previously admitted unfenced target is attributed to a retired exact query and selected by reviewed policy +- **THEN** its queue row SHALL remain present with all attribution, retry, deduplication, scan, reservation, and coverage history preserved while its status becomes unclaimable `cold` + +#### Scenario: Historical records remain unchanged +- **WHEN** a stale-query cold transition is applied +- **THEN** existing target scans, findings, credentials, keycheck results, completed queue rows, and coverage records SHALL NOT be deleted or rewritten diff --git a/openspec/changes/cold-policy-stale-backlog/specs/target-queue-policy-holds/spec.md b/openspec/changes/cold-policy-stale-backlog/specs/target-queue-policy-holds/spec.md new file mode 100644 index 0000000..757bc42 --- /dev/null +++ b/openspec/changes/cold-policy-stale-backlog/specs/target-queue-policy-holds/spec.md @@ -0,0 +1,78 @@ +## ADDED Requirements + +### Requirement: Cold queue rows are not claimable +The system SHALL represent policy-held target work with `target_queue.status='cold'`, and no scanner, legacy queue consumer, or Docker resolver SHALL claim a cold row. + +#### Scenario: Scanner checks active backlog +- **WHEN** a queue row has status `cold` +- **THEN** the row SHALL be excluded from pending, due-deferred, retry, and periodic resolver admission + +#### Scenario: Worker completes ordinary work +- **WHEN** a scan result is ingested +- **THEN** its queue disposition SHALL remain limited to ordinary lifecycle outcomes and SHALL NOT create a cold status + +### Requirement: Policy holds are exact and audited +The system SHALL transition queue rows into or out of cold state only through a reviewed, hash-fenced, append-only policy event under stopped-source lifecycle authority. + +#### Scenario: Eligible stale row is held +- **WHEN** an exact manifest entry still matches an unfenced `pending` or `deferred` row whose source query is absent from canonical policy +- **THEN** the system SHALL atomically record the audit event and change only the row status and update timestamp to `cold` + +#### Scenario: Selected row has an active fence +- **WHEN** a selected row has a queue lease, resolver lease, current claim, active result reservation, or submitted content lease +- **THEN** the policy apply SHALL fail closed without partially applying the manifest + +#### Scenario: Exact apply is repeated +- **WHEN** the same reviewed manifest is applied again after a successful transition +- **THEN** the system SHALL report an idempotent duplicate without creating a second state transition + +### Requirement: Cold state is explicitly reversible +The system SHALL retain the exact prior queue status in policy audit history and SHALL require a reviewed reactivation manifest to restore a cold row. + +#### Scenario: Cold row is rediscovered normally +- **WHEN** enqueue, synchronization, rediscovery, retry, or completion logic encounters an existing cold row +- **THEN** the row SHALL remain cold and its historical attribution SHALL remain unchanged + +#### Scenario: Reviewed cold event is reactivated +- **WHEN** an exact reactivation manifest references an unreversed cold event and all row fences still match +- **THEN** the system SHALL restore the audited prior `pending` or `deferred` status and append a linked reactivation event + +### Requirement: Stale-query selection follows canonical policy +The system SHALL evaluate query staleness using case-sensitive exact source, platform, and query policy derived from canonical configuration. + +#### Scenario: Configured query remains active +- **WHEN** a queue row's exact source/platform/query triple remains configured +- **THEN** automatic stale-policy planning SHALL NOT select the row + +#### Scenario: Attribution cannot be classified safely +- **WHEN** query attribution is null, blank, operational, non-rotation, or belongs to an unknown source/platform pair +- **THEN** automatic planning SHALL skip and report the row rather than inferring retirement + +#### Scenario: Initial Docker stale cohort is planned +- **WHEN** DockerHub policy planning is scoped to platform `docker` +- **THEN** it SHALL include eligible pending/deferred resolver anchors and digest rows attributed to removed exact queries and SHALL expose only aggregate counts plus non-sensitive manifest fields + +### Requirement: Docker resolvers honor current query policy +The system SHALL prevent completed or retryable Docker resolver anchors attributed to retired non-null queries from generating new digest queue rows. + +#### Scenario: Periodic stale anchor becomes due +- **WHEN** a completed Docker resolver anchor is due but its exact query is not in the configured DockerHub allowlist +- **THEN** resolver admission SHALL leave the anchor unclaimed + +#### Scenario: Configured anchor becomes due +- **WHEN** a Docker resolver anchor's exact query remains configured and all ordinary claim fences pass +- **THEN** resolver admission SHALL preserve the existing retry and periodic behavior + +### Requirement: Source pause preserves backlog authority +Pausing a source SHALL remove its worker from the active core without deleting or rewriting that source's queue or historical records. + +#### Scenario: GitHub Actions is paused +- **WHEN** canonical runtime configuration is loaded after this change +- **THEN** GitHub Actions SHALL be absent from the supervisor core and disabled at both supervisor-source and source configuration layers while its configured query and persisted backlog remain intact + +### Requirement: Cold work is separately observable +Operational queue summaries SHALL report cold rows separately and SHALL NOT include them in active pending or deferred backlog counts. + +#### Scenario: Queue state is summarized +- **WHEN** an operator inspects canonical target queue status +- **THEN** the summary SHALL expose a distinct cold count without representing those rows as retryable or claimable work diff --git a/openspec/changes/cold-policy-stale-backlog/tasks.md b/openspec/changes/cold-policy-stale-backlog/tasks.md new file mode 100644 index 0000000..b80de7f --- /dev/null +++ b/openspec/changes/cold-policy-stale-backlog/tasks.md @@ -0,0 +1,21 @@ +## 1. Durable Policy State + +- [x] 1.1 Add the `cold` target queue lifecycle state, append-only policy-event schema, indexes, additive migration marker, and schema validation coverage. +- [x] 1.2 Implement atomic, fenced, idempotent cold and reactivation database transitions that preserve all non-lifecycle queue authority. + +## 2. Reviewed Policy Operations + +- [x] 2.1 Implement exact canonical query-policy derivation and privacy-safe stale-row/reversal manifest planning. +- [x] 2.2 Add stopped-source migration CLI dry-run/apply paths with manifest/config/policy/selection hash validation and bounded deterministic scope. + +## 3. Runtime Enforcement + +- [x] 3.1 Preserve cold rows across ordinary queue enqueue/synchronization and expose cold separately in canonical queue summaries. +- [x] 3.2 Pass configured DockerHub query policy into retry/periodic resolver admission and exclude retired-query anchors. +- [x] 3.3 Remove GitHub Actions from the active core and disable both supervisor and source layers without altering its query or persisted backlog. + +## 4. Verification And Deployment + +- [x] 4.1 Add focused unit and PostgreSQL integration tests for claim exclusion, exact selection, fenced transitions, idempotency, reversal, Docker resolver filtering, cold preservation, observability, and GHA pause. +- [x] 4.2 Run targeted test suites and strict OpenSpec validation with no forbidden application bytecode artifacts. +- [x] 4.3 Coordinately stop runtime, apply the additive schema migration and reviewed Docker stale-query cold manifest, remove temporary artifacts, restart canonically, and verify active backlog and pipeline health. diff --git a/openspec/changes/expand-openai-ecosystem-discovery/.openspec.yaml b/openspec/changes/expand-openai-ecosystem-discovery/.openspec.yaml new file mode 100644 index 0000000..701445b --- /dev/null +++ b/openspec/changes/expand-openai-ecosystem-discovery/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-08-26 diff --git a/openspec/changes/expand-openai-ecosystem-discovery/design.md b/openspec/changes/expand-openai-ecosystem-discovery/design.md new file mode 100644 index 0000000..3002cb1 --- /dev/null +++ b/openspec/changes/expand-openai-ecosystem-discovery/design.md @@ -0,0 +1,78 @@ +## Context + +The exact `openai` rollout proved that bounded source queries can restore unseen credential supply without changing scanner or keycheck authority: 18 DockerHub targets produced 24 genuinely new OpenAI credentials, all with explicit terminal API outcomes. None were usable, so increasing that broad query is not justified. Historical production attribution instead shows usable OpenAI outcomes behind narrower agent, chatbot, and conversation ecosystems, while source semantics differ enough that one shared keyword list is inefficient. + +The existing exact-query override mechanism already validates and applies `pages`, `per_page`, and `max_targets`. This change can therefore remain configuration-only plus focused contract tests. Runtime configuration is immutable-authority covered, so deployment must use coordinated stop/start and must not modify persisted query state directly. + +## Goals / Non-Goals + +**Goals:** +- Test nine high-signal OpenAI integration and deployable-application queries across the sources where their search semantics fit. +- Bound discovery admission as well as scan claims for every new query. +- Run each new query once promptly after deployment without directly editing query state. +- Attribute the canary through durable scans, candidates, provider results, and projections. +- Keep or remove each query based on its own measured useful yield and operational cost. + +**Non-Goals:** +- Expanding broad generic terms such as `gpt`, `llm`, `chatgpt`, `ai`, or `model`. +- Increasing global scan concurrency, source worker counts, or updated-target promotion limits. +- Changing detector routing, OpenAI checker classification, known-credential caching, or projection semantics. +- Re-enabling inactive broad sources or guaranteeing a valid funded credential. + +## Decisions + +### Use source-specific first-wave queries + +The first wave is: +- GitHub: `OPENAI_API_KEY`, `api.openai.com`, `openai-agents`. +- GitLab: `openai-api`, `openai-agents`, `librechat`. +- DockerHub: `librechat`, `lobechat`, `openai-proxy`. + +GitHub can search README content, so direct environment and endpoint signatures are appropriate. GitLab project search is metadata-oriented, so branded slug terms are used. DockerHub searches repository metadata and then resolves immutable image digests, so deployable project and proxy names are used. + +Alternative: add the same list to all sources. Rejected because it lengthens every rotation and ignores source search semantics. Alternative: expand historically broad terms. Rejected because those cohorts produced volume without usable OpenAI outcomes. + +### Bound every query independently + +Bounds are: +- GitHub `OPENAI_API_KEY`: `pages=1`, `per_page=25`, `max_targets=5`. +- GitHub `api.openai.com`: `1/25/5`. +- GitHub `openai-agents`: `1/50/5`. +- GitLab `openai-api`: `1/50/5`. +- GitLab `openai-agents`: `1/50/5`. +- GitLab `librechat`: `1/25/5`. +- DockerHub `librechat`, `lobechat`, and `openai-proxy`: each `2/10/10`. + +Page and page-size limits bound fetched/admitted identities; `max_targets` separately bounds claims in the active cycle. The existing global three-slot limit, GitLab one-updated-target-per-cycle cap, 24-hour update cooldown, and Docker digest requirement remain unchanged. + +### Place the wave at each stopped source's current rotation index + +After a coordinated runtime stop, capture each source's persisted `query_index` and insert that source's three-query block at the same index. Restarting then exercises the block naturally. Successful discovery advances through the block; failures and backlog-only work retain the current query under existing semantics. State files are never edited. + +Alternative: append and wait for a full rotation. Rejected because DockerHub cycles can be long and attribution would be delayed. Alternative: edit persisted state. Rejected because state is runtime authority and direct edits would weaken recovery evidence. + +### Evaluate individual query funnels + +The canary reports fetched, new/updated admissions, durable scan outcomes, findings, genuinely new OpenAI credentials, `api_check` versus `cached_status`, explicit provider outcomes, first-alive/usable counts, projection drain, and runtime health. Aggregate volume alone cannot justify retention. + +## Risks / Trade-offs + +- [README signatures produce placeholders] -> Count genuinely new credentials and explicit API outcomes; remove queries with only placeholder/dead yield. +- [A query admits more work than its claim cap] -> Keep `pages × per_page` small and inspect the exact query-attributed queue until terminal. +- [Docker scans are expensive or inaccessible] -> Cap each query at 20 repositories and 10 claims, retain digest authority, and classify registry failures separately. +- [Three added terms lengthen source rotations] -> Keep only terms that add distinct credential or usable yield after the canary. +- [Config deployment triggers authority fail-close] -> Stop coordinately before editing and restart only through `start_runtime.ps1`. + +## Migration Plan + +1. Add focused tests for exact source membership, uniqueness, and all nine bounds. +2. Coordinately stop the runtime and capture persisted source indices. +3. Insert each three-query block at its source's current index; do not edit state files. +4. Run focused and full regression suites plus strict OpenSpec validation with bytecode writes disabled. +5. Start through the authoritative runtime script and verify PostgreSQL, pipeline, core sources, keychecks, and recorder. +6. Observe each exact query cohort through terminal scans and keycheck projection, then retain or remove each query based on measured evidence. +7. Roll back any low-value query by removing that query and override during a coordinated stop/start. No schema or data rollback is required. + +## Open Questions + +None. A second wave remains gated on this canary's per-query useful-yield evidence. diff --git a/openspec/changes/expand-openai-ecosystem-discovery/proposal.md b/openspec/changes/expand-openai-ecosystem-discovery/proposal.md new file mode 100644 index 0000000..c8272d0 --- /dev/null +++ b/openspec/changes/expand-openai-ecosystem-discovery/proposal.md @@ -0,0 +1,24 @@ +## Why + +The bounded exact `openai` canary restored fresh credential supply but produced no usable credentials, while historical production evidence shows that narrower ecosystem and integration terms can reach different cohorts. A small source-specific keyword wave can test those higher-signal surfaces without expanding broad generic discovery or scan concurrency. + +## What Changes + +- Add three source-specific OpenAI ecosystem queries to each of GitHub, GitLab, and DockerHub. +- Apply exact per-query page, page-size, and claim bounds so every new query is independently constrained. +- Preserve existing query rotation, target deduplication, revision-aware rescan limits, Docker digest authority, scan concurrency, and keycheck behavior. +- Run one controlled production canary per new query and measure discovery, scans, new OpenAI credentials, explicit API outcomes, and usable yield. +- Retain, revise, or remove individual queries based on measured bounded evidence rather than fetched volume. + +## Capabilities + +### New Capabilities +- `openai-ecosystem-discovery`: Source-specific bounded discovery for OpenAI integration signatures and deployable ecosystem projects, with per-query canary evidence. + +### Modified Capabilities + +None. + +## Impact + +The change affects `app/config.yaml`, focused query-configuration tests, authority-managed source rotation, and production canary operations. Existing allowlisted query override code is reused unchanged. There is no schema migration, new dependency, detector change, global concurrency increase, or credential recheck policy change. diff --git a/openspec/changes/expand-openai-ecosystem-discovery/specs/openai-ecosystem-discovery/spec.md b/openspec/changes/expand-openai-ecosystem-discovery/specs/openai-ecosystem-discovery/spec.md new file mode 100644 index 0000000..aea8337 --- /dev/null +++ b/openspec/changes/expand-openai-ecosystem-discovery/specs/openai-ecosystem-discovery/spec.md @@ -0,0 +1,73 @@ +## ADDED Requirements + +### Requirement: Source-specific OpenAI ecosystem queries +The system SHALL include the approved first-wave OpenAI ecosystem queries only in the source rotations whose search semantics match those queries. + +#### Scenario: GitHub searches integration signatures +- **WHEN** GitHub reaches the first-wave positions in its normal rotation +- **THEN** it SHALL search `OPENAI_API_KEY`, `api.openai.com`, and `openai-agents` + +#### Scenario: GitLab searches branded project metadata +- **WHEN** GitLab reaches the first-wave positions in its normal rotation +- **THEN** it SHALL search `openai-api`, `openai-agents`, and `librechat` + +#### Scenario: DockerHub searches deployable ecosystems +- **WHEN** DockerHub reaches the first-wave positions in its normal rotation +- **THEN** it SHALL search `librechat`, `lobechat`, and `openai-proxy` + +#### Scenario: Broad generic expansion is excluded +- **WHEN** the first-wave configuration is evaluated +- **THEN** it SHALL NOT add new broad variants of `gpt`, `llm`, `chatgpt`, `ai`, or `model` + +### Requirement: Independent bounded query policies +Each first-wave query SHALL have an exact allowlisted override that bounds both discovery volume and scan claims without changing source defaults or other query policies. + +#### Scenario: GitHub signature queries are bounded +- **WHEN** GitHub builds arguments for `OPENAI_API_KEY` or `api.openai.com` +- **THEN** it SHALL use one page of 25 results and claim at most five targets + +#### Scenario: GitHub agent query is bounded +- **WHEN** GitHub builds arguments for `openai-agents` +- **THEN** it SHALL use one page of 50 results and claim at most five targets + +#### Scenario: GitLab queries are bounded +- **WHEN** GitLab builds arguments for a first-wave query +- **THEN** it SHALL use one page, the configured 25- or 50-result page size, and claim at most five targets + +#### Scenario: DockerHub queries are bounded +- **WHEN** DockerHub builds arguments for a first-wave query +- **THEN** it SHALL use at most two pages of ten repositories and claim at most ten targets + +#### Scenario: Existing authority limits remain unchanged +- **WHEN** any first-wave query runs +- **THEN** global scan concurrency, revision-aware promotion limits, cooldowns, and Docker digest requirements SHALL remain authoritative + +### Requirement: Natural rotation and failure behavior +The first-wave queries SHALL use the existing persisted source rotation without direct query-state modification. + +#### Scenario: Successful query advances +- **WHEN** a first-wave discovery cycle completes successfully +- **THEN** the source SHALL advance through the existing persisted rotation semantics + +#### Scenario: Failed query is retained +- **WHEN** first-wave discovery fails before successful completion +- **THEN** the source SHALL retain that query according to existing failure semantics + +#### Scenario: Backlog work does not masquerade as discovery +- **WHEN** a source drains existing backlog while a first-wave query is current +- **THEN** canary attribution SHALL distinguish backlog-only cycles from the actual discovery cycle + +### Requirement: Per-query end-to-end canary evidence +Operators SHALL evaluate each first-wave query through durable source, scan, candidate, provider-result, and projection evidence without exposing targets or credential values. + +#### Scenario: Query cohort reaches terminal accounting +- **WHEN** a first-wave query admits new or updated targets +- **THEN** operators SHALL verify queue dispositions, scan completion, candidate completion, and projection drain for that exact query cohort + +#### Scenario: Useful yield is measured separately +- **WHEN** a first-wave cohort creates OpenAI candidates +- **THEN** genuinely new credentials and their explicit API outcomes SHALL be reported separately from cached-known occurrences + +#### Scenario: Retention decision uses measured value +- **WHEN** the bounded canary is complete +- **THEN** each query SHALL be retained, revised, or removed using its distinct credential yield, usable outcomes, and operational error cost rather than fetched count alone diff --git a/openspec/changes/expand-openai-ecosystem-discovery/tasks.md b/openspec/changes/expand-openai-ecosystem-discovery/tasks.md new file mode 100644 index 0000000..99e50b4 --- /dev/null +++ b/openspec/changes/expand-openai-ecosystem-discovery/tasks.md @@ -0,0 +1,17 @@ +## 1. Source-specific configuration + +- [x] 1.1 Coordinately stop the runtime, capture persisted source query indices, and insert each three-query block at its current index. +- [x] 1.2 Add exact allowlisted bounds for all nine queries without changing source-wide defaults or existing safety policies. +- [x] 1.3 Add focused configuration tests for source membership, uniqueness, exact bounds, and excluded broad expansion. + +## 2. Verification + +- [x] 2.1 Run focused and full regression suites with bytecode writes disabled. +- [x] 2.2 Run strict OpenSpec validation and verify implementation against the artifacts. + +## 3. Production canary + +- [x] 3.1 Start the authority-managed runtime and verify PostgreSQL, pipeline, recorder, keychecks, and all core sources. +- [x] 3.2 Observe one actual discovery cycle for each first-wave query and verify configured fetch and claim bounds. +- [x] 3.3 Follow every exact query cohort through queue disposition, scan/candidate completion, provider outcomes, and projection drain. +- [x] 3.4 Record per-query useful-yield evidence, retain or remove low-value terms, and update the parking lot. diff --git a/openspec/changes/fix-custom-provider-detector-compatibility/.openspec.yaml b/openspec/changes/fix-custom-provider-detector-compatibility/.openspec.yaml new file mode 100644 index 0000000..d28e909 --- /dev/null +++ b/openspec/changes/fix-custom-provider-detector-compatibility/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-17 diff --git a/openspec/changes/fix-custom-provider-detector-compatibility/design.md b/openspec/changes/fix-custom-provider-detector-compatibility/design.md new file mode 100644 index 0000000..234d536 --- /dev/null +++ b/openspec/changes/fix-custom-provider-detector-compatibility/design.md @@ -0,0 +1,51 @@ +## Context + +TruffleHog custom detector entries combine every regex in one detector through match permutation, so multiple regex fields are conjunctive rather than alternative. The Xai and ZaiGLM policies each place context-before and context-after patterns in one detector, making ordinary one-direction matches disappear. Existing tests compile each expression with Python and use `any(...)`, which does not exercise TruffleHog's actual configuration semantics. + +The scanner already canonicalizes `CustomRegex` findings from `ExtraData.name`, and downstream candidate routing and keycheckers depend on the canonical names `Xai` and `ZaiGLM`. + +## Goals / Non-Goals + +**Goals:** +- Make both context directions independently executable for Xai and ZAI/GLM. +- Preserve canonical persisted detector names and existing keycheck routing. +- Exercise the complete policy through the real TruffleHog CLI without network verification. +- Keep the policy compatible with the pinned fork on Windows and Linux. + +**Non-Goals:** +- Change provider verification endpoints or keycheck behavior. +- Add new key formats or broaden the existing regex bounds. +- Require TruffleHog to be installed for pure unit-test environments. +- Perform live provider requests or use real credentials. + +## Decisions + +### Split alternatives into uniquely named detectors + +Keep the context-before expression under the existing canonical name and move the context-after expression into `XaiContextAfter` or `ZaiGLMContextAfter`. Duplicate names are not used because the custom detector registry can collapse same-name entries; a synthetic CLI probe confirmed unique names execute both alternatives. + +Alternative considered: combine both alternatives into one regex. This was rejected because TruffleHog takes the first capture group as the secret, and Go regex does not support branch-reset groups needed to keep one capture position across both directions. + +### Canonicalize compatibility aliases centrally + +Extend custom detector normalization with a small alias map from the context-after names to `Xai` and `ZaiGLM`. This keeps candidate routing, keychecker gates, dashboards, deduplication, and persisted detector values unchanged. + +Alternative considered: teach every downstream consumer the new names. This would widen the change and create divergent persisted identities for one provider. + +### Add an optional real-CLI regression test + +The test resolves TruffleHog from an explicit environment override, the existing Windows path, or `PATH`. When available, it scans deterministic high-entropy synthetic fixtures with the complete production YAML, `--no-verification`, and `--no-update`, then checks canonical finding names and candidate services. It skips only when no binary is available, while pure tests continue to validate policy structure and alias normalization everywhere. + +## Risks / Trade-offs + +- [Risk] A CI environment without TruffleHog can skip the integration gate. -> Keep structural unit coverage and run the real-CLI test in Windows and Linux release jobs. +- [Risk] Native Xai can duplicate the custom result for its exact 80-character format. -> Existing candidate identity deduplication remains authoritative; the regression fixture uses a shorter supported overlay form to isolate custom behavior. +- [Risk] A future TruffleHog release can change custom detector output fields or flags. -> The real-CLI test asserts `CustomRegex`, `ExtraData.name`, normalization, and candidate routing as one contract. + +## Migration Plan + +Deploy the YAML, scanner alias map, and tests together, rebuild the runtime image, then run the offline compatibility test against both worker binaries before enabling scans. Rollback restores the prior YAML and alias map; no persisted data migration is required. + +## Open Questions + +None. diff --git a/openspec/changes/fix-custom-provider-detector-compatibility/proposal.md b/openspec/changes/fix-custom-provider-detector-compatibility/proposal.md new file mode 100644 index 0000000..d29092a --- /dev/null +++ b/openspec/changes/fix-custom-provider-detector-compatibility/proposal.md @@ -0,0 +1,23 @@ +## Why + +The Xai and ZaiGLM custom detector policies model alternative context directions as separate regex entries, but TruffleHog combines entries within one detector as an AND condition. This silently prevents the broader Xai overlay and ordinary ZAI/GLM source detection, while the current unit tests incorrectly model the entries as OR alternatives. + +## What Changes + +- Express each alternative Xai and ZAI/GLM context direction as an independently executable custom detector while preserving the normalized provider names consumed by routing and keychecks. +- Add an offline CLI compatibility test that runs the configured TruffleHog binary with the complete custom detector policy and synthetic high-entropy fixtures. +- Verify that custom findings normalize and route to the expected Xai and ZAI keycheck services without performing provider verification requests. +- Keep the existing native detector, result bundle, candidate, and keycheck contracts unchanged. + +## Capabilities + +### New Capabilities +- `custom-provider-detection-compatibility`: Defines executable compatibility requirements for external custom detector policies and their normalized keycheck routing. + +### Modified Capabilities + +None. + +## Impact + +The change affects `app/trufflehog-custom-detectors.yaml`, scanner finding normalization/routing tests, and the provider detector compatibility test surface. It introduces no production API, schema, dependency, or persisted-data changes. diff --git a/openspec/changes/fix-custom-provider-detector-compatibility/specs/custom-provider-detection-compatibility/spec.md b/openspec/changes/fix-custom-provider-detector-compatibility/specs/custom-provider-detection-compatibility/spec.md new file mode 100644 index 0000000..cb38914 --- /dev/null +++ b/openspec/changes/fix-custom-provider-detector-compatibility/specs/custom-provider-detection-compatibility/spec.md @@ -0,0 +1,30 @@ +## ADDED Requirements + +### Requirement: Alternative provider contexts execute independently +The custom detector policy SHALL detect supported Xai and ZAI/GLM credentials when provider context appears either before or after the credential, without requiring both context directions in one input chunk. + +#### Scenario: Provider context appears before the credential +- **WHEN** an offline scan processes a bounded synthetic credential preceded by its supported provider context +- **THEN** the policy emits one corresponding custom provider finding + +#### Scenario: Provider context appears after the credential +- **WHEN** an offline scan processes a bounded synthetic credential followed by its supported provider context +- **THEN** the policy emits one corresponding custom provider finding + +### Requirement: Alternative detector names normalize canonically +The scanner MUST normalize all compatibility-only custom detector aliases to the existing canonical `Xai` or `ZaiGLM` detector identity before persistence and candidate extraction. + +#### Scenario: Context-after alias is emitted +- **WHEN** TruffleHog emits `CustomRegex` with a context-after compatibility name in `ExtraData.name` +- **THEN** the scanner retains `CustomRegex` as the original detector and exposes the canonical provider detector name downstream + +### Requirement: Compatibility is tested through the real CLI +The compatibility suite SHALL run the complete configured custom detector policy through an available TruffleHog executable using deterministic synthetic credentials, disabled verification, and disabled update checks. + +#### Scenario: Compatible executable is available +- **WHEN** a configured Windows or Linux TruffleHog executable scans the compatibility fixtures +- **THEN** both context directions produce canonical findings and route to the expected keycheck candidate services without network verification + +#### Scenario: Executable is unavailable +- **WHEN** no TruffleHog executable is available in a general unit-test environment +- **THEN** the real-CLI test is explicitly skipped while policy-structure and normalization unit tests still execute diff --git a/openspec/changes/fix-custom-provider-detector-compatibility/tasks.md b/openspec/changes/fix-custom-provider-detector-compatibility/tasks.md new file mode 100644 index 0000000..a080d5c --- /dev/null +++ b/openspec/changes/fix-custom-provider-detector-compatibility/tasks.md @@ -0,0 +1,16 @@ +## 1. Detector Policy + +- [x] 1.1 Split Xai context-before and context-after alternatives into uniquely named single-regex detector entries. +- [x] 1.2 Split ZaiGLM context-before and context-after alternatives into uniquely named single-regex detector entries. +- [x] 1.3 Canonicalize the compatibility-only detector names before persistence and candidate extraction. + +## 2. Compatibility Coverage + +- [x] 2.1 Add pure unit coverage for detector policy structure and canonical alias normalization. +- [x] 2.2 Add an optional real-TruffleHog CLI regression test for both context directions and candidate routing. + +## 3. Verification + +- [x] 3.1 Run the focused provider and scanner unit tests. +- [x] 3.2 Run the real CLI compatibility test with the current Windows binary and a Linux binary where available. +- [x] 3.3 Validate the OpenSpec change and confirm the patch is formatting-clean. diff --git a/openspec/changes/fix-keycheck-accounting-visibility/.openspec.yaml b/openspec/changes/fix-keycheck-accounting-visibility/.openspec.yaml new file mode 100644 index 0000000..38f7628 --- /dev/null +++ b/openspec/changes/fix-keycheck-accounting-visibility/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-06-22 diff --git a/openspec/changes/fix-keycheck-accounting-visibility/design.md b/openspec/changes/fix-keycheck-accounting-visibility/design.md new file mode 100644 index 0000000..c2264d6 --- /dev/null +++ b/openspec/changes/fix-keycheck-accounting-visibility/design.md @@ -0,0 +1,82 @@ +## Context + +The scanner runs continuously and appends findings to `found_secrets.jsonl` and `scanner_active.db`. Keycheckers classify provider credentials into current-state status files under `runtime/keychecks//` and opportunistically write rows to `keycheck_results` for dashboard visibility. + +The current behavior has several failure modes: + +- Hourly keychecks can replay a multi-GB `found_secrets.jsonl`, delaying or blocking later services in the batch. +- Some checkers implement custom input readers, so global tail behavior does not apply consistently. +- Known keys are skipped before writing a new occurrence row, so repeated source/query hits for an already alive key are invisible in DB/dashboard history. +- Per-result DB writes compete with scanner writes and can fail under SQLite locks, causing file state and DB observations to diverge. +- Dashboard views mix file current-state, DB current-state, historical occurrences, and provider-specific usable access semantics. + +## Goals / Non-Goals + +**Goals:** + +- Make current-state status files remain the authoritative per-service status store. +- Record historical occurrences for known keys without forcing API rechecks. +- Make hourly keychecks bounded and incremental enough to keep up with continuous scanning. +- Make DB writes resilient and explainable, with visible lag/error indicators. +- Give operators dashboard presets for current usable keys, historical usable occurrences, no-quota/limited states, and pipeline health. + +**Non-Goals:** + +- Replace provider status files with the SQLite DB. +- Revalidate every known alive key on every hourly cycle. +- Guarantee zero SQLite lock contention while scanner writers are active. +- Redesign detector extraction or provider-specific validity semantics beyond accounting and visibility. + +## Decisions + +1. Keep status files as current-state truth. + + Rationale: checkers already compact/move keys between `Alive`, `NoBalance`, `Dead`, `Network`, and related files. Replacing this would be riskier than making DB observations catch up. + + Alternative considered: make `keycheck_results` the source of truth. Rejected because the active DB is large, frequently locked, and dashboard reads must not block scanner writes. + +2. Introduce occurrence recording for skipped known keys. + + When a checker sees a candidate that is already known in status files or checked files, it should write a lightweight occurrence row with the cached status, source line, and finding attribution. It must not call provider APIs unless retry flags or recheck flags require it. + + Alternative considered: only record fresh API checks. Rejected because this hides repeated source/query yield for already alive keys. + +3. Use a shared bounded/incremental input reader for all checkers. + + Checkers should use common reader helpers rather than hand-rolled full-file loops. The minimum implementation can use a tail window; the target implementation should store per-service high-watermark offsets so hourly runs neither replay old data nor miss data outside a fixed tail window. + + Alternative considered: keep reading full JSONL and rely on skip sets. Rejected because the input is already multi-GB and causes long stalls. + +4. Centralize DB recording or make per-checker DB writes lock-tolerant. + + The preferred direction is batching result/occurrence rows through `keycheck_runner` after each service completes. A smaller intermediate step is to avoid schema initialization on every single keycheck write and retry lock failures with bounded backoff. + + Alternative considered: ignore DB write failures because files are authoritative. Rejected because dashboard and source attribution depend on DB visibility. + +5. Distinguish access tiers from raw provider statuses. + + Dashboard should classify provider statuses into operator-facing tiers such as `usable_llm`, `alive_unproven_llm`, `no_quota`, and `quota_limited`, while still allowing exact status filtering for values like `BEDROCK`, `VERTEX`, `VALID_RATE_LIMITED`, and `ALIVE`. + +## Risks / Trade-offs + +- Cached occurrence rows could be mistaken for fresh provider rechecks -> Label occurrence rows with a result source such as `cached_status` versus `api_check`. +- Tail windows can miss old-but-newly-unchecked lines after downtime or file rewrites -> Prefer high-watermark offsets and detect file truncation/rotation. +- DB batching can still fail if SQLite is locked for extended periods -> Keep files authoritative and surface DB write lag/errors in dashboard. +- Provider semantics differ: e.g. Gemini `VALID_RATE_LIMITED`, AWS `BEDROCK`, GCP `VERTEX` -> Keep exact statuses available and use access tiers only as an additional view. +- Rechecking known alive keys too often can spend quota or trigger provider limits -> Occurrence recording must not imply revalidation. + +## Migration Plan + +1. Add shared input high-watermark/tail behavior to all keycheckers, starting with hand-rolled readers. +2. Add cached occurrence recording for known keys using existing status maps. +3. Add batched DB write path or harden lock retry behavior. +4. Update dashboard to show file current-state, DB observation freshness, and historical/current presets separately. +5. Backfill/repair missing occurrence attribution from current status files and recent findings where safe. + +Rollback: disable cached occurrence writes and fall back to existing status-file behavior; status files remain unchanged. + +## Open Questions + +- Should high-watermark state live in `runtime/keychecks//state.json` or a shared `runtime/state/keycheck_offsets.json`? +- Should cached occurrences be written for every repeated finding or deduped per service/key/finding/source per day? +- Which dashboard panel should be considered the primary operator view: file current-state or DB latest-current-state? diff --git a/openspec/changes/fix-keycheck-accounting-visibility/proposal.md b/openspec/changes/fix-keycheck-accounting-visibility/proposal.md new file mode 100644 index 0000000..0727cd1 --- /dev/null +++ b/openspec/changes/fix-keycheck-accounting-visibility/proposal.md @@ -0,0 +1,27 @@ +## Why + +Keycheck current-state files, DB observations, and dashboard views can diverge, making it unclear whether usable provider keys are still being found and which sources produced them. This is urgent because the scanner is running continuously, but large input files, known-key skipping, and SQLite lock behavior can hide fresh usable findings from operator-visible stats. + +## What Changes + +- Add reliable keycheck accounting for fresh checks and known-key occurrences. +- Track current-state file counts separately from DB-observed validation rows. +- Ensure hourly keychecks process recent findings efficiently without replaying multi-GB JSONL inputs from the beginning. +- Preserve source/query/finding attribution even when a key was already classified as alive, dead, no-balance, or limited. +- Surface keycheck pipeline health, write lag, skipped-known counts, and usable/no-quota status in dashboard views. +- Reduce DB lock impact on keycheck result recording so file state and DB state remain explainably consistent. + +## Capabilities + +### New Capabilities +- `keycheck-accounting`: Defines reliable current-state, historical occurrence, and dashboard visibility behavior for keycheck results. + +### Modified Capabilities + +## Impact + +- `app/keycheck_runner.py` and `app/keycheckers/*`: keycheck execution, input reading, skip behavior, and result recording. +- `app/scanner_db.py`: keycheck DB writes, lock handling, and occurrence recording. +- `app/dashboard.py`: operator-facing keycheck and usable-key reporting. +- Runtime files under `runtime/keychecks/`: current-state status files remain authoritative but gain clearer relationship to DB observations. +- Runtime DB `scanner_active.db`: keycheck rows and attribution semantics become more complete and auditable. diff --git a/openspec/changes/fix-keycheck-accounting-visibility/specs/keycheck-accounting/spec.md b/openspec/changes/fix-keycheck-accounting-visibility/specs/keycheck-accounting/spec.md new file mode 100644 index 0000000..eac72c6 --- /dev/null +++ b/openspec/changes/fix-keycheck-accounting-visibility/specs/keycheck-accounting/spec.md @@ -0,0 +1,76 @@ +## ADDED Requirements + +### Requirement: Current-state status files remain authoritative +The system SHALL keep per-service keycheck status files as the authoritative current-state classification for keys. + +#### Scenario: Key status changes after recheck +- **WHEN** a checker revalidates a key and receives a new status +- **THEN** the key MUST be removed from other status files for that service and written to the status file for the new status + +#### Scenario: Dashboard compares files and DB +- **WHEN** dashboard displays keycheck totals +- **THEN** it MUST be clear whether each count comes from current-state status files or from DB observation rows + +### Requirement: Known key occurrences are recorded +The system SHALL record an occurrence when a checker sees a candidate that is already known in service status files or checked files. + +#### Scenario: Known alive key appears in a new finding +- **WHEN** a key already classified as alive appears in a new scanner finding +- **THEN** the system MUST record the new source/query/finding occurrence without requiring a provider API recheck + +#### Scenario: Known dead key appears in a new finding +- **WHEN** a key already classified as dead appears in a new scanner finding +- **THEN** the system MUST record the new source/query/finding occurrence with a cached dead status + +#### Scenario: Cached occurrence is distinguishable from API recheck +- **WHEN** an occurrence row is written without calling the provider API +- **THEN** the row MUST indicate that the status came from cached current-state classification + +### Requirement: Keycheck input processing is bounded and consistent +The system SHALL avoid replaying the full scanner JSONL input on every hourly keycheck run. + +#### Scenario: Hourly keychecks run on a multi-GB input file +- **WHEN** `found_secrets.jsonl` is large +- **THEN** each checker MUST process only a bounded recent range or an incremental range since its last processed offset + +#### Scenario: Checker has a custom input loop +- **WHEN** a checker reads scanner findings +- **THEN** it MUST use shared keycheck input-reading behavior or implement equivalent high-watermark/tail semantics + +#### Scenario: Input file rotates or shrinks +- **WHEN** a stored high-watermark offset is larger than the current input file size +- **THEN** the system MUST reset the offset safely and continue processing without crashing + +### Requirement: Keycheck DB observation writes are resilient +The system SHALL make keycheck DB observation writes resilient to active scanner DB contention. + +#### Scenario: SQLite database is temporarily locked +- **WHEN** a keycheck result or occurrence is ready to record and SQLite is locked +- **THEN** the system MUST retry with bounded backoff before reporting a DB write failure + +#### Scenario: DB write fails after retries +- **WHEN** all DB write retries fail +- **THEN** the status file write MUST remain intact and the failure MUST be visible in logs or dashboard health + +#### Scenario: Schema initialization would contend with active writers +- **WHEN** a checker records a single result row +- **THEN** it MUST NOT run schema initialization or migration DDL as part of that per-result write path + +### Requirement: Dashboard exposes keycheck pipeline health +The dashboard SHALL expose keycheck pipeline health and freshness separately from provider status counts. + +#### Scenario: DB observations lag behind status files +- **WHEN** status files are newer than the latest DB keycheck row +- **THEN** dashboard MUST show that DB observation data is stale relative to file current-state + +#### Scenario: Keycheck run is stuck on a service +- **WHEN** the keychecks process has not advanced past a service for longer than expected +- **THEN** dashboard or supervisor-visible status MUST make the stuck service and elapsed time visible + +#### Scenario: Operator wants current usable keys +- **WHEN** an operator selects current usable key view +- **THEN** dashboard MUST use provider-specific access tiers while retaining exact status filters such as `BEDROCK`, `VERTEX`, `ALIVE`, and `VALID_RATE_LIMITED` + +#### Scenario: Operator wants historical source yield +- **WHEN** an operator selects historical yield view +- **THEN** dashboard MUST include cached known-key occurrences so source/query yield is not lost after rechecks or known-key skips diff --git a/openspec/changes/fix-keycheck-accounting-visibility/tasks.md b/openspec/changes/fix-keycheck-accounting-visibility/tasks.md new file mode 100644 index 0000000..63f0e13 --- /dev/null +++ b/openspec/changes/fix-keycheck-accounting-visibility/tasks.md @@ -0,0 +1,36 @@ +## 1. Input Processing + +- [x] 1.1 Add shared keycheck input state storage for per-service file path, file size, inode/signature if available, and last processed byte offset. +- [x] 1.2 Extend shared keycheck input reader to support high-watermark processing with safe reset on file truncation or rotation. +- [x] 1.3 Migrate hand-rolled readers in OpenAI, OpenRouter, and Gemini to the shared input reader. +- [x] 1.4 Keep a bounded tail fallback for first run or missing state, with clear logging of the active input mode. + +## 2. Known-Key Occurrence Recording + +- [x] 2.1 Add a shared helper that resolves cached status for a key from checked/status files without provider API calls. +- [x] 2.2 Add occurrence recording for skipped known keys, including service, cached status, source line, finding payload, and detector. +- [x] 2.3 Mark cached occurrence rows distinctly from API-check rows in metadata. +- [x] 2.4 Deduplicate cached occurrences per service/key/finding/source to avoid unbounded repeated rows. + +## 3. DB Write Reliability + +- [x] 3.1 Ensure per-result keycheck DB writes do not run schema initialization or migration DDL. +- [x] 3.2 Add bounded lock retry/backoff for keycheck result and occurrence writes. +- [x] 3.3 Add keycheck runner counters for DB write success, retry, and failure counts per service. +- [x] 3.4 Log DB write failures without preventing status-file updates. + +## 4. Dashboard Visibility + +- [x] 4.1 Add dashboard panel comparing status-file current-state counts with latest DB observation timestamps. +- [x] 4.2 Add keycheck pipeline health panel showing last completed service, current/stuck service, run duration, and DB write errors. +- [x] 4.3 Add current usable-key view based on provider-specific access tiers and exact statuses. +- [x] 4.4 Add historical source-yield view that includes cached known-key occurrences. +- [x] 4.5 Label cached-status rows separately from fresh API-check rows in validation tables. + +## 5. Verification + +- [x] 5.1 Add a smoke test or script that runs a checker over a small fixture and verifies status-file writes plus DB occurrence rows. +- [x] 5.2 Verify OpenAI, OpenRouter, Gemini, AWS, and GCP checkers use bounded/incremental input processing. +- [x] 5.3 Verify a known alive key found in a new source/query creates a cached occurrence without an API recheck. +- [x] 5.4 Verify dashboard can show current-state file counts even when DB observations are stale. +- [x] 5.5 Run `python -m py_compile` on changed Python modules. diff --git a/openspec/changes/harden-dockerhub-search-pagination/.openspec.yaml b/openspec/changes/harden-dockerhub-search-pagination/.openspec.yaml new file mode 100644 index 0000000..1ea7e36 --- /dev/null +++ b/openspec/changes/harden-dockerhub-search-pagination/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-09 diff --git a/openspec/changes/harden-dockerhub-search-pagination/design.md b/openspec/changes/harden-dockerhub-search-pagination/design.md new file mode 100644 index 0000000..e3d33e5 --- /dev/null +++ b/openspec/changes/harden-dockerhub-search-pagination/design.md @@ -0,0 +1,70 @@ +## Context + +Managed DockerHub source cycles configure an explicit account pool, but repository search bypasses it and calls the Hub search endpoint anonymously. Docker Hub rejects anonymous result windows beyond 200 entries; an isolated authenticated probe using the existing Hub bearer flow returned 100 results for pages 3, 20, 21, and 30 at a page size of 100. The standard search path also schedules every page before learning the reported result count and currently accepts successful pages when another expected page fails, so the source cycle completes and advances its numeric query cursor with incomplete discovery. + +The account manager already provides secret-safe accounts, cached Hub bearer tokens, account rotation, endpoint-keyed cooldowns, and persisted auth events. The implementation must reuse those controls without changing tag resolution, Registry access, or immutable scan behavior. + +## Goals / Non-Goals + +**Goals:** + +- Authenticate standard and recent DockerHub repository searches through the configured account pool. +- Fail closed when an explicit pool has no usable account. +- Support at most 30 authenticated search pages and avoid requests beyond the result count reported by page one. +- Give each page exactly one additional attempt for transient transport or server failures. +- Return a complete expected page set or fail the source cycle without enqueueing partial results or advancing its query cursor. +- Keep logs, exceptions, and auth events free of credentials and bearer values. + +**Non-Goals:** + +- Adding discovery queries, changing active page sizes, changing repository refresh cadence, or altering scan budgets. +- Changing Docker tag selection, Registry bearer authentication, layer scanning, immutable target identity, or scan retry policy. +- Guaranteeing that Docker Hub will indefinitely support 30 pages; the local cap remains a safety bound, not an external SLA. + +## Decisions + +### Reuse Hub bearer authentication with a search-specific endpoint identity + +Add a repository-search response helper that follows the existing tag-response pattern but uses the `hub_search` endpoint key. It obtains cached or fresh Hub bearer tokens, sends `Authorization: Bearer ...`, refreshes once after a 401, and rotates across the configured accounts for 401, 403, or 429 responses. Search cooldowns remain separate from `hub_tags` and `registry` cooldowns. The shared token cache remains per account because the same Hub token is valid for both Hub endpoints. + +When the manager represents an explicit pool and no account is usable, the helper fails before making an anonymous request. A non-explicit legacy invocation may retain anonymous behavior for direct CLI compatibility. + +Alternative considered: add a second login mechanism or cookie session. Rejected because the existing `/v2/auth/token` bearer flow was verified against authenticated search through page 30 and avoids another credential path. + +### Fetch page one before parallel remainder + +Raise the code-level page cap from 20 to 30. Standard discovery fetches page one first, validates its payload, derives the expected page count from its reported `count`, and submits only pages 2 through `min(requested, expected, 30)` concurrently. This removes ambiguous post-hoc suppression of failed pages and avoids requesting pages objectively outside the reported result set. + +Alternative considered: keep launching all pages concurrently and ignore failures above the largest successful count. Rejected because a failed early page can make the inferred boundary unreliable, and unnecessary out-of-range requests consume account budget. + +### Bound page retries inside the request primitive + +The authenticated search helper gives each page at most two search GETs across transient retry, bearer refresh, and account rotation. Each GET calls `api_request` with one total network attempt; the outer page budget supplies the single bounded retry for `408`, `500`, `502`, `503`, `504`, or account-specific failures. A legacy anonymous invocation has no account handling and calls `api_request` with two total attempts directly. This keeps the page-wide budget independent of global proxy retry settings and prevents it from resetting during account rotation. + +### Treat incomplete pagination as a source-cycle transport failure + +Introduce a DockerHub discovery transport error analogous to the existing GitLab error. Any expected page that remains unavailable after its bounded request/account handling aborts the complete search result before tag resolution or enqueue. The configured-source runner records a failed cycle and returns without advancing the query cursor; it does not crash the long-running source process. A later cycle retries the same query, and queue uniqueness keeps successful rediscovery idempotent. + +### Keep discovery authentication isolated from scan behavior + +The change only replaces repository-search HTTP calls and their error propagation. Tag/manifest resolution continues to use its existing endpoint identities, retries, caches, resolver states, and immutable digest constraints. + +## Risks / Trade-offs + +- [Thirty authenticated pages increase search request volume] -> Preserve the hard cap, configured worker bounds, and account rotation; this change does not raise active source page settings. +- [Page-one count can change while later pages are fetched] -> Treat page one as the cycle snapshot boundary; all pages within that boundary must still succeed, and DB deduplication handles overlap caused by result movement. +- [A single page outage now rejects otherwise usable pages] -> This is intentional complete-or-fail behavior; one retry limits transient loss, and the unchanged cursor retries the query later. +- [Fetching page one serially adds one request latency before parallel work] -> It prevents unnecessary pages and provides an authoritative expected set, which is more valuable than the small latency saving. +- [All accounts can be temporarily unavailable] -> Fail closed and persist endpoint-specific auth state rather than silently reverting to the anonymous 200-result window. + +## Migration Plan + +1. Canonically stop the supervisor and verify all managed children are down. +2. Deploy the scanner, runner, and focused regression tests without changing source query configuration. +3. Run focused DockerHub authentication/pagination/cursor tests and the broader relevant scanner suites. +4. Canonically restart the supervisor and verify PostgreSQL, pipeline workers, DockerHub source worker, and restart counters. +5. Roll back by restoring the previous code under a canonical stop/start if authenticated search causes an operational regression; no data migration is required. + +## Open Questions + +None. Authenticated access through page 30 and the explicit-pool behavior have been verified or are covered by deterministic tests. diff --git a/openspec/changes/harden-dockerhub-search-pagination/proposal.md b/openspec/changes/harden-dockerhub-search-pagination/proposal.md new file mode 100644 index 0000000..b844e5c --- /dev/null +++ b/openspec/changes/harden-dockerhub-search-pagination/proposal.md @@ -0,0 +1,27 @@ +## Why + +DockerHub repository discovery is currently anonymous even when a managed account pool is configured, so searches are limited to the anonymous 200-result window and partial page failures can silently advance the query rotation. Authenticated probing confirms the configured Hub bearer flow can retrieve at least 30 pages of 100 results, making reliable deeper pagination available without new credentials or dependencies. + +## What Changes + +- Authenticate every managed DockerHub repository-search mode through the configured account pool and existing Hub bearer-token flow. +- **BREAKING**: when an explicit DockerHub account pool is configured, fail closed if no account can authenticate instead of falling back to anonymous search. +- Support an authenticated search window of up to 30 pages while retaining a bounded code-level limit. +- Retry transient page failures once, rotate accounts for account-specific failures, and reject an incomplete expected page set rather than enqueueing partial discovery results. +- Preserve the current query cursor when pagination fails, while continuing to advance it after complete or objectively exhausted pagination. +- Keep tag resolution, Registry authentication, immutable-digest deduplication, scan retries, and queue disposition unchanged. + +## Capabilities + +### New Capabilities +- `dockerhub-search-pagination`: Authenticated, bounded, complete-or-fail DockerHub repository-search pagination using the managed account pool. + +### Modified Capabilities + +None. + +## Impact + +- Affects DockerHub search/authentication in `app/scanner.py`, source-cycle failure propagation in `app/console_runner.py`, and focused scanner/runner tests. +- Reuses existing configured DockerHub accounts, Hub bearer tokens, cooldowns, and auth-event persistence; no new external dependency or credential format is introduced. +- Active query lists, refresh cadence, scan budgets, cold/failed target policy, and layer-aware scanning are out of scope. diff --git a/openspec/changes/harden-dockerhub-search-pagination/specs/dockerhub-search-pagination/spec.md b/openspec/changes/harden-dockerhub-search-pagination/specs/dockerhub-search-pagination/spec.md new file mode 100644 index 0000000..88435c0 --- /dev/null +++ b/openspec/changes/harden-dockerhub-search-pagination/specs/dockerhub-search-pagination/spec.md @@ -0,0 +1,68 @@ +## ADDED Requirements + +### Requirement: Managed repository search is authenticated +The system SHALL authenticate every standard and recent DockerHub repository-search request through the configured DockerHub account pool using a Hub bearer token. + +#### Scenario: Configured account performs search +- **WHEN** a managed DockerHub source cycle searches for repositories with an available account +- **THEN** the system sends the search request with that account's bearer authorization and records success under the `hub_search` endpoint identity + +#### Scenario: Explicit pool has no usable account +- **WHEN** an explicit DockerHub account pool is configured but no account can authenticate or leave cooldown +- **THEN** the system fails the repository search without making an anonymous fallback request + +#### Scenario: Search authorization is rejected +- **WHEN** a search request receives an account-specific 401, 403, or 429 response +- **THEN** the system refreshes a rejected bearer once where applicable and rotates to another usable account within the configured pool + +### Requirement: Authenticated pagination is bounded +The system SHALL support up to 30 DockerHub search pages per query and SHALL cap larger requested page counts at 30. + +#### Scenario: Thirty-page authenticated search +- **WHEN** a query requests 30 pages and the first page reports at least 30 pages of results +- **THEN** the system requests the complete page range from 1 through 30 + +#### Scenario: Request exceeds safety cap +- **WHEN** a query requests more than 30 pages +- **THEN** the system limits the search to pages 1 through 30 and records that the requested range was capped + +#### Scenario: Reported result set is shorter +- **WHEN** page one reports fewer results than the requested page range would contain +- **THEN** the system requests only the pages required by that reported count + +### Requirement: Page acquisition is bounded and complete +The system SHALL give each expected search page at most two transient transport/server attempts and SHALL not return partial repository results when any expected page remains unavailable. + +#### Scenario: Transient failure recovers +- **WHEN** an expected page receives a retryable transport error or transient HTTP status on its first attempt and succeeds on its second attempt +- **THEN** the system includes that page and completes discovery without another transient attempt + +#### Scenario: Expected page remains unavailable +- **WHEN** an expected page still fails after bounded retry and account handling +- **THEN** the system raises a DockerHub discovery transport failure before tag resolution or repository enqueue + +#### Scenario: Every expected page succeeds +- **WHEN** all expected pages return valid payloads +- **THEN** the system combines their repositories in page order and proceeds with existing deduplication and resolution behavior + +### Requirement: Failed pagination preserves query rotation +The system SHALL record incomplete DockerHub pagination as a failed source cycle and SHALL keep the current query cursor unchanged. + +#### Scenario: Source cycle receives pagination failure +- **WHEN** repository discovery raises a DockerHub discovery transport failure +- **THEN** the source cycle finishes with failed status, enqueues no partial search result, and selects the same query for the next cycle + +#### Scenario: Complete source cycle succeeds +- **WHEN** repository discovery and the remaining source cycle complete normally +- **THEN** the existing query-advance policy remains unchanged + +### Requirement: Search authentication is secret-safe and isolated +The system MUST NOT expose account credentials or bearer tokens through search logs, errors, or auth events, and SHALL preserve existing tag, Registry, immutable-digest, and scan-retry behavior. + +#### Scenario: Search request fails +- **WHEN** an authenticated search request fails or exhausts the account pool +- **THEN** emitted diagnostics identify only the safe endpoint/status category without including usernames, credentials, bearer values, or request authorization headers + +#### Scenario: Repository search implementation changes +- **WHEN** authenticated search pagination is deployed +- **THEN** existing DockerHub tag resolution, Registry authentication, target deduplication, and scan retry contracts remain unchanged diff --git a/openspec/changes/harden-dockerhub-search-pagination/tasks.md b/openspec/changes/harden-dockerhub-search-pagination/tasks.md new file mode 100644 index 0000000..c695082 --- /dev/null +++ b/openspec/changes/harden-dockerhub-search-pagination/tasks.md @@ -0,0 +1,21 @@ +## 1. Runtime Safety + +- [x] 1.1 Canonically stop the live supervisor and verify all managed child processes are down before editing `app` + +## 2. Authenticated Search Implementation + +- [x] 2.1 Make Hub bearer acquisition endpoint-aware and add a `hub_search` response path with explicit-pool fail-closed behavior, account rotation, and two-attempt transient request bounds +- [x] 2.2 Raise the search safety cap to 30 pages and make standard pagination fetch page one first, derive the expected range, and reject any incomplete expected page set +- [x] 2.3 Route recent-mode repository search through the same authenticated bounded response path +- [x] 2.4 Add a DockerHub discovery transport failure path that records a failed source cycle without advancing the query cursor + +## 3. Regression Coverage + +- [x] 3.1 Cover bearer authorization, token refresh, account rotation/cooldown, and explicit-pool no-fallback behavior without exposing secrets +- [x] 3.2 Cover the 30-page cap, page-one count boundary, ordered complete results, one transient retry, and rejection of unresolved partial pagination +- [x] 3.3 Cover failed-cycle cursor retention and successful-cycle compatibility +- [x] 3.4 Run focused and broader relevant scanner/runner test suites + +## 4. Runtime Verification + +- [x] 4.1 Canonically restart the supervisor and verify PostgreSQL, pipeline workers, DockerHub source health, restart counters, and absence of app bytecode artifacts diff --git a/openspec/changes/improve-core-scan-coverage/.openspec.yaml b/openspec/changes/improve-core-scan-coverage/.openspec.yaml new file mode 100644 index 0000000..7f2cf9b --- /dev/null +++ b/openspec/changes/improve-core-scan-coverage/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-08-28 diff --git a/openspec/changes/improve-core-scan-coverage/design.md b/openspec/changes/improve-core-scan-coverage/design.md new file mode 100644 index 0000000..0badc9b --- /dev/null +++ b/openspec/changes/improve-core-scan-coverage/design.md @@ -0,0 +1,120 @@ +## Context + +DockerHub discovery currently inspects a bounded tag page but stops after the first eligible digest. Different tags often alias the same manifest or share the same ordered layers, so increasing the tag count without content-aware selection would mostly multiply duplicate work. The Hub tag response identifies platform digests but does not expose their layers; those come from the Docker Registry v2 manifest API. + +GitHub and GitLab updated-target admission currently uses provider timestamps. A claimed scan then points TruffleHog at a mutable repository URL with a rolling age boundary and a depth cap. The timestamp is useful for deciding that something may have changed, but it neither identifies the exact ref nor proves which commit was scanned. Existing PostgreSQL reservations provide the fence under which an immutable plan can be attached. + +## Goals / Non-Goals + +**Goals:** + +- Spend each repository's Docker budget on up to three distinct ordered layer graphs. +- Keep Docker tag, manifest, request, cache, and emitted-target counts explicitly bounded. +- Bind GitHub and GitLab scans to a provider-resolved ref and commit SHA after a fenced claim. +- Use the last successfully covered SHA as the incremental boundary for the same ref. +- Preserve exact plan and coverage evidence through retries, worker failure, and mid-scan updates. +- Keep first-scan history bounded while making every later delta independent of rolling age and depth limits. + +**Non-Goals:** + +- Persist a global historical inventory of Docker layers. +- Guarantee discovery of a branch that appears and disappears between metadata polling cycles. +- Replace repository metadata search with a global push-event feed. +- Remove source API rate limits, target timeouts, result bounds, or queue admission bounds. +- Claim that a bounded first scan covered history older than its configured baseline depth. + +## Decisions + +### Resolve platform manifests before selecting Docker targets + +For each bounded Hub tag candidate, the resolver will choose the requested platform child digest, obtain a short-lived pull token for the public repository, and fetch that child manifest from the Docker Registry v2 API. A candidate identity contains its tag, update time, immutable platform manifest digest, and ordered layer digest tuple. + +The top-level multi-platform index digest will not be emitted when a matching child digest exists. Registry responses must be JSON manifests of bounded size with valid `sha256` layer digests. Token requests use a fixed Docker authentication origin and repository pull scope; credentials are never placed in cache records or logs. + +Alternatives rejected: + +- Comparing tag names or manifest digests alone, because different manifests can still contain the same layer chain. +- Pulling every candidate image before selection, because manifest metadata is sufficient and much cheaper. +- Persisting every layer immediately, because within-resolution novelty provides the requested bounded diversity without a new authoritative subsystem. + +### Select up to three graphs deterministically + +The source setting `docker_images_per_repository` is clamped to one through three. After exact ordered-tuple deduplication, selection uses stable tie breaking: + +1. The newest resolved graph. +2. The remaining graph that contributes the most layers not present in the selected union, with recency as the tie breaker. +3. The oldest remaining distinct graph. + +If fewer distinct graphs exist, fewer targets are emitted. Reordered layer tuples remain distinct because layer order changes the image filesystem. Selected targets retain the existing immutable `repository@sha256:...` identity, so queue deduplication and Docker scan execution do not change. + +Only complete graph-selection results are stored in the disposable tag cache. A partial manifest-resolution failure can return successfully resolved targets for the current cycle, but it is not cached as complete and is reported as partial coverage. + +### Resolve and bind a Git plan after claim + +Repository timestamps remain coarse admission signals. Once PostgreSQL has fenced a queue row and result reservation, the worker resolves the source-provided ref hint or the repository's current default branch through the GitHub or GitLab API. The resolver returns a normalized ref and exact head SHA. + +The reservation is then bound transactionally to an immutable plan containing ref, head SHA, optional covered base SHA, plan mode, and baseline bounds. Binding requires the active reservation and claim lease token. A missing or malformed revision is a retryable source failure; the worker does not silently scan a mutable URL and does not advance exact coverage. + +Metadata search can identify only repository-level activity, so its exact scope is the provider-resolved default branch. Event-backed discovery may supply a more specific ref. Polling cannot guarantee capture of transient or unadvertised refs; that limitation remains explicit. + +Alternatives rejected: + +- Resolving one SHA for every search result before admission, because that would spend scarce API quota on records that are never claimed. +- Encoding SHA in queue target identity, because it would create unbounded rows and bypass existing changed-target coalescing. +- Copying a timestamp into a covered field at claim, because failed work would look complete. + +### Execute pinned baselines and deltas + +An initial or ref-changed plan scans the exact head with the configured first-scan depth bound. A same-ref plan with a different successfully covered head scans the pinned head with `--since-commit ` and omits rolling age and maximum-depth limits. A plan whose resolved head already equals the covered head produces an exact no-op result. + +The installed TruffleHog binary's ability to accept a commit SHA as `--branch` is a deployment contract and will be covered by a local repository contract test. If the covered base is unavailable after a force push or ref recreation, execution falls back to a pinned bounded baseline and labels the result as such; it never reports an incremental range as covered when the base was not usable. + +This delta means all commits reachable from the pinned head after the covered boundary, not merely the final filesystem diff. An add-then-delete sequence in separate new commits therefore remains visible. + +### Advance covered SHA only during fenced successful ingestion + +`target_queue` stores the last successfully covered ref and head SHA. `result_reservations` stores the immutable claimed plan, and normalized scan metadata stores the executed plan and whether execution remained pinned. On fenced ingestion with queue disposition `done`, the plan in metadata must match the reservation. Only a successful pinned baseline, successful incremental scan, or exact no-op advances or confirms the covered head. Failed, deferred, unbound, or mutable fallback work leaves coverage unchanged. + +Because the claimed head is immutable, a newer provider update during execution remains discoverable after completion and can create a later plan from the just-covered head. + +## Risks / Trade-offs + +- [Registry manifest calls increase Docker API traffic] -> Keep tag candidates and emitted graphs bounded, reuse one scoped token per repository resolution, and do not cache partial results as complete. +- [Many tags alias one graph] -> Deduplicate exact ordered layer tuples before queue insertion. +- [A source API is unavailable after claim] -> Produce a retryable source failure and refund through the existing bounded lifecycle without changing coverage. +- [TruffleHog SHA branch behavior differs by version] -> Add a local contract test against the configured binary and fail closed when immutable pinning is unsupported. +- [Force-pushed base is no longer reachable] -> Retry as a pinned bounded baseline and label the loss of incremental continuity. +- [Default-branch resolution misses non-default branch activity] -> Record the resolved scope honestly and allow event-backed ref hints; a complete ref-event feed remains future work. +- [First-scan depth remains bounded] -> Treat the first head as the future delta baseline without claiming unbounded historical coverage. +- [Three Docker graphs can triple downstream work] -> Clamp the per-repository setting to three and retain all existing source, queue, timeout, and output bounds. + +## Migration Plan + +1. Add nullable Git coverage and reservation-plan columns through idempotent PostgreSQL schema initialization. +2. Deploy graph resolution, plan binding, and tests with explicit configuration gates. +3. Enable `docker_images_per_repository: 3` for DockerHub while retaining the existing twenty-tag candidate bound. +4. Enable exact Git planning for GitHub and GitLab; legacy rows start with no covered SHA and receive a pinned bounded baseline on their next admitted scan. +5. Observe partial manifest resolution, distinct graph counts, exact no-ops, baseline resets, delta scans, failures, and strict usable yield. + +Rollback is configuration-first. Set Docker images per repository back to one and disable exact Git planning; nullable schema additions remain inert and require no destructive migration. + +## Open Questions + +- Whether production evidence supports increasing the Docker candidate tag page beyond twenty without exhausting Hub rate limits. +- Whether a later change should snapshot all advertised refs or consume a dedicated push-event feed for complete non-default-branch coverage. + +## Implementation Evidence + +Implementation completed and locally validated on 2026-08-28: + +- Docker resolution uses the bounded Registry v2 bearer flow, immutable platform-child digests, ordered-layer graph deduplication, and deterministic newest/novel/oldest selection. Partial graph resolution emits only proven targets, retains the repository for retry, and is not cached as complete. +- PostgreSQL reservations bind one canonical exact Git plan under the active queue/reservation lease fence. Covered ref/head state advances in the same fenced transaction as successful result ingestion after exact plan and execution-evidence comparison. +- GitHub and GitLab resolve either an explicit branch ref or the provider default branch to an exact commit. Baseline, delta, no-op, and continuity-reset execution remain pinned to the bound head. +- The checked-in core configuration enables three Docker graphs and exact Git planning with a baseline depth of 100, two ref-resolution attempts, a 10-second shared timeout, and a 1 MiB response limit. +- `python -B -m pytest -p no:cacheprovider -q tests/test_exact_git_scan_planning.py tests/test_validated_high_scanner_fixes.py::DockerTagIdentityTests tests/test_runtime_safety_layer.py tests/test_pipeline_cutover_invariants.py tests/test_migration_runtime_safety.py tests/test_pipeline_postgres_integration.py::PipelinePostgresIntegrationTests::test_exact_git_plan_binding_and_coverage_are_fenced` completed with `141 passed`. +- The PostgreSQL integration scenario covers idempotent/conflicting binding, baseline to delta to no-op progression, durable exact metadata, a provider update observed during an active scan, successful fenced coverage advancement, and a post-bind worker refund that cannot advance coverage. +- The configured local TruffleHog binary passed the exact-SHA `--branch` contract test included in the focused suite. +- `python -B -m py_compile app/scanner.py app/scanner_db.py app/console_runner.py tests/test_exact_git_scan_planning.py tests/test_pipeline_postgres_integration.py` completed successfully. +- `openspec validate improve-core-scan-coverage --strict` reported the change as valid. + +Retained rollout limits are twenty Docker tag candidates, at most three emitted distinct graphs, 8 MiB per Registry manifest, 1,000 index descriptors, 2,048 layers, bounded source/queue/result limits, and PostgreSQL-only exact Git plan binding. No production migration, process restart, or rollout was performed as part of implementation. Deployment still requires the offline idempotent runtime-safety migration before restarting sources. Configuration-first rollback remains `docker_images_per_repository: 1` plus disabling `exact_git_planning_enabled`; nullable schema additions may remain in place. diff --git a/openspec/changes/improve-core-scan-coverage/proposal.md b/openspec/changes/improve-core-scan-coverage/proposal.md new file mode 100644 index 0000000..edf72d8 --- /dev/null +++ b/openspec/changes/improve-core-scan-coverage/proposal.md @@ -0,0 +1,27 @@ +## Why + +The core DockerHub, GitHub, and GitLab sources spend most of their scan budget on repeated image contents or mutable repository snapshots, while recent strict-usable yield remains near zero. The scanner needs to cover materially different Docker layers and bind Git work to exact revisions so that additional work buys new evidence rather than another pass over the same surface. + +## What Changes + +- Select up to three Docker images per repository whose ordered layer graphs are distinct, instead of stopping at the first eligible tag. +- Prefer the newest graph, a graph adding the most not-yet-selected layers, and an older divergent graph while continuing to deduplicate immutable digest targets. +- Resolve Git discovery observations into immutable ref and commit identities before scanning. +- Scan all commits introduced since the last successfully covered commit for that ref, rather than relying on a moving repository URL, age cutoff, and depth cap. +- Persist immutable Git scan plans and successfully covered heads; never advance exact coverage on a failed or unpinned fallback scan. +- Bound API enumeration and Docker/Git expansion through explicit configuration and report partial or inexact coverage honestly. + +## Capabilities + +### New Capabilities + +- `docker-layer-graph-selection`: Bounded selection of materially distinct platform-specific Docker image layer graphs. +- `git-ref-delta-scanning`: Immutable, per-ref Git scan planning and successful incremental coverage tracking. + +### Modified Capabilities + +None. + +## Impact + +The change affects Docker Hub tag and registry-manifest resolution, GitHub and GitLab metadata resolution, source-cycle configuration, PostgreSQL queue/reservation/scan state, TruffleHog command construction, and focused scanner/runtime integration tests. It adds bounded registry and source API requests but does not change external service APIs or credential output formats. diff --git a/openspec/changes/improve-core-scan-coverage/specs/docker-layer-graph-selection/spec.md b/openspec/changes/improve-core-scan-coverage/specs/docker-layer-graph-selection/spec.md new file mode 100644 index 0000000..18b136e --- /dev/null +++ b/openspec/changes/improve-core-scan-coverage/specs/docker-layer-graph-selection/spec.md @@ -0,0 +1,49 @@ +## ADDED Requirements + +### Requirement: Platform-specific layer graphs are resolved +The system SHALL resolve each bounded Docker tag candidate to the requested platform child manifest and SHALL represent its image contents as the ordered sequence of valid layer digests. + +#### Scenario: Multi-platform tag contains the requested platform +- **WHEN** a tag exposes a matching `linux/amd64` child manifest +- **THEN** the system uses that child manifest digest and its ordered layers rather than the top-level index digest + +#### Scenario: Candidate manifest is malformed +- **WHEN** a registry response is oversized, malformed, or contains invalid layer identities +- **THEN** the candidate is not represented as a resolved layer graph + +### Requirement: Docker selection covers distinct graphs +The system SHALL emit no more than the configured one-to-three image targets per repository and SHALL NOT emit two candidates with identical ordered layer digest sequences. + +#### Scenario: Tags alias one graph +- **WHEN** multiple tags resolve to the same ordered layer sequence +- **THEN** only the newest alias remains eligible for selection + +#### Scenario: Three or more graphs are available +- **WHEN** at least three distinct graphs resolve successfully +- **THEN** the system selects the newest graph, the remaining graph adding the most not-yet-selected layers, and the oldest remaining distinct graph + +#### Scenario: Fewer graphs are available +- **WHEN** fewer distinct graphs resolve successfully than the configured maximum +- **THEN** the system emits only the distinct graphs that exist + +#### Scenario: Layer order differs +- **WHEN** two manifests contain the same layer identities in a different order +- **THEN** the system treats them as distinct graphs + +### Requirement: Selected Docker targets remain immutable +The system SHALL emit each selected target as the canonical requested-platform manifest digest identity `repository@sha256:`. + +#### Scenario: Selected tag moves later +- **WHEN** a tag is republished after discovery +- **THEN** the queued target continues to identify the originally selected platform manifest digest + +### Requirement: Docker graph resolution is bounded and honest +The system SHALL retain hard tag-candidate, selected-graph, response-size, retry, and timeout bounds and SHALL NOT cache partial graph resolution as complete. + +#### Scenario: Some manifest requests fail +- **WHEN** at least one candidate graph resolves and another candidate fails transiently +- **THEN** the system may emit the resolved targets for the cycle but records partial resolution and does not write a complete positive cache entry + +#### Scenario: Every manifest request fails transiently +- **WHEN** no candidate graph can be resolved because registry metadata is unavailable +- **THEN** the repository remains retryable through the existing deferred-resolution lifecycle diff --git a/openspec/changes/improve-core-scan-coverage/specs/git-ref-delta-scanning/spec.md b/openspec/changes/improve-core-scan-coverage/specs/git-ref-delta-scanning/spec.md new file mode 100644 index 0000000..fca8de4 --- /dev/null +++ b/openspec/changes/improve-core-scan-coverage/specs/git-ref-delta-scanning/spec.md @@ -0,0 +1,68 @@ +## ADDED Requirements + +### Requirement: Git scans are bound to an exact revision +The system SHALL resolve a normalized ref and exact commit SHA for each claimed GitHub or GitLab repository before invoking TruffleHog and SHALL bind that immutable plan to the active reservation. + +#### Scenario: Repository search supplies no ref hint +- **WHEN** a claimed repository came from metadata search without an exact ref +- **THEN** the system resolves the provider's current default branch and its exact head SHA + +#### Scenario: Discovery supplies an exact ref hint +- **WHEN** an event-backed target includes a valid branch ref +- **THEN** the system resolves and binds that specific ref instead of substituting the default branch + +#### Scenario: Revision lookup fails +- **WHEN** the provider API cannot return a valid ref and commit SHA +- **THEN** the claim receives a bounded retryable source failure and no exact coverage state advances + +### Requirement: Git updates scan every newly introduced commit +The system SHALL scan the exact claimed head after the last successfully covered head for the same ref and SHALL NOT apply rolling age or maximum-depth limits to that incremental range. + +#### Scenario: Same ref advances +- **WHEN** ref `R` was successfully covered at commit `A` and now resolves to descendant commit `D` +- **THEN** the scan is pinned to `D` with `A` as its boundary and includes commits introduced between them + +#### Scenario: Secret is added and then deleted in the delta +- **WHEN** one newly introduced commit adds a secret and a later newly introduced commit removes it +- **THEN** both commits remain in scan scope even though the final filesystem snapshot is clean + +#### Scenario: Head is unchanged +- **WHEN** the resolved head equals the successfully covered head for the same ref +- **THEN** the system records an exact no-op without launching a redundant repository scan + +### Requirement: Git baseline and discontinuity handling remain pinned +The system SHALL use a pinned bounded baseline for a first-seen ref or an unusable incremental base and SHALL identify that mode without claiming unbounded historical coverage. + +#### Scenario: Ref has no covered head +- **WHEN** an exact ref is claimed without prior successful coverage +- **THEN** the system scans its pinned head using the configured baseline depth bound and establishes that head as the future delta boundary on success + +#### Scenario: Covered base is unavailable +- **WHEN** force push, ref recreation, or remote history removal makes the covered SHA unusable +- **THEN** the system falls back to a pinned bounded baseline and records the continuity reset + +### Requirement: Git coverage advances only after successful fenced work +The system SHALL update a queue row's covered ref and head only when successful ingestion applies a matching immutable reservation plan. + +#### Scenario: Exact scan succeeds +- **WHEN** a pinned baseline or delta result is ingested with queue disposition `done` and its plan matches the active reservation +- **THEN** the queue's covered ref and head advance to the claimed head + +#### Scenario: Exact scan fails or is deferred +- **WHEN** execution fails, times out, loses its fence, or receives a deferred disposition +- **THEN** the previously covered ref and head remain unchanged + +#### Scenario: Remote advances during a scan +- **WHEN** a newer commit appears after the worker binds its immutable head +- **THEN** successful completion advances coverage only to the bound head and leaves the newer update eligible for later discovery + +### Requirement: Exact Git scope is observable +The system SHALL durably record the executed ref, head, base, scan mode, baseline bound, and whether immutable execution was preserved. + +#### Scenario: Operator inspects an incremental scan +- **WHEN** an exact delta result is committed +- **THEN** its normalized scan metadata identifies the covered range without exposing source credentials + +#### Scenario: Metadata discovery observes repository-level activity +- **WHEN** no branch-specific event exists +- **THEN** observability identifies the provider-resolved default-branch scope rather than implying coverage of every repository ref diff --git a/openspec/changes/improve-core-scan-coverage/tasks.md b/openspec/changes/improve-core-scan-coverage/tasks.md new file mode 100644 index 0000000..3b6ba6a --- /dev/null +++ b/openspec/changes/improve-core-scan-coverage/tasks.md @@ -0,0 +1,20 @@ +## 1. Docker Layer Graph Selection + +- [x] 1.1 Add and clamp `docker_images_per_repository` configuration through source argument construction and all Docker tag-resolution call sites. +- [x] 1.2 Implement bounded Docker Registry token, platform-manifest, and ordered-layer resolution with strict response validation. +- [x] 1.3 Implement deterministic distinct-graph selection, immutable child-digest targets, partial-result handling, and cache-version invalidation. +- [x] 1.4 Add focused tests for aliases, platform children, novelty and age selection, malformed manifests, partial failures, and resolution plumbing. + +## 2. Exact Git Revision Planning + +- [x] 2.1 Add idempotent PostgreSQL fields and fenced methods for immutable reservation plans and last successfully covered Git ref/head. +- [x] 2.2 Implement bounded GitHub and GitLab ref/head resolvers with default-branch and explicit-ref handling. +- [x] 2.3 Extend Git scan execution for exact no-op, pinned baseline, and unbounded-by-age/depth delta modes, including continuity-reset fallback. +- [x] 2.4 Bind plans after claims, include them in result metadata, and advance covered heads only during matching successful ingestion. +- [x] 2.5 Add focused tests for plan resolution, command construction, unchanged heads, failure/refund fencing, successful coverage, force-push fallback, and mid-scan updates. + +## 3. Configuration And Verification + +- [x] 3.1 Enable three distinct Docker graphs and exact Git planning for the core sources with bounded documented defaults. +- [x] 3.2 Run focused Docker, Git, queue, schema, lifecycle, and OpenSpec validation suites. +- [x] 3.3 Record implementation evidence and any retained rollout limits in the change artifacts. diff --git a/openspec/changes/improve-worker-operator-cheatsheets/.openspec.yaml b/openspec/changes/improve-worker-operator-cheatsheets/.openspec.yaml new file mode 100644 index 0000000..1aca8b9 --- /dev/null +++ b/openspec/changes/improve-worker-operator-cheatsheets/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-30 diff --git a/openspec/changes/improve-worker-operator-cheatsheets/design.md b/openspec/changes/improve-worker-operator-cheatsheets/design.md new file mode 100644 index 0000000..5dae18f --- /dev/null +++ b/openspec/changes/improve-worker-operator-cheatsheets/design.md @@ -0,0 +1,54 @@ +## Context + +The packaged CLI persists a private JSON configuration after `install`, but first installation accepts credentials only through `--token`. Windows has a root launcher, the Linux image relies on its entrypoint, and assembled Linux packages have no root launcher. `attach` already provides live coherent status, but operators reasonably look for a `watch` command. Active documentation mixes platform-specific syntax and server-capacity administration with local lifecycle commands. + +## Goals / Non-Goals + +**Goals:** + +- Make the same lifecycle vocabulary available from the package root on Windows and native Linux and through a short Docker helper. +- Keep the device token out of command arguments and shell history during installation. +- Verify start, clean stop, attach, status, and watch behavior before publishing copy-paste cheatsheets. +- Remove server cap `0` from routine worker maintenance instructions. + +**Non-Goals:** + +- Change the worker protocol, server API, assignment cancellation, or capacity semantics. +- Replace the installed private JSON configuration or expose its token. +- Add an updater, public artifact registry, or service-manager integration for native Linux. + +## Decisions + +### Treat `watch` as the live human status alias + +`watch` uses the same authenticated local control stream and coherent status refresh as `attach`. It accepts the same optional bounded `--follow-seconds` argument and detaches without stopping the worker. Reusing the control path avoids a second polling implementation and behaves consistently where an OS `watch` utility is absent. + +### Accept a strict YAML installation document + +`install --config PATH` reads an exact YAML mapping containing `server`, `token`, and `parallelism`. `PATH=-` reads a bounded document from standard input for Docker. A path must be a private regular file; YAML uses `safe_load`, rejects aliases/extra fields through exact shape validation, and never changes the durable installed JSON schema. Direct `--server` and `--token` remain supported for compatibility but cannot be combined with `--config`. + +### Add only a Linux package-root launcher + +Assembled non-Windows packages receive executable `truf-worker` and `run-worker` shell launchers equivalent to the Windows command files. Docker keeps its existing entrypoint; the launcher also enables release tooling to export the assembled package for native Linux without inventing another client implementation. + +### Separate active instructions from evidence reports + +The general quickstart links concise Windows, Linux, and Docker cheatsheets and retains conceptual guidance. Each platform sheet starts from its artifact or Compose root and contains installation plus start, stop, attach, status, and watch commands. Historical validation reports remain unchanged even where they record earlier cap-zero experiments. + +## Risks / Trade-offs + +- [A YAML file leaves a plaintext token on disk] -> Require private permissions, label it one-time installation input, and instruct deletion after successful install; the durable token remains in the existing private state. +- [Standard input cannot prove source-file permissions] -> Bound and validate its content and document `chmod 600` on the redirected host file. +- [Watch and attach appear redundant] -> Document watch as a discoverable alias rather than maintaining distinct semantics. +- [Native Linux lacks automatic daemon startup] -> Keep start/stop in the portable CLI and explicitly leave systemd installation out of scope. + +## Migration Plan + +1. Add parser, YAML input, launcher, and tests without changing existing command forms. +2. Build fresh Windows and Linux/Docker artifacts and run isolated lifecycle checks. +3. Publish the updated general guide and platform sheets only after those checks pass. +4. Roll back by restoring the previous artifact; installed schema-2 JSON remains compatible. + +## Open Questions + +None. diff --git a/openspec/changes/improve-worker-operator-cheatsheets/proposal.md b/openspec/changes/improve-worker-operator-cheatsheets/proposal.md new file mode 100644 index 0000000..b1ccaa0 --- /dev/null +++ b/openspec/changes/improve-worker-operator-cheatsheets/proposal.md @@ -0,0 +1,23 @@ +## Why + +Remote-worker instructions do not provide independently verified copy-paste workflows for Windows, native Linux, and Docker. They also document a nonexistent `watch` command, require a device token in the install command line, claim a native Linux launcher that is not packaged, and recommend server cap `0` for routine local lifecycle operations. + +## What Changes + +- Add a bounded `watch` lifecycle command alongside `start`, `stop`, `attach`, and `status`. +- Allow first-time installation to read the server origin, device token, and parallelism from a small YAML document instead of process arguments. +- Package a native Linux launcher and verify the lifecycle command surface on Windows, native Linux, and Docker. +- Keep the complete operator guide concise while adding separate copy-paste cheatsheets for each supported environment. +- Remove routine cap `0` instructions; local graceful stop drains the selected worker without changing server scheduling for other devices. + +## Capabilities + +### New Capabilities + +- `worker-operator-lifecycle`: Cross-platform worker lifecycle commands, private YAML installation input, and verified platform cheatsheets. + +### Modified Capabilities + +## Impact + +This affects `worker_cli.py`, worker package launchers, the Linux worker image, CLI/package tests, packaged lifecycle verification, and remote-worker operator documentation. The worker protocol, server API, stored private configuration schema, and assignment capacity model remain unchanged. diff --git a/openspec/changes/improve-worker-operator-cheatsheets/specs/worker-operator-lifecycle/spec.md b/openspec/changes/improve-worker-operator-cheatsheets/specs/worker-operator-lifecycle/spec.md new file mode 100644 index 0000000..e10b820 --- /dev/null +++ b/openspec/changes/improve-worker-operator-cheatsheets/specs/worker-operator-lifecycle/spec.md @@ -0,0 +1,41 @@ +## ADDED Requirements + +### Requirement: Cross-platform lifecycle command surface +The packaged worker SHALL expose start, stop, attach, status, and watch operations with equivalent local-control semantics on Windows, native Linux, and Docker. + +#### Scenario: Operator watches a running worker +- **WHEN** an operator runs `watch` for a bounded interval +- **THEN** the CLI emits coherent live status and detaches without stopping the worker + +#### Scenario: Operator stops a worker +- **WHEN** an operator requests a graceful stop while the worker can complete its local drain +- **THEN** the CLI returns a clean shutdown receipt without requiring a server-side assignment-cap change + +### Requirement: Installation credentials can come from YAML +The worker SHALL accept a bounded strict YAML installation document containing the HTTPS server origin, device token, and positive local parallelism instead of requiring those values in process arguments. + +#### Scenario: Install from a private YAML file +- **WHEN** an operator invokes `install --config` with a private valid YAML file +- **THEN** the worker verifies the package and persists the existing private installed configuration without exposing the token in argv + +#### Scenario: Install from redirected standard input +- **WHEN** a Docker operator redirects a valid YAML document to `install --config -` +- **THEN** the worker performs the same installation without placing the token in the Compose or container command + +#### Scenario: Reject ambiguous installation input +- **WHEN** YAML input is malformed, has extra fields, exceeds its byte bound, or is combined with direct server/token arguments +- **THEN** installation fails before writing worker configuration + +### Requirement: Native Linux package is directly operable +An assembled native Linux worker package SHALL include an executable package-root launcher for the same operator CLI used by Windows and Docker. + +#### Scenario: Linux operator runs from the extracted package root +- **WHEN** the operator invokes `./truf-worker status` +- **THEN** the integrity-checking bootstrap runs with the package-local application and dependencies + +### Requirement: Platform cheatsheets are executable and capacity-independent +The active operator documentation SHALL provide separate Windows, native Linux, and Docker copy-paste cheatsheets whose lifecycle commands are verified against the corresponding packaged artifact and do not instruct routine use of server cap `0`. + +#### Scenario: Operator follows one platform sheet +- **WHEN** an operator starts from the documented artifact or Compose root +- **THEN** installation, start, status, attach, watch, and clean stop require only the documented local files and commands diff --git a/openspec/changes/improve-worker-operator-cheatsheets/tasks.md b/openspec/changes/improve-worker-operator-cheatsheets/tasks.md new file mode 100644 index 0000000..3b93a0c --- /dev/null +++ b/openspec/changes/improve-worker-operator-cheatsheets/tasks.md @@ -0,0 +1,16 @@ +## 1. Operator CLI + +- [x] 1.1 Add strict private YAML and standard-input configuration to worker installation +- [x] 1.2 Add the bounded watch alias over the existing attach control stream +- [x] 1.3 Add an executable native Linux package-root launcher + +## 2. Platform Documentation + +- [x] 2.1 Create separate copy-paste cheatsheets for Windows, native Linux, and Docker +- [x] 2.2 Update the general quickstart and operations runbook and remove routine cap-zero guidance + +## 3. Verification + +- [x] 3.1 Add focused CLI, package, YAML-security, and documentation tests +- [x] 3.2 Build fresh Windows and Linux/Docker artifacts and verify install, start, status, attach, watch, and clean stop +- [x] 3.3 Run focused tests and strict OpenSpec validation and record the checked command matrix diff --git a/openspec/changes/improve-worker-operator-cheatsheets/validation.md b/openspec/changes/improve-worker-operator-cheatsheets/validation.md new file mode 100644 index 0000000..cd7f446 --- /dev/null +++ b/openspec/changes/improve-worker-operator-cheatsheets/validation.md @@ -0,0 +1,79 @@ +# Worker Cheatsheet Validation - 2026-09-30 + +## Scope + +Validation used isolated fake device tokens, private local state roots, a local +TLS no-work fixture, unique Docker names/volumes, and fresh artifacts. It did not +contact production, issue assignments, or use production credentials. + +## Initial audit + +- Windows package exposed install, start, stop, status, attach, logs, history, + and doctor; `watch` was rejected as an invalid command. +- The Linux image exposed the same command set, but an assembled native Linux + package had no package-root launcher or preparation workflow. +- First installation required token-bearing process arguments. Later lifecycle + commands already read the private installed JSON configuration. +- Active quickstart and operations instructions recommended server cap zero for + routine local maintenance. +- An unreachable-server negative check made `stop --timeout 120` return a + non-drained receipt with exit code 2 instead of reporting false success. + +## Fresh artifact identities + +| Artifact | Identity | +| --- | --- | +| Final Windows ZIP SHA-256 | `210eb61e8d6c35b014b39b18b4dd7e28e1e057a11d57790188f7904487bac004` | +| Windows package manifest | `4572e349cead890c4efdc113e4d41285981086db7c0783fb67493fdeb6bac04c` | +| Linux/Docker image | `sha256:8bc99d9e7e5f5f364de9b7d2b30100942b5ce3d9170a64e5abf0068ffc3d02c4` | +| Linux package manifest | `19907f29382bd2b5a1de13fd53a829e97a5626fbacbed8f4286071ba4d5a9bec` | + +The final Windows ZIP was rebuilt after the last documentation correction. Its +full lifecycle run used the same package-manifest identity; the rebuild changed +only packaged operator-document bytes outside worker code authority. + +## Checked command matrix + +| Environment | YAML install | Doctor | Start | Status | Attach | Watch | Stop | +| --- | --- | --- | --- | --- | --- | --- | --- | +| Windows package | pass | pass | pass | pass | pass | pass | `drained=true`, `exit_code=0` | +| Native Linux package under `/opt` | pass | pass | pass | pass | pass | pass | `drained=true`, `exit_code=0` | +| Docker image with persistent volume | stdin pass | pass | pass | pass | pass | pass | `drained=true`, `exit_code=0` | + +The stopped Docker container reported `status=exited`, `exit=0`, and +`oom=false`. `attach` and `watch` detached without stopping the verified worker +on every environment. + +## Documentation result + +The active general guide and operations runbook contain no cap-zero routine and +no token-bearing install command. Separate Windows, native Linux, and Docker +cheatsheets contain copy-paste install, start, stop, attach, status, and watch +commands. Historical dated validation reports retain factual records of earlier +cap-zero experiments and are not active instructions. + +## Final gates + +- Focused worker/package/documentation matrix: `157 passed, 3 skipped`. +- Windows packaged `watch --help`: pass. +- Linux image launcher, preparation script, packaged cheatsheets, and shell + syntax smoke: pass. +- Worker Compose rendering: pass. +- Python compilation: pass. +- Strict OpenSpec validation: pass. + +## Release build + +The final `dist/release-20260930-linux` release includes both native Linux and +Docker Linux artifacts plus all platform sheets under `cheatsheets/`. All +entries in `SHA256SUMS.txt` passed verification. The Docker bundle also contains +the sheets and both `workerctl` helpers, and all eight internal checksums passed. +The native archive contained no absolute paths, parent traversal, or links; its +launchers retained mode 0755 and passed an extracted `/opt` preparation and CLI +smoke test. + +| Release artifact | SHA-256 | +| --- | --- | +| Native Linux package | `e44717c9e84fc73d1d0734189da5b809c1c2271648868c05caea954bf46f6ebc` | +| Docker Linux image archive | `80a59f180e7c1e4e5861427d94b44605a54ba08ed12198c63c3ae98c35678f63` | +| Trusted Linux package manifest | `a3e8b73855d3b0854c5891cb5a10ff892aa0929e24046d2ce6fd28a245317e82` | diff --git a/openspec/changes/optimize-dockerhub-discovery-rotation/.openspec.yaml b/openspec/changes/optimize-dockerhub-discovery-rotation/.openspec.yaml new file mode 100644 index 0000000..1ea7e36 --- /dev/null +++ b/openspec/changes/optimize-dockerhub-discovery-rotation/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-09 diff --git a/openspec/changes/optimize-dockerhub-discovery-rotation/design.md b/openspec/changes/optimize-dockerhub-discovery-rotation/design.md new file mode 100644 index 0000000..196ac49 --- /dev/null +++ b/openspec/changes/optimize-dockerhub-discovery-rotation/design.md @@ -0,0 +1,91 @@ +## Context + +Managed DockerHub repository search is authenticated and bounded to two GET attempts per page, but standard discovery still builds one all-pages result before database admission. An unavailable expected page therefore discards successful pages and leaves the numeric query cursor on the same keyword. Database deduplication happens only after pagination, so known repositories cannot currently stop deeper requests. + +DockerHub search results are ordered but not snapshot-stable. A delayed retry of page N cannot reconstruct the exact earlier result window, so the design must combine idempotent page admission with periodic deep coverage rather than claim snapshot completeness. Repository anchors and immutable digest scan targets already have durable PostgreSQL identities, but none of the existing scan, resolver, projection, or source-cycle tables is a valid leaseable queue for failed discovery-page work. + +The active source uses a single DockerHub writer, a numeric main query cursor, and additive JSON state. The new query list is appended to preserve that cursor. Runtime configuration and application files may only be changed while the canonical supervisor is stopped. + +## Goals / Non-Goals + +**Goals:** + +- Persist each valid DockerHub page before requesting or acting on later pages. +- Stop ordinary pagination after two consecutive nonempty pages containing only identities known before the current pass. +- Use a 30-page, 100-result ceiling for every normal and deep query pass. +- Give each exact query a deep pass that bypasses seen-page stopping when 72 hours have elapsed since its last durable dispatch. +- Delegate unavailable page/query work to a durable, bounded, fenced retry queue without blocking main keyword rotation. +- Add the confirmed 12 product/framework queries and disable only periodic re-resolution of completed repository anchors. +- Preserve authenticated fail-closed requests, the existing two-GET page budget, cold/failed target exclusion, and immutable-digest error retries. + +**Non-Goals:** + +- Treating DockerHub pagination as a stable snapshot or guaranteeing successful external acquisition every 72 hours. +- Re-enabling cold/failed targets, rescanning successful immutable digests, or changing tag/layer selection, keychecks, scan workers, or scan budgets. +- Reusing scan/resolver queues for HTTP discovery work. +- Changing GitHub, GitLab, HuggingFace, or other source behavior. + +## Decisions + +### Admit repository pages incrementally + +The managed PostgreSQL DockerHub path will fetch page one first, validate `count`, and process the expected range sequentially. A narrow database operation will normalize and deduplicate the page, determine preexisting identities, insert bare repository anchors using the existing unresolved-anchor semantics, and return safe counts plus internal normalized identities needed for the current-pass knownness decision. It will not invoke resolver or scan claiming. The existing resolver gate runs once after pagination, not once per page. + +Sequential acquisition is selected over the current parallel remainder because requests for page three and beyond must be avoidable after pages one and two prove fully known. It also bounds in-memory results and makes page-level durability explicit. The authenticated page helper and its two-total-GET budget remain unchanged. + +Knownness is evaluated before page admission. A page increments the streak only when it is nonempty and every normalized repository existed before the pass. Identities inserted on an earlier page in the same pass do not count as preexisting if they appear again after result movement. Empty pages end the available range. Lookup uncertainty fails open by resetting the streak and continuing deeper. + +Before each managed pass, source state records an incomplete marker keyed by exact query and effective policy hash. A matching marker forces deep behavior on the next attempt and is cleared only after a durable `completed`, `completed_with_retries`, or `query_invalid` outcome. This prevents a failed pass from treating pages it admitted before the failure as old-enough evidence for an early stop after restart. + +### Delegate gaps to a PostgreSQL discovery retry queue + +Add a dedicated `discovery_retry_queue`; existing target, resolver, projection, outbox, and source-cycle tables have incompatible lifecycle and foreign-key semantics. Work is coalesced by a deterministic key over source, exact query, effective policy hash, pass kind, and page/range. Rows move through `pending -> leased -> deleted`, with retryable failure returning to `pending`, expired leases reclaimable, and removed/mismatched policy work moved to `held`. + +Claims use bounded `FOR UPDATE SKIP LOCKED` selection and random lease tokens. Completion and retry transitions require the exact row, owner, and token. A stale worker cannot acknowledge or replace newer work. The worker renews the same fenced lease before each page in a multi-page retry, so a bounded per-page lease cannot expire merely because an entire range takes longer than one lease interval. Retryable failures use exponential backoff with a cap and are never silently dropped at a maximum attempt count. Provider cooldown refunds the dispatch attempt only when the claim has made no remote request and uses the trusted retry time. Diagnostics store fixed safe categories, never authorization material or arbitrary response bodies. + +If page one is unavailable, enqueue query-level work because the expected range is unknown. If a later page fails, enqueue page work and continue when account availability permits. If the account pool is exhausted before a remaining tail can be attempted, coalesce the tail into range work instead of creating one row per unattempted page. Successful pages remain admitted. The source cycle may advance its main query only after every observed gap is durably delegated; retry persistence failure keeps the old cursor and marks the cycle failed. + +At most one due retry item is processed per normal source-loop iteration, independently of the main cursor. Main query state is saved before retry work can affect the next iteration. Retry success admits repositories before fenced acknowledgement; retry failure changes only retry state. This prevents a persistent page outage from head-of-line blocking the 61-keyword rotation. + +### Represent partial success explicitly + +Add `completed_with_retries` as a source-cycle outcome for a pass whose successful pages were admitted and whose gaps were durably delegated. The numeric query cursor advances for `completed`, `completed_with_retries`, and `query_invalid`, but not for `failed`, `source_failed`, or `backlog_only`. Invalid payloads, database admission failure, and retry-enqueue failure remain hard cycle failures rather than endlessly retryable transport work. + +Alternative considered: keep the keyword pinned while retaining successful pages. Rejected because it preserves head-of-line blocking. Alternative considered: advance after logging a failed page without durable work. Rejected because a moving search window can make the gap unrecoverable. + +### Schedule deep passes by exact query + +DockerHub source state gains a versioned map keyed by exact query, not numeric list position. Each record contains the effective search policy hash and UTC deep-dispatch timestamp. A query is deep-due when the record is absent, malformed, policy-mismatched, or at least 72 hours old. Its next main rotation pass ignores only the two-known-page stop; page-one count, 30-page cap, 100-result size, authentication, retries, and incremental admission remain mandatory. + +The dispatch timestamp is written only after successful page admission and durable delegation of any gaps. Scheduling from dispatch time avoids deep-pass storms during prolonged provider failure. Appended queries are immediately due without invalidating existing query timestamps. Removed queries are pruned from state and their retry work is held. This provides one deep dispatch per exact query on its first normal selection after the 72-hour boundary; it is not an external-success SLA. + +State-file loss causes conservative early deep passes, not missed passes. PostgreSQL remains authoritative for retry work and repository identity. + +### Apply the confirmed discovery policy + +Set DockerHub defaults to 30 pages and 100 results per page and align existing per-query page overrides so every query has the same acquisition ceiling. Append exactly the 12 confirmed terms, preserving existing order and numeric cursor safety. + +Set `docker_repository_refresh_max_per_cycle` to zero. This disables periodic claims of completed/resolved repository anchors only. `refresh_registry` remains enabled, so keyword search still runs; pending initial anchors and partial/error resolver rows remain eligible; new immutable digests discovered through those paths are scanned; immutable scan errors retain their existing retries. + +## Risks / Trade-offs + +- [The first pass of 12 new terms can inspect up to 36,000 search rows and create a large resolver backlog] -> Keep existing resolver, scan-worker, admission, and pipeline bounds; append terms rather than force a manual pass; observe backlog after natural rotation. +- [Two known pages do not prove all deeper pages are known] -> Treat stopping as an optimization and bypass it per exact query every 72 hours. +- [A delayed page retry sees a moving result window] -> Admit retries idempotently and rely on recurring deep passes for eventual coverage rather than snapshot claims. +- [A retry table can grow during a prolonged outage] -> Coalesce deterministic work keys, use range rows for unattempted tails, bound claims per loop, expose pending/oldest-age metrics, and never advance when durable delegation itself fails. +- [Sequential pages increase latency for a truly deep pass] -> Normal passes usually stop early; deep passes deliberately trade latency for bounded coverage and remain capped at 30 pages. +- [Disabling completed-anchor refresh can miss a new digest in a repository that no longer appears in search] -> This is the user's temporary policy choice; initial and failed resolution continue, and the switch can be restored independently. +- [JSON deep-state corruption can trigger extra load] -> Validate schema strictly, key by exact query/policy, and fail toward an early deep pass rather than suppressing coverage. + +## Migration Plan + +1. Create and validate the OpenSpec artifacts and deterministic tests before touching the live runtime. +2. Canonically stop the supervisor and verify all managed workers and PostgreSQL are down. +3. Apply the additive PostgreSQL schema migration, scanner/runner orchestration, source-state validation, tests, and configuration changes. +4. Run focused unit/SQL/migration/integration tests, then the broader relevant scanner and runtime-safety suites. +5. Canonically start the supervisor; verify PostgreSQL migration authority, pipeline readiness, DockerHub authentication, worker restart counters, retry/deep-state health, and absence of bytecode artifacts. +6. Roll back under another canonical stop by restoring code/config and leaving the additive retry table dormant; repository and immutable-target inserts are idempotent and require no destructive rollback. + +## Open Questions + +None. Keyword scope, 30x100 policy, 72-hour deep cadence, disabled completed-anchor refresh, preserved scan retries, and independent retry backlog were explicitly confirmed. diff --git a/openspec/changes/optimize-dockerhub-discovery-rotation/proposal.md b/openspec/changes/optimize-dockerhub-discovery-rotation/proposal.md new file mode 100644 index 0000000..080481b --- /dev/null +++ b/openspec/changes/optimize-dockerhub-discovery-rotation/proposal.md @@ -0,0 +1,31 @@ +## Why + +DockerHub discovery currently downloads its full configured page range before database deduplication, and one exhausted page discards every successful page while blocking keyword rotation. The authenticated 30-page window makes deeper discovery possible, but it needs incremental persistence, seen-page stopping, and durable retry delegation to remain efficient and avoid silent gaps or head-of-line blocking. + +## What Changes + +- **BREAKING**: replace complete-or-fail DockerHub pagination with page-level durable persistence; successful pages remain admitted when another page exhausts its two request attempts. +- Stop an ordinary query pass after two consecutive nonempty pages whose repository identities were already known before the pass. +- Run every DockerHub query with an effective ceiling of 30 pages and 100 results per page. +- Bypass seen-page stopping for a full deep pass of each exact query at least once per 72-hour scheduling interval. +- Delegate failed page/query acquisition to a fenced PostgreSQL discovery retry backlog before allowing the main keyword rotation to advance. +- Add 12 DockerHub product/framework queries: `open-webui`, `ragflow`, `dify`, `flowise`, `crewai`, `n8n`, `langflow`, `autogen`, `browser-use`, `openhands`, `anythingllm`, and `agent-zero`. +- Temporarily disable periodic re-resolution of completed DockerHub repository anchors while preserving initial resolution, partial/error resolver retries, immutable-digest scan retries, and normal keyword discovery. +- Preserve authenticated fail-closed search, the two-GET page budget, target uniqueness, cold/failed target policy, and secret-safe diagnostics. + +## Capabilities + +### New Capabilities +- `dockerhub-incremental-discovery`: Incremental DockerHub page admission, seen-page stopping, 72-hour deep passes, and durable failed-page retry work. + +### Modified Capabilities + +None. + +## Impact + +- Affects DockerHub pagination and authentication integration in `app/scanner.py`, source-cycle orchestration/state in `app/console_runner.py`, PostgreSQL schema and retry claims in `app/scanner_db.py`, and DockerHub settings in `app/config.yaml`. +- Adds a PostgreSQL discovery retry queue with bounded leases, fencing, backoff, and configured-query/policy validation. +- Changes DockerHub query rotation from failure-blocking to durable retry delegation and adds an initial bounded backlog from 12 new deep searches. +- Requires focused unit, SQL-shape, migration, PostgreSQL integration, source-state, configuration, and runtime health verification. +- Does not add dependencies or credential formats and does not change Docker tag selection, Registry authentication, layer scanning, scan workers, keychecks, or immutable-digest retry policy. diff --git a/openspec/changes/optimize-dockerhub-discovery-rotation/specs/dockerhub-incremental-discovery/spec.md b/openspec/changes/optimize-dockerhub-discovery-rotation/specs/dockerhub-incremental-discovery/spec.md new file mode 100644 index 0000000..b0b4456 --- /dev/null +++ b/openspec/changes/optimize-dockerhub-discovery-rotation/specs/dockerhub-incremental-discovery/spec.md @@ -0,0 +1,164 @@ +## ADDED Requirements + +### Requirement: DockerHub pages are admitted incrementally +The system SHALL validate and durably admit every successful DockerHub repository-search page before relying on later-page acquisition, while preserving target identity and cold/failed-target exclusion. + +#### Scenario: Successful page precedes a later failure +- **WHEN** an expected DockerHub page is valid and a later expected page exhausts its request budget +- **THEN** repositories from the successful page remain idempotently admitted and the later failure cannot roll them back + +#### Scenario: Page admission fails +- **WHEN** repository normalization or durable page admission fails +- **THEN** the source cycle fails, does not treat the page as complete, and does not advance the main query cursor + +#### Scenario: Resolver processing follows pagination +- **WHEN** one or more pages admit repository anchors +- **THEN** the existing Docker resolver gate runs at most once after the pass rather than once per page + +### Requirement: Ordinary discovery stops after consecutive known pages +The system SHALL stop an ordinary DockerHub query before requesting another page after two consecutive nonempty pages contain only repository identities that existed before the current pass. + +#### Scenario: First two pages were previously known +- **WHEN** pages one and two are nonempty and every normalized repository identity existed before the pass +- **THEN** the pass completes without requesting page three + +#### Scenario: Page contains a new repository +- **WHEN** either of the last two pages contains a repository identity not known before the pass +- **THEN** the consecutive-known counter resets and discovery continues within the effective range + +#### Scenario: Current-pass duplicate appears later +- **WHEN** a repository first admitted earlier in the same pass appears on a later page +- **THEN** that identity is not treated as preexisting evidence for the later page's known-page stop + +#### Scenario: Knownness cannot be determined +- **WHEN** the database cannot establish complete pre-pass knownness for a valid page +- **THEN** discovery fails open by continuing deeper rather than stopping early + +#### Scenario: Previous pass ended incompletely +- **WHEN** an exact query and policy have a retained incomplete-pass marker +- **THEN** the next selected pass bypasses known-page stopping until a durable pass outcome clears the marker + +### Requirement: Deep discovery bypasses seen-page stopping every 72 hours +The system SHALL make each exact configured DockerHub query deep-due no later than 72 hours after its last durable deep dispatch and SHALL bypass the consecutive-known-page stop on that query's next selected pass. + +#### Scenario: Exact query reaches its deep interval +- **WHEN** at least 72 hours have elapsed since the query's last durable deep dispatch +- **THEN** its next selected pass processes the available expected range without stopping on known pages + +#### Scenario: Query is new or policy changes +- **WHEN** an exact configured query has no valid state or its effective search-policy hash changes +- **THEN** that query is immediately deep-due without resetting unrelated query schedules + +#### Scenario: Deep pass has durable gaps +- **WHEN** valid pages are admitted and unavailable pages are durably delegated to retry work +- **THEN** the deep dispatch timestamp advances while completion remains represented by the outstanding retry work + +#### Scenario: Provider prevents acquisition +- **WHEN** neither successful pages nor durable retry delegation can establish a deep dispatch +- **THEN** the system does not advance that query's deep timestamp + +### Requirement: DockerHub page gaps use a durable retry backlog +The system SHALL persist unavailable DockerHub query/page work in a dedicated PostgreSQL retry backlog before allowing the main keyword rotation to advance. + +#### Scenario: Page one is unavailable +- **WHEN** page one exhausts its bounded request/account handling and the expected range is unknown +- **THEN** the system durably enqueues query-level retry work + +#### Scenario: Later page is unavailable +- **WHEN** a later expected page exhausts its bounded request/account handling +- **THEN** the system durably enqueues page-level work and retains every successfully admitted page + +#### Scenario: Remaining tail cannot be attempted +- **WHEN** provider/account exhaustion prevents remote attempts for a known remaining page range +- **THEN** the system coalesces the unattempted range into bounded retry work instead of creating unbounded individual rows + +#### Scenario: Durable delegation succeeds +- **WHEN** every observed acquisition gap has durable retry work +- **THEN** the cycle records `completed_with_retries` and advances the main keyword independently of retry processing + +#### Scenario: Durable delegation fails +- **WHEN** retry work cannot be persisted authoritatively +- **THEN** the cycle fails and retains the current keyword cursor + +### Requirement: Discovery retry claims are bounded and fenced +The system SHALL coalesce retry work by source, exact query, effective policy, pass kind, and page/range; SHALL process bounded due work with expiring leases; and MUST reject stale acknowledgements. + +#### Scenario: Concurrent workers claim due work +- **WHEN** multiple workers attempt to claim the same due retry row +- **THEN** at most one receives the active lease token + +#### Scenario: Lease expires +- **WHEN** a worker fails to complete work before its lease expires +- **THEN** the row becomes reclaimable without deleting its attempt history + +#### Scenario: Multi-page retry remains active +- **WHEN** a leased query or range retry is about to request another page +- **THEN** the worker renews the same owner-and-token fence before acquisition and stops if renewal is rejected + +#### Scenario: Stale worker finishes +- **WHEN** a worker presents an obsolete owner or lease token +- **THEN** it cannot acknowledge, delete, defer, or hold the newer work + +#### Scenario: Retryable acquisition fails again +- **WHEN** leased work encounters another retryable failure +- **THEN** it returns to pending with bounded exponential backoff and is not silently dropped at an attempt limit + +#### Scenario: Provider cooldown follows partial remote progress +- **WHEN** an earlier page in the same claim made a remote request before a later local provider cooldown +- **THEN** the dispatch attempt is not refunded + +#### Scenario: Query is removed or policy is obsolete +- **WHEN** retry work no longer matches an exact configured query and effective policy +- **THEN** it is held and cannot execute against stale discovery policy + +### Requirement: Retry work does not block main keyword rotation +The system SHALL process at most a bounded amount of due discovery retry work per source-loop iteration independently of the saved main query cursor. + +#### Scenario: Persistent page failure exists +- **WHEN** a retry row remains unavailable across multiple attempts +- **THEN** ordinary configured keywords continue rotating while the row follows its own backoff + +#### Scenario: Retry succeeds +- **WHEN** leased page or range work returns valid repositories +- **THEN** repositories are durably admitted before the fenced retry acknowledgement + +### Requirement: DockerHub search policy uses the confirmed breadth +The system SHALL use an effective ceiling of 30 pages and 100 results per page for every configured DockerHub query and SHALL include exactly the 12 confirmed new product/framework terms in addition to the existing ordered query set. + +#### Scenario: Ordinary pass reaches known content +- **WHEN** a normal 30-by-100 query pass reaches two consecutive preexisting-known pages +- **THEN** it stops early despite the larger configured ceiling + +#### Scenario: Deep pass remains novel +- **WHEN** a due deep pass continues to return pages containing new repositories +- **THEN** it processes up to the page-one result boundary or the 30-page safety cap + +#### Scenario: Query list is expanded +- **WHEN** the configuration is loaded after deployment +- **THEN** `open-webui`, `ragflow`, `dify`, `flowise`, `crewai`, `n8n`, `langflow`, `autogen`, `browser-use`, `openhands`, `anythingllm`, and `agent-zero` each appear once at the ordered tail + +### Requirement: Completed repository refresh is disabled without changing retries +The system SHALL disable periodic re-resolution of successfully completed DockerHub repository anchors while retaining normal search, initial/partial resolver work, and immutable-digest error retries. + +#### Scenario: Completed anchor becomes periodically due +- **WHEN** a resolved repository anchor reaches its prior refresh interval +- **THEN** periodic policy does not claim it solely for refresh + +#### Scenario: Initial or partial anchor is due +- **WHEN** an unresolved or retryable partial repository anchor is due +- **THEN** the existing resolver remains eligible to process it + +#### Scenario: Immutable digest scan fails retryably +- **WHEN** an immutable Docker image scan meets the existing retry conditions +- **THEN** its current bounded retry and backoff behavior remains unchanged + +### Requirement: Incremental discovery remains authenticated and secret-safe +The system MUST preserve explicit-pool fail-closed Hub bearer authentication and MUST NOT persist or emit credentials, bearer values, authorization headers, raw response bodies, arbitrary exception text, or repository targets through retry/deep-state diagnostics. + +#### Scenario: Search or retry fails +- **WHEN** DockerHub page acquisition or retry processing reports an error +- **THEN** diagnostics contain only bounded status/category/count metadata required for operation + +#### Scenario: Explicit account pool is unavailable +- **WHEN** no configured account can perform repository search +- **THEN** normal and retry acquisition fail closed without anonymous fallback diff --git a/openspec/changes/optimize-dockerhub-discovery-rotation/tasks.md b/openspec/changes/optimize-dockerhub-discovery-rotation/tasks.md new file mode 100644 index 0000000..eccbdb9 --- /dev/null +++ b/openspec/changes/optimize-dockerhub-discovery-rotation/tasks.md @@ -0,0 +1,30 @@ +## 1. Runtime Safety + +- [x] 1.1 Canonically stop the live supervisor and verify all managed workers and PostgreSQL are down before editing application or configuration files + +## 2. Durable Discovery Storage + +- [x] 2.1 Add and validate the additive PostgreSQL discovery retry queue schema, required indexes, migration metadata, and bounded lifecycle fields +- [x] 2.2 Implement strict page admission plus retry enqueue, claim, reclaim, backoff, hold, and fenced completion database operations +- [x] 2.3 Add SQL-shape and PostgreSQL migration/integration coverage for idempotency, concurrent claims, expired leases, and stale-token rejection + +## 3. Incremental DockerHub Discovery + +- [x] 3.1 Refactor managed DockerHub pagination to validate, deduplicate, and durably admit successful pages sequentially without invoking the resolver per page +- [x] 3.2 Implement two-consecutive-preexisting-page stopping with current-pass duplicate protection and fail-open knownness handling +- [x] 3.3 Implement per-query policy-hashed 72-hour deep scheduling that bypasses only seen-page stopping +- [x] 3.4 Delegate page-one, later-page, and unavailable-tail failures to durable retry work and advance the main cursor only after authoritative delegation +- [x] 3.5 Process bounded retry work independently of main rotation with lease fencing, backoff, policy/query validation, and safe diagnostics + +## 4. Discovery Policy + +- [x] 4.1 Configure every DockerHub query for a 30-page by 100-result ceiling and append the exact 12 confirmed product/framework queries once +- [x] 4.2 Disable periodic completed-anchor digest refresh while preserving initial/partial resolver and immutable-digest scan retries +- [x] 4.3 Add permanent configuration regressions for query order/count, effective breadth, disabled periodic refresh, and unchanged retry/resource policy + +## 5. Verification And Deployment + +- [x] 5.1 Run focused pagination, retry-queue, source-state, migration, and configuration tests without creating application bytecode +- [x] 5.2 Run broader relevant scanner, queue, runtime-safety, and PostgreSQL integration suites and complete an independent read-only review +- [x] 5.3 Strictly validate the OpenSpec change and verify no secret-bearing diagnostics or unrelated behavior changes +- [x] 5.4 Canonically start the runtime and verify PostgreSQL readiness, migrations, pipeline workers, DockerHub authentication/worker health, retry/deep state, restart counters, and absence of application bytecode diff --git a/openspec/changes/prune-low-yield-discovery-keywords/.openspec.yaml b/openspec/changes/prune-low-yield-discovery-keywords/.openspec.yaml new file mode 100644 index 0000000..34f54d2 --- /dev/null +++ b/openspec/changes/prune-low-yield-discovery-keywords/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-05 diff --git a/openspec/changes/prune-low-yield-discovery-keywords/design.md b/openspec/changes/prune-low-yield-discovery-keywords/design.md new file mode 100644 index 0000000..e29c955 --- /dev/null +++ b/openspec/changes/prune-low-yield-discovery-keywords/design.md @@ -0,0 +1,75 @@ +## Context + +The configured sources contain 543 query occurrences covering 133 normalized terms. Canonical PostgreSQL lineage links discovery scans to credentials through both candidate and result records, and the dashboard already defines the strict `usable_llm` tier. Applying that rule across every provider identified 38 non-operational terms with zero historical strict-usable linkage despite 117,492 scans and 1,890.6 cumulative scanner-hours. The same terms consumed 18,641 scans and 317.1 scanner-hours in the latest 30-day window. + +The runtime configuration is code-authority protected. Query rotation state, target queues, scan history, and credential history are separate persisted authorities and must not be rewritten to deploy this change. + +## Goals / Non-Goals + +**Goals:** +- Remove only globally zero-yield terms with enough exposure to support a conservative decision. +- Apply one decision consistently anywhere the exact normalized term is configured. +- Preserve operational source sentinels and all terms with any demonstrated strict-usable linkage. +- Remove stale per-query overrides and verify deterministic post-prune query sets. +- Deploy through the coordinated authority lifecycle and verify normal source rotation. + +**Non-Goals:** +- Deleting, reprioritizing, or rewriting existing target backlog or historical records. +- Optimizing for one provider, broad `alive`, raw findings, or candidate volume. +- Changing detector routing, keycheck classification, source concurrency, scan limits, or query-state files. +- Claiming that a retired term can never produce a useful credential in the future. + +## Decisions + +### Use all-provider strict-usable evidence + +A term is eligible only when no credential linked to that term has ever reached the canonical dashboard `usable_llm` tier and no earliest-origin credential attributed to it has reached that tier. Candidate/result scan links are unioned before attribution so migrated and resolver-routed credentials are not lost. + +The decision is global by case-normalized exact term. If a term produced one strict-usable credential for any provider or source, it remains configured everywhere. This is more conservative than pruning source-term pairs independently and avoids removing cross-provider terms such as `groq`, `llm`, `chat`, `rag`, or `langchain`. + +Alternative: use broad `status_group=alive` or OpenAI-only yield. Rejected because broad alive contains unproven and historically misclassified statuses, while provider-only analysis can remove terms that work for another provider. + +### Require meaningful exposure + +A zero-yield term qualifies when either it has at least 30 linked credential observations, or it has at least 200 completed scan events and 20 cumulative scanner-hours. The credential branch tests precision; the cost branch catches terms that repeatedly consume work without reaching candidate intake. The threshold is applied to all retained history, with the latest 30-day cost recorded as corroborating evidence. + +Alternative: remove every zero-yield term. Rejected because recent and low-sample terms have insufficient evidence. Those terms remain canaries. + +### Exempt source-operational sentinels + +`gharchive`, `gharchive-files`, and `gists` are sole query tokens used to operate dedicated sources rather than interchangeable discovery keywords. They remain even though they have no strict-usable attribution. Emptying those lists would disable or invalidate source operation rather than merely prune a search term. + +### Retire the approved cohort consistently + +Remove these 32 terms from GitHub, GitLab, DockerHub, npm, PyPI, and package-git: `autonomous`, `benchmarks`, `claw`, `code-assistant`, `codegen`, `dspy`, `embedding`, `embeddings`, `eval`, `evals`, `gateway`, `grok`, `haystack`, `inference`, `inference-api`, `knowledge`, `llamaindex`, `model`, `model-router`, `models`, `ollama`, `orchestration`, `prompts`, `replicate`, `rerank`, `reranker`, `retrieval`, `router`, `tokenizer`, `tool-use`, `vector`, and `vllm`. + +Remove `chatgpt`, `gpt`, and `moonshot` from those six sources and Postman. Remove `openai` from GitHub, GitLab, DockerHub, and Postman. Remove `dashscope-intl.aliyuncs.com` and `generativelanguage.googleapis.com` from Postman. + +This removes 219 occurrences. Resulting list sizes are GitHub 64, GitLab 46, DockerHub 46, npm 43, PyPI 43, package-git 43, and Postman 33. + +### Keep deployment configuration-only + +Delete the three `openai` query overrides together with the query entries. Existing rotation reads `query_index` modulo the current list length, so no persisted state edit is needed. Runtime is stopped before editing and restarted only after tests and strict OpenSpec validation. + +Alternative: rewrite query indices or purge queued targets attributed to removed terms. Rejected because both mutate independent durable authority and are unnecessary for preventing future discovery. + +## Risks / Trade-offs + +- [Historical zero yield may not predict future supply] -> Keep low-sample terms, retain all historical evidence, and make rollback a configuration-only restoration. +- [Earliest-origin attribution can hide useful rediscovery] -> Require zero strict-usable linkage across every scan link in addition to zero origin yield. +- [Large list reduction changes rotation cadence] -> Verify exact list sizes and allow normal modulo-based state handling; do not edit source state. +- [Completed OpenAI rollout previously required the literal term] -> Record the requirement retirement explicitly and retain higher-signal bounded ecosystem queries. +- [Authority drift during a live edit] -> Use coordinated stop, test, and canonical start rather than relying on fail-close shutdown. + +## Migration Plan + +1. Add configuration contract tests for the exact retired set, retained sentinels, uniqueness, post-prune sizes, and absence of orphaned overrides. +2. Stop the authenticated runtime coordinately. +3. Remove the 219 query occurrences and three matching overrides from `app/config.yaml`; do not edit state or queue data. +4. Run focused query tests, configuration/runtime safety tests as applicable, and strict OpenSpec validation. +5. Restart through `start_runtime.ps1` and verify authenticated supervisor, PostgreSQL, pipeline readiness, source processes, and query-list loading. +6. Roll back by restoring the configuration entries and overrides through the same coordinated lifecycle if source health regresses. + +## Open Questions + +None. diff --git a/openspec/changes/prune-low-yield-discovery-keywords/proposal.md b/openspec/changes/prune-low-yield-discovery-keywords/proposal.md new file mode 100644 index 0000000..bd2221e --- /dev/null +++ b/openspec/changes/prune-low-yield-discovery-keywords/proposal.md @@ -0,0 +1,23 @@ +## Why + +Current discovery rotations spend substantial scanner time on query terms that have accumulated meaningful exposure without linking to a single strict-usable credential for any provider. Removing only this globally zero-yield cohort reduces avoidable discovery and scan work while preserving every query with demonstrated usable yield. + +## What Changes + +- Remove 38 sufficiently exposed, globally zero-yield search terms from the configured GitHub, GitLab, DockerHub, npm, PyPI, package-git, and Postman rotations. +- Remove query-scoped overrides whose corresponding query is retired. +- Preserve source-operational sentinel queries and every term linked to at least one historical strict-usable credential for any provider. +- Preserve historical queue rows, scan results, credential lineage, deduplication state, and all runtime concurrency limits. +- Define a repeatable evidence rule for future pruning instead of using raw findings or provider-specific yield alone. + +## Capabilities + +### New Capabilities +- `discovery-keyword-pruning`: Evidence-based, all-provider retirement of sufficiently tested zero-yield discovery terms. + +### Modified Capabilities +- `openai-discovery-coverage`: Retire the literal `openai` core query after its bounded rollout produced no strict-usable credential for any provider. + +## Impact + +The change affects `app/config.yaml`, focused query-configuration tests, and authority-managed source rotation after a coordinated restart. It removes 219 configured query occurrences but introduces no schema migration, dependency, queue rewrite, credential recheck, detector change, or scan-concurrency change. diff --git a/openspec/changes/prune-low-yield-discovery-keywords/specs/discovery-keyword-pruning/spec.md b/openspec/changes/prune-low-yield-discovery-keywords/specs/discovery-keyword-pruning/spec.md new file mode 100644 index 0000000..399cf49 --- /dev/null +++ b/openspec/changes/prune-low-yield-discovery-keywords/specs/discovery-keyword-pruning/spec.md @@ -0,0 +1,61 @@ +## ADDED Requirements + +### Requirement: All-provider strict-yield pruning rule +The system SHALL retire a discovery term only when canonical lineage shows zero historical strict-usable credential linkage for every provider and the term has meaningful measured exposure. + +#### Scenario: Any strict-usable linkage preserves a term +- **WHEN** any credential linked to a configured term has ever met the canonical `usable_llm` rule +- **THEN** that term SHALL remain in every configured source rotation + +#### Scenario: Credential exposure qualifies a zero-yield term +- **WHEN** a term has zero strict-usable linkage and at least 30 linked credential observations +- **THEN** the term SHALL qualify for retirement + +#### Scenario: Scanner-cost exposure qualifies a zero-yield term +- **WHEN** a term has zero strict-usable linkage, at least 200 scan events, and at least 20 cumulative scanner-hours +- **THEN** the term SHALL qualify for retirement + +#### Scenario: Low-exposure zero-yield term remains a canary +- **WHEN** a zero-yield term satisfies neither exposure condition +- **THEN** it SHALL remain configured until more evidence is available + +### Requirement: Approved global retirement cohort +The system SHALL omit the approved 38-term zero-yield cohort from every source rotation where each exact term was configured. + +#### Scenario: Shared broad-source cohort is removed +- **WHEN** GitHub, GitLab, DockerHub, npm, PyPI, or package-git loads its query rotation +- **THEN** it SHALL omit `autonomous`, `benchmarks`, `claw`, `code-assistant`, `codegen`, `dspy`, `embedding`, `embeddings`, `eval`, `evals`, `gateway`, `grok`, `haystack`, `inference`, `inference-api`, `knowledge`, `llamaindex`, `model`, `model-router`, `models`, `ollama`, `orchestration`, `prompts`, `replicate`, `rerank`, `reranker`, `retrieval`, `router`, `tokenizer`, `tool-use`, `vector`, and `vllm` + +#### Scenario: Cross-source zero-yield terms are removed +- **WHEN** an affected rotation is loaded +- **THEN** `chatgpt`, `gpt`, and `moonshot` SHALL be absent from GitHub, GitLab, DockerHub, npm, PyPI, package-git, and Postman, and `openai` SHALL be absent from GitHub, GitLab, DockerHub, and Postman + +#### Scenario: Zero-yield Postman signatures are removed +- **WHEN** Postman loads its query rotation +- **THEN** `dashscope-intl.aliyuncs.com` and `generativelanguage.googleapis.com` SHALL be absent + +#### Scenario: Post-prune list sizes are deterministic +- **WHEN** canonical configuration is loaded +- **THEN** query counts SHALL be GitHub 64, GitLab 46, DockerHub 46, npm 43, PyPI 43, package-git 43, and Postman 33 + +### Requirement: Operational and historical authority is preserved +Keyword retirement SHALL stop future discovery for the retired terms without deleting or rewriting source state, target queues, scans, findings, credentials, or results. + +#### Scenario: Dedicated source sentinels remain +- **WHEN** archive and gist source rotations are loaded +- **THEN** `gharchive`, `gharchive-files`, and `gists` SHALL remain as their sole configured query tokens + +#### Scenario: Persisted rotation index remains valid +- **WHEN** an existing query index exceeds a shortened query list +- **THEN** normal modulo-based rotation SHALL select a valid configured query without a state-file edit + +#### Scenario: Existing backlog remains intact +- **WHEN** the pruned configuration is deployed +- **THEN** previously admitted targets and all historical attribution records SHALL remain unchanged + +### Requirement: Retired query overrides are removed +The canonical configuration SHALL NOT retain a query override for a retired query. + +#### Scenario: Literal OpenAI overrides are absent +- **WHEN** GitHub, GitLab, and DockerHub configuration is loaded +- **THEN** each source SHALL omit the `openai` query override while preserving overrides for retained bounded queries diff --git a/openspec/changes/prune-low-yield-discovery-keywords/specs/openai-discovery-coverage/spec.md b/openspec/changes/prune-low-yield-discovery-keywords/specs/openai-discovery-coverage/spec.md new file mode 100644 index 0000000..f8414ec --- /dev/null +++ b/openspec/changes/prune-low-yield-discovery-keywords/specs/openai-discovery-coverage/spec.md @@ -0,0 +1,37 @@ +## MODIFIED Requirements + +### Requirement: Query-scoped safety bounds +The system SHALL support exact-query overrides for configured queries only, limited to `pages`, `per_page`, and `max_targets`, without changing source-wide defaults for other queries. + +#### Scenario: Configured query receives bounded arguments +- **WHEN** a source builds arguments for a configured query with an exact override +- **THEN** it SHALL apply that query's configured page, page-size, and target bounds + +#### Scenario: Retired query has no override +- **WHEN** a query is removed from a source rotation +- **THEN** the source SHALL NOT retain an override for that query + +#### Scenario: Ordinary query retains source defaults +- **WHEN** the same source builds arguments for any query without an override +- **THEN** it SHALL retain the source-wide page, page-size, and target values + +#### Scenario: Invalid override fails closed +- **WHEN** a query override is not a mapping or contains a key outside the allowlist +- **THEN** argument construction SHALL fail before discovery or queue mutation + +## REMOVED Requirements + +### Requirement: Exact OpenAI core discovery +**Reason**: The completed bounded rollout produced 43 linked origin credentials and no strict-usable credential for any provider, meeting the approved global retirement rule. + +**Migration**: Remove `openai` from GitHub, GitLab, and DockerHub rotations and allow normal modulo-based query rotation to continue without editing persisted source state. + +### Requirement: Source-specific rollout limits +**Reason**: The exact-query rollout is complete and its query is being retired, so source-specific `openai` execution bounds are no longer active policy. + +**Migration**: Remove the three matching `openai` overrides while retaining the generic exact-query override mechanism and all overrides for configured ecosystem queries. + +### Requirement: End-to-end canary evidence +**Reason**: The exact-query canary reached terminal evidence and its measured all-provider strict yield is captured by the pruning decision. + +**Migration**: Evaluate future keyword retirement under `discovery-keyword-pruning` using canonical all-provider lineage and measured exposure. diff --git a/openspec/changes/prune-low-yield-discovery-keywords/tasks.md b/openspec/changes/prune-low-yield-discovery-keywords/tasks.md new file mode 100644 index 0000000..a7180dd --- /dev/null +++ b/openspec/changes/prune-low-yield-discovery-keywords/tasks.md @@ -0,0 +1,19 @@ +## 1. Configuration Contract + +- [x] 1.1 Update focused query tests to assert the exact retired cohort, retained sentinels, retained productive terms, and deterministic list sizes. +- [x] 1.2 Assert that retired queries have no orphaned query overrides while retained bounded ecosystem queries remain unchanged. + +## 2. Runtime Configuration + +- [x] 2.1 Stop the authenticated runtime through the coordinated lifecycle before changing authority-covered configuration. +- [x] 2.2 Remove the approved 219 query occurrences and three `openai` overrides from `app/config.yaml` without modifying persisted state or backlog data. + +## 3. Verification + +- [x] 3.1 Run focused query/configuration tests and verify canonical configuration loads with the required query sets and counts. +- [x] 3.2 Run strict OpenSpec validation and relevant supervisor/runtime safety tests. + +## 4. Deployment + +- [x] 4.1 Start the runtime through `start_runtime.ps1` and verify authenticated supervisor and managed PostgreSQL readiness. +- [x] 4.2 Verify pipeline readiness, normal source processes, keychecks, and post-prune query loading without queue mutations. diff --git a/openspec/changes/rescan-updated-core-targets/.openspec.yaml b/openspec/changes/rescan-updated-core-targets/.openspec.yaml new file mode 100644 index 0000000..e685d45 --- /dev/null +++ b/openspec/changes/rescan-updated-core-targets/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-08-25 diff --git a/openspec/changes/rescan-updated-core-targets/design.md b/openspec/changes/rescan-updated-core-targets/design.md new file mode 100644 index 0000000..f747024 --- /dev/null +++ b/openspec/changes/rescan-updated-core-targets/design.md @@ -0,0 +1,87 @@ +## Context + +GitHub, GitLab, and HuggingFace discovery return stable target identities together with remote update timestamps. The runner currently discards those timestamps and the PostgreSQL queue permanently deduplicates targets by `(source, normalized_target)`. A completed repository or Space is therefore never scanned again even when its content changes. + +Production discovery is already saturated: less than one percent of fetched records are new target identities and the core queue is often empty. The change must restore changed-content coverage without turning every rediscovery into a rescan, growing one queue row per revision, or creating an unbounded backlog. + +## Goals / Non-Goals + +**Goals:** +- Preserve a bounded remote content-update signal for GitHub, GitLab, and HuggingFace targets. +- Rescan a completed target only when discovery observes content newer than the revision covered by its latest claim. +- Coalesce multiple remote updates into one mutable queue row and one pending follow-up. +- Bound changed-target admission and expose it separately from new-target admission. +- Roll out without treating every legacy row as changed. + +**Non-Goals:** +- Requeue terminal failures or actively leased/deferred targets. +- Rescan every known target on a timer. +- Change DockerHub's digest-based identity and refresh behavior. +- Guarantee that provider activity timestamps always represent content changes. +- Enable inactive historical sources or enlarge global scan concurrency. + +## Decisions + +### Carry one normalized discovery record + +The runner will represent eligible discovery results internally as a target plus an optional UTC `remote_modified_at`. GitHub uses `pushed_at`, GitLab uses `last_activity_at`, and HuggingFace uses `lastModified`. Missing, malformed, or non-monotonic timestamps remain valid discovery results but cannot trigger an updated-target rescan. + +HuggingFace discovery will use a newest-modified feed so old Spaces changed recently are observable. Identity-only known-page stopping will be disabled when updated-target rescans are enabled; hard page and result limits remain the discovery bound. + +Alternatives rejected: +- Repository `updated_at` on GitHub, because metadata-only edits are not content pushes. +- A HEAD-SHA request per repository, because it multiplies API traffic and rate-limit exposure. +- Revision-aware early stopping, because a known first page does not prove later pages contain no changed targets. + +### Keep one queue row and two remote timestamps + +`target_queue` will gain nullable `remote_modified_at` and `scan_remote_modified_at` columns. Discovery monotonically advances `remote_modified_at`. Claiming atomically copies the currently observed value into `scan_remote_modified_at`, recording what that scan covers. + +The separate claim snapshot is required because a remote update can arrive while a scan is running. On completion, a newer observed timestamp remains ahead of the claimed timestamp and is eligible for exactly one later scan. Encoding revisions into `normalized_target` was rejected because it would grow queue rows and weaken queue authority. + +For a legacy completed row with no claim snapshot, `completed_at` is the rollout baseline. It is eligible only when the first valid remote timestamp observed is newer than that completion. This prevents a migration surge while still admitting updates that occurred after the historical scan. + +### Observe and requeue atomically under a hard budget + +A PostgreSQL transaction will upsert discovery observations and requeue at most `updated_rescan_max_per_cycle` eligible rows for one source. Eligibility requires: +- `status='done'`; +- a strictly newer observed timestamp than `scan_remote_modified_at`, or than `completed_at` for a legacy row; +- elapsed `updated_rescan_cooldown_seconds` since completion; +- no lease, reservation, claim, or resolver authority. + +Eligible rows are locked with `FOR UPDATE SKIP LOCKED`. Requeue resets only retry/completion scheduling fields required for a normal pending claim. Failed, pending, deferred, in-progress, unresolved, unchanged, and invalid-timestamp rows are never promoted by this path. + +The initial production setting is one updated target per source cycle. Existing backlog-first behavior remains enabled, so a source drains its admitted work before discovery can admit more; `refresh_registry` is not enabled by this change. + +Alternatives rejected: +- Global `requeue_done=True`, because it requeues unchanged rows on every cycle. +- Comparing only remote time with local completion time forever, because provider and host clocks differ and a mid-scan update can be lost. +- A separate maintenance queue, because it duplicates existing lease, reservation, and completion authority. + +### Account for updated targets separately + +`source_cycles` will gain `queued_updated_count`. `queued_new_count` keeps its current meaning. Source logs and dashboard aggregation will report changed-target admissions separately so rollout volume and yield can be audited. + +## Risks / Trade-offs + +- [GitLab activity can change without a repository push] -> Use a one-target-per-cycle cap and cooldown; report updated admissions separately. +- [Provider clock skew] -> Require strict monotonicity and use the claimed remote timestamp after the first revision-aware scan. +- [More discovery API traffic after disabling identity-only early stop] -> Keep existing hard page/per-page limits and source intervals. +- [Changed targets consume capacity without useful findings] -> Start at one per cycle and compare updated-target yield before increasing the cap. +- [Schema rollout while runtime is active] -> Stop the authority-managed runtime, apply schema through the normal initialization path, run tests, then restart through `start_runtime.ps1`. +- [Rollback leaves nullable columns] -> Disable the feature in source configuration; nullable columns and metrics are backward-compatible and can remain. + +## Migration Plan + +1. Add nullable queue columns, the source-cycle counter, and a partial eligibility index through idempotent schema initialization. +2. Deploy code and tests with updated-target rescans disabled by default. +3. Enable GitHub, GitLab, and HuggingFace with a cap of one and a conservative cooldown; leave DockerHub unchanged. +4. Restart the authority-managed runtime and verify discovery, queue authority, projection, and source health. +5. Observe changed-target admission, completion, findings, and worker occupancy before changing any cap. + +Rollback is configuration-first: disable updated-target rescans and restart the managed runtime. Existing pending work completes under normal queue semantics; no destructive data migration is required. + +## Open Questions + +- Whether production evidence supports different cooldowns per source after the initial canary. +- Whether a later change should add provider-specific immutable revisions when APIs can supply them without extra requests. diff --git a/openspec/changes/rescan-updated-core-targets/proposal.md b/openspec/changes/rescan-updated-core-targets/proposal.md new file mode 100644 index 0000000..7535531 --- /dev/null +++ b/openspec/changes/rescan-updated-core-targets/proposal.md @@ -0,0 +1,28 @@ +## Why + +GitHub, GitLab, and HuggingFace discovery continuously observe recently changed repositories and Spaces, but the queue permanently suppresses every URL or Space ID after its first scan. Production consequently fetched roughly 290,000 discovery results in the latest 24 hours while admitting less than one percent as targets, leaving scan capacity underused and ignoring secrets added to already-known projects. + +## What Changes + +- Preserve the remote content-update timestamp supplied by GitHub, GitLab, and HuggingFace discovery. +- Requeue a completed target only when the observed remote content timestamp is newer than its last completed scan. +- Bound changed-target admission per source cycle and enforce a per-target cooldown. +- Keep active, deferred, failed, and unchanged targets untouched. +- Expose changed-target requeue counts separately from newly discovered target counts. +- Leave DockerHub digest-based refresh behavior unchanged. + +## Capabilities + +### New Capabilities +- `updated-target-rescan`: Safely detect and rescan remotely changed core repository and Space targets under bounded rollout controls. + +### Modified Capabilities + +None. + +## Impact + +- Discovery metadata and target preparation in `app/scanner.py` and `app/console_runner.py`. +- PostgreSQL target queue state, migrations, cycle accounting, and dashboard observability in `app/scanner_db.py` and `app/dashboard.py`. +- Source configuration for GitHub, GitLab, and HuggingFace. +- Focused queue, discovery, and PostgreSQL integration tests. diff --git a/openspec/changes/rescan-updated-core-targets/specs/updated-target-rescan/spec.md b/openspec/changes/rescan-updated-core-targets/specs/updated-target-rescan/spec.md new file mode 100644 index 0000000..c999777 --- /dev/null +++ b/openspec/changes/rescan-updated-core-targets/specs/updated-target-rescan/spec.md @@ -0,0 +1,87 @@ +## ADDED Requirements + +### Requirement: Discovery preserves remote content recency +The system SHALL preserve a normalized remote content-update timestamp for GitHub, GitLab, and HuggingFace discovery records and SHALL keep DockerHub digest-based identity behavior unchanged. + +#### Scenario: Source-specific remote timestamp is retained +- **WHEN** GitHub supplies `pushed_at`, GitLab supplies `last_activity_at`, or HuggingFace supplies `lastModified` +- **THEN** the queue observation stores the valid UTC timestamp with the target identity + +#### Scenario: Missing or malformed remote timestamp +- **WHEN** a discovery result has no valid remote content-update timestamp +- **THEN** the target remains eligible for normal new-target admission but MUST NOT trigger an updated-target rescan + +#### Scenario: DockerHub discovery +- **WHEN** DockerHub resolves an image tag +- **THEN** the existing digest identity and refresh behavior remain authoritative without updated-target promotion + +### Requirement: Only changed completed targets are promoted +The system SHALL promote a known target only when it is completed, its observed remote timestamp is strictly newer than the remote timestamp covered by its latest scan, and its configured cooldown has elapsed. + +#### Scenario: Completed target changed after its covered revision +- **WHEN** discovery observes a newer valid remote timestamp for a completed target after cooldown +- **THEN** the target becomes pending for one normal fenced scan + +#### Scenario: Unchanged target is rediscovered +- **WHEN** discovery observes the same or an older remote timestamp +- **THEN** the completed target remains unchanged and unclaimable + +#### Scenario: Legacy completed target is first observed +- **WHEN** a completed target has no claimed remote timestamp +- **THEN** it is promoted only if the observed remote timestamp is strictly newer than its last completion time + +#### Scenario: Non-completed target is rediscovered +- **WHEN** the target is failed, pending, deferred, in progress, unresolved, leased, claimed, or reserved +- **THEN** updated-target discovery MUST NOT alter its lifecycle or authority fields + +#### Scenario: Update arrives during a scan +- **WHEN** discovery records a newer remote timestamp after the active claim captured its scan timestamp +- **THEN** completion preserves the newer observation and permits one bounded follow-up scan after cooldown + +### Requirement: Updated-target admission is bounded +The system SHALL enforce a source-configured hard maximum of updated-target promotions per discovery cycle and a per-target cooldown. + +#### Scenario: Eligible changes exceed the cycle budget +- **WHEN** more completed changed targets are eligible than the configured maximum +- **THEN** at most the configured maximum are promoted and the remainder stay eligible for later cycles + +#### Scenario: Cooldown has not elapsed +- **WHEN** a changed completed target was scanned within the configured cooldown +- **THEN** it remains completed until a later eligible cycle + +#### Scenario: Concurrent discovery cycles +- **WHEN** multiple workers observe the same changed target concurrently +- **THEN** transactional row fencing permits at most one promotion for the covered revision + +### Requirement: Updated discovery remains capable of seeing changed known targets +The system SHALL NOT use identity-only known-page stopping for a source while updated-target rescans are enabled and SHALL keep discovery bounded by explicit page and result limits. + +#### Scenario: Known identities appear on an early page +- **WHEN** an early discovery page contains only known target identities +- **THEN** discovery continues within its configured hard page limit so later changed targets can be observed + +#### Scenario: HuggingFace Space was created long ago and recently updated +- **WHEN** a known Space has a recent `lastModified` value +- **THEN** update-sorted HuggingFace discovery can observe it independently of creation time + +### Requirement: Claims record the covered remote revision +The system SHALL atomically snapshot the newest observed remote timestamp when a target is claimed. + +#### Scenario: Revision-aware target is claimed +- **WHEN** a pending target receives a valid lease and reservation +- **THEN** its scan-covered timestamp equals the newest remote timestamp observed before that claim + +#### Scenario: Claim fails before authority is committed +- **WHEN** capacity or fencing prevents the claim +- **THEN** the scan-covered timestamp MUST NOT advance + +### Requirement: Updated-target activity is separately observable +The system SHALL report updated-target promotions separately from newly discovered target admissions in durable cycle metrics, source logs, and dashboard summaries. + +#### Scenario: Cycle admits new and updated targets +- **WHEN** a discovery cycle inserts new identities and promotes changed completed identities +- **THEN** `queued_new_count` and `queued_updated_count` record the respective counts without overlap + +#### Scenario: No changed targets are promoted +- **WHEN** a cycle observes only unchanged or ineligible known targets +- **THEN** `queued_updated_count` is zero diff --git a/openspec/changes/rescan-updated-core-targets/tasks.md b/openspec/changes/rescan-updated-core-targets/tasks.md new file mode 100644 index 0000000..1c02e5f --- /dev/null +++ b/openspec/changes/rescan-updated-core-targets/tasks.md @@ -0,0 +1,24 @@ +## 1. Queue State And Atomic Admission + +- [x] 1.1 Add idempotent PostgreSQL schema support for observed and scan-covered remote timestamps, updated-target cycle counts, and eligibility indexing. +- [x] 1.2 Implement atomic discovery observation and bounded promotion for changed completed targets while preserving all non-eligible lifecycle authority. +- [x] 1.3 Snapshot the observed remote timestamp only when a legacy or slot-first target claim commits. + +## 2. Discovery And Runner Integration + +- [x] 2.1 Preserve GitHub `pushed_at`, GitLab `last_activity_at`, and HuggingFace `lastModified` through target preparation. +- [x] 2.2 Make HuggingFace discovery newest-modified and disable identity-only page stopping when updated rescans are enabled. +- [x] 2.3 Wire source-specific enablement, per-cycle caps, and cooldowns without changing DockerHub or enabling refresh-while-backlogged discovery. + +## 3. Observability + +- [x] 3.1 Persist and log `queued_updated_count` separately from new-target admission. +- [x] 3.2 Add updated-target admission to dashboard source-cycle summaries. + +## 4. Verification And Rollout + +- [x] 4.1 Add focused discovery and runner tests for timestamp preservation, invalid timestamps, known-page behavior, and DockerHub isolation. +- [x] 4.2 Add PostgreSQL integration tests for legacy baselines, changed and unchanged targets, cooldown and cap enforcement, concurrent fencing, claim snapshots, and mid-scan updates. +- [x] 4.3 Run focused and full regression suites with bytecode writes disabled and run strict OpenSpec validation. +- [x] 4.4 Restart the authority-managed runtime and verify PostgreSQL, pipeline, source, queue, and recorder health. +- [x] 4.5 Observe a one-per-cycle production canary and compare changed-target admission, completion, findings, and worker occupancy before increasing any cap. diff --git a/openspec/changes/resolve-ambiguous-key-providers/.openspec.yaml b/openspec/changes/resolve-ambiguous-key-providers/.openspec.yaml new file mode 100644 index 0000000..f161d5c --- /dev/null +++ b/openspec/changes/resolve-ambiguous-key-providers/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-08-16 diff --git a/openspec/changes/resolve-ambiguous-key-providers/design.md b/openspec/changes/resolve-ambiguous-key-providers/design.md new file mode 100644 index 0000000..7b0239f --- /dev/null +++ b/openspec/changes/resolve-ambiguous-key-providers/design.md @@ -0,0 +1,78 @@ +## Context + +The scanner already persists exact and ambiguous routing hints for overlapping Qwen, DeepSeek, and Kimi `sk-...` findings. Provider workers, however, claim service-specific candidates and reject any hint that is not exactly their own service, so an ambiguous candidate can be repeatedly left unconsumed and quarantined. A ZAI detector already exists, but candidate extraction and a ZAI keychecker do not. + +The authoritative runtime uses PostgreSQL candidate leases and transactional result completion. Compatibility JSONL and status files are projections and must not control routing. + +## Goals / Non-Goals + +**Goals:** + +- Resolve one ambiguous credential sequentially across only its compatible providers. +- Stop at the first response that proves the credential belongs to a provider. +- Persist the successful result under the provider that recognized the credential. +- Add direct ZAI extraction plus model-list authentication and a minimal generation/billing probe through the global and China APIs. +- Preserve fenced candidate completion, capacity accounting, and restart safety. + +**Non-Goals:** + +- Redesign credential storage or secret-retention policy. +- Probe every supported provider for every unknown string. +- Run provider probes for one ambiguous credential in parallel. +- Automatically retry credentials whose current status is configured as terminal. + +## Decisions + +### Use one virtual resolver candidate + +Findings with a single strong provider hint continue to produce that provider's existing candidate. Findings with an ambiguous generic-key hint produce one `provider_resolver` candidate, deduplicated by credential within a staged scan bundle. The resolver owns one normal PostgreSQL lease and invokes compatible provider adapters in order, avoiding sibling candidates and cross-worker races. + +This is preferred to enqueueing one active candidate per provider because the latter requires new coordination state, can spend quota concurrently, and complicates exact queue-capacity release. + +### Keep route selection bounded and deterministic + +Persisted provider evidence limits the compatible set. The default fallback order is `deepseek,zai,qwen,kimi`; a provider identified by the originating detector is moved to the front when it belongs to the compatible set. The order is configurable, deduplicated, and never expanded beyond the supported generic-key provider set. + +Provider-specific formats such as `sk-sp-...`, `zai-...`, and ZAI's dotted key form remain direct routes when the finding evidence is unambiguous. + +### Normalize adapter outcomes + +Each adapter returns its existing detailed status plus a resolver outcome: + +- `match`: a successful authenticated response or provider-specific account/quota response proves ownership. +- `no_match`: the provider definitively rejects the credential as invalid. +- `retry`: network, server, generic rate-limit, malformed, or otherwise inconclusive responses. + +The resolver continues past `no_match` and may continue past `retry` to find a later positive match. If no provider matches, any retryable attempt keeps the result unresolved; only an all-`no_match` route is exhausted. + +### Reassign a matched candidate during fenced completion + +When a resolver result names a matched provider, `complete_keycheck_candidate` obtains or creates the canonical credential row for that provider, reassigns the leased candidate to it, and writes the result/current state under the matched service in the same transaction. The existing provider-key fingerprint, event fence, projection reservation, and capacity accounting remain unchanged. No schema migration is required. + +Legacy Qwen, DeepSeek, or Kimi candidates carrying an ambiguous persisted hint delegate to the same resolver so explicitly retried old candidates do not return to the unconsumed quarantine loop. + +### Authenticate ZAI through model listing and prove usability + +The ZAI adapter first calls authenticated `GET /models` on `https://api.z.ai/api/paas/v4` and `https://open.bigmodel.cn/api/paas/v4`, then sends a one-token `POST /chat/completions` probe to the fixed `glm-5.2` target. A key is `VALID` only when that probe succeeds. Model-list authentication still proves provider ownership when the probe reports quota, balance, permission, model access, or transient failures, but those outcomes are persisted outside the alive set. HTTP, documented ZAI business codes, and bounded message markers distinguish invalid authentication, recognized quota/balance restrictions, rate limits, permission restrictions, and transient failures. + +## Risks / Trade-offs + +- [A transient response from an early provider could hide a later match if probing stopped] -> Continue through the bounded compatible set while retaining the transient outcome if nobody matches. +- [The same text could theoretically be valid at more than one compatible gateway] -> Deterministic first-match ordering is explicit and recorded with all preceding attempts. +- [All-provider fallback increases requests for weak-context findings] -> Restrict it to detector-qualified generic-key formats and one sequential resolver candidate. +- [Existing quarantined candidates are not silently mutated] -> Make legacy candidates resolver-aware; operators can explicitly retry affected quarantine records through the existing review path. +- [Provider API behavior may change] -> Keep ZAI endpoints configurable and cover response classification with mocked regression tests. +- [The usability probe consumes provider resources] -> Request one output token from one deterministic chat model and stop after the first conclusive authenticated endpoint. + +## Migration Plan + +1. Deploy extraction, resolver, ZAI adapter, runner registration, and transactional service reassignment together. +2. Restart the supervised runtime so the lifecycle code manifest and service registry are rebuilt atomically. +3. Verify new ambiguous candidates are owned by `provider_resolver` and matched rows are projected under the actual provider. +4. Explicitly retry only relevant legacy provider-routing quarantine records after the new behavior is active. + +Rollback requires stopping the runtime and restoring the previous code/config manifest. No database schema rollback is needed. + +## Open Questions + +None. diff --git a/openspec/changes/resolve-ambiguous-key-providers/proposal.md b/openspec/changes/resolve-ambiguous-key-providers/proposal.md new file mode 100644 index 0000000..7fe2486 --- /dev/null +++ b/openspec/changes/resolve-ambiguous-key-providers/proposal.md @@ -0,0 +1,27 @@ +## Why + +Generic `sk-...` credentials can match several supported providers, while the current exact-hint routing leaves ambiguous findings unconsumed and can eventually quarantine them without testing a compatible provider. The keycheck pipeline needs ordered provider resolution and ZAI coverage so a credential is attributed to the first provider that positively recognizes it. + +## What Changes + +- Add durable, sequential resolution for credentials whose format or finding context permits multiple providers. +- Distinguish provider mismatch from authenticated match and retryable probe failure. +- Stop remaining provider attempts after the first positive match while retaining auditable attempt outcomes. +- Add a ZAI provider checker with a minimal generation/billing probe and include ZAI in compatible generic-key routing. +- Keep explicit single-provider findings on their existing direct validation path. + +## Capabilities + +### New Capabilities +- `ambiguous-provider-resolution`: Ordered, durable validation of one ambiguous credential across compatible providers until one positively matches or all definitive routes are exhausted. +- `zai-key-validation`: Extraction, probing, classification, persistence, and runtime registration for ZAI API credentials. + +### Modified Capabilities + +None. + +## Impact + +- Affects scanner provider hints, keycheck candidate extraction, PostgreSQL queue/schema operations, provider checker orchestration, status projection, runtime configuration, and lifecycle authority manifests. +- Adds a ZAI checker module using the existing HTTP and PostgreSQL keycheck infrastructure; model-list authentication alone does not qualify a key as alive. +- Requires regression coverage for route ordering, retry behavior, atomic match resolution, candidate deduplication, and ZAI response classification; the existing free-form service and credential tables require no schema migration. diff --git a/openspec/changes/resolve-ambiguous-key-providers/specs/ambiguous-provider-resolution/spec.md b/openspec/changes/resolve-ambiguous-key-providers/specs/ambiguous-provider-resolution/spec.md new file mode 100644 index 0000000..b8b668f --- /dev/null +++ b/openspec/changes/resolve-ambiguous-key-providers/specs/ambiguous-provider-resolution/spec.md @@ -0,0 +1,52 @@ +## ADDED Requirements + +### Requirement: Ambiguous credentials use one sequential resolver +The system SHALL represent one ambiguous credential occurrence as one leased resolver candidate and SHALL probe only the compatible provider set in deterministic order. + +#### Scenario: Weak generic-key context +- **WHEN** a detector-qualified generic `sk-...` finding has no single strong provider attribution +- **THEN** the system SHALL enqueue one resolver candidate rather than independently active candidates for every compatible provider + +#### Scenario: Strong provider attribution +- **WHEN** a finding has one strong provider-specific format or context signal +- **THEN** the system SHALL retain the direct provider route without invoking unrelated provider adapters + +### Requirement: Resolver outcomes control progression +The resolver SHALL distinguish positive match, definitive provider mismatch, and retryable uncertainty. + +#### Scenario: Provider rejects credential +- **WHEN** a provider definitively reports invalid authentication for an ambiguous credential +- **THEN** the resolver SHALL record that attempt and continue to the next compatible provider + +#### Scenario: Provider response is inconclusive +- **WHEN** a provider attempt fails because of a network error, server error, or otherwise inconclusive response +- **THEN** the resolver SHALL NOT classify that attempt as a definitive provider mismatch + +#### Scenario: All providers reject credential +- **WHEN** every compatible provider definitively rejects the credential +- **THEN** the resolver SHALL record an exhausted unresolved result after the final attempt + +### Requirement: First positive match terminates resolution +The resolver SHALL stop after the first response that proves the credential belongs to a provider. + +#### Scenario: Later provider recognizes credential +- **WHEN** earlier providers reject a credential and a later provider positively recognizes it +- **THEN** the resolver SHALL stop without calling subsequent providers and SHALL retain the ordered attempt evidence + +### Requirement: Matched result uses actual provider authority +A positively resolved candidate SHALL be completed transactionally under the provider that recognized it. + +#### Scenario: Resolver candidate matches another service +- **WHEN** a leased resolver candidate receives a positive ZAI result +- **THEN** the same fenced transaction SHALL associate the candidate and current state with the canonical ZAI credential and project the result as service `zai` + +#### Scenario: Completion loses its lease fence +- **WHEN** the candidate lease no longer matches during provider reassignment +- **THEN** no result, current-state update, or partial service reassignment SHALL be committed + +### Requirement: Legacy ambiguous candidates remain recoverable +Existing generic-provider candidates with persisted ambiguous routing evidence SHALL use the resolver when explicitly retried. + +#### Scenario: Retried legacy Qwen candidate +- **WHEN** an old Qwen candidate carries an ambiguous generic-provider hint and is retried +- **THEN** the Qwen worker SHALL delegate it to the shared resolver instead of leaving it unconsumed again diff --git a/openspec/changes/resolve-ambiguous-key-providers/specs/zai-key-validation/spec.md b/openspec/changes/resolve-ambiguous-key-providers/specs/zai-key-validation/spec.md new file mode 100644 index 0000000..2999e19 --- /dev/null +++ b/openspec/changes/resolve-ambiguous-key-providers/specs/zai-key-validation/spec.md @@ -0,0 +1,49 @@ +## ADDED Requirements + +### Requirement: ZAI findings produce keycheck candidates +The scanner SHALL create ZAI keycheck candidates for detector-qualified `zai-...`, compatible `sk-...`, and bounded dotted ZAI/Zhipu key forms. + +#### Scenario: Existing ZaiGLM detector finding +- **WHEN** the `ZaiGLM` custom detector emits a bounded API credential +- **THEN** candidate extraction SHALL preserve its finding attribution and route it to ZAI or the ambiguous resolver according to persisted provider evidence + +### Requirement: ZAI validation proves generation availability +The ZAI checker SHALL authenticate with a model-list request and SHALL require a bounded one-token `glm-5.2` generation probe before classifying a credential as valid and alive. + +#### Scenario: Global ZAI credential +- **WHEN** the global ZAI `/models` endpoint accepts the credential and the bounded generation probe succeeds +- **THEN** the checker SHALL record a valid authenticated ZAI result with bounded model and probe metadata + +#### Scenario: China Zhipu credential +- **WHEN** the global endpoint rejects a credential but the configured China endpoint accepts it and its bounded generation probe succeeds +- **THEN** the checker SHALL record the credential as ZAI with the successful endpoint region + +#### Scenario: Model listing succeeds but generation is unavailable +- **WHEN** `/models` authenticates the credential but the generation probe reports quota, balance, permission, transient, or inconclusive failure +- **THEN** the checker SHALL preserve the authenticated ZAI match but SHALL NOT classify the credential as valid or write it to the alive set + +#### Scenario: Model listing omits GLM 5.2 +- **WHEN** `/models` authenticates the credential but does not advertise `glm-5.2` +- **THEN** the checker SHALL still probe the fixed `glm-5.2` target and SHALL NOT substitute another model + +### Requirement: ZAI responses are classified by protocol evidence +The checker SHALL classify HTTP status and documented ZAI business error codes without treating inconclusive failures as invalid credentials. + +#### Scenario: Authentication rejected +- **WHEN** ZAI returns HTTP 401 or an authentication-failure business code +- **THEN** the attempt SHALL be classified as a definitive provider mismatch or dead direct credential + +#### Scenario: Authenticated balance or plan restriction +- **WHEN** ZAI returns a provider-specific balance, usage-plan, or permission response during model listing or the generation probe +- **THEN** the attempt SHALL be marked as belonging to ZAI with the corresponding limited, no-balance, or restricted status + +#### Scenario: Network or server failure +- **WHEN** the request fails in transit or ZAI returns a server error +- **THEN** the checker SHALL classify the attempt as retryable rather than dead + +### Requirement: ZAI participates in normal runtime accounting +The ZAI checker SHALL use the existing PostgreSQL lease, result, current-state, projection, and summary infrastructure. + +#### Scenario: Scheduled ZAI work exists +- **WHEN** the unified keycheck scheduler detects claimable ZAI candidates +- **THEN** it SHALL launch the ZAI checker with the same authority and bounded-slice controls used for other providers diff --git a/openspec/changes/resolve-ambiguous-key-providers/tasks.md b/openspec/changes/resolve-ambiguous-key-providers/tasks.md new file mode 100644 index 0000000..6bd90a1 --- /dev/null +++ b/openspec/changes/resolve-ambiguous-key-providers/tasks.md @@ -0,0 +1,27 @@ +## 1. Routing And Candidate Extraction + +- [x] 1.1 Extend provider evidence and persisted hints to include ZAI and weak-context generic-key ambiguity. +- [x] 1.2 Route ambiguous findings to one deduplicated `provider_resolver` candidate while preserving direct strong-provider candidates. +- [x] 1.3 Add bounded ZAI key formats and `ZaiGLM` service extraction. + +## 2. Resolver And ZAI Checker + +- [x] 2.1 Implement shared deterministic provider resolution with match, no-match, and retry outcomes. +- [x] 2.2 Add the PostgreSQL `provider_resolver` checker and legacy ambiguous-candidate delegation. +- [x] 2.3 Add the ZAI `/models` authentication checker with global/China endpoint and business-code classification. +- [x] 2.4 Require a bounded ZAI generation/billing probe before assigning `VALID`, while preserving authenticated non-alive outcomes. +- [x] 2.5 Pin the ZAI usability probe to `glm-5.2` without model-list fallback. + +## 3. Transactional Persistence And Runtime + +- [x] 3.1 Reassign a positively matched resolver candidate to the canonical provider credential during fenced completion. +- [x] 3.2 Register resolver and ZAI services, capabilities, status projections, configuration, and lifecycle runtime behavior. + +## 4. Verification + +- [x] 4.1 Add extraction, route ordering, short-circuit, retry, ZAI classification, and service-reassignment regression tests. +- [x] 4.2 Run targeted and existing keycheck/scanner test suites with bytecode writes disabled. +- [x] 4.3 Restart the supervised runtime and verify READY status plus live resolver/ZAI queue behavior. +- [x] 4.4 Add regression coverage for successful generation, no-balance, limited, restricted, and inconclusive ZAI probes. +- [x] 4.5 Run the affected suites and verify the supervised runtime plus one live ZAI recheck. +- [x] 4.6 Verify the fixed `glm-5.2` target with regression tests and one live recheck. diff --git a/openspec/changes/restore-openai-discovery-coverage/.openspec.yaml b/openspec/changes/restore-openai-discovery-coverage/.openspec.yaml new file mode 100644 index 0000000..701445b --- /dev/null +++ b/openspec/changes/restore-openai-discovery-coverage/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-08-26 diff --git a/openspec/changes/restore-openai-discovery-coverage/design.md b/openspec/changes/restore-openai-discovery-coverage/design.md new file mode 100644 index 0000000..ff532ab --- /dev/null +++ b/openspec/changes/restore-openai-discovery-coverage/design.md @@ -0,0 +1,68 @@ +## Context + +The OpenAI checker, candidate router, scheduler, and projection path are healthy, but the fresh input funnel collapsed after old target backlogs drained. In the last six days, core sources produced only 27 OpenAI findings and 11 new OpenAI credentials. A read-only exact-query probe found one unseen GitHub repository, 27 safely rescan-eligible GitLab projects, and 20 unseen repositories in the first 20 DockerHub results. + +The current source configuration accepts one plain string query per cycle and applies one source-wide page/target policy. Adding `openai` without a query-specific Docker bound could resolve and enqueue up to 200 repositories in one cycle. Runtime source state is authority-managed and must not be edited casually. + +## Goals / Non-Goals + +**Goals:** +- Restore explicit OpenAI-oriented discovery in the three query-driven core sources. +- Bound only the exact `openai` query while preserving every other query's current limits. +- Make the first post-deployment cycle deterministic without directly modifying runner state. +- Preserve queue, revision, digest, pipeline, and keycheck authority guarantees. +- Measure whether the added coverage yields new OpenAI credentials and usable checks. + +**Non-Goals:** +- Reclassifying provider API outcomes or weakening the successful-generation requirement. +- Re-enabling `package_git` or other broad inactive sources. +- Rechecking known credentials, changing HuggingFace discovery, or increasing global scan concurrency. +- Guaranteeing that newly discovered credentials are valid or funded. + +## Decisions + +### Add an exact provider query to existing rotations + +GitHub, GitLab, and DockerHub each receive one literal `openai` query. The term remains a normal persisted rotation entry, so completed cycles advance naturally and failures retain the query under existing semantics. + +For rollout, each entry is inserted at that source's current persisted query index while the runtime is coordinately stopped. The first restarted cycle therefore exercises `openai` without mutating state files; the next successful cycle advances to the query that previously occupied that index. + +Alternative: append the term and wait for a full rotation. Rejected because Docker cycles can be long and the canary would be delayed and difficult to attribute. + +### Apply a small allowlisted query override + +`build_args_from_source_config()` merges only `pages`, `per_page`, and `max_targets` from `query_overrides.`. Overrides are exact string matches, non-mutating, and validated before use. Unknown keys or non-mapping override shapes fail closed. + +Initial bounds: +- GitHub `openai`: one page, five scan claims. +- GitLab `openai`: one page, five scan claims; changed-target promotion remains capped at one. +- DockerHub `openai`: two ten-result pages, twenty scan claims. + +The Docker page limit bounds discovery admission to at most 20 repositories before tag resolution. `max_targets` separately bounds claims in that source cycle. Other queries continue using their source-wide page and target values. + +Alternative: temporarily reduce source-wide pages. Rejected because it would silently reduce all-provider coverage after rollback mistakes. Alternative: add a dedicated Docker source alias. Rejected because it would duplicate lifecycle and queue ownership code. + +### Preserve downstream authority and deduplication + +The change ends at target discovery arguments. Existing normalized identity deduplication, revision-aware completed-target promotion, Docker digest resolution, result-bundle fencing, credential deduplication, cached-known occurrence handling, and API keycheck rules remain authoritative. + +## Risks / Trade-offs + +- [Exact search still produces many known targets] -> Persist separate new/updated counts and evaluate the full candidate funnel, not fetched volume. +- [Docker results could create expensive scans] -> Bound search to 20 repositories and claims to 20 while retaining the global three-slot limit. +- [Query override could accidentally alter unrelated settings] -> Allow only three numeric discovery/claim keys and test that ordinary queries retain source defaults. +- [Config edit triggers immutable-authority shutdown] -> Use coordinated stop/start scripts and verify PostgreSQL, pipeline workers, and every core source after deployment. +- [No valid OpenAI keys appear] -> Treat zero usable outcomes as yield evidence, not as proof of pipeline failure, provided candidates complete with explicit API outcomes. + +## Migration Plan + +1. Add query-override parsing and focused tests. +2. Add exact queries and bounded overrides at each source's current persisted index. +3. Run focused and full regression suites plus strict OpenSpec validation. +4. Coordinately restart the authority-managed runtime. +5. Observe exactly the first `openai` source cycle for GitHub, GitLab, and DockerHub; verify limits, queue admission, scan completion, candidate checks, and pipeline drain. +6. Keep the query in normal rotation if safety bounds hold. Roll back by removing the query and overrides, then coordinately restart; no data migration is required. + +## Open Questions + +None. Further query weighting or expansion depends on measured canary yield. diff --git a/openspec/changes/restore-openai-discovery-coverage/proposal.md b/openspec/changes/restore-openai-discovery-coverage/proposal.md new file mode 100644 index 0000000..11ef96c --- /dev/null +++ b/openspec/changes/restore-openai-discovery-coverage/proposal.md @@ -0,0 +1,24 @@ +## Why + +OpenAI credential discovery fell from hundreds of new identities to almost none after historical Docker and package backlogs drained. The core GitHub, GitLab, and DockerHub discovery rotations do not contain the literal `openai` query, even though a read-only production probe showed that exact query exposes previously unseen supply. + +## What Changes + +- Add exact `openai` discovery to the GitHub, GitLab, and DockerHub core query rotations. +- Support narrowly allowlisted per-query bounds so the exact query can use smaller page and target limits without reducing coverage for every other query. +- Bound the first and recurring exact-query windows to one GitHub page, one GitLab page, and two DockerHub pages with source-appropriate scan limits. +- Preserve existing target deduplication, revision-aware rescan limits, queue authority, and Docker digest requirements. +- Measure the exact-query canary from discovery through scans, OpenAI candidates, API checks, and usable outcomes. + +## Capabilities + +### New Capabilities +- `openai-discovery-coverage`: Exact provider-term discovery with query-scoped bounds and observable production rollout. + +### Modified Capabilities + +None. + +## Impact + +The change affects `app/config.yaml`, query argument construction in `app/console_runner.py`, focused runner/config tests, source-cycle behavior for GitHub/GitLab/DockerHub, and production canary operations. It adds no dependency or schema migration and does not alter HuggingFace, detector routing, keycheck classification, or Docker target identity. diff --git a/openspec/changes/restore-openai-discovery-coverage/specs/openai-discovery-coverage/spec.md b/openspec/changes/restore-openai-discovery-coverage/specs/openai-discovery-coverage/spec.md new file mode 100644 index 0000000..d60b603 --- /dev/null +++ b/openspec/changes/restore-openai-discovery-coverage/specs/openai-discovery-coverage/spec.md @@ -0,0 +1,57 @@ +## ADDED Requirements + +### Requirement: Exact OpenAI core discovery +The system SHALL include the literal `openai` query in the normal GitHub, GitLab, and DockerHub core discovery rotations. + +#### Scenario: Exact provider term is rotated +- **WHEN** each supported source reaches its configured exact-query position +- **THEN** the source SHALL execute discovery using the literal `openai` term and persist normal cycle attribution + +#### Scenario: Successful exact-query cycle advances +- **WHEN** an exact-query source cycle completes successfully +- **THEN** the source SHALL advance to its next configured query through the existing persisted rotation + +#### Scenario: Failed exact-query cycle is retained +- **WHEN** exact-query discovery fails before a successful cycle completion +- **THEN** the source SHALL retain the same query according to existing failure semantics + +### Requirement: Query-scoped safety bounds +The system SHALL support exact-query overrides for only `pages`, `per_page`, and `max_targets`, without changing source-wide defaults for other queries. + +#### Scenario: Exact query receives bounded arguments +- **WHEN** a source builds arguments for `openai` +- **THEN** it SHALL apply that source's configured page, page-size, and target overrides + +#### Scenario: Ordinary query retains source defaults +- **WHEN** the same source builds arguments for any query without an override +- **THEN** it SHALL retain the source-wide page, page-size, and target values + +#### Scenario: Invalid override fails closed +- **WHEN** a query override is not a mapping or contains a key outside the allowlist +- **THEN** argument construction SHALL fail before discovery or queue mutation + +### Requirement: Source-specific rollout limits +The initial production policy SHALL constrain GitHub and GitLab exact discovery to one page each and DockerHub exact discovery to two pages of ten results, while preserving the existing global scan limit and changed-target promotion cap. + +#### Scenario: Docker exact discovery is bounded +- **WHEN** DockerHub executes the exact `openai` query +- **THEN** it SHALL request at most two pages of ten repositories and claim at most twenty targets in that cycle + +#### Scenario: Revision-aware source remains bounded +- **WHEN** GitLab exact discovery observes multiple changed completed projects +- **THEN** updated-target promotion SHALL remain capped by the existing one-per-cycle policy + +#### Scenario: Docker target authority is unchanged +- **WHEN** DockerHub exact discovery returns repository names +- **THEN** only targets satisfying the existing immutable digest requirement SHALL reach scanning + +### Requirement: End-to-end canary evidence +The rollout SHALL be evaluated from source discovery through durable scan completion, candidate completion, provider result, and projection drain without exposing credential or target values. + +#### Scenario: Safe canary completes +- **WHEN** the first exact-query cycles run after deployment +- **THEN** operators SHALL verify source limits, new and updated admissions, queue dispositions, pipeline completion, and runtime health using aggregate evidence + +#### Scenario: Useful yield is reported accurately +- **WHEN** exact-query scans create OpenAI candidates +- **THEN** operators SHALL report genuinely new credentials and their explicit API outcomes separately from cached-known occurrences and stale legacy file state diff --git a/openspec/changes/restore-openai-discovery-coverage/tasks.md b/openspec/changes/restore-openai-discovery-coverage/tasks.md new file mode 100644 index 0000000..ace8506 --- /dev/null +++ b/openspec/changes/restore-openai-discovery-coverage/tasks.md @@ -0,0 +1,16 @@ +## 1. Query-scoped controls + +- [x] 1.1 Add validated allowlisted per-query overrides for pages, page size, and maximum targets. +- [x] 1.2 Add focused tests proving exact-query overrides apply and ordinary queries remain unchanged. + +## 2. OpenAI discovery coverage + +- [x] 2.1 Add literal `openai` queries and source-specific safety bounds for GitHub, GitLab, and DockerHub. +- [x] 2.2 Test canonical configuration coverage, Docker isolation, and rollout limits. + +## 3. Verification and rollout + +- [x] 3.1 Run focused and full regression suites with bytecode writes disabled. +- [x] 3.2 Run strict OpenSpec validation and verify implementation against artifacts. +- [x] 3.3 Coordinately restart the authority-managed runtime and verify PostgreSQL, pipeline, and core sources. +- [x] 3.4 Observe the first bounded exact-query cycles and measure admissions, scan outcomes, OpenAI candidates, API outcomes, and pipeline drain. diff --git a/openspec/changes/retire-zero-alive-keywords/.openspec.yaml b/openspec/changes/retire-zero-alive-keywords/.openspec.yaml new file mode 100644 index 0000000..2e24cfa --- /dev/null +++ b/openspec/changes/retire-zero-alive-keywords/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-07 diff --git a/openspec/changes/retire-zero-alive-keywords/design.md b/openspec/changes/retire-zero-alive-keywords/design.md new file mode 100644 index 0000000..369fa06 --- /dev/null +++ b/openspec/changes/retire-zero-alive-keywords/design.md @@ -0,0 +1,81 @@ +## Context + +The earlier global pruning pass removed 38 terms with zero strict-usable yield but deliberately preserved their existing backlog. The later cold-policy change added an audited reversible hold and applied it only to the retired Docker cohort. The current canonical configuration still has 324 exact source/query pairs across 95 query texts. + +A fresh PostgreSQL dry-run attributed scans and candidates through immutable `target_scans.query` and `keycheck_candidates.query`, then linked candidates by `credential_id` to every historical `keycheck_results.status_group='alive'` result. After excluding dedicated source sentinels, 38 exact source/query pairs each have at least 1,000 successful scans, zero ever-alive credentials, and zero pending candidate checks. They account for 114,226 successful scans and 75,551 currently unfenced pending/deferred queue rows. + +## Goals / Non-Goals + +**Goals:** +- Retire source/query pairs only after substantial completed exposure and fully matured zero-alive evidence. +- Preserve a visible canonical rejection record with the evidence and decision reason. +- Stop future discovery and existing claimable work for the rejected pairs. +- Preserve every historical and queue authority record and make queue holds reversible. +- Apply the decision without racing workers or partially transitioning a reviewed cohort. + +**Non-Goals:** +- Delete queue rows, scans, findings, candidates, credentials, keycheck results, reservations, or coverage. +- Treat source sentinels such as `gharchive`, `gharchive-files`, `gists`, `logs`, or `spaces` as discovery keywords. +- Retire a query globally because it failed in one source. +- Automatically reevaluate or reactivate rejected pairs during normal runtime. +- Claim that a rejected pair can never become productive in the future. + +## Decisions + +### Evaluate exact source/query pairs + +The decision unit is the case-sensitive exact `(source, query)` pair. A keyword that produced an alive credential in DockerHub does not justify retaining a large npm backlog when the npm pair itself has substantial zero-alive evidence. + +Alternative: retire only globally zero-alive query text. Rejected because the dry-run would hold only 3,626 rows and preserve most demonstrated source-specific waste. + +### Require 1,000 successful scans and mature keychecks + +A pair qualifies only when it has at least 1,000 ended scans with status `clean`, `found`, or `degraded`, no credential linked through any of its candidate occurrences has ever had a historical `status_group='alive'` result, and no candidate for the pair remains pending or leased. Historical alive status is intentionally used instead of only current status so a once-working credential permanently proves yield. + +The 1,000-scan floor is deliberately stricter than a 100-scan cut. The latter selected 131 pairs and 187,917 rows but is too weak for rare useful credentials. Pair attribution uses immutable scan/candidate query snapshots for evidence; queue transition uses the row's current exact query because that is the durable admission attribution. + +Alternative: use raw finding count, unique candidates, current status only, or the dashboard strict-usable tier. Rejected because the operator requested actual `alive`, findings do not prove provider utility, current-only status forgets historical success, and pending checks make zero yield unresolved. + +### Preserve a canonical rejected registry + +Each removed pair remains under canonical `query_policy.rejected` with status `rejected_zero_alive`, evidence cutoff, successful scan count, finding count, unique credential count, pending count, ever-alive count, and reviewed queue count. Active query lists and rejected entries must be disjoint. The registry is evidence and operator visibility; existing append-only queue policy events remain the state-transition audit. + +Alternative: leave comments beside removed YAML entries. Rejected because comments are not machine-checkable and cannot fence future accidental reintroduction. + +### Use the existing cold lifecycle + +After configuration removal, the existing hash-fenced stopped-source manifest flow selects only unfenced `pending` or `deferred` rows whose exact query is absent from active policy. Each selected row becomes `cold`; no target value appears in review output, and no other row field or linked record changes. Active/fenced rows make apply fail closed. Reactivation remains possible only through the existing reviewed reverse action. + +### Approve the exact cohort + +- DockerHub: `OR`, `agent`. +- GitHub: `coding`, `memory`. +- npm: `OR`, `agent`, `agents`, `ai`, `assistant`, `benchmark`, `bot`, `chat`, `chats`, `completion`, `completions`, `conversation`, `gemini`, `groq`, `langchain`, `llm`, `mcp`, `open`, `openrouter`, `prompt`, `rag`, `semantic`, `studio`, `xai`. +- package-git: `agent`, `bot`, `completion`, `llm`, `open`, `semantic`, `studio`. +- Postman: `XAI_API_KEY`. +- PyPI: `langchain`, `open`. + +GitLab has no pair meeting the 1,000-scan rule. Dedicated source sentinel queries remain untouched. + +## Risks / Trade-offs + +- [Rare future yield is lost] -> Preserve evidence, history, and reversible cold events; reactivation requires explicit review. +- [Current queue query may differ from an older scan query after rediscovery] -> Use immutable scan/candidate attribution for yield evidence, exact current queue attribution for holding, and preserve append-only reversal evidence. +- [A candidate becomes alive between dry-run and apply] -> Regenerate and verify evidence immediately before apply; fail if the approved zero-alive cohort drifts. +- [A worker owns a selected row] -> Stop sources and fail the entire manifest on any queue, resolver, reservation, or blob fence. +- [Configuration accidentally reintroduces a rejected pair] -> Validate active/rejected disjointness in configuration tests and policy loading. + +## Migration Plan + +1. Add canonical rejected-query evidence and exact configuration tests for the approved 38-pair cohort. +2. Verify runtime sources are stopped and cleanly stop the current maintenance database before changing authority-covered configuration. +3. Remove each rejected pair from its active source list and retain it in `query_policy.rejected` with reviewed evidence. +4. Restart maintenance PostgreSQL, rerun the zero-alive evidence query, and require the exact cohort and queue counts to match the reviewed decision. +5. Generate bounded private cold manifests per affected source/platform and apply them through the existing cluster lock and stopped-source action. +6. Verify 75,551 rows became cold, no selected row remains claimable, history counts are unchanged, and policy events reconcile exactly. +7. Cleanly stop maintenance PostgreSQL, start the canonical runtime, and verify retained queries, pipeline workers, sources, and queue movement. +8. Roll back through the existing reviewed reactivation manifests plus restoration of the active query entries. + +## Open Questions + +None. diff --git a/openspec/changes/retire-zero-alive-keywords/proposal.md b/openspec/changes/retire-zero-alive-keywords/proposal.md new file mode 100644 index 0000000..3187934 --- /dev/null +++ b/openspec/changes/retire-zero-alive-keywords/proposal.md @@ -0,0 +1,26 @@ +## Why + +The remaining discovery rotations still contain source/query pairs that have each completed at least 1,000 successful scans without ever producing an `alive` credential. Their retained pending and deferred work consumes most of the avoidable backlog even though the existing audited `cold` lifecycle can preserve it reversibly. + +## What Changes + +- Retire only exact source/query pairs with at least 1,000 successful scans, zero historically `alive` linked credentials, and zero pending candidate checks at the evidence cutoff. +- Keep a durable rejected-query registry with the source, exact query, evidence counters, cutoff, and reason `rejected_zero_alive` while removing each pair from active rotation. +- Preserve dedicated source sentinel queries and evaluate the same query independently in different sources. +- Move every eligible unfenced pending or deferred target attributed to the rejected pairs into the existing audited, reversible `cold` state instead of deleting queue or history rows. +- Fail closed if attribution, evidence, policy identity, queue selection, or runtime quiescence changes between review and apply. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `discovery-keyword-pruning`: Add source/query-specific retirement based on substantial zero-`alive` evidence and retain explicit rejected-query evidence. +- `target-queue-policy-holds`: Apply the existing audited cold lifecycle to every reviewed source scope in the retired cohort. + +## Impact + +The change affects `app/config.yaml`, query-policy validation and tests, private reviewed cold manifests, and `target_queue` policy events. The approved cohort contains 38 source/query pairs and 75,551 currently eligible queue rows. No target, scan, finding, credential, result, reservation, deduplication, or coverage record is deleted. diff --git a/openspec/changes/retire-zero-alive-keywords/specs/discovery-keyword-pruning/spec.md b/openspec/changes/retire-zero-alive-keywords/specs/discovery-keyword-pruning/spec.md new file mode 100644 index 0000000..bb96baf --- /dev/null +++ b/openspec/changes/retire-zero-alive-keywords/specs/discovery-keyword-pruning/spec.md @@ -0,0 +1,56 @@ +## ADDED Requirements + +### Requirement: Source-specific zero-alive retirement rule +The system SHALL retire an exact source/query pair only when it has at least 1,000 successful completed scans, zero historically alive linked credentials, and zero pending candidate checks at the reviewed evidence cutoff. + +#### Scenario: Historical alive result preserves the pair +- **WHEN** any credential linked through a candidate occurrence for the exact source/query pair has ever produced `status_group='alive'` +- **THEN** that source/query pair SHALL remain active regardless of the credential's current status + +#### Scenario: Pending candidate preserves the pair +- **WHEN** an otherwise zero-alive source/query pair has a pending or leased candidate check +- **THEN** the pair SHALL remain active until the candidate evidence matures + +#### Scenario: Insufficient scan exposure preserves the pair +- **WHEN** a zero-alive mature pair has fewer than 1,000 successful completed scans +- **THEN** the pair SHALL remain active + +#### Scenario: Exact pair qualifies for retirement +- **WHEN** the exact pair has at least 1,000 successful completed scans, zero historical alive credentials, and zero pending or leased candidates +- **THEN** the pair SHALL qualify for reviewed retirement without affecting the same query in another source + +### Requirement: Rejected query evidence remains visible +Canonical configuration SHALL retain a machine-checkable rejection record for every retired source/query pair while keeping rejected pairs absent from active query rotation. + +#### Scenario: Rejected pair is loaded +- **WHEN** canonical query policy is validated +- **THEN** every rejected entry SHALL identify its exact source, query, status `rejected_zero_alive`, evidence cutoff, successful scans, findings, unique credentials, pending candidates, historical alive credentials, and reviewed queue count + +#### Scenario: Active and rejected policy overlaps +- **WHEN** the same exact source/query pair appears in both active rotation and rejected evidence +- **THEN** policy validation SHALL fail closed + +#### Scenario: Rejection evidence is incomplete +- **WHEN** a rejected entry omits or weakens the approved zero-alive evidence fields +- **THEN** policy validation SHALL fail closed + +## MODIFIED Requirements + +### Requirement: Operational and historical authority is preserved +Keyword retirement SHALL stop future discovery and SHALL permit existing unfenced pending/deferred targets attributed to retired exact source/query pairs to enter an audited, reversible cold state without deleting or rewriting historical authority. + +#### Scenario: Dedicated source sentinels remain +- **WHEN** archive, gist, CI-log, or HuggingFace source rotations are loaded +- **THEN** their operational sentinel queries SHALL remain configured and SHALL NOT be evaluated as interchangeable discovery keywords + +#### Scenario: Persisted rotation index remains valid +- **WHEN** an existing query index exceeds a shortened query list +- **THEN** normal modulo-based rotation SHALL select a valid configured query without a state-file edit + +#### Scenario: Existing backlog is preserved but held +- **WHEN** a previously admitted unfenced target is attributed to a retired exact source/query pair and selected by reviewed policy +- **THEN** its queue row SHALL remain present with all attribution, retry, deduplication, scan, reservation, and coverage history preserved while its status becomes unclaimable `cold` + +#### Scenario: Historical records remain unchanged +- **WHEN** a zero-alive policy cold transition is applied +- **THEN** existing target scans, findings, candidates, credentials, keycheck results, completed queue rows, and coverage records SHALL NOT be deleted or rewritten diff --git a/openspec/changes/retire-zero-alive-keywords/specs/target-queue-policy-holds/spec.md b/openspec/changes/retire-zero-alive-keywords/specs/target-queue-policy-holds/spec.md new file mode 100644 index 0000000..fa3b683 --- /dev/null +++ b/openspec/changes/retire-zero-alive-keywords/specs/target-queue-policy-holds/spec.md @@ -0,0 +1,37 @@ +## ADDED Requirements + +### Requirement: Reviewed zero-alive backlog is held across source scopes +The system SHALL apply the existing exact audited cold lifecycle to every approved source/platform/query scope in a reviewed zero-alive cohort. + +#### Scenario: Eligible rejected-query row is selected +- **WHEN** an unfenced `pending` or `deferred` row has an exact source/platform/query absent from active policy and present in approved rejected evidence +- **THEN** the reviewed manifest SHALL be permitted to transition the row to `cold` + +#### Scenario: Rejected cohort contains a fenced row +- **WHEN** any selected row has an active queue, resolver, reservation, or Docker content lease +- **THEN** apply SHALL fail closed without partially transitioning that manifest + +#### Scenario: Retired pair is rediscovered +- **WHEN** ordinary enqueue or rediscovery encounters a cold target previously attributed to a rejected pair +- **THEN** the target SHALL remain cold until an explicit reviewed reactivation + +## MODIFIED Requirements + +### Requirement: Stale-query selection follows canonical policy +The system SHALL evaluate query staleness using case-sensitive exact source, platform, and query policy derived from canonical active and rejected configuration. + +#### Scenario: Configured query remains active +- **WHEN** a queue row's exact source/platform/query triple remains configured in active policy +- **THEN** automatic stale-policy planning SHALL NOT select the row + +#### Scenario: Attribution cannot be classified safely +- **WHEN** query attribution is null, blank, operational, non-rotation, or belongs to an unknown source/platform pair +- **THEN** automatic planning SHALL skip and report the row rather than inferring retirement + +#### Scenario: Rejected source cohort is planned +- **WHEN** policy planning is scoped to an affected source/platform from the reviewed zero-alive cohort +- **THEN** it SHALL include only eligible pending/deferred rows attributed to exact rejected queries and SHALL expose no target values + +#### Scenario: Rejected evidence and active policy disagree +- **WHEN** rejected evidence does not match the canonical active-query omission or approved evidence identity +- **THEN** planning and apply SHALL fail closed diff --git a/openspec/changes/retire-zero-alive-keywords/tasks.md b/openspec/changes/retire-zero-alive-keywords/tasks.md new file mode 100644 index 0000000..cfdac3b --- /dev/null +++ b/openspec/changes/retire-zero-alive-keywords/tasks.md @@ -0,0 +1,22 @@ +## 1. Canonical Rejection Policy + +- [x] 1.1 Add strict machine-checkable rejected-query evidence validation with active/rejected disjointness and sentinel protection. +- [x] 1.2 Record the approved 38 source/query pairs and evidence in canonical configuration while removing them from active rotations. +- [x] 1.3 Make reviewed stale-row planning select only exact registered rejected pairs when a rejection registry is present. + +## 2. Verification + +- [x] 2.1 Add focused unit tests for evidence validation, exact source-specific retention, malformed evidence, overlap, and sentinel rejection. +- [x] 2.2 Extend PostgreSQL policy-hold tests for multi-source exact rejected selection, fences, idempotency, history preservation, and reactivation. +- [x] 2.3 Run focused configuration, migration, queue, integration, and strict OpenSpec validation. + +## 3. Reviewed Queue Transition + +- [x] 3.1 Stop maintenance PostgreSQL before editing authority-covered configuration, restart it afterward, and rerun the aggregate evidence cutoff. +- [x] 3.2 Generate and review bounded privacy-safe cold manifests for every affected source/platform scope. +- [x] 3.3 Apply the exact manifests with sources stopped and verify counts, append-only policy events, unclaimability, and unchanged history. + +## 4. Deployment + +- [x] 4.1 Cleanly stop maintenance PostgreSQL and start the canonical runtime with the reduced rotations. +- [x] 4.2 Verify authenticated database/pipeline/source health, rejected-pair absence, retained query loading, and normal queue movement. diff --git a/openspec/changes/run-bounded-docker-depth-experiment/.openspec.yaml b/openspec/changes/run-bounded-docker-depth-experiment/.openspec.yaml new file mode 100644 index 0000000..1ea7e36 --- /dev/null +++ b/openspec/changes/run-bounded-docker-depth-experiment/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-09 diff --git a/openspec/changes/run-bounded-docker-depth-experiment/design.md b/openspec/changes/run-bounded-docker-depth-experiment/design.md new file mode 100644 index 0000000..039f9aa --- /dev/null +++ b/openspec/changes/run-bounded-docker-depth-experiment/design.md @@ -0,0 +1,134 @@ +## Context + +DockerHub discovery now durably admits pages and rotates across 61 queries, but repository anchors are physically deduplicated by `(source, normalized_target)` and retain only the first inserting query. The resolver is FIFO, runs in batches of 100 only after scanable image work drains, and silently clamps image selection to three graphs. The current backlog therefore cannot provide fair keyword evidence or answer whether versions four through ten add useful findings. + +The runtime is PostgreSQL-final-cutover and safety-critical. Queue leases, resolver tokens, content reservations, quarantine holds, and existing cold-policy events must remain authoritative. Normal Docker image scans already traverse every layer; this change varies immutable image/version depth and records manifest positions without enabling broad layer-fallback mode. + +## Goals / Non-Goals + +**Goals:** + +- Run one durable experiment bounded to 1,200 unique immutable image targets and approximately 24-48 hours of observed throughput. +- Give every configured Docker query equal breadth before spending capacity on deeper versions. +- Preserve many-to-many discovery provenance while retaining one physical queue row and scan per normalized image target. +- Hold non-cohort repository anchors reversibly without bypassing active fences or unrelated hold policies. +- Keep ordinary and experiment image depths independently authoritative and validate them before runtime side effects. +- Persist enough immutable manifest and selection evidence to compare ranks 1-3 with ranks 4-10 and attribute findings to exact layers where evidence exists. + +**Non-Goals:** + +- Do not change Docker target normalization or merge repository aliases. +- Do not enable broad layer-only scanning or change the eight-layer recovery budget. +- Do not infer a layer for findings that lack an exact Docker layer digest. +- Do not resurrect failed, quarantined, already-completed, or independently cold targets automatically. +- Do not reconstruct historical many-to-many provenance; only fresh observations are experiment-eligible. +- Do not make experiment mutation APIs available on SQLite/file-queue execution. + +## Decisions + +### Fresh relational provenance + +Page admission will upsert `docker_repository_query_provenance` in the same transaction that admits repository anchors. Its key is `(source, query, repository_queue_id)` and it records first/last observation, search rank, cycle/policy evidence, and observation count. Existing queue attribution may be seeded as `legacy_queue`, but only a fresh complete observation under the experiment policy is eligible. + +This preserves physical deduplication while allowing one repository to belong to several keywords. A JSON column on `target_queue` was rejected because concurrent page admissions, ranking, indexing, and foreign-key validation require relational updates. + +### Durable experiment state machine + +The experiment uses PostgreSQL tables and compare-and-swap transitions: + +`collecting -> planned -> holding -> resolving -> active -> draining -> completed -> released` + +Any phase can enter `held` on configuration drift, incomplete query coverage, capacity conflict, or stale fencing. Rows carry a config hash, exact ordered-query hash, selector version/hash, counters, and timestamps. Every transaction that can change experiment provenance, repository membership, resolver state, target/binding/reservation state, queue state, or experiment-owned policy events first locks the experiment row. Cohort/target/queue locks follow that aggregate lock and global capacity is acquired last. Authority validation holds the same experiment lock while reading its multi-table snapshot, which provides a stable READ COMMITTED protocol without changing isolation for ordinary workers. + +### Fair bounded cohort + +Planning requires one complete fresh deep observation pass for all 61 configured queries. After that pass, each query contributes up to ten eligible previously unseen repositories; a query with fewer or zero new repositories remains in the authority with its exact desired and selected counts, and historical targets are never substituted. Repository selection is deterministic by discovery rank and stable queue identity. Scheduling proceeds round-robin across the available memberships by repository rank and query ordinal, never by global FIFO. + +Every selected breadth membership must terminate with either an eligible physical rank-one immutable target or exact hashed image-unavailability evidence before activation or completion. When fresh resolver evidence excludes an existing terminal, quarantined, independently cold, or fenced immutable candidate, the resolver records only bounded hashed skip evidence and compacts the next eligible graph/alias to the experiment rank. If no eligible graph remains, it deterministically advances to the next fresh ranked repository observation under the pinned generation/policy without reactivating its queue row. Once that fresh replacement pool is exhausted, only that membership becomes terminal `skipped` with `no_eligible_physical_target`; it consumes no target/capacity slot and remains separately visible as image-level scarcity. Missing, malformed, or mutated skip evidence moves the experiment to `held`. A query that had no eligible repository at planning time has no breadth membership to satisfy and remains explicitly reported as unavailable rather than missing work. + +For each query: + +- up to ten available fresh repositories contribute their newest distinct immutable image; +- when the query has at least one repository, one of them, chosen deterministically by the largest valid distinct layer-graph count with stable tie breaks, contributes image ranks 2-10. + +The theoretical maximum remains `61 * (10 + 9) = 1,159`; honest query scarcity lowers the actual planned maximum. A separate transactional ceiling of 1,200 unique immutable queue targets protects against planner defects and concurrency. Shared repositories/images consume physical capacity once while retaining all query selection rows. + +The dispatch order is breadth rank 1 for every query, then deep ranks 2-3 across every query, then deep ranks 4-10 across every query. These are completion barriers: no wave-two target can be reserved until every physical wave-one target has a latest terminal experiment binding, and wave three similarly waits for wave two. Refunding a terminal attempt returns its target to pending and reopens the earlier barrier. A target shared across selections retains its earliest wave and is scanned once. This makes partial experiment results interpretable even if runtime is stopped early. + +### Reversible holds + +The planner creates an explicit reviewed experiment hold manifest. Existing fenced cold APIs apply it only to unresolved Docker repository anchors with no active lease, resolver token, content reservation, quarantine, or unrelated hold. Every event stores the prior queue state and experiment identity. Release reverses only unreversed events belonging to this experiment. + +Newly observed non-cohort anchors are admitted directly into the same authorized experiment hold after planning. Far-future retry timestamps and destructive deletion were rejected because they obscure ownership and cannot be safely reversed. + +The reviewed hold/reactivation set is bounded at 250,000 rows. This is above the strict maximum configured search surface of `61 * 30 * 100 = 183,000` rows. Selection uses a `LIMIT 250001` fail-closed overflow check and application locks deterministic 500-row parts in one transaction, so PostgreSQL parameter limits cannot silently truncate the reviewed set. Experiment authority continuously validates every owned event's queue state, config/policy/manifest hashes, audit hash, and reversal state. + +Release is never automatic and a reviewed reactivation manifest can be generated or applied only from `completed`. Holding, resolving, active, draining, and held experiments cannot transition directly to `released`. + +### Dedicated resolver lane and atomic completion + +Experiment repositories use a dedicated round-robin resolver claim lane that does not wait for the ordinary image queue to empty and runs before ordinary backlog suppression. It bypasses only discovery and scan-queue-empty gating; PostgreSQL final-cutover, enabled state, API authentication, and shared rate-limit stops remain mandatory. Resolver completion atomically validates its owner/generation/token, persists immutable image targets, manifest graphs/layers, selection rank/reason, experiment target membership, and the unique-target counter. + +Resolver leases are renewed under that exact owner/generation/token before and between bounded tag/manifest network stages and verified again after remote work. A stale or expired token performs no write. Authority validation atomically returns interrupted resolver memberships to pending and records `held(stale_resolver_fence)`; only that reason may automatically return to `resolving` on a later claim, and only after a complete clean authority snapshot proves every resolver fence absent. Consumed non-conclusive remote attempts use a code-pinned 300-second exponential backoff capped at 3,600 seconds. On the third consumed attempt, only that membership terminates: breadth work becomes exact hashed `remote_unavailable_after_attempt_limit` scarcity, while a deep probe keeps its already selected images and closes at the achieved depth. Existing clean `held(resolver_attempt_limit)` rows from the earlier policy are terminalized by the same rule on the next fully validated claim. Evidence, capacity, configuration, and authority conflicts remain fail-closed experiment holds; a shared cooldown that made no remote request refunds the claim attempt. + +A resolver-attempt refund is never automatic. A one-time offline reviewed manifest may refund exactly two attempts only when the private immutable log snapshot contains exactly two target-bound instances of the retired local zero-graph limit defect. The manifest contains only queue/member IDs and hashes of target identity, prior error, entry evidence, log, and authority; it never contains raw targets. Application requires exact SHA approval, stopped sources, the unchanged attempt-limit hold, unfenced membership state, and an additive append-only audit row unique to that membership and recovery kind. Real partial/remote failures and every membership without exact old-bug evidence remain untouched. + +The offline reviewed disposition protocol remains available for a legacy persisted `resolver_attempt_limit` hold before runtime resumes. It snapshots the deterministic fresh replacement search as IDs, ranks, conflict codes, and target-identity hashes. Exact-SHA application either replaces the held membership with the first unchanged eligible fresh repository and resets only that membership's attempts, or terminally records scarcity. Normal runtime no longer serializes the whole pilot on target-local remote exhaustion. + +Only previously unseen pending immutable targets are experiment-eligible. Conflicts with `done`, `failed`, `quarantined`, or another cold policy cause deterministic fresh replacement and, after that pool is exhausted, an explicitly evidenced terminal membership skip. Historical work is never automatically reactivated. + +The reviewed cohort hash covers all 61 query authorities, each query's desired count of ten, its actual selected count from zero through ten, and every selected repository identity. Repository exclusions and replacements do not rewrite it. Once every selected breadth membership has either rank-one evidence or exact terminal skip evidence, a separate immutable runtime-selection hash freezes effective repository identities, terminal scarcity evidence, and one graph-informed deep-probe choice for each query with at least one image-bearing membership. An all-skipped query has no deep probe. Claims continuously validate both hashes, so terminal evidence and `is_deep_probe` cannot mutate after selection. + +### Config is the source of truth + +`docker_images_per_repository` remains the ordinary FIFO resolver depth and is fixed at the reviewed production value of three while the experiment is configured. `docker_depth_experiment.deep_images_per_repository` independently fixes the dedicated experiment lane at ten. The shared selector accepts an explicit validated limit through ten without a hidden clamp, but disabling experiment activation never raises ordinary FIFO work above three. Enabled experiment configuration rejects ordinary periodic repository refresh (`docker_repository_refresh_max_per_cycle` must be zero), preventing the ordinary resolver from mutating experiment repository authority. + +The experiment mapping enables provenance collection even while `enabled` is false. Both collection and activation therefore require managed PostgreSQL final-cutover, search mode, immutable digests, exact ordered queries, one representable effective pages/per-page pass policy, strict query overrides, and valid Docker platform/candidate settings. Validation runs before secrets, state, database, network, or worker initialization. The canonical semantic config hash includes ordered effective query policies, the code-pinned collection generation, and all platform selector inputs, but excludes operational `enabled`; false-to-true activation therefore preserves the reviewed frozen hash while enabled state remains a separately validated claim gate. Configuration/order/selector/generation hash drift moves an active experiment to `held` rather than silently changing its cohort, and disabling in `holding`, `resolving`, `active`, or `draining` moves it to `held`. + +The pinned collection generation is persisted on each discovery pass. While the experiment mapping remains operationally disabled for reviewed collection, Docker discovery continues but ordinary Docker resolver and scan admission remain paused so an unbounded historical backlog cannot pin query rotation or consume fresh cohort candidates. After reviewed activation, that collection-only barrier is removed and the frozen experiment lanes own resolution and scanning. Until PostgreSQL contains one complete deep pass for the pinned generation, policy, and ordered-query authority, every query is forced through deep provenance collection regardless of a pre-migration 72-hour runner-state marker. Terminal page evidence includes total count and must be coherent with the page's absolute result bound; undercounts cannot complete a pass. + +### Deterministic image selection beyond three + +The existing first-three semantics remain: newest distinct graph, maximum marginal layer novelty, then oldest distinct graph. Ranks four through ten repeatedly choose maximum marginal layer novelty against already selected graphs, with temporal distance, recency, normalized target, and graph hash as stable tie breakers. Invalid/duplicate graphs do not consume a rank. + +### Manifest and finding attribution + +Every selected image persists its immutable manifest identity and ordered descriptors. Layer position 1 is from the base; `layer_count - position + 1` is from the top. Duplicate layer digests at different positions remain separate rows. + +During result ingestion, an exact finding `Data.Docker.layer` digest is joined to all matching positions in that selected image and written to a finding-layer relation. Findings without exact evidence remain explicitly unattributed. Existing finding identity metadata is not rewritten. + +Exact attribution is fail-closed at a code-pinned limit of 100,000 projected finding-position rows per experiment result bundle. Ingestion counts every matching duplicate position before inserting scan evidence; overflow quarantines the whole bundle without truncation, inference, or a partial database commit. Reports aggregate position fan-out in SQL and read all report relations from one read-only repeatable snapshot. + +### Reservation-bound reporting + +Experiment scan bindings are created atomically with queue reservations, so historical scans of the same target cannot leak into the pilot. Reports join experiment query, repository, selection, unique target, reservation, target scan, findings, layer positions, and frozen/current keycheck outcomes. + +Capacity or quarantine saturation changes the experiment to `held` in the same experiment-row-locked transaction that aborts admission. There is no committed interval in which an aborted experiment admission remains active. Result readiness, renewal, refund, quarantine, and ingestion acquire the same experiment aggregate lock before reservation, queue, binding, or capacity transitions. + +Reports expose aggregates only. Per-query attribution credits shared physical scans to every observing query; global totals deduplicate experiment target and scan IDs. Marginal identity yield is assigned to the minimum image rank, and rank buckets are 1-3 versus 4-10. Unattributed layer findings and incomplete keychecks remain visible rather than being dropped. + +## Risks / Trade-offs + +- [A query has fewer than ten fresh repositories after the complete pinned deep pass] -> Freeze its exact available count, substitute no historical targets, and report the shortage against the desired count of ten. +- [The resolver cannot find ten distinct graphs for a deep probe] -> Record actual deep depth and continue with fewer depth selections; each breadth membership still requires rank one or exact terminal image-unavailability evidence. +- [Shared layers/findings multiply through many-to-many joins] -> Preaggregate physical scan/finding identities before query attribution and report both physical and attributed totals. +- [Native scanner output lacks a layer digest] -> Count it as unattributed and do not infer a position. +- [Concurrent resolver completions exceed 1,200] -> Serialize counter updates under the locked experiment row and recheck before insert. +- [Existing queue work competes with the pilot] -> Apply reviewed reversible holds and use experiment-only claim ordering while leaving unrelated active fences untouched. +- [Runtime/config changes invalidate evidence] -> Hash exact ordered queries, selector code version, and experiment config; transition fail-closed to `held` on drift. +- [Migration fails or rollout must be reverted] -> Use additive schema only, keep activation disabled during migration, and roll back configuration without deleting provenance or experiment evidence. + +## Migration Plan + +1. Canonically stop the supervised runtime and apply the additive migration with experiment activation disabled. +2. Start the runtime to collect fresh many-to-many provenance through one complete deep observation pass for all 61 queries. +3. Canonically stop, verify quiescence, generate and review the deterministic cohort/hold manifest, and activate the frozen experiment definition. +4. Start runtime, resolve and dispatch the cohort in round-robin waves, and monitor the 1,200 ceiling, leases, restart counters, and quarantine. +5. Transition to draining when all selections are terminal, then generate the immutable aggregate report. +6. Keep non-cohort rows cold until a reviewed post-experiment decision; release by exact experiment event identity when approved. + +Rollback disables new experiment claims and moves the experiment to `held`. Existing reservations finish through normal fencing. Additive provenance, manifest, and audit evidence remains intact for diagnosis and later resumption. + +## Open Questions + +None. The experiment limits, fairness rule, fresh-only eligibility, hold behavior, attribution model, and fail-closed rollout are fixed by the reviewed plan. diff --git a/openspec/changes/run-bounded-docker-depth-experiment/proposal.md b/openspec/changes/run-bounded-docker-depth-experiment/proposal.md new file mode 100644 index 0000000..853b4f2 --- /dev/null +++ b/openspec/changes/run-bounded-docker-depth-experiment/proposal.md @@ -0,0 +1,27 @@ +## Why + +DockerHub discovery admitted a large deduplicated repository backlog, but FIFO resolution and the hidden three-image selector cap cannot produce a fair, bounded comparison across all configured keywords. A controlled 24-48 hour experiment is needed to measure keyword yield and the marginal value of image versions beyond the current first three without allowing the backlog to monopolize runtime capacity. + +## What Changes + +- Add durable many-to-many DockerHub keyword-to-repository provenance so deduplicated targets retain every discovery attribution. +- Add a reversible experiment hold and a round-robin cohort planner that takes up to ten genuinely fresh repositories per configured keyword after one complete pinned deep pass, preserving honest shortfalls without substituting historical targets. +- Scan one current image from each cohort repository and up to ten distinct image graphs from one version-rich deep probe per keyword. +- Enforce a global experiment ceiling of 1,200 unique immutable images and keep non-cohort/new repositories cold until reviewed reactivation. +- Replace the hidden three-image clamp with explicit fail-closed configuration validation and deterministic selection beyond the third image. +- Persist image rank, selection reason, manifest layer graph, layer digest, and base/top layer positions for per-keyword experiment reporting. +- Report first-three versus later-version yield using deduplicated findings, credential identities, verification outcomes, scan time, and coverage. + +## Capabilities + +### New Capabilities +- `docker-depth-experiment`: Durable provenance, bounded fair cohort scheduling, configurable image depth, reversible holds, layer attribution, and experiment reporting. + +### Modified Capabilities + +## Impact + +- PostgreSQL schema and managed migrations for Docker discovery provenance, experiment cohorts, image selection metadata, and durable reporting state. +- DockerHub page admission, repository resolver scheduling, immutable target creation, and finding attribution in `app/scanner_db.py`, `app/scanner.py`, and `app/console_runner.py`. +- DockerHub configuration and validation in `app/config.yaml` and runner argument construction. +- Focused unit/PostgreSQL integration tests plus canonical offline migration and supervised runtime rollout. diff --git a/openspec/changes/run-bounded-docker-depth-experiment/specs/docker-depth-experiment/spec.md b/openspec/changes/run-bounded-docker-depth-experiment/specs/docker-depth-experiment/spec.md new file mode 100644 index 0000000..dd8cbea --- /dev/null +++ b/openspec/changes/run-bounded-docker-depth-experiment/specs/docker-depth-experiment/spec.md @@ -0,0 +1,252 @@ +## ADDED Requirements + +### Requirement: Authoritative Docker image depth configuration +The system SHALL obtain separate ordinary and experiment Docker images-per-repository limits from validated configuration. While this experiment is configured, the ordinary FIFO resolver limit SHALL remain three and the dedicated experiment deep limit SHALL remain ten. The system SHALL reject invalid or incompatible collection configuration before secrets, state, database, network, or worker initialization. + +#### Scenario: Disabled activation collection rollout +- **WHEN** experiment activation is disabled while provenance collection remains configured +- **THEN** Docker discovery SHALL persist fresh provenance while ordinary Docker resolver and scan admission remain paused, and the configured ordinary resolver depth SHALL remain three for later non-collection operation + +#### Scenario: Reviewed false-to-true activation +- **WHEN** a disabled collection is reviewed and operational `enabled` changes from false to true without another configuration change +- **THEN** the frozen semantic configuration hash SHALL remain unchanged and enabled state SHALL be enforced separately at each activation or claim boundary + +#### Scenario: Experiment depth ten +- **WHEN** the dedicated experiment resolver receives the validated deep limit of 10 +- **THEN** it SHALL be allowed to select up to ten deterministic distinct image graphs without a hidden lower clamp + +#### Scenario: Invalid depth +- **WHEN** Docker image depth is boolean, non-integer, below 1, above 10, or incompatible with the configured candidate-tag depth +- **THEN** startup SHALL fail before any runtime side effect + +#### Scenario: Mixed discovery policies +- **WHEN** query overrides produce different effective pages or per-page policy values for experiment queries +- **THEN** startup SHALL fail because the current experiment pass authority represents one discovery policy + +#### Scenario: Incompatible ordinary refresh +- **WHEN** experiment activation is enabled with periodic ordinary repository refresh greater than zero +- **THEN** startup SHALL fail before runtime side effects + +### Requirement: Durable many-to-many discovery provenance +The system SHALL record every fresh Docker query-to-repository observation in the same transaction as page admission while preserving one physical repository queue identity. + +#### Scenario: Repository observed by two queries +- **WHEN** two Docker queries observe the same normalized repository anchor +- **THEN** the system SHALL retain two provenance relations and one repository queue row + +#### Scenario: Page admission rolls back +- **WHEN** provenance persistence or repository admission fails +- **THEN** neither the page admission nor its provenance and retry progress SHALL commit partially + +#### Scenario: Page admission races authority validation +- **WHEN** page ingestion and experiment validation execute concurrently at READ COMMITTED +- **THEN** both SHALL serialize on the experiment authority row and validation SHALL NOT observe a half-committed page, hold event, queue, binding, or reservation transition + +#### Scenario: Pre-migration deep marker +- **WHEN** runner state contains a deep-dispatch marker but PostgreSQL has no complete deep pass for the pinned collection generation, policy, and ordered queries +- **THEN** each configured query SHALL be forced through one generation-current deep provenance pass before the normal 72-hour policy resumes + +#### Scenario: Underreported result count +- **WHEN** a Docker discovery page reports a total count below its current absolute result bound or incoherent with an empty continuation +- **THEN** that page SHALL fail or be delegated and SHALL NOT provide terminal pass evidence + +### Requirement: Fresh complete cohort eligibility +The experiment SHALL use only fresh observations made under its pinned policy and SHALL remain in collecting state until one complete deep observation pass covers every configured query. It SHALL then preserve every query authority row and select up to ten eligible previously unscanned repositories per query, including an explicit actual count of zero when the completed pass produced none. + +#### Scenario: Query has insufficient candidates +- **WHEN** the complete pinned deep pass leaves a configured query with fewer than ten fresh eligible repositories +- **THEN** the cohort SHALL retain exactly the available fresh repositories, SHALL NOT substitute historical or previously scanned targets, and SHALL report the unavailable count against the desired quota of ten + +#### Scenario: Collection pass is incomplete +- **WHEN** no complete pinned deep pass covers all 61 configured queries +- **THEN** planning SHALL remain unavailable even if partial observations exist + +#### Scenario: Legacy attribution exists +- **WHEN** a repository has only historical first-inserter queue attribution +- **THEN** that evidence SHALL NOT satisfy fresh experiment eligibility + +### Requirement: Bounded fair experiment cohort +The system SHALL select up to ten fresh repositories per configured query after the complete pass, resolve one newest distinct image from each selected repository when an eligible unseen image exists, and select image ranks 2 through 10 from one deterministic version-rich image-bearing repository for each applicable query, subject to a transactional ceiling of 1,200 unique immutable image targets. A selected repository that exhausts fresh replacements without an eligible image SHALL remain explicit terminal image-level scarcity rather than causing historical substitution. + +#### Scenario: Complete 61-query authority +- **WHEN** all 61 query rows are planned from a complete pinned deep pass +- **THEN** the planned selections SHALL be no more than 1,159 before cross-query deduplication, honest scarcity SHALL reduce rather than inflate that count, and physical experiment targets SHALL never exceed 1,200 + +#### Scenario: Conclusive breadth zero +- **WHEN** a breadth repository resolves conclusively with no eligible immutable image +- **THEN** the resolver SHALL deterministically select the next fresh ranked repository and, when that pool is exhausted, SHALL terminally mark only that membership `skipped` with exact hashed `no_eligible_physical_target` evidence without consuming a physical target slot + +#### Scenario: Complete breadth authority +- **WHEN** the experiment activates or completes +- **THEN** every selected breadth membership SHALL have either a valid rank-one selection bound to an eligible physical experiment target or exact terminal image-unavailability evidence, while zero-member queries SHALL remain explicit nonmissing repository scarcity evidence + +#### Scenario: Query has no image-bearing membership +- **WHEN** every selected repository for a query terminates with valid image-unavailability evidence +- **THEN** that query SHALL have no deep probe and SHALL remain separately reportable without blocking activation + +#### Scenario: Concurrent shared image selection +- **WHEN** concurrent query selections resolve to the same normalized immutable image +- **THEN** the image SHALL consume one physical target slot and SHALL retain every query selection relation + +### Requirement: Round-robin experiment scheduling +The system SHALL schedule breadth and depth work by pinned query ordinal rather than repository FIFO. + +#### Scenario: Breadth precedes depth +- **WHEN** experiment targets become claimable +- **THEN** ranks 2-3 SHALL remain unclaimable until every physical rank-one target has a terminal latest experiment binding, and ranks 4-10 SHALL similarly wait for ranks 2-3 + +#### Scenario: Earlier wave is refunded +- **WHEN** a terminal attempt in an earlier wave is refunded and its physical target returns to pending +- **THEN** the earlier completion barrier SHALL reopen and later waves SHALL stop until its replacement attempt completes terminally + +#### Scenario: Partial execution +- **WHEN** runtime stops before the experiment completes +- **THEN** completed work SHALL remain evenly attributable to the earliest unfinished round-robin wave + +### Requirement: Deterministic distinct graph selection +The resolver SHALL preserve the existing first-three selection semantics and SHALL select ranks 4 through 10 deterministically by marginal layer novelty and stable temporal/identity tie breaks. + +#### Scenario: Duplicate tags share a graph +- **WHEN** multiple tags resolve to an identical ordered layer graph +- **THEN** the graph SHALL consume at most one image rank + +#### Scenario: Fewer than ten valid graphs +- **WHEN** a deep repository has fewer than ten valid distinct image graphs +- **THEN** the resolver SHALL record the actual depth without inserting invalid or duplicate replacements + +### Requirement: Reversible fenced backlog hold +The system SHALL place non-cohort Docker repository anchors into a durable experiment-scoped cold state only when no active lease, resolver token, reservation, quarantine, or unrelated hold prevents the transition, and SHALL preserve the exact prior state for reviewed reactivation. + +#### Scenario: Unfenced non-cohort anchor +- **WHEN** an eligible non-cohort unresolved repository is covered by the reviewed experiment hold policy +- **THEN** it SHALL become cold with a durable policy event and SHALL be ignored by ordinary resolver claims + +#### Scenario: Independently fenced anchor +- **WHEN** a repository has an active or unrelated safety fence +- **THEN** the experiment SHALL leave it unchanged and record the hold conflict + +#### Scenario: Reviewed release +- **WHEN** the experiment hold is released +- **THEN** the experiment SHALL already be completed and only unreversed cold events owned by that experiment SHALL restore their exact prior queue states + +#### Scenario: Unsafe early release +- **WHEN** a reviewed release is requested from holding, resolving, active, draining, or held +- **THEN** release SHALL be rejected without changing owned cold events or experiment state + +#### Scenario: Full reviewed search surface +- **WHEN** a reviewed hold or reactivation covers more than 100,000 rows up to the strict `61 * 30 * 100` search maximum +- **THEN** deterministic bounded parts SHALL cover every row, aggregate counts/hashes SHALL remain authoritative, and overflow beyond the reviewed 250,000-row ceiling SHALL fail rather than omit rows + +### Requirement: Fresh target safety +The experiment SHALL scan only previously unseen immutable image targets and SHALL NOT automatically reactivate completed, failed, quarantined, or independently cold targets. + +#### Scenario: Selected image already completed +- **WHEN** a resolver selection conflicts with an immutable target already in done state +- **THEN** the system SHALL record hashed skip evidence and deterministically continue to the next eligible fresh graph/alias without requeueing the completed target or retrying the identical selected set + +#### Scenario: Selected image is quarantined +- **WHEN** a resolver selection conflicts with quarantined work +- **THEN** the system SHALL skip it and continue to the next eligible fresh candidate or replacement repository and SHALL NOT bypass quarantine + +#### Scenario: No safe replacement remains +- **WHEN** every fresh candidate is terminal, independently cold, quarantined, fenced, or otherwise ineligible +- **THEN** only that repository membership SHALL become terminal `skipped` with exact hashed image-unavailability evidence, no experiment target or capacity slot SHALL be consumed, and no candidate SHALL be reactivated + +#### Scenario: Terminal scarcity evidence drifts +- **WHEN** terminal repository skip state lacks its exact reason, repository identity, ordinal, or canonical evidence hash +- **THEN** experiment authority validation SHALL move the experiment to held state before activation or further mutation + +### Requirement: Durable manifest and layer attribution +The system SHALL persist each selected immutable manifest and its ordered layer descriptors, including positions from base and top, and SHALL associate findings only when an exact layer digest is present. + +#### Scenario: Exact layer digest finding +- **WHEN** a finding reports a Docker layer digest present at one or more manifest positions +- **THEN** the system SHALL persist every exact matching base/top position for that image + +#### Scenario: Finding has no layer digest +- **WHEN** scanner evidence does not identify an exact layer digest +- **THEN** the report SHALL count the finding as layer-unattributed and SHALL NOT infer a position + +### Requirement: Reservation-bound experiment evidence +The system SHALL bind experiment targets to scan reservations atomically and SHALL use those bindings to exclude historical or unrelated scans from experiment results. + +#### Scenario: Experiment target is reserved +- **WHEN** an experiment target receives scan capacity and a queue lease +- **THEN** its experiment binding, reservation, and queue transition SHALL commit atomically + +#### Scenario: Reservation is retried +- **WHEN** the same experiment target requires a fenced retry +- **THEN** reporting SHALL preserve each bound attempt while deduplicating final physical target totals + +#### Scenario: Admission capacity is saturated +- **WHEN** experiment admission cannot reserve pipeline or quarantine capacity +- **THEN** the aborted admission intent and experiment `held` transition SHALL commit atomically under the same experiment authority lock + +### Requirement: Safe experiment reporting +The system SHALL produce aggregate physical and per-query reports comparing image ranks 1-3 with ranks 4-10, including deduplicated findings, credential identities, keycheck outcomes, scan duration/errors, layer positions, overlap, coverage, and marginal minimum-rank yield without exposing secret material. + +#### Scenario: Image belongs to multiple queries +- **WHEN** one physical image selection is attributed to multiple queries +- **THEN** global totals SHALL count it once while each relevant query report SHALL receive attribution and overlap SHALL be explicit + +#### Scenario: Identity repeats at later rank +- **WHEN** a detector-secret or credential identity first appears at rank 2 and appears again at rank 7 +- **THEN** marginal yield SHALL assign that identity to rank 2 and SHALL NOT recount it as new in ranks 4-10 + +#### Scenario: Report contains sensitive evidence +- **WHEN** aggregate reporting reads findings or keycheck records +- **THEN** output SHALL contain only approved IDs, hashes, enums, counts, durations, and positions and SHALL NOT contain raw credentials or secret-bearing excerpts + +### Requirement: Fail-closed experiment authority +Experiment collection and activation SHALL run only on managed PostgreSQL final-cutover with search-mode immutable-digest discovery. The canonical semantic configuration hash SHALL exclude operational `enabled` and include the pinned collection generation, exact ordered effective query policies, and Docker platform filter, OS, architecture, and candidate-count values. Enabled state SHALL be validated independently. The original cohort plan hash SHALL remain immutable across deterministic exclusions/replacements, and a second frozen runtime-selection hash SHALL bind effective repositories, terminal image-scarcity evidence, and the executed deep-probe choice before activation. Experiment authority validation and every experiment-owned mutation SHALL use an experiment-row-first locked protocol. An active experiment SHALL transition to held state when ordered queries, selector version, configuration hash, runtime-selection hash, terminal skip evidence, owned hold-event/queue history, capacity, or fencing invariants drift. + +#### Scenario: Query list changes during execution +- **WHEN** the configured ordered query hash differs from the pinned experiment hash +- **THEN** new experiment claims SHALL stop and the experiment SHALL enter held state + +#### Scenario: File-queue fallback is active +- **WHEN** PostgreSQL final-cutover authority is unavailable +- **THEN** experiment activation and mutation SHALL be rejected + +#### Scenario: Executed deep probe drifts +- **WHEN** an executed `is_deep_probe` choice differs from the frozen runtime-selection hash +- **THEN** new claims SHALL stop and the experiment SHALL enter held state + +#### Scenario: Disabled during holding +- **WHEN** operational `enabled` becomes false while the experiment is in holding +- **THEN** the experiment SHALL transition fail-closed to held + +#### Scenario: Long remote graph resolution +- **WHEN** candidate manifest resolution spans the original 300-second lease +- **THEN** the worker SHALL renew before and between bounded remote stages under its exact owner/generation/token, verify the token after remote work, and perform no completion write after lease loss + +#### Scenario: Runtime stops during resolver work +- **WHEN** authority validation finds an expired resolver fence left by an interrupted runtime +- **THEN** it SHALL atomically return the affected membership to pending and enter `held(stale_resolver_fence)`, and a later claim MAY resume `resolving` only after full authority validation succeeds and no resolver fence remains; no other held reason SHALL automatically resume + +#### Scenario: Repeated non-conclusive remote failures +- **WHEN** a resolver membership consumes three non-conclusive remote attempts +- **THEN** only that membership SHALL terminate instead of retrying forever or pinning the breadth barrier +- **AND** breadth work SHALL record exact hashed `remote_unavailable_after_attempt_limit` scarcity without a target slot, while deep work SHALL preserve already selected images and close at its achieved depth + +#### Scenario: Systemic resolver conflict reaches its ceiling +- **WHEN** repeated evidence, capacity, configuration, or authority conflicts reach their safety ceiling +- **THEN** the experiment SHALL remain fail-closed in held state and SHALL NOT misclassify the conflict as target-local remote scarcity + +#### Scenario: Legacy attempt-limit hold resumes under the simplified policy +- **WHEN** a fully validated claim encounters exactly one unfenced membership persisted as `held(resolver_attempt_limit)` by the earlier policy +- **THEN** it SHALL terminalize only that membership with exact hashed remote-unavailable evidence, clear the legacy experiment hold, and continue resolving the remaining cohort + +#### Scenario: Reviewed disposition of a genuine attempt-limit hold +- **WHEN** an operator reviews an exact hash-only manifest for the single membership held by genuine remote failures and approves its SHA while sources are stopped +- **THEN** the system SHALL atomically use the first unchanged eligible fresh replacement repository and reset only that membership's attempts, or SHALL terminally record exact `remote_unavailable_after_attempt_limit` scarcity when the reviewed fresh replacement pool is exhausted +- **AND** it SHALL append a one-time immutable audit row, resume the experiment, preserve the global three-attempt policy, consume no target slot for a skip, and never expose or reactivate historical target identity + +#### Scenario: Reviewed refund of attempts consumed by a retired local defect +- **WHEN** stopped-source offline review proves exactly two target-bound occurrences of the retired zero-graph limit defect for a membership and an operator approves the exact private manifest SHA +- **THEN** the system SHALL refund exactly two attempts, append a one-time immutable audit row, clear only that membership's obsolete local error, and resume the attempt-limit-held experiment without exposing the target identity +- **AND** memberships containing only partial or other remote failures SHALL remain unchanged, and the same recovery kind SHALL never refund a membership twice + +#### Scenario: Owned hold history drifts +- **WHEN** an experiment-owned hold event, queue state, config/policy/manifest hash, audit hash, or reversal is changed outside the reviewed protocol +- **THEN** continuous authority validation SHALL stop mutation and hold the experiment before new rows or events commit diff --git a/openspec/changes/run-bounded-docker-depth-experiment/tasks.md b/openspec/changes/run-bounded-docker-depth-experiment/tasks.md new file mode 100644 index 0000000..6e4e345 --- /dev/null +++ b/openspec/changes/run-bounded-docker-depth-experiment/tasks.md @@ -0,0 +1,52 @@ +## 1. Schema And Migration + +- [x] 1.1 Add migration marker and SQLite/PostgreSQL schema for Docker query provenance, immutable manifests/layers, experiment authority, query/repository/selection/target/binding relations, and finding-layer attribution +- [x] 1.2 Register required columns, keys, foreign keys, indexes, integer-width conversions, constraints, migration quiescence checks, and idempotent validation +- [x] 1.3 Add legacy first-inserter provenance seeding that is explicitly ineligible for fresh experiment coverage + +## 2. Configuration And Selection + +- [x] 2.1 Add strict side-effect-free Docker image-depth and experiment configuration validation with exact ordered-query/config/selector hashes +- [x] 2.2 Remove the hidden three-image clamp and implement deterministic distinct graph selection through rank ten while preserving first-three semantics +- [x] 2.3 Add production configuration for the disabled collection rollout and the pinned 61-query, up-to-10-fresh-repository, 10-image, 1,200-target experiment + +## 3. Discovery Provenance + +- [x] 3.1 Persist main-pass query-to-repository provenance and stable search rank atomically with each Docker discovery page +- [x] 3.2 Persist retry-pass provenance with the original query/policy evidence under retry lease fencing +- [x] 3.3 Require one complete pinned deep pass, then freeze each of all 61 queries with its exact zero-to-ten eligible unseen repository count without historical substitution + +## 4. Cohort And Holds + +- [x] 4.1 Implement deterministic sparse round-robin cohort planning, one deep-probe choice per nonempty query, immutable desired/actual count hashes, and serialized unique-target capacity accounting +- [x] 4.2 Generate and validate an explicit experiment hold manifest for non-cohort unresolved Docker anchors +- [x] 4.3 Apply/release experiment-owned cold events through existing fence-aware APIs without changing independently fenced targets +- [x] 4.4 Hold newly observed non-cohort anchors under the activated reviewed experiment policy + +## 5. Resolver And Scheduling + +- [x] 5.1 Add a dedicated fenced experiment resolver lane ordered by repository round and query ordinal +- [x] 5.2 Persist selected immutable targets, ranks/reasons, manifest identity, and ordered layer descriptors atomically with resolver completion +- [x] 5.3 Reject or replace conflicts with done, failed, quarantined, or independently cold immutable targets without implicit reactivation +- [x] 5.4 Bind experiment targets atomically to scan reservations and dispatch breadth, ranks 2-3, then ranks 4-10 round-robin +- [x] 5.5 Fail closed to held state on query/config/selector drift, capacity conflict, stale token, or non-PostgreSQL authority + +## 6. Evidence And Reporting + +- [x] 6.1 Associate exact Docker finding layer digests with all matching base/top manifest positions during atomic result ingestion +- [x] 6.2 Add a secret-safe aggregate experiment report separating physical totals from per-query attribution and ranks 1-3 from ranks 4-10 +- [x] 6.3 Report deduplicated findings/credentials, frozen and current verification outcomes, marginal minimum-rank yield, scan duration/errors, overlap, coverage, and unattributed findings + +## 7. Verification + +- [x] 7.1 Add focused unit tests for strict configuration, side-effect ordering, rank 4-10 selection, deduplication, and deterministic tie breaks +- [x] 7.2 Add schema/migration and PostgreSQL integration tests for provenance atomicity, fair planning, global cap concurrency, hold/reactivation fencing, resolver completion, reservation binding, and ingestion +- [x] 7.3 Add reporting and secret-sentinel tests for shared attribution, rank boundaries, marginal identity yield, layer positions, retries, and unattributed evidence +- [x] 7.4 Run focused tests, PostgreSQL integration, broad relevant suites, strict OpenSpec validation, and independent review + +## 8. Controlled Rollout + +- [x] 8.1 Canonically stop runtime, apply the additive migration with experiment disabled, and restart to collect fresh provenance +- [x] 8.2 Verify one complete fresh observation pass for every query, then canonically stop and generate/review the cohort and hold manifest +- [x] 8.3 Activate the frozen experiment, restart canonically, and verify PostgreSQL/pipeline/auth/worker health, fair cohort dispatch, capacity ceiling, leases, retries, restart counters, and bytecode absence +- [x] 8.4 Leave the experiment running toward drain/report while keeping non-cohort rows reversibly cold for a later reviewed decision diff --git a/openspec/changes/run-bounded-rank1-breadth-experiment/.openspec.yaml b/openspec/changes/run-bounded-rank1-breadth-experiment/.openspec.yaml new file mode 100644 index 0000000..2b596d1 --- /dev/null +++ b/openspec/changes/run-bounded-rank1-breadth-experiment/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-12 diff --git a/openspec/changes/run-bounded-rank1-breadth-experiment/design.md b/openspec/changes/run-bounded-rank1-breadth-experiment/design.md new file mode 100644 index 0000000..cb2abe2 --- /dev/null +++ b/openspec/changes/run-bounded-rank1-breadth-experiment/design.md @@ -0,0 +1,134 @@ +## Context + +The bounded Docker-depth experiment showed that all five currently alive +credentials were already present in the newest selected image. Image ranks +2-10 consumed about 7.5 scanner-hours, added twelve globally new historical +credentials, and added no currently alive credential. The next experiment +therefore spends a larger bounded budget on repository breadth while scanning +only the newest eligible immutable image from each selected repository. + +The complete frozen 61-query discovery pass contains enough fresh provenance +for this cohort. Most non-cohort repository anchors are currently held by the +depth experiment and can become eligible only after that experiment completes +and its exact reviewed cold events are reversed. PostgreSQL remains the sole +authority for cohort, leases, reservations, policy events, capacity, and +reporting evidence. + +## Goals / Non-Goals + +**Goals:** + +- Select exactly 2,000 previously unscanned physical repositories from the + existing complete frozen discovery pass without using yield. +- Balance selection across the 52 keywords that have an eligible remaining + pool: 38 repositories each, then one additional repository for the first 24 + still-eligible keywords in pinned query order. +- Physically deduplicate repositories across keywords while retaining their + frozen many-to-many keyword provenance. +- Resolve and scan at most one newest eligible immutable image per selected + repository under the existing experiment authority and fencing protocol. +- Hand authority over only after the depth experiment is completed and its + reviewed release has restored every owned non-cohort row. +- Report globally deduplicated credential and currently-alive yield, scanner + cost, and secret-safe per-keyword attribution. + +**Non-Goals:** + +- Do not scan older image ranks or reinterpret `rank1` as a filesystem layer. +- Do not infer an adaptive-depth trigger from the zero depth-only alive sample. +- Do not reactivate scanned, failed, quarantined, independently cold, fenced, + or incompletely released repository rows. +- Do not run the depth and breadth experiments concurrently. +- Do not expose repositories, image targets, credentials, hashes, or raw + scanner evidence in operator output. + +## Decisions + +### Versioned experiment profiles + +The existing Docker experiment tables and worker protocol are reused. The +reviewed selector version identifies one of two exact code-pinned profiles: +the existing depth profile remains unchanged, while the breadth profile fixes +61 queries, a per-query ceiling of 39, shallow/deep image depth of one, and a +global physical target limit of 2,000. Configuration validation accepts only +one complete reviewed profile; arbitrary mixtures remain invalid. + +The breadth profile's theoretical capacity is the global physical limit, not +`query_count * repositories_per_query`, because cross-keyword physical +deduplication and the global stop are part of planning. Schema bounds are +widened only enough to store the reviewed profile. Runtime authority still +compares every persisted value with the exact supplied reviewed profile. + +### Deterministic balanced physical selection + +Planning reads the existing complete frozen deep discovery pass and orders each +query's eligible repository observations by search rank and stable queue ID. +It walks pinned queries round-robin, taking the next candidate whose physical +queue ID has not already been selected, until each query owns at most 39 +repositories or the global count reaches exactly 2,000. Duplicate observations +are skipped within the selecting query rather than consuming quota. + +Given the reviewed frozen pool this produces 38 repositories for each of 52 +nonempty keywords and a 39th for the first 24 of those keywords in pinned order; +the nine empty keywords remain explicit zero rows. Manifest generation fails +closed unless it reaches exactly 2,000 unique physical repositories with this +distribution. All fresh query observations remain in relational provenance and +report attribution joins the selected physical repository back to that frozen +evidence rather than duplicating scan work. + +### Exact prior-release eligibility + +A repository with no policy-event history remains eligible under the existing +rules. A repository with history is eligible only when every event belongs to +a completed/released Docker experiment and forms an exact cold/reactivate pair: +the reverse event names the cold event, restores its recorded prior state, and +matches experiment, config, policy, manifest, and audit evidence. Unreversed, +unrelated, malformed, or mixed history is ineligible and causes no automatic +repair. The old depth cohort is excluded independently through its persisted +experiment membership. + +This exception permits the newly reviewed experiment to hold and later release +rows that were safely released by the prior experiment without weakening the +append-only policy audit chain. + +### Reviewed authority handoff + +The current depth experiment must first become `completed`. Runtime is stopped +canonically, the terminal aggregate report and release manifest are generated, +and exact-SHA reviewed release restores only that experiment's owned cold rows. +The breadth cohort and hold manifests are then generated and applied while +sources remain stopped. Activation rejects any other unreleased, nonterminal, +or fenced Docker experiment authority. + +The breadth experiment uses the existing state machine, experiment-row-first +locking, resolver tokens, target reservations, finite retries, capacity +accounting, and reversible holds. Since both shallow and deep image limits are +one, every image-bearing membership produces only selection rank one and a +single dispatch wave. + +### Secret-safe decision report + +The terminal report treats target-scoped finding fingerprints as location +evidence, not secret novelty. Primary outcomes are globally deduplicated +credentials, currently alive credentials, and each per scanner-hour. It also +reports physical repository/image coverage, scan states, keycheck completeness, +and per-keyword attribution from frozen provenance. No raw identity or target +material is emitted. + +## Risks / Trade-offs + +- [A selected repository has no eligible image] -> Use the existing deterministic + fresh replacement path; fail the reviewed cohort if exact 2,000 repository + ownership cannot be preserved before activation. +- [Cross-keyword overlap biases ownership] -> Use pinned query-order round-robin, + preserve all frozen query provenance, and distinguish physical totals from + keyword attribution credits. +- [Prior policy history is malformed] -> Leave the row ineligible and fail + closed rather than guessing or rewriting history. +- [Two experiment authorities overlap] -> Reject planning/activation until the + prior experiment is completed, reviewed, released, and fence-free. +- [The cohort is large] -> Keep the hard 2,000-target ceiling, existing shared + capacity limits, two Docker workers, and finite retry/hold behavior. +- [Migration or rollout fails] -> Keep schema changes additive where possible, + stop before authority mutation, and retain manifests and audit rows for a + deterministic retry. diff --git a/openspec/changes/run-bounded-rank1-breadth-experiment/proposal.md b/openspec/changes/run-bounded-rank1-breadth-experiment/proposal.md new file mode 100644 index 0000000..0cf3e53 --- /dev/null +++ b/openspec/changes/run-bounded-rank1-breadth-experiment/proposal.md @@ -0,0 +1,50 @@ +## Why + +The completed portion of the Docker depth pilot found every currently alive +credential in the newest selected image and no additional alive credential in +older image ranks. A larger but still bounded rank-1 cohort is needed to test +whether spending the same capacity on repository breadth produces better +credential and currently-alive yield than blanket image depth. + +## What Changes + +- Add a separate rank-1-only Docker breadth experiment over exactly 2,000 + previously unscanned physical repository anchors from the existing complete + frozen discovery pass, excluding the current depth-pilot cohort. +- Balance the cohort without using prior yield: each of the 52 keywords with an + eligible remaining pool receives 38 repositories and 24 deterministically + selected keywords receive one additional repository; the 9 exhausted + keywords retain explicit zero coverage. +- Deduplicate physical repositories across keywords while preserving all fresh + keyword provenance, and scan at most one newest eligible immutable image per + selected repository. +- Keep the new experiment fail-closed until the current depth experiment is + terminal and its cold rows have passed reviewed release; never run two Docker + experiment authorities concurrently. +- Persist a reviewable immutable cohort plan before activation and retain the + existing lease, reservation, fencing, finite-retry, capacity, and reversible + hold guarantees. +- Report globally deduplicated credentials, currently alive credentials, + repository/image coverage, and both yield measures per scanner-hour, with + per-keyword attribution and no secret or target material. + +## Capabilities + +### New Capabilities + +- `docker-rank1-breadth-experiment`: A balanced, deterministic, physically + deduplicated 2,000-repository rank-1 experiment with reviewed authority + handoff, bounded execution, and secret-safe yield reporting. + +### Modified Capabilities + +## Impact + +- Docker experiment validation, cohort planning, authority handoff, resolver + admission, rank-1 target scheduling, and aggregate reporting. +- Managed PostgreSQL experiment state and audit evidence, with migrations only + where the existing depth-experiment schema cannot represent the new plan. +- Docker experiment configuration and focused unit/PostgreSQL integration + coverage. +- Runtime operations require canonical stop, reviewed release of the completed + depth experiment, reviewed activation of this change, and canonical restart. diff --git a/openspec/changes/run-bounded-rank1-breadth-experiment/specs/docker-rank1-breadth-experiment/spec.md b/openspec/changes/run-bounded-rank1-breadth-experiment/specs/docker-rank1-breadth-experiment/spec.md new file mode 100644 index 0000000..e3f500e --- /dev/null +++ b/openspec/changes/run-bounded-rank1-breadth-experiment/specs/docker-rank1-breadth-experiment/spec.md @@ -0,0 +1,97 @@ +## ADDED Requirements + +### Requirement: Reviewed rank-one breadth authority +The system SHALL support a code-pinned rank-one breadth profile of 61 ordered +queries, at most 39 owned repositories per query, one image per repository, and +exactly 2,000 unique physical repository selections and target slots. It SHALL +reject arbitrary limit combinations and SHALL preserve the existing depth +profile unchanged. + +#### Scenario: Breadth profile is valid +- **WHEN** configuration supplies the exact reviewed breadth selector and limits +- **THEN** validation SHALL produce a distinct immutable authority hash with a physical target ceiling of 2,000 + +#### Scenario: Profile limits are mixed +- **WHEN** configuration combines a selector or limit from different reviewed profiles +- **THEN** startup SHALL fail before secrets, database mutation, network work, or workers initialize + +#### Scenario: Prior authority is not released +- **WHEN** another Docker experiment is nonterminal, unreleased, or fenced +- **THEN** breadth cohort application and activation SHALL fail without changing either experiment + +### Requirement: Exact balanced physical cohort +The system SHALL deterministically select exactly 2,000 previously unscanned +physical repository anchors from the existing complete frozen discovery pass, +excluding every repository in the depth cohort. Selection SHALL be independent +of prior finding or credential yield. + +#### Scenario: Reviewed frozen pool is selected +- **WHEN** 52 keywords have sufficient eligible repositories and nine have none +- **THEN** each nonempty keyword SHALL own 38 unique repositories, the first 24 still-eligible keywords in pinned order SHALL own one additional repository, and empty keywords SHALL remain explicit zero rows + +#### Scenario: Repository appears under several keywords +- **WHEN** the next candidate is already physically owned by an earlier round-robin selection +- **THEN** it SHALL consume neither another physical slot nor the selecting keyword's quota, and selection SHALL continue to that keyword's next eligible candidate + +#### Scenario: Exact cohort cannot be formed +- **WHEN** deterministic eligible selection cannot reach exactly 2,000 unique repositories with the reviewed distribution +- **THEN** manifest generation or application SHALL fail closed and no partial experiment cohort SHALL activate + +### Requirement: Released policy history eligibility +The system SHALL admit a previously held repository only when its complete +policy-event history consists exclusively of authority-valid cold/reactivate +pairs owned by completed and released Docker experiments. + +#### Scenario: Prior hold was exactly released +- **WHEN** every prior cold event has one matching reverse event that restores its recorded state and matches experiment, config, policy, manifest, and audit evidence +- **THEN** the unfenced unscanned repository MAY participate in the new reviewed cohort or hold manifest + +#### Scenario: Prior history is incomplete or unrelated +- **WHEN** any policy event is unreversed, malformed, belongs to an unreleased experiment, or represents an unrelated policy +- **THEN** the repository SHALL remain ineligible and its queue and event history SHALL remain unchanged + +### Requirement: Rank-one-only execution +The system SHALL resolve at most the newest eligible immutable image for each +selected repository and SHALL create no image selection above rank one. + +#### Scenario: Repository has an eligible newest image +- **WHEN** its dedicated resolver completes under a valid owner, generation, token, and experiment authority +- **THEN** exactly one rank-one selection MAY consume one physical experiment target slot + +#### Scenario: Candidate image is unsafe +- **WHEN** the newest candidate is previously scanned, failed, quarantined, independently cold, fenced, or otherwise ineligible +- **THEN** existing deterministic safe replacement rules SHALL apply without reactivating historical work or selecting an older image rank for depth + +#### Scenario: Breadth work is dispatched +- **WHEN** rank-one targets become available +- **THEN** they SHALL use one round-robin dispatch wave with existing reservation, capacity, retry, and terminal-binding guarantees + +### Requirement: Reviewed handoff and reversible holds +The system SHALL activate the breadth experiment only through a stopped-runtime +reviewed handoff after the depth experiment completes and releases its owned +cold rows. Breadth-owned non-cohort holds SHALL remain exactly reversible. + +#### Scenario: Canonical handoff succeeds +- **WHEN** runtime is stopped, the depth experiment is completed and fence-free, its exact release SHA is approved, and the breadth cohort and hold SHAs are approved +- **THEN** release and breadth activation SHALL commit through their existing experiment-row-first fenced protocols before runtime restarts + +#### Scenario: Breadth release is reviewed +- **WHEN** the breadth experiment later completes and its exact reactivation manifest is approved +- **THEN** only unreversed breadth-owned cold events SHALL restore their recorded prior states + +### Requirement: Secret-safe breadth reporting +The system SHALL report physical coverage, globally deduplicated credential and +currently-alive yield, per-keyword attribution, scan cost, and both yield +measures per scanner-hour without exposing secret or target material. + +#### Scenario: Credential repeats across keywords +- **WHEN** one credential is found in a repository attributed to multiple frozen keyword observations +- **THEN** global totals SHALL count it once while each eligible keyword attribution MAY receive an explicit non-additive credit + +#### Scenario: Report measures novelty +- **WHEN** findings repeat across target-scoped locations +- **THEN** the decision report SHALL distinguish finding locations from detector-secret identities and credential identities and SHALL use credentials and currently-alive credentials as primary outcomes + +#### Scenario: Report reads sensitive evidence +- **WHEN** aggregate reporting accesses findings or keycheck rows +- **THEN** output SHALL omit raw credentials, repositories, image targets, URLs, hashes, excerpts, DSNs, and configuration identities diff --git a/openspec/changes/run-bounded-rank1-breadth-experiment/tasks.md b/openspec/changes/run-bounded-rank1-breadth-experiment/tasks.md new file mode 100644 index 0000000..934ca28 --- /dev/null +++ b/openspec/changes/run-bounded-rank1-breadth-experiment/tasks.md @@ -0,0 +1,35 @@ +## 1. Reviewed Profile And Schema + +- [ ] 1.1 Add exact versioned depth and rank-one breadth profile validation while preserving the existing depth profile hashes and behavior. +- [ ] 1.2 Widen PostgreSQL/SQLite experiment bounds only to the reviewed 39-repository and 2,000-target profile. +- [ ] 1.3 Make persisted resolver authority dynamic from the exact reviewed profile instead of hardcoded depth constants. + +## 2. Deterministic Cohort + +- [ ] 2.1 Select fresh candidates from the existing complete frozen pass while excluding the prior depth cohort and unsafe queue state. +- [ ] 2.2 Validate exact fully reversed prior Docker-experiment policy history without accepting unrelated or incomplete events. +- [ ] 2.3 Implement physically deduplicated pinned-query round-robin planning to exactly 2,000 repositories with 38/39/0 keyword ownership. +- [ ] 2.4 Preserve the old depth plan and manifest bytes and add strict breadth plan/manifest validation. +- [ ] 2.5 Revalidate every reviewed cohort row and its provenance/history under locks before application. + +## 3. Runtime Authority + +- [ ] 3.1 Enforce that no other Docker experiment is nonterminal, unreleased, or fenced at breadth application and activation. +- [ ] 3.2 Reuse reversible holds for safely released rows and retain exact append-only policy-event authority. +- [ ] 3.3 Resolve at most one newest immutable image per repository and prohibit breadth selections above rank one. +- [ ] 3.4 Keep experiment-row-first leases, reservations, capacity, finite retries, and terminal target reconciliation unchanged. + +## 4. Reporting And Verification + +- [ ] 4.1 Add secret-safe breadth reporting for globally deduplicated credentials, alive credentials, keyword attribution, coverage, scan-hours, and yield per scanner-hour. +- [ ] 4.2 Add focused unit tests for profile validation, exact balanced physical selection, old-profile compatibility, and prior-release history. +- [ ] 4.3 Add focused PostgreSQL integration coverage for authority handoff, locked cohort application, rank-one dispatch, and reversible holds. +- [ ] 4.4 Run focused tests and strict OpenSpec validation. + +## 5. Reviewed Launch + +- [ ] 5.1 Canonically stop runtime and verify sources, workers, leases, reservations, and experiment fences are quiescent. +- [ ] 5.2 Complete and report the two remaining depth targets or apply only an existing reviewed terminal protocol. +- [ ] 5.3 Generate and approve the depth release manifest, then verify every owned cold event is exactly reversed. +- [ ] 5.4 Generate, review, and apply the exact 2,000-repository breadth cohort and hold manifests. +- [ ] 5.5 Update the reviewed runtime profile, canonically restart, and verify secret-safe live progress with no hold, exhausted retry, or fence anomaly. diff --git a/openspec/changes/simplify-dashboard/.openspec.yaml b/openspec/changes/simplify-dashboard/.openspec.yaml new file mode 100644 index 0000000..d658936 --- /dev/null +++ b/openspec/changes/simplify-dashboard/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-08-02 diff --git a/openspec/changes/simplify-dashboard/design.md b/openspec/changes/simplify-dashboard/design.md new file mode 100644 index 0000000..e44b324 --- /dev/null +++ b/openspec/changes/simplify-dashboard/design.md @@ -0,0 +1,52 @@ +## Context + +The current Streamlit dashboard has six visible pages. PostgreSQL is authoritative for scans, findings, queues, and keychecks, but several pages still read compatibility files or filter a pre-limited client-side frame. The dashboard is read-only, loopback-only, supervisor-managed, and disabled by default; those safety boundaries must remain. + +## Goals / Non-Goals + +**Goals:** +- Present the operational summary and finding lookup on one understandable page. +- Apply one selected UTC time window consistently to all historical statistics. +- Use PostgreSQL for every result, queue, and validation value. +- Accept convenient pasted lookup values without querying or rendering raw secret columns. +- Keep queries bounded and preserve degraded behavior when PostgreSQL is unavailable. + +**Non-Goals:** +- Add scanner lifecycle controls or expose the dashboard beyond loopback. +- Add a REST service, database migration, or write path. +- Preserve retired queue-file and compatibility-file diagnostics in the visible UI. +- Turn the dashboard into a raw log or raw credential browser. + +## Decisions + +1. **Keep Streamlit and replace only the visible information architecture.** The existing supervisor launch, dependency isolation, read-only connection, redaction, and health checks are valuable. A new frontend/backend split would add operational surface without improving this local dashboard. + +2. **Render one page with four sections.** The order is lookup, time-window KPIs, source/alive breakdowns, then compact runtime health. The lookup is placed first because it is a direct operator task; all secondary diagnostics are omitted or collapsed. + +3. **Use bound UTC timestamps in SQL before aggregation and limits.** Presets cover 1 hour, 24 hours, 7 days, and 30 days; custom start/end values are converted to UTC. This replaces the current newest-5000-rows client filter. + +4. **Use only authoritative stores.** Historical metrics come from `source_cycles`, `target_scans`, `findings`, `keycheck_credentials`, `keycheck_current_state`, and `keycheck_results`. Current backlog and pipeline state come from PostgreSQL. Supervisor status and `scan_limiter.db` remain valid operational sidecars. Retired queue files, global runner state, TSV summaries, and JSONL metrics are not rendered. + +5. **Normalize lookup input before database access.** A 64-hex value is treated as a digest; numeric and UID-like values are searched by indexed identity; credential-like text is SHA-256 hashed in memory and only the digest reaches SQL. Short or structured non-secret metadata uses escaped `LIKE` predicates. No query selects raw payload columns, and the submitted raw value is never echoed. + +6. **Lookup overrides statistical filters.** Search results always show current validation state and all bounded matching origins regardless of the selected reporting period or access-tier preset. This avoids hiding a valid credential behind an unrelated dashboard filter. + +7. **Retain bounded rendering and query failure isolation.** Lookup origins and breakdowns have explicit limits. Query failure rolls back the PostgreSQL transaction and degrades only the affected section. + +## Risks / Trade-offs + +- **Raw pasted credentials traverse the local Streamlit session before hashing** -> Keep loopback-only authority, hash immediately, never log/echo/persist the input, and offer SHA-256 lookup as the safest path. +- **Broad metadata lookup can be expensive** -> Escape wildcard characters, require a useful minimum length, search indexed exact identities first, and cap returned origins. +- **Removing advanced pages hides forensic diagnostics** -> PostgreSQL and log files remain available to engineering tools; the operator dashboard intentionally prioritizes clarity. +- **Existing helper functions may remain temporarily unused** -> Remove the visible routes first and prune only when tests establish that no safety helper depends on them. + +## Migration Plan + +1. Add and test the new query/lookup helpers. +2. Switch `main()` to the single page and disable Streamlit telemetry. +3. Run focused dashboard tests and a live supervisor-launched desktop/mobile smoke test. +4. Keep dashboard startup disabled by default. Rollback is a source revert; no persisted data changes are involved. + +## Open Questions + +None required for the initial implementation. diff --git a/openspec/changes/simplify-dashboard/proposal.md b/openspec/changes/simplify-dashboard/proposal.md new file mode 100644 index 0000000..e114e0f --- /dev/null +++ b/openspec/changes/simplify-dashboard/proposal.md @@ -0,0 +1,29 @@ +## Why + +The existing six-page dashboard mixes PostgreSQL authority with retired compatibility files, producing contradictory queue counts and incomplete time-window statistics. Operators need one fast, obvious view that answers the same questions as the current manual status checks and can locate a finding without exposing raw credentials. + +## What Changes + +- Replace the visible multi-page dashboard with one PostgreSQL-authoritative observability page. +- Add accurate preset and custom time windows applied in SQL before aggregation or row limits. +- Show scanner activity, findings, errors, new and current alive credentials, queue state, and compact runtime health in one view. +- Add unified lookup by pasted credential, SHA-256 identity, finding ID/UID, target, path, commit, or other redacted metadata. +- Hash credential-like lookup input immediately and never query or render raw secret columns. +- Remove legacy queue-file, global runner-state, TSV-gated, log-browsing, and advanced/debug panels from the visible interface. +- Keep the dashboard read-only, loopback-only, and supervisor-managed. + +## Capabilities + +### New Capabilities +- `single-page-observability`: Accurate time-window scanner statistics, current runtime state, alive credential summaries, and safe finding lookup on one page. + +### Modified Capabilities + +None. + +## Impact + +- Primary implementation: `app/dashboard.py`. +- Focused behavior and security coverage: `tests/test_dashboard_behavior.py` and `tests/test_dashboard_secret_guard.py`. +- PostgreSQL remains authoritative; no database migration, mutation endpoint, new service, or external API is introduced. +- Existing supervisor lifecycle and disabled-by-default startup policy remain unchanged. diff --git a/openspec/changes/simplify-dashboard/specs/single-page-observability/spec.md b/openspec/changes/simplify-dashboard/specs/single-page-observability/spec.md new file mode 100644 index 0000000..195cddb --- /dev/null +++ b/openspec/changes/simplify-dashboard/specs/single-page-observability/spec.md @@ -0,0 +1,72 @@ +## ADDED Requirements + +### Requirement: Single-page operator view +The dashboard SHALL present finding lookup, reporting metrics, source results, alive credential results, and current runtime health on one page without legacy page navigation. + +#### Scenario: Operator opens the dashboard +- **WHEN** the supervisor-launched dashboard session connects +- **THEN** one page presents the primary lookup and observability sections in operational priority order + +#### Scenario: Retired diagnostics remain hidden +- **WHEN** the page renders successfully +- **THEN** it does not present compatibility queue files, global runner state, TSV-gated summaries, raw logs, or advanced/debug navigation + +### Requirement: Accurate reporting window +The dashboard SHALL apply the selected start and end timestamps in PostgreSQL before aggregating or limiting historical rows. + +#### Scenario: Preset period is selected +- **WHEN** the operator selects 1 hour, 24 hours, 7 days, or 30 days +- **THEN** all historical KPI and breakdown queries use that exact UTC interval + +#### Scenario: Custom period is selected +- **WHEN** the operator supplies a valid custom start and end +- **THEN** the page reports data only from the normalized custom interval + +#### Scenario: Invalid custom period is supplied +- **WHEN** the end is not later than the start +- **THEN** the dashboard explains the error and does not run historical aggregation queries + +### Requirement: PostgreSQL-authoritative summary +The dashboard SHALL derive scanner, finding, queue, candidate, and validation statistics from PostgreSQL and SHALL label current runtime values separately from period values. + +#### Scenario: Period contains activity +- **WHEN** scans and keychecks exist in the selected interval +- **THEN** the page shows scanned targets, findings, errors, newly alive credentials, current alive credentials, and grouped source/provider results + +#### Scenario: No period activity exists +- **WHEN** no matching historical rows exist +- **THEN** the page renders zero-valued KPIs and clear empty states without falling back to compatibility files + +### Requirement: Safe unified finding lookup +The dashboard SHALL locate current validation state and finding origins using a pasted credential, SHA-256 identity, finding ID/UID, masked value, target, path, commit, or other redacted metadata without selecting raw database payload columns. + +#### Scenario: Raw credential-like value is pasted +- **WHEN** the operator submits a credential-like value +- **THEN** the dashboard hashes it in memory and sends only its SHA-256 identity to PostgreSQL + +#### Scenario: Exact identity is pasted +- **WHEN** the operator submits a finding ID, finding UID, fingerprint, or SHA-256 identity +- **THEN** indexed exact predicates locate matching current status and bounded origins independently of reporting filters + +#### Scenario: Non-secret metadata is pasted +- **WHEN** the operator submits a sufficiently specific target, path, commit, detector, source, or query fragment +- **THEN** escaped metadata predicates return bounded redacted matches + +#### Scenario: Match has validation state +- **WHEN** a finding or credential is linked to current keycheck state +- **THEN** the result includes service, provider status, status group, checked time, source, query, target, and available location metadata + +### Requirement: Read-only safety boundaries +The simplified dashboard SHALL remain supervisor-authorized, loopback-only, read-only, redacted, and bounded. + +#### Scenario: Dashboard issues database queries +- **WHEN** any page section loads or a lookup is submitted +- **THEN** no mutation statement or forbidden raw payload column is requested + +#### Scenario: PostgreSQL query fails +- **WHEN** a query times out or the connection enters an error transaction +- **THEN** the dashboard rolls back and renders a bounded degraded message instead of failing the process + +#### Scenario: Dashboard is launched without authority +- **WHEN** the process lacks canonical supervisor and loopback launch markers +- **THEN** startup is refused before argument parsing or database access diff --git a/openspec/changes/simplify-dashboard/tasks.md b/openspec/changes/simplify-dashboard/tasks.md new file mode 100644 index 0000000..eac894f --- /dev/null +++ b/openspec/changes/simplify-dashboard/tasks.md @@ -0,0 +1,17 @@ +## 1. Authoritative Data Model + +- [x] 1.1 Add validated preset/custom UTC reporting-window helpers. +- [x] 1.2 Add PostgreSQL-authoritative KPI, source activity, alive breakdown, queue, and runtime queries. +- [x] 1.3 Add safe lookup normalization and bounded current-status/origin queries. + +## 2. Single-Page Interface + +- [x] 2.1 Build the one-page lookup, KPI, breakdown, and runtime layout with clear empty/error states. +- [x] 2.2 Replace visible legacy navigation and compatibility panels with the single-page route. +- [x] 2.3 Disable Streamlit usage telemetry while retaining loopback supervisor authority and disabled-by-default startup. + +## 3. Verification + +- [x] 3.1 Add focused tests for time-window validation, SQL-before-limit behavior, lookup hashing, identity search, and raw-column exclusion. +- [x] 3.2 Run dashboard and security tests, then the broader relevant suite. +- [x] 3.3 Launch through the supervisor and smoke-test desktop/mobile rendering, search, accurate 24-hour totals, and clean shutdown. diff --git a/openspec/changes/stabilize-and-widen-scanner-coverage/.openspec.yaml b/openspec/changes/stabilize-and-widen-scanner-coverage/.openspec.yaml new file mode 100644 index 0000000..e767a17 --- /dev/null +++ b/openspec/changes/stabilize-and-widen-scanner-coverage/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-06-15 diff --git a/openspec/changes/stabilize-and-widen-scanner-coverage/design.md b/openspec/changes/stabilize-and-widen-scanner-coverage/design.md new file mode 100644 index 0000000..a540222 --- /dev/null +++ b/openspec/changes/stabilize-and-widen-scanner-coverage/design.md @@ -0,0 +1,89 @@ +## Context + +> Supersession, 2026-07-27: the approved cumulative-lag/memory/backpressure architecture supersedes this change's earlier no-new-queue and conservative-concurrency constraints. PostgreSQL is now sole authority, sources hand off S:-backed durable bundles, JSONL/status files are asynchronous projections, normal keychecks use PostgreSQL candidates, and the fair global scan limit is three. Process recycling, GC trimming, and reduced concurrency are not correctness mechanisms. + +The scanner runs multiple supervised source processes that share runtime state, queue files, logs, and an observability database. Recent logs show source restarts caused by Windows `PermissionError` during state-file replacement, while dashboard and supervisor read the same state files for status display. Coverage is also limited by conservative artifact size caps and by discovery inputs that are broad but not always high-signal. + +The original incremental constraints below remain historical context only where contradicted by the supersession above. + +## Goals / Non-Goals + +**Goals:** +- Prevent transient Windows state-file read/write races from crashing source processes. +- Increase scan coverage through configurable size limit bumps while preserving conservative concurrency. +- Improve discovery signal by focusing metadata searches on provider, host, and framework terms. +- Strengthen `package_git` as a discovery backbone through better repository URL extraction and canonicalization. +- Make GitHub Actions and GitLab CI scanning use better parsed seeds and modestly larger per-cycle target sets. +- Continue provider-specific detector and keychecker additions with safe validation and context routing. + +**Non-Goals:** +- No deferred/deep queue in this change. +- No new checked/skipped classification model in this change. +- No scheduler rewrite or central scoring engine. +- No broad generic `api_key`, `secret`, or `token` terms in normal repo/package metadata discovery by default. +- No removal of existing command-line entry points or queue file formats. + +## Decisions + +### Use retrying unique-temp state writes instead of locking + +State writes will use a unique temporary file name and retry `os.replace` on transient Windows permission failures. This keeps the existing JSON state model and avoids cross-process lock files or SQLite migration. + +Alternatives considered: +- Lock files: rejected for now because they add another failure mode and require all readers/writers to cooperate. +- SQLite state: rejected as too large for this incremental reliability fix. +- Ignoring failed state writes: rejected because auth/query state must remain observable. + +### Increase limits through configuration first + +Package, Postman, and CI artifact limits will be increased in config while keeping source worker counts conservative. The implementation will not introduce deferred queues or new outcome classes; oversized skips remain visible through existing logs and target scan records. + +Alternatives considered: +- Remove limits entirely: rejected because large archives can exhaust disk, CPU, and scan slots. +- Add deep-lane queues: deferred to a separate design because it needs stronger classification semantics. + +### Keep metadata discovery provider-focused + +Repository/package metadata search should use provider names, API hosts, framework terms, and ecosystem terms. Exact secret variable terms should be reserved for code/artifact-oriented searches where content is actually searched. + +Alternatives considered: +- Add `api_key` globally: rejected because most sources search names/descriptions/readmes and this produces low-signal security-tool/tutorial results. + +### Improve existing CI seed flow before increasing volume + +CI sources should first parse more existing seed formats from DB records, package candidates, and findings. After parsing improves, `ci_seed_scan_limit` and per-cycle target counts can be raised modestly. + +Alternatives considered: +- Increase CI volume immediately: rejected because current logs show many unparseable/known seeds, so raw volume would mostly amplify waste. + +### Add provider support as detector plus keychecker pairs + +Provider expansion should follow the Qwen/DashScope pattern: contextual detector, safe keychecker, and routing safeguards for ambiguous `sk-...` formats. + +Alternatives considered: +- Detector-only additions: rejected for providers where validation is feasible because they increase unverified noise. + +## Risks / Trade-offs + +- Windows state retry may hide a persistent file access problem for a few hundred milliseconds -> surface the final error after bounded retries. +- Higher artifact size caps increase runtime and disk pressure -> keep workers conservative and rely on existing scan slot limits. +- Provider-focused queries may miss generic projects that leak keys -> use package/artifact/code-oriented paths for exact env var searches instead of metadata search. +- CI seed parsing improvements may still leave many stale/known targets -> postpone TTL/revisit policy until outcome semantics are revisited. +- Context routing can misclassify ambiguous keys when context is weak -> only route away from a provider when strong provider-specific context is present. + +## Migration Plan + +1. Apply state write retry first and monitor source restarts. +2. Increase size caps in config and monitor disk usage, scan duration, and skipped/error counts. +3. Adjust discovery query lists and verify target volume remains healthy. +4. Improve `package_git` metadata URL extraction/canonicalization. +5. Improve CI seed parsing, then raise CI limits modestly. +6. Add the next provider detector/keychecker pair using the established pattern. + +Rollback is straightforward for each step: revert the code change for state writes, restore previous config caps/query lists, or disable individual CI/provider changes. + +## Open Questions + +- Should oversized artifacts remain checked under current semantics, or should that become a separate future change? +- Which provider detector/keychecker pair should be prioritized after Qwen/DashScope? +- What disk and scan-duration thresholds should trigger reducing size caps again? diff --git a/openspec/changes/stabilize-and-widen-scanner-coverage/proposal.md b/openspec/changes/stabilize-and-widen-scanner-coverage/proposal.md new file mode 100644 index 0000000..fcdea82 --- /dev/null +++ b/openspec/changes/stabilize-and-widen-scanner-coverage/proposal.md @@ -0,0 +1,28 @@ +## Why + +The scanner is now broad enough that small reliability issues and conservative defaults are limiting throughput: source uptime resets from Windows state-file races, large artifacts are skipped too aggressively, and CI/package discovery often spends cycles on low-value or already-known targets. This change stabilizes source execution and widens coverage incrementally without introducing new queues or a scheduler rewrite. + +## What Changes + +- Make runner state persistence resilient to transient Windows file-lock races so sources do not crash when supervisor or dashboard reads state concurrently. +- Increase configurable size limits for package, Postman, and CI artifacts in a controlled way while keeping worker counts conservative. +- Refine discovery queries so repository/package metadata searches focus on provider, host, and framework terms instead of generic secret words. +- Improve `package_git` discovery quality by broadening metadata extraction and canonicalization while preserving the current queue model. +- Improve CI source target selection by parsing more seed formats and scanning more relevant GitHub Actions and GitLab CI repositories per cycle. +- Continue the provider-specific detector plus keychecker pattern, including context-based routing for generic key formats. + +## Capabilities + +### New Capabilities +- `scanner-runtime-stability`: resilient source state persistence and restart behavior for supervised scanner processes. +- `scan-coverage-sizing`: configurable artifact size coverage for packages, Postman artifacts, and CI logs/artifacts. +- `source-discovery-targeting`: higher-signal source queries, package git discovery, and CI seed target selection. +- `provider-key-validation`: provider-specific custom detectors, keycheckers, and context routing for ambiguous key formats. + +### Modified Capabilities + +## Impact + +- Affected code: `app/console_runner.py`, `app/supervisor.py`, `app/dashboard.py`, `app/scanner.py`, `app/scanner_db.py`, `app/config.yaml`, and selected `app/keycheckers/**` modules. +- Affected runtime data: `runtime/state/runner_state_*.json`, `runtime/queues/*`, scanner logs, target scan records, and keycheck results. +- No breaking changes to command-line entry points or existing queue file formats are intended. diff --git a/openspec/changes/stabilize-and-widen-scanner-coverage/specs/provider-key-validation/spec.md b/openspec/changes/stabilize-and-widen-scanner-coverage/specs/provider-key-validation/spec.md new file mode 100644 index 0000000..fecea70 --- /dev/null +++ b/openspec/changes/stabilize-and-widen-scanner-coverage/specs/provider-key-validation/spec.md @@ -0,0 +1,48 @@ +## ADDED Requirements + +### Requirement: PostgreSQL Candidate And Current-State Authority +Normal provider workers SHALL claim one fenced PostgreSQL candidate at a time and SHALL transactionally insert the result, update `keycheck_current_state`, complete that candidate, and create a projection job. + +#### Scenario: Compatibility files lag or fail +- **WHEN** keycheck JSONL or status projection is delayed or quarantined +- **THEN** the committed PostgreSQL result and current state SHALL remain authoritative and unrelated candidates SHALL continue + +#### Scenario: Explicit compatibility input +- **WHEN** an operator selects `--input-mode jsonl` +- **THEN** the bounded legacy JSONL reader MAY be used, but `--input` SHALL be rejected in normal PostgreSQL mode + +### Requirement: Provider Support Uses Detector And Keychecker Pair +New provider API key support SHALL include both detection and validation when a safe validation endpoint is available. + +#### Scenario: Provider has safe validation endpoint +- **WHEN** support is added for a provider with a non-generating authentication or model-list endpoint +- **THEN** the change SHALL include a detector or detector routing rule and a keychecker for that provider + +#### Scenario: Provider has no safe validation endpoint +- **WHEN** a provider does not have a safe validation endpoint +- **THEN** the detector MAY be added only with explicit documentation that validation is unavailable + +### Requirement: Ambiguous Key Formats Use Context Routing +The scanner SHALL use nearby provider-specific context to route ambiguous key formats to the correct keychecker when possible. + +#### Scenario: Qwen context around generic key +- **WHEN** an `sk-...` key is found near Qwen or DashScope context +- **THEN** the key SHALL be routed away from unrelated generic `sk-...` checkers such as DeepSeek when the context does not also identify that provider + +#### Scenario: Weak context around generic key +- **WHEN** an ambiguous key has no strong provider-specific context +- **THEN** the scanner SHALL avoid speculative rerouting that would suppress the existing detector result + +### Requirement: Keycheckers Avoid Token-Generating Probes By Default +Provider keycheckers SHALL prefer safe non-generating validation endpoints where available. + +#### Scenario: Provider exposes model-list endpoint +- **WHEN** a provider exposes an authenticated model-list or account-status endpoint +- **THEN** the keychecker SHALL use that endpoint before considering any generation-style probe + +### Requirement: Keycheck Results Remain Linkable To Findings +Provider keycheckers SHALL emit detector names and result metadata that can be linked back to scanner findings. + +#### Scenario: Custom detector result is validated +- **WHEN** a keychecker validates a key from a custom detector finding +- **THEN** the result SHALL include a stable detector name and source metadata sufficient for DB link repair diff --git a/openspec/changes/stabilize-and-widen-scanner-coverage/specs/scan-coverage-sizing/spec.md b/openspec/changes/stabilize-and-widen-scanner-coverage/specs/scan-coverage-sizing/spec.md new file mode 100644 index 0000000..cb0c928 --- /dev/null +++ b/openspec/changes/stabilize-and-widen-scanner-coverage/specs/scan-coverage-sizing/spec.md @@ -0,0 +1,33 @@ +## ADDED Requirements + +### Requirement: Configurable Larger Artifact Coverage +The scanner SHALL allow package, Postman, GitHub Actions, and GitLab CI artifact size limits to be raised through configuration without code changes. + +#### Scenario: Package artifact limit is increased +- **WHEN** npm or PyPI `max_artifact_size_mb` is configured to a larger value +- **THEN** package scans SHALL use the configured size limit for download and scan decisions + +#### Scenario: CI artifact limits are increased +- **WHEN** GitHub Actions or GitLab CI artifact archive and file limits are configured to larger values +- **THEN** CI scans SHALL use those configured limits for artifact download and extraction decisions + +### Requirement: Conservative Concurrency Preserved +The scanner SHALL preserve per-source worker controls so larger artifact limits do not automatically increase concurrent heavy scans. + +#### Scenario: Size limits increase with unchanged workers +- **WHEN** artifact size limits are raised in configuration and worker counts are unchanged +- **THEN** the scanner SHALL keep using the configured worker counts for the affected source + +### Requirement: Oversized Artifact Visibility +The scanner SHALL record existing oversized-artifact skip reasons in logs and target scan records using the current result model. + +#### Scenario: Artifact remains over configured limit +- **WHEN** an artifact exceeds the configured size limit +- **THEN** the scanner SHALL record a skipped result with the size-limit reason using existing logging and target scan recording paths + +### Requirement: No New Queue Semantics +The scanner SHALL NOT introduce a deferred or deep artifact queue as part of this change. + +#### Scenario: Artifact is too large for current run +- **WHEN** an artifact is skipped because it exceeds the configured limit +- **THEN** the scanner SHALL handle it through the current scan result and queue behavior without creating a new queue type diff --git a/openspec/changes/stabilize-and-widen-scanner-coverage/specs/scanner-runtime-stability/spec.md b/openspec/changes/stabilize-and-widen-scanner-coverage/specs/scanner-runtime-stability/spec.md new file mode 100644 index 0000000..c6afe4f --- /dev/null +++ b/openspec/changes/stabilize-and-widen-scanner-coverage/specs/scanner-runtime-stability/spec.md @@ -0,0 +1,40 @@ +## ADDED Requirements + +### Requirement: Durable Asynchronous Result Handoff Supersedes Source Publication +PostgreSQL SHALL be the sole result authority. A fair global permit count of three SHALL end after a pre-reserved result bundle is fsynced and atomically renamed on `S:`, without covering database ingest, JSONL projection, or keychecks. + +#### Scenario: Projector or database is blocked after bundle handoff +- **WHEN** a source completes the durable ready rename +- **THEN** its scan permit SHALL be released while the bundle remains recoverable and downstream backlog applies capacity-based admission pressure + +### Requirement: Correctness Does Not Depend On Recycling +The runtime SHALL NOT use GC trimming, source lifetime limits, private-memory restarts, periodic recycling, reduced concurrency, or time-based restarts to preserve correctness. + +#### Scenario: Runtime memory grows after warmup +- **WHEN** a managed process reports increasing private memory +- **THEN** admission and durable queue capacity SHALL provide backpressure without recycling the process or reducing the configured three scan permits + +### Requirement: Resilient Runner State Writes +The scanner SHALL persist runner state using a unique temporary file per write attempt and SHALL retry replacement when the operating system reports a transient file access error. + +#### Scenario: Concurrent state read during write +- **WHEN** a supervised source writes `runner_state_.json` while supervisor or dashboard reads the same state file +- **THEN** the source process SHALL retry the replacement and continue without crashing when the file becomes available within the retry window + +#### Scenario: Persistent state write failure +- **WHEN** the state file cannot be replaced after the bounded retry window +- **THEN** the source process SHALL surface the final write error instead of silently discarding state changes + +### Requirement: State Writes Avoid Shared Temp Path Contention +The scanner SHALL avoid using a single shared `.tmp` path for repeated state writes from supervised processes. + +#### Scenario: Multiple state write attempts overlap +- **WHEN** two state write attempts occur close together for the same state file +- **THEN** each attempt SHALL use a distinct temporary file path before replacing the final state file + +### Requirement: Restart Noise Reduction +The supervisor SHALL no longer restart sources due only to transient state-file replacement races that resolve within the retry window. + +#### Scenario: GitLab state replacement race resolves +- **WHEN** the GitLab source hits a transient Windows file lock while saving state +- **THEN** the source SHALL complete the state save after retry and its `up` timer SHALL not reset because of that transient race diff --git a/openspec/changes/stabilize-and-widen-scanner-coverage/specs/source-discovery-targeting/spec.md b/openspec/changes/stabilize-and-widen-scanner-coverage/specs/source-discovery-targeting/spec.md new file mode 100644 index 0000000..c428bb5 --- /dev/null +++ b/openspec/changes/stabilize-and-widen-scanner-coverage/specs/source-discovery-targeting/spec.md @@ -0,0 +1,48 @@ +## ADDED Requirements + +### Requirement: Metadata Discovery Uses High-Signal Queries +Repository, package, and image metadata discovery SHALL prefer provider names, API hosts, framework names, and ecosystem terms over generic secret-related words. + +#### Scenario: Repo metadata query list is reviewed +- **WHEN** default repository or package metadata queries are configured +- **THEN** broad terms such as `api_key`, `secret`, and `token` SHALL NOT be added by default unless the source searches content rather than metadata + +#### Scenario: Provider query is configured +- **WHEN** a provider such as Qwen, DashScope, Groq, or OpenRouter is targeted +- **THEN** provider names, API hostnames, and framework terms SHALL be eligible for metadata discovery queries + +### Requirement: Exact Secret Terms Reserved For Content-Oriented Sources +Exact environment variable and API host searches SHALL be used for content-oriented sources such as Postman/API artifacts, code search, and CI artifacts rather than generic metadata searches. + +#### Scenario: Env var search term is added +- **WHEN** an exact term such as `DASHSCOPE_API_KEY` is added to discovery +- **THEN** it SHALL be applied to a source that can inspect file or artifact content + +### Requirement: Package Git Repository Canonicalization +`package_git` discovery SHALL canonicalize repository URLs from package metadata before queueing targets. + +#### Scenario: Package metadata contains issue URL +- **WHEN** package metadata contains a repository-like URL ending in `/issues` +- **THEN** `package_git` discovery SHALL normalize it to the canonical repository URL when possible + +#### Scenario: Package metadata contains repository URL variants +- **WHEN** package metadata contains `.git`, branch, tree, or homepage variants for the same repository +- **THEN** `package_git` discovery SHALL avoid queueing duplicate normalized repository targets + +### Requirement: CI Seed Parsing From Existing Data +GitHub Actions and GitLab CI target discovery SHALL parse repository/project seeds from existing scanner DB records, package git candidates, findings, and target scan metadata where possible. + +#### Scenario: Seed record contains package git target JSON +- **WHEN** a CI source examines a package git candidate or target scan record with repository metadata +- **THEN** it SHALL derive a GitHub repository or GitLab project seed when the URL provider matches the CI source + +#### Scenario: Seed record cannot identify repository +- **WHEN** no repository or project can be derived from a seed record +- **THEN** the CI source SHALL count it as unparseable and continue processing other seeds + +### Requirement: CI Scan Volume Is Configurable +CI source scan volume SHALL remain controlled by existing per-source configuration values. + +#### Scenario: CI seed limit is raised +- **WHEN** `ci_seed_scan_limit` or `ci_max_repos_per_cycle` is increased in configuration +- **THEN** the CI source SHALL use the configured value without requiring code changes diff --git a/openspec/changes/stabilize-and-widen-scanner-coverage/tasks.md b/openspec/changes/stabilize-and-widen-scanner-coverage/tasks.md new file mode 100644 index 0000000..f59e8d0 --- /dev/null +++ b/openspec/changes/stabilize-and-widen-scanner-coverage/tasks.md @@ -0,0 +1,48 @@ +## 1. Runtime Stability + +- [x] 1.1 Update `console_runner.save_state()` to write through a unique temp file per attempt. +- [x] 1.2 Add bounded retry/backoff around `os.replace` for transient `PermissionError` and related Windows access errors. +- [x] 1.3 Ensure failed state replacement after retries still surfaces the final error. +- [x] 1.4 Add or run a focused smoke test that simulates state writes while the state file is repeatedly read. + +## 2. Artifact Size Coverage + +- [x] 2.1 Increase package artifact size limits in `config.yaml` while keeping npm/PyPI worker counts unchanged. +- [x] 2.2 Increase Postman artifact size limits in `config.yaml` while keeping Postman worker counts unchanged. +- [x] 2.3 Increase GitHub Actions and GitLab CI artifact archive/file limits in `config.yaml` while keeping CI worker counts unchanged. +- [x] 2.4 Verify oversized artifacts still record existing skipped reasons in logs and target scan records. + +## 3. Metadata Discovery Targeting + +- [x] 3.1 Review default repository/package/image metadata query lists and keep generic `api_key`, `secret`, and `token` terms out of those defaults. +- [x] 3.2 Add or retain provider/framework metadata queries for high-signal discovery terms such as Qwen, DashScope, Groq, OpenRouter, LiteLLM, LangChain, and LlamaIndex. +- [x] 3.3 Ensure exact env var/API host terms are used only for content-oriented discovery paths such as Postman/API artifacts, code-like artifact search, or CI artifacts. + +## 4. Package Git Discovery + +- [x] 4.1 Extend package metadata extraction to inspect repository, homepage, bugs, and related package metadata fields for GitHub/GitLab repository URLs. +- [x] 4.2 Canonicalize package-derived repository URLs by removing `.git`, issue paths, branch/tree paths, and other non-repository suffixes when possible. +- [x] 4.3 Deduplicate package git targets by normalized repository URL before queueing. +- [x] 4.4 Gradually increase `package_git.pages` and verify target volume, duplicate rate, and source runtime remain acceptable. + +## 5. CI Seed Selection + +- [x] 5.1 Improve GitHub Actions seed parsing from scanner DB target scans, findings, and package git candidate records. +- [x] 5.2 Improve GitLab CI seed parsing from scanner DB target scans, findings, and package git candidate records. +- [x] 5.3 Verify `skipped_unparseable` counts decrease for CI source discovery. +- [x] 5.4 Increase `ci_seed_scan_limit` and `ci_max_repos_per_cycle` modestly after seed parsing improves. +- [x] 5.5 Verify CI sources still respect configured worker and artifact limits. + +## 6. Provider Validation Pattern + +- [x] 6.1 Keep Qwen/DashScope context routing from sending strong Qwen-context `sk-...` keys to unrelated generic checkers. +- [x] 6.2 Pick the next provider candidate with a safe non-generating validation endpoint. +- [x] 6.3 Add the next provider using the detector plus keychecker pattern. +- [x] 6.4 Ensure new provider keycheck results include stable detector names and metadata for DB link repair. + +## 7. Verification + +- [x] 7.1 Run Python compilation checks for modified Python modules. +- [x] 7.2 Validate OpenSpec specs and task status for this change. +- [x] 7.3 Run targeted smoke commands for state persistence, package discovery, and CI seed discovery. +- [x] 7.4 Inspect supervisor status and recent logs after deployment to confirm source restarts and skipped/unparseable counts improved. diff --git a/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/.openspec.yaml b/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/.openspec.yaml new file mode 100644 index 0000000..41c30ba --- /dev/null +++ b/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-08-19 diff --git a/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/design.md b/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/design.md new file mode 100644 index 0000000..0d09b46 --- /dev/null +++ b/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/design.md @@ -0,0 +1,85 @@ +## Context + +DockerHub TruffleHog commands currently run under the project's supervisor, scan-slot lease, and Windows Job containment while TruffleHog also starts its own overseer process. Runtime evidence shows frequent exit-code-1 runs that emit `running source` but not `finished scanning`; the current fallback classifies these runs as non-retryable `command_exit` failures. + +Controlled runs of previously affected images completed with exit code 0 under the same Windows Job when TruffleHog's embedded overseer was bypassed with `--local-dev`. The flag changes process lifecycle only; it does not enable verification, alter detectors, or relax containment. + +## Goals / Non-Goals + +**Goals:** + +- Give the existing external supervisor sole ownership of Docker scan lifecycle. +- Require positive completion evidence for a successful Docker TruffleHog process. +- Retry incomplete Docker runs through the existing bounded target retry policy. +- Preserve findings emitted before an incomplete process exit. +- Roll out and replay historical failures in bounded, observable stages. + +**Non-Goals:** + +- Upgrade or replace the installed TruffleHog binary. +- Change non-Docker TruffleHog commands. +- Change detector selection, provider routing, or keycheck behavior. +- Treat every Docker diagnostic as retryable. +- Replay the full historical failure set before the canary is healthy. + +## Decisions + +### Bypass the embedded overseer only for DockerHub + +`scan_docker_image` will add `--local-dev` while retaining `--no-update`. The project already provides restart policy, process-tree containment, timeout enforcement, and immutable executable authority, so the embedded updater/overseer is redundant. A Docker-only rollout limits behavioral scope and makes the canary attributable. + +Alternative considered: upgrade TruffleHog first. Rejected for this change because an official current binary also retained the overseer behavior in controlled tests, while an upgrade changes detectors and Docker internals at the same time. + +### Make normal completion explicit + +Diagnostic parsing will record whether the exact JSON message `finished scanning` was observed. For Docker, exit code 0 without this marker is an incomplete run rather than success. A nonzero unexplained exit remains a failure even if the marker exists, but it is retryable because the wrapper lifecycle did not terminate cleanly. + +Alternative considered: trust exit code alone. Rejected because the observed coverage gap is specifically caused by ambiguous process exits and partial output. + +### Reuse the existing bounded retry state machine + +An unexplained Docker exit without completion evidence will use a distinct `command_incomplete` class with `retryable=true`. An unexplained exit after a completion marker will use `wrapper_exit` with `retryable=true`. Existing `target_retry_max_attempts=3` and exponential delay remain authoritative; no unbounded or immediate retry loop is introduced. + +Alternative considered: retry every Docker exit code 1 without changing command lifecycle. Rejected because controlled retries were inconsistent and repeatedly downloaded/scanned the same image without removing the triggering lifecycle race. + +### Preserve partial findings and fail closed + +Findings parsed before an incomplete exit remain in the durable result. The target is not marked clean or done until a complete run succeeds. This preserves useful evidence without claiming full image coverage. + +### Bound internal Docker parallelism + +DockerHub will pass a source-configured TruffleHog concurrency of 4 and use a 600-second target timeout. The existing two Docker workers and 6 GiB per-process Windows Job limit remain unchanged. This replaces up to 32 aggregate internal workers across two image processes with at most 8, reducing decompression and chunking pressure while preserving source-level parallelism. + +Alternative considered: increase the memory cap. Rejected because a production OOM image completed within the existing 6 GiB Job in 239 seconds at concurrency 4; increasing the cap would raise host-wide risk without addressing amplification. + +### Keep detector context timeouts target-scoped and nonfatal + +The exact TruffleHog diagnostic `a detector ignored the context timeout` will be retained as a nonfatal `detector_timeout` warning. A completed RC=0 scan containing only these diagnostics is degraded, not failed. Other timeout diagnostics keep their existing retryable error behavior. + +### Separate canary, replay, and binary upgrade + +The runtime will first deploy the command and classification changes. The canary will compare unexplained Docker exit rate, completion-marker rate, source restarts, and queue health against the existing baseline. Historical exact-signature failures will be requeued in bounded batches only after the canary is healthy. A pinned official TruffleHog upgrade remains a follow-up change. + +## Risks / Trade-offs + +- [The hidden `--local-dev` flag changes in a future binary] -> Keep executable hash pinning and add command-level regression coverage before any binary upgrade. +- [Completion logging changes upstream] -> Treat a missing marker as retryable and bounded, not as success or an infinite retry. +- [More retries increase Docker traffic] -> Keep the existing three-attempt cap and delay policy; replay historical failures in small batches. +- [Lower concurrency increases scan duration] -> Raise the Docker target timeout to 600 seconds and retain two source workers. +- [A detector context timeout omits some detector coverage] -> Persist it as degraded status rather than silently calling the image clean. +- [Partial findings are duplicated across attempts] -> Rely on existing finding and credential identities for deduplication while preserving each scan event. +- [Docker-only behavior diverges from other sources] -> Use the canary to validate the lifecycle decision before considering a common TruffleHog command policy. + +## Migration Plan + +1. Add focused command, diagnostic, and queue-policy tests. +2. Deploy the Docker-only lifecycle change and restart the managed runtime so immutable code authority is refreshed. +3. Observe at least 100 completed Docker attempts or two hours, whichever is longer. +4. Require no stale scan leases, no new unexplained terminal RC=1 failures, and normal pipeline drain before replay. +5. Requeue a small exact-signature batch, verify completion and deduplication, then increase batches conservatively. +6. Roll back by removing the Docker-only flag/classification change and restarting the supervisor; replayed rows remain ordinary auditable scan events. + +## Open Questions + +- What batch size gives acceptable registry traffic during historical replay? Determine from the canary's average duration and Docker rate-limit headroom. +- Should `--local-dev` later become the common policy for every externally supervised TruffleHog source? Decide in a separate change using per-source evidence. diff --git a/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/proposal.md b/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/proposal.md new file mode 100644 index 0000000..96c3f13 --- /dev/null +++ b/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/proposal.md @@ -0,0 +1,30 @@ +## Why + +DockerHub scans frequently terminate with exit code 1 after emitting `running source` but before `finished scanning`. These incomplete runs are currently treated as terminal failures after one attempt, which leaves a material coverage gap even though the scanner runtime and target are often healthy. + +## What Changes + +- Run DockerHub TruffleHog scans without TruffleHog's redundant embedded overseer while retaining the existing external supervisor and Windows Job containment. +- Record whether TruffleHog emitted its normal completion marker. +- Classify an unexplained Docker exit without the completion marker as an incomplete transient run instead of a permanent target failure. +- Reuse the existing bounded target retry policy for incomplete runs. +- Bound Docker's internal TruffleHog concurrency and allow enough time for a contained full-image scan. +- Treat TruffleHog's exact detector context-timeout diagnostic as degraded detector coverage rather than a failed image scan. +- Add a controlled replay path for historical failures matching this exact signature after the canary is healthy. +- Keep the TruffleHog binary upgrade out of this change so lifecycle behavior can be measured independently. + +## Capabilities + +### New Capabilities +- `docker-scan-lifecycle`: Defines completion, containment, retry, and replay behavior for DockerHub TruffleHog scans. + +### Modified Capabilities + +None. + +## Impact + +- Affects Docker command construction and TruffleHog diagnostic classification in `app/scanner.py`. +- Affects Docker target completion disposition in the existing PostgreSQL queue flow. +- Adds focused scanner policy tests and runtime canary checks. +- Does not change provider keycheck behavior, non-Docker scan commands, or the installed TruffleHog binary. diff --git a/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/specs/docker-scan-lifecycle/spec.md b/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/specs/docker-scan-lifecycle/spec.md new file mode 100644 index 0000000..edac3ec --- /dev/null +++ b/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/specs/docker-scan-lifecycle/spec.md @@ -0,0 +1,74 @@ +## ADDED Requirements + +### Requirement: External Docker scan lifecycle ownership +The system SHALL bypass TruffleHog's embedded overseer for DockerHub image scans while retaining the existing external supervisor, scan-slot lease, timeout, output bounds, and Windows Job containment. + +#### Scenario: Docker command construction +- **WHEN** the system constructs a TruffleHog command for a DockerHub image +- **THEN** the command includes both `--local-dev` and `--no-update` + +#### Scenario: Non-Docker command construction +- **WHEN** the system constructs a TruffleHog command for a non-Docker source +- **THEN** this Docker-only capability does not add `--local-dev` + +### Requirement: Explicit Docker scan completion +The system SHALL record whether TruffleHog emitted the exact normal completion message `finished scanning`, and SHALL require both that marker and exit code 0 before treating process execution as complete. + +#### Scenario: Normal completion +- **WHEN** a Docker TruffleHog process exits with code 0 after emitting `finished scanning` +- **THEN** diagnostic metadata records completed execution and no lifecycle error is added + +#### Scenario: Missing completion marker +- **WHEN** a Docker TruffleHog process exits without emitting `finished scanning` +- **THEN** the result is classified as an incomplete retryable run and is not treated as clean or done + +#### Scenario: Nonzero exit after completion marker +- **WHEN** a Docker TruffleHog process emits `finished scanning` but exits nonzero without a more specific diagnostic +- **THEN** the result is classified as a retryable wrapper exit rather than successful execution + +### Requirement: Bounded retry for incomplete Docker runs +The system SHALL route incomplete Docker lifecycle failures through the existing bounded target retry policy and SHALL preserve the terminal attempt limit. + +#### Scenario: Retry remains available +- **WHEN** an incomplete Docker run occurs before the configured maximum target attempt +- **THEN** the queue defers the target using the configured retry delay + +#### Scenario: Attempt limit is reached +- **WHEN** an incomplete Docker run occurs at the configured maximum target attempt +- **THEN** the queue records a terminal failed target and does not create an unbounded retry loop + +### Requirement: Partial finding preservation +The system SHALL retain findings emitted before an incomplete Docker process exit without representing the target as fully scanned. + +#### Scenario: Findings precede incomplete exit +- **WHEN** TruffleHog emits one or more findings and then exits before complete execution is confirmed +- **THEN** those findings remain durable while the target receives retryable incomplete disposition + +### Requirement: Bounded Docker internal parallelism +The system SHALL pass source-configured internal concurrency to Docker TruffleHog commands while retaining the existing source worker and Windows Job limits. + +#### Scenario: Docker canary resource settings +- **WHEN** the configured DockerHub source starts an image scan +- **THEN** TruffleHog runs with internal concurrency 4 and a target timeout of 600 seconds + +### Requirement: Nonfatal detector context timeout +The system SHALL retain the exact diagnostic `a detector ignored the context timeout` as degraded detector coverage rather than a fatal image-scan error. + +#### Scenario: Completed scan with detector timeout +- **WHEN** a Docker scan emits the detector context-timeout diagnostic, emits `finished scanning`, and exits with code 0 +- **THEN** the target result contains a `detector_timeout` warning and no lifecycle error + +#### Scenario: Other timeout diagnostic +- **WHEN** a Docker scan emits a different timeout diagnostic +- **THEN** the existing retryable timeout error policy remains in effect + +### Requirement: Controlled historical replay +The system SHALL replay historical Docker failures matching the exact incomplete-exit signature only in bounded batches after lifecycle canary criteria pass. + +#### Scenario: Canary has not passed +- **WHEN** lifecycle health has not met the defined canary criteria +- **THEN** historical terminal failures are not mass-requeued + +#### Scenario: Canary has passed +- **WHEN** lifecycle health meets the defined canary criteria and a bounded replay batch is selected +- **THEN** only exact-signature Docker failures in that batch are returned to the pending queue diff --git a/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/tasks.md b/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/tasks.md new file mode 100644 index 0000000..159fa49 --- /dev/null +++ b/openspec/changes/stabilize-dockerhub-trufflehog-lifecycle/tasks.md @@ -0,0 +1,22 @@ +## 1. Docker Lifecycle Policy + +- [x] 1.1 Add `--local-dev` only to Docker TruffleHog command construction while retaining existing containment and `--no-update`. +- [x] 1.2 Record the `finished scanning` marker and classify missing completion, unexplained RC=1, and wrapper exit outcomes with explicit retryable error classes. +- [x] 1.3 Confirm incomplete outcomes use the existing three-attempt queue policy and preserve partial findings. +- [x] 1.4 Add source-configured Docker internal concurrency 4 and target timeout 600 seconds. +- [x] 1.5 Classify the exact detector context-timeout message as nonfatal degraded coverage. + +## 2. Regression Coverage + +- [x] 2.1 Add command-construction tests proving Docker receives `--local-dev` and non-Docker commands do not. +- [x] 2.2 Add diagnostic tests for complete RC=0, missing-marker RC=0, incomplete RC=1, wrapper-exit RC=1, and unchanged non-Docker behavior. +- [x] 2.3 Add queue-policy coverage for deferred attempts, terminal attempt exhaustion, and partial-finding preservation. +- [x] 2.4 Run focused scanner tests and the relevant broader regression suite with bytecode writes disabled. +- [x] 2.5 Add resource-plumbing and detector-timeout regression coverage, then rerun affected tests. + +## 3. Validation And Rollout + +- [x] 3.1 Run strict OpenSpec validation and verify the implementation against the change artifacts. +- [x] 3.2 Restart the managed runtime and confirm PostgreSQL, pipeline workers, scan slots, and Docker source health. +- [x] 3.3 Observe the Docker lifecycle canary for at least 100 attempts or two hours before historical replay. +- [x] 3.4 Requeue one bounded batch of exact-signature historical failures and verify completion, deduplication, and queue drain. diff --git a/openspec/changes/stabilize-gitlab-discovery-retries/.openspec.yaml b/openspec/changes/stabilize-gitlab-discovery-retries/.openspec.yaml new file mode 100644 index 0000000..f774115 --- /dev/null +++ b/openspec/changes/stabilize-gitlab-discovery-retries/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-08-20 diff --git a/openspec/changes/stabilize-gitlab-discovery-retries/design.md b/openspec/changes/stabilize-gitlab-discovery-retries/design.md new file mode 100644 index 0000000..3292d5a --- /dev/null +++ b/openspec/changes/stabilize-gitlab-discovery-retries/design.md @@ -0,0 +1,55 @@ +## Context + +GitLab project discovery calls `api_request` without an explicit retry budget. Direct requests therefore receive one attempt, and `ApiRequestError` escapes `run_configured_source`, which terminates the supervised child. The supervisor restarts the child and the persisted query position prevents confirmed target loss, but every transient 30-second read timeout creates avoidable churn and delay. + +## Goals / Non-Goals + +**Goals:** + +- Retry idempotent GitLab project discovery requests with a small bounded budget. +- Keep the GitLab child alive when the bounded network budget is exhausted. +- Persist a failed source cycle without advancing the query or damaging auth state. +- Preserve existing rate-limit and authentication handling. + +**Non-Goals:** + +- Change TruffleHog scan commands or target retry policy. +- Retry non-idempotent requests. +- Hide persistent GitLab outages or loop without delay. +- Change other source APIs in this change. + +## Decisions + +### Configure attempts at the GitLab source boundary + +GitLab source configuration will provide `discovery_request_attempts: 3` and `discovery_retry_delay: 5`. These values flow only into project discovery calls and use the existing `api_request` retry implementation. + +Alternative considered: change the direct-request default globally. Rejected because it would silently alter every source and API call without source-specific evidence. + +### Convert exhausted discovery transport errors into failed cycles + +GitLab project discovery will wrap exhausted request transport failures in a dedicated `GitLabDiscoveryTransportError`. `run_configured_source` will catch only that error for GitLab, roll back any open database transaction, finish the source cycle as failed, retain the current query, and return to the normal configured-source cooldown. It will not mark the token invalid or terminate the process. Payload and programming errors retain fail-fast behavior. + +Alternative considered: rely on supervisor restart. Rejected because process restart is expensive error handling for an ordinary transient network condition. + +### Preserve rate-limit and auth paths + +HTTP 429, 401, and 403 responses continue through the existing GitLab API and auth-pool policy. Only transport failures and the existing retryable HTTP status set use the discovery retry budget. + +## Risks / Trade-offs + +- [A persistent outage lengthens a failed cycle] -> Bound attempts to three and delay to five seconds. +- [A failed page could cause partial discovery ambiguity] -> Discard the cycle's fetched list on failure and retain the current query for the next cycle. +- [A catch could hide programming errors] -> Catch only `ApiRequestError` for the GitLab source; all other exceptions retain fail-fast behavior. +- [Auth state could be corrupted] -> Do not invoke token cooldown or invalidation for transport-only failures. + +## Migration Plan + +1. Add retry and failed-cycle tests using injected timeout responses. +2. Deploy the GitLab-only configuration and error handling. +3. Restart the managed runtime and observe at least one full query rotation or an injected exhaustion test. +4. Roll back by removing the GitLab source settings and narrow exception handler. + +## Open Questions + +- Whether the same direct-request policy should later be adopted by other sources requires separate evidence. diff --git a/openspec/changes/stabilize-gitlab-discovery-retries/proposal.md b/openspec/changes/stabilize-gitlab-discovery-retries/proposal.md new file mode 100644 index 0000000..d5c230c --- /dev/null +++ b/openspec/changes/stabilize-gitlab-discovery-retries/proposal.md @@ -0,0 +1,27 @@ +## Why + +One transient GitLab project-search timeout currently terminates the entire supervised source process. Four single-attempt read timeouts caused four avoidable GitLab restarts in the latest runtime window even though the token remained healthy and the same query succeeded after restart. + +## What Changes + +- Retry transient GitLab discovery requests with a small bounded attempt count and delay. +- Treat exhausted discovery transport failures as a failed source cycle with backoff instead of terminating the supervised child. +- Preserve the current query and authentication state so a failed cycle can resume without a coverage gap. +- Add focused retry, exhaustion, state, and non-GitLab isolation coverage. +- Keep TruffleHog process lifecycle changes outside this change. + +## Capabilities + +### New Capabilities + +- `gitlab-discovery-resilience`: Defines bounded transport retry and nonfatal cycle behavior for GitLab project discovery. + +### Modified Capabilities + +None. + +## Impact + +- Affects GitLab discovery requests and config-cycle error handling in `app/scanner.py` and `app/console_runner.py`. +- Adds source configuration for the retry budget and delay. +- Does not change GitLab token validity, rate-limit rotation, target scan policy, or other source APIs. diff --git a/openspec/changes/stabilize-gitlab-discovery-retries/specs/gitlab-discovery-resilience/spec.md b/openspec/changes/stabilize-gitlab-discovery-retries/specs/gitlab-discovery-resilience/spec.md new file mode 100644 index 0000000..4e303cf --- /dev/null +++ b/openspec/changes/stabilize-gitlab-discovery-retries/specs/gitlab-discovery-resilience/spec.md @@ -0,0 +1,34 @@ +## ADDED Requirements + +### Requirement: Bounded GitLab discovery transport retries +The system SHALL retry idempotent GitLab project discovery requests after transient transport failures using a source-configured bounded attempt count and delay. + +#### Scenario: Transient timeout succeeds on retry +- **WHEN** a GitLab project discovery request times out and a later attempt within the configured budget succeeds +- **THEN** discovery continues using the successful response without restarting the source + +#### Scenario: Retry budget is bounded +- **WHEN** every GitLab project discovery attempt fails transiently +- **THEN** the request stops after the configured attempt count and does not retry indefinitely + +### Requirement: Exhausted discovery is a nonfatal source cycle +The system SHALL record an exhausted GitLab discovery transport failure as a failed cycle without terminating the supervised source child. + +#### Scenario: Discovery attempts are exhausted +- **WHEN** GitLab project discovery exhausts its transport attempt budget +- **THEN** the cycle is finished as failed and the configured source loop remains alive + +#### Scenario: Query position is retained +- **WHEN** a GitLab discovery cycle fails from an exhausted transport error +- **THEN** the current query is not advanced and is eligible for the next cycle + +### Requirement: Discovery failure isolation +The system SHALL keep transport-only GitLab discovery failures separate from authentication, rate-limit, scanner, and non-GitLab source policy. + +#### Scenario: Token remains healthy after transport failure +- **WHEN** GitLab discovery fails only because of a network transport error +- **THEN** the active token is not marked invalid or rate limited + +#### Scenario: Other source behavior is unchanged +- **WHEN** a non-GitLab source encounters an API request failure +- **THEN** this GitLab-only cycle handling does not alter its existing behavior diff --git a/openspec/changes/stabilize-gitlab-discovery-retries/tasks.md b/openspec/changes/stabilize-gitlab-discovery-retries/tasks.md new file mode 100644 index 0000000..7fb7e95 --- /dev/null +++ b/openspec/changes/stabilize-gitlab-discovery-retries/tasks.md @@ -0,0 +1,17 @@ +## 1. GitLab Discovery Policy + +- [x] 1.1 Add source-configured GitLab discovery request attempts and retry delay. +- [x] 1.2 Pass the bounded settings only to GitLab project discovery requests. +- [x] 1.3 Convert exhausted GitLab discovery transport errors into failed cycles without advancing query or auth state. + +## 2. Regression Coverage + +- [x] 2.1 Test timeout-then-success and bounded exhaustion behavior. +- [x] 2.2 Test failed-cycle persistence, query retention, token health, and non-GitLab isolation. +- [x] 2.3 Run focused and broader regression suites with bytecode writes disabled. + +## 3. Validation And Rollout + +- [x] 3.1 Run strict OpenSpec validation and verify implementation against artifacts. +- [x] 3.2 Restart the managed runtime and confirm GitLab source, PostgreSQL, and pipeline health. +- [x] 3.3 Observe a GitLab discovery canary and confirm transient API errors no longer restart the source. diff --git a/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/.openspec.yaml b/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/.openspec.yaml new file mode 100644 index 0000000..f774115 --- /dev/null +++ b/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-08-20 diff --git a/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/design.md b/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/design.md new file mode 100644 index 0000000..8b64995 --- /dev/null +++ b/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/design.md @@ -0,0 +1,63 @@ +## Context + +GitLab repository scans use the shared TruffleHog Git command under the project's external supervisor, scan-slot leases, timeout enforcement, and Windows Job containment. Unlike Docker, Git commands still launch TruffleHog's embedded overseer. Twelve recent GitLab scans emitted no fatal diagnostic and no `finished scanning` marker, exited code 1, and were committed as non-retryable terminal failures after one attempt. + +## Goals / Non-Goals + +**Goals:** + +- Give the existing external runtime sole lifecycle ownership for GitLab TruffleHog processes. +- Require positive completion evidence for GitLab process success. +- Retry incomplete GitLab scans through the existing three-attempt target policy. +- Preserve partial findings and isolate behavior from other Git sources. +- Validate with a GitLab-only canary and bounded exact-signature replay. + +**Non-Goals:** + +- Change GitLab discovery API behavior. +- Change GitHub, package Git, or generic ad-hoc Git commands. +- Upgrade TruffleHog or change detector selection. +- Make clone-unavailable repositories retry forever. + +## Decisions + +### Enable local process mode only for configured GitLab scans + +`scan_git_repo` will accept an explicit lifecycle flag. GitLab source configuration enables it, causing the command to add `--local-dev`; callers that do not opt in retain the current command. + +Alternative considered: enable `--local-dev` for every Git source. Rejected because current production evidence is GitLab-specific and a narrow canary is easier to attribute and roll back. + +### Generalize explicit completion without changing existing defaults + +The diagnostic parser will accept an explicit completion-required option while preserving Docker as completion-required by default. GitLab local-process scans will enable that option. Missing completion or unexplained nonzero exit becomes retryable `command_incomplete`; a nonzero unexplained exit after completion becomes retryable `wrapper_exit`. + +Alternative considered: classify every Git RC=1 as retryable. Rejected because completion evidence distinguishes lifecycle interruption from explicit permanent diagnostics. + +### Reuse target retries and finding identity + +The existing maximum of three attempts, exponential delay, queue dispositions, finding UID assignment, and durable deduplication remain authoritative. Findings emitted before interruption are retained without marking the target complete. + +### Stage rollout and replay + +A controlled A/B run must first show that Git `--local-dev` completes a previously affected public target. Production then observes at least 100 GitLab attempts or two hours before a small exact-signature historical replay. + +## Risks / Trade-offs + +- [The flag behaves differently for Git than Docker] -> Require controlled A/B evidence and a GitLab-only canary. +- [Retries increase clone traffic] -> Keep the existing attempt cap and replay only exact-signature targets in small batches. +- [Completion logging changes upstream] -> Treat missing evidence as bounded incomplete coverage rather than success. +- [Shared Git scanner plumbing leaks behavior] -> Use an explicit source-configured flag and assert non-GitLab command isolation. +- [Partial findings repeat] -> Retain scan events while relying on existing stable finding identities downstream. + +## Migration Plan + +1. Run controlled baseline and local-process scans without exposing findings. +2. Add command, completion, retry, and isolation regression tests. +3. Enable the flag only for GitLab and restart the managed runtime. +4. Observe at least 100 GitLab attempts or two hours with no new unexplained terminal RC=1. +5. Replay one bounded exact-signature batch and verify queue disposition and deduplication. +6. Roll back by disabling the GitLab lifecycle flag and restarting the source runtime. + +## Open Questions + +- Whether the policy should later become common to all externally supervised Git scans remains a separate change. diff --git a/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/proposal.md b/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/proposal.md new file mode 100644 index 0000000..9470760 --- /dev/null +++ b/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/proposal.md @@ -0,0 +1,28 @@ +## Why + +Recent GitLab repository scans produced 12 exit-code-1 outcomes without a fatal diagnostic or `finished scanning` marker. All were treated as non-retryable terminal failures after one attempt, leaving the same process-lifecycle coverage gap previously observed and corrected for Docker scans. + +## What Changes + +- Run GitLab TruffleHog Git scans without TruffleHog's redundant embedded overseer while retaining external supervision, scan-slot control, timeout enforcement, and Windows Job containment. +- Require the normal completion marker before treating a GitLab scan process as complete. +- Classify incomplete or unexplained GitLab process exits as retryable through the existing three-attempt target policy. +- Preserve findings emitted before an incomplete exit. +- Run a GitLab-only canary before replaying a bounded exact-signature historical batch. +- Keep GitHub, package Git, and other Git scan behavior unchanged. + +## Capabilities + +### New Capabilities + +- `gitlab-scan-lifecycle`: Defines external lifecycle ownership, explicit completion, bounded retry, and controlled replay for GitLab repository scans. + +### Modified Capabilities + +None. + +## Impact + +- Affects GitLab command construction and TruffleHog diagnostic disposition in `app/scanner.py` and source plumbing in `app/console_runner.py`. +- Adds focused scanner and queue-policy regression coverage. +- Does not change GitLab discovery requests, non-GitLab commands, detector selection, keychecks, or the installed TruffleHog binary. diff --git a/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/specs/gitlab-scan-lifecycle/spec.md b/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/specs/gitlab-scan-lifecycle/spec.md new file mode 100644 index 0000000..bf2eb6f --- /dev/null +++ b/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/specs/gitlab-scan-lifecycle/spec.md @@ -0,0 +1,56 @@ +## ADDED Requirements + +### Requirement: External GitLab scan lifecycle ownership +The system SHALL bypass TruffleHog's embedded overseer for configured GitLab repository scans while retaining external supervision, scan-slot leases, timeout enforcement, output bounds, and Windows Job containment. + +#### Scenario: GitLab command construction +- **WHEN** the GitLab source constructs a TruffleHog Git command with external lifecycle enabled +- **THEN** the command includes both `--local-dev` and `--no-update` + +#### Scenario: Non-GitLab command construction +- **WHEN** another source constructs a TruffleHog Git command without external lifecycle enabled +- **THEN** the command does not gain `--local-dev` or GitLab completion policy + +### Requirement: Explicit GitLab scan completion +The system SHALL require both exit code 0 and the exact `finished scanning` marker before treating an externally managed GitLab TruffleHog process as complete. + +#### Scenario: Normal GitLab completion +- **WHEN** the process exits code 0 after emitting `finished scanning` +- **THEN** completion metadata is recorded and no lifecycle error is added + +#### Scenario: Missing GitLab completion marker +- **WHEN** the process exits without emitting `finished scanning` +- **THEN** the result is classified as retryable `command_incomplete` and is not treated as complete + +#### Scenario: Nonzero exit after completion marker +- **WHEN** the process emits `finished scanning` and exits nonzero without a more specific fatal diagnostic +- **THEN** the result is classified as retryable `wrapper_exit` + +### Requirement: Bounded retry for incomplete GitLab scans +The system SHALL route incomplete GitLab lifecycle outcomes through the existing bounded target retry policy. + +#### Scenario: Retry remains available +- **WHEN** an incomplete GitLab scan occurs before the configured maximum target attempt +- **THEN** the queue defers the target using the configured retry delay + +#### Scenario: Attempt limit is reached +- **WHEN** an incomplete GitLab scan occurs at the maximum target attempt +- **THEN** the queue records a terminal failed target without an unbounded loop + +### Requirement: Partial GitLab finding preservation +The system SHALL retain findings emitted before an incomplete GitLab process exit without representing the repository as fully scanned. + +#### Scenario: Findings precede incomplete exit +- **WHEN** TruffleHog emits findings and then exits before completion is confirmed +- **THEN** those findings remain durable while the target receives retryable incomplete disposition + +### Requirement: Controlled GitLab lifecycle rollout +The system SHALL enable and replay GitLab lifecycle behavior only through bounded, observable stages. + +#### Scenario: Canary has not passed +- **WHEN** the GitLab lifecycle canary has not reached its observation threshold +- **THEN** historical terminal failures are not mass-requeued + +#### Scenario: Canary has passed +- **WHEN** the canary is healthy and a bounded exact-signature batch is selected +- **THEN** only targets in that batch are returned to the pending queue diff --git a/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/tasks.md b/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/tasks.md new file mode 100644 index 0000000..01e6a23 --- /dev/null +++ b/openspec/changes/stabilize-gitlab-trufflehog-lifecycle/tasks.md @@ -0,0 +1,19 @@ +## 1. GitLab Lifecycle Policy + +- [x] 1.1 Run controlled baseline and `--local-dev` scans for previously affected public GitLab targets. +- [x] 1.2 Add source-configured external lifecycle command construction only for GitLab scans. +- [x] 1.3 Generalize explicit completion parsing and classify missing completion or wrapper exit as retryable. +- [x] 1.4 Confirm partial findings use the existing three-attempt queue policy and stable identities. + +## 2. Regression Coverage + +- [x] 2.1 Add command tests proving GitLab receives `--local-dev` while other Git sources do not. +- [x] 2.2 Add completion, incomplete exit, wrapper exit, partial finding, and queue exhaustion tests. +- [x] 2.3 Run focused and broader regression suites with bytecode writes disabled. + +## 3. Validation And Rollout + +- [x] 3.1 Run strict OpenSpec validation and verify implementation against artifacts. +- [x] 3.2 Restart the managed runtime and confirm GitLab command, source, scan-slot, PostgreSQL, and pipeline health. +- [x] 3.3 Observe at least 100 GitLab attempts or two hours before historical replay. +- [x] 3.4 Requeue one bounded exact-signature historical batch and verify completion, deduplication, and queue drain. diff --git a/openspec/changes/support-fifty-remote-assignments/.openspec.yaml b/openspec/changes/support-fifty-remote-assignments/.openspec.yaml new file mode 100644 index 0000000..1aca8b9 --- /dev/null +++ b/openspec/changes/support-fifty-remote-assignments/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-30 diff --git a/openspec/changes/support-fifty-remote-assignments/design.md b/openspec/changes/support-fifty-remote-assignments/design.md new file mode 100644 index 0000000..a0e050a --- /dev/null +++ b/openspec/changes/support-fifty-remote-assignments/design.md @@ -0,0 +1,70 @@ +## Context + +Remote admission uses `result_bundle_max_event_bytes` for both the hard upload limit and the initial bundle reservation, and reserves twice that value for projection. With production's 64 MiB hard limit, one unresolved assignment consumes 64 MiB of bundle capacity and 128 MiB of projection capacity. Historical production data shows much smaller payloads, but the hard limit must remain available for outliers. + +The production projector is singleton-owned. This allows one projection job at a time to expand from its baseline reservation into the protected projection headroom without allowing many simultaneous expansions to exhaust the backlog. + +## Goals / Non-Goals + +**Goals:** +- Admit up to 50 unresolved remote assignments atomically across all users. +- Charge 2 MiB baseline bundle and projection reservations per remote assignment. +- Preserve the 64 MiB hard bundle limit and accept larger-than-baseline valid results with capacity backpressure. +- Keep all capacity changes durable, replay-safe, and recoverable. + +**Non-Goals:** +- Change worker protocol, bundle format, assignment TTLs, or local scanner parallelism. +- Preallocate RAM or disk for logical capacity counters. +- Remove per-user assignment caps or disk-free safety checks. + +## Decisions + +### Persist a distinct bundle reservation + +Add `reserved_bundle_bytes` to result reservations. `declared_bundle_bytes` remains the immutable hard upload bound; `reserved_bundle_bytes` is the amount charged to `pipeline_capacity.bundle_bytes`. Existing rows are backfilled with their declared value so migration does not change outstanding accounting. + +Alternative considered: derive the reservation from current configuration during release. This is rejected because configuration may change while an assignment is unresolved. + +### Use a 2 MiB baseline for both bundle and projection accounting + +Add `remote_assignment_reserve_bytes`, set to 2 MiB in production. Remote admission charges this value for `reserved_bundle_bytes` and `reserved_projection_bytes`; local reservations retain their existing worst-case behavior. + +Candidate reservations remain bounded by the execution snapshot. Production keycheck item and byte capacity will be sized for 50 such reservations. + +### Expand bundle accounting when actual size is known + +Bundle acceptance atomically grows `reserved_bundle_bytes` to `actual_bytes` before publishing the acceptance receipt. Growth must remain within configured bundle capacity. If capacity is unavailable, acceptance remains unresolved and the durable worker retries the same upload; replay cannot double-charge capacity. + +### Expand projection accounting before append + +After serialization determines the exact aggregate projection bytes, the singleton projector atomically grows the leased job's `capacity_bytes` when required. Growth may use configured projection headroom. If unavailable, the job returns to pending without quarantine or partial append and retries later. + +Alternative considered: treat 2 MiB as a hard projection limit. This is rejected because a valid bundle below the 64 MiB hard limit must not become a deterministic projection failure merely for exceeding its baseline reservation. + +### Enforce a global unresolved-assignment cap + +Add `remote_assignment_max_active`, set to 50. Admission counts unresolved remote reservations while holding the same transactional locks used for per-user quota enforcement. The 51st claim receives normal no-work backpressure. Existing per-user caps still apply, so the effective limit is the minimum of global capacity, global assignment cap, and the user's cap. + +### Production capacity values + +Keep the 64 MiB hard result limit, 512 MiB bundle capacity, 256 MiB projection capacity, and 128 MiB projection headroom. Fifty 2 MiB baselines consume 100 MiB on each byte axis. Increase keycheck capacity to at least 100,000 items and 100 MiB; use 131,072 items and 128 MiB for bounded operational headroom. Set the intended production user's cap to 50. + +## Risks / Trade-offs + +- [Several near-maximum bundles arrive together] → Bundle acceptance atomically expands capacity and later uploads retry without losing their local durable bundle. +- [Projection exceeds its baseline] → The singleton projector expands actual capacity before any append and defers without quarantine when headroom is busy. +- [Configuration changes with unresolved work] → Every reservation persists charged bundle/projection values; reconciliation sums persisted values rather than current defaults. +- [Fifty remote scans overload the small server on result return] → Scanning remains worker-side; API ingestion and projection retain byte, item, disk-free, singleton, and backpressure limits. +- [Deployment rollback sees migrated rows] → The additive column is backfilled with prior declared values and remains compatible with restored conservative configuration; rollback code must be from a build that understands the migrated schema. + +## Migration Plan + +1. Add and backfill `reserved_bundle_bytes`, then reconcile `pipeline_capacity` from durable rows. +2. Deploy code with the new configuration fields at conservative values. +3. Apply production capacity values and set the selected worker user cap to 50 through typed administration. +4. Verify 50 baseline admissions in a bounded test, then reconcile all reservations, bundles, projection jobs, and capacity to zero. +5. Roll back by setting the global and user caps to the prior value, draining unresolved work, and restoring conservative reservation settings; do not remove the additive column. + +## Open Questions + +None. diff --git a/openspec/changes/support-fifty-remote-assignments/proposal.md b/openspec/changes/support-fifty-remote-assignments/proposal.md new file mode 100644 index 0000000..fd893df --- /dev/null +++ b/openspec/changes/support-fifty-remote-assignments/proposal.md @@ -0,0 +1,22 @@ +## Why + +Remote admission currently reserves each assignment at the absolute 64 MiB bundle limit and 128 MiB projection limit. Production results are much smaller (239 KiB maximum bundle and 273 KiB maximum projection across 6,733 observed remote results), so worst-case reservation blocks useful concurrency long before physical resources are under pressure. + +## What Changes + +- Keep the 64 MiB hard limit for an individual result bundle. +- Reserve 2 MiB of bundle and projection capacity when issuing a remote assignment, independently of the hard result limit. +- Permit a valid result or projection larger than 2 MiB to atomically acquire additional capacity based on its actual size, with retryable backpressure when capacity is temporarily unavailable. +- Add an atomic global limit of 50 unresolved remote assignments in addition to existing per-user caps. +- Size production pipeline and keycheck accounting so 50 baseline reservations fit without consuming emergency headroom. + +## Capabilities + +### New Capabilities +- `remote-assignment-capacity`: Bounded global remote concurrency, separate hard result limits and baseline reservations, and actual-size overflow accounting. + +### Modified Capabilities + +## Impact + +This affects remote assignment admission and result settlement in `worker_assignment.py` and `scanner_db.py`, projection capacity handling in `jsonl_projector.py`, runtime configuration validation, PostgreSQL schema migration, production capacity settings, and integration tests. Worker protocol and bundle format remain unchanged. diff --git a/openspec/changes/support-fifty-remote-assignments/specs/remote-assignment-capacity/spec.md b/openspec/changes/support-fifty-remote-assignments/specs/remote-assignment-capacity/spec.md new file mode 100644 index 0000000..8cbfe76 --- /dev/null +++ b/openspec/changes/support-fifty-remote-assignments/specs/remote-assignment-capacity/spec.md @@ -0,0 +1,60 @@ +## ADDED Requirements + +### Requirement: Hard result limit is separate from baseline reservation +The system SHALL retain the configured hard result-bundle byte limit while charging each new remote assignment a separately configured 2 MiB baseline bundle reservation and 2 MiB baseline projection reservation. + +#### Scenario: Remote assignment is issued +- **WHEN** an eligible worker claims an assignment with available global and user quota +- **THEN** the assignment advertises the unchanged hard bundle limit and pipeline accounting charges only the configured baseline bundle and projection reservations + +#### Scenario: Local assignment is issued +- **WHEN** the server admits a local scan +- **THEN** its existing worst-case capacity reservation behavior remains unchanged + +### Requirement: Actual bundle size expands accounting safely +The system SHALL atomically grow a remote reservation's charged bundle bytes to the validated actual bundle size before issuing an acceptance receipt. + +#### Scenario: Bundle exceeds baseline with capacity available +- **WHEN** a valid bundle is larger than 2 MiB but does not exceed the hard result limit and aggregate bundle capacity is available +- **THEN** the system increases the persisted reservation and pipeline capacity charge exactly once and accepts the bundle + +#### Scenario: Bundle exceeds baseline without capacity available +- **WHEN** a valid bundle requires additional bundle capacity that is temporarily unavailable +- **THEN** the system does not issue an acceptance receipt and permits the worker to retry the identical durable upload later + +### Requirement: Actual projection size expands accounting safely +The system SHALL grow a leased projection job's capacity to its serialized aggregate byte size before appending output. + +#### Scenario: Projection exceeds baseline with headroom available +- **WHEN** serialization exceeds the baseline projection reservation and configured projection capacity is available +- **THEN** the system atomically increases the job and pipeline charge before appending output + +#### Scenario: Projection exceeds baseline without headroom available +- **WHEN** the additional projection capacity is temporarily unavailable +- **THEN** the system returns the untouched job to pending without quarantine and retries it later + +### Requirement: Global remote assignment limit +The system SHALL enforce a configured global maximum of 50 unresolved remote assignments atomically in addition to each user's assignment cap. + +#### Scenario: Fiftieth assignment is admitted +- **WHEN** 49 unresolved remote assignments exist and all other admission constraints are open +- **THEN** one additional assignment is admitted + +#### Scenario: Fifty-first assignment is refused +- **WHEN** 50 unresolved remote assignments exist +- **THEN** another claim receives normal no-work backpressure without creating a reservation + +#### Scenario: Assignment resolves +- **WHEN** an unresolved assignment receives a terminal resolution or expires +- **THEN** its global slot becomes available for a subsequent claim + +### Requirement: Capacity survives migration and reconciliation +The system MUST preserve exact capacity accounting for reservations created before and after deployment. + +#### Scenario: Existing reservation is migrated +- **WHEN** the additive schema migration encounters a reservation without a distinct bundle reservation value +- **THEN** it backfills the value from the existing declared bundle bytes + +#### Scenario: Capacity is reconciled +- **WHEN** pipeline capacity is rebuilt from durable state +- **THEN** it sums each reservation's persisted bundle, projection, candidate, job, and quarantine charges exactly once diff --git a/openspec/changes/support-fifty-remote-assignments/tasks.md b/openspec/changes/support-fifty-remote-assignments/tasks.md new file mode 100644 index 0000000..ddc6178 --- /dev/null +++ b/openspec/changes/support-fifty-remote-assignments/tasks.md @@ -0,0 +1,23 @@ +## 1. Capacity Model + +- [x] 1.1 Add configuration validation for the 2 MiB remote baseline reservation and global active-assignment limit +- [x] 1.2 Add and backfill persisted remote bundle reservation bytes in the runtime-safety schema +- [x] 1.3 Charge persisted baseline bundle and projection bytes during remote admission while preserving local admission behavior +- [x] 1.4 Enforce the global unresolved remote-assignment limit atomically with existing per-user quota + +## 2. Actual-Size Expansion + +- [x] 2.1 Expand bundle capacity atomically to validated actual bytes before remote acceptance +- [x] 2.2 Expand projection-job capacity atomically before append and defer without quarantine when capacity is unavailable +- [x] 2.3 Update refunds, cleanup, quarantine, and reconciliation to use persisted charged bytes + +## 3. Production Configuration + +- [x] 3.1 Configure a 2 MiB baseline, global limit 50, and keycheck capacity for 50 assignments +- [x] 3.2 Update operator documentation with the distinct hard-limit, baseline-reservation, and backpressure semantics + +## 4. Verification + +- [x] 4.1 Add unit and PostgreSQL integration coverage for baseline admission, global quota, bundle expansion, projection expansion, retries, migration, and reconciliation +- [x] 4.2 Run targeted worker, pipeline, migration, and packaged-worker tests +- [x] 4.3 Validate a bounded 50-assignment production candidate and reconcile capacity and pipeline debt before rollout diff --git a/openspec/changes/support-fifty-remote-assignments/validation.md b/openspec/changes/support-fifty-remote-assignments/validation.md new file mode 100644 index 0000000..eee587a --- /dev/null +++ b/openspec/changes/support-fifty-remote-assignments/validation.md @@ -0,0 +1,27 @@ +## Production Validation + +Validated and deployed on `sec` on 2026-09-30 through the fail-closed +`deploy_capacity50.ps1` workflow. + +- Baseline runtime image: `sha256:46f1cf1b92d1a7d93d06f690309d8c7eca45f64dc45ca310861c70fb419bb035`. +- Deployed runtime image: `sha256:17b21c62f559220cd60e7972826fb439e5538c7c9443c4fdd1c3615751de5f8d`. +- Active config SHA-256: `2cf2ac66b58f1b39c1e45dd0d562b9b6375be42fa2289cc9bbcbd1a310951ee7`. +- Rollback tag: `truf-local:runtime-pre-capacity50-20260930`. +- The additive `20260930_33_remote_assignment_capacity` migration and + `reserved_bundle_bytes` backfill completed with zero null rows. +- Runtime health was `ACTIVE`/`READY`; runtime and edge had zero restarts and no OOM. +- Runtime control returned to `normal`; all bundle, projection, keycheck, + quarantine, and unresolved-assignment counters were zero. +- The 64 MiB hard limit, 2 MiB baseline, global cap 50, 131072 keycheck items, + and 128 MiB keycheck bytes were active. +- A production test user's cap was changed from 2 to 50 through typed admin, + verified, and restored to 2. No test user retains the expanded cap. +- The active Windows worker contacted the new Worker API after rollout. +- The exact 50th-admitted/51st-refused behavior remains covered by the real + PostgreSQL integration test; production was not flooded with synthetic work. + +The live run also exercised rollback before cutover and after a rejected +migration precondition. Both restored the original image/config, healthy runtime +and edge, normal control state, and zero pipeline debt. The final workflow now +quiesces singleton pipeline workers and releases their exact fenced leases before +the schema migration. diff --git a/openspec/config.yaml b/openspec/config.yaml new file mode 100644 index 0000000..392946c --- /dev/null +++ b/openspec/config.yaml @@ -0,0 +1,20 @@ +schema: spec-driven + +# Project context (optional) +# This is shown to AI when creating artifacts. +# Add your tech stack, conventions, style guides, domain knowledge, etc. +# Example: +# context: | +# Tech stack: TypeScript, React, Node.js +# We use conventional commits +# Domain: e-commerce platform + +# Per-artifact rules (optional) +# Add custom rules for specific artifacts. +# Example: +# rules: +# proposal: +# - Keep proposals under 500 words +# - Always include a "Non-goals" section +# tasks: +# - Break tasks into chunks of max 2 hours diff --git a/openspec/parking-lot.md b/openspec/parking-lot.md new file mode 100644 index 0000000..50a04fc --- /dev/null +++ b/openspec/parking-lot.md @@ -0,0 +1,96 @@ +# Runtime Reliability Parking Lot + +Last reviewed: 2026-08-27 + +This file records observed follow-up work that is intentionally outside completed changes. Evidence must be refreshed before opening a new change. + +## P1: GitLab discovery exits on one transient API timeout + +- Status: implemented by `stabilize-gitlab-discovery-retries`; all 9 tasks complete and the change is ready for archive. +- Evidence: four GitLab source restarts during the current runtime window matched four `GET /api/v4/projects` read timeouts. +- Current behavior: direct API calls get one attempt; an exhausted discovery request escapes the source cycle and exits the supervised child. +- Impact: avoidable source churn and discovery delay. The saved query position is retained, so no confirmed target loss was observed. +- Candidate direction: bounded direct-request retries plus a source-cycle network disposition that does not terminate the child. +- Validation: inject timeout-then-success and timeout-exhaustion cases; verify query state, auth state, backoff, and no source restart. + +## P1: GitLab scans have an incomplete RC=1 coverage gap + +- Status: resolved by `stabilize-gitlab-trufflehog-lifecycle`; the production canary and bounded historical replay completed successfully. +- Evidence: 12 of 175 recent GitLab attempts exited RC=1 without a fatal diagnostic or `finished scanning` marker. +- Resolution evidence: 120 process attempts passed the rollout gate with 117 RC=0 completion markers and zero recurrence of the old signature. A three-target exact-signature replay completed cleanly on the first new attempt for every target; a further 30-minute soak reached 142 attempts and 139 RC=0 completion markers with the old signature still at zero. A later three-target sample produced two clean completions and one explicit retryable `trufflehog` deferral instead of the old terminal classification. +- Current behavior: all 12 were classified as non-retryable `command_exit` and became terminal failures after one attempt. +- Impact: these repositories have no confirmed complete scan. +- Candidate direction: run controlled Git `--local-dev` A/B tests, then consider extending explicit completion and bounded retry semantics beyond Docker. +- Validation: canary completion-marker rate, unexplained RC=1 rate, partial-finding preservation, retry exhaustion, and non-Git source isolation. + +## P1: Runtime-wide external Windows termination has no autonomous recovery + +- Evidence: at `2026-08-20 19:59:40 MSK`, PostgreSQL and unrelated source children were terminated together with Windows exception `0x40010004`; PostgreSQL shut down after its startup process also failed, and the supervisor was no longer alive to recover it. +- Impact: the pipeline remained offline until the next authority-checked `start_runtime.ps1` launch. +- Current state: PostgreSQL completed WAL recovery without corruption, all core sources and pipeline children restarted, and the post-recovery soak showed no child restarts or database failures. +- Candidate direction: run the authority-checked runtime under an external Windows Service or Task Scheduler watchdog that can restart a dead supervisor without weakening singleton or cluster-identity checks. + +## Resolved: Updated core targets were permanently suppressed + +- Status: resolved by `rescan-updated-core-targets`; the change is ready for archive. +- Evidence: the first production canary admitted three GitHub and three GitLab updated targets across six separate cycles, never exceeding the configured one-per-cycle cap. All six scans completed cleanly with exact remote revision snapshots, zero errors, zero findings, and applied queue completion. +- Isolation: DockerHub recorded no revision observations or updated admissions and continues to use digest identity. HuggingFace established its newest-modified revision baseline without an updated-target surge. +- Rollout decision: retain the cap at one per source cycle and the 24-hour per-target cooldown until longer-running yield data justifies a change. + +## Resolved: Core discovery omitted an exact OpenAI query + +- Status: resolved by `restore-openai-discovery-coverage`; the change is ready for archive. +- Evidence: the first bounded cycles fetched 44 GitHub and 42 GitLab results, admitted one changed GitLab target, and discovered 18 new DockerHub repository identities under the exact `openai` query. Source bounds remained one GitHub/GitLab page with at most five claims and two DockerHub pages with at most 20 claims. +- Funnel: all 18 DockerHub admissions reached terminal queue state (16 done, two explicit registry-access failures). Their scans produced 347 findings, including 53 OpenAI finding rows representing 24 distinct credentials. All 24 were genuinely new and received an authoritative API result: 20 quota/no-balance, four invalid/revoked, and zero alive. +- Durability: all 18 scan projections and 53 OpenAI keycheck projections completed with released capacity; the OpenAI candidate and publication backlogs drained to zero. +- Rollout decision: retain the bounded rotating query. It restored missing supply but did not produce a usable credential in the initial sample, so broader limits are not justified yet. + +## Resolved: Bounded OpenAI ecosystem keyword wave + +- Status: implemented by `expand-openai-ecosystem-discovery`; all nine source-specific queries completed one bounded production cycle. +- Git evidence: the three GitHub queries fetched 6, 0, and 3 results without new or updated admissions. GitLab `openai-api` and `openai-agents` admitted nothing; `librechat` admitted one bounded updated target whose scan completed cleanly without findings. +- Docker evidence: `librechat`, `lobechat`, and `openai-proxy` admitted 20, 19, and 16 immutable-digest targets. All 55 reached terminal queue state: 50 done and five explicit `auth_invalid` failures. Exact query scans produced 597 finding rows. +- Credential funnel: `librechat` produced five genuinely new credentials (GCP two, Gemini one, GitHub one, OpenAI one); `lobechat` produced two (AWS one, GitHub one); `openai-proxy` produced none. All seven received authoritative API outcomes and none was alive or usable. The OpenAI credential was explicitly invalid/revoked. +- Durability: every exact cohort candidate completed and all 147 cohort scan/keycheck projection jobs completed with released capacity. +- Rollout decision: retain the bounded wave provisionally without increasing any page, claim, worker, or global concurrency limit. Reassess after the next natural rotation; remove zero-yield terms if they remain empty rather than expanding this wave. + +## P2: Heavy Docker images can exceed bounded resources + +- Evidence: in the 197-attempt Docker canary, six scans reached the 600-second timeout, three hit explicit Windows `VirtualAlloc`/paging limits, and one had a mixed stream/timeout outcome. +- Current behavior: failures are explicit, partial findings are retained, and retryable outcomes use the bounded queue policy. The old unexplained RC=1 signature remained at zero. +- Impact: a small set of large images does not receive confirmed full coverage. +- Candidate direction: a separate heavy-image lane with lower concurrency and a larger target budget; do not raise global limits without host-level measurements. +- Validation: completion gain, host commit usage, scan-slot fairness, source throughput, and retry amplification. + +## P2: Historical Docker replay remains intentionally gradual + +- Evidence: the first exact-signature batch of 10 produced seven completed targets, one terminal registry-access failure, and two explicit deferred retries, with no old RC=1 recurrence. A second six-target batch produced four completed targets, one terminal registry-access failure, and one timeout deferral; all 41 emitted finding UIDs were globally unique and the timeout retained 35 partial findings. +- Current behavior: the historical terminal set has not been mass-requeued. +- Candidate direction: increase exact-signature batches conservatively while monitoring registry traffic and heavy-image failures. + +## P2: GitHub manifests are not mined for Docker image references + +- Evidence: 1,954 retained `github_archive_files` artifacts contained 158 unique image references, but only about 48 plausible namespaced Docker Hub repositories were incremental after filtering bases, local names, and existing queue identities. +- Current behavior: changed compose/workflow files are scanned for secrets only; `image:`, `FROM`, and `docker://` references do not feed Docker resolution. The bounded `github_archive_files` source is currently disabled. +- Impact: a small but potentially fresher source of Docker repositories is omitted. The measured one-time opportunity is much smaller than the existing unresolved Docker repository backlog. +- Candidate direction: after Docker resolver throughput is healthy, parse exact image references from changed compose, Kubernetes/Helm, workflow, and Dockerfile artifacts; resolve concrete tags directly to immutable platform digests and filter common base images. +- Validation: incremental repositories and digests, API calls per admitted target, source freshness, strict usable-key yield, duplicate rate, and GitHub quota cost. + +## P2: Background supervisor launch nonce can parse as an option + +- Evidence: one authority-checked launch on 2026-08-25 generated a URL-safe nonce beginning with an option-like prefix; `argparse` reported `argument --launch-nonce: expected one argument`. An immediate canonical retry succeeded. +- Impact: a rare transient startup refusal; no PostgreSQL or queue mutation occurred. +- Candidate direction: emit the hidden value as `--launch-nonce=` and add a command-construction test with a leading-hyphen nonce. + +## P3: GitLab removed/private repository churn + +- Evidence: 10 recent clone-error scan events came from four targets; three targets exhausted all three retries. +- Current behavior: clone preparation errors are retryable even when a repository appears deleted, private, or deletion-scheduled. +- Impact: bounded but avoidable repeated work. +- Candidate direction: distinguish permanent not-found/access outcomes from transient clone transport failures before retrying. + +## Needs Fresh Validation: Projector recovery/index path + +- Earlier investigation suggested a rollback/missing-index weakness in projector recovery. +- Current state is healthy: ingester and projector report `ready`, with no active lease error. +- Before creating a change, reproduce or recover the original query/error evidence and determine whether the issue still exists. diff --git a/pytest.ini b/pytest.ini new file mode 100644 index 0000000..5ee6477 --- /dev/null +++ b/pytest.ini @@ -0,0 +1,2 @@ +[pytest] +testpaths = tests diff --git a/start_core_runtime.ps1 b/start_core_runtime.ps1 new file mode 100644 index 0000000..1279687 --- /dev/null +++ b/start_core_runtime.ps1 @@ -0,0 +1,28 @@ +$ErrorActionPreference = 'Stop' +throw 'Docker development copy: runtime control is disabled until container isolation is ready. See DOCKER_MIGRATION.md.' + +$Root = Split-Path -Parent $MyInvocation.MyCommand.Path +$RuntimeBootstrap = Join-Path $Root 'app\runtime_bootstrap.py' +$Supervisor = (Resolve-Path -LiteralPath (Join-Path $Root 'app\supervisor.py')).Path +$Config = Join-Path $Root 'app\config.yaml' +$FreezeCounters = Join-Path $Root 'start_freeze_counters.ps1' +$CoreSources = 'gitlab,dockerhub,huggingface' + +try { + & $FreezeCounters +} catch { + Write-Warning "Freeze diagnostics did not start: $($_.Exception.Message)" +} + +python -I -S -B $RuntimeBootstrap postgres-runtime -- verify --config $Config +if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } + +python -I -S -B $RuntimeBootstrap supervisor -- ` + --runtime-bootstrap-entrypoint $Supervisor ` + --config $Config ` + --sources $CoreSources ` + --background ` + --no-dashboard ` + --with-postgres +$SupervisorExit = $LASTEXITCODE +exit $SupervisorExit diff --git a/start_freeze_counters.ps1 b/start_freeze_counters.ps1 new file mode 100644 index 0000000..908bcc2 --- /dev/null +++ b/start_freeze_counters.ps1 @@ -0,0 +1,122 @@ +param( + [string]$Name = 'TrufFreezeCounters', + [string]$OutputRoot = 'D:\truf\runtime\freeze-diagnostics' +) + +$ErrorActionPreference = 'Stop' +throw 'Docker development copy: runtime control is disabled until container isolation is ready. See DOCKER_MIGRATION.md.' + +function Start-FallbackMonitor { + $monitor = Join-Path $PSScriptRoot 'monitor_runtime_lag.ps1' + if (-not (Test-Path -LiteralPath $monitor)) { + throw 'Fallback freeze monitor script is absent' + } + Start-Process ` + -FilePath "$env:SystemRoot\System32\WindowsPowerShell\v1.0\powershell.exe" ` + -ArgumentList @( + '-NoProfile', '-ExecutionPolicy', 'Bypass', '-File', $monitor, + '-OutputPath', (Join-Path $OutputRoot 'freeze-fallback-v3.csv'), + '-IntervalSeconds', '5' + ) ` + -WindowStyle Hidden | Out-Null +} + +if (-not (Test-Path -LiteralPath $OutputRoot)) { + New-Item -ItemType Directory -Path $OutputRoot | Out-Null +} + +function Read-CounterTable([string]$Language) { + $values = (Get-ItemProperty -LiteralPath "HKLM:\SOFTWARE\Microsoft\Windows NT\CurrentVersion\Perflib\$Language").Counter + $result = @{} + for ($index = 0; $index -lt $values.Count; $index += 2) { + $result[[int]$values[$index]] = $values[$index + 1] + } + return $result +} + +$english = Read-CounterTable '009' +$localized = Read-CounterTable '019' +$indexByEnglishName = @{} +foreach ($entry in $english.GetEnumerator()) { + $indexByEnglishName[[string]$entry.Value] = [int]$entry.Key +} + +function Counter-Path([string]$Object, [string]$Counter, [string]$Instance = '') { + if (-not $indexByEnglishName.ContainsKey($Object) -or -not $indexByEnglishName.ContainsKey($Counter)) { + return $null + } + $objectName = $localized[$indexByEnglishName[$Object]] + $counterName = $localized[$indexByEnglishName[$Counter]] + if (-not $objectName -or -not $counterName) { + return $null + } + if ($Instance) { + return "\$objectName($Instance)\$counterName" + } + return "\$objectName\$counterName" +} + +$requested = @( + @('Processor', '% Processor Time', '_Total'), + @('System', 'Processor Queue Length', ''), + @('System', 'Context Switches/sec', ''), + @('System', 'Processes', ''), + @('System', 'Threads', ''), + @('Memory', 'Available MBytes', ''), + @('Memory', '% Committed Bytes In Use', ''), + @('Memory', 'Pages Input/sec', ''), + @('Memory', 'Page Reads/sec', ''), + @('Memory', 'Pool Nonpaged Bytes', ''), + @('Paging File', '% Usage', '_Total'), + @('PhysicalDisk', 'Current Disk Queue Length', '*'), + @('PhysicalDisk', 'Avg. Disk sec/Transfer', '*'), + @('PhysicalDisk', 'Disk Read Bytes/sec', '*'), + @('PhysicalDisk', 'Disk Write Bytes/sec', '*'), + @('PhysicalDisk', 'Disk Reads/sec', '*'), + @('PhysicalDisk', 'Disk Writes/sec', '*'), + @('Process', '% Processor Time', '*'), + @('Process', 'Private Bytes', '*'), + @('Process', 'Working Set', '*'), + @('Process', 'Handle Count', '*'), + @('Process', 'Thread Count', '*'), + @('Process', 'IO Read Bytes/sec', '*'), + @('Process', 'IO Write Bytes/sec', '*'), + @('Process', 'Page Faults/sec', '*'), + @('GPU Engine', 'Utilization Percentage', '*'), + @('GPU Process Memory', 'Local Usage', '*'), + @('GPU Process Memory', 'Non Local Usage', '*') +) + +$counters = @( + foreach ($definition in $requested) { + $path = Counter-Path $definition[0] $definition[1] $definition[2] + if ($path) { $path } + } +) +if (-not $counters.Count) { + throw 'No localized performance counters could be resolved' +} + +& logman.exe query $Name *> $null +$exists = $LASTEXITCODE -eq 0 +if (-not $exists) { + & logman.exe create counter $Name ` + -f bincirc ` + -max 512 ` + -si 00:00:01 ` + -o (Join-Path $OutputRoot 'freeze-counters') ` + -c $counters *> $null + if ($LASTEXITCODE -ne 0) { + Start-FallbackMonitor + return + } +} + +& logman.exe start $Name *> $null +if ($LASTEXITCODE -ne 0) { + & logman.exe query $Name *> $null + if ($LASTEXITCODE -ne 0) { + Start-FallbackMonitor + return + } +} diff --git a/start_runtime.ps1 b/start_runtime.ps1 new file mode 100644 index 0000000..584a5ba --- /dev/null +++ b/start_runtime.ps1 @@ -0,0 +1,21 @@ +$ErrorActionPreference = 'Stop' +throw 'Docker development copy: runtime control is disabled until container isolation is ready. See DOCKER_MIGRATION.md.' + +$Root = Split-Path -Parent $MyInvocation.MyCommand.Path +$RuntimeBootstrap = Join-Path $Root 'app\runtime_bootstrap.py' +$Supervisor = (Resolve-Path -LiteralPath (Join-Path $Root 'app\supervisor.py')).Path +$Config = Join-Path $Root 'app\config.yaml' +$FreezeCounters = Join-Path $Root 'start_freeze_counters.ps1' + +try { + & $FreezeCounters +} catch { + Write-Warning "Freeze diagnostics did not start: $($_.Exception.Message)" +} + +python -I -S -B $RuntimeBootstrap postgres-runtime -- verify --config $Config +if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } + +python -I -S -B $RuntimeBootstrap supervisor -- --runtime-bootstrap-entrypoint $Supervisor --config $Config --background --no-dashboard --with-postgres +$SupervisorExit = $LASTEXITCODE +exit $SupervisorExit diff --git a/stop_runtime.ps1 b/stop_runtime.ps1 new file mode 100644 index 0000000..36dc834 --- /dev/null +++ b/stop_runtime.ps1 @@ -0,0 +1,11 @@ +$ErrorActionPreference = 'Stop' +throw 'Docker development copy: runtime control is disabled until container isolation is ready. See DOCKER_MIGRATION.md.' + +$Root = Split-Path -Parent $MyInvocation.MyCommand.Path +$RuntimeBootstrap = Join-Path $Root 'app\runtime_bootstrap.py' +$Supervisor = (Resolve-Path -LiteralPath (Join-Path $Root 'app\supervisor.py')).Path +$Config = Join-Path $Root 'app\config.yaml' + +python -I -S -B $RuntimeBootstrap supervisor -- --runtime-bootstrap-entrypoint $Supervisor --config $Config --stop-background --with-postgres +$SupervisorExit = $LASTEXITCODE +exit $SupervisorExit diff --git a/tests/container_e2e.py b/tests/container_e2e.py new file mode 100644 index 0000000..5973ce1 --- /dev/null +++ b/tests/container_e2e.py @@ -0,0 +1,3515 @@ +"""Offline, disposable-volume container E2E driver. Never run against user data. + +Install this file in the TEST image at /opt/truf/tests/container_e2e.py. All +commands use `python -I -S -B /opt/truf/tests/container_e2e.py MODE`. The image +must run as 10001:10001, with a fresh provisioned /data named volume, read-only +application code, network_mode: none, and /run/truf as a 0700 UID-10001 tmpfs. + +Integration order (the provisioner creates directories, not a database): + 1. prepare: copy the full Linux profile, commit the known-fake Git fixture, + and check the native TruffleHog Git command. No database is opened. + 2. Start the REAL entrypoint: python -I -S -B app/container_runtime.py run + --config /data/config/e2e.yaml. `run` performs real first initialization. + Explicit `initialize --config /data/config/e2e.yaml` before run is also OK. + 3. assert-pipeline --timeout 180: wait for the real scanner/ingester/projector + and save private artifact identities in /data/fixture/result.json. + 4. keycheck-fixture --timeout 180 (optional): run the actual OpenAI provider + through child_bootstrap, replacing ONLY requests.Session.request. + 5. Recreate the container with the SAME named volume, config, and isolation. + Do not prepare again. assert-persisted waits for the new one-shot GitHub + pass and requires unchanged output counts, identities, offsets and bytes. + 6. If step 4 ran, keycheck-fixture again must make zero HTTP requests and + produce no duplicate keycheck results or projections. + 7. assert-remote-recovery proves cross-device quota serialization, fixed + expiry, recovery, and stale fencing without contacting a source. + 8. prepare-remote-transport streams one canonical synthetic bundle through + the Worker API, reconciles publication/ready failures, and waits for the + real ingester and projector to durably consume and remove it. + 9. prepare-dockerhub-canary proves a bounded synthetic DockerHub discovery, + protocol-2 result, expiry, replay, and drain flow using test-only provider + transport substitution. + 10. Recreate the container again. assert-remote-transport-replay and + finish-dockerhub-canary prove receipts survive restart, former expiry, + and spool cleanup. + +Only /data/fixture and the new /data/config/e2e.yaml are written by this driver; +the real application owns its database and output writes. stdout is one JSON +object containing counts and hashes, including on failure. No child output, +credentials, findings, control tokens, or exception messages are printed. +""" + +import argparse +from contextlib import redirect_stderr, redirect_stdout +import copy +import hashlib +import json +import logging +import math +import os +from pathlib import Path +import re +import runpy +import socket +import stat +import subprocess +import sys +import tempfile +import time +from urllib.parse import quote + + +APP = Path(__file__).resolve().parents[1] / 'app' +DATA = Path('/data') +FIXTURE = DATA / 'fixture' +CONFIG = DATA / 'config/e2e.yaml' +BASE_CONFIG = APP / 'config.linux.yaml' +RUNTIME = DATA / 'runtime-linux' +PG_DATA = DATA / 'postgres-linux' +BUNDLES = DATA / 'scanner-result-bundles' +CONTROL = Path('/run/truf/control') +REPO = FIXTURE / 'repo' +TARGET_FILE = RUNTIME / 'queues/e2e-targets.txt' +TARGET = 'https://gitlab.com/truf-e2e/container-pipeline-fixture.git' +CORE_SOURCES = ('gitlab', 'dockerhub', 'huggingface') +REMOTE_TARGET = 'https://gitlab.com/truf-e2e/container-fixture.git' +FIXTURE_REPOSITORY = 'file:///data/fixture/repo' +PREPARED = FIXTURE / 'prepared.json' +RESULT = FIXTURE / 'result.json' +REMOTE_TRANSPORT = FIXTURE / 'remote-transport.json' +REMOTE_FULL = FIXTURE / 'remote-full.json' +DOCKERHUB_CANARY = FIXTURE / 'dockerhub-canary.json' +REMOTE_CLIENT_BUNDLES = FIXTURE / 'remote-client-bundles' +REMOTE_LOCAL_BUNDLES = FIXTURE / 'remote-local-bundles' +MAX_BYTES = 4 * 1024 * 1024 +WORKERS = ('result-ingester', 'jsonl-projector', 'janitor') +DOCKERHUB_CANARY_QUERY = 'truf-e2e-dockerhub-canary' +DOCKERHUB_CANARY_DISCOVERY_TOKEN = 'test-only-dockerhub-discovery-token' +DOCKERHUB_CANARY_REPOSITORIES = tuple( + 'truf-e2e/canary-' + name for name in ('permanent', 'retryable', 'success', 'pending') +) +DOCKERHUB_CANARY_TARGETS = tuple( + repository + '@sha256:' + marker * 64 + for repository, marker in zip(DOCKERHUB_CANARY_REPOSITORIES, 'abcd') +) +DOCKERHUB_CANARY_PER_PAGE = len(DOCKERHUB_CANARY_REPOSITORIES) +BASE_ENVIRONMENT = { + 'PATH': '/usr/local/bin:/usr/bin:/bin', + 'LANG': 'C.UTF-8', 'LC_ALL': 'C.UTF-8', + 'PYTHONDONTWRITEBYTECODE': '1', 'PYTHONNOUSERSITE': '1', 'PYTHONPATH': '', + 'HTTP_PROXY': '', 'http_proxy': '', 'HTTPS_PROXY': '', 'https_proxy': '', + 'ALL_PROXY': '', 'all_proxy': '', 'NO_PROXY': '*', 'no_proxy': '*', +} +KEYCHECK_CHILD_ENVIRONMENT = frozenset({ + 'SCANNER_SUPERVISED', 'TRUF_SUPERVISOR_INSTANCE_FILE', + 'TRUF_SUPERVISOR_INSTANCE_ID', 'TRUF_SUPERVISOR_TOKEN', + 'TRUF_SUPERVISOR_CONFIG_SHA256', 'TRUF_SUPERVISOR_SHA256', + 'TRUF_SUPERVISOR_CODE_MANIFEST_SHA256', 'TRUF_SUPERVISOR_DSN_SHA256', + 'TRUF_SUPERVISOR_CHILD_KIND', 'TRUF_MANAGED_POSTGRES_DSN', + 'SCANNER_DB_URL', 'DATABASE_URL', 'KEYCHECK_DB_URL', 'KEYCHECK_SERVICE', + 'KEYCHECK_INPUT_MODE', 'KEYCHECK_OUTPUT_DIR', 'KEYCHECK_STATE_DIR', + 'KEYCHECK_PROVIDER_SLICE_KEYS', 'KEYCHECK_DB_INLINE', 'KEYCHECK_PROXY_FILE', + 'KEYCHECK_RESULT_PROJECTION_RESERVE_BYTES', 'KEYCHECK_PROJECTION_MAX_ITEMS', + 'KEYCHECK_PROJECTION_MAX_BYTES', +}) + + +class E2EFailure(AssertionError): + """Only fixed, credential-free check names may be used as messages.""" + + +class NotReady(E2EFailure): + pass + + +def require(condition, check): + if not condition: + raise E2EFailure(check) + + +def neutralize_environment(mode): + preserved = { + name: os.environ[name] + for name in KEYCHECK_CHILD_ENVIRONMENT + if mode == '_keycheck-child' and name in os.environ + } + os.environ.clear() + os.environ.update(BASE_ENVIRONMENT) + os.environ.update(preserved) + tempfile.tempdir = None + + +def json_bytes(value): + return json.dumps( + value, ensure_ascii=False, sort_keys=True, separators=(',', ':'), + ).encode('utf-8') + + +def digest(value): + return hashlib.sha256(value).hexdigest() + + +def synthetic_openai_token(): + # Identical known-fake canary to tests/test_synthetic_llm_pipeline.py. + alphabet = 'abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789' + + def material(label): + seed = hashlib.sha512(label.encode('ascii')).hexdigest() + return ''.join( + alphabet[int(seed[index:index + 2], 16) % len(alphabet)] + for index in range(0, len(seed), 2) + )[:24] + + return ( + 'sk-proj-' + material('local-openai-canary-prefix') + 'T3BlbkFJ' + + material('local-openai-canary-suffix') + ) + + +def fixture_config(base): + config = copy.deepcopy(base) + global_config = config['global'] + global_config.update({ + 'root_dir': '/opt/truf', 'project_dir': '/opt/truf/app', + 'runtime_dir': RUNTIME.as_posix(), 'postgres_data_dir': PG_DATA.as_posix(), + 'postgres_bin_dir': '/usr/lib/postgresql/16/bin', + 'result_bundle_dir': BUNDLES.as_posix(), 'control_dir': CONTROL.as_posix(), + 'secrets_file': '/data/config/secrets.yaml', + 'work_dir': '/data/scanner-work', 'database_url': '', + 'dashboard_db_url': '', 'api_proxy_enabled': False, + 'download_proxy_enabled': False, 'sync_file_queues': False, + 'max_active_scans': 1, 'opportunistic_scan_slots': 0, + 'min_free_gb': 0, + 'result_bundle_min_free_bytes': 0, 'loop': False, + 'detectors': 'OpenAI', 'exclude_detectors': '', 'drop_detectors': '', + 'no_verification': True, 'trufflehog_config': '', + }) + for name, source in config['sources'].items(): + source['enabled'] = name in CORE_SOURCES + if 'auth_pool' in source: + source['auth_pool'] = '' + config['sources']['gitlab'].update({ + 'enabled': True, 'mode': 'custom', 'target_file': TARGET_FILE.as_posix(), + 'queries': ['e2e'], 'query_overrides': {}, 'workers': 1, + 'timeout': 60, 'max_commit_age_days': 0, + 'exact_git_planning_enabled': False, 'updated_target_rescan_enabled': False, + 'external_trufflehog_lifecycle': True, 'auth_pool': '', + }) + supervisor = config['supervisor'] + supervisor.update({ + 'enabled_sources': list(CORE_SOURCES), 'interactive': False, 'autostart': True, + 'postgres_stable_ready_sec': 2, 'postgres_health_interval_sec': 1, + 'interval': 3600, + }) + supervisor.setdefault('defaults', {}).update({ + 'enabled': False, 'once': True, 'repeat': False, 'restart': False, + }) + supervisor['sources']['gitlab'].update({ + 'enabled': True, 'once': True, + 'env': { + 'GOMAXPROCS': '1', 'GIT_CONFIG_NOSYSTEM': '1', + 'GIT_CONFIG_GLOBAL': '/dev/null', 'GIT_ALLOW_PROTOCOL': 'file', + 'GIT_CONFIG_COUNT': '2', + 'GIT_CONFIG_KEY_0': 'url.' + FIXTURE_REPOSITORY + '.insteadOf', + 'GIT_CONFIG_VALUE_0': TARGET, + 'GIT_CONFIG_KEY_1': 'url.' + FIXTURE_REPOSITORY + '.insteadOf', + 'GIT_CONFIG_VALUE_1': REMOTE_TARGET, + }, + }) + for name in ('result_ingester', 'jsonl_projector', 'janitor'): + supervisor[name]['enabled'] = True + supervisor['docker_shadow']['enabled'] = False + supervisor['dashboard']['enabled'] = False + config['keychecks'].update({'enabled': False, 'autostart': False}) + return config + + +def require_container(config_path): + require(sys.platform == 'linux', 'linux_test_container_required') + require(os.getuid() == os.geteuid() == 10001 and os.getgid() == 10001, + 'uid_10001_required') + require(APP == Path('/opt/truf/app'), 'test_image_path_required') + require(Path(config_path) == CONFIG, 'fixture_config_path_required') + require(sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode, + 'isolated_python_required') + require({name for _, name in socket.if_nameindex()} == {'lo'}, + 'network_none_required') + require(not os.access(APP, os.W_OK), 'readonly_application_required') + run = CONTROL.parent.stat(follow_symlinks=False) + require(stat.S_ISDIR(run.st_mode) and stat.S_IMODE(run.st_mode) == 0o700 + and run.st_uid == 10001, 'private_run_directory_required') + with open('/proc/self/mountinfo', 'rb') as handle: + mounts = handle.read(1024 * 1024 + 1) + require(len(mounts) <= 1024 * 1024, 'mountinfo_byte_bound') + require(any( + len(fields) > 6 and fields[4] == b'/run/truf' and b'-' in fields + and fields[fields.index(b'-') + 1] == b'tmpfs' + for fields in (line.split() for line in mounts.splitlines()) + ), 'run_tmpfs_required') + os.umask(0o077) + + +def read_bytes(path, limit=MAX_BYTES, private=True): + from runtime_security import reject_reparse_components, require_private_file + + reject_reparse_components(str(path)) + if private: + require_private_file(str(path)) + details = path.stat(follow_symlinks=False) + require(stat.S_ISREG(details.st_mode) and details.st_size <= limit, + 'bounded_regular_file_required') + with path.open('rb') as handle: + value = handle.read(limit + 1) + require(len(value) <= limit, 'file_byte_bound') + return value + + +def write_new(path, payload): + from runtime_security import fsync_directory, require_private_directory + + require_private_directory(str(path.parent), create=False) + descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(descriptor, 'wb') as handle: + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + fsync_directory(str(path.parent)) + + +def fixture_environment(): + # Git must not read user/global configuration, hooks, or credentials. + return { + 'PATH': '/usr/local/bin:/usr/bin:/bin', 'HOME': str(FIXTURE), + 'TMPDIR': str(FIXTURE / 'tmp'), 'LANG': 'C.UTF-8', 'GOMAXPROCS': '1', + 'GIT_CONFIG_NOSYSTEM': '1', 'GIT_CONFIG_GLOBAL': '/dev/null', + 'GIT_TERMINAL_PROMPT': '0', 'GIT_ALLOW_PROTOCOL': 'file', + 'GIT_OPTIONAL_LOCKS': '0', 'NO_PROXY': '*', + 'GIT_CONFIG_COUNT': '2', + 'GIT_CONFIG_KEY_0': 'url.' + FIXTURE_REPOSITORY + '.insteadOf', + 'GIT_CONFIG_VALUE_0': TARGET, + 'GIT_CONFIG_KEY_1': 'url.' + FIXTURE_REPOSITORY + '.insteadOf', + 'GIT_CONFIG_VALUE_1': REMOTE_TARGET, + } + + +def git(*arguments): + command = [ + '/usr/bin/git', '-c', 'user.name=Container E2E Fixture', + '-c', 'user.email=fixture@example.invalid', '-c', 'commit.gpgsign=false', + '-c', 'core.hooksPath=/dev/null', *arguments, + ] + env = fixture_environment() + env.update({ + 'GIT_AUTHOR_DATE': '2026-01-01T00:00:00+00:00', + 'GIT_COMMITTER_DATE': '2026-01-01T00:00:00+00:00', + }) + completed = subprocess.run( + command, cwd=REPO, env=env, stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=15, check=False, + ) + require(completed.returncode == 0, 'fixture_git_failed') + require(len(completed.stdout) <= 65536, 'fixture_git_output_bound') + return completed.stdout.strip() + + +def native_fixture_scan(config): + from runtime_security import require_trusted_native_executable + + command = [ + require_trusted_native_executable(config['global']['trufflehog_path']), + 'git', TARGET, '--json', '--no-update', '--local-dev', + '--include-detectors', 'OpenAI', '--no-verification', '--concurrency', '1', + ] + # Capture even the native self-check in private storage, never the terminal. + with tempfile.TemporaryFile(dir=FIXTURE) as output, tempfile.TemporaryFile(dir=FIXTURE) as errors: + completed = subprocess.run( + command, cwd=FIXTURE, env=fixture_environment(), stdin=subprocess.DEVNULL, + stdout=output, stderr=errors, timeout=60, check=False, + ) + require(completed.returncode == 0, 'native_scan_exit') + output.seek(0) + errors.seek(0) + raw = output.read(MAX_BYTES + 1) + diagnostics = errors.read(MAX_BYTES + 1) + require(max(len(raw), len(diagnostics)) <= MAX_BYTES, 'native_scan_output_bound') + findings = [json.loads(line) for line in raw.splitlines() if line.strip()] + require(len(findings) == 1, 'native_scan_finding_count') + require(findings[0].get('DetectorName') == 'OpenAI' + and findings[0].get('Raw') == synthetic_openai_token() + and findings[0].get('Verified') is False, 'native_scan_fixture_identity') + finished = 0 + for line in diagnostics.splitlines(): + try: + value = json.loads(line) + except ValueError: + continue + if isinstance(value, dict) and value.get('msg') == 'finished scanning': + finished += 1 + require(finished == 1, 'native_scan_completion_marker') + return {'counts': {'native_findings': 1, 'native_finished': 1}, + 'hashes': {'native_command_sha256': digest(json_bytes(command))}} + + +def require_fresh_volume(): + from runtime_security import require_private_directory + + # Only the provisioner's exact empty proxy file may exist in the fresh tree. + proxy = RUNTIME / 'proxy.txt' + entries = 0 + for root in (RUNTIME, PG_DATA, BUNDLES, DATA / 'scanner-work', FIXTURE): + if not root.exists(): + continue + require_private_directory(str(root), create=False) + for current, directories, files in os.walk(root, followlinks=False): + entries += len(directories) + len(files) + require(entries <= 512, 'fresh_volume_required') + for name in files: + path = Path(current) / name + require(path == proxy, 'fresh_volume_required') + details = path.lstat() + require(stat.S_ISREG(details.st_mode) + and stat.S_IMODE(details.st_mode) == 0o600 + and details.st_uid == details.st_gid == 10001 + and details.st_size == 0 and details.st_nlink == 1, + 'fresh_volume_required') + for name in directories: + require(Path(current) / name != proxy, 'fresh_volume_required') + require_private_directory(os.path.join(current, name), create=False) + + +def prepare(): + import yaml + from runtime_security import require_private_directory, write_private_json_exclusive + + require_private_directory(str(DATA), create=False) + require_private_directory(str(CONFIG.parent), create=False) + require(not os.path.lexists(CONFIG) and not os.path.lexists(PREPARED), + 'fixture_already_prepared') + require(not os.path.lexists(PG_DATA / 'PG_VERSION'), 'fresh_volume_required') + require_fresh_volume() + require_private_directory(str(FIXTURE), create=True) + require_private_directory(str(FIXTURE / 'tmp'), create=True) + require_private_directory(str(REPO), create=True) + config = fixture_config(yaml.safe_load(read_bytes(BASE_CONFIG, private=False))) + payload = ('OPENAI_API_KEY=' + synthetic_openai_token() + '\n').encode('ascii') + git('init', '--quiet', '--initial-branch=main', '--template=') + write_new(REPO / 'synthetic.env', payload) + git('add', '--', 'synthetic.env') + git('commit', '--quiet', '-m', 'Known-fake offline OpenAI fixture') + commit = git('rev-parse', 'HEAD').decode('ascii') + require(re.fullmatch(r'[a-f0-9]{40}', commit), 'fixture_git_commit') + write_new(TARGET_FILE, (TARGET + '\n').encode('ascii')) + native = native_fixture_scan(config) + config_bytes = yaml.safe_dump(config, sort_keys=False).encode('utf-8') + write_new(CONFIG, config_bytes) + marker = { + 'schema': 1, 'repo_commit': commit, 'config_sha256': digest(config_bytes), + 'fixture_sha256': digest(payload), 'native': native, + } + write_private_json_exclusive(str(PREPARED), marker) + return { + 'counts': {'ok': 1, 'prepared': 1, **native['counts']}, + 'hashes': {'config_sha256': marker['config_sha256'], **native['hashes']}, + } + + +def load_fixture(): + import yaml + + marker = json.loads(read_bytes(PREPARED)) + require(marker.get('schema') == 1, 'prepared_fixture_required') + config_bytes = read_bytes(CONFIG) + require(digest(config_bytes) == marker['config_sha256'], 'fixture_config_changed') + config = yaml.safe_load(config_bytes) + require(config == fixture_config(yaml.safe_load(read_bytes(BASE_CONFIG, private=False))), + 'fixture_config_contract_changed') + payload = ('OPENAI_API_KEY=' + synthetic_openai_token() + '\n').encode('ascii') + require(read_bytes(REPO / 'synthetic.env') == payload + and marker['fixture_sha256'] == digest(payload), 'fixture_material_changed') + require(read_bytes(TARGET_FILE) == (TARGET + '\n').encode('ascii'), + 'fixture_target_changed') + require(git('rev-parse', 'HEAD').decode('ascii') == marker['repo_commit'] + and git('rev-list', '--all', '--count') == b'1' + and git('status', '--porcelain', '--untracked-files=all') == b'', + 'fixture_repository_changed') + return config, marker + + +def database_environment(): + import yaml + + password = read_bytes(DATA / 'postgres-password', limit=4096).decode('utf-8').rstrip('\n') + require(len(password) >= 32 and not any(char.isspace() for char in password), + 'generated_postgres_password_required') + require(yaml.safe_load(read_bytes(DATA / 'config/secrets.yaml')) == {}, + 'empty_provider_secrets_required') + url = 'postgresql://truf:' + quote(password, safe='') + '@127.0.0.1:5432/truf' + expected = { + 'TRUF_POSTGRES_DB': 'truf', 'TRUF_POSTGRES_USER': 'truf', + 'TRUF_POSTGRES_PORT': '5432', 'TRUF_POSTGRES_PASSWORD': password, + 'SCANNER_DB_URL': url, 'DATABASE_URL': url, 'TRUF_MANAGED_POSTGRES_DSN': url, + } + # docker exec does not inherit the environment derived inside PID 1. + for name, value in expected.items(): + require(name not in os.environ or os.environ[name] == value, + 'conflicting_database_environment') + require(not any(name.upper().startswith('PG') for name in os.environ), + 'libpq_overrides_forbidden') + os.environ.update(expected) + os.environ.update({ + 'TRUF_DB_CONNECT_TIMEOUT_SEC': '2', 'TRUF_DB_STATEMENT_TIMEOUT_MS': '3000', + 'TRUF_DB_LOCK_TIMEOUT_MS': '1000', 'TRUF_DB_IDLE_TRANSACTION_TIMEOUT_MS': '10000', + 'TRUF_DB_TCP_USER_TIMEOUT_MS': '3000', + }) + return url + + +def control_snapshot(marker, url): + from supervisor import get_control_snapshot + from supervisor_instance import load_instance_metadata + from runtime_security import require_private_directory + + require_private_directory(str(CONTROL.parent), create=False) + try: + metadata = load_instance_metadata(str(CONTROL / 'supervisor.instance.json')) + snapshot = get_control_snapshot(metadata) + except (OSError, ValueError, RuntimeError): + raise NotReady('authenticated_control_not_ready') from None + require(metadata['config_path'] == str(CONFIG) + and metadata['config_sha256'] == marker['config_sha256'] + and metadata['canonical_dsn_sha256'] == digest(url.encode('utf-8')), + 'supervisor_authority_mismatch') + require(snapshot.get('activation_state') not in ('STOPPING', 'FAILED_HOLD'), + 'supervisor_failed_hold') + if snapshot.get('activation_state') != 'ACTIVE': + raise NotReady('supervisor_not_active') + postgres = snapshot.get('postgres') or {} + if postgres.get('state') != 'READY' or postgres.get('ready') is not True: + raise NotReady('postgres_not_ready') + signatures = snapshot.get('signature') or [] + require(all(isinstance(row, (list, tuple)) and len(row) == 9 for row in signatures), + 'worker_signature_shape') + workers = {row[0]: row for row in signatures} + require(len(workers) == len(signatures) + and set(workers) == {'gitlab', *WORKERS}, 'unexpected_worker_set') + for name, row in workers.items(): + require(row[1] != 'failed' and not row[6] and not row[8], + 'worker_failed_or_restarted') + if name in WORKERS: + require(not row[5], 'pipeline_worker_restarted') + if tuple(row[1:3]) != ('running', 'running') or not row[3] or row[7]: + raise NotReady('pipeline_workers_not_running') + gitlab = workers['gitlab'] + if gitlab[1] != 'waiting': + raise NotReady('gitlab_discovery_cycle_not_finished') + require(gitlab[2] == 'running' and gitlab[3] is None + and gitlab[4] == 0 and gitlab[5] == 1 and not gitlab[7], + 'gitlab_discovery_cycle_failed') + require((snapshot.get('dashboard') or {}).get('status') == 'disabled', + 'dashboard_must_be_disabled') + return metadata + + +def rows(db, statement, parameters=()): + require(statement.lstrip().startswith('SELECT '), 'read_only_assertions_required') + result = [dict(row) for row in db.conn.execute(statement, parameters).fetchall()] + db.conn.commit() + return result + + +async def invoke_bundle_upload( + service, token, reservation_id, body, *, content_length=None, + payload_sha256=None, disconnect_after=None, +): + from types import SimpleNamespace + import worker_api + + class Request: + def __init__(self): + self.headers = { + 'authorization': 'Bearer ' + token, + 'content-type': 'application/octet-stream', + 'content-length': str(len(body) if content_length is None else content_length), + 'x-truf-payload-sha256': payload_sha256 or digest(body), + } + self.path_params = {'reservation_id': str(reservation_id)} + self.app = SimpleNamespace(state=SimpleNamespace(worker_service=service)) + + async def stream(self): + sent = 0 + while sent < len(body): + chunk = body[sent:sent + 257] + sent += len(chunk) + yield chunk + if disconnect_after is not None and sent >= disconnect_after: + raise ConnectionError('synthetic disconnect') + + try: + response = await worker_api.upload_bundle(Request()) + except worker_api.WorkerAPIError as exc: + return exc.status_code, exc.code, None + except worker_api.ResultBundleError: + return 400, 'invalid_bundle', None + except worker_api.ScanEventConflictError: + return 409, 'reservation_conflict', None + return response.status_code, None, json.loads(bytes(response.body)) + + +async def invoke_worker_claim(service, token, request_id, build): + from types import SimpleNamespace + import worker_api + + body = json_bytes({'request_id': request_id, 'build': build}) + + class Request: + headers = { + 'authorization': 'Bearer ' + token, + 'content-type': 'application/json', + } + app = SimpleNamespace(state=SimpleNamespace(worker_service=service)) + + async def stream(self): + for offset in range(0, len(body), 257): + yield body[offset:offset + 257] + + response = await worker_api.claim(Request()) + body = bytes(response.body) + return response.status_code, json.loads(body) if body else None + + +async def invoke_terminal_report(service, token, reservation_id, payload): + from types import SimpleNamespace + import worker_api + + body = json_bytes(payload) + + class Request: + headers = { + 'authorization': 'Bearer ' + token, + 'content-type': 'application/json', + } + path_params = {'reservation_id': str(reservation_id)} + app = SimpleNamespace(state=SimpleNamespace(worker_service=service)) + + async def stream(self): + yield body + + try: + response = await worker_api.terminal_report(Request()) + except worker_api.WorkerAPIError as exc: + return exc.status_code, exc.code, None + except worker_api.ScanEventConflictError: + return 409, 'reservation_conflict', None + return response.status_code, None, json.loads(bytes(response.body)) + + +def remote_device_token(role): + require(role in ('expired', 'winner'), 'remote_fixture_role') + return 'container-e2e-' + role + '-device-token' + + +def remote_source_args(config): + from types import SimpleNamespace + + source = config['sources']['gitlab'] + return SimpleNamespace( + platform='gitlab', exact_git_planning_enabled=True, + workers=1, timeout=int(source['timeout']), save_dir=str(FIXTURE), + detectors='OpenAI', exclude_detectors='', drop_detectors=[], + no_verification=True, + trufflehog_config=str(APP / 'trufflehog-custom-detectors.yaml'), token='', + external_trufflehog_lifecycle=True, + scan_full_history=False, max_depth=0, git_baseline_depth=1, + max_commit_age_days=0, commit_lookup_pages=1, + skip_if_commit_lookup_fails=True, + result_bundle_max_event_bytes=MAX_BYTES, + result_bundle_max_items=4, result_bundle_max_total_bytes=4 * MAX_BYTES, + projection_backlog_max_items=8, projection_backlog_max_bytes=8 * MAX_BYTES, + projection_backlog_headroom_bytes=2 * MAX_BYTES, + keycheck_queue_max_items=8, keycheck_queue_max_bytes=4 * MAX_BYTES, + pipeline_quarantine_max_items=4, pipeline_quarantine_max_bytes=4 * MAX_BYTES, + keycheck_candidates_per_event=8, keycheck_candidate_bytes_per_event=MAX_BYTES, + target_retry_max_attempts=3, target_retry_base_delay_sec=60, + target_retry_max_delay_sec=600, target_timeout_retry_delay_sec=300, + max_active_scans=1, admission_resolution_attempts=2, + admission_resolution_seconds=2, admission_resolution_retry_delay_sec=0.01, + target_claim_order='oldest', git_ref_resolution_attempts=1, + git_ref_resolution_timeout_sec=1, git_ref_resolution_max_bytes=1 << 20, + strict_git_provider_token_filter=True, trufflehog_stdout_max_mb=2, + trufflehog_stderr_max_mb=1, trufflehog_max_findings_per_target=100, + trufflehog_job_memory_limit_bytes=0, trufflehog_windows_job_cpu_weight=0, + trufflehog_windows_memory_priority=0, trufflehog_diagnostic_max_lines=200, + trufflehog_diagnostic_max_line_chars=2048, + trufflehog_diagnostic_max_line_bytes=2048, + trufflehog_diagnostic_max_errors=20, trufflehog_diagnostic_max_warnings=20, + trufflehog_diagnostic_max_unclassified=10, + ) + + +def remote_package_manifest(metadata, capabilities=None): + from lifecycle_authority import ( + GIT_MANIFEST_NAME, REMOTE_WORKER_CODE_AUTHORITY_FILES, + TRUFFLEHOG_MANIFEST_NAME, + ) + from result_bundle import FORMAT_VERSION + from scan_execution import PROTOCOL_VERSION, local_platform_tag + + authority = metadata['code_manifest'] + require(set(REMOTE_WORKER_CODE_AUTHORITY_FILES) <= set(authority['files']), + 'remote_code_authority_fixture') + policy = APP / 'trufflehog-custom-detectors.yaml' + policy_sha256 = digest(read_bytes(policy, private=False)) + files = { + name: { + 'path': 'app/' + name, + 'sha256': authority['files'][name]['sha256'], + } + for name in REMOTE_WORKER_CODE_AUTHORITY_FILES + } + executables = authority['executables'] + capabilities = capabilities or [{ + 'source': 'gitlab', 'platform': 'gitlab', + 'planning_kind': 'exact_git_v1', + }] + return { + 'schema': 3, 'protocol_version': PROTOCOL_VERSION, + 'bundle_format_version': FORMAT_VERSION, + 'platform_tag': local_platform_tag(), + 'capabilities': [dict(value) for value in capabilities], + 'app_root': 'app', 'files': files, + 'executables': { + TRUFFLEHOG_MANIFEST_NAME: { + 'path': 'bin/trufflehog', + 'sha256': executables[TRUFFLEHOG_MANIFEST_NAME]['sha256'], + }, + GIT_MANIFEST_NAME: { + 'path': 'runtime/git/bin/git', + 'sha256': executables[GIT_MANIFEST_NAME]['sha256'], + }, + }, + 'assets': {'detector_policy': { + 'path': 'app/trufflehog-custom-detectors.yaml', + 'sha256': policy_sha256, + }}, + 'runtime_trees': {'git': { + 'path': 'runtime/git', + 'sha256': digest(json_bytes({ + 'git': executables[GIT_MANIFEST_NAME]['sha256'], + })), + 'file_count': 1, + }}, + } + + +def remote_worker_service(config, marker, metadata, url): + import scanner_db + from worker_api import WorkerService + from worker_assignment import RemoteGitAssignmentBuilder + from worker_package import worker_package_build_compatibility + + def planner(args, db_url, source, claim, scan_kwargs, remote_credential=None): + del args, scan_kwargs + require(source == 'gitlab' and claim['target'] == REMOTE_TARGET, + 'remote_fixture_plan_target') + planning = scanner_db.ScannerDB(db_url=db_url, initialize=False) + try: + resolution = { + 'provider': 'gitlab', 'repo_url': REMOTE_TARGET, + 'repo_path': 'truf-e2e/container-fixture', 'branch': 'main', + 'ref': 'refs/heads/main', 'head_sha': marker['repo_commit'], + 'ref_source': 'provider_default', + } + return planning.bind_git_scan_plan( + claim['reservation_id'], claim['claim_lease_token'], resolution, 1, + remote_credential=remote_credential, + ) + finally: + planning.close() + + package = remote_package_manifest(metadata) + builder = RemoteGitAssignmentBuilder( + url, str(BUNDLES), {'gitlab': remote_source_args(config)}, + {'container-e2e': {'package_manifest': package, 'sources': ['gitlab']}}, + metadata['instance_id'], assignment_ttl_seconds=60, planner=planner, + ) + return ( + WorkerService(url, str(BUNDLES), builder, max_bundle_bytes=MAX_BYTES), + worker_package_build_compatibility(package), + ) + + +def execute_fixture_assignment(config, metadata, assignment, *, compare_local): + from console_runner import apply_global_config + from lifecycle_authority import build_code_manifest, code_manifest_sha256 + from paths import apply_path_config + from result_bundle import BundleReservation + from runtime_security import ensure_private_directory + from scan_execution import ( + PACKAGE_DETECTOR_POLICY, execute_planned_claim, stage_scan_result_in_scope, + ) + import scanner + + stage = 'reservation' + reservation = BundleReservation.from_mapping(assignment['reservation']) + require(reservation.target == REMOTE_TARGET and assignment['scan_kwargs'].get( + 'trufflehog_config') == PACKAGE_DETECTOR_POLICY, 'remote_execution_assignment') + stage = 'directories' + ensure_private_directory(str(REMOTE_CLIENT_BUNDLES), reject_reparse=True) + if compare_local: + ensure_private_directory(str(REMOTE_LOCAL_BUNDLES), reject_reparse=True) + stage = 'config' + apply_global_config(apply_path_config(config, str(CONFIG))['global']) + scanner.initialize_scanner_runtime(preflight_complete=True, register_cleanup=False) + scan_kwargs = dict(assignment['scan_kwargs']) + policy_path = str(APP / 'trufflehog-custom-detectors.yaml') + scan_kwargs['trufflehog_config'] = policy_path + client_manifest = build_code_manifest( + str(APP), scanner.scan_config.trufflehog_path, (policy_path,), + git_path=metadata['code_manifest']['executables']['git']['path'], + ) + limits = assignment['limits'] + environment = fixture_environment() + previous = {name: os.environ.get(name) for name in environment} + try: + os.environ.update(environment) + stage = 'authority' + with scanner.client_scan_launch_authority( + client_manifest, code_manifest_sha256(client_manifest), + ): + local = None + if compare_local: + stage = 'local_scan' + with scanner.client_scan_execution_policy(assignment['scan_policy']): + result = scanner.scan_target_result( + reservation.target, reservation.platform, + reservation.scan_event_id, scan_kwargs, + ) + stage = 'local_bundle' + local = stage_scan_result_in_scope( + result, reservation, str(REMOTE_LOCAL_BUNDLES), + assignment['event_scan_options'], assignment['queue_policy'], + attempts=int(assignment['reservation']['attempts']), + candidate_max_items=int(limits['candidate_max_items']), + candidate_max_bytes=int(limits['candidate_max_bytes']), + ) + stage = 'remote_scan_bundle' + remote = execute_planned_claim( + reservation, str(REMOTE_CLIENT_BUNDLES), scan_kwargs, + assignment['event_scan_options'], assignment['queue_policy'], + assignment['scan_policy'], + attempts=int(assignment['reservation']['attempts']), + candidate_max_items=int(limits['candidate_max_items']), + candidate_max_bytes=int(limits['candidate_max_bytes']), + ) + except E2EFailure: + raise + except Exception as exc: + trace = exc.__traceback__ + origins = [] + while trace: + origins.append(trace.tb_frame.f_code.co_name) + trace = trace.tb_next + origin = '_'.join([type(exc).__name__, *origins[-3:]]) + origin = re.sub(r'[^a-z0-9_]', '_', origin.lower())[:80] + raise E2EFailure('remote_execution_' + stage + '_' + origin) from None + finally: + for name, value in previous.items(): + if value is None: + os.environ.pop(name, None) + else: + os.environ[name] = value + + stage = 'bundle_read' + try: + remote_path = REMOTE_CLIENT_BUNDLES / remote.relative_path + payload = read_bytes(remote_path) + except E2EFailure: + raise + except Exception: + raise E2EFailure('remote_execution_' + stage + '_exception') from None + parity_sha256 = '' + if compare_local: + stage = 'parity' + try: + helpers = runpy.run_path(str(Path(__file__).with_name('parity_helpers.py'))) + local_evidence = helpers['normalized_bundle_evidence']( + str(REMOTE_LOCAL_BUNDLES / local.relative_path), + ) + remote_evidence = helpers['normalized_bundle_evidence'](str(remote_path)) + except Exception: + raise E2EFailure('remote_execution_' + stage + '_exception') from None + require(helpers['bundle_evidence_difference_paths']( + local_evidence, remote_evidence, + ) == [], 'remote_local_bundle_parity') + require(local.queue_status == remote.queue_status == 'done' + and not local.first_error and not remote.first_error, + 'remote_local_disposition_parity') + parity_sha256 = digest(json_bytes(remote_evidence)) + return payload, remote.as_dict(), parity_sha256 + + +def verify_projection_region(payload, expected, append, state, job, records): + offset = int(append['byte_offset']) + end = offset + len(expected) + require(0 <= offset <= end <= len(payload) + and payload[offset:end] == expected, 'projection_region_bytes_differ') + require(append['state'] == 'appended' and append['job_id'] == job['id'] + and append['stream_name'] == state['stream_name'] + and append['event_id'] == job['event_id'] + and append['event_hash'] == job['event_hash'], 'projection_region_identity') + require(append['byte_length'] == len(expected) + and append['payload_sha256'] == digest(expected) + and append['record_count'] == records, 'projection_region_length') + require(append['generation'] == state['generation'] == state['current_generation'], + 'projection_region_generation') + require(state['committed_offset'] == len(payload) + and state['committed_offset'] >= end, + 'projection_region_committed_size') + require(state['last_append_id'] >= append['id'], + 'projection_region_cursor_coverage') + + +def verify_projection(payload, expected, append, state, job, records): + require(payload == expected, 'projection_bytes_differ') + require(append['state'] == 'appended' + and append['job_id'] == job['id'] + and append['stream_name'] == state['stream_name'] + and append['event_id'] == job['event_id'] + and append['event_hash'] == job['event_hash'], 'projection_append_identity') + require(append['byte_offset'] == 0 and append['byte_length'] == len(payload) + and append['payload_sha256'] == digest(payload) + and append['record_count'] == records, 'projection_append_bytes') + require(append['generation'] == state['generation'] == state['current_generation'] == 0 + and state['committed_offset'] == len(payload) + and state['last_append_id'] == append['id'] + and state['last_job_id'] == job['id'] + and state['last_event_id'] == job['event_id'] + and state['last_event_hash'] == job['event_hash'], 'projection_cursor_identity') + + +def pipeline_snapshot(config, marker, url, checked): + import scanner_db + from keycheck_candidates import candidate_uid, extract_candidates + + metadata = control_snapshot(marker, url) + db = scanner_db.ScannerDB(db_url=url, initialize=False) + if not db.enabled: + raise NotReady('database_not_ready') + try: + require(db.conn.is_postgres, 'postgres_required') + db.set_application_name('truf-container-e2e:read-only') + db.conn.execute('SET default_transaction_read_only = on') + db.conn.commit() + identity = rows(db, '''SELECT current_database() AS database, current_user AS username, + current_setting('data_directory') AS data_directory, + current_setting('server_version_num') AS server_version''')[0] + require(identity['database'] == identity['username'] == 'truf' + and identity['data_directory'] == str(PG_DATA) + and int(identity['server_version']) // 10000 == 16, 'postgres_identity') + for worker in ('result_ingester', 'jsonl_projector'): + if not db.pipeline_worker_health(worker, metadata['instance_id'])['healthy']: + raise NotReady('pipeline_lease_not_ready') + db.require_runtime_safety_schema() + db.require_final_cutover() + migrations = rows(db, 'SELECT version, code_sha256 FROM runtime_schema_migrations ORDER BY version') + require(len(migrations) == len(scanner_db.PIPELINE_MIGRATION_VERSIONS) == 33 + and {row['version'] for row in migrations} == set(scanner_db.PIPELINE_MIGRATION_VERSIONS), + 'fresh_migration_versions') + sql_hash = digest(scanner_db.PIPELINE_SCHEMA_SQL.encode('utf-8')) + require(next(row['code_sha256'] for row in migrations + if row['version'] == scanner_db.PIPELINE_MIGRATION_VERSIONS[-1]) == sql_hash, + 'latest_migration_sql_hash') + cutover = db.final_cutover_status() + require(cutover is not None, 'valid_final_cutover_required') + expected_counts = { + 'target_queue': 1, 'result_reservations': 1, 'result_bundles': 1, + 'target_scans': 1, 'scan_result_compat': 1, 'findings': 1, + 'finding_compat_payloads': 1, 'finding_uid_map': 1, + 'keycheck_candidates': 1, 'keycheck_credentials': 1, + 'projection_jobs': 1 + checked, 'projection_appends': 2 + checked, + 'keycheck_results': checked, 'keycheck_current_state': checked, + 'keycheck_event_map': checked, 'errors': 0, 'pipeline_quarantine': 0, + 'projection_append_audit': 0, 'projection_rotations': 0, + 'scan_publication_outbox': 0, + } + counts = { + name: int(rows(db, f'SELECT COUNT(*) AS count FROM {name}')[0]['count']) + for name in expected_counts + } + for name, expected in expected_counts.items(): + require(counts[name] <= expected, 'unexpected_or_duplicate_' + name) + if counts != expected_counts: + raise NotReady('pipeline_output_counts_not_ready') + queue = rows(db, 'SELECT * FROM target_queue LIMIT 2')[0] + reservation = rows(db, 'SELECT * FROM result_reservations LIMIT 2')[0] + scan = rows(db, 'SELECT * FROM target_scans LIMIT 2')[0] + finding = rows(db, 'SELECT * FROM findings LIMIT 2')[0] + candidate = rows(db, 'SELECT * FROM keycheck_candidates LIMIT 2')[0] + credential = rows(db, 'SELECT * FROM keycheck_credentials LIMIT 2')[0] + bundle = db.result_bundle_for_reservation(reservation['id']) + jobs = rows(db, 'SELECT * FROM projection_jobs ORDER BY id LIMIT 3') + if (queue['status'] != 'done' or reservation['state'] != 'acknowledged' + or bundle['state'] != 'acknowledged' + or any(job['status'] != 'completed' for job in jobs)): + raise NotReady('pipeline_durable_completion_not_ready') + for row in (queue, reservation, scan, finding, candidate): + require(row['source'] == 'gitlab' and row['query'] == 'e2e' + and row['target'] == TARGET, 'pipeline_fixture_attribution') + require(queue['target_scan_id'] == scan['id'] + and queue['attempts'] == 1 + and queue['lease_token'] is None + and queue['current_result_reservation_id'] is None + and queue['claim_event_id'] is None and not queue['last_error'], 'queue_completion') + require(scan['queue_id'] == reservation['queue_id'] == queue['id'] + and scan['result_reservation_id'] == reservation['id'] + and scan['scan_event_id'] == reservation['scan_event_id'] == bundle['scan_event_id'] + and scan['scan_event_hash'] == bundle['scan_event_hash'] + and scan['claim_lease_token'] == reservation['claim_lease_token'] + and scan['queue_completion_applied'] == 1 + and scan['queue_completion_disposition'] == 'applied', 'scan_reservation_identity') + require(scan['raw_result_storage'] == 'normalized_v2' and scan['raw_result_json'] is None + and scan['status'] == 'found' and scan['scan_type'] == 'gitlab' + and scan['findings_count'] == 1 and scan['verified_findings_count'] == 0 + and scan['error_count'] == 0, 'normalized_scan_result') + confirmed = db.confirm_scan_event(scan['scan_event_id'], scan['scan_event_hash']) + require(confirmed and confirmed['id'] == scan['id'], 'confirmed_scan_event') + require(bundle['target_scan_id'] == scan['id'] and bundle['format_version'] == 2 + and bundle['bundle_id'] == reservation['bundle_id'] + and bundle['finding_count'] == bundle['candidate_count'] == 1 + and bundle['error_count'] == 0 and bundle['actual_bytes'] > 0 + and reservation['bundle_credit_released'] == 1 + and reservation['candidate_credit_transferred'] == 1 + and reservation['projection_credit_transferred'] == 1, 'bundle_completion') + relative = Path(bundle['relative_path']) + require(not relative.is_absolute() and '..' not in relative.parts, 'bundle_path') + require(not os.path.lexists(BUNDLES / relative), 'acknowledged_bundle_not_removed') + require(finding['target_scan_id'] == scan['id'] and finding['detector_name'] == 'OpenAI' + and not finding['verified'] and finding['file_path'] == 'synthetic.env' + and finding['commit_hash'] == marker['repo_commit'], 'native_git_finding_provenance') + rebuilt = db.reconstruct_scan_result(scan['id'], max_bytes=MAX_BYTES, max_findings=2, max_errors=0) + require(rebuilt and len(rebuilt['findings']) == 1 and rebuilt['errors'] == [], + 'reconstructed_scan_result') + scan_meta = rebuilt.get('scan_meta') or {} + require(scan_meta.get('trufflehog_returncode') == 0 + and scan_meta.get('trufflehog_finished') is True + and scan_meta.get('command_timed_out') is False, 'native_pipeline_completion') + options = rebuilt.get('scan_options') or {} + require(options.get('no_verification') is True and options.get('detectors') == 'OpenAI' + and not options.get('exclude_detectors') and not options.get('trufflehog_config') + and not options.get('max_commit_age_days'), 'native_pipeline_scan_options') + rebuilt_finding = rebuilt['findings'][0] + require(rebuilt_finding.get('Raw') == synthetic_openai_token() + and rebuilt_finding.get('finding_uid') == finding['finding_uid'], 'synthetic_finding_identity') + specs = list(extract_candidates(rebuilt_finding)) + require(len(specs) == 1 and specs[0].service == 'openai', 'candidate_extraction') + spec = specs[0] + require(candidate['finding_id'] == finding['id'] + and candidate['finding_uid'] == finding['finding_uid'] + and candidate['target_scan_id'] == scan['id'] + and candidate['scan_event_id'] == scan['scan_event_id'] + and candidate['credential_id'] == credential['id'] + and candidate['service'] == candidate['routed_service'] == credential['service'] == 'openai' + and candidate['secret_hash'] == finding['secret_hash'] == spec.secret_hash + and candidate['candidate_uid'] == candidate_uid( + scan['scan_event_id'], finding['finding_uid'], 'openai', spec.credential_hash), + 'candidate_exact_attribution') + for name in ('credential_hash', 'provider_key_hash', 'candidate_kind', 'secret_text', 'secret_json'): + require(credential[name] == getattr(spec, name), 'candidate_credential_identity') + require(candidate['state'] == ('completed' if checked else 'pending') + and candidate['attempts'] == checked and candidate['lease_token'] is None + and candidate['capacity_released'] == checked, 'candidate_completion_state') + compat = rows(db, 'SELECT * FROM scan_result_compat LIMIT 2')[0] + compat_bytes = compat['metadata_json'].encode('utf-8') + require(compat['metadata_sha256'] == digest(compat_bytes) + and compat['metadata_bytes'] == len(compat_bytes), 'scan_metadata_hash') + payload_row = rows(db, 'SELECT * FROM finding_compat_payloads LIMIT 2')[0] + require(payload_row['finding_id'] == finding['id'] and not payload_row['payload_omitted'] + and payload_row['payload_sha256'] == finding['raw_payload_sha256'], 'finding_payload_hash') + require(rows(db, 'SELECT finding_id FROM finding_uid_map WHERE finding_uid = ?', + (finding['finding_uid'],)) == [{'finding_id': finding['id']}], 'finding_uid_mapping') + scan_job = next(job for job in jobs if job['job_kind'] == 'scan_event') + require(scan_job['target_scan_id'] == scan['id'] + and scan_job['event_id'] == scan['scan_event_id'] + and scan_job['event_hash'] == scan['scan_event_hash'] + and scan_job['required_stream_mask'] == 3, 'scan_projection_job') + projections = [ + (scan_job, 'scan_results', 'scan_results.jsonl', json_bytes(rebuilt) + b'\n'), + (scan_job, 'found_secrets', 'found_secrets.jsonl', json_bytes(rebuilt_finding) + b'\n'), + ] + ids = { + 'queue_id': queue['id'], 'reservation_id': reservation['id'], + 'bundle_id': bundle['bundle_id'], 'scan_id': scan['id'], + 'scan_event_id': scan['scan_event_id'], 'finding_id': finding['id'], + 'finding_uid': finding['finding_uid'], 'candidate_id': candidate['id'], + 'candidate_uid': candidate['candidate_uid'], 'credential_id': credential['id'], + 'scan_projection_id': scan_job['id'], + } + hashes = { + 'migrations_sha256': digest(json_bytes(migrations)), 'pipeline_sql_sha256': sql_hash, + 'cutover_sha256': digest(json_bytes(cutover)), + 'scan_event_sha256': scan['scan_event_hash'], 'scan_metadata_sha256': digest(compat_bytes), + 'finding_payload_sha256': payload_row['payload_sha256'], + 'fixture_sha256': marker['fixture_sha256'], 'config_sha256': marker['config_sha256'], + } + if checked: + result = db.keycheck_result_for_projection(candidate['keycheck_result_id']) + require(result and result['status'] == 'INVALID_OR_REVOKED' + and result['status_group'] == 'dead' and result['result_source'] == 'api_check' + and result['link_status'] == 'linked' and result['service'] == 'openai' + and result['candidate_id'] == candidate['id'] + and result['credential_id'] == credential['id'] + and result['finding_id'] == finding['id'] + and result['finding_uid'] == finding['finding_uid'] + and result['target_scan_id'] == scan['id'] + and result['key_hash'] == spec.provider_key_hash + and result['secret_hash'] == spec.secret_hash, 'real_provider_completion') + current = rows(db, 'SELECT * FROM keycheck_current_state LIMIT 2')[0] + require(current['last_result_id'] == result['id'] and current['state_version'] == 1 + and current['credential_id'] == credential['id'] + and current['status'] == 'INVALID_OR_REVOKED' + and current['status_group'] == 'dead', 'keycheck_current_state') + require(rows(db, 'SELECT keycheck_result_id FROM keycheck_event_map WHERE event_id = ?', + (result['event_id'],)) == [{'keycheck_result_id': result['id']}], + 'keycheck_event_mapping') + key_job = next(job for job in jobs if job['job_kind'] == 'keycheck_event') + require(key_job['keycheck_result_id'] == result['id'] + and key_job['event_id'] == result['event_id'] + and key_job['required_stream_mask'] == 8 + and key_job['event_hash'] == digest( + result['metadata_json'].encode('utf-8') + result['event_id'].encode('ascii')), + 'keycheck_projection_job') + projected = {key: result[key] for key in ( + 'event_id', 'service', 'status', 'status_group', 'checked_at', 'key_hash', + 'secret_hash', 'key_masked', 'finding_uid', 'source', 'message', 'result_source', + )} + projected.update({'detector': result['detector_name'], + 'metadata': json.loads(result['metadata_json'])}) + projections.append((key_job, 'keycheck:openai:results', 'openai/openaiResults.jsonl', + json_bytes(projected) + b'\n')) + ids.update({'keycheck_result_id': result['id'], 'keycheck_event_id': result['event_id'], + 'keycheck_projection_id': key_job['id']}) + for job, stream, relative_path, expected in projections: + require(job['capacity_released'] == 1, 'projection_capacity_not_released') + state = db.projection_stream_state(stream) + append = db.projection_append_for_job(job['id'], stream) + require(state and append and state['base_relative_path'] == relative_path, + 'projection_stream_path') + root = RUNTIME / ('keychecks' if stream.startswith('keycheck:') else 'results') + payload = read_bytes(root / relative_path) + verify_projection(payload, expected, append, state, job, 1) + hashes[stream + '_sha256'] = digest(payload) + hashes[stream + '_ledger_sha256'] = digest(json_bytes({ + 'append': append, 'cursor': state, + })) + counts[stream + '_bytes'] = len(payload) + capacity = db.pipeline_capacity_snapshot() + for name in ('bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'quarantine_items', 'quarantine_bytes'): + require(capacity[name] == 0, 'pipeline_capacity_not_drained') + require(capacity['keycheck_items'] == 1 - checked + and capacity['keycheck_bytes'] == (0 if checked else candidate['capacity_bytes']), + 'keycheck_capacity_accounting') + hashes['artifact_ids_sha256'] = digest(json_bytes(ids)) + counts.update({'queue_attempts': queue['attempts'], + 'migration_versions': len(migrations), 'healthy_pipeline_leases': 2, + 'pipeline_workers': len(WORKERS), 'native_finished': 1}) + return {'schema': 1, 'checked': checked, 'counts': counts, 'hashes': hashes, 'ids': ids, + 'origin_instance_sha256': digest(metadata['instance_id'].encode('utf-8'))} + finally: + db.close() + + +def wait_for_pipeline(config, marker, url, checked, timeout): + deadline = time.monotonic() + timeout + last_check = 'pipeline_timeout' + while time.monotonic() < deadline: + try: + return pipeline_snapshot(config, marker, url, checked) + except NotReady as exc: + last_check = str(exc) + time.sleep(min(0.5, max(0, deadline - time.monotonic()))) + raise E2EFailure('timeout_' + last_check) + + +def run_local_pipeline_fixture(config, marker, url): + from types import SimpleNamespace + + import console_runner + from lifecycle_authority import supervised_child_environment + import scanner_db + + metadata = control_snapshot(marker, url) + db = scanner_db.ScannerDB(db_url=url, initialize=False) + try: + queue = rows(db, 'SELECT source, status FROM target_queue LIMIT 2') + require(queue == [{'source': 'gitlab', 'status': 'pending'}], + 'pipeline_fixture_queue_required') + finally: + db.close() + + authority_environment = supervised_child_environment(metadata, url, 'scanner') + metadata = None + os.environ.update(fixture_environment()) + os.environ.update(authority_environment) + try: + console_runner.run_config_mode(SimpleNamespace( + config=str(CONFIG), source='gitlab', once=True, show_state=False, + cleanup_only=False, cooldown=0, + ), runtime_role='scanner') + finally: + for name in authority_environment: + os.environ.pop(name, None) + authority_environment = None + return {'counts': {'ok': 1, 'local_pipeline_scans': 1}, 'hashes': {}} + + +def compare_persisted(before, after, require_restart=False): + require(before.get('schema') == after.get('schema') == 1, 'result_baseline_schema') + for name in ('checked', 'counts', 'hashes', 'ids'): + require(before[name] == after[name], 'persisted_' + name + '_changed') + if require_restart: + require(before['origin_instance_sha256'] != after['origin_instance_sha256'], + 'container_recreation_required') + + +def assert_remote_recovery(config, marker, url): + del config + import threading + from process_identity import current_process_identity + from result_bundle import FORMAT_VERSION + from scan_execution import ( + PACKAGE_DETECTOR_POLICY, PROTOCOL_VERSION, QueueDispositionPolicy, + remote_execution_identity, + ) + import scanner_db + from unittest import mock + + metadata = control_snapshot(marker, url) + db = scanner_db.ScannerDB(db_url=url, initialize=False) + connections = [] + try: + require(db.enabled and db.conn.is_postgres, 'remote_postgres_required') + db.set_application_name('truf-container-e2e:remote-recovery') + db.require_runtime_safety_schema() + db.require_final_cutover() + require(db.pipeline_worker_health( + 'result_ingester', metadata['instance_id'], + )['healthy'], 'remote_ingester_lease_required') + require(rows(db, 'SELECT COUNT(*) AS count FROM remote_worker_users')[0]['count'] == 0, + 'remote_fixture_database_not_fresh') + + capacity_names = ( + 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'keycheck_items', 'keycheck_bytes', 'quarantine_items', 'quarantine_bytes', + ) + baseline = db.pipeline_capacity_snapshot() + baseline = {name: int(baseline[name]) for name in capacity_names} + require(not any(baseline.values()), 'remote_capacity_not_drained') + limits = { + 'bundle_items': 2, 'bundle_bytes': 16384, + 'projection_items': 2, 'projection_bytes': 16384, + 'keycheck_items': 2, 'keycheck_bytes': 16384, + 'quarantine_items': 1, 'quarantine_bytes': 16384, + } + query = 'remote-recovery' + targets = ( + 'https://github.com/truf-e2e/remote-race.git', + 'https://github.com/truf-e2e/remote-spare.git', + ) + require(db.enqueue_targets('github', 'github', query, targets) == 2, + 'remote_fixture_enqueue') + + user_key = 'container-e2e-remote-user' + device_keys = ('container-e2e-device-a', 'container-e2e-device-b') + token_hashes = tuple(digest(name.encode('ascii')) for name in device_keys) + devices = [ + db.provision_remote_worker_device(user_key, key, token_hash, 1) + for key, token_hash in zip(device_keys, token_hashes) + ] + require(devices[0]['user_id'] == devices[1]['user_id'] + and devices[0]['device_id'] != devices[1]['device_id'] + and all(not device['revoked'] for device in devices), + 'remote_shared_user_fixture') + + scan_kwargs = { + 'timeout_sec': 30.0, + 'trufflehog_config': PACKAGE_DETECTOR_POLICY, + } + execution_limits = {'candidate_max_items': 10, 'candidate_max_bytes': 4096} + scan_policy = { + 'drop_detectors': [], + 'strict_git_provider_token_filter': True, + 'trufflehog_stdout_max_mb': 32, + 'trufflehog_stderr_max_mb': 8, + 'result_bundle_max_event_bytes': 1024 * 1024, + 'trufflehog_max_findings_per_target': 20000, + 'trufflehog_job_memory_limit_bytes': 0, + 'trufflehog_windows_job_cpu_weight': 0, + 'trufflehog_windows_memory_priority': 0, + 'trufflehog_diagnostic_max_lines': 2000, + 'trufflehog_diagnostic_max_line_chars': 8192, + 'trufflehog_diagnostic_max_line_bytes': 8192, + 'trufflehog_diagnostic_max_errors': 200, + 'trufflehog_diagnostic_max_warnings': 200, + 'trufflehog_diagnostic_max_unclassified': 20, + } + effective, execution = remote_execution_identity( + 'github', scan_kwargs, scan_kwargs, QueueDispositionPolicy(), + execution_limits, scan_policy, + ) + snapshot = { + 'schema': 1, + 'compatibility': { + 'protocol_version': PROTOCOL_VERSION, + 'bundle_format_version': FORMAT_VERSION, + 'platform_tag': 'linux-x86_64', + 'code_manifest_sha256': digest(b'remote-recovery-code'), + 'effective_config_sha256': effective, + 'detector_policy_sha256': digest(b'remote-recovery-policy'), + }, + 'execution': execution, + 'planning': { + 'kind': 'exact_git_v1', 'git_baseline_depth': 100, + 'git_ref_resolution_attempts': 2, + 'git_ref_resolution_timeout_sec': 10.0, + 'git_ref_resolution_max_bytes': 1 << 20, + }, + 'credential_ref': {'source': 'github', 'auth_entry': 'primary'}, + } + producer = current_process_identity() + requests = [{ + 'reservation_token': digest(f'remote-request-{index}'.encode('ascii')), + 'bundle_id': digest(f'remote-bundle-{index}'.encode('ascii')), + 'scan_event_id': digest(f'remote-event-{index}'.encode('ascii')), + 'device': devices[index], 'token_sha256': token_hashes[index], + 'client_compat_sha256': digest(f'remote-client-{index}'.encode('ascii')), + } for index in range(2)] + + def claim(connection, request): + return connection.reserve_and_claim_target( + 'github', 'github', producer, metadata['instance_id'], + 4096, 4096, 0, 0, lease_seconds=90, max_attempts=3, + capacity_limits=limits, + reservation_token=request['reservation_token'], + bundle_id=request['bundle_id'], + scan_event_id=request['scan_event_id'], claim_order='oldest', + final_cutover=True, remote_assignment={ + 'user_id': request['device']['user_id'], + 'device_id': request['device']['device_id'], + 'effective_config_sha256': effective, + 'client_compat_sha256': request['client_compat_sha256'], + 'token_sha256': request['token_sha256'], + 'result_upload_body_timeout_seconds': 1800, + 'execution_snapshot': snapshot, + }, + ) + + barrier = threading.Barrier(2) + claimed = [None, None] + failures = [] + + def race(index): + connection = scanner_db.ScannerDB(db_url=url, initialize=False) + connections.append(connection) + try: + connection.set_application_name( + f'truf-container-e2e:remote-race-{index}' + ) + barrier.wait(timeout=10) + claimed[index] = claim(connection, requests[index]) + except Exception: + failures.append(index) + finally: + connection.close() + + threads = [threading.Thread(target=race, args=(index,)) for index in range(2)] + for thread in threads: + thread.start() + for thread in threads: + thread.join(30) + require(not failures and all(not thread.is_alive() for thread in threads), + 'remote_quota_race_failed') + winners = [index for index, value in enumerate(claimed) if value is not None] + require(len(winners) == 1, 'remote_quota_not_atomic') + winner_index = winners[0] + loser_index = 1 - winner_index + winner = requests[winner_index] + active_claim = claimed[winner_index] + + intents = rows(db, '''SELECT reservation_token, state FROM admission_intents + WHERE reservation_token IN (?, ?) ORDER BY reservation_token''', + tuple(request['reservation_token'] for request in requests)) + require({row['state'] for row in intents} == {'committed', 'aborted'}, + 'remote_intent_race_resolution') + require(rows(db, '''SELECT COUNT(*) AS count FROM result_reservations + WHERE assignment_kind = 'remote' AND remote_user_id = ? + AND remote_resolved_at IS NULL''', (winner['device']['user_id'],))[0]['count'] == 1, + 'remote_shared_quota_count') + current_capacity = db.pipeline_capacity_snapshot() + expected_delta = { + 'bundle_items': 1, 'bundle_bytes': 4096, + 'projection_items': 1, 'projection_bytes': 4096, + 'keycheck_items': 0, 'keycheck_bytes': 0, + 'quarantine_items': 0, 'quarantine_bytes': 0, + } + require(all(int(current_capacity[name]) == baseline[name] + expected_delta[name] + for name in capacity_names), 'remote_capacity_single_charge') + + db.close() + db = scanner_db.ScannerDB(db_url=url, initialize=False) + db.set_application_name('truf-container-e2e:remote-reconcile') + recovered = db.reconcile_remote_assignment_request( + winner['reservation_token'], winner['device']['device_id'], + winner['token_sha256'], + ) + rejected = db.reconcile_remote_assignment_request( + requests[loser_index]['reservation_token'], + requests[loser_index]['device']['device_id'], + requests[loser_index]['token_sha256'], + ) + require(recovered['state'] == 'committed' + and recovered['claim'] == active_claim + and recovered['execution_snapshot'] == snapshot + and recovered['git_plan'] is None and recovered['receipt'] is None + and rejected == {'state': 'aborted'}, 'remote_lost_reply_recovery') + + before_lower = rows(db, '''SELECT remote_expires_at, producer_lease_expires_at + FROM result_reservations WHERE id = ?''', + (active_claim['reservation_id'],))[0] + lowered = db.provision_remote_worker_device( + user_key, winner['device']['device_key'], winner['token_sha256'], 0, + ) + authenticated = db.authenticate_remote_worker(winner['token_sha256']) + require(lowered['active_assignment_cap'] == 0 + and authenticated['active_assignment_cap'] == 0, + 'remote_lower_cap_not_applied') + require(claim(db, winner) == active_claim, + 'remote_lower_cap_cancelled_existing_claim') + require(not db.renew_result_claim( + active_claim['reservation_id'], active_claim['claim_lease_token'], 3600, + ), 'remote_claim_was_renewed') + require(rows(db, '''SELECT remote_expires_at, producer_lease_expires_at + FROM result_reservations WHERE id = ?''', + (active_claim['reservation_id'],))[0] == before_lower, + 'remote_contact_changed_deadline') + + resolution = { + 'provider': 'github', 'repo_url': active_claim['target'], + 'repo_path': 'truf-e2e/remote-race', 'branch': 'main', + 'ref': 'refs/heads/main', 'head_sha': 'a' * 40, + 'ref_source': 'provider_default', + } + credential = { + 'device_id': winner['device']['device_id'], + 'token_sha256': winner['token_sha256'], + } + plan = db.bind_git_scan_plan( + active_claim['reservation_id'], active_claim['claim_lease_token'], + resolution, 100, remote_credential=credential, + ) + require(plan['mode'] == 'baseline' + and db.remote_bound_git_scan_plan( + active_claim['reservation_id'], winner['device']['device_id'], + active_claim['claim_lease_token'], winner['token_sha256'], + ) == plan, 'remote_plan_recovery') + lease = rows(db, '''SELECT r.remote_issued_at, r.remote_expires_at, + r.producer_lease_expires_at, q.leased_at, q.lease_expires_at + FROM result_reservations r JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ?''', (active_claim['reservation_id'],))[0] + require(lease['remote_issued_at'] == lease['leased_at'] + and lease['remote_expires_at'] == lease['producer_lease_expires_at'] + == lease['lease_expires_at'], 'remote_dependent_lease_alignment') + require((scanner_db.parse_time(lease['remote_expires_at']) + - scanner_db.parse_time(lease['remote_issued_at'])).total_seconds() == 90, + 'remote_fixed_lease_window') + + with mock.patch.object( + scanner_db, 'utc_now_iso', return_value=lease['remote_expires_at'], + ): + expired = db.reap_expired_remote_assignments(limit=10) + require(len(expired) == 1 and expired[0]['resolution'] == 'expired' + and expired[0]['reservation_id'] == active_claim['reservation_id'], + 'remote_expiry_reaper') + expired_state = rows(db, '''SELECT r.state, r.remote_resolution_kind, + q.status, q.attempts, q.lease_token, q.current_result_reservation_id + FROM result_reservations r JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ?''', (active_claim['reservation_id'],))[0] + require(expired_state == { + 'state': 'refunded', 'remote_resolution_kind': 'expired', + 'status': 'pending', 'attempts': 0, 'lease_token': None, + 'current_result_reservation_id': None, + }, 'remote_expiry_refund_state') + require(all(int(db.pipeline_capacity_snapshot()[name]) == baseline[name] + for name in capacity_names), 'remote_expiry_capacity_release') + + db.provision_remote_worker_device( + user_key, winner['device']['device_key'], winner['token_sha256'], 1, + ) + reclaim_request = dict(winner) + reclaim_request.update({ + 'reservation_token': digest(b'remote-reclaim-request'), + 'bundle_id': digest(b'remote-reclaim-bundle'), + 'scan_event_id': digest(b'remote-reclaim-event'), + }) + reclaimed = claim(db, reclaim_request) + require(reclaimed is not None and reclaimed['queue_id'] == active_claim['queue_id'] + and reclaimed['reservation_id'] != active_claim['reservation_id'] + and reclaimed['claim_lease_token'] != active_claim['claim_lease_token'], + 'remote_same_worker_reclaim') + try: + db.bind_git_scan_plan( + active_claim['reservation_id'], active_claim['claim_lease_token'], + resolution, 100, remote_credential=credential, + ) + except scanner_db.ScanEventConflictError: + pass + else: + raise E2EFailure('remote_stale_plan_not_fenced') + try: + db.remote_bound_git_scan_plan( + active_claim['reservation_id'], winner['device']['device_id'], + active_claim['claim_lease_token'], winner['token_sha256'], + ) + except scanner_db.ScanEventConflictError: + pass + else: + raise E2EFailure('remote_stale_plan_recovery_not_fenced') + try: + db.report_remote_prebundle_failure( + active_claim['reservation_id'], winner['device']['device_id'], + winner['token_sha256'], { + 'failure_code': 'client_process_failed', 'detail': 'stale fixture', + }, + ) + except scanner_db.ScanEventConflictError: + pass + else: + raise E2EFailure('remote_stale_report_not_fenced') + current_fence = rows(db, '''SELECT status, lease_token, + current_result_reservation_id, claim_event_id + FROM target_queue WHERE id = ?''', (reclaimed['queue_id'],))[0] + require(current_fence == { + 'status': 'in_progress', 'lease_token': reclaimed['claim_lease_token'], + 'current_result_reservation_id': reclaimed['reservation_id'], + 'claim_event_id': reclaimed['scan_event_id'], + }, 'remote_stale_response_changed_reclaim') + + report = { + 'failure_code': 'client_process_failed', + 'detail': 'synthetic remote recovery fixture', + } + receipt = db.report_remote_prebundle_failure( + reclaimed['reservation_id'], winner['device']['device_id'], + winner['token_sha256'], report, + ) + require(receipt['resolution'] == 'prebundle_report' + and db.report_remote_prebundle_failure( + reclaimed['reservation_id'], winner['device']['device_id'], + winner['token_sha256'], report, + ) == receipt, 'remote_terminal_report_replay') + try: + db.report_remote_prebundle_failure( + reclaimed['reservation_id'], winner['device']['device_id'], + winner['token_sha256'], { + 'failure_code': 'client_storage_failed', 'detail': 'conflict fixture', + }, + ) + except scanner_db.ScanEventConflictError: + pass + else: + raise E2EFailure('remote_conflicting_report_not_fenced') + require(all(int(db.pipeline_capacity_snapshot()[name]) == baseline[name] + for name in capacity_names), 'remote_terminal_capacity_release') + require(rows(db, '''SELECT COUNT(*) AS count FROM result_reservations + WHERE assignment_kind = 'remote' AND remote_user_id = ? + AND remote_resolved_at IS NULL''', (winner['device']['user_id'],))[0]['count'] == 0, + 'remote_quota_not_released_once') + final_queue = rows(db, '''SELECT status, attempts, lease_token, + current_result_reservation_id FROM target_queue WHERE id = ?''', + (reclaimed['queue_id'],))[0] + require(final_queue == { + 'status': 'pending', 'attempts': 0, 'lease_token': None, + 'current_result_reservation_id': None, + }, 'remote_terminal_queue_release') + evidence = { + 'race_devices': 2, 'admitted': 1, 'expired': 1, + 'same_worker_reclaims': 1, 'terminal_replays': 1, + } + return { + 'counts': {'ok': 1, **evidence}, + 'hashes': {'remote_recovery_sha256': digest(json_bytes(evidence))}, + } + finally: + for connection in connections: + connection.close() + db.close() + + +def prepare_remote_transport(config, marker, url, timeout): + del config + import asyncio + from process_identity import current_process_identity + from result_bundle import bundle_partial_path, bundle_ready_path + from runtime_security import ensure_private_directory, write_private_json_exclusive + from scanner import stage_result_bundle + import scanner_db + from unittest import mock + from worker_api import WorkerService + + metadata = control_snapshot(marker, url) + db = scanner_db.ScannerDB(db_url=url, initialize=False) + stage = 'fixture' + try: + db.set_application_name('truf-container-e2e:remote-transport') + capacity_names = ( + 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'keycheck_items', 'keycheck_bytes', 'quarantine_items', 'quarantine_bytes', + ) + baseline = {name: int(db.pipeline_capacity_snapshot()[name]) for name in capacity_names} + require(not any(baseline.values()), 'remote_transport_capacity_not_drained') + before_scans = rows(db, 'SELECT COUNT(*) AS count FROM target_scans')[0]['count'] + + device_key = 'container-e2e-device-a' + token_sha256 = digest(device_key.encode('ascii')) + fixture = rows(db, '''SELECT u.id AS user_id, d.id AS device_id + FROM remote_worker_users u JOIN remote_worker_devices d ON d.user_id = u.id + WHERE u.user_key = ? AND d.device_key = ?''', + ('container-e2e-remote-user', device_key)) + require(len(fixture) == 1, 'remote_transport_device_fixture') + device = db.provision_remote_worker_device( + 'container-e2e-remote-user', device_key, token_sha256, 1, + ) + require(device['user_id'] == fixture[0]['user_id'] + and device['device_id'] == fixture[0]['device_id'], + 'remote_transport_device_identity') + prior = rows(db, '''SELECT remote_execution_snapshot_json, + remote_effective_config_sha256 + FROM result_reservations + WHERE assignment_kind = 'remote' AND remote_user_id = ? + ORDER BY id LIMIT 1''', (device['user_id'],)) + require(len(prior) == 1, 'remote_transport_snapshot_fixture') + snapshot = json.loads(prior[0]['remote_execution_snapshot_json']) + effective = str(prior[0]['remote_effective_config_sha256']) + + stage = 'claim' + reservation_token = digest(b'remote-transport-request') + bundle_id = digest(b'remote-transport-bundle') + scan_event_id = digest(b'remote-transport-event') + declared_bytes = 16384 + limits = { + 'bundle_items': 1, 'bundle_bytes': declared_bytes, + 'projection_items': 1, 'projection_bytes': declared_bytes, + 'keycheck_items': 1, 'keycheck_bytes': declared_bytes, + 'quarantine_items': 1, 'quarantine_bytes': declared_bytes, + } + claimed = db.reserve_and_claim_target( + 'github', 'github', current_process_identity(), metadata['instance_id'], + declared_bytes, declared_bytes, 0, 0, lease_seconds=300, + max_attempts=3, capacity_limits=limits, + reservation_token=reservation_token, bundle_id=bundle_id, + scan_event_id=scan_event_id, claim_order='oldest', final_cutover=True, + remote_assignment={ + 'user_id': device['user_id'], 'device_id': device['device_id'], + 'effective_config_sha256': effective, + 'client_compat_sha256': digest(b'remote-transport-client'), + 'token_sha256': token_sha256, + 'result_upload_body_timeout_seconds': 1800, + 'execution_snapshot': snapshot, + }, + ) + require(claimed is not None, 'remote_transport_claim') + repo_path = str(claimed['target']).split('github.com/', 1)[-1] + if repo_path.endswith('.git'): + repo_path = repo_path[:-4] + resolution = { + 'provider': 'github', 'repo_url': claimed['target'], + 'repo_path': repo_path, 'branch': 'main', 'ref': 'refs/heads/main', + 'head_sha': 'b' * 40, 'ref_source': 'provider_default', + } + plan = db.bind_git_scan_plan( + claimed['reservation_id'], claimed['claim_lease_token'], resolution, 100, + remote_credential={ + 'device_id': device['device_id'], 'token_sha256': token_sha256, + }, + ) + stage = 'bundle_plan' + plan_sha256 = digest(scanner_db.canonical_git_scan_plan_bytes(plan)) + exact_scope = { + key: plan.get(key) for key in ( + 'provider', 'ref', 'head_sha', 'base_sha', 'mode', + 'baseline_depth', 'ref_source', + ) + } + stage = 'bundle_directory' + ensure_private_directory(str(REMOTE_CLIENT_BUNDLES), reject_reparse=True) + stage = 'bundle_stage' + staged = stage_result_bundle( + { + 'scan_event_id': claimed['scan_event_id'], + 'target': claimed['target'], 'scan_type': 'github', + 'timestamp': '2026-09-18T00:00:00+00:00', + 'scan_started_at': '2026-09-18T00:00:00+00:00', + 'duration_sec': 1.0, 'findings': [], 'errors': [], + 'git_scan_plan': plan, + 'git_scan_execution': { + 'mode': plan['mode'], 'pinned': True, 'success': True, + 'continuity_reset': False, 'plan_sha256': plan_sha256, + }, + 'scan_meta': {'exact_git_scope': exact_scope}, + }, + claimed, str(REMOTE_CLIENT_BUNDLES), {}, + {'queue_status': 'done', 'queue_error': None, 'available_after': None}, + candidate_max_items=0, candidate_max_bytes=0, + ) + stage = 'bundle_read' + client_ready = Path(bundle_ready_path(str(REMOTE_CLIENT_BUNDLES), bundle_id)) + payload = read_bytes(client_ready, limit=declared_bytes) + payload_sha256 = digest(payload) + require(staged.actual_bytes == len(payload) and len(payload) < declared_bytes, + 'remote_transport_client_bundle_bound') + + service = WorkerService( + url, str(BUNDLES), lambda *_args: None, max_bundle_bytes=declared_bytes, + ) + server_ready = bundle_ready_path(str(BUNDLES), bundle_id) + server_partial = bundle_partial_path(str(BUNDLES), bundle_id, reservation_token) + + def upload(body=payload, **kwargs): + return asyncio.run(invoke_bundle_upload( + service, device_key, claimed['reservation_id'], body, **kwargs, + )) + + stage = 'negative_upload' + status, code, _ = upload( + payload[:-1], content_length=len(payload), + payload_sha256=digest(payload[:-1]), + ) + require((status, code) == (400, 'length_mismatch'), + 'remote_transport_truncation_not_rejected') + + corrupted = bytearray(payload) + corrupted[len(corrupted) // 2] ^= 1 + status, code, _ = upload(bytes(corrupted)) + require((status, code) == (400, 'invalid_bundle'), + 'remote_transport_invalid_bundle_not_rejected') + try: + upload(disconnect_after=257) + except ConnectionError: + pass + else: + raise E2EFailure('remote_transport_disconnect_not_rejected') + require(not os.path.lexists(server_ready) and not os.path.lexists(server_partial), + 'remote_transport_failed_upload_not_cleaned') + + stage = 'deadline' + with mock.patch.object( + scanner_db, 'utc_now_iso', return_value=claimed['remote_expires_at'], + ): + status, code, _ = upload() + require((status, code) == (410, 'assignment_expired') + and not os.path.lexists(server_ready), + 'remote_transport_deadline_crossing_accepted') + + stage = 'publication' + with mock.patch.object( + service, 'accept_ready', side_effect=RuntimeError('synthetic ready failure'), + ): + try: + upload() + except RuntimeError: + pass + else: + raise E2EFailure('remote_transport_ready_crash_not_injected') + unresolved = rows(db, '''SELECT state, remote_resolution_kind + FROM result_reservations WHERE id = ?''', (claimed['reservation_id'],))[0] + require(unresolved == {'state': 'scanning', 'remote_resolution_kind': None} + and os.path.isfile(server_ready), + 'remote_transport_publication_not_recoverable') + + service = WorkerService( + url, str(BUNDLES), lambda *_args: None, max_bundle_bytes=declared_bytes, + ) + + stage = 'ready_commit' + committed_receipts = [] + accept_ready = service.accept_ready + + def accept_then_crash(*args): + committed_receipts.append(accept_ready(*args)) + raise RuntimeError('synthetic lost ready commit acknowledgement') + + with mock.patch.object(service, 'accept_ready', side_effect=accept_then_crash): + try: + upload() + except RuntimeError: + pass + else: + raise E2EFailure('remote_transport_commit_crash_not_injected') + require(len(committed_receipts) == 1 + and committed_receipts[0]['resolution'] == 'bundle_accepted', + 'remote_transport_ready_commit_not_durable') + + stage = 'retry' + + async def simultaneous_retry(): + return await asyncio.gather(*( + invoke_bundle_upload( + service, device_key, claimed['reservation_id'], payload, + payload_sha256=payload_sha256, + ) for _ in range(2) + )) + + retries = asyncio.run(simultaneous_retry()) + require(all(status == 200 and code is None for status, code, _ in retries) + and retries[0][2] == retries[1][2] == committed_receipts[0], + 'remote_transport_simultaneous_retry') + receipt = retries[0][2] + require(receipt['resolution'] == 'bundle_accepted' + and receipt['payload_sha256'] == payload_sha256, + 'remote_transport_receipt_identity') + status, code, _ = asyncio.run(invoke_bundle_upload( + service, device_key, claimed['reservation_id'], payload + b'x', + )) + require((status, code) == (409, 'resolution_conflict'), + 'remote_transport_conflicting_replay') + + stage = 'ingestion' + deadline = time.monotonic() + timeout + complete = None + while time.monotonic() < deadline: + state = rows(db, '''SELECT r.state, r.remote_resolution_kind, + r.remote_resolution_json, b.state AS bundle_state, + q.status AS queue_status, q.target_scan_id + FROM result_reservations r + JOIN result_bundles b ON b.reservation_id = r.id + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ?''', (claimed['reservation_id'],)) + capacity = db.pipeline_capacity_snapshot() + if ( + len(state) == 1 and state[0]['state'] == 'acknowledged' + and state[0]['bundle_state'] == 'acknowledged' + and state[0]['queue_status'] == 'done' + and state[0]['target_scan_id'] is not None + and not os.path.lexists(server_ready) + and all(int(capacity[name]) == baseline[name] for name in capacity_names) + ): + complete = state[0] + break + time.sleep(min(0.25, max(0, deadline - time.monotonic()))) + require(complete is not None, 'remote_transport_ingestion_timeout') + require(json.loads(complete['remote_resolution_json']) == receipt, + 'remote_transport_receipt_changed_during_ingestion') + require(rows(db, 'SELECT COUNT(*) AS count FROM target_scans WHERE scan_event_id = ?', + (scan_event_id,))[0]['count'] == 1 + and rows(db, 'SELECT COUNT(*) AS count FROM scan_result_compat WHERE target_scan_id = ?', + (complete['target_scan_id'],))[0]['count'] == 1 + and rows(db, 'SELECT COUNT(*) AS count FROM target_scans')[0]['count'] == before_scans + 1, + 'remote_transport_not_ingested_once') + require(rows(db, '''SELECT COUNT(*) AS count FROM result_reservations + WHERE assignment_kind = 'remote' AND remote_user_id = ? + AND remote_resolved_at IS NULL''', (device['user_id'],))[0]['count'] == 0, + 'remote_transport_quota_not_released') + status_receipt = service.status({ + 'device_id': device['device_id'], 'token_sha256': token_sha256, + }, claimed['reservation_id']) + require(status_receipt == receipt, 'remote_transport_status_receipt') + + stage = 'evidence' + evidence = { + 'schema': 1, 'reservation_id': claimed['reservation_id'], + 'bundle_id': bundle_id, 'scan_event_id': scan_event_id, + 'payload_sha256': payload_sha256, 'receipt': receipt, + 'target_scan_id': complete['target_scan_id'], + 'before_scans': before_scans, + } + write_private_json_exclusive(str(REMOTE_TRANSPORT), evidence) + public = { + 'truncated_rejected': 1, 'invalid_rejected': 1, + 'disconnect_rejected': 1, 'deadline_rejected': 1, + 'publication_recovered': 1, 'simultaneous_replays': 2, + 'commit_reply_recovered': 1, 'ingested': 1, + } + return { + 'counts': {'ok': 1, **public}, + 'hashes': {'remote_transport_sha256': digest(json_bytes(public))}, + } + except E2EFailure: + raise + except Exception: + raise E2EFailure('remote_transport_' + stage + '_exception') from None + finally: + db.close() + + +def assert_remote_transport_replay(config, marker, url, timeout): + del config + import asyncio + from result_bundle import bundle_ready_path + import worker_api + from unittest import mock + + deadline = time.monotonic() + timeout + while True: + try: + control_snapshot(marker, url) + break + except NotReady: + if time.monotonic() >= deadline: + raise + time.sleep(min(0.25, max(0, deadline - time.monotonic()))) + evidence = json.loads(read_bytes(REMOTE_TRANSPORT)) + require(evidence.get('schema') == 1, 'remote_transport_evidence_required') + db = worker_api.ScannerDB(db_url=url, initialize=False) + try: + db.set_application_name('truf-container-e2e:remote-transport-replay') + owner = rows(db, '''SELECT d.device_key, d.token_sha256, r.remote_expires_at, + r.remote_resolution_json, r.state, r.queue_id + FROM result_reservations r + JOIN remote_worker_devices d ON d.id = r.remote_device_id + WHERE r.id = ?''', (evidence['reservation_id'],)) + require(len(owner) == 1 and owner[0]['state'] == 'acknowledged', + 'remote_transport_restart_state') + client_ready = Path(bundle_ready_path( + str(REMOTE_CLIENT_BUNDLES), evidence['bundle_id'], + )) + payload = read_bytes(client_ready, limit=16384) + require(digest(payload) == evidence['payload_sha256'], + 'remote_transport_client_bundle_changed') + service = worker_api.WorkerService( + url, str(BUNDLES), lambda *_args: None, max_bundle_bytes=16384, + ) + with mock.patch.object(worker_api, 'utc_now_iso', return_value='9999-12-31T23:59:59+00:00'): + status, code, receipt = asyncio.run(invoke_bundle_upload( + service, owner[0]['device_key'], evidence['reservation_id'], payload, + )) + require(status == 200 and code is None and receipt == evidence['receipt'], + 'remote_transport_restart_replay') + status, code, _ = asyncio.run(invoke_bundle_upload( + service, owner[0]['device_key'], evidence['reservation_id'], payload + b'x', + )) + require((status, code) == (409, 'resolution_conflict'), + 'remote_transport_restart_conflict') + require(json.loads(owner[0]['remote_resolution_json']) == evidence['receipt'] + and service.status({ + 'device_id': rows(db, '''SELECT remote_device_id FROM result_reservations + WHERE id = ?''', (evidence['reservation_id'],))[0]['remote_device_id'], + 'token_sha256': owner[0]['token_sha256'], + }, evidence['reservation_id']) == evidence['receipt'], + 'remote_transport_restart_receipt_changed') + require(not os.path.lexists(bundle_ready_path(str(BUNDLES), evidence['bundle_id'])) + and rows(db, 'SELECT COUNT(*) AS count FROM target_scans WHERE scan_event_id = ?', + (evidence['scan_event_id'],))[0]['count'] == 1 + and rows(db, '''SELECT COUNT(*) AS count FROM target_scans + WHERE id = ? AND scan_event_id = ?''', + (evidence['target_scan_id'], evidence['scan_event_id']))[0]['count'] == 1 + and rows(db, 'SELECT COUNT(*) AS count FROM result_bundles WHERE reservation_id = ?', + (evidence['reservation_id'],))[0]['count'] == 1, + 'remote_transport_restart_exactly_once') + capacity = db.pipeline_capacity_snapshot() + require(not any(int(capacity[name]) for name in ( + 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'keycheck_items', 'keycheck_bytes', 'quarantine_items', 'quarantine_bytes', + )), 'remote_transport_restart_capacity') + public = {'receipt_replayed': 1, 'conflict_rejected': 1, 'exactly_once': 1} + return { + 'counts': {'ok': 1, **public}, + 'hashes': {'remote_transport_replay_sha256': digest(json_bytes(public))}, + } + finally: + db.close() + + +def prepare_remote_full_race(config, marker, url): + import asyncio + from result_bundle import bundle_partial_path, bundle_ready_path + from runtime_security import ensure_private_directory, write_private_json_exclusive + import scanner_db + import worker_api + from unittest import mock + + metadata = control_snapshot(marker, url) + baseline = json.loads(read_bytes(RESULT)) + require(baseline.get('schema') == 1 and baseline.get('checked') == 1, + 'remote_full_checked_local_baseline') + require(not os.path.lexists(REMOTE_FULL), 'remote_full_evidence_exists') + ensure_private_directory(str(REMOTE_CLIENT_BUNDLES), reject_reparse=True) + ensure_private_directory(str(REMOTE_LOCAL_BUNDLES), reject_reparse=True) + db = scanner_db.ScannerDB(db_url=url, initialize=False) + stage = 'database' + try: + db.set_application_name('truf-container-e2e:remote-full-prepare') + capacity_names = ( + 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'keycheck_items', 'keycheck_bytes', 'quarantine_items', 'quarantine_bytes', + ) + require(not any(int(db.pipeline_capacity_snapshot()[name]) for name in capacity_names), + 'remote_full_initial_capacity') + user_key = 'container-e2e-full-user' + require(not rows(db, 'SELECT id FROM remote_worker_users WHERE user_key = ?', + (user_key,)), 'remote_full_user_not_fresh') + require(db.enqueue_targets('gitlab', 'gitlab', 'e2e', [REMOTE_TARGET]) == 1, + 'remote_full_requeue') + + devices = {} + stage = 'devices' + for role in ('expired', 'winner'): + token = remote_device_token(role) + device = db.provision_remote_worker_device( + user_key, 'container-e2e-full-' + role, + digest(token.encode('ascii')), 1, + ) + require(device and not device['revoked'], 'remote_full_device_provision') + devices[role] = device + + stage = 'service' + service, build = remote_worker_service(config, marker, metadata, url) + + def claim(role): + status, response = asyncio.run(invoke_worker_claim( + service, remote_device_token(role), + digest(('remote-full-' + role + '-request').encode('ascii')), build, + )) + require(status == 201 and set(response) == {'assignment'}, + 'remote_full_claim_transport') + assignment = response['assignment'] + require(assignment['reservation']['target'] == REMOTE_TARGET + and assignment['reservation']['remote_device_id'] == devices[role]['device_id'], + 'remote_full_claim_identity') + return assignment + + stage = 'expired_claim' + expired_assignment = claim('expired') + expired = expired_assignment['reservation'] + stage = 'expired_scan' + expired_payload, expired_bundle, _ = execute_fixture_assignment( + config, metadata, expired_assignment, compare_local=False, + ) + if ( + expired_bundle['finding_count'], expired_bundle['candidate_count'], + expired_bundle['error_count'], expired_bundle['queue_status'], + ) != (1, 1, 0, 'done'): + status = re.sub(r'[^a-z0-9_]', '_', expired_bundle['queue_status'])[:20] + category = re.sub( + r'[^a-z0-9_]', '_', expired_bundle['source_failure_category'], + )[:30] + error_text = str(expired_bundle['first_error'] or '').lower() + error_classes = '_'.join( + label for marker, label in ( + ('protocol', 'protocol'), ('error preparing repo', 'prepare_repo'), + ('clone', 'clone'), ('branch', 'branch'), ('reference', 'reference'), + ('commit', 'commit'), ('exit status', 'exit_status'), + ('authentication', 'auth'), ('certificate', 'certificate'), + ('config', 'config'), ('permission', 'permission'), + ('no such file', 'missing_file'), ('deadline', 'deadline'), + ('timed out', 'timeout'), ('scan-slot', 'scan_slot'), + ('authority', 'authority'), ('supervisor', 'supervisor'), + ('lifecycle', 'lifecycle'), ('executable', 'executable'), + ('completion', 'completion'), ('error running scan', 'running_scan'), + ('unable', 'unable'), ('failed', 'failed'), ('private', 'private'), + ) if marker in error_text + )[:40] or 'unclassified' + raise E2EFailure( + f'remote_full_expired_counts_{expired_bundle["finding_count"]}_' + f'{expired_bundle["candidate_count"]}_{expired_bundle["error_count"]}_' + f'{status}_{int(expired_bundle["source_failure"])}_{category}_{error_classes}' + ) + stage = 'partial_upload' + try: + asyncio.run(invoke_bundle_upload( + service, remote_device_token('expired'), expired['reservation_id'], + expired_payload, disconnect_after=257, + )) + except ConnectionError: + pass + else: + raise E2EFailure('remote_full_disconnect_not_observed') + expired_partial = bundle_partial_path( + str(BUNDLES), expired['bundle_id'], expired['reservation_token'], + ) + require(not os.path.lexists(expired_partial) + and not os.path.lexists(bundle_ready_path(str(BUNDLES), expired['bundle_id'])), + 'remote_full_partial_not_cleaned') + + stage = 'exact_deadline' + with mock.patch.object( + worker_api, 'utc_now_iso', return_value=expired['remote_expires_at'], + ): + status, code, _ = asyncio.run(invoke_bundle_upload( + service, remote_device_token('expired'), expired['reservation_id'], + expired_payload, + )) + require((status, code) == (410, 'assignment_expired'), + 'remote_full_exact_expiry_upload') + stage = 'expiry_reaper' + with mock.patch.object( + scanner_db, 'utc_now_iso', return_value=expired['remote_expires_at'], + ): + expiry_receipts = service.reap() + require(len(expiry_receipts) == 1 + and expiry_receipts[0]['reservation_id'] == expired['reservation_id'] + and expiry_receipts[0]['resolution'] == 'expired' + and expiry_receipts[0]['resolved_at'] == expired['remote_expires_at'], + 'remote_full_exact_expiry_reap') + + stage = 'winner_claim' + winner_assignment = claim('winner') + winner = winner_assignment['reservation'] + require(winner['queue_id'] == expired['queue_id'] + and winner['reservation_id'] != expired['reservation_id'] + and winner['reservation_token'] != expired['reservation_token'] + and winner['claim_lease_token'] != expired['claim_lease_token'] + and winner['scan_event_id'] != expired['scan_event_id'] + and winner['bundle_id'] != expired['bundle_id'], + 'remote_full_reissue_identity') + stage = 'stale_report' + status, code, _ = asyncio.run(invoke_terminal_report( + service, remote_device_token('expired'), expired['reservation_id'], { + 'failure_code': 'client_process_failed', + 'detail': 'synthetic stale issuance fixture', + }, + )) + require((status, code) == (409, 'reservation_conflict'), + 'remote_full_stale_report_fence') + + stage = 'winner_scan' + winner_payload, winner_bundle, parity_sha256 = execute_fixture_assignment( + config, metadata, winner_assignment, compare_local=True, + ) + if ( + winner_bundle['finding_count'], winner_bundle['candidate_count'], + winner_bundle['error_count'], winner_bundle['queue_status'], bool(parity_sha256), + ) != (1, 1, 0, 'done', True): + status = re.sub(r'[^a-z0-9_]', '_', winner_bundle['queue_status'])[:20] + raise E2EFailure( + f'remote_full_winner_counts_{winner_bundle["finding_count"]}_' + f'{winner_bundle["candidate_count"]}_{winner_bundle["error_count"]}_' + f'{status}_{int(bool(parity_sha256))}' + ) + plan = winner_assignment['scan_kwargs']['git_plan'] + expired_plan = expired_assignment['scan_kwargs']['git_plan'] + require(plan == expired_plan and plan['ref'] == 'refs/heads/main' + and plan['head_sha'] == marker['repo_commit'] + and plan['mode'] == 'baseline' and plan['baseline_depth'] == 1, + 'remote_full_exact_plans') + stage = 'plan_bindings' + bound = rows(db, '''SELECT id, git_scan_plan_json, git_scan_plan_sha256 + FROM result_reservations WHERE id IN (?, ?) ORDER BY id''', + (expired['reservation_id'], winner['reservation_id'])) + plan_sha256 = digest(scanner_db.canonical_git_scan_plan_bytes(plan)) + require(len(bound) == 2 + and all(json.loads(row['git_scan_plan_json']) == plan + and row['git_scan_plan_sha256'] == plan_sha256 for row in bound), + 'remote_full_separate_plan_bindings') + stage = 'capacity' + capacity = db.pipeline_capacity_snapshot() + expected_capacity = { + 'bundle_items': 1, 'bundle_bytes': MAX_BYTES, + 'projection_items': 1, 'projection_bytes': 2 * MAX_BYTES, + 'keycheck_items': 8, 'keycheck_bytes': MAX_BYTES, + 'quarantine_items': 0, 'quarantine_bytes': 0, + } + require(all(int(capacity[name]) == value + for name, value in expected_capacity.items()), + 'remote_full_reissue_capacity') + interim = db.admin_remote_worker_snapshot(limit=200) + workers = { + row['device_key']: row for row in interim['workers'] + if row['user_key'] == user_key + } + require(set(workers) == { + 'container-e2e-full-expired', 'container-e2e-full-winner', + } and { + key: int(workers['container-e2e-full-expired'][key]) + for key in ('unfinished_count', 'completed_count', 'failed_count', 'expired_count') + } == { + 'unfinished_count': 0, 'completed_count': 0, + 'failed_count': 0, 'expired_count': 1, + } and { + key: int(workers['container-e2e-full-winner'][key]) + for key in ('unfinished_count', 'completed_count', 'failed_count', 'expired_count') + } == { + 'unfinished_count': 1, 'completed_count': 0, + 'failed_count': 0, 'expired_count': 0, + }, 'remote_full_interim_statistics') + + stage = 'evidence' + evidence = { + 'schema': 1, 'user_key': user_key, + 'origin_instance_sha256': digest(metadata['instance_id'].encode('utf-8')), + 'queue_id': winner['queue_id'], + 'expired': { + 'reservation_id': expired['reservation_id'], + 'bundle_id': expired['bundle_id'], 'scan_event_id': expired['scan_event_id'], + 'payload_sha256': digest(expired_payload), + 'lease_sha256': digest(expired['claim_lease_token'].encode('utf-8')), + }, + 'winner': { + 'reservation_id': winner['reservation_id'], + 'bundle_id': winner['bundle_id'], 'scan_event_id': winner['scan_event_id'], + 'payload_sha256': digest(winner_payload), + 'lease_sha256': digest(winner['claim_lease_token'].encode('utf-8')), + }, + 'plan': plan, 'plan_sha256': plan_sha256, + 'parity_sha256': parity_sha256, + 'baseline_keycheck_result_id': baseline['ids']['keycheck_result_id'], + } + write_private_json_exclusive(str(REMOTE_FULL), evidence) + public = { + 'real_claims': 2, 'native_scans': 3, 'partial_disconnects': 1, + 'exact_expiries': 1, 'stale_reports': 1, 'reissues': 1, + 'parity_comparisons': 1, 'pending_restart': 1, + } + return { + 'counts': {'ok': 1, **public}, + 'hashes': { + 'remote_full_prepare_sha256': digest(json_bytes(public)), + 'remote_full_parity_sha256': parity_sha256, + }, + } + except E2EFailure: + raise + except Exception as exc: + trace = exc.__traceback__ + origins = [] + while trace: + origins.append(trace.tb_frame.f_code.co_name) + trace = trace.tb_next + origin = '_'.join([type(exc).__name__, *origins[-3:]]) + origin = re.sub(r'[^a-z0-9_]', '_', origin.lower())[:70] + raise E2EFailure('remote_full_prepare_' + stage + '_' + origin) from None + finally: + db.close() + + +def remote_full_snapshot(config, marker, url, evidence, checked): + from keycheck_candidates import candidate_uid, extract_candidates + from result_bundle import bundle_ready_path + import scanner_db + + metadata = control_snapshot(marker, url) + require(digest(metadata['instance_id'].encode('utf-8')) + != evidence['origin_instance_sha256'], 'remote_full_runtime_restart') + db = scanner_db.ScannerDB(db_url=url, initialize=False) + try: + db.set_application_name('truf-container-e2e:remote-full-snapshot') + for worker in ('result_ingester', 'jsonl_projector'): + if not db.pipeline_worker_health(worker, metadata['instance_id'])['healthy']: + raise NotReady('remote_full_pipeline_lease') + reservations = rows(db, '''SELECT r.*, b.reservation_id AS result_bundle_id, + b.state AS bundle_state, b.actual_bytes AS bundle_actual_bytes, + b.finding_count AS bundle_finding_count, + b.error_count AS bundle_error_count, + b.candidate_count AS bundle_candidate_count + FROM result_reservations r + JOIN remote_worker_users u ON u.id = r.remote_user_id + LEFT JOIN result_bundles b ON b.reservation_id = r.id + WHERE u.user_key = ? ORDER BY r.id''', (evidence['user_key'],)) + require(len(reservations) == 2, 'remote_full_reservation_count') + by_id = {row['id']: row for row in reservations} + expired = by_id[evidence['expired']['reservation_id']] + winner = by_id[evidence['winner']['reservation_id']] + require(expired['state'] == 'refunded' + and expired['remote_resolution_kind'] == 'expired' + and expired['bundle_credit_released'] == 1 + and expired['projection_credit_transferred'] == 0 + and expired['candidate_credit_transferred'] == 0 + and expired['result_bundle_id'] is None, + 'remote_full_expired_credit_state') + if winner['state'] != 'acknowledged' or winner['bundle_state'] != 'acknowledged': + raise NotReady('remote_full_ingestion') + require(winner['remote_resolution_kind'] == 'bundle_accepted' + and winner['bundle_credit_released'] == 1 + and winner['projection_credit_transferred'] == 1 + and winner['candidate_credit_transferred'] == 1 + and winner['bundle_finding_count'] == winner['bundle_candidate_count'] == 1 + and winner['bundle_error_count'] == 0 + and 0 < int(winner['bundle_actual_bytes']) < MAX_BYTES, + 'remote_full_winner_credit_state') + receipt = json.loads(winner['remote_resolution_json']) + require(receipt['resolution'] == 'bundle_accepted' + and receipt['reservation_id'] == winner['id'] + and receipt['bundle_id'] == evidence['winner']['bundle_id'] + and receipt['scan_event_id'] == evidence['winner']['scan_event_id'] + and receipt['payload_sha256'] == evidence['winner']['payload_sha256'], + 'remote_full_receipt_identity') + require(not os.path.lexists(bundle_ready_path(str(BUNDLES), expired['bundle_id'])) + and not os.path.lexists(bundle_ready_path(str(BUNDLES), winner['bundle_id'])), + 'remote_full_server_spool_cleanup') + + scans = rows(db, '''SELECT s.* FROM target_scans s + JOIN result_reservations r ON r.id = s.result_reservation_id + JOIN remote_worker_users u ON u.id = r.remote_user_id + WHERE u.user_key = ? ORDER BY s.id''', (evidence['user_key'],)) + require(len(scans) == 1 and scans[0]['result_reservation_id'] == winner['id'], + 'remote_full_authoritative_scan_count') + scan = scans[0] + queue = rows(db, 'SELECT * FROM target_queue WHERE id = ?', + (evidence['queue_id'],))[0] + require(queue['status'] == 'done' and queue['target_scan_id'] == scan['id'] + and queue['attempts'] == 1 and queue['lease_token'] is None + and queue['current_result_reservation_id'] is None + and queue['claim_event_id'] is None and not queue['last_error'], + 'remote_full_queue_disposition') + require(queue['covered_ref'] == evidence['plan']['ref'] + and queue['covered_head'] == evidence['plan']['head_sha'], + 'remote_full_winning_plan_coverage') + require(scan['scan_event_id'] == evidence['winner']['scan_event_id'] + and scan['status'] == 'found' and scan['findings_count'] == 1 + and scan['error_count'] == 0 + and scan['queue_completion_disposition'] == 'applied', + 'remote_full_scan_result') + rebuilt = db.reconstruct_scan_result( + scan['id'], max_bytes=MAX_BYTES, max_findings=2, max_errors=0, + ) + require(rebuilt and len(rebuilt['findings']) == 1 and rebuilt['errors'] == [], + 'remote_full_reconstructed_result') + finding = rows(db, 'SELECT * FROM findings WHERE target_scan_id = ?', + (scan['id'],)) + candidate = rows(db, 'SELECT * FROM keycheck_candidates WHERE target_scan_id = ?', + (scan['id'],)) + require(len(finding) == len(candidate) == 1, 'remote_full_normalized_counts') + finding, candidate = finding[0], candidate[0] + rebuilt_finding = rebuilt['findings'][0] + specs = list(extract_candidates(rebuilt_finding)) + require(len(specs) == 1 and specs[0].service == 'openai' + and rebuilt_finding['Raw'] == synthetic_openai_token() + and finding['file_path'] == 'synthetic.env' + and finding['commit_hash'] == marker['repo_commit'] + and candidate['finding_id'] == finding['id'] + and candidate['finding_uid'] == finding['finding_uid'] + and candidate['candidate_uid'] == candidate_uid( + scan['scan_event_id'], finding['finding_uid'], 'openai', + specs[0].credential_hash, + ), 'remote_full_finding_candidate_identity') + require(rebuilt['git_scan_plan'] == evidence['plan'] + and rebuilt['git_scan_execution']['success'] is True + and rebuilt['git_scan_execution']['pinned'] is True + and rebuilt['git_scan_execution']['plan_sha256'] == evidence['plan_sha256'] + and rebuilt['scan_meta']['exact_git_scope']['ref'] == evidence['plan']['ref'] + and rebuilt['scan_meta']['exact_git_scope']['head_sha'] == evidence['plan']['head_sha'], + 'remote_full_persisted_exact_metadata') + + scan_jobs = rows(db, 'SELECT * FROM projection_jobs WHERE target_scan_id = ?', + (scan['id'],)) + if len(scan_jobs) != 1 or scan_jobs[0]['status'] != 'completed': + raise NotReady('remote_full_scan_projection') + scan_job = scan_jobs[0] + require(scan_job['event_id'] == scan['scan_event_id'] + and scan_job['event_hash'] == scan['scan_event_hash'] + and scan_job['required_stream_mask'] == 3 + and scan_job['capacity_released'] == 1, + 'remote_full_scan_projection_lineage') + projection_hashes = {} + for stream, relative, expected in ( + ('scan_results', 'scan_results.jsonl', json_bytes(rebuilt) + b'\n'), + ('found_secrets', 'found_secrets.jsonl', json_bytes(rebuilt_finding) + b'\n'), + ): + state = db.projection_stream_state(stream) + append = db.projection_append_for_job(scan_job['id'], stream) + if not state or not append: + raise NotReady('remote_full_scan_projection_append') + require(state['base_relative_path'] == relative, + 'remote_full_scan_projection_path') + payload = read_bytes(RUNTIME / 'results' / relative) + verify_projection_region(payload, expected, append, state, scan_job, 1) + projection_hashes[stream + '_sha256'] = digest(expected) + + if checked: + if candidate['state'] != 'completed': + raise NotReady('remote_full_keycheck_completion') + require(candidate['attempts'] == 1 and candidate['lease_token'] is None + and candidate['capacity_released'] == 1 + and candidate['result_projection_credit_transferred'] == 1, + 'remote_full_keycheck_candidate_credit') + result = db.keycheck_result_for_projection(candidate['keycheck_result_id']) + require(result and result['status'] == 'INVALID_OR_REVOKED' + and result['status_group'] == 'dead' + and result['result_source'] == 'cached_status' + and result['link_status'] == 'linked' + and result['candidate_id'] == candidate['id'] + and result['finding_id'] == finding['id'] + and result['finding_uid'] == finding['finding_uid'] + and result['target_scan_id'] == scan['id'], + 'remote_full_detailed_keycheck_result') + result_metadata = json.loads(result['metadata_json']) + require(result_metadata['cached_status'] is True + and result_metadata['cached_state_version'] == 1 + and result_metadata['cached_result_id'] + == evidence['baseline_keycheck_result_id'] + and result_metadata['cached_result_source'] == 'api_check', + 'remote_full_cached_keycheck_lineage') + current = rows(db, 'SELECT * FROM keycheck_current_state WHERE credential_id = ?', + (candidate['credential_id'],))[0] + require(current['last_result_id'] == evidence['baseline_keycheck_result_id'] + and current['state_version'] == 1 + and current['status'] == 'INVALID_OR_REVOKED' + and current['status_group'] == 'dead', + 'remote_full_keycheck_current_state') + require(rows(db, 'SELECT keycheck_result_id FROM keycheck_event_map WHERE event_id = ?', + (result['event_id'],)) == [{'keycheck_result_id': result['id']}], + 'remote_full_keycheck_event_map') + key_jobs = rows(db, 'SELECT * FROM projection_jobs WHERE keycheck_result_id = ?', + (result['id'],)) + if len(key_jobs) != 1 or key_jobs[0]['status'] != 'completed': + raise NotReady('remote_full_keycheck_projection') + key_job = key_jobs[0] + require(key_job['event_id'] == result['event_id'] + and key_job['required_stream_mask'] == 8 + and key_job['capacity_released'] == 1, + 'remote_full_keycheck_projection_lineage') + projected = {key: result[key] for key in ( + 'event_id', 'service', 'status', 'status_group', 'checked_at', 'key_hash', + 'secret_hash', 'key_masked', 'finding_uid', 'source', 'message', + 'result_source', + )} + projected.update({'detector': result['detector_name'], + 'metadata': result_metadata}) + expected = json_bytes(projected) + b'\n' + stream = 'keycheck:openai:results' + state = db.projection_stream_state(stream) + append = db.projection_append_for_job(key_job['id'], stream) + if not state or not append: + raise NotReady('remote_full_keycheck_projection_append') + payload = read_bytes(RUNTIME / 'keychecks/openai/openaiResults.jsonl') + verify_projection_region(payload, expected, append, state, key_job, 1) + projection_hashes[stream + '_sha256'] = digest(expected) + else: + require(candidate['state'] == 'pending' and candidate['attempts'] == 0 + and candidate['capacity_released'] == 0 + and candidate['result_projection_credit_transferred'] == 0, + 'remote_full_keycheck_pending_state') + + unreleased_jobs = rows(db, '''SELECT job_kind, status, capacity_items, + capacity_bytes FROM projection_jobs + WHERE capacity_released = 0 ORDER BY id''') + if unreleased_jobs: + raise NotReady('remote_full_projection_drain') + capacity = db.pipeline_capacity_snapshot() + for name in ( + 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'quarantine_items', 'quarantine_bytes', + ): + require(int(capacity[name]) == 0, + 'remote_full_capacity_' + name + '_' + str(int(capacity[name]))) + require(int(capacity['keycheck_items']) == (0 if checked else 1) + and int(capacity['keycheck_bytes']) + == (0 if checked else int(candidate['capacity_bytes'])), + 'remote_full_keycheck_capacity') + + admin = db.admin_remote_worker_snapshot(limit=200) + workers = { + row['device_key']: row for row in admin['workers'] + if row['user_key'] == evidence['user_key'] + } + require(set(workers) == { + 'container-e2e-full-expired', 'container-e2e-full-winner', + }, 'remote_full_admin_worker_identity') + expected_statistics = { + 'container-e2e-full-expired': (0, 0, 0, 1), + 'container-e2e-full-winner': (0, 1, 0, 0), + } + for device_key, expected in expected_statistics.items(): + actual = tuple(int(workers[device_key][name]) for name in ( + 'unfinished_count', 'completed_count', 'failed_count', 'expired_count', + )) + require(actual == expected, 'remote_full_admin_statistics') + assignments = [row for row in admin['assignments'] + if row['user_key'] == evidence['user_key']] + require(len(assignments) == 2 + and {row['outcome'] for row in assignments} == {'completed', 'expired'} + and sum(int(row['accepted']) for row in assignments) == 1 + and sum(int(row['ingested']) for row in assignments) == 1, + 'remote_full_admin_assignment_counts') + return { + 'scan_id': scan['id'], 'candidate_id': candidate['id'], + 'receipt_id': receipt['receipt_id'], 'projection_hashes': projection_hashes, + } + finally: + db.close() + + +def wait_for_remote_full(config, marker, url, evidence, checked, timeout): + deadline = time.monotonic() + timeout + last_check = 'remote_full_timeout' + while time.monotonic() < deadline: + try: + return remote_full_snapshot(config, marker, url, evidence, checked) + except NotReady as exc: + last_check = str(exc) + time.sleep(min(0.25, max(0, deadline - time.monotonic()))) + raise E2EFailure('timeout_' + last_check) + + +def dockerhub_canary_args(config, *, discovery): + from console_runner import build_args_from_source_config + + source = copy.deepcopy(config['sources']['dockerhub']) + source.update({ + 'enabled': True, 'mode': 'search', + 'queries': [DOCKERHUB_CANARY_QUERY], 'query_overrides': {}, + 'pages': 1, 'per_page': DOCKERHUB_CANARY_PER_PAGE, + 'workers': 1, 'max_targets': len(DOCKERHUB_CANARY_TARGETS), + 'timeout': 60, 'sync_file_queues': False, + 'tag_fetch_workers': 1, 'tag_retry_count': 1, 'tag_retry_delay': 1, + 'tag_resolve_limit': len(DOCKERHUB_CANARY_TARGETS), + 'docker_images_per_repository': 1, + 'docker_platform_filter_enabled': True, + 'docker_platform_os': 'linux', 'docker_platform_arch': 'amd64', + 'docker_platform_candidate_tags': len(DOCKERHUB_CANARY_TARGETS), + 'docker_repository_refresh_interval_sec': 0, + 'docker_repository_refresh_max_per_cycle': 0, + 'docker_content_scan_mode': 'full', + 'detectors': 'OpenAI', 'exclude_detectors': '', 'drop_detectors': [], + 'no_verification': True, 'external_trufflehog_lifecycle': True, + 'trufflehog_config': str(APP / 'trufflehog-custom-detectors.yaml'), + 'token': DOCKERHUB_CANARY_DISCOVERY_TOKEN if discovery else '', + 'docker_username': '', 'docker_token': '', 'auth_pool': '', + }) + args = build_args_from_source_config( + 'dockerhub', source, copy.deepcopy(config['global']), + DOCKERHUB_CANARY_QUERY, auth_entry=None, + ) + args.max_active_scans = 1 + args.result_bundle_max_event_bytes = MAX_BYTES + args.result_bundle_max_items = 1 + args.result_bundle_max_total_bytes = MAX_BYTES + args.projection_backlog_max_items = 1 + args.projection_backlog_max_bytes = 4 * MAX_BYTES + args.projection_backlog_headroom_bytes = 2 * MAX_BYTES + args.keycheck_queue_max_items = 1 + args.keycheck_queue_max_bytes = MAX_BYTES + args.keycheck_candidates_per_event = 1 + args.keycheck_candidate_bytes_per_event = 1024 + args.pipeline_quarantine_max_items = 1 + args.pipeline_quarantine_max_bytes = MAX_BYTES + return args + + +def dockerhub_canary_transport(): + from types import SimpleNamespace + + calls = {'search': [], 'tags': []} + targets = dict(zip(DOCKERHUB_CANARY_REPOSITORIES, DOCKERHUB_CANARY_TARGETS)) + + def search(query, page, **kwargs): + require( + query == DOCKERHUB_CANARY_QUERY and page == 1 + and kwargs == { + 'per_page': DOCKERHUB_CANARY_PER_PAGE, + 'sort_by': 'updated_at', 'sort_order': 'desc', + 'request_timeout': 15, + } + and not calls['search'], + 'dockerhub_canary_search_bound', + ) + calls['search'].append((query, page, dict(kwargs))) + return { + 'page': 1, 'total_count': len(DOCKERHUB_CANARY_REPOSITORIES), + 'repositories': [ + {'repo_name': repository} + for repository in DOCKERHUB_CANARY_REPOSITORIES + ], + } + + def tags(repository, *args, **kwargs): + require( + repository in targets and repository not in calls['tags'] + and args == ( + None, 1, 1, 1, True, 'linux', 'amd64', + len(DOCKERHUB_CANARY_TARGETS), + ) + and kwargs == {'return_outcome': True}, + 'dockerhub_canary_tag_bound', + ) + calls['tags'].append(repository) + return SimpleNamespace( + tags=(targets[repository],), status='ok', remote_attempted=True, + retry_at=None, error='', + ) + + return search, tags, calls + + +def dockerhub_canary_service(config, metadata, url): + from worker_api import WorkerService + from worker_assignment import RemoteAssignmentBuilder + from worker_package import worker_package_build_compatibility + + capability = { + 'source': 'dockerhub', 'platform': 'docker', + 'planning_kind': 'docker_direct_v1', + } + package = remote_package_manifest(metadata, [capability]) + args = dockerhub_canary_args(config, discovery=False) + bundle_body_timeout = int( + config['supervisor']['worker_api']['bundle_body_timeout_seconds'] + ) + assignment_ttl = int(args.timeout) + bundle_body_timeout + 60 + builder = RemoteAssignmentBuilder( + url, str(BUNDLES), {'dockerhub': args}, + {'dockerhub-canary': { + 'package_manifest': package, 'sources': ['dockerhub'], + }}, + metadata['instance_id'], assignment_ttl_seconds=assignment_ttl, + ) + return ( + WorkerService(url, str(BUNDLES), builder, max_bundle_bytes=MAX_BYTES), + worker_package_build_compatibility(package), args, + ) + + +def require_dockerhub_canary_assignment(assignment): + reservation = assignment['reservation'] + snapshot = assignment['execution_snapshot'] + plan = assignment['execution_plan'] + require( + reservation['source'] == 'dockerhub' + and reservation['platform'] == 'docker' + and reservation['target'] in DOCKERHUB_CANARY_TARGETS + and '@sha256:' in reservation['target'] + and snapshot['planning'] == {'kind': 'docker_direct_v1'} + and snapshot['credential_ref'] == { + 'source': 'dockerhub', 'auth_entry': '', + } + and plan == { + 'kind': 'docker_direct_v1', + 'execution_target': reservation['target'], 'bound_plan': None, + } + and assignment['scan_kwargs'] == assignment['event_scan_options'], + 'dockerhub_canary_assignment_identity', + ) + encoded = json.dumps(assignment, ensure_ascii=True, sort_keys=True).lower() + require( + DOCKERHUB_CANARY_DISCOVERY_TOKEN not in encoded + and all(name not in encoded for name in ( + 'docker_registry_auth', 'access_probe', 'public_access_proof', + 'proof_freshness', 'docker_layer_work', 'docker_layer_plan', + )), + 'dockerhub_canary_assignment_minimal', + ) + + +def stage_dockerhub_canary_result(metadata, assignment, result_kind): + from lifecycle_authority import build_code_manifest, code_manifest_sha256 + from result_bundle import BundleReservation, bundle_ready_path + import scan_execution + import scanner + + diagnostics = { + 'permanent': json.dumps({ + 'level': 'error', 'msg': 'provider access failed', + 'error': 'pull access denied', + }), + 'retryable': json.dumps({ + 'level': 'error', 'msg': 'provider request failed', + 'error': 'HTTP 503 service unavailable', + }), + 'success': json.dumps({ + 'level': 'info-0', 'logger': 'trufflehog', + 'msg': 'finished scanning', + }), + } + require(result_kind in diagnostics, 'dockerhub_canary_result_kind') + reservation = BundleReservation.from_mapping(assignment['reservation']) + result = {'findings': [], 'errors': []} + policy_path = str(APP / 'trufflehog-custom-detectors.yaml') + client_manifest = build_code_manifest( + str(APP), scanner.scan_config.trufflehog_path, (policy_path,), + git_path=metadata['code_manifest']['executables']['git']['path'], + ) + with scanner.client_scan_launch_authority( + client_manifest, code_manifest_sha256(client_manifest), + ), scanner.client_scan_execution_policy( + assignment['scan_policy'], + ), scanner.client_remote_execution_binding('docker_direct_v1'): + scanner.apply_trufflehog_diagnostics( + result, diagnostics[result_kind], 0 if result_kind == 'success' else 1, + 'docker', + ) + result.update({ + 'target': reservation.target, 'scan_type': 'docker', + 'scan_event_id': reservation.scan_event_id, + 'scan_started_at': '2026-09-20T00:00:00+00:00', + 'duration_sec': 0.0, 'timestamp': '2026-09-20T00:00:00+00:00', + }) + classification = { + 'error_class': str(result.get('error_class') or ''), + 'retryable': bool(result.get('retryable', False)), + 'skipped': bool(result.get('skipped')), + } + staged = scan_execution.stage_scan_result_in_scope( + result, reservation, str(REMOTE_CLIENT_BUNDLES), + assignment['event_scan_options'], assignment['queue_policy'], + attempts=int(assignment['reservation'].get('attempts') or 0), + candidate_max_items=int(assignment['limits']['candidate_max_items']), + candidate_max_bytes=int(assignment['limits']['candidate_max_bytes']), + ) + payload = read_bytes(Path(bundle_ready_path( + str(REMOTE_CLIENT_BUNDLES), reservation.bundle_id, + )), limit=MAX_BYTES) + require( + staged.actual_bytes == len(payload) and 0 < len(payload) < MAX_BYTES, + 'dockerhub_canary_bundle_bound', + ) + expected = { + 'permanent': ('docker_registry_access', False, True, 'done', 0), + 'retryable': ('remote_transient', True, False, 'deferred', 1), + 'success': ('', False, False, 'done', 0), + }[result_kind] + require( + ( + classification['error_class'], classification['retryable'], + classification['skipped'], staged.queue_status, staged.error_count, + ) == expected, + 'dockerhub_canary_worker_disposition', + ) + return payload, staged, classification + + +def dockerhub_canary_result_snapshot(db, reservation_id, result_kind, receipt): + state = rows(db, '''SELECT r.state, r.remote_resolution_json, + b.state AS bundle_state, q.status AS queue_status, q.target_scan_id + FROM result_reservations r + JOIN result_bundles b ON b.reservation_id = r.id + JOIN target_queue q ON q.id = r.queue_id + WHERE r.id = ?''', (reservation_id,)) + if ( + len(state) != 1 or state[0]['state'] != 'acknowledged' + or state[0]['bundle_state'] != 'acknowledged' + or state[0]['target_scan_id'] is None + ): + raise NotReady('dockerhub_canary_ingestion') + expected_queue = 'deferred' if result_kind == 'retryable' else 'done' + require(state[0]['queue_status'] == expected_queue, + 'dockerhub_canary_queue_disposition') + require(json.loads(state[0]['remote_resolution_json']) == receipt, + 'dockerhub_canary_receipt_persistence') + scan = rows(db, 'SELECT * FROM target_scans WHERE id = ?', + (state[0]['target_scan_id'],))[0] + rebuilt = db.reconstruct_scan_result( + scan['id'], max_bytes=MAX_BYTES, max_findings=1, max_errors=2, + ) + require(rebuilt and rebuilt['target'] in DOCKERHUB_CANARY_TARGETS + and rebuilt['scan_type'] == 'docker', + 'dockerhub_canary_reconstructed_identity') + if result_kind == 'permanent': + require(rebuilt.get('error_class') == 'docker_registry_access' + and rebuilt.get('retryable') is False + and rebuilt.get('skipped') == 'Docker image is unavailable to the worker' + and rebuilt['errors'] == [], + 'dockerhub_canary_permanent_classification') + elif result_kind == 'retryable': + require(rebuilt.get('error_class') == 'remote_transient' + and rebuilt.get('retryable') is True + and len(rebuilt['errors']) == 1, + 'dockerhub_canary_retryable_classification') + else: + require(not rebuilt['errors'] and not rebuilt.get('skipped'), + 'dockerhub_canary_success_classification') + jobs = rows(db, 'SELECT * FROM projection_jobs WHERE target_scan_id = ?', + (scan['id'],)) + if len(jobs) != 1 or jobs[0]['status'] != 'completed': + raise NotReady('dockerhub_canary_projection') + job = jobs[0] + expected_stream_mask = 5 if result_kind == 'retryable' else 1 + require(job['job_kind'] == 'scan_event' + and job['event_id'] == scan['scan_event_id'] + and job['event_hash'] == scan['scan_event_hash'] + and job['required_stream_mask'] == expected_stream_mask + and job['capacity_released'] == 1, + 'dockerhub_canary_projection_lineage') + state_row = db.projection_stream_state('scan_results') + append = db.projection_append_for_job(job['id'], 'scan_results') + if not state_row or not append: + raise NotReady('dockerhub_canary_projection_append') + payload = read_bytes(RUNTIME / 'results' / state_row['base_relative_path']) + expected = json_bytes(rebuilt) + b'\n' + verify_projection_region(payload, expected, append, state_row, job, 1) + return { + 'scan_id': scan['id'], 'projection_job_id': job['id'], + 'projection_sha256': digest(expected), + 'projection_completed_at': job['completed_at'], + } + + +def wait_for_dockerhub_canary_result( + db, reservation_id, result_kind, receipt, timeout, +): + deadline = time.monotonic() + timeout + last_check = 'dockerhub_canary_pipeline' + while time.monotonic() < deadline: + try: + return dockerhub_canary_result_snapshot( + db, reservation_id, result_kind, receipt, + ) + except NotReady as exc: + last_check = str(exc) + time.sleep(min(0.1, max(0, deadline - time.monotonic()))) + raise E2EFailure('timeout_' + last_check) + + +def dockerhub_canary_operation_id(label): + value = digest(('dockerhub-canary:' + label).encode('ascii'))[:32] + return '-'.join((value[:8], value[8:12], value[12:16], value[16:20], value[20:])) + + +def dockerhub_canary_cycle(db, args, label): + run_id = db.start_run( + 'container-e2e-dockerhub-canary', selected_source='dockerhub', + selected_platform='docker', config_path=str(CONFIG), + enabled_sources=['dockerhub'], + ) + cycle_id = db.start_source_cycle( + run_id, 'dockerhub', 'docker', 'search', DOCKERHUB_CANARY_QUERY, 1, 1, + ) + return run_id, cycle_id + + +def prepare_dockerhub_canary(config, marker, url, timeout): + import asyncio + from datetime import datetime, timedelta, timezone + from runtime_security import ensure_private_directory, write_private_json_exclusive + import console_runner + import scanner_db + from unittest import mock + + metadata = control_snapshot(marker, url) + require(not os.path.lexists(DOCKERHUB_CANARY), + 'dockerhub_canary_evidence_exists') + ensure_private_directory(str(REMOTE_CLIENT_BUNDLES), reject_reparse=True) + db = scanner_db.ScannerDB(db_url=url, initialize=False) + stage = 'database' + try: + db.set_application_name('truf-container-e2e:dockerhub-canary-prepare') + require(db.enabled and db.conn.is_postgres, 'dockerhub_canary_postgres') + db.require_runtime_safety_schema() + db.require_final_cutover() + control = db.runtime_control_state() + require(control['drain_state'] == 'normal' + and not control['effective_discovery_paused'] + and not control['effective_dispatch_paused'], + 'dockerhub_canary_initial_control') + require(not rows(db, 'SELECT id FROM target_queue WHERE source = ? AND query = ?', + ('dockerhub', DOCKERHUB_CANARY_QUERY)), + 'dockerhub_canary_queue_not_fresh') + authority_columns = rows(db, '''SELECT table_name, column_name + FROM information_schema.columns + WHERE table_schema = current_schema() + ORDER BY table_name, ordinal_position''') + require(not any( + forbidden in ( + str(column['table_name']) + '.' + str(column['column_name']) + ).lower() + for column in authority_columns + for forbidden in ('access_probe', 'public_access_proof', 'proof_freshness') + ), 'dockerhub_canary_access_state_forbidden') + + stage = 'discovery' + discovery_args = dockerhub_canary_args(config, discovery=True) + search, tags, calls = dockerhub_canary_transport() + run_id, cycle_id = dockerhub_canary_cycle(db, discovery_args, 'initial') + with mock.patch.object( + console_runner, 'fetch_dockerhub_search_page', new=search, + ), mock.patch.object( + console_runner, 'fetch_dockerhub_tags', new=tags, + ): + metrics = console_runner.run_discovery_cycle( + discovery_args, db, run_id, cycle_id, 'dockerhub', + ) + db.finish_run(run_id) + require(metrics['cycle_status'] == 'completed' + and metrics['fetched_count'] == len(DOCKERHUB_CANARY_REPOSITORIES) + and metrics['queued_new_count'] == len(DOCKERHUB_CANARY_REPOSITORIES) + and metrics['discovery_pages_fetched'] == 1 + and metrics['scan_requested_count'] == 0 + and len(calls['search']) == 1 and not calls['tags'], + 'dockerhub_canary_discovery_result') + + stage = 'resolution' + resolver_now = ( + datetime.now(timezone.utc) + timedelta(seconds=7200) + ).isoformat(timespec='seconds') + with mock.patch.object( + console_runner, 'fetch_dockerhub_tags', new=tags, + ), mock.patch.object( + scanner_db, 'utc_now_iso', return_value=resolver_now, + ): + resolved = console_runner.resolve_due_docker_queue_targets( + db, 'dockerhub', discovery_args, + ) + if ( + resolved != len(DOCKERHUB_CANARY_TARGETS) + or calls['tags'] != list(DOCKERHUB_CANARY_REPOSITORIES) + ): + raise E2EFailure( + f'dockerhub_canary_resolution_result_{resolved}_{len(calls["tags"])}' + ) + queue = rows(db, '''SELECT target, status, resolver_state FROM target_queue + WHERE source = ? AND query = ? ORDER BY id''', + ('dockerhub', DOCKERHUB_CANARY_QUERY)) + repositories = [row for row in queue if '@sha256:' not in row['target']] + targets = [row for row in queue if '@sha256:' in row['target']] + require([row['target'] for row in repositories] + == list(DOCKERHUB_CANARY_REPOSITORIES) + and all(row['status'] == 'done' and row['resolver_state'] == 'resolved' + for row in repositories) + and [row['target'] for row in targets] == list(DOCKERHUB_CANARY_TARGETS) + and all(row['status'] == 'pending' for row in targets), + 'dockerhub_canary_immutable_queue') + + stage = 'worker' + user_key = 'container-e2e-dockerhub-canary-user' + device_key = 'container-e2e-dockerhub-canary-device' + device_token = 'container-e2e-dockerhub-canary-token' + device = db.provision_remote_worker_device( + user_key, device_key, digest(device_token.encode('ascii')), 1, + ) + require(device['active_assignment_cap'] == 1 and not device['revoked'], + 'dockerhub_canary_device') + service, build, assignment_args = dockerhub_canary_service( + config, metadata, url, + ) + require(assignment_args.workers == assignment_args.max_active_scans == 1 + and not assignment_args.token and not assignment_args.docker_username + and not assignment_args.docker_token and not assignment_args.auth_name, + 'dockerhub_canary_assignment_configuration') + + def claim(label): + request_id = digest(('dockerhub-canary:' + label).encode('ascii')) + status, response = asyncio.run(invoke_worker_claim( + service, device_token, request_id, build, + )) + require(status == 201 and set(response or {}) == {'assignment'}, + 'dockerhub_canary_claim_transport') + assignment = response['assignment'] + require_dockerhub_canary_assignment(assignment) + require(assignment['reservation']['remote_device_id'] == device['device_id'], + 'dockerhub_canary_claim_device') + return request_id, assignment + + def upload(assignment, result_kind): + payload, staged, classification = stage_dockerhub_canary_result( + metadata, assignment, result_kind, + ) + status, code, receipt = asyncio.run(invoke_bundle_upload( + service, device_token, + assignment['reservation']['reservation_id'], payload, + )) + require(status == 201 and code is None + and receipt['resolution'] == 'bundle_accepted' + and receipt['payload_sha256'] == digest(payload), + 'dockerhub_canary_upload') + snapshot = wait_for_dockerhub_canary_result( + db, assignment['reservation']['reservation_id'], result_kind, + receipt, timeout, + ) + return payload, staged, classification, receipt, snapshot + + stage = 'permanent' + lost_request_id, lost_assignment = claim('permanent-lost-response') + status, replayed = asyncio.run(invoke_worker_claim( + service, device_token, lost_request_id, build, + )) + require(status == 201 and replayed == {'assignment': lost_assignment}, + 'dockerhub_canary_lost_claim_replay') + permanent = upload(lost_assignment, 'permanent') + + stage = 'retryable' + _, retryable_assignment = claim('retryable') + retryable = upload(retryable_assignment, 'retryable') + + stage = 'expiry_claim' + _, expired_assignment = claim('unfinished-expiry') + expired_payload, _, _, = stage_dockerhub_canary_result( + metadata, expired_assignment, 'success', + ) + expired = expired_assignment['reservation'] + state = db.runtime_control_state() + db.start_runtime_drain( + expected_revision=state['revision'], actor='test:container-e2e', + operation_id=dockerhub_canary_operation_id('expiry-drain-start'), + ) + status_value = service.status({ + 'device_id': device['device_id'], + 'token_sha256': digest(device_token.encode('ascii')), + }, expired['reservation_id']) + require(status_value['state'] == 'scanning', + 'dockerhub_canary_drain_status') + + stage = 'paused_discovery' + paused_run, paused_cycle = dockerhub_canary_cycle( + db, discovery_args, 'paused', + ) + with mock.patch.object( + console_runner, 'fetch_dockerhub_search_page', new=search, + ), mock.patch.object( + console_runner, 'fetch_dockerhub_tags', new=tags, + ): + paused = console_runner.run_discovery_cycle( + discovery_args, db, paused_run, paused_cycle, 'dockerhub', + ) + db.finish_run(paused_run) + require(paused['cycle_status'] == 'paused' + and len(calls['search']) == 1 + and len(calls['tags']) == len(DOCKERHUB_CANARY_TARGETS), + 'dockerhub_canary_drain_discovery_gate') + + stage = 'expiry' + with mock.patch.object( + scanner_db, 'utc_now_iso', return_value=expired['remote_expires_at'], + ): + expiry_receipts = service.reap() + require(len(expiry_receipts) == 1 + and expiry_receipts[0]['reservation_id'] == expired['reservation_id'] + and expiry_receipts[0]['resolution'] == 'expired', + 'dockerhub_canary_expiry_receipt') + status, code, _ = asyncio.run(invoke_bundle_upload( + service, device_token, expired['reservation_id'], expired_payload, + )) + require((status, code) == (409, 'resolution_conflict'), + 'dockerhub_canary_stale_upload') + first_drained = db.runtime_control_state() + require(first_drained['drain_state'] == 'drained' + and first_drained['effective_discovery_paused'] + and first_drained['effective_dispatch_paused'], + 'dockerhub_canary_expiry_drain') + pending_after_expiry = rows(db, '''SELECT target FROM target_queue + WHERE source = ? AND query = ? AND status = 'pending' + AND target LIKE '%@sha256:%' ORDER BY id''', + ('dockerhub', DOCKERHUB_CANARY_QUERY)) + require(len(pending_after_expiry) == 2, + 'dockerhub_canary_expiry_pending_targets') + db.cancel_runtime_drain( + expected_revision=first_drained['revision'], actor='test:container-e2e', + operation_id=dockerhub_canary_operation_id('expiry-drain-cancel'), + ) + + stage = 'success_claim' + success_request_id, success_assignment = claim('success') + require(success_assignment['reservation']['queue_id'] == expired['queue_id'], + 'dockerhub_canary_expired_target_reclaim') + success_payload, success_staged, success_classification = ( + stage_dockerhub_canary_result(metadata, success_assignment, 'success') + ) + success = success_assignment['reservation'] + state = db.runtime_control_state() + drain_start = db.start_runtime_drain( + expected_revision=state['revision'], actor='test:container-e2e', + operation_id=dockerhub_canary_operation_id('upload-drain-start'), + ) + require(service.status({ + 'device_id': device['device_id'], + 'token_sha256': digest(device_token.encode('ascii')), + }, success['reservation_id'])['state'] == 'scanning', + 'dockerhub_canary_upload_drain_status') + + stage = 'drain_upload' + status, code, success_receipt = asyncio.run(invoke_bundle_upload( + service, device_token, success['reservation_id'], success_payload, + )) + require(status == 201 and code is None + and success_receipt['resolution'] == 'bundle_accepted', + 'dockerhub_canary_drain_upload') + status, code, replay_receipt = asyncio.run(invoke_bundle_upload( + service, device_token, success['reservation_id'], success_payload, + )) + require(status == 200 and code is None and replay_receipt == success_receipt, + 'dockerhub_canary_drain_replay') + status, code, _ = asyncio.run(invoke_bundle_upload( + service, device_token, success['reservation_id'], success_payload + b'x', + )) + require((status, code) == (409, 'resolution_conflict'), + 'dockerhub_canary_drain_conflict') + reservations_before = rows(db, '''SELECT COUNT(*) AS count + FROM result_reservations WHERE remote_user_id = ?''', + (device['user_id'],))[0]['count'] + blocked_status, blocked = asyncio.run(invoke_worker_claim( + service, device_token, + digest(b'dockerhub-canary-drain-blocked-claim'), build, + )) + reservations_after = rows(db, '''SELECT COUNT(*) AS count + FROM result_reservations WHERE remote_user_id = ?''', + (device['user_id'],))[0]['count'] + require(blocked_status == 204 and blocked is None + and reservations_before == reservations_after, + 'dockerhub_canary_drain_dispatch_gate') + + stage = 'drain_pipeline' + success_snapshot = wait_for_dockerhub_canary_result( + db, success['reservation_id'], 'success', success_receipt, timeout, + ) + state = db.runtime_control_state() + if state['drain_state'] == 'draining': + completed = db.reconcile_runtime_drain() + require(completed is not None, 'dockerhub_canary_drain_reconcile') + state = db.runtime_control_state() + require(state['drain_state'] == 'drained', + 'dockerhub_canary_final_drained') + pending = rows(db, '''SELECT target FROM target_queue + WHERE source = ? AND query = ? AND status = 'pending' + AND target LIKE '%@sha256:%' ORDER BY id''', + ('dockerhub', DOCKERHUB_CANARY_QUERY)) + require(pending == [{'target': DOCKERHUB_CANARY_TARGETS[-1]}], + 'dockerhub_canary_final_pending_target') + drain_operation = rows(db, '''SELECT completed_at FROM runtime_operations + WHERE operation_id = ?''', (drain_start['operation_id'],))[0] + require(success_snapshot['projection_completed_at'] >= drain_operation['completed_at'], + 'dockerhub_canary_projection_during_drain') + db.cancel_runtime_drain( + expected_revision=state['revision'], actor='test:container-e2e', + operation_id=dockerhub_canary_operation_id('upload-drain-cancel'), + ) + capacity = db.pipeline_capacity_snapshot() + require(not any(int(capacity[name]) for name in ( + 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'keycheck_items', 'keycheck_bytes', 'quarantine_items', + 'quarantine_bytes', + )), 'dockerhub_canary_capacity_release') + + stage = 'evidence' + evidence = { + 'schema': 1, + 'origin_instance_sha256': digest(metadata['instance_id'].encode('utf-8')), + 'user_key': user_key, 'device_key': device_key, + 'success_request_id': success_request_id, + 'build': build, + 'success': { + 'reservation_id': success['reservation_id'], + 'bundle_id': success['bundle_id'], + 'scan_event_id': success['scan_event_id'], + 'payload_sha256': digest(success_payload), + 'remote_expires_at': success['remote_expires_at'], + 'receipt': success_receipt, + 'scan_id': success_snapshot['scan_id'], + 'projection_job_id': success_snapshot['projection_job_id'], + }, + 'expired': { + 'reservation_id': expired['reservation_id'], + 'receipt': expiry_receipts[0], + }, + 'classifications': { + 'permanent': permanent[2], 'retryable': retryable[2], + 'success': success_classification, + }, + } + write_private_json_exclusive(str(DOCKERHUB_CANARY), evidence) + public = { + 'search_pages': 1, + 'cohort_targets': len(DOCKERHUB_CANARY_TARGETS), + 'digest_resolutions': len(calls['tags']), + 'protocol2_assignments': 4, 'lost_claim_replays': 1, + 'worker_provider_failures': 2, 'accepted_uploads': 3, + 'expired_assignments': 1, 'stale_uploads_rejected': 1, + 'drain_cycles': 2, 'drain_replays': 1, + 'drain_conflicts_rejected': 1, 'pending_targets': len(pending), + 'projected_results': 3, + } + return { + 'counts': {'ok': 1, **public}, + 'hashes': { + 'dockerhub_canary_prepare_sha256': digest(json_bytes(public)), + 'dockerhub_canary_projection_sha256': success_snapshot['projection_sha256'], + }, + } + except E2EFailure: + raise + except Exception as exc: + trace = exc.__traceback__ + origins = [] + while trace: + origins.append(trace.tb_frame.f_code.co_name) + trace = trace.tb_next + origin = '_'.join([type(exc).__name__, *origins[-3:]]) + origin = re.sub(r'[^a-z0-9_]', '_', origin.lower())[:80] + raise E2EFailure( + 'dockerhub_canary_prepare_' + stage + '_' + origin + ) from None + finally: + db.close() + + +def finish_dockerhub_canary(config, marker, url, timeout): + del timeout + import asyncio + from result_bundle import bundle_ready_path + import scanner_db + import worker_api + from unittest import mock + + metadata = control_snapshot(marker, url) + evidence = json.loads(read_bytes(DOCKERHUB_CANARY)) + require(evidence.get('schema') == 1 + and digest(metadata['instance_id'].encode('utf-8')) + != evidence['origin_instance_sha256'], + 'dockerhub_canary_restart_evidence') + service, build, _args = dockerhub_canary_service(config, metadata, url) + require(build == evidence['build'], 'dockerhub_canary_restart_build') + success = evidence['success'] + path = Path(bundle_ready_path(str(REMOTE_CLIENT_BUNDLES), success['bundle_id'])) + payload = read_bytes(path, limit=MAX_BYTES) + require(digest(payload) == success['payload_sha256'], + 'dockerhub_canary_restart_payload') + + status, response = asyncio.run(invoke_worker_claim( + service, 'container-e2e-dockerhub-canary-token', + evidence['success_request_id'], build, + )) + require(status == 200 and response == {'resolution': success['receipt']}, + 'dockerhub_canary_restart_claim_receipt') + with mock.patch.object( + worker_api, 'utc_now_iso', return_value='9999-12-31T23:59:59+00:00', + ): + status, code, receipt = asyncio.run(invoke_bundle_upload( + service, 'container-e2e-dockerhub-canary-token', + success['reservation_id'], payload, + )) + require(status == 200 and code is None and receipt == success['receipt'], + 'dockerhub_canary_restart_receipt_replay') + status, code, _ = asyncio.run(invoke_bundle_upload( + service, 'container-e2e-dockerhub-canary-token', + success['reservation_id'], payload + b'x', + )) + require((status, code) == (409, 'resolution_conflict'), + 'dockerhub_canary_restart_conflict') + + db = scanner_db.ScannerDB(db_url=url, initialize=False) + try: + db.set_application_name('truf-container-e2e:dockerhub-canary-finish') + assignments = rows(db, '''SELECT r.id, r.state, r.remote_resolution_kind, + r.remote_resolution_json + FROM result_reservations r + JOIN remote_worker_users u ON u.id = r.remote_user_id + WHERE u.user_key = ? ORDER BY r.id''', (evidence['user_key'],)) + require(len(assignments) == 4 + and sum(row['remote_resolution_kind'] == 'bundle_accepted' + for row in assignments) == 3 + and sum(row['remote_resolution_kind'] == 'expired' + for row in assignments) == 1, + 'dockerhub_canary_restart_assignments') + scans = rows(db, '''SELECT id, scan_event_id, scan_event_hash + FROM target_scans WHERE scan_event_id = ?''', (success['scan_event_id'],)) + bundles = rows(db, '''SELECT reservation_id FROM result_bundles + WHERE reservation_id = ?''', (success['reservation_id'],)) + jobs = rows(db, 'SELECT * FROM projection_jobs WHERE target_scan_id = ?', + (success['scan_id'],)) + appends = rows(db, '''SELECT * FROM projection_appends + WHERE job_id = ? ORDER BY id''', (success['projection_job_id'],)) + require(len(scans) == len(bundles) == len(jobs) == len(appends) == 1 + and scans[0]['id'] == success['scan_id'] + and bundles[0]['reservation_id'] == success['reservation_id'] + and jobs[0]['id'] == success['projection_job_id'] + and jobs[0]['job_kind'] == 'scan_event' + and jobs[0]['status'] == 'completed' + and jobs[0]['event_id'] == scans[0]['scan_event_id'] + and jobs[0]['event_hash'] == scans[0]['scan_event_hash'] + and jobs[0]['required_stream_mask'] == 1 + and jobs[0]['capacity_released'] == 1 + and appends[0]['stream_name'] == 'scan_results' + and appends[0]['event_id'] == jobs[0]['event_id'] + and appends[0]['event_hash'] == jobs[0]['event_hash'], + 'dockerhub_canary_restart_exactly_once') + require(not os.path.lexists(bundle_ready_path( + str(BUNDLES), success['bundle_id'], + )), 'dockerhub_canary_restart_spool_cleanup') + capacity = db.pipeline_capacity_snapshot() + require(not any(int(capacity[name]) for name in ( + 'bundle_items', 'bundle_bytes', 'projection_items', 'projection_bytes', + 'keycheck_items', 'keycheck_bytes', 'quarantine_items', + 'quarantine_bytes', + )), 'dockerhub_canary_restart_capacity') + finally: + db.close() + public = { + 'runtime_restarts': 1, 'lost_claim_receipts': 1, + 'receipt_replays': 1, 'conflicts_rejected': 1, + 'exactly_once': 1, 'former_expiry_replays': 1, + } + return { + 'counts': {'ok': 1, **public}, + 'hashes': { + 'dockerhub_canary_finish_sha256': digest(json_bytes(public)), + }, + } + + +def fixture_transport(expected_id, calls): + import requests + + def request(session, method, url, **kwargs): + del session + common = sys.modules.get('keycheck_common') + active = getattr(common, '_ACTIVE_DB_CANDIDATE', None) + require(active and active['id'] == expected_id and active.get('lease_token') + and not active.get('_completed'), 'provider_active_candidate_fence') + require(not calls and str(method).upper() == 'GET' + and url == 'https://api.openai.com/v1/models' + and kwargs.get('headers') == {'Authorization': 'Bearer ' + synthetic_openai_token()} + and set(kwargs) <= {'headers', 'proxies', 'params', 'timeout', 'allow_redirects'} + and not kwargs.get('proxies') and not kwargs.get('params'), + 'unexpected_provider_network') + calls.append(1) + response = requests.Response() + response.status_code = 401 + response.url = url + response.headers['Content-Type'] = 'application/json' + response._content = b'{"error":{"message":"Known-fake offline fixture","type":"invalid_request_error","code":"invalid_api_key"}}' + return response + + return request + + +def keycheck_child(bootstrap): + # This is a test transport wrapper, not an alternate production bootstrap. + # main() below still authenticates the provider, verifies its immutable + # manifest, and invokes the actual leaf entrypoint and database candidate loop. + from unittest import mock + import requests + + baseline = json.loads(read_bytes(RESULT)) + calls = [] + sys.argv = [str(APP / 'child_bootstrap.py'), 'keycheck-provider', + 'keycheckers/openai/Keycheck.py', '--', '--proxy-file', str(FIXTURE / 'no-proxy')] + with mock.patch.object(requests.Session, 'request', new=fixture_transport(baseline['ids']['candidate_id'], calls)): + try: + bootstrap['main']() + except SystemExit as exc: + require(exc.code in (None, 0), 'provider_exit') + expected_calls = 1 - baseline['checked'] + require(len(calls) == expected_calls, 'provider_request_count') + return {'counts': {'ok': 1, 'http_requests': len(calls)}, 'hashes': {}} + + +def run_keycheck_process(config, marker, url, expected_http_requests, timeout): + from lifecycle_authority import strip_supervisor_credentials, supervised_child_environment + + metadata = control_snapshot(marker, url) + env = os.environ.copy() + strip_supervisor_credentials(env) + for name in list(env): + if name.startswith('KEYCHECK_') or name.lower().endswith('_proxy'): + env.pop(name) + env.update(supervised_child_environment(metadata, url, 'keycheck-provider')) + output = str(RUNTIME / 'keychecks/openai') + env.update({ + 'SCANNER_DB_URL': url, 'DATABASE_URL': url, 'KEYCHECK_DB_URL': url, + 'KEYCHECK_SERVICE': 'openai', 'KEYCHECK_INPUT_MODE': 'postgres', + 'KEYCHECK_OUTPUT_DIR': output, 'KEYCHECK_STATE_DIR': output, + 'KEYCHECK_PROVIDER_SLICE_KEYS': '1', 'KEYCHECK_DB_INLINE': '0', + 'KEYCHECK_PROXY_FILE': str(FIXTURE / 'no-proxy'), + 'KEYCHECK_RESULT_PROJECTION_RESERVE_BYTES': str(config['global']['keycheck_result_projection_reserve_bytes']), + 'KEYCHECK_PROJECTION_MAX_ITEMS': str(config['global']['projection_backlog_max_items']), + 'KEYCHECK_PROJECTION_MAX_BYTES': str(config['global']['projection_backlog_max_bytes']), + }) + require(not os.path.lexists(FIXTURE / 'no-proxy'), 'fixture_proxy_forbidden') + completed = subprocess.run( + [sys.executable, '-I', '-S', '-B', str(Path(__file__).resolve()), '_keycheck-child'], + cwd=FIXTURE, env=env, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, + stderr=subprocess.PIPE, timeout=min(90, timeout), check=False, + ) + require(completed.returncode == 0 and len(completed.stdout) <= 4096, 'provider_subprocess_failed') + summary = json.loads(completed.stdout) + require(summary == {'counts': { + 'ok': 1, 'http_requests': expected_http_requests, + }, 'hashes': {}}, + 'provider_subprocess_summary') + return summary + + +def run_keycheck(config, marker, url, baseline, timeout): + summary = run_keycheck_process( + config, marker, url, 1 - baseline['checked'], timeout, + ) + after = wait_for_pipeline(config, marker, url, 1, timeout) + if baseline['checked']: + compare_persisted(baseline, after) + else: + for name, value in baseline['ids'].items(): + require(after['ids'][name] == value, 'keycheck_changed_scanner_identity') + for name, value in baseline['hashes'].items(): + if name != 'artifact_ids_sha256': + require(after['hashes'][name] == value, 'keycheck_changed_scanner_projection') + after['origin_instance_sha256'] = baseline['origin_instance_sha256'] + from runtime_security import atomic_write_private_json + atomic_write_private_json(str(RESULT), after) + return {'counts': {'ok': 1, 'http_requests': summary['counts']['http_requests'], **after['counts']}, + 'hashes': after['hashes']} + + +def _finish_remote_full_race(config, marker, url, timeout): + import asyncio + from worker_api import WorkerService + + evidence = json.loads(read_bytes(REMOTE_FULL)) + require(evidence.get('schema') == 1, 'remote_full_evidence_required') + service = WorkerService( + url, str(BUNDLES), lambda *_args: None, max_bundle_bytes=MAX_BYTES, + ) + payloads = {} + for role in ('expired', 'winner'): + item = evidence[role] + path = REMOTE_CLIENT_BUNDLES / 'ready' / item['bundle_id'][:2] / ( + item['bundle_id'] + '.trb' + ) + payloads[role] = read_bytes(path) + require(digest(payloads[role]) == item['payload_sha256'], + 'remote_full_client_bundle_changed') + + async def race_uploads(): + barrier = asyncio.Barrier(2) + + async def upload(role): + await barrier.wait() + return await invoke_bundle_upload( + service, remote_device_token(role), + evidence[role]['reservation_id'], payloads[role], + ) + + return await asyncio.gather(upload('expired'), upload('winner')) + + expired_result, winner_result = asyncio.run(race_uploads()) + require(expired_result[:2] == (409, 'resolution_conflict') + and expired_result[2] is None, + 'remote_full_concurrent_stale_upload') + require(winner_result[0] == 201 and winner_result[1] is None + and winner_result[2]['resolution'] == 'bundle_accepted' + and winner_result[2]['payload_sha256'] == evidence['winner']['payload_sha256'], + 'remote_full_concurrent_winner_upload') + + before_keycheck = wait_for_remote_full( + config, marker, url, evidence, False, timeout, + ) + summary = run_keycheck_process(config, marker, url, 0, timeout) + complete = wait_for_remote_full( + config, marker, url, evidence, True, timeout, + ) + require(before_keycheck['scan_id'] == complete['scan_id'] + and before_keycheck['candidate_id'] == complete['candidate_id'] + and complete['receipt_id'] == winner_result[2]['receipt_id'], + 'remote_full_processing_identity') + public = { + 'concurrent_uploads': 2, 'authoritative_receipts': 1, + 'authoritative_scans': 1, 'expired_losers': 1, + 'completed_winners': 1, 'cached_keychecks': 1, + 'provider_http_requests': summary['counts']['http_requests'], + 'projection_streams': len(complete['projection_hashes']), + } + return { + 'counts': {'ok': 1, **public}, + 'hashes': { + 'remote_full_finish_sha256': digest(json_bytes(public)), + 'remote_full_parity_sha256': evidence['parity_sha256'], + **complete['projection_hashes'], + }, + } + + +def finish_remote_full_race(config, marker, url, timeout): + try: + return _finish_remote_full_race(config, marker, url, timeout) + except E2EFailure: + raise + except Exception as exc: + trace = exc.__traceback__ + origins = [] + while trace: + origins.append(trace.tb_frame.f_code.co_name) + trace = trace.tb_next + origin = '_'.join([type(exc).__name__, *origins[-3:]]) + origin = re.sub(r'[^a-z0-9_]', '_', origin.lower())[:90] + raise E2EFailure('remote_full_finish_' + origin) from None + + +def main(argv=None): + class Parser(argparse.ArgumentParser): + def error(self, message): + raise E2EFailure('invalid_arguments') + + try: + parser = Parser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument('mode', choices=( + 'prepare', 'run-local-pipeline', 'assert-pipeline', + 'keycheck-fixture', 'assert-persisted', + 'assert-remote-recovery', 'prepare-remote-transport', + 'assert-remote-transport-replay', 'prepare-remote-full-race', + 'finish-remote-full-race', 'prepare-dockerhub-canary', + 'finish-dockerhub-canary', '_keycheck-child', + )) + parser.add_argument('--config', default=str(CONFIG)) + parser.add_argument('--timeout', type=float, default=180) + args = parser.parse_args(argv) + require(math.isfinite(args.timeout) and 1 <= args.timeout <= 600, 'bounded_timeout_required') + require_container(args.config) + neutralize_environment(args.mode) + # Application/provider logging can contain masked secrets or source data. + # The only terminal output from this driver is the explicit safe summary. + logging.disable(logging.CRITICAL) + with open(os.devnull, 'w', encoding='utf-8') as quiet, redirect_stdout(quiet), redirect_stderr(quiet): + bootstrap = runpy.run_path(str(APP / 'child_bootstrap.py')) + if args.mode == '_keycheck-child': + bootstrap['_authenticate']('keycheck-provider') + bootstrap['_enable_dependency_paths']('keycheck-provider' if args.mode == '_keycheck-child' else 'scanner') + sys.path.insert(0, str(APP)) + if args.mode == 'prepare': + summary = prepare() + else: + config, marker = load_fixture() + url = database_environment() + if args.mode == '_keycheck-child': + summary = keycheck_child(bootstrap) + elif args.mode == 'run-local-pipeline': + summary = run_local_pipeline_fixture(config, marker, url) + elif args.mode == 'assert-pipeline': + snapshot = wait_for_pipeline(config, marker, url, 0, args.timeout) + if RESULT.exists(): + compare_persisted(json.loads(read_bytes(RESULT)), snapshot) + else: + from runtime_security import write_private_json_exclusive + write_private_json_exclusive(str(RESULT), snapshot) + summary = {'counts': {'ok': 1, **snapshot['counts']}, 'hashes': snapshot['hashes']} + elif args.mode == 'assert-remote-recovery': + summary = assert_remote_recovery(config, marker, url) + elif args.mode == 'prepare-remote-transport': + summary = prepare_remote_transport( + config, marker, url, args.timeout, + ) + elif args.mode == 'assert-remote-transport-replay': + summary = assert_remote_transport_replay( + config, marker, url, args.timeout, + ) + elif args.mode == 'prepare-remote-full-race': + summary = prepare_remote_full_race(config, marker, url) + elif args.mode == 'finish-remote-full-race': + summary = finish_remote_full_race( + config, marker, url, args.timeout, + ) + elif args.mode == 'prepare-dockerhub-canary': + summary = prepare_dockerhub_canary( + config, marker, url, args.timeout, + ) + elif args.mode == 'finish-dockerhub-canary': + summary = finish_dockerhub_canary( + config, marker, url, args.timeout, + ) + else: + baseline = json.loads(read_bytes(RESULT)) + require(baseline.get('checked') in (0, 1), 'pipeline_baseline_required') + snapshot = wait_for_pipeline(config, marker, url, baseline['checked'], args.timeout) + compare_persisted(baseline, snapshot, require_restart=args.mode == 'assert-persisted') + if args.mode == 'keycheck-fixture': + summary = run_keycheck(config, marker, url, baseline, args.timeout) + else: + summary = {'counts': {'ok': 1, **snapshot['counts']}, 'hashes': snapshot['hashes']} + require(set(summary) == {'counts', 'hashes'} + and all(type(value) is int and value >= 0 for value in summary['counts'].values()) + and all(isinstance(value, str) and re.fullmatch(r'[a-f0-9]{64}', value) + for value in summary['hashes'].values()), 'invalid_public_summary') + except Exception as exc: + check = str(exc) if isinstance(exc, E2EFailure) else 'helper_exception' + if not re.fullmatch(r'[a-z_][a-z0-9_]{0,120}', check): + check = 'helper_exception' + summary = {'counts': {'ok': 0, check: 1}, 'hashes': {}} + print(json.dumps(summary, sort_keys=True, separators=(',', ':'))) + return 0 if summary['counts']['ok'] else 1 + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/tests/container_import_stop_e2e.py b/tests/container_import_stop_e2e.py new file mode 100644 index 0000000..1a7d1c2 --- /dev/null +++ b/tests/container_import_stop_e2e.py @@ -0,0 +1,124 @@ +"""Real PostgreSQL lifecycle regression on a fresh, disposable, offline volume. + +Only truf-import-stop-test-<32 hex digits> volumes are accepted. Provision the +volume with container_runtime.py first. No scanner, provider, or user data is +opened. --expect-schema-refusal records the pre-fix behavior and withdraws the +synthetic fault before cleanup; the default requires identity-safe shutdown. +""" + +import argparse +import json +import os +from pathlib import Path +import re +import runpy +import socket +import subprocess +import sys +import time + + +def require(value, check): + if not value: + raise AssertionError(check) + + +def emit(**values): + print(json.dumps(values, sort_keys=True), flush=True) + + +def main(): + parser = argparse.ArgumentParser(allow_abbrev=False) + parser.add_argument('--expect-schema-refusal', action='store_true') + args = parser.parse_args() + require(sys.platform == 'linux' and os.geteuid() == 10001, 'test_identity') + require(Path(__file__) == Path('/opt/truf/tests/container_import_stop_e2e.py'), 'test_image') + require({name for _, name in socket.if_nameindex()} == {'lo'}, 'offline_test') + mounts = [line.split() for line in Path('/proc/self/mountinfo').read_text().splitlines()] + data = [row for row in mounts if row[4] == '/data' or row[4].startswith('/data/')] + require(len(data) == 1 and data[0][4] == '/data' + and re.fullmatch(r'/var/lib/docker/volumes/truf-import-stop-test-[a-f0-9]{32}/_data', data[0][3]) + and data[0][data[0].index('-') + 1] == 'ext4', 'disposable_test_volume') + os.umask(0o077) + runtime = runpy.run_path('/opt/truf/app/container_runtime.py') + runtime['require_container']() + require(not any(Path('/data/postgres-linux').iterdir()), 'fresh_cluster') + require(not Path('/data/runtime-linux/postgres/cluster_identity.json').exists(), 'fresh_identity') + config_path = runtime['DEFAULT_CONFIG'] + config = runtime['prepare_environment'](config_path) + import postgres_runtime as pg + import psycopg + from runtime_security import ClusterAuthorityLock + + def cli(action): + child = subprocess.Popen(runtime['_bootstrap_command']( + 'postgres-runtime', action, '--config', str(config_path)), + stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) + try: + code = child.wait(timeout=60) + except subprocess.TimeoutExpired: + emit(stage=action, status='FAILED_HOLD', reason='lifecycle_child_timeout') + # The child, not a timeout, owns compensation of an uncertain start. + code = child.wait() + require(code == 0, 'lifecycle_cli_' + action) + + def connect(): + return psycopg.connect(os.environ['SCANNER_DB_URL'], autocommit=True, + connect_timeout=3, options='-c statement_timeout=5000') + + def stop(backend, stage): + started = time.monotonic() + result = backend.stop() + emit(stage=stage, completed=result.completed, stopped=result.stopped, + elapsed_ms=round((time.monotonic() - started) * 1000), + recovering_refusal='online identity is unavailable' in result.detail) + return result.completed is True and result.stopped is True + + def cleanup(backend): + while not stop(backend, 'cleanup'): + emit(status='FAILED_HOLD', reason='stop_unconfirmed') + time.sleep(5) + require(backend.probe().kind == pg.ProbeKind.STOPPED, 'offline_proof') + backend.close() + + cli('initialize-empty') + cli('maintenance-start') + with ClusterAuthorityLock(config, endpoint_dsn=os.environ['SCANNER_DB_URL']): + backend = pg.PostgresBackend(config) + try: + require(backend.probe().kind == pg.ProbeKind.READY, 'handoff_ready') + require(stop(backend, 'healthy_handoff'), 'healthy_handoff_stop') + finally: + cleanup(backend) + + cli('maintenance-start') + with ClusterAuthorityLock(config, endpoint_dsn=os.environ['SCANNER_DB_URL']): + backend = pg.PostgresBackend(config) + fault_installed = False + stopped = False + try: + require(backend.probe().kind == pg.ProbeKind.READY, 'second_handoff_ready') + with connect() as connection: + connection.execute('GRANT CREATE ON SCHEMA public TO PUBLIC') + fault_installed = True + require(backend.probe().kind != pg.ProbeKind.READY, 'unsafe_schema_not_ready') + stopped = stop(backend, 'schema_failure_handoff') + require(stopped is not args.expect_schema_refusal, 'schema_failure_stop_expectation') + finally: + # Withdraw only this synthetic fault, never weaken a production gate. + if fault_installed and not stopped: + with connect() as connection: + connection.execute('REVOKE CREATE ON SCHEMA public FROM PUBLIC') + cleanup(backend) + require(not Path('/data/initialized.json').exists(), 'no_application_initialization') + emit(status='passed', expected_schema_refusal=args.expect_schema_refusal) + return 0 + + +if __name__ == '__main__': + try: + result = main() + except BaseException as error: + emit(status='failed', error_type=type(error).__name__) + result = 1 + raise SystemExit(result) diff --git a/tests/container_projection_recovery_e2e.py b/tests/container_projection_recovery_e2e.py new file mode 100644 index 0000000..7d1e85f --- /dev/null +++ b/tests/container_projection_recovery_e2e.py @@ -0,0 +1,668 @@ +"""Real PG16 regression; run only in the trusted, offline disposable test image. + +Main provisions a fresh truf-projection-loss-test-<32 hex> volume separately. +Run python -u -I -S -B /opt/truf/tests/container_projection_recovery_e2e.py +--budget-seconds 300. The budget is cooperative: it NEVER kills PostgreSQL or +releases uncertain lifecycle authority. A FAILED_HOLD may outlive that budget. + +The actual schema/migration helpers, queries, transactions and locks are used. +Only the recovery module's manifest pin is substituted in memory for this fresh +fixture. Fault adapters raise only AFTER real SQL execution or real commit. +No accepted journal is deleted, no old output is rebuilt, and no application +initialized marker, supervisor, scanner, ingester or provider is started. +""" + +import argparse +import contextlib +import hashlib +import json +import os +from pathlib import Path +import re +import runpy +import signal +import socket +import subprocess +import sys +import time +from types import SimpleNamespace + + +OUTPUT = sys.stdout +STOPPED = False +IMAGE_PATH = Path('/opt/truf/tests/container_projection_recovery_e2e.py') +STAMP = '2026-09-16T00:00:00+00:00' +APPENDS = 38024 +OLD_OFFSET = 97783145 + + +def require(value, check): + if not value: + raise AssertionError(check) + + +def emit(**values): + try: + print(json.dumps(values, sort_keys=True), file=OUTPUT, flush=True) + except BaseException: + pass + + +def safe_error(error): + try: + line, state, current = 0, None, error + for _ in range(8): + tb = current.__traceback__ + while tb is not None: + if tb.tb_frame.f_code.co_filename == str(IMAGE_PATH): + line = tb.tb_lineno + tb = tb.tb_next + candidate = getattr(current, 'sqlstate', None) + if state is None and isinstance(candidate, str) and re.fullmatch(r'[A-Z0-9]{5}', candidate): + state = candidate + current = current.__cause__ or current.__context__ + if current is None: + break + name = type(error).__name__ + if not re.fullmatch(r'[A-Za-z_][A-Za-z0-9_]{0,63}', name or ''): + name = 'other' + return {'error_type': next((i for i, kind in enumerate( + (TimeoutError, OSError, ValueError, RuntimeError, KeyboardInterrupt, AssertionError, + TypeError, KeyError, AttributeError, ImportError, LookupError), 1) + if isinstance(error, kind)), 0), 'line': line, 'sqlstate': state, 'error_class': name} + except BaseException: + return {'error_type': 0, 'line': 0, 'sqlstate': None, 'error_class': 'other'} + + +def hold_pause(): + try: + time.sleep(2) + except BaseException: + pass + + +def stop_confirmed(backend, pg): + """Retain this exact backend and the caller's two locks until positive stop.""" + global STOPPED + attempts = 0 + while True: + try: + attempts += 1 + result = backend.stop() + require(result.completed is True and result.stopped is True, 'stop_result') + require(backend.probe().kind == pg.ProbeKind.STOPPED, 'stopped_probe') + backend.close() + STOPPED = True + emit(stage='stop', stopped=True, attempts=attempts) + return + except BaseException as error: + emit(stage='stop', failed_hold=True, attempts=attempts, **safe_error(error)) + hold_pause() + + +def initialize_empty(runtime, config_path): + """The real lifecycle child owns initialization compensation; never kill it.""" + try: + child = subprocess.Popen(runtime._bootstrap_command( + 'postgres-runtime', 'initialize-empty', '--config', str(config_path)), + stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) + except BaseException as error: + emit(stage='initialize_empty', failed_hold=True, **safe_error(error)) + return None + attempts = 0 + while True: + try: + code = child.wait(timeout=10) + emit(stage='initialize_empty', completed=code == 0, exit_code=code) + return code + except BaseException as error: + attempts += 1 + emit(stage='initialize_empty', failed_hold=True, attempts=attempts, **safe_error(error)) + hold_pause() + + +def checkpoint(runtime, deadline): + require(runtime._shutdown_requested is False, 'shutdown_requested') + if time.monotonic() >= deadline: + raise TimeoutError('driver_budget') + + +def identifier(name): + require(isinstance(name, str) and re.fullmatch(r'[a-z_][a-z0-9_]*', name), 'fixture_identifier') + return '"' + name + '"' + + +def digest(payload): + return hashlib.sha256(payload).hexdigest() + + +def metadata(connection): + row = connection.execute("""SELECT pg_catalog.row_to_json(s) AS stream, + pg_catalog.row_to_json(c) AS cursor FROM public.projection_streams s + JOIN public.projection_cursors c USING (stream_name) + WHERE s.stream_name = 'found_secrets'""").fetchone() + parsed = {} + for key in ('stream', 'cursor'): + value = row[key] + if isinstance(value, str): + value = json.loads(value) + require(isinstance(value, dict), 'metadata_object') + parsed[key] = value + return parsed + + +def proof(connection, check): + """Independent, bounded-fixture whole-row digests; never return raw rows.""" + tables = connection.execute("""SELECT c.relname AS name FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = 'public' AND c.relkind = 'r' ORDER BY c.relname""").fetchall() + result = {'tables': {}, 'sequences': {'public': {}}} + for table in tables: + check() + name = table['name'] + where = " WHERE t.stream_name <> 'found_secrets'" if name in ('projection_streams', 'projection_cursors') else '' + row = connection.execute("""SELECT count(*) AS rows, pg_catalog.encode(pg_catalog.sha256( + pg_catalog.convert_to(COALESCE(string_agg(h, '' ORDER BY h), ''), 'UTF8')), 'hex') AS sha256 + FROM (SELECT pg_catalog.encode(pg_catalog.sha256(pg_catalog.convert_to( + pg_catalog.row_to_json(t)::text, 'UTF8')), 'hex') COLLATE "C" AS h FROM public.""" + + identifier(name) + ' AS t' + where + ') AS hashes').fetchone() + require(0 <= row['rows'] <= 100000, 'bounded_fixture') + result['tables'][name] = row + sequences = connection.execute("""SELECT c.relname AS name FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = 'public' AND c.relkind = 'S' ORDER BY c.relname""").fetchall() + for row in sequences: + result['sequences']['public'][row['name']] = connection.execute( + 'SELECT last_value, is_called FROM public.' + identifier(row['name'])).fetchone() + return result + + +def journal_snapshot(runtime, recovery): + from container_import import _input, _json + + path = runtime.DATA / 'config' / recovery.JOURNAL_NAME + with _input(path, runtime) as (handle, before): + require(0 < before[4] <= recovery.MAX_JOURNAL, 'journal_bound') + raw = handle.read(recovery.MAX_JOURNAL + 1) + value = _json(raw) + require(raw == recovery._encoded(value) and value['record']['state'] == 'PREPARED' + and value['sha256'] == digest(recovery._encoded(value['record'])), 'durable_journal') + return raw, before, value + + +class ObservedConnection: + """Delegate everything to psycopg; inject only after a real operation.""" + def __init__(self, connection, runtime, recovery, check, *, after_update=False, lost_ack=False): + self.connection, self.runtime, self.recovery, self.check = connection, runtime, recovery, check + self.after_update, self.lost_ack = after_update, lost_ack + self.updates, self.commits, self.injected = 0, 0, False + + def __getattr__(self, name): + return getattr(self.connection, name) + + def execute(self, query, *args, **kwargs): + result = self.connection.execute(query, *args, **kwargs) + sql = ' '.join(query.split()).upper() if isinstance(query, str) else '' + if sql.startswith('UPDATE '): + self.updates += 1 + if self.after_update and sql.startswith('UPDATE PUBLIC.PROJECTION_STREAMS '): + require(result.rowcount == 1, 'real_first_update') + journal_snapshot(self.runtime, self.recovery) + self.after_update, self.injected = False, True + raise OSError('synthetic_transport_after_real_update') + self.check() + return result + + @contextlib.contextmanager + def transaction(self, *args, **kwargs): + with self.connection.transaction(*args, **kwargs) as transaction: + yield transaction + self.commits += 1 + if self.lost_ack: + self.lost_ack, self.injected = False, True + raise OSError('synthetic_ack_loss_after_real_commit') + + +def seed_history(connection, runtime, importer): + """Seed real production tables; historical byte proofs need no old files.""" + with connection.transaction(): + connection.execute("""INSERT INTO public.target_scans( + id, scan_event_id, scan_event_hash, source, target, normalized_target, scan_type, + status, ended_at, findings_count, raw_result_storage, created_at) + SELECT n, lpad(to_hex(n), 32, '0'), encode(sha256(convert_to('event-' || n, 'UTF8')), 'hex'), + 'fixture', 'fixture:' || n, 'fixture:' || n, 'fixture', 'found', %s, 1, 'normalized_v2', %s + FROM generate_series(1, 38024) AS g(n)""", (STAMP, STAMP)) + connection.execute("""INSERT INTO public.findings( + id, target_scan_id, source, target, detector_name, detector_type, verified, + raw_secret, redacted_secret, secret_hash, finding_uid, created_at) + SELECT id, id, 'fixture', target, 'Fixture', 'fixture', 0, + 'synthetic-only-' || id, 'fixture', encode(sha256(convert_to('synthetic-only-' || id, 'UTF8')), 'hex'), + encode(sha256(convert_to('finding-' || id, 'UTF8')), 'hex'), %s FROM public.target_scans""", (STAMP,)) + connection.execute("""INSERT INTO public.scan_result_compat( + target_scan_id, schema_version, metadata_json, metadata_sha256, metadata_bytes, reconstruction_status, created_at) + SELECT id, 2, '{}', encode(sha256(convert_to('{}', 'UTF8')), 'hex'), 2, 'exact', %s + FROM public.target_scans""", (STAMP,)) + connection.execute("""INSERT INTO public.finding_compat_payloads( + finding_id, raw_value, extension_json, payload_sha256, payload_bytes, payload_omitted, created_at) + SELECT id, raw_secret, '{}', encode(sha256(convert_to(raw_secret, 'UTF8')), 'hex'), + octet_length(raw_secret), 0, %s FROM public.findings""", (STAMP,)) + connection.execute("""INSERT INTO public.projection_jobs( + id, job_kind, event_id, event_hash, target_scan_id, status, required_stream_mask, + capacity_items, capacity_bytes, capacity_released, attempts, created_at, updated_at, completed_at) + SELECT id, 'scan_event', scan_event_id, scan_event_hash, id, 'completed', 2, + 1, 65536, 1, 1, %s, %s, %s FROM public.target_scans""", (STAMP, STAMP, STAMP)) + connection.execute("""WITH positions AS ( + SELECT id, event_id, event_hash, (id - 1) / 2716 AS generation, + CASE WHEN id > 35308 THEN 97783145::bigint ELSE 134217728::bigint END AS bytes, + (id - 1) %% 2716 AS ordinal FROM public.projection_jobs) + INSERT INTO public.projection_appends(id, job_id, stream_name, event_id, event_hash, + generation, byte_offset, byte_length, payload_sha256, record_count, state, prepared_at, appended_at) + SELECT id, id, 'found_secrets', event_id, event_hash, generation, + bytes * ordinal / 2716, bytes * (ordinal + 1) / 2716 - bytes * ordinal / 2716, + encode(sha256(convert_to('historical-output-' || id, 'UTF8')), 'hex'), + 1, 'appended', %s, %s FROM positions""", (STAMP, STAMP)) + connection.execute("""INSERT INTO public.projection_rotations( + id, stream_name, from_generation, to_generation, source_bytes, segment_relative_path, state, created_at, completed_at) + SELECT n + 1, 'found_secrets', n, n + 1, 134217728, + 'found_secrets.g' || lpad(n::text, 6, '0') || '.jsonl', 'completed', %s, %s + FROM generate_series(0, 12) AS g(n)""", (STAMP, STAMP)) + for i in range(18): + provider = f'p{i:02d}' + for suffix, filename in (('results', 'Results.jsonl'), ('status', 'Checked.txt')): + if suffix == 'status' and i >= 13: + continue + name = f'keycheck:{provider}:{suffix}' + connection.execute("""INSERT INTO public.projection_streams( + stream_name, base_relative_path, current_generation, rotation_bytes, max_generations, created_at, updated_at) + VALUES (%s, %s, 0, 33554432, 16, %s, %s)""", (name, f'{provider}/{provider}{filename}', STAMP, STAMP)) + connection.execute("""INSERT INTO public.projection_cursors(stream_name, generation, committed_offset, updated_at) + VALUES (%s, 0, %s, %s)""", (name, 1000 + i if suffix == 'status' else 0, STAMP)) + connection.execute("UPDATE public.projection_streams SET current_generation = 13 WHERE stream_name = 'found_secrets'") + connection.execute("""UPDATE public.projection_cursors SET generation = 13, committed_offset = 97783145, + last_append_id = 38024, last_job_id = 38024, + last_event_id = (SELECT event_id FROM public.projection_jobs WHERE id = 38024), + last_event_hash = (SELECT event_hash FROM public.projection_jobs WHERE id = 38024) + WHERE stream_name = 'found_secrets'""") + for worker in ('result_ingester', 'jsonl_projector'): + connection.execute("""INSERT INTO public.pipeline_leases(worker_name, generation, lease_token, + supervisor_instance_id, owner_pid, owner_creation_time, owner_executable, state, + acquired_at, heartbeat_at, lease_expires_at, updated_at) + VALUES (%s, 7, 'synthetic-expired', 'synthetic-expired', 999999, 'synthetic', + '/synthetic/expired', 'ready', %s, %s, %s, %s)""", (worker, STAMP, STAMP, STAMP, STAMP)) + for table in ('target_scans', 'findings', 'projection_jobs', 'projection_appends', 'projection_rotations'): + connection.execute('SELECT pg_catalog.setval(pg_catalog.pg_get_serial_sequence(%s, %s), ' + + '(SELECT max(id) FROM public.' + identifier(table) + '), true)', ('public.' + table, 'id')) + status_files = [] + for i in range(13): + provider = f'p{i:02d}' + directory = runtime.DATA / 'runtime-linux/keychecks' / provider + directory.mkdir(mode=0o700) + path = directory / f'{provider}Checked.txt' + importer._write(runtime, path, b'synthetic status snapshot\n') + status_files.append(path) + return status_files + + +def fixture_manifest(connection, runtime, importer, recovery, baseline, status_files, system_identifier): + # Required source labels are inert format metadata, never opened as paths. + paths = list(status_files) + for name in sorted(importer.REQUIRED_FILES): + path = runtime.DATA / name + if not os.path.lexists(path): + path.parent.mkdir(mode=0o700, parents=True, exist_ok=True) + importer._write(runtime, path, b'') + paths.append(path) + files = [] + for path in paths: + with importer._input(path, runtime) as (handle, before): + raw = handle.read(4096) + require(len(raw) == before[4], 'status_fixture_bound') + files.append({'path': path.relative_to(runtime.DATA).as_posix(), 'size': len(raw), 'sha256': digest(raw)}) + # Historical bytes are intentionally not materialized; this is not a restore test. + files.append({'path': 'runtime-linux/results/found_secrets.jsonl', 'size': OLD_OFFSET, + 'sha256': digest(b'intentionally-unavailable-synthetic-output')}) + value = {'format': 'truf-windows-snapshot-v1', + 'source': {'root': r'D:\truf', 'postgres_data_dir': r'S:\postgres-data', + 'supervisor_stopped': True, 'postgres_stopped': True}, + 'database': {'version_num': connection.execute("SELECT current_setting('server_version_num')::int AS version").fetchone()['version'], + 'system_identifier': '1' if system_identifier != '1' else '2', + 'database_name': 'synthetic_source', 'user_name': 'synthetic_source', 'port': 15432, + 'data_directory': r'S:\postgres-data', 'bytes': 6, 'sha256': digest(b'PGDMPx'), + 'table_counts': {name: item['rows'] + int(name in ('projection_streams', 'projection_cursors')) + for name, item in baseline['tables'].items()}, + 'sequence_states': baseline['sequences'], 'sequence_count': len(baseline['sequences']['public'])}, + 'archive': {'bytes': sum(item['size'] for item in files) + 10240, + 'sha256': digest(b'synthetic archive identity')}, 'files': files} + raw = importer._encoded(value) + pin = digest(raw) + importer._manifest(raw, pin) + importer._write(runtime, runtime.DATA / 'config/windows-import-manifest.json', raw) + recovery.APPROVED_MANIFEST_SHA256 = pin + return pin + + +def exercise_projector(connection, db, runtime, config, check): + from jsonl_projector import JsonlProjector + from migrate_runtime_safety import initialize_projection_cursors_from_existing_files, require_legacy_cutover_clear + + # This is a real offline cutover after recovery, not a mocked readiness gate. + db.require_runtime_safety_schema() + cursors = initialize_projection_cursors_from_existing_files(db, config) + legacy = require_legacy_cutover_clear(db, config) + migrations = db.conn.execute('SELECT version, code_sha256 FROM runtime_schema_migrations ORDER BY version').fetchall() + db.conn.commit() + db.record_final_cutover({'legacy': legacy, 'projection_cursors': cursors, + 'schema_migrations': [dict(row) for row in migrations]}) + db.require_final_cutover() + check() + event_id, event_hash, finding_uid = 'f' * 32, digest(b'future-event'), digest(b'future-finding') + with connection.transaction(): + scan_id = connection.execute("""INSERT INTO public.target_scans( + scan_event_id, scan_event_hash, target, normalized_target, scan_type, status, ended_at, + findings_count, raw_result_storage, created_at) + VALUES (%s, %s, 'fixture:future', 'fixture:future', 'fixture', 'found', %s, 1, 'normalized_v2', %s) + RETURNING id""", (event_id, event_hash, STAMP, STAMP)).fetchone()['id'] + connection.execute("""INSERT INTO public.scan_result_compat(target_scan_id, schema_version, + metadata_json, metadata_sha256, metadata_bytes, reconstruction_status, created_at) + VALUES (%s, 2, '{}', %s, 2, 'exact', %s)""", (scan_id, digest(b'{}'), STAMP)) + finding_id = connection.execute("""INSERT INTO public.findings(target_scan_id, detector_name, + detector_type, verified, raw_secret, redacted_secret, finding_uid, created_at) + VALUES (%s, 'Fixture', 'fixture', 0, 'synthetic-future', 'fixture', %s, %s) RETURNING id""", + (scan_id, finding_uid, STAMP)).fetchone()['id'] + connection.execute("""INSERT INTO public.finding_compat_payloads(finding_id, raw_value, + extension_json, payload_sha256, payload_bytes, payload_omitted, created_at) + VALUES (%s, 'synthetic-future', '{}', %s, 16, 0, %s)""", (finding_id, digest(b'synthetic-future'), STAMP)) + job_id = connection.execute("""INSERT INTO public.projection_jobs(job_kind, event_id, event_hash, + target_scan_id, status, required_stream_mask, capacity_items, capacity_bytes, created_at, updated_at) + VALUES ('scan_event', %s, %s, %s, 'pending', 2, 1, 65536, %s, %s) RETURNING id""", + (event_id, event_hash, scan_id, STAMP, STAMP)).fetchone()['id'] + connection.execute("""UPDATE public.pipeline_capacity SET projection_items = projection_items + 1, + projection_bytes = projection_bytes + 65536 WHERE id = 1""") + worker = JsonlProjector(db, str(runtime.DATA / 'runtime-linux/results'), 'projection-loss-e2e', + keycheck_dir=str(runtime.DATA / 'runtime-linux/keychecks')) + try: + worker.start() + require(worker.process_one() is True, 'projector_processed') + require(worker.process_one() is False, 'projector_idle') + row = connection.execute("""SELECT a.generation, a.byte_offset, a.byte_length, + a.payload_sha256, a.state, j.status, j.capacity_released + FROM public.projection_appends a JOIN public.projection_jobs j ON j.id = a.job_id + WHERE a.job_id = %s AND a.stream_name = 'found_secrets'""", (job_id,)).fetchone() + path = runtime.DATA / 'runtime-linux/results/found_secrets.jsonl' + runtime.private_path(path) + require(path.stat().st_size < 65536, 'bounded_new_output') + raw = path.read_bytes() + require(row and row['generation'] == 14 and row['byte_offset'] == 0 + and row['byte_length'] == len(raw) and row['payload_sha256'] == digest(raw) + and row['state'] == 'appended' and row['status'] == 'completed' and row['capacity_released'] == 1, + 'projector_fenced_append') + require(json.loads(raw)['finding_uid'] == finding_uid, 'projector_real_payload') + cursor = metadata(connection)['cursor'] + require(cursor['generation'] == 14 and cursor['committed_offset'] == len(raw) + and cursor['last_job_id'] == job_id, 'projector_cursor') + capacity = connection.execute('SELECT projection_items, projection_bytes FROM public.pipeline_capacity WHERE id = 1').fetchone() + require(capacity == {'projection_items': 0, 'projection_bytes': 0}, 'projector_capacity') + emit(stage='projector', passed=True, generation=14, records=1, bytes=len(raw)) + finally: + worker.stop() + + +def run_cases(connection, connect, resources, runtime, config, identity, initialize_lock, authority_lock, check): + import container_import as importer + import container_projection_recovery as recovery + from runtime_security import PrivateFileLock + from scanner_db import ScannerDB, migrate_runtime_safety_schema + + schema_db = ScannerDB(db_url=os.environ['SCANNER_DB_URL'], initialize=False) + resources.append(schema_db) + require(schema_db.enabled and schema_db.conn.is_postgres, 'real_schema_connection') + try: + migrate_runtime_safety_schema(schema_db, initialize_base=True) + schema_db.require_runtime_safety_schema() + finally: + schema_db.close() + emit(stage='schema', passed=True) + check() + status_files = seed_history(connection, runtime, importer) + emit(stage='seed', passed=True) + baseline = proof(connection, check) + emit(stage='proof', passed=True, tables=len(baseline['tables']), sequences=len(baseline['sequences']['public'])) + before = metadata(connection) + require(before['stream']['current_generation'] == before['cursor']['generation'] == 13 + and before['cursor']['committed_offset'] == OLD_OFFSET, 'reviewed_before') + require(baseline['tables']['projection_appends']['rows'] == APPENDS + and baseline['tables']['projection_rotations']['rows'] == 13 + and baseline['tables']['projection_append_audit']['rows'] == 0, 'reviewed_history') + original_pin = recovery.APPROVED_MANIFEST_SHA256 + pin = fixture_manifest(connection, runtime, importer, recovery, baseline, status_files, identity['system_identifier']) + journal_path = runtime.DATA / 'config' / recovery.JOURNAL_NAME + status_before = {path: (path.stat().st_ino, path.read_bytes()) for path in status_files} + emit(stage='fixture', passed=True, streams=34, appends=APPENDS, rotations=13, + tables=len(baseline['tables']), sequences=len(baseline['sequences']['public'])) + + def recover(observed): + return recovery.recover_found_secrets_projection(runtime, observed, + system_identifier=identity['system_identifier'], manifest_sha256=pin, + initialize_lock=initialize_lock, authority_lock=authority_lock) + + def refused(observed): + try: + recover(observed) + except recovery.ProjectionRecoveryError as error: + if not observed.injected and (observed.after_update or observed.lost_ack): + emit(stage='fault_not_reached', passed=False, **safe_error(error)) + return + raise AssertionError('expected_recovery_refusal') + + def alone(): + while True: + check() + connection.execute('SELECT pg_catalog.pg_stat_clear_snapshot()') + count = connection.execute("""SELECT count(*) AS count FROM pg_catalog.pg_stat_activity + WHERE backend_type = 'client backend' AND pid <> pg_backend_pid()""").fetchone()['count'] + if count == 0: + return + time.sleep(0.05) + + try: + alone() + contender = connect() + resources.append(contender) + try: + observed = ObservedConnection(connection, runtime, recovery, check) + refused(observed) + require(observed.updates == 0, 'other_client_no_updates') + with contender.transaction(): + contender.execute('LOCK TABLE public.projection_streams IN ROW EXCLUSIVE MODE') + observed = ObservedConnection(connection, runtime, recovery, check) + refused(observed) + require(observed.updates == 0, 'writer_contention_no_updates') + finally: + contender.close() + alone() + with PrivateFileLock(str(runtime.DATA / 'runtime-linux/results/.jsonl-projector.lock')): + observed = ObservedConnection(connection, runtime, recovery, check) + refused(observed) + require(observed.updates == 0, 'projector_file_lock_no_updates') + require(metadata(connection) == before and proof(connection, check) == baseline + and not os.path.lexists(journal_path), 'contention_unchanged') + emit(stage='contention', passed=True, cases=3, updates=0) + + observed = ObservedConnection(connection, runtime, recovery, check, after_update=True) + refused(observed) + require(observed.injected and observed.updates == 1 and observed.commits == 0, 'rollback_fault_window') + require(int(connection.info.transaction_status) == 0 and metadata(connection) == before + and proof(connection, check) == baseline, 'both_rows_rolled_back') + journal = journal_snapshot(runtime, recovery) + require(journal[2]['record']['before'] == before, 'journal_before') + emit(stage='rollback', passed=True, updates=1, commits=0, journal_retained=True) + + observed = ObservedConnection(connection, runtime, recovery, check, lost_ack=True) + refused(observed) + require(observed.injected and observed.updates == 2 and observed.commits == 1, 'real_commit_before_ack_loss') + after = metadata(connection) + require(after == journal[2]['record']['after'] and proof(connection, check) == baseline + and journal_snapshot(runtime, recovery) == journal, 'committed_despite_ack_loss') + emit(stage='lost_ack', passed=True, updates=2, commits=1, history_unchanged=True) + + observed = ObservedConnection(connection, runtime, recovery, check) + result = recover(observed) + require(result['status'] == 'already-committed' and observed.updates == 0 + and result['journal_sha256'] == digest(journal[0]) and metadata(connection) == after + and proof(connection, check) == baseline and journal_snapshot(runtime, recovery) == journal, + 'idempotent_readonly_retry') + require({path: (path.stat().st_ino, path.read_bytes()) for path in status_files} == status_before, + 'other_output_unchanged') + emit(stage='retry', passed=True, updates=0, journal_unchanged=True, history_unchanged=True, sequences_unchanged=True) + + writer_db = ScannerDB(db_url=os.environ['SCANNER_DB_URL'], initialize=False) + resources.append(writer_db) + require(writer_db.enabled and writer_db.conn.is_postgres, 'real_writer_connection') + try: + exercise_projector(connection, writer_db, runtime, config, check) + finally: + writer_db.close() + alone() + advanced, advanced_proof = metadata(connection), proof(connection, check) + future_path = runtime.DATA / 'runtime-linux/results/found_secrets.jsonl' + future_bytes, future_inode = future_path.read_bytes(), future_path.stat().st_ino + observed = ObservedConnection(connection, runtime, recovery, check) + refused(observed) + require(observed.updates == 0 and advanced['cursor']['committed_offset'] > 0 + and metadata(connection) == advanced and proof(connection, check) == advanced_proof + and (future_path.read_bytes(), future_path.stat().st_ino) == (future_bytes, future_inode) + and journal_snapshot(runtime, recovery) == journal, 'future_output_not_rewound') + emit(stage='future_output', passed=True, updates=0, advanced_cursor_retained=True, journal_unchanged=True) + finally: + recovery.APPROVED_MANIFEST_SHA256 = original_pin + + +def main(): + global OUTPUT + # Native pg_ctl also inherits fd 1/2: Python stream redirection alone is insufficient. + OUTPUT = os.fdopen(os.dup(1), 'w', encoding='utf-8', buffering=1) + sink = os.open(os.devnull, os.O_WRONLY) + try: + os.dup2(sink, 1) + os.dup2(sink, 2) + finally: + os.close(sink) + parser = argparse.ArgumentParser(allow_abbrev=False) + parser.add_argument('--budget-seconds', type=int, default=300) + args = parser.parse_args() + require(30 <= args.budget_seconds <= 3600, 'budget') + started = time.monotonic() + deadline = started + args.budget_seconds + require(sys.platform == 'linux' and os.geteuid() == os.getuid() == 10001 + and os.getegid() == os.getgid() == 10001, 'test_identity') + require(Path(__file__) == IMAGE_PATH and sys.flags.isolated and sys.flags.no_site + and sys.flags.dont_write_bytecode, 'test_image') + require({name for _, name in socket.if_nameindex()} == {'lo'}, 'offline_test') + raw_mounts = Path('/proc/self/mountinfo').read_bytes() + require(len(raw_mounts) <= 1024 * 1024, 'mount_bound') + mounts = [line.split() for line in raw_mounts.decode('utf-8', errors='strict').splitlines()] + data = [row for row in mounts if len(row) > 6 and (row[4] == '/data' or row[4].startswith('/data/'))] + require(len(data) == 1 and data[0][4] == '/data' and '-' in data[0] + and re.fullmatch(r'/var/lib/docker/volumes/truf-projection-loss-test-[a-f0-9]{32}/_data', data[0][3]) + and data[0][data[0].index('-') + 1] == 'ext4', 'fresh_test_volume') + runtime = SimpleNamespace(**runpy.run_path('/opt/truf/app/container_runtime.py')) + runtime.require_container() + runtime.private_path(IMAGE_PATH.parent, directory=True) + runtime.private_path(IMAGE_PATH) + runtime._shutdown_requested = False + for sig in (signal.SIGTERM, signal.SIGINT, signal.SIGHUP): + signal.signal(sig, lambda *_: setattr(runtime, '_shutdown_requested', True)) + + def fresh(): + runtime.private_path(runtime.DATA / 'postgres-linux', directory=True) + require(not any((runtime.DATA / 'postgres-linux').iterdir()), 'empty_pgdata') + require(not any((runtime.DATA / 'runtime-linux/results').iterdir()), 'empty_results') + for name in ('runtime-linux/postgres/cluster_identity.json', 'initialized.json', + 'config/windows-import-manifest.json', 'config/found-secrets-loss-g13-g14.prepared.json'): + require(not os.path.lexists(runtime.DATA / name), 'fresh_fixture') + + fresh() + config_path = runtime.DEFAULT_CONFIG + config = runtime.prepare_environment(config_path) + os.environ.update(TRUF_DB_STATEMENT_TIMEOUT_MS='120000', TRUF_DB_LOCK_TIMEOUT_MS='5000', + TRUF_DB_IDLE_TRANSACTION_TIMEOUT_MS='300000') + import postgres_runtime as pg + import psycopg + from psycopg.rows import dict_row + from runtime_security import ClusterAuthorityLock, PrivateFileLock + + def connect(): + return psycopg.connect(os.environ['SCANNER_DB_URL'], autocommit=True, row_factory=dict_row, + connect_timeout=5, application_name='truf-projection-loss-e2e', tcp_user_timeout=30000, + options='-c search_path=public -c statement_timeout=120000 -c lock_timeout=5000 ' + '-c idle_in_transaction_session_timeout=300000 -c row_security=off ' + '-c log_min_error_statement=panic -c log_min_messages=panic ' + '-c log_statement=none -c log_min_duration_statement=-1') + + resources = [] + check = lambda: checkpoint(runtime, deadline) + with PrivateFileLock(str(runtime.INITIALIZE_LOCK)) as initialize_lock: + fresh() + authority_lock = ClusterAuthorityLock(config, endpoint_dsn=os.environ['SCANNER_DB_URL']) + code = initialize_empty(runtime, config_path) + while True: + try: + authority_lock.acquire() + break + except BaseException as error: + emit(stage='authority', failed_hold=True, **safe_error(error)) + hold_pause() + try: + backend = None + while backend is None: + try: + backend = pg.PostgresBackend(config, stop_timeout_sec=60) + except BaseException as error: + emit(stage='backend', failed_hold=True, **safe_error(error)) + hold_pause() + try: + require(code == 0, 'initialize_empty_exit') + check() + require(backend.probe().kind == pg.ProbeKind.STOPPED, 'initial_stopped_probe') + identity = pg.verify_cluster_identity(config) + require(identity['pg_major'] == 16 and identity['data_directory'] == '/data/postgres-linux', 'bound_fixture') + require(backend.start().accepted is True, 'direct_backend_start') + while True: + check() + probe = backend.probe() + if probe.kind == pg.ProbeKind.READY: + break + require(probe.kind == pg.ProbeKind.RECOVERING, 'authenticated_start') + time.sleep(0.1) + emit(stage='start', ready=True) + connection = connect() + resources.append(connection) + count = connection.execute("""SELECT count(*) AS count FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = 'public' AND c.relkind IN ('r','p','f')""").fetchone()['count'] + require(count == 0, 'virgin_schema') + run_cases(connection, connect, resources, runtime, config, identity, initialize_lock, authority_lock, check) + require(not os.path.lexists(runtime.INITIALIZED), 'no_application_marker') + finally: + try: + for resource in reversed(resources): + try: + resource.close() + except BaseException as error: + emit(stage='client_close', failed_hold=True, **safe_error(error)) + finally: + stop_confirmed(backend, pg) + finally: + authority_lock.release() + emit(stage='finished', passed=True, stopped=STOPPED, application_initialized=False, + elapsed_ms=round((time.monotonic() - started) * 1000)) + return 0 + + +if __name__ == '__main__': + try: + result = main() + except BaseException as error: + emit(stage='finished', passed=False, stopped=STOPPED, **safe_error(error)) + result = 1 + raise SystemExit(result) diff --git a/tests/container_unit.py b/tests/container_unit.py new file mode 100644 index 0000000..a87d20f --- /dev/null +++ b/tests/container_unit.py @@ -0,0 +1,508 @@ +"""Explicit offline regressions for the Docker test target. + +Run as UID 10001 with read-only /opt/truf/app, writable /tmp tmpfs, and +Docker --network none: python -I -S -B /opt/truf/tests/container_unit.py. +No /data mount, runtime configuration, credentials, PostgreSQL server, external +Git source, or provider entrypoint is needed. SQLite fixtures stay in /tmp; +native OwnedProcess children and loopback control sockets are intentional. + +--list and --check-selection inspect source with the stdlib only, including +on the host. -k only narrows the reviewed selection; pytest arguments and +additional paths/plugins are not accepted. New test files are never discovered. +""" + +import sys + +sys.dont_write_bytecode = True + +import argparse +import ast +from contextlib import ExitStack +import os +from pathlib import Path +import runpy +import stat +import tempfile +from unittest import mock + + +TESTS = Path(__file__).resolve().parent +ROOT = TESTS.parent +APP = ROOT / 'app' + +SELECTION = { + 'test_db_backend_safety.py': ( + 'PostgresTimeoutTests', + ), + 'test_docker_foundation.py': ( + 'PortablePathTests', + 'DockerStagingTests::test_native_checkout_default_is_derived_from_this_copy', + 'DockerStagingTests::test_python_entrypoints_refuse_before_application_imports', + 'DockerStagingTests::test_all_copied_powershell_tools_have_static_refusals', + 'DockerStagingTests::test_docker_context_exceptions_are_explicit_source_files_only', + ), + # Linux metadata and runtime I/O are mocked; filesystem fixtures stay in tmp_path. + 'test_container_runtime.py': ( + 'test_private_path_accepts_exact_owner_only_mode', + 'test_private_path_rejects_nonabsolute_or_parent_traversal_before_stat', + 'test_private_path_rejects_symlink_at_every_component', + 'test_private_path_fails_closed_on_missing_or_uninspectable_components', + 'test_private_path_rejects_wrong_owner_group_type_or_exact_mode', + 'test_container_requires_private_image_native_data_and_dedicated_tmpfs', + 'test_container_refuses_before_mount_inventory_when_identity_gate_fails', + 'test_root_is_permitted_only_for_explicit_provisioning', + 'test_container_rejects_nonprivate_required_paths', + 'test_container_rejects_missing_shared_or_host_style_storage', + 'test_new_private_file_is_exclusive_nofollow_and_durable', + 'test_new_private_file_never_overwrites_existing_file', + 'test_fresh_provision_generates_private_password_once_without_initializing_pg', + 'test_provision_refuses_nonempty_or_partial_volume_without_repair', + 'test_provision_lock_conflict_never_creates_layout_or_credentials', + 'test_provision_does_not_regenerate_incomplete_marked_volume', + 'test_provision_adds_only_managed_files_to_legacy_marked_volume', + 'test_provision_refuses_wrong_existing_managed_files_path', + 'test_bad_provision_marker_prevents_environment_or_application_imports', + 'test_prepare_environment_scrubs_dsn_and_runtime_overrides_before_imports', + 'test_prepare_rejects_invalid_generated_password_before_imports', + 'test_prepare_refuses_config_escaping_fixed_storage_contract', + 'test_only_default_image_config_or_private_data_config_is_accepted', + 'test_bootstrap_target_is_the_real_runtime_safety_migration_module', + 'test_initialize_publishes_marker_only_after_migration_and_confirmed_stop', + 'test_failed_initialization_always_stops_maintenance_and_never_marks_success', + 'test_shutdown_during_initialization_never_publishes_early_marker', + 'test_partial_initialization_is_not_adopted_or_repaired', + 'test_initialized_marker_must_match_bound_cluster_identity', + 'test_health_does_not_import_database_or_control_before_valid_marker', + 'test_health_requires_active_supervisor_and_explicit_postgres_ready', + 'test_health_rejects_missing_or_unready_pipeline_processes_before_db', + 'test_health_requires_each_enabled_pipeline_worker', + 'test_health_allows_explicitly_disabled_janitor', + 'test_health_requires_worker_api_only_when_explicitly_enabled', + 'test_strict_health_requires_enabled_worker_api_before_database', + 'test_strict_health_probes_exact_unauthorized_worker_endpoint', + 'test_strict_worker_probe_fails_closed_before_database', + 'test_health_rejects_failed_or_uncertain_managed_source', + 'test_health_checks_readonly_schema_cutover_leases_identity_then_storage', + 'test_health_fails_closed_and_closes_database_on_each_database_gate', + 'test_health_requires_writable_nonfull_persistent_storage', + 'test_secret_import_refuses_active_runtime_before_reading_stdin', + 'test_secret_import_cluster_lock_conflict_precedes_stopped_check_and_stdin', + 'test_secret_import_is_locked_atomic_private_and_never_prints_credentials', + 'test_invalid_secret_input_is_bounded_and_does_not_leak_or_replace', + 'test_secret_import_validation_error_clears_candidate_traceback_locals', + 'test_secret_import_rejects_config_hash_drift_before_temporary_write', + 'test_secret_import_rechecks_config_hash_before_replacement', + 'test_write_new_clears_secret_payload_from_traceback_locals', + 'test_initialize_rejects_config_hash_drift_before_mutation', + 'test_secret_import_error_removes_temporary_file_and_preserves_old_credentials', + 'test_main_container_gate_precedes_every_action', + 'test_main_provision_does_not_enter_nonroot_environment_or_database', + 'test_main_import_secrets_uses_preliminary_environment_validation', + 'test_main_initialize_binds_validated_config_hash', + 'test_main_run_rechecks_config_hash_before_exec', + 'test_main_snapshot_requires_approved_digest_and_default_config', + 'test_main_snapshot_dispatch_preserves_module_signal_state_without_initialize', + 'test_manifest_digest_is_rejected_for_other_actions', + 'test_worker_api_requirement_flag_is_restricted_to_health', + 'test_strict_status_dispatches_worker_api_requirement', + 'test_status_stdout_contains_readiness_but_no_credentials', + ), + 'test_runtime_document.py': ( + 'RuntimeDocumentTests', + ), + 'test_managed_files.py': ( + 'ManagedFileConfigurationTests', + 'ManagedFileTraversalTests', + 'ManagedFileTraversalLinuxTests', + ), + 'test_operations_schema.py': ('OperationsSchemaTests',), + 'test_operations_control.py': ('OperationsControlTests',), + 'test_host_agent_protocol.py': ( + 'HostAgentProtocolTests', + 'HostAgentClientTests', + 'HostAgentServerTests', + ), + 'test_host_agent_linux.py': ( + 'HostAgentLinuxTests', + ), + 'test_host_agent_apply.py': ( + 'HostAgentApplyTests', + ), + 'test_host_agent_deploy.py': ( + 'HostAgentDeployTests', + 'HostAgentInstallerTests', + ), + 'test_host_agent_lifecycle.py': ( + 'HostAgentLifecycleTests', + ), + 'test_host_agent_reconcile.py': ( + 'HostAgentReconcileTests', + ), + 'test_host_agent_runtime.py': ( + 'HostAgentRuntimeTests', + ), + 'test_host_agent_state.py': ( + 'HostAgentStateTests', + ), + 'test_runtime_document_io.py': ( + 'RuntimeDocumentIOTests', + ), + 'test_operations_service.py': ( + 'OperationsServiceTests', + ), + 'test_owned_process.py': ( + 'OwnedProcessTests', + 'WindowsOwnedProcessTests', + 'StaticProcessSafetyTests', + ), + 'test_owned_process_linux.py': ( + 'LinuxIdentityTests', + 'StartupHandshakeTests', + 'LinuxContainmentTests', + 'LinuxOwnedProcessIntegrationTests', + ), + 'test_owned_process_boundary.py': ( + 'OwnedProcessHostBoundaryTests', + 'CredentialBoundaryTests', + ), + 'test_runtime_bootstrap_authority.py': ( + 'RuntimeBootstrapAuthorityTests::test_authenticated_docker_shadow_dispatches_exact_entrypoint', + 'RuntimeBootstrapAuthorityTests::test_provider_bootstrap_consumes_one_required_separator_before_provider_parse', + 'RuntimeBootstrapAuthorityTests::test_provider_bootstrap_rejects_missing_and_duplicate_separators', + 'RuntimeBootstrapAuthorityTests::test_authenticated_entrypoint_exception_is_classified_as_runtime_failure', + 'RuntimeBootstrapAuthorityTests::test_command_builders_use_exact_isolation_flag_order', + 'RuntimeBootstrapAuthorityTests::test_direct_mutating_supervisor_fails_before_config_env_locks_or_children', + ), + 'test_supervisor_foreground_shutdown.py': ( + 'test_signal_callback_is_only_an_idempotent_flag_assignment', + 'test_receipt_parity_and_finalization_order', + 'test_windows_does_not_register_signals', + 'test_mutating_launch_gates_precede_config_locks_children_and_signals', + 'test_non_owning_dispatch_does_not_register_signals', + 'test_pending_term_cannot_admit_startup', + 'test_background_activation_wait_consumes_term', + 'test_admission_rejects_pending_term_before_checkpoint', + 'test_control_activation_cannot_reopen_pending_shutdown', + 'test_background_activation_callback_failure_is_sticky', + 'test_term_between_pipeline_starts_admits_no_more_children', + 'test_checkpoints_exit_without_polling_or_starting', + 'test_term_between_source_polls_rejects_next_poll', + 'test_foreground_waits_consume_term_without_input_or_next_refresh', + 'test_metadata_publication_failure_leaves_closed_memory_gates', + 'test_repeated_term_checkpoint_does_not_republish_or_rearm_retry', + 'test_repeated_shutdown_does_not_discard_confirmed_stop_proof', + 'test_initial_cleanup_interruption_retains_authority_and_failure', + 'test_interrupted_cleanup_lock_acquisition_retains_authority', + 'test_partially_published_activation_uses_full_cleanup', + 'test_coordinated_shutdown_interruption_closes_gate_before_propagating', + 'test_failed_hold_survives_interruption_and_logging_failure', + 'test_failed_hold_does_not_release_with_unconfirmed_children', + 'test_receipt_write_failure_is_nonzero', + 'test_late_cleanup_failures_are_in_receipt', + 'test_failure_latched_by_draining_control_worker_is_in_receipt', + 'test_terminal_foreground_failure_is_sticky_without_changing_restart_policy', + 'test_foreground_source_failure_survives_cleanup_status_reset', + 'test_foreground_missing_autostart_is_nonzero', + 'test_system_exit_status_is_preserved', + 'test_interrupt_message_failure_cannot_publish_success', + 'test_authority_drift_failure_remains_sticky', + 'test_preactivation_close_failure_is_sticky', + 'test_noninteractive_keeps_existing_completion_policy', + 'test_stopper_requires_posix_receipt_and_preserves_windows_crosscheck', + 'test_unconfirmed_child_defers_all_postgres_actions', + 'test_children_precede_postgres_and_stop_close_share_budget', + 'test_postgres_wait_consumes_close_budget', + ), + 'test_supervisor_startup_rollback.py': ( + 'test_unconfirmed_launch_rollback_retains_owner_until_stop_retry', + 'test_retained_rollback_refuses_start_schedule_and_poll', + 'test_confirmed_launch_rollback_clears_owner_and_allows_explicit_retry', + 'test_constructor_failure_without_returned_owner_still_closes_log', + 'test_rollback_interruption_propagates_without_losing_owner', + 'test_log_cleanup_error_does_not_undo_confirmed_exit', + 'test_log_pump_with_failed_thread_start_can_close_on_rollback_retry', + 'test_rollback_pending_closes_admission_before_failed_hold_publication', + 'test_owned_child_uncertain_defers_postgres_until_retained_retry_confirms_exit', + 'test_main_retains_locks_and_control_until_startup_rollback_is_confirmed', + 'test_locked_stale_metadata_recovery_uses_exact_identity_not_pid_or_timestamp', + 'test_foreground_reconciliation_uses_existing_exact_identity_api', + 'test_live_or_uncertain_owner_metadata_is_never_replaced', + 'test_metadata_bound_to_another_instance_path_is_not_removed', + 'test_reconciliation_error_or_interruption_preserves_metadata_for_retry', + 'test_changed_instance_during_matching_removal_fails_closed', + ), + 'test_supervisor_managed_postgres_gate.py': ('SupervisorManagedPostgresGateTests',), + 'test_observer_only_coordinated_shutdown.py': ('ObserverOnlyCoordinatedShutdownTests',), + 'test_discovery_producer_supervisor.py': ('DiscoveryProducerSupervisorTests',), + 'test_distributed_core_profile.py': ( + 'test_default_distributed_core_profile_is_exact', + 'test_keychecks_are_independent_from_discovery_profile', + 'test_core_wrapper_selects_only_discovery_producers', + ), + 'test_discovery_only_cycle.py': ('DiscoveryOnlyCycleTests',), + 'test_discovery_request_budgets.py': ( + 'test_each_discovery_page_has_one_shared_attempt_deadline', + 'test_discovery_proxy_uses_source_read_timeout_and_at_most_three_attempts', + 'test_late_discovery_success_fails_closed_without_advancing_page', + ), + 'test_supervisor_safety.py': ( + 'AuthenticatedControlTests', 'DependencyFailureTests', 'DashboardManagerTests', + ), + 'test_postgres_runtime.py': ( + 'PostgresRuntimePathTests', + 'PostgresControllerTests', + 'PostgresBackendTests', + 'PostgresEntrypointTests', + ), + 'test_container_security.py': ( + 'NativeExecutablePolicyTests', + 'WindowsNativePolicyTests', + 'ContainerLifecyclePolicyTests', + 'LinuxFilesystemSecurityTests', + ), + 'test_runtime_security.py': ('RuntimeSecurityTests',), + 'test_postgres_empty_initialization.py': ( + 'NativePostgresPathTests', + 'InitializeEmptyTests', + 'InitializationEntrypointTests::test_cli_dispatches_to_lock_owning_initializer_after_preflight', + 'InitializationEntrypointTests::test_container_refusal_and_full_layout_gate_precede_application_imports', + 'InitializationEntrypointTests::test_bounded_maintenance_stop_requires_proof_and_never_closes_a_supplied_backend', + 'InitializationEntrypointTests::test_bootstrap_cannot_replace_an_existing_identity', + 'MaintenanceEntrypointTests', + ), + 'test_process_identity_linux.py': ('LinuxProcessIdentityTests',), + 'test_container_migration_paths.py': ( + 'LegacySpoolPathTests', + 'ImmutableNativeHardeningTests', + 'LegacyCutoverChecksTests', + 'NormalMigrationContractTests', + ), + 'test_container_provider_portability.py': ('AskpassTests', 'ProviderDeadlineTests'), + 'test_container_e2e_helpers.py': ('ContainerE2EHelperTests', 'ContainerE2EFreshVolumeTests'), + 'test_container_import.py': ('ContainerImportTests',), + 'test_container_import_config.py': ('ContainerImportConfigTests',), + 'test_container_projection_recovery.py': ('ProjectionRecoveryTests',), + 'test_result_bundle_v2.py': ('ResultBundleV2Tests',), + 'test_pipeline_cutover_invariants.py': ('PipelineCutoverInvariantTests',), + 'test_custom_provider_detector_compatibility.py': ( + 'CustomProviderDetectorPolicyTests', 'CustomProviderDetectorCLITests', + ), + 'test_scan_execution.py': ('ScanExecutionTests',), + 'test_synthetic_llm_pipeline.py': ('SyntheticLLMPipelineTests',), + 'test_worker_api.py': ( + 'WorkerServiceTests', 'WorkerAPIRouteTests', 'WorkerSlotRecoveryTests', + ), + 'test_worker_api_runtime.py': ( + 'ConfiguredWorkerServiceTests', 'WorkerSupervisorWiringTests', + ), + 'test_worker_assignment.py': ('RemoteGitAssignmentBuilderTests',), + 'test_worker_assignment_runner.py': ( + 'WorkerAssignmentRunnerProtocolTests', 'WorkerSlotRunnerTests', + ), + 'test_worker_runner_handoff_linux.py': ('LinuxWorkerRunnerHandoffTests',), + 'test_worker_contracts.py': ('WorkerContractTests',), + 'test_worker_cli.py': ('WorkerCLITests',), + 'test_worker_local_state.py': ('WorkerLocalStateTests',), + 'test_worker_supervisor.py': ('WorkerSupervisorTests',), + 'test_worker_package.py': ('WorkerPackageTests',), + 'test_multisource_execution_snapshot.py': ('MultisourceExecutionSnapshotTests',), + 'test_remote_direct_credentials.py': ( + 'test_direct_child_environment_preserves_operator_provider_settings', + 'test_direct_scanner_entries_reject_credentials_and_skip_docker_pool', + 'test_anonymous_docker_bearer_never_reads_account_pool', + 'test_direct_anonymous_docker_auth_failure_is_permanent', + 'test_huggingface_discovery_errors_never_include_response_body', + 'test_huggingface_missing_repository_is_permanent_and_not_retryable', + 'test_direct_provider_access_failures_are_permanent', + ), + 'test_remote_worker_db.py': ('RemoteWorkerDBTests',), + 'test_admin_api.py': ('AdminAPITests',), + 'test_edge_deployment.py': ( + 'test_updater_persists_canonical_state_and_expires_automatically', + 'test_updater_reuses_persisted_applied_digest_without_reloading', + 'test_updater_rejects_corrupt_state_before_commands_or_snippet_changes', + 'test_updater_rejects_symlinked_snippet_when_supported', + 'test_updater_rejects_non_ip_or_unsafe_addresses_without_commands', + 'test_updater_renders_deterministically_and_bounds_matcher_lines', + 'test_updater_rolls_back_state_and_snippet_when_reload_fails', + 'test_updater_rolls_back_without_reload_when_validation_fails', + 'test_fail2ban_filter_counts_only_redacted_supplied_bad_credentials', + 'test_fail2ban_jail_is_persistent_bounded_and_admin_only', + 'test_caddy_routes_and_failure_log_are_closed_and_redacted', + 'test_edge_compose_only_opts_runtime_namespace_into_https_publication', + 'test_edge_image_and_startup_require_pinned_caddy_hash_and_random_prefix', + 'test_edge_e2e_image_inputs_are_exact_and_host_orchestrator_stays_host_only', + 'test_production_server_omits_trufflehog_but_worker_and_test_retain_it', + 'test_production_runbook_requires_fixed_host_agent_and_runtime_paths', + 'test_real_caddy_e2e_crawls_all_admin_routes_and_detail', + ), +} + +HELPER_SOURCES = ('owned_process_helper.py', 'container_e2e.py', 'container_import_stop_e2e.py', + 'container_projection_recovery_e2e.py', 'parity_helpers.py') + + +def selected_nodes(): + """Resolve only the named declarations, without importing any test or app.""" + result = [] + for filename, selectors in SELECTION.items(): + path = TESTS / filename + if path.resolve().parent != TESTS or not path.is_file(): + raise SystemExit('container-unit: missing or nonlocal selected source: ' + filename) + tree = ast.parse(path.read_text(encoding='utf-8'), filename=str(path)) + compile(tree, str(path), 'exec') + for selector in selectors: + body = tree.body + declaration = None + for name in selector.split('::'): + matches = [node for node in body + if isinstance(node, (ast.ClassDef, ast.FunctionDef)) and node.name == name] + if len(matches) != 1: + raise SystemExit('container-unit: missing or ambiguous selection: ' + filename + '::' + selector) + declaration = matches[0] + body = declaration.body + if isinstance(declaration, ast.ClassDef): + members = [node.name for node in body + if isinstance(node, ast.FunctionDef) and node.name.startswith('test_')] + nodes = [filename + '::' + selector + '::' + member for member in members] + elif isinstance(declaration, ast.FunctionDef) and declaration.name.startswith('test_'): + nodes = [filename + '::' + selector] + else: + nodes = [] + if not nodes: + raise SystemExit('container-unit: empty test selection: ' + filename + '::' + selector) + result.extend(nodes) + if not result or len(result) != len(set(result)): + raise SystemExit('container-unit: empty or duplicate selection') + for filename in HELPER_SOURCES: + path = TESTS / filename + if path.resolve().parent != TESTS or not path.is_file(): + raise SystemExit('container-unit: missing or nonlocal helper source: ' + filename) + compile(ast.parse(path.read_text(encoding='utf-8'), filename=str(path)), str(path), 'exec') + return result + + +def main(argv=None): + if not (sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode) or sys.flags.optimize: + raise SystemExit('container-unit requires python -I -S -B without optimization') + parser = argparse.ArgumentParser(description=__doc__, allow_abbrev=False, exit_on_error=False) + modes = parser.add_mutually_exclusive_group() + modes.add_argument('--list', action='store_true', help='print exact leaf selectors without importing tests') + modes.add_argument('--check-selection', action='store_true', help='validate selected source AST and node IDs only') + parser.add_argument('-k', dest='keyword', metavar='EXPRESSION', help='narrow the allowlist using pytest -k') + try: + args, extra = parser.parse_known_args(argv) + except argparse.ArgumentError: + parser.error('invalid runner arguments') + if extra or (args.keyword is not None and (args.list or args.check_selection)): + parser.error('only --list, --check-selection, or -k EXPRESSION is supported') + if args.keyword is not None and (not args.keyword.strip() or len(args.keyword) > 512 + or any(char in args.keyword for char in '\x00\r\n')): + parser.error('-k requires a nonempty single-line expression of at most 512 characters') + + nodes = selected_nodes() + if args.list: + for node in nodes: + print('tests/' + node) + return 0 + if args.check_selection: + print(f'container-unit: AST OK; {len(SELECTION)} modules, {len(nodes)} test definitions, ' + 'no runner skips (parametrizations expand in pytest)') + return 0 + + if sys.platform != 'linux' or os.geteuid() != 10001 or TESTS != Path('/opt/truf/tests'): + raise SystemExit('container-unit execution requires Linux UID 10001 at /opt/truf/tests') + if not APP.is_dir() or os.access(APP, os.W_OK): + raise SystemExit('container-unit requires readable, read-only /opt/truf/app') + + # A fresh allowlist also removes mixed-case/future credentials, not just known + # TRUF_/SCANNER_/SCAN_/TRUFFLEHOG_/KEYCHECK_/PG*/DATABASE_URL/proxy names. + os.environ.clear() + os.environ.update({ + 'PATH': '/usr/local/bin:/usr/bin:/bin', + 'LANG': 'C.UTF-8', 'LC_ALL': 'C.UTF-8', + 'PYTHONDONTWRITEBYTECODE': '1', 'PYTHONNOUSERSITE': '1', 'PYTHONPATH': '', + 'PYTEST_DISABLE_PLUGIN_AUTOLOAD': '1', 'PYTEST_ADDOPTS': '', 'PYTEST_PLUGINS': '', + }) + os.umask(0o077) + with tempfile.TemporaryDirectory(prefix='container-unit-', dir='/tmp') as temporary: + private = Path(temporary).resolve() + details = private.stat() + if private.parent != Path('/tmp') or details.st_uid != 10001 or stat.S_IMODE(details.st_mode) != 0o700: + raise SystemExit('container-unit temporary directory is not private beneath /tmp') + for name in ('ALLTEMP', 'TEMP', 'TMP', 'TMPDIR', 'HOME'): + os.environ[name] = str(private) + for name, child in (('XDG_CONFIG_HOME', 'config'), ('XDG_CACHE_HOME', 'cache'), ('XDG_DATA_HOME', 'data')): + os.environ[name] = str(private / child) + tempfile.tempdir = None + with tempfile.NamedTemporaryFile(prefix='probe-') as probe: + actual = Path(probe.name).resolve() + if (Path(tempfile.gettempdir()).resolve() != private or actual.parent != private + or stat.S_IMODE(actual.stat().st_mode) != 0o600): + raise SystemExit('container-unit tempfile escaped its private directory') + + bootstrap = runpy.run_path(str(APP / 'child_bootstrap.py')) + bootstrap['_enable_dependency_paths']('supervisor') + sys.path.insert(0, str(APP)) + import pytest + import psycopg + import requests + import paths + + # Only scanner's import-time defaults are relocated. Restore paths before + # collection so portable-path and security checks exercise real policy. + blocked_import = AssertionError('container-unit forbids import-time runtime I/O') + with ExitStack() as imports: + imports.enter_context(mock.patch.multiple(paths, APP_DIR=str(private / 'app'), CANONICAL_ROOT=str(private))) + for target in ('builtins.open', 'os.open', 'os.mkdir', 'os.makedirs', 'sqlite3.connect', + 'subprocess.Popen.__init__', 'threading.Thread.start', 'socket.socket.__init__', + 'requests.Session.request', 'psycopg.connect'): + imports.enter_context(mock.patch(target, side_effect=blocked_import)) + import scanner + + # Filesystem/process mocks have ended: tests exercise real ACLs and locks. + os.chdir(ROOT) + sys.path.insert(1, str(TESTS)) + + def loopback_only(event, arguments): + if event in ('socket.getaddrinfo', 'socket.gethostbyname', 'socket.gethostbyaddr'): + host = arguments[0] + elif event in ('socket.connect', 'socket.bind', 'socket.sendto', 'socket.getnameinfo'): + address = arguments[0] if event == 'socket.getnameinfo' else arguments[-1] + host = address[0] if isinstance(address, tuple) and address else None + else: + return + if host not in ('127.0.0.1', '::1', 'localhost'): + raise AssertionError('container-unit permits loopback sockets only') + + sys.addaudithook(loopback_only) + + class Scope: + def pytest_collection_modifyitems(self, items): + for item in items: + node = item.nodeid.removeprefix('tests/').split('[', 1)[0] + if node not in nodes: + raise pytest.UsageError('container-unit collected a test outside its allowlist') + + options = [ + '-c', '/dev/null', '--rootdir', str(ROOT), '--noconftest', + '-p', 'no:cacheprovider', '-o', 'addopts=', '--basetemp', str(private / 'pytest'), + '--tb=short', '--show-capture=no', '-ra', + ] + if args.keyword is not None: + options.append('-k=' + args.keyword) + options.extend(str(TESTS / node) for node in nodes) + # Tests may replace these fences with their own scoped mocks, never real + # PostgreSQL or provider transports. Native local processes stay real. + with mock.patch.object(psycopg, 'connect', side_effect=AssertionError('external PostgreSQL is forbidden')), \ + mock.patch.object(requests.Session, 'request', side_effect=AssertionError('provider HTTP is forbidden')): + return int(pytest.main(options, plugins=[Scope()])) + + +if __name__ == '__main__': + try: + code = main() + except Exception as error: + print('container-unit bootstrap failed (' + type(error).__name__ + '); details withheld', file=sys.stderr) + code = 1 + raise SystemExit(code) diff --git a/tests/edge_e2e_backend.py b/tests/edge_e2e_backend.py new file mode 100644 index 0000000..b007956 --- /dev/null +++ b/tests/edge_e2e_backend.py @@ -0,0 +1,433 @@ +"""Private PostgreSQL and loopback backend for the standalone edge E2E.""" + +import sys + +sys.dont_write_bytecode = True + +import hashlib +import json +import os +from pathlib import Path +import runpy +import signal +import stat +import subprocess +import threading +import time +from types import SimpleNamespace + + +APP = Path('/opt/truf/app') +DATA = Path('/data') +CONTROL = DATA / 'control' +POSTGRES = DATA / 'postgres' +SOCKET = DATA / 'postgres-socket' +BUNDLES = DATA / 'bundles' +DB_PORT = 55433 +DB_URL = f'postgresql://truf@127.0.0.1:{DB_PORT}/edge_e2e' +BACKEND_PORT = 8766 +SUPERVISOR_ID = 'standalone-edge-e2e' +TARGET = 'https://gitlab.com/truf-edge-e2e/repository.git' +MAX_OUTPUT = 1024 * 1024 + + +def require(condition, label): + if not condition: + raise RuntimeError('standalone edge E2E backend: ' + label) + + +def private_directory(path, create=False): + path = Path(path) + if create: + path.mkdir(mode=0o700, parents=True, exist_ok=True) + os.chmod(path, 0o700) + details = path.stat(follow_symlinks=False) + require( + stat.S_ISDIR(details.st_mode) and details.st_uid == os.getuid() + and stat.S_IMODE(details.st_mode) == 0o700, + 'private directory', + ) + return path + + +def write_json(path, value): + payload = json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + b'\n' + require(len(payload) <= MAX_OUTPUT, 'control payload bound') + temporary = Path(str(path) + '.tmp') + try: + temporary.unlink() + except FileNotFoundError: + pass + descriptor = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(descriptor, 'wb') as handle: + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + os.replace(temporary, path) + + +def run(command, timeout=120): + completed = subprocess.run( + command, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, + stderr=subprocess.PIPE, timeout=timeout, check=False, + env={ + 'PATH': '/usr/lib/postgresql/16/bin:/usr/local/bin:/usr/bin:/bin', + 'HOME': str(DATA / 'home'), 'LANG': 'C.UTF-8', 'LC_ALL': 'C.UTF-8', + }, + ) + require( + len(completed.stdout) <= MAX_OUTPUT and len(completed.stderr) <= MAX_OUTPUT, + 'native command output bound', + ) + require(completed.returncode == 0, 'native command failed: ' + Path(command[0]).name) + return completed + + +def start_postgres(): + private_directory(DATA) + for path in (CONTROL, SOCKET, BUNDLES, DATA / 'home'): + private_directory(path, create=True) + require(not (POSTGRES / 'PG_VERSION').exists(), 'PostgreSQL volume is not fresh') + private_directory(POSTGRES, create=True) + run([ + '/usr/lib/postgresql/16/bin/initdb', '--pgdata', str(POSTGRES), + '--username=truf', '--auth=trust', '--encoding=UTF8', '--no-locale', + ]) + with open(POSTGRES / 'pg_hba.conf', 'a', encoding='ascii') as handle: + handle.write('\n# Disposable internal E2E network only.\nhost edge_e2e truf 0.0.0.0/0 trust\n') + run([ + '/usr/lib/postgresql/16/bin/pg_ctl', '-D', str(POSTGRES), '-w', 'start', + '-l', str(DATA / 'postgres.log'), + '-o', f'-k {SOCKET} -h 0.0.0.0 -p {DB_PORT}', + ]) + run([ + '/usr/lib/postgresql/16/bin/createdb', '-h', str(SOCKET), + '-p', str(DB_PORT), '-U', 'truf', 'edge_e2e', + ]) + + +def stop_postgres(): + if (POSTGRES / 'postmaster.pid').exists(): + try: + run([ + '/usr/lib/postgresql/16/bin/pg_ctl', '-D', str(POSTGRES), + '-w', '-m', 'fast', 'stop', + ], timeout=30) + except Exception: + pass + + +def source_args(platform): + return SimpleNamespace( + platform=platform, exact_git_planning_enabled=platform == 'gitlab', + workers=1, timeout=60, save_dir=str(DATA), detectors='', + exclude_detectors='', drop_detectors='', no_verification=True, + trufflehog_config=str(APP / 'trufflehog-custom-detectors.yaml'), token='', + scan_full_history=False, max_depth=25, git_baseline_depth=25, + max_commit_age_days=0, commit_lookup_pages=1, + skip_if_commit_lookup_fails=True, result_bundle_max_event_bytes=1 << 20, + result_bundle_max_items=20, result_bundle_max_total_bytes=32 << 20, + projection_backlog_max_items=20, projection_backlog_max_bytes=32 << 20, + projection_backlog_headroom_bytes=2 << 20, keycheck_queue_max_items=100, + keycheck_queue_max_bytes=8 << 20, pipeline_quarantine_max_items=20, + pipeline_quarantine_max_bytes=8 << 20, keycheck_candidates_per_event=50, + keycheck_candidate_bytes_per_event=1 << 20, target_retry_max_attempts=3, + target_retry_base_delay_sec=60, target_retry_max_delay_sec=600, + target_timeout_retry_delay_sec=300, max_active_scans=1, + admission_resolution_attempts=2, admission_resolution_seconds=1, + admission_resolution_retry_delay_sec=0.01, target_claim_order='oldest', + git_ref_resolution_attempts=1, git_ref_resolution_timeout_sec=1, + git_ref_resolution_max_bytes=1 << 20, + ) + + +def package_manifest(): + from lifecycle_authority import ( + GIT_MANIFEST_NAME, REMOTE_WORKER_CODE_AUTHORITY_FILES, + TRUFFLEHOG_MANIFEST_NAME, + ) + from result_bundle import FORMAT_VERSION + from scan_execution import PROTOCOL_VERSION + + policy_digest = hashlib.sha256( + (APP / 'trufflehog-custom-detectors.yaml').read_bytes(), + ).hexdigest() + files = { + name: {'path': f'app/{name}', 'sha256': '1' * 64} + for name in REMOTE_WORKER_CODE_AUTHORITY_FILES + } + return { + 'schema': 3, + 'protocol_version': PROTOCOL_VERSION, + 'bundle_format_version': FORMAT_VERSION, + 'platform_tag': 'linux-x86_64', + 'capabilities': [ + { + 'source': 'gitlab', 'platform': 'gitlab', + 'planning_kind': 'exact_git_v1', + }, + { + 'source': 'dockerhub', 'platform': 'docker', + 'planning_kind': 'docker_direct_v1', + }, + { + 'source': 'huggingface', 'platform': 'huggingface', + 'planning_kind': 'huggingface_space_v1', + }, + ], + 'app_root': 'app', + 'files': files, + 'executables': { + TRUFFLEHOG_MANIFEST_NAME: { + 'path': 'bin/trufflehog', 'sha256': '2' * 64, + }, + GIT_MANIFEST_NAME: { + 'path': 'runtime/git/bin/git', 'sha256': '3' * 64, + }, + }, + 'assets': { + 'detector_policy': { + 'path': 'app/trufflehog-custom-detectors.yaml', + 'sha256': policy_digest, + }, + }, + 'runtime_trees': { + 'git': {'path': 'runtime/git', 'sha256': '4' * 64, 'file_count': 1}, + }, + } + + +def worker_build(): + from worker_package import worker_package_build_compatibility + + return worker_package_build_compatibility(package_manifest()) + + +class Harness: + def __init__(self): + self.stop = threading.Event() + self.ingester_ready = threading.Event() + self.ingester_error = [] + self.server = None + self.tokens = { + name: str(os.environ.get(environment) or '') + for name, environment in { + 'good': 'TRUF_EDGE_E2E_GOOD_TOKEN', + 'wrong': 'TRUF_EDGE_E2E_WRONG_TOKEN', + 'revoked': 'TRUF_EDGE_E2E_REVOKED_TOKEN', + }.items() + } + self.edge_marker = str(os.environ.get('TRUF_ADMIN_EDGE_MARKER') or '') + require( + all(32 <= len(value) <= 512 for value in self.tokens.values()) + and len(set(self.tokens.values())) == 3, + 'test token configuration', + ) + require( + len(self.edge_marker) == 64 + and all(character in '0123456789abcdef' for character in self.edge_marker), + 'edge marker configuration', + ) + + def initialize_database(self): + from scanner_db import ScannerDB, migrate_runtime_safety_schema + + db = ScannerDB(db_url=DB_URL, initialize=False) + require(db.enabled and db.conn.is_postgres, 'PostgreSQL connection') + try: + migrate_runtime_safety_schema(db, initialize_base=True) + db.record_final_cutover({'fixture': 'standalone-edge-e2e-v1'}) + for name, token in self.tokens.items(): + result = db.provision_remote_worker_device( + 'edge-user-' + name, 'edge-device-' + name, + hashlib.sha256(token.encode('utf-8')).hexdigest(), 1, + ) + require(result['device_key'] == 'edge-device-' + name, 'device provisioning') + require( + db.set_remote_worker_device_revoked('edge-device-revoked', True), + 'revoked fixture', + ) + require( + db.enqueue_targets('gitlab', 'gitlab', 'standalone-edge-e2e', [TARGET]) == 1, + 'target enqueue', + ) + finally: + db.close() + + @staticmethod + def planner(args, db_url, source, claim, scan_kwargs, remote_credential=None): + from scanner_db import ScannerDB + + resolution = { + 'provider': 'gitlab', 'repo_url': TARGET, + 'repo_path': 'truf-edge-e2e/repository', 'branch': 'main', + 'ref': 'refs/heads/main', 'head_sha': 'a' * 40, + 'ref_source': 'provider_default', + } + db = ScannerDB(db_url=db_url, initialize=False) + try: + return db.bind_git_scan_plan( + claim['reservation_id'], claim['claim_lease_token'], resolution, + 25, remote_credential=remote_credential, + ) + finally: + db.close() + + def assignment_builder(self): + from worker_assignment import RemoteGitAssignmentBuilder + + return RemoteGitAssignmentBuilder( + DB_URL, str(BUNDLES), { + 'gitlab': source_args('gitlab'), + 'dockerhub': source_args('docker'), + 'huggingface': source_args('huggingface'), + }, + { + 'linux': { + 'package_manifest': package_manifest(), + 'sources': ['gitlab', 'dockerhub', 'huggingface'], + }, + }, + SUPERVISOR_ID, assignment_ttl_seconds=600, planner=self.planner, + ) + + def ingester_loop(self): + from result_ingester import ResultIngester + from scanner_db import ScannerDB + + db = ScannerDB(db_url=DB_URL, initialize=False) + ingester = None + try: + ingester = ResultIngester( + db, str(BUNDLES), SUPERVISOR_ID, lease_seconds=30, + ).start() + self.ingester_ready.set() + heartbeat = time.monotonic() + while not self.stop.is_set(): + progressed = ingester.process_one() + if time.monotonic() - heartbeat >= 5: + require(ingester.heartbeat(), 'ingester heartbeat') + heartbeat = time.monotonic() + if not progressed: + self.stop.wait(0.05) + except Exception as exc: + self.ingester_error.append(type(exc).__name__) + self.ingester_ready.set() + self.stop.set() + finally: + if ingester is not None: + ingester.stop('edge E2E stopping' if self.ingester_error else '') + db.close() + + def app(self): + from admin_api import AdminService + from worker_api import WorkerService, create_worker_app + + service = WorkerService( + DB_URL, str(BUNDLES), self.assignment_builder(), + max_bundle_bytes=32 << 20, claim_retry_after_seconds=1, + ) + admin = AdminService( + DB_URL, 'https://localhost', self.edge_marker, + db_factory=service.db_factory, + ) + app = create_worker_app( + service, reaper_interval_seconds=30, admin_service=admin, + ) + marker = self.edge_marker.encode('ascii') + + class BackendEvidence: + async def __call__(self, scope, receive, send): + if scope.get('type') == 'http': + path = str(scope.get('path') or '') + marker_headers = [ + value for name, value in scope.get('headers', ()) + if name.lower() == b'x-truf-admin-edge' + ] + operator_headers = [ + value for name, value in scope.get('headers', ()) + if name.lower() == b'x-truf-admin-operator' + ] + if path.startswith('/admin-internal'): + write_json(CONTROL / 'last-admin-request.json', { + 'schema': 1, 'path': path, + 'marker_authorized': marker_headers == [marker], + 'operators': [ + value.decode('ascii', errors='strict') + for value in operator_headers + ], + }) + elif path.startswith('/api/v1/worker/'): + write_json(CONTROL / 'last-worker-request.json', { + 'schema': 1, 'admin_headers_absent': ( + not marker_headers and not operator_headers + ), + }) + await app(scope, receive, send) + + return BackendEvidence() + + def run(self): + import uvicorn + + self.initialize_database() + ingester = threading.Thread( + target=self.ingester_loop, name='result-ingester', daemon=True, + ) + ingester.start() + require( + self.ingester_ready.wait(30) and not self.ingester_error, + 'ingester startup', + ) + write_json(CONTROL / 'ready.json', { + 'schema': 1, 'backend': 'private-network', 'postgres': 'fresh', + 'sources': ['gitlab', 'dockerhub', 'huggingface'], + }) + self.server = uvicorn.Server(uvicorn.Config( + self.app(), host='0.0.0.0', port=BACKEND_PORT, + access_log=False, log_level='warning', server_header=False, + proxy_headers=False, + )) + try: + self.server.run() + finally: + self.server = None + self.stop.set() + ingester.join(15) + require(not self.ingester_error, 'result ingester failure') + + +def main(): + require( + sys.platform == 'linux' and os.getuid() == os.getgid() == 10001, + 'Linux UID 10001 required', + ) + require( + sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode, + 'isolated Python required', + ) + os.umask(0o077) + sys.path.insert(0, str(APP)) + bootstrap = runpy.run_path(str(APP / 'child_bootstrap.py')) + bootstrap['_enable_dependency_paths']('supervisor') + start_postgres() + harness = Harness() + + def terminate(_signum, _frame): + if harness.server is not None: + harness.server.should_exit = True + harness.stop.set() + + signal.signal(signal.SIGTERM, terminate) + signal.signal(signal.SIGINT, terminate) + try: + harness.run() + finally: + harness.stop.set() + stop_postgres() + + +if __name__ == '__main__': + main() diff --git a/tests/edge_e2e_client.py b/tests/edge_e2e_client.py new file mode 100644 index 0000000..4fd8c49 --- /dev/null +++ b/tests/edge_e2e_client.py @@ -0,0 +1,428 @@ +"""Same-IP HTTPS assertions for the standalone edge E2E orchestrator.""" + +import sys + +sys.dont_write_bytecode = True + +import base64 +import hashlib +import http.client +import json +import os +from pathlib import Path +import re +import runpy +import socket +import ssl +import time +import uuid +from urllib.parse import urlencode + + +APP = Path('/opt/truf/app') +TESTS = Path('/opt/truf/tests') +CERTIFICATE = Path('/opt/truf/tests/fixtures/worker_tls_cert.pem') +MAX_BODY = 1024 * 1024 +FAILURE_STAGE = 'startup' +SECURITY_HEADERS = { + 'cache-control': 'no-store', + 'referrer-policy': 'same-origin', + 'content-security-policy': "default-src 'none'", + 'x-content-type-options': 'nosniff', + 'strict-transport-security': 'max-age=', +} + + +def require(condition, label): + if not condition: + raise RuntimeError('standalone edge E2E client: ' + label) + + +def environment(name, minimum=1): + value = str(os.environ.get(name) or '') + require(len(value) >= minimum, 'missing test environment') + return value + + +PREFIX = environment('TRUF_ADMIN_PREFIX', 64) +ADMIN_USER = environment('TRUF_ADMIN_USER') +ADMIN_PASSWORD = environment('TRUF_EDGE_E2E_ADMIN_PASSWORD', 16) +CONNECT_HOST = environment('TRUF_EDGE_E2E_CONNECT_HOST') +DB_URL = f'postgresql://truf@{CONNECT_HOST}:55433/edge_e2e' +TOKENS = { + name: environment(variable, 32) + for name, variable in { + 'good': 'TRUF_EDGE_E2E_GOOD_TOKEN', + 'wrong': 'TRUF_EDGE_E2E_WRONG_TOKEN', + 'revoked': 'TRUF_EDGE_E2E_REVOKED_TOKEN', + }.items() +} + + +def basic(user=ADMIN_USER, password=ADMIN_PASSWORD): + payload = base64.b64encode(f'{user}:{password}'.encode('utf-8')).decode('ascii') + return 'Basic ' + payload + + +def request(method, path, *, headers=None, body=None, expected=None, secure=True): + context = ssl.create_default_context(cafile=str(CERTIFICATE)) + connection = http.client.HTTPSConnection( + 'localhost', 443, context=context, timeout=10, + ) + connection._create_connection = lambda _address, timeout=None, source_address=None: ( + socket.create_connection((CONNECT_HOST, 443), timeout, source_address) + ) + try: + connection.request(method, path, body=body, headers=headers or {}) + response = connection.getresponse() + payload = response.read(MAX_BODY + 1) + require(len(payload) <= MAX_BODY, 'response body bound') + response_headers = {name.lower(): value for name, value in response.getheaders()} + finally: + connection.close() + if expected is not None: + require(response.status == expected, f'{method} {path} status {response.status}') + if secure: + for name, fragment in SECURITY_HEADERS.items(): + require(fragment in response_headers.get(name, ''), f'{name} on {path}') + return response.status, response_headers, payload + + +def json_request(path, token, request_id): + from edge_e2e_backend import worker_build + + payload = json.dumps( + {'request_id': request_id, 'build': worker_build()}, + ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + return request( + 'POST', path, + headers={ + 'Authorization': 'Bearer ' + token, + 'Content-Type': 'application/json', + 'X-Truf-Admin-Edge': 'spoofed-edge-marker', + 'X-Truf-Admin-Operator': 'spoofed-operator', + }, + body=payload, + secure=False, + ) + + +def db_rows(statement, values=()): + from scanner_db import ScannerDB + + db = ScannerDB(db_url=DB_URL, initialize=False) + require(db.enabled and db.conn.is_postgres, 'client PostgreSQL connection') + try: + rows = db.conn.execute(statement, values).fetchall() + db.conn.commit() + return [tuple(dict(row).values()) for row in rows] + finally: + db.close() + + +def assignment_state(reservation_id): + return { + 'reservation': db_rows( + '''SELECT id, queue_id, state, assignment_kind, remote_user_id, + remote_device_id, remote_resolution_kind, remote_receipt_id, + remote_expires_at, updated_at + FROM result_reservations WHERE id = ?''', + (reservation_id,), + ), + 'queue': db_rows( + '''SELECT q.id, q.status, q.attempts, q.lease_token, + q.current_result_reservation_id, q.claim_event_id, q.updated_at + FROM target_queue q JOIN result_reservations r ON r.queue_id = q.id + WHERE r.id = ?''', + (reservation_id,), + ), + 'capacity': db_rows( + '''SELECT bundle_items, bundle_bytes, projection_items, projection_bytes, + keycheck_items, keycheck_bytes, updated_at + FROM pipeline_capacity WHERE id = 1''', + ), + } + + +def reservation_id(): + rows = db_rows( + '''SELECT r.id FROM result_reservations r + JOIN remote_worker_devices d ON d.id = r.remote_device_id + WHERE d.device_key = 'edge-device-good' + ORDER BY r.id DESC LIMIT 1''', + ) + require(len(rows) == 1 and int(rows[0][0]) > 0, 'good assignment lookup') + return int(rows[0][0]) + + +def worker_status(reservation, token=TOKENS['good'], expected=200): + return request( + 'GET', f'/api/v1/worker/assignments/{reservation}', + headers={ + 'Authorization': 'Bearer ' + token, + 'X-Truf-Admin-Edge': 'spoofed-edge-marker', + 'X-Truf-Admin-Operator': 'spoofed-operator', + }, expected=expected, secure=False, + ) + + +def admin_request( + path='', *, authorization=True, expected=200, method='GET', body=None, + origin=None, secure=True, +): + headers = {'X-Truf-Admin-Operator': 'spoofed-operator'} + if authorization: + headers['Authorization'] = basic() + if origin is not None: + headers['Origin'] = origin + if body is not None: + headers['Content-Type'] = 'application/x-www-form-urlencoded' + return request( + method, f'/{PREFIX}/{path}', headers=headers, body=body, expected=expected, + secure=secure, + ) + + +def baseline(): + global FAILURE_STAGE + FAILURE_STAGE = 'public_routes' + checked = 0 + for path in ( + '/unknown', '/dashboard', '/metrics', '/admin-internal', + '/api/v1/private', '/api/v1/worker/private', + ): + request('GET', path, expected=404, secure=False) + checked += 1 + + FAILURE_STAGE = 'worker_challenge' + missing_payload = b'{"build":{},"request_id":"00000000000000000000000000000000"}' + status, headers, _ = request( + 'POST', '/api/v1/worker/claim', + headers={'Content-Type': 'application/json'}, body=missing_payload, expected=401, + secure=False, + ) + require(headers.get('www-authenticate') == 'Bearer', 'missing worker Bearer challenge') + checked += 1 + + FAILURE_STAGE = 'admin_challenge' + for path in ('', 'admin.css'): + status, headers, _ = admin_request( + path, authorization=False, expected=401, secure=False, + ) + require( + headers.get('www-authenticate') == 'Basic realm="truf-admin"', + 'missing admin Basic challenge', + ) + checked += 1 + + FAILURE_STAGE = 'worker_auth' + status, _, _ = json_request( + '/api/v1/worker/claim', TOKENS['revoked'], '1' * 32, + ) + require(status == 401, 'revoked worker token') + checked += 1 + + FAILURE_STAGE = 'worker_api' + status, _, payload = json_request( + '/api/v1/worker/claim', TOKENS['good'], '2' * 32, + ) + require(status == 201, 'valid worker claim') + value = json.loads(payload.decode('utf-8')) + reservation = int(value['assignment']['reservation']['reservation_id']) + require(reservation > 0, 'valid worker reservation') + checked += 1 + + before = assignment_state(reservation) + worker_status(reservation, TOKENS['wrong'], expected=404) + after = assignment_state(reservation) + require(after == before, 'wrong-device request changed authoritative assignment state') + checked += 1 + worker_status(reservation) + checked += 1 + + FAILURE_STAGE = 'admin_page_render' + _, _, page = admin_request() + checked += 1 + require(b'Workers / Dispatch' in page, 'protected admin page') + match = re.search(rb'name="csrf_token" value="([A-Za-z0-9_-]{32,128})"', page) + require(match is not None, 'admin CSRF token') + csrf = match.group(1).decode('ascii') + FAILURE_STAGE = 'admin_asset' + _, _, css = admin_request('admin.css') + require(b'color-scheme' in css, 'protected admin asset') + checked += 1 + + users_before = db_rows('SELECT user_key, active_assignment_cap, disabled_at FROM remote_worker_users ORDER BY id') + FAILURE_STAGE = 'admin_cross_site' + cross_site_operation_id = str(uuid.uuid4()) + cross_site = urlencode({ + 'csrf_token': csrf, 'user_key': 'cross-site-user', + 'active_assignment_cap': '2', 'operation_id': cross_site_operation_id, + }).encode('ascii') + admin_request( + 'users/create', method='POST', body=cross_site, + origin='https://cross-site.invalid', expected=403, + ) + require( + db_rows('SELECT user_key, active_assignment_cap, disabled_at FROM remote_worker_users ORDER BY id') + == users_before, + 'cross-site mutation changed authoritative state', + ) + checked += 1 + + FAILURE_STAGE = 'admin_same_site' + same_site_operation_id = str(uuid.uuid4()) + same_site = urlencode({ + 'csrf_token': csrf, 'user_key': 'same-site-user', + 'active_assignment_cap': '2', 'operation_id': same_site_operation_id, + }).encode('ascii') + _, mutation_headers, _ = admin_request( + 'users/create', method='POST', body=same_site, + origin='https://localhost', expected=303, + ) + require(mutation_headers.get('location') == '../', 'same-origin mutation redirect') + require( + db_rows("SELECT user_key, active_assignment_cap FROM remote_worker_users WHERE user_key = 'same-site-user'") + == [('same-site-user', 2)], + 'same-origin mutation persistence', + ) + FAILURE_STAGE = 'admin_operation_actor' + require( + db_rows( + '''SELECT actor, action, target_kind, target_ref, status + FROM runtime_operations WHERE operation_id = ?''', + (same_site_operation_id,), + ) == [(ADMIN_USER, 'workers.user.create', 'worker-admin', 'same-site-user', 'succeeded')], + 'same-origin durable operation actor attribution', + ) + FAILURE_STAGE = 'admin_audit_actor' + require( + db_rows( + '''SELECT actor, action, target_kind, target_ref, result + FROM runtime_audit_events WHERE operation_id = ? ORDER BY id''', + (same_site_operation_id,), + ) == [ + (ADMIN_USER, 'workers.user.create', 'worker-admin', 'same-site-user', 'accepted'), + (ADMIN_USER, 'workers.user.create', 'worker-admin', 'same-site-user', 'succeeded'), + ], + 'same-origin durable audit actor attribution', + ) + checked += 1 + + FAILURE_STAGE = 'admin_route_crawl' + admin_routes = ( + 'overview', 'search', 'supervisor', 'logs', 'config', 'secrets', + 'files', 'operations', 'audit', + f'operations/{same_site_operation_id}', + ) + for route in admin_routes: + route_label = ( + 'operation_detail' if route.startswith('operations/') else route + ) + FAILURE_STAGE = 'admin_route_' + route_label + route_status, _, route_page = admin_request(route, expected=None) + FAILURE_STAGE = f'admin_route_{route_label}_{route_status}' + require(route_status in (200, 503), 'protected admin route status') + if route_status == 200: + FAILURE_STAGE = 'admin_route_' + route_label + '_body' + require(b'' in route_page.lower(), 'protected admin route body') + checked += 1 + + FAILURE_STAGE = 'admin_worker_status' + worker_status(reservation) + checked += 1 + return { + 'mode': 'baseline', 'checked_responses': checked, + 'reservation_id': reservation, 'operation_id': same_site_operation_id, + } + + +def bad_auth(): + attempts = ( + ('truf-admin', 'definitely-wrong-one', '198.51.100.17'), + ('not-the-admin', 'definitely-wrong-two', '2001:db8::17'), + ) + for user, password, spoofed in attempts: + request( + 'GET', f'/{PREFIX}/', + headers={ + 'Authorization': basic(user, password), + 'X-Forwarded-For': spoofed, + 'Forwarded': f'for="[{spoofed}]";proto=http;host=spoofed.invalid', + }, + expected=401, + ) + return {'mode': 'bad-auth', 'attempts': len(attempts)} + + +def assert_ban(): + global FAILURE_STAGE + FAILURE_STAGE = 'ban_lookup' + reservation = reservation_id() + FAILURE_STAGE = 'ban_admin' + admin_request(expected=403) + FAILURE_STAGE = 'ban_worker' + worker_status(reservation) + return {'mode': 'assert-ban', 'reservation_id': reservation} + + +def assert_unban(): + global FAILURE_STAGE + FAILURE_STAGE = 'unban_lookup' + reservation = reservation_id() + FAILURE_STAGE = 'unban_admin' + _, _, page = admin_request() + require(b'Workers / Dispatch' in page, 'admin restoration') + FAILURE_STAGE = 'unban_worker' + worker_status(reservation) + return {'mode': 'assert-unban', 'reservation_id': reservation} + + +def probe(): + status, headers, _ = request('GET', '/edge-e2e-probe', secure=False) + missing = [ + name for name, fragment in SECURITY_HEADERS.items() + if fragment not in headers.get(name, '') + ] + return { + 'mode': 'probe', + 'ready': status == 404 and not missing, + 'status': status, + 'missing': missing, + } + + +def main(): + require( + sys.platform == 'linux' and sys.flags.isolated and sys.flags.no_site + and sys.flags.dont_write_bytecode, + 'isolated Linux Python required', + ) + sys.path.insert(0, str(APP)) + sys.path.insert(0, str(TESTS)) + bootstrap = runpy.run_path(str(APP / 'child_bootstrap.py')) + bootstrap['_enable_dependency_paths']('supervisor') + modes = { + 'baseline': baseline, 'bad-auth': bad_auth, 'assert-ban': assert_ban, + 'assert-unban': assert_unban, 'probe': probe, + } + require(len(sys.argv) == 2 and sys.argv[1] in modes, 'mode') + try: + result = modes[sys.argv[1]]() + except Exception: + stage = FAILURE_STAGE + if not re.fullmatch(r'[a-z][a-z0-9_]{0,39}', stage): + stage = 'runtime' + print(json.dumps({ + 'mode': sys.argv[1], 'failure_stage': stage, + 'tls': 'validated-localhost-certificate', + }, ensure_ascii=True, sort_keys=True, separators=(',', ':'))) + return 1 + result['tls'] = 'validated-localhost-certificate' + print(json.dumps(result, ensure_ascii=True, sort_keys=True, separators=(',', ':'))) + return 0 + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/tests/fixtures/worker_tls_cert.pem b/tests/fixtures/worker_tls_cert.pem new file mode 100644 index 0000000..8e99c98 --- /dev/null +++ b/tests/fixtures/worker_tls_cert.pem @@ -0,0 +1,20 @@ +-----BEGIN CERTIFICATE----- +MIIDRjCCAi6gAwIBAgIUcRnJpULcek4E2YM83/qYtCdWE+IwDQYJKoZIhvcNAQEL +BQAwFDESMBAGA1UEAwwJbG9jYWxob3N0MB4XDTI2MDkxODEyNDcxNFoXDTM2MDkx +NTEyNDcxNFowFDESMBAGA1UEAwwJbG9jYWxob3N0MIIBIjANBgkqhkiG9w0BAQEF +AAOCAQ8AMIIBCgKCAQEA7rm79Bmbz51sKhYYRevsAmjnSE8ilpQgFBD82VuDd7TB +7kVL0Z+3xzMJO78XIlM0EPGLgXsc9JujNUmcLmFronMUVyeHjr1mtg+kFodyusDO +QXjbh74ZNafGJlfxWp6s7abJVfTXE/1or6OAXHqQgBgrkkEXmwVx34Dn9AuAO4m1 +jS9cbLA4jETLE/62un23sqIaP1QWZz6cSdxCCdV+NaF98RoKVbhMRD9EY0IkWpW9 +bAG0a7YgEanJ+41yZSmdthPrrAir96AWqHhJ0w4zbxydUBySj6btl+wTAG7b67ht +becWRzWsjg1kGSFUb2nyon9gUQDKz7xMiKbvQ1EfdwIDAQABo4GPMIGMMB0GA1Ud +DgQWBBSp05qZeXtLzt1VzZIDa3kZPaWifjAfBgNVHSMEGDAWgBSp05qZeXtLzt1V +zZIDa3kZPaWifjAUBgNVHREEDTALgglsb2NhbGhvc3QwDwYDVR0TAQH/BAUwAwEB +/zAOBgNVHQ8BAf8EBAMCAqQwEwYDVR0lBAwwCgYIKwYBBQUHAwEwDQYJKoZIhvcN +AQELBQADggEBAEK30uxZSWC+oQypgFGtAOQh5WUJt2tdVkdfxNTUA9HhNssWAXYT +ipS9ZyuTeqHN93vxs1zeJbmTeCvzpv2oc+E2/B6+LhQopXQdb62NUyCWt3JqLPUk +antYpitkvcfSGP+L4k3K3VMbW3nFn/qxirMlKTBjPLAkhu7RxrqdKREydCLWDp2c +flYrmnoj4C4OqKOuEHCxxLOKwPhbS3w5QvwZyEwkwOsRNfyruMPzHk3u8erhWRYs +Vlrlqb1ZHMX2mfJu2HLbYUEckXpXbZ4NVEXdbEGxSC0RV2F0urqqCmzaOc0R979/ +Fw4NsLOW/LKAk+pFKT2a4v274bE3oDwHEKI= +-----END CERTIFICATE----- diff --git a/tests/fixtures/worker_tls_key.pem b/tests/fixtures/worker_tls_key.pem new file mode 100644 index 0000000..6e0ab99 --- /dev/null +++ b/tests/fixtures/worker_tls_key.pem @@ -0,0 +1,29 @@ +# Public test fixture only; never deploy this key. +-----BEGIN PRIVATE KEY----- +MIIEvgIBADANBgkqhkiG9w0BAQEFAASCBKgwggSkAgEAAoIBAQDuubv0GZvPnWwq +FhhF6+wCaOdITyKWlCAUEPzZW4N3tMHuRUvRn7fHMwk7vxciUzQQ8YuBexz0m6M1 +SZwuYWuicxRXJ4eOvWa2D6QWh3K6wM5BeNuHvhk1p8YmV/FanqztpslV9NcT/Wiv +o4BcepCAGCuSQRebBXHfgOf0C4A7ibWNL1xssDiMRMsT/ra6fbeyoho/VBZnPpxJ +3EIJ1X41oX3xGgpVuExEP0RjQiRalb1sAbRrtiARqcn7jXJlKZ22E+usCKv3oBao +eEnTDjNvHJ1QHJKPpu2X7BMAbtvruG1t5xZHNayODWQZIVRvafKif2BRAMrPvEyI +pu9DUR93AgMBAAECggEAGVRYY/208YzI0EJd5aAIVPfKf8/zDI+/mOwggp4vedZn +s/S4Ubqk+5wnq6Y1Odgi2yALPFF9cLAe22WUY7tvJI0Z/bwHqcE2PQpgel95cI/1 +PS8p+TeAt0Jg8j/nL5qhyz7PeAZYLRp4hBemDqnr28Y0wU+UxfGy872wCXiQQkDA +K/bZTzzAsFVHpM1lfB/o2X7i1B8S88eYOwlxL7zVQOHLKw3tsPnp1vIWS6YgEYfY +TrAvaIe8+FMUPDEhf/gVcXmvDxGtJ0tdHVu70crIJgFWQncota7dV2iue0Jt/zmI +GUrZ28ljL5Xt1u8OTAGXVFDfPjmXFNgHX4di/SgqWQKBgQD8AQ4utGstAsWTawuq +2mojjYFi3SUEqG0yrJxygQ96oMieNSZLbdsKVyVN06Zt3NimdoKAG+D9MjVZxmfp +HpuiOyo3fnQHMyrTak00NfE7zTT+DVL/V68Bt8IWIBxt4BTdMammiz7dowzWYYDb +s+6YTXzuqnT3/ROFrrk6rJloCwKBgQDygsceZsurGldMZWeLGfW4+2bGnMjqvTC2 +12RyHY4OK43hNl7/XDbvAu97XN+yO7hVj9UNulItwWyWns4GmlMGs08Y9l4jBQ/S +bVJfMzX0Zop7fwOHIwYn2YMMLmXYtijJD02ICISobDHFodkx+K9ZFXJtXTqZHh0q +dvdR1KKNxQKBgQCc/CNToPzrC0D9dr/L7UgVYb9qUQ0Qe8Oav8Ct7Awyfhq7w6xZ +bNP4+xS4CNMyuVMVT9o36CYeVLq7dEejB3g4ddb0vweUvKE/FoeFsNzYPht27+H2 +Qy84SLrVgad0IxWcPaXLpA7DjyEeI5tcQhiuNAdRvkojejpBGvk0vfTKxQKBgBnv +lJ4Svlt5QLbh7XX5+8ah1HcPU4mPXENhu9Nch9HKJK1eZECJOzLKrJQT9bSZIHi+ +HjoOoDVWh2eAamZYYOLJkH8J8j1qkCugF3wo/O87fDoC9nygaUsfvx0xZSENMkV2 +hoMy7gUZNSV+zrzCbPZpDcjWfKrdhp8BBChTRmNFAoGBAPR2QWB/nemeuEtRV/yp +vMRvu8De2bV3n8uW6SwLv8U38yoBQfAjdnBbp068pmEL/KVmTmim6s2Hm89d20ie +xJMy5CoAHaEuxab7lDs8G67djV8V6tR0I+g796y62MWccUHhc4rNMhSrZvWKPF1M +N/4p2pyOhln0ggY0FseDh0Xt +-----END PRIVATE KEY----- diff --git a/tests/owned_process_helper.py b/tests/owned_process_helper.py new file mode 100644 index 0000000..f1dfab4 --- /dev/null +++ b/tests/owned_process_helper.py @@ -0,0 +1,159 @@ +import os +import subprocess +import sys +import time + +sys.dont_write_bytecode = True + + +CREATE_NO_WINDOW = 0x08000000 + + +def write_value(path, value): + with open(path, "w", encoding="ascii") as output: + output.write(str(value)) + output.flush() + + +def wait_for_file(path, timeout=5): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if os.path.exists(path): + return True + time.sleep(0.02) + return False + + +def main(): + mode = sys.argv[1] + if mode == "forked-proxy": + from pathlib import Path + + sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "app")) + from owned_process import OwnedProcess + + marker_dir, release_path = Path(sys.argv[2]), sys.argv[3] + child = OwnedProcess( + [sys.executable, "-I", "-S", "-B", os.path.abspath(__file__), + "linger", str(marker_dir / "payload-0.pid")], + stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + ) + write_value(marker_dir / "host-1.pid", child.host_pid) + write_value(marker_dir / "payload-2.pid", os.getpid()) + if not wait_for_file(marker_dir / "payload-0.pid"): + return 90 + child._control_lock.acquire() + fork_pid = os.fork() + if fork_pid == 0: + del child + os._exit(0) + child._control_lock.release() + write_value(marker_dir / "payload-1.pid", fork_pid) + _, status = os.waitpid(fork_pid, 0) + if status != 0 or child.poll() is not None: + return 92 + write_value(marker_dir / "fork-detached", 1) + while not os.path.exists(release_path): + time.sleep(0.02) + os._exit(23) + + if mode == "adopted-observer": + from pathlib import Path + + marker_dir, release_path = Path(sys.argv[2]), sys.argv[3] + owner = subprocess.Popen( + [sys.executable, "-I", "-S", "-B", os.path.abspath(__file__), + "nested-owned", "1", str(marker_dir), str(marker_dir / "orphan-release")], + stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + ) + write_value(marker_dir / "payload-2.pid", os.getpid()) + owner.wait() + while not os.path.exists(release_path): + time.sleep(0.02) + return 23 + + if mode == "nested-owned": + from pathlib import Path + + sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "app")) + from owned_process import OwnedProcess + + depth, marker_dir, release_path = int(sys.argv[2]), Path(sys.argv[3]), sys.argv[4] + if depth: + child = OwnedProcess( + [sys.executable, "-I", "-S", "-B", os.path.abspath(__file__), + mode, str(depth - 1), str(marker_dir), ""], + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + ) + write_value(marker_dir / f"host-{depth}.pid", child.host_pid) + write_value(marker_dir / f"payload-{depth}.pid", os.getpid()) + while not release_path or not os.path.exists(release_path): + time.sleep(0.02) + os._exit(23) + + if mode == "linger": + write_value(sys.argv[2], os.getpid()) + while True: + time.sleep(1) + + if mode == "escaped-session": + from pathlib import Path + + marker_dir = Path(sys.argv[2]) + child = subprocess.Popen( + [sys.executable, "-I", "-S", "-B", os.path.abspath(__file__), + "linger", str(marker_dir / "escaped.pid")], + stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, start_new_session=True, + ) + write_value(marker_dir / "payload-0.pid", os.getpid()) + if not wait_for_file(marker_dir / "escaped.pid"): + child.kill() + return 90 + while True: + time.sleep(1) + + if mode in ("tree", "spawn-and-exit"): + ready_path = sys.argv[2] + child = subprocess.Popen( + [sys.executable, os.path.abspath(__file__), "linger", ready_path], + stdin=subprocess.DEVNULL, + close_fds=True, + creationflags=CREATE_NO_WINDOW if os.name == "nt" else 0, + ) + if not wait_for_file(ready_path): + child.terminate() + return 90 + if mode == "spawn-and-exit": + return int(sys.argv[3]) + while True: + time.sleep(1) + + if mode == "marker": + write_value(sys.argv[2], os.getpid()) + return 0 + + if mode == "stdio": + sys.stdout.write("owned stdout\n") + sys.stdout.flush() + sys.stderr.write("owned stderr\n") + sys.stderr.flush() + if len(sys.argv) > 3 and sys.argv[3] == "wait-stdin": + sys.stdin.buffer.read() + return int(sys.argv[2]) + + if mode == "partial": + sys.stdout.write("partial stdout\n") + sys.stdout.flush() + sys.stderr.write("partial stderr\n") + sys.stderr.flush() + while True: + time.sleep(1) + + return 91 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/packaged_worker_e2e_server.py b/tests/packaged_worker_e2e_server.py new file mode 100644 index 0000000..709fdb6 --- /dev/null +++ b/tests/packaged_worker_e2e_server.py @@ -0,0 +1,626 @@ +"""Isolated HTTPS/PostgreSQL harness for real packaged worker clients.""" + +import sys + +sys.dont_write_bytecode = True + +import hashlib +import json +import os +from pathlib import Path +import runpy +import signal +import stat +import subprocess +import threading +import time +from types import SimpleNamespace +from urllib.parse import urlsplit + + +APP = Path('/opt/truf/app') +DATA = Path('/data') +CONTROL = DATA / 'control' +POSTGRES = DATA / 'postgres' +SOCKET = DATA / 'postgres-socket' +BUNDLES = DATA / 'bundles' +FIXTURE = Path('/fixture') +PORT = 8443 +DB_PORT = 55432 +DB_URL = f'postgresql://truf@127.0.0.1:{DB_PORT}/packaged_worker_e2e' +MAX_CONTROL_BYTES = 1024 * 1024 +COMPLETION_REQUIREMENTS = frozenset(( + 'normalized row counts', 'queue completion', 'remote reservation completion', + 'bundle completion', 'scan completion', 'native findings', 'candidate routing', + 'exact fixture commits', 'server bundle spool cleanup', 'bundle capacity release', + 'source coverage', 'direct assignment planning', 'exact fixture targets', +)) + + +def fail(message): + raise RuntimeError('packaged worker E2E: ' + message) + + +def require(condition, message): + if not condition: + fail(message) + + +def private_directory(path, create=False): + path = Path(path) + if create: + path.mkdir(mode=0o700, parents=True, exist_ok=True) + os.chmod(path, 0o700) + details = path.stat(follow_symlinks=False) + require(stat.S_ISDIR(details.st_mode) and details.st_uid == os.getuid(), 'private directory') + require(stat.S_IMODE(details.st_mode) == 0o700, 'private directory mode') + return path + + +def write_json(path, value): + payload = json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ).encode('ascii') + require(len(payload) <= MAX_CONTROL_BYTES, 'control payload bound') + temporary = Path(str(path) + '.tmp') + try: + temporary.unlink() + except FileNotFoundError: + pass + descriptor = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(descriptor, 'wb') as handle: + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + os.replace(temporary, path) + + +def read_json(path): + details = Path(path).stat(follow_symlinks=False) + require(stat.S_ISREG(details.st_mode) and details.st_size <= MAX_CONTROL_BYTES, 'fixture JSON bound') + with open(path, 'rb') as handle: + payload = handle.read(MAX_CONTROL_BYTES + 1) + require(len(payload) <= MAX_CONTROL_BYTES, 'fixture JSON bound') + value = json.loads(payload.decode('ascii')) + require(isinstance(value, dict), 'fixture JSON shape') + return value + + +def run(command, timeout=60): + completed = subprocess.run( + command, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, + stderr=subprocess.PIPE, timeout=timeout, check=False, + env={ + 'PATH': '/usr/lib/postgresql/16/bin:/usr/local/bin:/usr/bin:/bin', + 'HOME': str(DATA / 'home'), 'LANG': 'C.UTF-8', 'LC_ALL': 'C.UTF-8', + }, + ) + if completed.returncode: + fail('native command failed: ' + Path(command[0]).name) + require(len(completed.stdout) <= MAX_CONTROL_BYTES and len(completed.stderr) <= MAX_CONTROL_BYTES, + 'native command output bound') + return completed + + +def start_postgres(): + private_directory(DATA) + for path in (CONTROL, SOCKET, BUNDLES, DATA / 'home'): + private_directory(path, create=True) + first = not (POSTGRES / 'PG_VERSION').exists() + if first: + private_directory(POSTGRES, create=True) + run([ + '/usr/lib/postgresql/16/bin/initdb', '--pgdata', str(POSTGRES), + '--username=truf', '--auth=trust', '--encoding=UTF8', '--no-locale', + ]) + run([ + '/usr/lib/postgresql/16/bin/pg_ctl', '-D', str(POSTGRES), '-w', 'start', + '-l', str(DATA / 'postgres.log'), + '-o', f'-k {SOCKET} -h 127.0.0.1 -p {DB_PORT}', + ]) + if first: + run([ + '/usr/lib/postgresql/16/bin/createdb', '-h', str(SOCKET), + '-p', str(DB_PORT), '-U', 'truf', 'packaged_worker_e2e', + ]) + return first + + +def stop_postgres(): + if (POSTGRES / 'postmaster.pid').exists(): + try: + run([ + '/usr/lib/postgresql/16/bin/pg_ctl', '-D', str(POSTGRES), + '-w', '-m', 'fast', 'stop', + ], timeout=30) + except Exception: + pass + + +def source_args(platform): + return SimpleNamespace( + platform=platform, exact_git_planning_enabled=platform == 'gitlab', + workers=2, timeout=240, save_dir=str(DATA), detectors='OpenAI', + exclude_detectors='', drop_detectors='', no_verification=True, + trufflehog_config=str(APP / 'trufflehog-custom-detectors.yaml'), + token='', scan_full_history=False, max_depth=25, git_baseline_depth=25, + max_commit_age_days=0, commit_lookup_pages=1, + skip_if_commit_lookup_fails=True, result_bundle_max_event_bytes=1 << 20, + result_bundle_max_items=20, result_bundle_max_total_bytes=32 << 20, + projection_backlog_max_items=20, projection_backlog_max_bytes=32 << 20, + projection_backlog_headroom_bytes=2 << 20, keycheck_queue_max_items=200, + keycheck_queue_max_bytes=8 << 20, pipeline_quarantine_max_items=20, + pipeline_quarantine_max_bytes=8 << 20, keycheck_candidates_per_event=50, + keycheck_candidate_bytes_per_event=1 << 20, target_retry_max_attempts=3, + target_retry_base_delay_sec=60, target_retry_max_delay_sec=600, + target_timeout_retry_delay_sec=300, max_active_scans=2, + admission_resolution_attempts=2, admission_resolution_seconds=1, + admission_resolution_retry_delay_sec=0.01, target_claim_order='oldest', + git_ref_resolution_attempts=1, git_ref_resolution_timeout_sec=1, + git_ref_resolution_max_bytes=1 << 20, + ) + + +class Harness: + def __init__(self, fixture): + self.fixture = fixture + self.targets = tuple(fixture.get('targets') or ()) + self.repositories = dict(fixture.get('repositories') or {}) + require(len(self.targets) == 2 and set(self.targets) == set(self.repositories), + 'exactly two fixture targets required') + self.direct_targets = dict(fixture.get('direct_targets') or {}) + require( + set(self.direct_targets) == {'dockerhub', 'huggingface'} + and all(isinstance(value, str) and value for value in self.direct_targets.values()), + 'direct fixture targets required', + ) + self.token = str(os.environ.get('TRUF_WORKER_E2E_TOKEN') or '') + self.phase = str(os.environ.get('TRUF_WORKER_E2E_PHASE') or '') + require(16 <= len(self.token) <= 512 and self.phase in ('windows', 'linux'), + 'phase credentials') + self.stop = threading.Event() + self.ingester_ready = threading.Event() + self.ingester_error = [] + self.claims = 0 + self.status_checks = 0 + self.git_reservation_ids = frozenset() + self.claim_lock = threading.Lock() + self.server = None + + def planner(self, args, db_url, source, claim, scan_kwargs, remote_credential=None): + from scanner_db import ScannerDB + + target = str(claim['target']) + repository = dict(self.repositories.get(target) or {}) + parsed = urlsplit(target) + repo_path = parsed.path.removeprefix('/').removesuffix('.git') + resolution = { + 'provider': 'gitlab', 'repo_url': target, 'repo_path': repo_path, + 'branch': 'main', 'ref': 'refs/heads/main', + 'head_sha': str(repository.get('head_sha') or ''), + 'ref_source': 'provider_default', + } + db = ScannerDB(db_url=db_url, initialize=False) + try: + return db.bind_git_scan_plan( + claim['reservation_id'], claim['claim_lease_token'], resolution, + 25, remote_credential=remote_credential, + ) + finally: + db.close() + + def initialize_database(self, first): + from scanner_db import ScannerDB, migrate_runtime_safety_schema + + db = ScannerDB(db_url=DB_URL, initialize=False) + require(db.enabled and db.conn.is_postgres, 'PostgreSQL connection') + try: + if first: + migrate_runtime_safety_schema(db, initialize_base=True) + db.record_final_cutover({'fixture': 'packaged-worker-e2e-v1'}) + provisioned = db.provision_remote_worker_device( + 'packaged-worker-e2e-' + self.phase, + 'packaged-worker-e2e-' + self.phase, + hashlib.sha256(self.token.encode('utf-8')).hexdigest(), 2, + ) + require(provisioned['active_assignment_cap'] == 2, 'device capacity') + require(db.enqueue_targets( + 'gitlab', 'gitlab', 'packaged-worker-e2e', self.targets, + ) == 2, 'target enqueue') + write_json(CONTROL / 'prepared.json', { + 'schema': 1, 'phase': self.phase, 'target_count': 2, + }) + else: + db.require_runtime_safety_schema() + db.require_final_cutover() + marker = read_json(CONTROL / 'prepared.json') + require(marker == {'schema': 1, 'phase': self.phase, 'target_count': 2}, + 'prepared marker') + finally: + db.close() + + def assignment_builder(self): + from worker_assignment import RemoteGitAssignmentBuilder + + return RemoteGitAssignmentBuilder( + DB_URL, str(BUNDLES), { + 'gitlab': source_args('gitlab'), + 'dockerhub': source_args('docker'), + 'huggingface': source_args('huggingface'), + }, + { + 'windows': { + 'package_manifest': FIXTURE / 'windows-manifest.json', + 'sources': ['gitlab', 'dockerhub', 'huggingface'], + }, + 'linux': { + 'package_manifest': FIXTURE / 'linux-manifest.json', + 'sources': ['gitlab', 'dockerhub', 'huggingface'], + }, + }, + 'packaged-worker-e2e-' + self.phase, + assignment_ttl_seconds=600, planner=self.planner, + ) + + def enqueue_direct_targets(self): + from scanner_db import ScannerDB + + db = ScannerDB(db_url=DB_URL, initialize=False) + try: + rows = db.conn.execute( + 'SELECT id FROM result_reservations ORDER BY id' + ).fetchall() + self.git_reservation_ids = frozenset(int(row['id']) for row in rows) + require(len(self.git_reservation_ids) == 2, 'initial Git reservations') + require(db.enqueue_targets( + 'dockerhub', 'docker', 'packaged-worker-e2e', + [self.direct_targets['dockerhub']], + ) == 1, 'Docker target enqueue') + require(db.enqueue_targets( + 'huggingface', 'huggingface', 'packaged-worker-e2e', + [self.direct_targets['huggingface']], + ) == 1, 'HuggingFace target enqueue') + finally: + db.close() + + def ingester_loop(self): + from result_ingester import ResultIngester + from scanner_db import ScannerDB + + db = ScannerDB(db_url=DB_URL, initialize=False) + ingester = None + try: + ingester = ResultIngester( + db, str(BUNDLES), 'packaged-worker-e2e-' + self.phase, + lease_seconds=30, + ).start() + self.ingester_ready.set() + heartbeat = time.monotonic() + while not self.stop.is_set(): + progressed = ingester.process_one() + if time.monotonic() - heartbeat >= 5: + require(ingester.heartbeat(), 'ingester heartbeat') + heartbeat = time.monotonic() + if not progressed: + self.stop.wait(0.05) + except Exception as exc: + self.ingester_error.append(type(exc).__name__) + self.ingester_ready.set() + self.stop.set() + finally: + if ingester is not None: + ingester.stop('harness stopping' if self.ingester_error else '') + db.close() + + def claim_complete(self): + with self.claim_lock: + self.claims += 1 + + def status_complete(self): + with self.claim_lock: + self.status_checks += 1 + if self.status_checks != 2: + return + require(self.claims == 2, 'status fencing before both claims') + write_json(CONTROL / 'outage.json', { + 'schema': 1, 'claim_count': 2, 'phase': self.phase, + }) + if self.server is not None: + self.server.should_exit = True + + def app(self, stop_after_claims): + from worker_api import WorkerService, create_worker_app + + service = WorkerService( + DB_URL, str(BUNDLES), self.assignment_builder(), + max_bundle_bytes=32 << 20, claim_retry_after_seconds=1, + ) + app = create_worker_app(service, reaper_interval_seconds=10) + if not stop_after_claims: + harness = self + + class DirectStatusFence: + async def __call__(self, scope, receive, send): + path = str(scope.get('path') or '') + parts = path.split('/') + reservation_id = 0 + if ( + scope.get('type') == 'http' + and scope.get('method') == 'GET' + and len(parts) == 6 + and parts[1:5] == ['api', 'v1', 'worker', 'assignments'] + ): + try: + reservation_id = int(parts[5]) + except ValueError: + reservation_id = 0 + if ( + reservation_id > 0 + and reservation_id not in harness.git_reservation_ids + and not (CONTROL / 'direct-ready').is_file() + ): + body = b'{"code":"fixture_not_ready"}' + await send({ + 'type': 'http.response.start', 'status': 503, + 'headers': [ + (b'content-type', b'application/json'), + (b'content-length', str(len(body)).encode('ascii')), + ], + }) + await send({'type': 'http.response.body', 'body': body}) + return + await app(scope, receive, send) + + return DirectStatusFence() + harness = self + + class StopAfterClaims: + async def __call__(self, scope, receive, send): + status = None + + async def wrapped(message): + nonlocal status + if message['type'] == 'http.response.start': + status = int(message['status']) + await send(message) + if message['type'] != 'http.response.body' or message.get('more_body'): + return + path = str(scope.get('path') or '') + if path == '/api/v1/worker/claim' and status == 201: + harness.claim_complete() + elif ( + scope.get('method') == 'GET' and status == 200 + and path.startswith('/api/v1/worker/assignments/') + and path.count('/') == 5 + ): + harness.status_complete() + + await app(scope, receive, wrapped) + + return StopAfterClaims() + + def evidence(self): + from scanner_db import ScannerDB + + db = ScannerDB(db_url=DB_URL, initialize=False) + try: + def rows(statement, values=()): + return [dict(row) for row in db.conn.execute(statement, values).fetchall()] + + counts = { + table: int(rows(f'SELECT COUNT(*) AS count FROM {table}')[0]['count']) + for table in ( + 'target_queue', 'result_reservations', 'result_bundles', + 'target_scans', 'scan_result_compat', 'findings', + 'finding_compat_payloads', 'finding_uid_map', + 'keycheck_candidates', 'keycheck_credentials', 'errors', + 'pipeline_quarantine', + ) + } + expected = { + name: (4 if name in { + 'target_queue', 'result_reservations', 'result_bundles', + 'target_scans', 'scan_result_compat', + } else 2) + for name in counts + } + expected.update({'errors': 0, 'pipeline_quarantine': 0}) + require(counts == expected, 'normalized row counts') + queue = rows('SELECT * FROM target_queue ORDER BY target') + reservations = rows('SELECT * FROM result_reservations ORDER BY id') + bundles = rows('SELECT * FROM result_bundles ORDER BY reservation_id') + scans = rows('SELECT * FROM target_scans ORDER BY target') + findings = rows('SELECT * FROM findings ORDER BY target') + candidates = rows('SELECT * FROM keycheck_candidates ORDER BY target') + require( + all(row['status'] == 'done' and row['attempts'] == 1 + and row['lease_token'] is None + and row['current_result_reservation_id'] is None for row in queue), + 'queue completion', + ) + require( + {(row['source'], row['target']) for row in queue} == { + *(('gitlab', target) for target in self.targets), + ('dockerhub', self.direct_targets['dockerhub']), + ('huggingface', self.direct_targets['huggingface']), + }, + 'exact fixture targets', + ) + require( + all(row['state'] == 'acknowledged' + and row['assignment_kind'] == 'remote' + and row['remote_resolution_kind'] == 'bundle_accepted' + and row['bundle_credit_released'] == 1 for row in reservations), + 'remote reservation completion', + ) + reservation_by_id = {int(row['id']): row for row in reservations} + for row in bundles: + source = reservation_by_id[int(row['reservation_id'])]['source'] + expected_findings = 1 if source == 'gitlab' else 0 + require( + row['state'] == 'acknowledged' and row['actual_bytes'] > 0 + and row['finding_count'] == row['candidate_count'] == expected_findings + and row['error_count'] == 0, + 'bundle completion', + ) + scan_contract = { + 'gitlab': ('gitlab', 'found', 1), + 'dockerhub': ('docker', 'clean', 0), + 'huggingface': ('huggingface', 'clean', 0), + } + for row in scans: + scan_type, status, finding_count = scan_contract.get(row['source'], (None, None, None)) + require( + row['scan_type'] == scan_type and row['status'] == status + and row['findings_count'] == finding_count and row['error_count'] == 0 + and row['queue_completion_applied'] == 1, + 'scan completion', + ) + require(all(row['detector_name'] == 'OpenAI' and not row['verified'] + and row['file_path'] == 'synthetic.env' for row in findings), + 'native findings') + require(all(row['service'] == row['routed_service'] == 'openai' + and row['state'] == 'pending' for row in candidates), + 'candidate routing') + expected_heads = { + target: str(self.repositories[target]['head_sha']) for target in self.targets + } + require({row['target']: row['commit_hash'] for row in findings} == expected_heads, + 'exact fixture commits') + source_counts = { + source: sum(row['source'] == source for row in reservations) + for source in ('gitlab', 'dockerhub', 'huggingface') + } + require( + source_counts == {'gitlab': 2, 'dockerhub': 1, 'huggingface': 1}, + 'source coverage', + ) + planning_counts = {} + expected_planning = { + 'gitlab': 'exact_git_v1', + 'dockerhub': 'docker_direct_v1', + 'huggingface': 'huggingface_space_v1', + } + for row in reservations: + snapshot = json.loads(row['remote_execution_snapshot_json']) + source = row['source'] + planning = snapshot['planning']['kind'] + require(planning == expected_planning[source], 'direct assignment planning') + if source != 'gitlab': + require( + snapshot['credential_ref'] == { + 'source': source, 'auth_entry': '', + }, + 'direct assignment planning', + ) + planning_counts[planning] = planning_counts.get(planning, 0) + 1 + require(not any(BUNDLES.rglob('*.trb')), 'server bundle spool cleanup') + capacity = db.pipeline_capacity_snapshot() + require(capacity['bundle_items'] == capacity['bundle_bytes'] == 0, + 'bundle capacity release') + return { + 'schema': 1, 'phase': self.phase, 'counts': counts, + 'claim_count': len(reservations), + 'detectors': sorted({row['detector_name'] for row in findings}), + 'candidate_services': sorted({row['service'] for row in candidates}), + 'secret_hashes': sorted(row['secret_hash'] for row in findings), + 'commit_hashes': sorted(row['commit_hash'] for row in findings), + 'receipt_count': len({row['remote_receipt_id'] for row in reservations}), + 'source_counts': source_counts, + 'planning_counts': planning_counts, + 'capacity': { + 'bundle_items': capacity['bundle_items'], + 'bundle_bytes': capacity['bundle_bytes'], + 'keycheck_items': capacity['keycheck_items'], + }, + } + finally: + db.close() + + def monitor_completion(self): + while not self.stop.wait(0.1): + if self.ingester_error: + return + try: + evidence = self.evidence() + except Exception as exc: + reason = type(exc).__name__ + message = str(exc) + prefix = 'packaged worker E2E: ' + if ( + isinstance(exc, RuntimeError) and message.startswith(prefix) + and message[len(prefix):] in COMPLETION_REQUIREMENTS + ): + reason = message[len(prefix):] + write_json(CONTROL / 'completion-wait.json', { + 'schema': 1, 'phase': self.phase, 'reason': reason, + }) + continue + try: + (CONTROL / 'completion-wait.json').unlink() + except FileNotFoundError: + pass + write_json(CONTROL / 'completed.json', evidence) + return + + def serve(self, stop_after_claims): + import uvicorn + + server = uvicorn.Server(uvicorn.Config( + self.app(stop_after_claims), host='0.0.0.0', port=PORT, + ssl_certfile=str(FIXTURE / 'worker_tls_cert.pem'), + ssl_keyfile=str(FIXTURE / 'worker_tls_key.pem'), + access_log=False, log_level='warning', server_header=False, + )) + self.server = server + server.run() + self.server = None + + def run(self, first): + self.initialize_database(first) + thread = threading.Thread(target=self.ingester_loop, name='result-ingester', daemon=True) + thread.start() + require(self.ingester_ready.wait(30) and not self.ingester_error, 'ingester startup') + self.serve(stop_after_claims=first) + require(first and (CONTROL / 'outage.json').is_file(), 'planned API outage') + while not (CONTROL / 'restore').is_file(): + require(not self.stop.wait(0.1), 'harness stopped before restore') + self.enqueue_direct_targets() + monitor = threading.Thread(target=self.monitor_completion, name='completion-monitor', daemon=True) + monitor.start() + self.serve(stop_after_claims=False) + self.stop.set() + thread.join(10) + monitor.join(2) + require(not self.ingester_error, 'result ingester failure') + + +def main(): + require(sys.platform == 'linux' and os.getuid() == os.getgid() == 10001, + 'Linux UID 10001 required') + require(sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode, + 'isolated Python required') + os.umask(0o077) + sys.path.insert(0, str(APP)) + bootstrap = runpy.run_path(str(APP / 'child_bootstrap.py')) + bootstrap['_enable_dependency_paths']('supervisor') + fixture = read_json(FIXTURE / 'fixture.json') + first = start_postgres() + harness = Harness(fixture) + + def terminate(_signum, _frame): + if harness.server is not None: + harness.server.should_exit = True + harness.stop.set() + + signal.signal(signal.SIGTERM, terminate) + signal.signal(signal.SIGINT, terminate) + try: + harness.run(first) + finally: + harness.stop.set() + stop_postgres() + + +if __name__ == '__main__': + main() diff --git a/tests/parity_helpers.py b/tests/parity_helpers.py new file mode 100644 index 0000000..ab0e6e1 --- /dev/null +++ b/tests/parity_helpers.py @@ -0,0 +1,126 @@ +import contextlib +import json +import os +from pathlib import Path +import re +import shutil +import subprocess + + +_DYNAMIC_FIELDS = { + 'duration_sec', 'finding_uid', 'scan_duration', 'scan_duration_ms', + 'scan_started_at', 'source_manager_worker_id', 'timestamp', 'trace_id', + 'transfer_duration_ms', 'worker_id', +} +_TEMP_COMPONENT = re.compile( + r'(?i)(?:trufflehog-run|trufflehog-\d+|docker-layer)-[^/\\\s"]+', +) +_ISO_TIMESTAMP = re.compile( + r'\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})', +) +_LOG_ID = re.compile( + r'("(?:trace|worker|source_manager_worker)_id"\s*:\s*")[^"]+("\s*[,}])', +) + + +def configured_trufflehog(): + candidates = ( + os.getenv('TRUF_TEST_TRUFFLEHOG'), + r'C:\Tools\trufflehog.exe', + shutil.which('trufflehog'), + ) + return next((Path(value) for value in candidates if value and Path(value).is_file()), None) + + +@contextlib.contextmanager +def native_streamed_command(command, timeout, env=None, **_kwargs): + import scanner + + completed = subprocess.run( + command, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + env=env, + check=False, + timeout=max(1.0, float(timeout)), + ) + with scanner.streamed_output_from_text( + completed.stdout.decode('utf-8', errors='replace'), + completed.stderr.decode('utf-8', errors='replace'), + completed.returncode, + ) as output: + yield output + + +def normalized_bundle_evidence(path): + from result_bundle import ResultBundleReader + + def normalize(value): + if isinstance(value, dict): + return { + key: normalize(item) + for key, item in sorted(value.items()) + if key not in _DYNAMIC_FIELDS + } + if isinstance(value, list): + return [normalize(item) for item in value] + if isinstance(value, str): + value = _TEMP_COMPONENT.sub('', value) + value = _ISO_TIMESTAMP.sub('', value) + value = _LOG_ID.sub(r'\1\2', value) + try: + parsed = json.loads(value) + except (TypeError, ValueError): + return value + if isinstance(parsed, (dict, list)): + return json.dumps( + normalize(parsed), ensure_ascii=True, sort_keys=True, + separators=(',', ':'), + ) + return value + return value + + def ordered(values): + normalized = [normalize(value) for value in values] + return sorted( + normalized, + key=lambda value: json.dumps( + value, ensure_ascii=True, sort_keys=True, separators=(',', ':'), + ), + ) + + reader = ResultBundleReader(path) + reader.validate() + return { + 'metadata': normalize(reader.metadata()), + 'findings': ordered(reader.iter_findings()), + 'errors': ordered(reader.iter_errors()), + 'candidates': ordered(reader.iter_candidates()), + } + + +def bundle_evidence_difference_paths(left, right, path='$', limit=50): + differences = [] + + def compare(first, second, current): + if len(differences) >= limit: + return + if type(first) is not type(second): + differences.append(current) + elif isinstance(first, dict): + for key in sorted(set(first) | set(second)): + if key not in first or key not in second: + differences.append(f'{current}.{key}') + else: + compare(first[key], second[key], f'{current}.{key}') + elif isinstance(first, list): + if len(first) != len(second): + differences.append(f'{current}.length') + for index, (first_item, second_item) in enumerate(zip(first, second)): + compare(first_item, second_item, f'{current}[{index}]') + elif first != second: + differences.append(current) + + compare(left, right, path) + return differences diff --git a/tests/requirements-browser.txt b/tests/requirements-browser.txt new file mode 100644 index 0000000..824ce70 --- /dev/null +++ b/tests/requirements-browser.txt @@ -0,0 +1 @@ +playwright==1.58.0 diff --git a/tests/test_admin_api.py b/tests/test_admin_api.py new file mode 100644 index 0000000..aeec950 --- /dev/null +++ b/tests/test_admin_api.py @@ -0,0 +1,4182 @@ +import asyncio +import hashlib +import html +import json +import os +from pathlib import Path +import re +import sys +import tempfile +import threading +from types import SimpleNamespace +import unittest +import uuid +from unittest import mock +from urllib.parse import urlencode, urljoin + + +ROOT = Path(__file__).resolve().parents[1] +APP_DIR = ROOT / 'app' +sys.path.insert(0, str(APP_DIR)) + +from starlette.testclient import TestClient +from starlette.datastructures import Headers +from starlette.requests import Request +from starlette.responses import Response + +from admin_api import ( + ADMIN_PREFIX, EDGE_MARKER_HEADER, OPERATOR_HEADER, SECURITY_HEADERS, + AdminAPIError, AdminService, _dispatch_managed_file_mutation, + _dispatch, _dispatch_runtime_document, _download_managed_file, _form_fields, + _managed_file_download_response, _ManagedFileStreamingResponse, + _parse_managed_file_content, + _trusted_operator, _validate_cap, +) +from managed_files import ( + ManagedFileAccessError, ManagedFileDirectoryEntry, ManagedFileDownload, + ManagedFileIdentity, ManagedFileLimits, ManagedFileListing, + ManagedFileMutation, ManagedFilePermissions, ManagedFileRoot, + ManagedFileRootRegistry, +) +from scanner_db import ( + RuntimeControlRevisionConflictError, RuntimeOperationIdentityConflictError, +) +from runtime_document import RuntimeDocumentError +from runtime_security import ensure_private_directory +from worker_api import WorkerAPIError, create_worker_app +from worker_contracts import ( + MAX_DIAGNOSTIC_BODY_BYTES, MAX_DIAGNOSTIC_LOG_BYTES, + make_body_material, make_log_material, +) + + +ORIGIN = 'https://admin.example.test' +MARKER = 'fixture-edge-marker-value-32bytes-minimum' +OPERATOR = 'fixture.operator' + + +class FakeWorkerService: + def __init__(self, root): + self.bundle_root = root + self.max_bundle_bytes = 1024 * 1024 + self.claim_retry_after_seconds = 5 + + def authenticate(self, authorization): + raise WorkerAPIError(401, 'unauthorized', 'worker credentials are invalid') + + def reap(self): + return [] + + +class AdminValidationTests(unittest.TestCase): + def test_assignment_cap_accepts_integer_zero(self): + self.assertEqual(_validate_cap(0), 0) + + +class FakeManagedFileTraversal: + def __init__(self): + self.files = { + 'guide.txt': b'fixture managed content', + 'nested/report.bin': b'\x00\xffreport', + } + self.calls = [] + self.closed = 0 + self.error_category = None + self.stream_downloads = False + self.last_snapshot = None + + @staticmethod + def _identity(content): + return ManagedFileIdentity(hashlib.sha256(content).hexdigest(), len(content)) + + def _fail(self): + if self.error_category: + raise ManagedFileAccessError(self.error_category) + + def list_directory(self, root_id, relative_path=None): + self._fail() + self.calls.append(('list', root_id, relative_path)) + prefix = '' if relative_path is None else relative_path + '/' + entries = {} + for path, content in self.files.items(): + if not path.startswith(prefix): + continue + remainder = path[len(prefix):] + name, separator, _tail = remainder.partition('/') + entries[name] = ( + ManagedFileDirectoryEntry(name, 'directory', None) + if separator else + ManagedFileDirectoryEntry(name, 'file', len(content)) + ) + values = tuple(entries[name] for name in sorted(entries)) + return ManagedFileListing(values, sum(len(item.name.encode()) for item in values)) + + def download_file(self, root_id, relative_path): + self._fail() + self.calls.append(('download', root_id, relative_path)) + if relative_path not in self.files: + raise ManagedFileAccessError('not_found') + content = self.files[relative_path] + if self.stream_downloads: + snapshot = FakeManagedFileSnapshot(content) + self.last_snapshot = snapshot + return ManagedFileDownload( + self._identity(content), snapshot=snapshot, + ) + return ManagedFileDownload(self._identity(content), content) + + def mutation_file_identity( + self, root_id, relative_path, operation, + *, require_private_sha256=None): + self._fail() + self.calls.append(( + 'identity', root_id, relative_path, operation.value, + require_private_sha256, + )) + if relative_path not in self.files: + raise ManagedFileAccessError('not_found') + return self._identity(self.files[relative_path]) + + def create_replace_file(self, root_id, relative_path, payload, *, expected_sha256): + self._fail() + self.calls.append(( + 'create_replace', root_id, relative_path, expected_sha256, + hashlib.sha256(payload).hexdigest(), len(payload), + )) + before_content = self.files.get(relative_path) + if expected_sha256 is None: + if before_content is not None: + raise ManagedFileAccessError('hash_conflict') + elif before_content is None: + raise ManagedFileAccessError('not_found') + elif self._identity(before_content).sha256 != expected_sha256: + raise ManagedFileAccessError('hash_conflict') + before = self._identity(before_content) if before_content is not None else None + after = self._identity(payload) + written = before != after + self.files[relative_path] = payload + return ManagedFileMutation(before, after, written) + + def delete_file(self, root_id, relative_path, *, expected_sha256): + self._fail() + self.calls.append(('delete', root_id, relative_path, expected_sha256)) + if relative_path not in self.files: + raise ManagedFileAccessError('not_found') + before = self._identity(self.files[relative_path]) + if before.sha256 != expected_sha256: + raise ManagedFileAccessError('hash_conflict') + del self.files[relative_path] + return ManagedFileMutation(before, None, True) + + def close(self): + self.closed += 1 + + +class FakeManagedFileSnapshot: + def __init__(self, content): + self.content = content + self.closed = 0 + + def chunks(self): + try: + for offset in range(0, len(self.content), 3): + yield self.content[offset:offset + 3] + finally: + self.close() + + def close(self): + if not self.closed: + self.closed = 1 + + +class RecordingDB: + enabled = True + calls = [] + fail_queue = False + source_operations = {} + worker_admin_operations = {} + document_operations = {} + apply_operations = {} + managed_file_operations = {} + audit_events = [] + audit_next_before_event_id = None + source_identity_conflict = False + completion_failures = 0 + control_identity_conflict = False + control_revision_conflict = False + document_completion_failures = 0 + managed_file_completion_failures = 0 + managed_file_execution_lock = threading.Lock() + + def __init__(self, **kwargs): + self.managed_file_execution_held = False + self.calls.append(('open', kwargs)) + + def close(self): + if self.managed_file_execution_held: + self.managed_file_execution_held = False + self.managed_file_execution_lock.release() + self.calls.append(('close',)) + + def acquire_runtime_managed_file_execution(self, operation_id): + if self.managed_file_execution_held: + raise RuntimeError('managed file execution lock is already held') + self.managed_file_execution_lock.acquire() + self.managed_file_execution_held = True + self.calls.append(('managed_file_execution_acquire', operation_id)) + return True + + def release_runtime_managed_file_execution(self, operation_id): + if not self.managed_file_execution_held: + raise RuntimeError('managed file execution lock is not held') + self.managed_file_execution_held = False + self.managed_file_execution_lock.release() + self.calls.append(('managed_file_execution_release', operation_id)) + return True + + def admin_remote_worker_snapshot(self, limit, filters=None): + filters = dict(filters or {}) + self.calls.append(('snapshot', limit, filters)) + common = { + 'queue_id': 5, 'user_key': 'fixture-user', + 'active_assignment_cap': 2, 'device_key': 'fixture-device', + 'source': 'gitlab', 'target': 'example/project@commit', + 'issued_at': '2026-09-20T00:01:00+00:00', + 'assignment_deadline_at': '2026-09-21T00:01:00+00:00', + 'remote_result_upload_body_timeout_seconds': 1800, + 'finished_at': None, 'duration_seconds': None, + 'accepted': False, 'ingested': False, 'assignment_code': None, + 'target_scan_id': None, 'scan_error_count': None, + 'first_error_summary': None, 'skipped_reason': None, + 'diagnostic_categories': None, 'diagnostic_codes': None, + 'primary_diagnostic': None, 'phase_started_at': None, + 'last_progress_at': None, 'last_progress_received_at': None, + 'scan_deadline_at': None, 'scan_remaining_seconds': None, + 'assignment_remaining_seconds': 3600, + 'ingestion_state': None, 'projection_state': None, + 'protocol_version': '2', 'bundle_format_version': '2', + 'platform_tag': 'linux-x86_64', 'code_manifest_sha256': '1' * 64, + 'detector_policy_sha256': '2' * 64, + 'effective_config_sha256': '3' * 64, + } + assignments = [{ + **common, 'reservation_id': 4, + 'assignment_outcome': 'unfinished', 'scan_outcome': 'unavailable', + 'diagnostic_count': 0, 'diagnostic_projection_version': None, + 'active_phase': 'scanning', 'phase_age_seconds': 17, + 'last_progress_age_seconds': 3, 'slot_id': 0, + 'scan_deadline_at': '2026-09-20T00:11:00+00:00', + 'scan_remaining_seconds': 500, + }, { + **common, 'reservation_id': 5, 'target_scan_id': 50, + 'assignment_outcome': 'accepted', 'scan_outcome': 'degraded', + 'finished_at': '2026-09-20T00:10:00+00:00', + 'scan_warning_class': 'detector_timeout', + 'scan_warning_summary': 'bounded detector output timed out', + 'accepted': True, 'ingested': True, 'diagnostic_count': 2, + 'diagnostic_projection_version': 1, + 'diagnostic_categories': 'rate_limit, scanner', + 'diagnostic_codes': 'provider.rate_limit, scanner.exit', + 'primary_diagnostic': 'scanner/scanner.exit', + 'active_phase': 'awaiting_receipt', 'phase_age_seconds': 4, + 'last_progress_age_seconds': 2, 'slot_id': 1, + 'ingestion_state': 'acknowledged', 'projection_state': 'completed', + }, { + **common, 'reservation_id': 6, + 'assignment_outcome': 'prebundle_failed', 'scan_outcome': 'unavailable', + 'finished_at': '2026-09-20T00:09:00+00:00', + 'diagnostic_count': 1, 'diagnostic_projection_version': 1, + 'diagnostic_categories': 'storage', 'diagnostic_codes': 'bundle.fsync', + 'primary_diagnostic': 'storage/bundle.fsync', 'active_phase': 'bundling', + 'phase_age_seconds': 9, 'last_progress_age_seconds': 8, 'slot_id': 0, + }, { + **common, 'reservation_id': 7, + 'assignment_outcome': 'expired', 'scan_outcome': 'unavailable', + 'finished_at': '2026-09-21T00:01:00+00:00', + 'diagnostic_count': 1, 'diagnostic_projection_version': 1, + 'diagnostic_categories': 'assignment_expired', + 'diagnostic_codes': 'assignment.expired', + 'primary_diagnostic': 'assignment_expired/assignment.expired', + 'active_phase': 'scanning', 'phase_age_seconds': 300, + 'last_progress_age_seconds': 280, 'slot_id': 0, + }, { + **common, 'reservation_id': 8, 'target_scan_id': 80, + 'assignment_outcome': 'accepted', 'scan_outcome': 'error', + 'finished_at': '2026-09-20T00:08:00+00:00', + 'accepted': True, 'diagnostic_count': 0, + 'diagnostic_projection_version': None, 'active_phase': None, + 'phase_age_seconds': None, 'last_progress_age_seconds': None, + 'slot_id': None, + 'remote_result_upload_body_timeout_seconds': None, + 'protocol_version': '1', + }, { + **common, 'reservation_id': 9, + 'assignment_outcome': 'unfinished', 'scan_outcome': 'unavailable', + 'diagnostic_count': 0, 'diagnostic_projection_version': 1, + 'active_phase': None, 'phase_age_seconds': None, + 'last_progress_age_seconds': None, 'slot_id': None, + }] + return { + 'filters': filters, + 'users': [{'user_key': 'fixture-user', 'active_assignment_cap': 2, 'disabled': False}], + 'workers': [{ + 'device_key': 'fixture-device', 'user_key': 'fixture-user', + 'active_assignment_cap': 2, 'active_slot_count': 1, + 'current_phases': 'scanning', + 'latest_progress_age_seconds': 3, + 'known_reasons': None, 'pending_local_recovery': False, + 'active_package_identity': 'linux-x86_64:' + ('1' * 12), + 'unfinished_count': 1, 'completed_count': 2, 'failed_count': 0, + 'expired_count': 0, 'last_contact_at': '2026-09-20T00:02:00+00:00', + 'revoked': False, + }], + 'assignments': assignments, + 'deferred_queue': [], + } + + def admin_worker_diagnostic_groups(self, limit, filters=None, *, occurrence_offset=0): + filters = dict(filters or {}) + self.calls.append(('diagnostic_groups', limit, filters, occurrence_offset)) + occurrences = [{ + 'diagnostic_uid': 'b' * 64, 'reservation_id': 5, + 'target_scan_id': 50, 'source': 'gitlab', + 'worker': 'fixture-device', 'user': 'fixture-user', + 'assignment_outcome': 'accepted', 'scan_outcome': 'error', + 'phase': 'scanning', 'kind': 'provider_http', + 'category': 'rate_limit', 'code': 'provider.rate_limit', + 'summary': 'rate limited', 'retryable': True, + 'occurred_at': '2026-09-20T00:03:00Z', + 'received_at': '2026-09-20T00:03:01Z', + }, { + 'diagnostic_uid': 'c' * 64, 'reservation_id': 6, + 'target_scan_id': None, 'source': 'gitlab', + 'worker': 'fixture-device', 'user': 'fixture-user', + 'assignment_outcome': 'prebundle_failed', 'scan_outcome': 'unavailable', + 'phase': 'scanning', 'kind': 'provider_http', + 'category': 'rate_limit', 'code': 'provider.rate_limit', + 'summary': 'rate limited again', 'retryable': True, + 'occurred_at': '2026-09-20T00:02:00Z', + 'received_at': '2026-09-20T00:02:01Z', + }] + return { + 'filters': filters, 'occurrence_limit': limit, + 'matched_occurrence_count': 201, 'page_occurrence_count': 2, + 'occurrence_offset': occurrence_offset, + 'has_previous': occurrence_offset > 0, + 'has_next': occurrence_offset == 0, + 'previous_occurrence_offset': ( + max(0, occurrence_offset - limit) if occurrence_offset else None + ), + 'next_occurrence_offset': limit if occurrence_offset == 0 else None, + 'truncated': occurrence_offset == 0, + 'groups': [{ + 'fingerprint': 'f' * 64, 'count': 2, + 'affected_assignment_count': 2, 'page_occurrence_count': 2, + 'affected_assignments': [5, 6], 'occurrences': occurrences, + }], + } + + def admin_worker_duration_metrics(self, limit, filters=None, *, offset=0): + filters = dict(filters or {}) + self.calls.append(('duration_metrics', limit, filters, offset)) + metrics = [{ + 'source': 'gitlab', 'phase': 'scanning', 'outcome': 'error', + 'sample_count': 12, 'sufficient': True, 'minimum_sample_count': 5, + 'p50_seconds': 10.0, 'p95_seconds': 18.0, 'p99_seconds': 21.0, + }, { + 'source': 'gitlab', 'phase': 'uploading', 'outcome': 'accepted', + 'sample_count': 2, 'sufficient': False, 'minimum_sample_count': 5, + 'p50_seconds': 3.0, 'p95_seconds': 4.0, 'p99_seconds': 4.0, + }] + return { + 'filters': filters, 'metrics': metrics, + 'total_group_count': 201, 'metric_limit': limit, + 'metric_offset': offset, 'page_group_count': len(metrics), + 'has_previous': offset > 0, 'has_next': offset == 0, + 'previous_metric_offset': max(0, offset - limit) if offset else None, + 'next_metric_offset': limit if offset == 0 else None, + 'truncated': offset == 0, + } + + @staticmethod + def _diagnostic_fixture(reservation_id=5, diagnostic_uid=None): + def material_value(material): + return { + 'encoding': material.encoding.value, + 'head': material.head, 'tail': material.tail, + 'original_size': material.original_size, + 'stored_size': material.stored_size, + 'sha256': material.sha256, 'truncated': material.truncated, + } + + material = material_value(make_body_material( + b'b' * MAX_DIAGNOSTIC_BODY_BYTES, + )) + truncated = material_value(make_log_material( + b'l' * (MAX_DIAGNOSTIC_LOG_BYTES + 1), + )) + diagnostic_uid = diagnostic_uid or ('b' * 64) + return { + 'schema': 1, 'diagnostic_uid': diagnostic_uid, + 'occurrence_id': 'fixture-occurrence', 'reservation_id': reservation_id, + 'scan_event_id': None, 'slot_id': 0, 'source': 'gitlab', + 'phase': 'scanning', 'kind': 'provider_http', + 'category': 'rate_limit', 'code': 'provider.rate_limit', + 'summary': 'provider returned a limit', 'retryable': True, + 'attempt': 1, 'assignment_outcome': None, 'scan_outcome': None, + 'occurred_at': '2026-09-20T00:03:00Z', + 'captured_at': '2026-09-20T00:03:01Z', 'received_at': None, + 'http': { + 'operation': 'GET', 'status_code': 429, + 'content_type': 'text/plain', 'request_id': 'fixture', + 'body': material, 'headers': None, + }, + 'process': { + 'name': 'trufflehog', 'exit_code': 1, 'signal': None, + 'timed_out': False, 'stdout': None, 'stderr': truncated, + }, + 'exception': None, + } + + def admin_worker_assignment_detail(self, reservation_id, **_kwargs): + self.calls.append(('assignment_detail', reservation_id, _kwargs)) + reservation_id = int(reservation_id) + states = { + 4: ('unfinished', 'unavailable', None, 'scanning'), + 5: ('accepted', 'error', '2026-09-20T00:06:00Z', 'awaiting_receipt'), + 6: ('prebundle_failed', 'unavailable', '2026-09-20T00:04:00Z', 'bundling'), + 7: ('expired', 'unavailable', '2026-09-20T00:11:00Z', 'scanning'), + 8: ('accepted', 'error', '2026-09-20T00:06:00Z', 'awaiting_receipt'), + } + if reservation_id not in states: + return None + assignment_outcome, scan_outcome, resolved_at, current_phase = states[reservation_id] + legacy = reservation_id == 8 + diagnostic_uid = ({5: 'b', 6: 'c', 7: 'd'}.get(reservation_id, 'b')) * 64 + envelope = self._diagnostic_fixture(reservation_id, diagnostic_uid) + canonical = json.dumps(envelope, ensure_ascii=True, sort_keys=True, separators=(',', ':')) + diagnostic_items = [] + if reservation_id in {5, 6, 7}: + diagnostic_items.append({ + 'diagnostic_uid': diagnostic_uid, 'phase': current_phase, + 'kind': 'provider_http', 'category': 'rate_limit', + 'code': 'provider.rate_limit', 'retryable': True, + 'summary': 'provider returned a limit', + 'occurred_at': '2026-09-20T00:03:00Z', + 'received_at': '2026-09-20T00:03:01Z', + 'envelope_sha256': '6' * 64, 'diagnostic': envelope, + 'canonical_envelope_json': canonical, + }) + timeline = [{ + 'timestamp': '2026-09-20T00:01:00Z', + 'kind': 'assignment', 'label': 'issued', + }] + if not legacy: + timeline.append({ + 'timestamp': '2026-09-20T00:02:00Z', 'kind': 'phase', + 'label': current_phase, 'sequence': 1, + 'received_at': '2026-09-20T00:02:01Z', + }) + if reservation_id == 5: + timeline.extend([ + {'timestamp': '2026-09-20T00:04:00Z', 'kind': 'transport', 'label': 'bundle received'}, + {'timestamp': '2026-09-20T00:05:00Z', 'kind': 'ingestion', 'label': 'bundle ingested'}, + {'timestamp': '2026-09-20T00:05:45Z', 'kind': 'settlement', 'label': 'queue settled'}, + {'timestamp': '2026-09-20T00:06:00Z', 'kind': 'projection', 'label': 'projection completed'}, + ]) + elif reservation_id == 6: + timeline.append({ + 'timestamp': resolved_at, 'kind': 'receipt', 'label': 'prebundle_report', + }) + elif reservation_id == 7: + timeline.append({ + 'timestamp': resolved_at, 'kind': 'receipt', 'label': 'expired', + }) + return { + 'schema': 1, + 'assignment': { + 'reservation_id': int(reservation_id), 'queue_id': 5, + 'source': 'gitlab', 'target': 'example/project@commit', + 'user': 'fixture-user', 'worker': 'fixture-device', + 'active_assignment_cap': 2, + 'assignment_outcome': assignment_outcome, + 'scan_outcome': scan_outcome, + 'issued_at': '2026-09-20T00:01:00Z', + 'resolved_at': resolved_at, + 'assignment_code': None, 'assignment_detail': None, + }, + 'deadlines': { + 'assignment_deadline_at': '2026-09-21T00:01:00Z', + 'scan_deadline_at': '2026-09-20T00:11:00Z', + 'upload_timeout_seconds': None if legacy else 1800, + 'upload_timeout_availability': ( + 'legacy/unavailable' if legacy + else 'persisted at assignment issuance' + ), + }, + 'package': { + 'protocol_version': 1 if legacy else 2, + 'bundle_format_version': 2, + 'platform_tag': 'linux-x86_64', + 'code_manifest_sha256': '1' * 64, + 'detector_policy_sha256': '2' * 64, + 'effective_config_sha256': '3' * 64, + }, + 'transport': { + 'resolution': None, 'receipt_id': ( + 'receipt-fixture' if resolved_at else None + ), + 'bundle_state': 'acknowledged' if reservation_id in {5, 8} else None, + 'bundle_ready_at': '2026-09-20T00:04:00Z' if reservation_id in {5, 8} else None, + 'bundle_committed_at': '2026-09-20T00:05:00Z' if reservation_id in {5, 8} else None, + 'bundle_acknowledged_at': '2026-09-20T00:05:30Z' if reservation_id in {5, 8} else None, + 'queue_status': 'done' if resolved_at else 'in_progress', + 'queue_settled_at': '2026-09-20T00:05:45Z' if reservation_id in {5, 8} else None, + 'projection_status': 'completed' if reservation_id in {5, 8} else None, + 'projection_completed_at': '2026-09-20T00:06:00Z' if reservation_id in {5, 8} else None, + }, + 'scan': { + 'available': reservation_id in {5, 8}, + 'target_scan_id': reservation_id * 10 if reservation_id in {5, 8} else None, + 'status': 'error' if reservation_id in {5, 8} else None, + 'started_at': '2026-09-20T00:02:00Z', + 'ended_at': '2026-09-20T00:04:00Z', + 'duration_seconds': 120.0, 'findings_count': 0, + 'verified_findings_count': 0, 'error_count': 1, + 'skipped_reason': None, + 'first_error_summary': 'legacy summary' if legacy else None, + 'warning_class': 'detector_timeout' if reservation_id == 5 else None, + 'warning_summary': ( + 'bounded detector output timed out' if reservation_id == 5 else None + ), + }, + 'timeline': timeline, + 'durations': [{ + 'phase': current_phase, 'duration_seconds': 120.0, + 'outcome': scan_outcome, + 'complete': resolved_at is not None, + 'ended_at': resolved_at or '2026-09-20T00:04:00Z', + 'authority': ( + 'assignment_resolution' if resolved_at else 'database_current_time' + ), + }], + 'progress': { + 'available': not legacy, + 'availability': ( + 'legacy/unavailable' if legacy else 'current' + ), + 'truncated': False, + 'total_event_count': 0 if legacy else 1, + 'omitted_older_event_count': 0, + 'current_phase': None if legacy else current_phase, + 'phase_started_at': None if legacy else '2026-09-20T00:02:00Z', + 'phase_age_seconds': None if legacy else 120.0, + 'last_progress_at': None if legacy else '2026-09-20T00:03:55Z', + 'last_progress_age_seconds': None if legacy else 5.0, + 'age_authority': ( + None if legacy else + 'assignment_resolution' if resolved_at + else 'database_current_time' + ), + 'events': [], + }, + 'diagnostics': { + 'availability': 'legacy/unavailable' if legacy else 'current', + 'projection_version': None if legacy or reservation_id == 4 else 1, + 'declared_count': None if legacy else len(diagnostic_items), + 'truncated': False, + 'items': diagnostic_items, + }, + 'legacy_evidence': { + 'available': legacy, 'explicitly_not_an_envelope': True, + 'truncated': False, + 'errors': [{ + 'id': 1, 'category': 'scanner', 'summary': 'legacy summary', + 'raw_error': 'exact legacy raw error', + 'created_at': '2026-09-20T00:04:00Z', + }] if legacy else [], + 'first_error_summary': 'legacy summary' if legacy else None, + }, + } + + def admin_worker_diagnostic_envelope(self, reservation_id, diagnostic_uid): + self.calls.append(('diagnostic_envelope', reservation_id, diagnostic_uid)) + if int(reservation_id) != 5 or diagnostic_uid != 'b' * 64: + return None + envelope = self._diagnostic_fixture(5, 'b' * 64) + canonical = json.dumps(envelope, ensure_ascii=True, sort_keys=True, separators=(',', ':')) + return {'canonical_json': canonical, 'sha256': '6' * 64} + + def admin_target_queue_health(self, sources): + self.calls.append(('queue_counts', tuple(sources))) + if self.fail_queue: + if self.fail_queue == 'malformed': + return { + 'counts': {}, 'degraded': False, 'stale': False, + 'truncated_statuses': 1, + } + if self.fail_queue == 'degraded': + return { + 'counts': {}, 'degraded': True, 'stale': True, + 'reason': 'bounded_count_query_failed', + 'retry_after_sec': 300, 'truncated_statuses': [], + } + raise RuntimeError('sensitive queue failure detail') + return { + 'counts': {'pending': 3, 'done': 100}, + 'degraded': True, 'stale': False, + 'reason': 'bounded_status_sample', + 'sample_limit_per_status': 100, 'timeout_ms': 5000, + 'sampled_rows': 103, 'retry_after_sec': 0, + 'truncated_statuses': ['done'], + } + + def runtime_drain_progress(self): + self.calls.append(('drain_progress',)) + return { + 'revision': 7, + 'discovery_paused': False, + 'dispatch_paused': True, + 'drain_state': 'draining', + 'effective_discovery_paused': True, + 'effective_dispatch_paused': True, + 'actor': 'fixture.operator', + 'operation_id': '00000000-0000-4000-8000-000000000001', + 'created_at': '2026-09-20T00:00:00+00:00', + 'updated_at': '2026-09-20T00:01:00+00:00', + 'live_remote_assignments': 2, + 'precommit_result_bundles': 1, + 'blocker_count': 3, + } + + def set_runtime_dispatch_paused( + self, paused, *, expected_revision, actor, operation_id, + ): + if self.control_revision_conflict: + raise RuntimeControlRevisionConflictError( + expected_revision, {'revision': expected_revision + 1}, + ) + if self.control_identity_conflict: + raise RuntimeOperationIdentityConflictError('fixture conflict') + self.calls.append(( + 'dispatch_paused', paused, expected_revision, actor, operation_id, + )) + return {'after': {'revision': expected_revision + 1}} + + def start_runtime_drain(self, *, expected_revision, actor, operation_id): + if self.control_revision_conflict: + raise RuntimeControlRevisionConflictError( + expected_revision, {'revision': expected_revision + 1}, + ) + if self.control_identity_conflict: + raise RuntimeOperationIdentityConflictError('fixture conflict') + self.calls.append(('drain_start', expected_revision, actor, operation_id)) + return {'after': {'revision': expected_revision + 1}} + + def cancel_runtime_drain(self, *, expected_revision, actor, operation_id): + if self.control_revision_conflict: + raise RuntimeControlRevisionConflictError( + expected_revision, {'revision': expected_revision + 1}, + ) + if self.control_identity_conflict: + raise RuntimeOperationIdentityConflictError('fixture conflict') + self.calls.append(('drain_cancel', expected_revision, actor, operation_id)) + return {'after': {'revision': expected_revision + 1}} + + def recent_runtime_operations( + self, limit, *, before_updated_at=None, before_operation_id=None, + ): + self.calls.append(( + 'recent_operations', limit, before_updated_at, before_operation_id, + )) + recorded = ( + list(type(self).apply_operations.values()) + + list(type(self).document_operations.values()) + + list(type(self).managed_file_operations.values()) + ) + if recorded: + return [dict(operation) for operation in recorded[:limit]] + return [{ + 'operation_id': '00000000-0000-4000-8000-000000000002', + 'actor': 'fixture.operator', 'action': 'restart', + 'target_ref': 'runtime', 'status': 'succeeded', + 'safe_category': None, 'safe_detail': None, + 'requested_at': '2026-09-20T00:00:00+00:00', + 'completed_at': '2026-09-20T00:00:05+00:00', + 'updated_at': '2026-09-20T00:00:05+00:00', + }] + + def set_runtime_discovery_paused( + self, paused, *, expected_revision, actor, operation_id, + ): + if self.control_revision_conflict: + raise RuntimeControlRevisionConflictError( + expected_revision, {'revision': expected_revision + 1}, + ) + self.calls.append(( + 'discovery_paused', paused, expected_revision, actor, operation_id, + )) + return {'after': {'revision': expected_revision + 1}} + + def create_runtime_source_operation( + self, *, operation_id, actor, source_id, source_action, + interval_seconds=None, mode=None, restart_enabled=None, + restart_delay_seconds=None, + ): + if self.source_identity_conflict: + raise RuntimeOperationIdentityConflictError('fixture conflict') + self.calls.append(( + 'source_operation_create', operation_id, actor, source_id, + source_action, interval_seconds, mode, restart_enabled, + restart_delay_seconds, + )) + existing = self.source_operations.get(operation_id) + if existing is not None: + return {**existing, 'replayed': True} + operation = { + 'operation_id': operation_id, 'status': 'running', 'replayed': False, + 'resulting_identity': None, + } + self.source_operations[operation_id] = operation + return dict(operation) + + def complete_runtime_source_operation( + self, operation_id, *, succeeded, outcome=None, + ): + self.calls.append(( + 'source_operation_complete', operation_id, succeeded, outcome, + )) + if type(self).completion_failures: + type(self).completion_failures -= 1 + raise RuntimeError('fixture completion failure') + operation = { + **self.source_operations[operation_id], + 'status': 'succeeded' if succeeded else 'failed', + 'resulting_identity': ( + {'outcome': outcome} if succeeded else None + ), + } + self.source_operations[operation_id] = operation + return dict(operation) + + def create_runtime_worker_admin_operation( + self, *, operation_id, actor, action, target_ref, request_sha256, + ): + self.calls.append(( + 'worker_admin_create', operation_id, actor, action, target_ref, + request_sha256, + )) + existing = self.worker_admin_operations.get(operation_id) + if existing is not None: + return {**existing, 'replayed': True} + operation = { + 'operation_id': operation_id, 'status': 'running', 'replayed': False, + 'resulting_identity': None, + } + self.worker_admin_operations[operation_id] = operation + return dict(operation) + + def complete_runtime_worker_admin_operation( + self, operation_id, *, succeeded, affected_count=None, + ): + self.calls.append(( + 'worker_admin_complete', operation_id, succeeded, affected_count, + )) + operation = { + **self.worker_admin_operations[operation_id], + 'status': 'succeeded' if succeeded else 'failed', + 'resulting_identity': ( + {'outcome': 'completed', 'affected_count': affected_count} + if succeeded else None + ), + } + self.worker_admin_operations[operation_id] = operation + return dict(operation) + + def create_runtime_document_operation(self, **kwargs): + self.calls.append(('document_operation_create', kwargs)) + operation_id = kwargs['operation_id'] + document = kwargs['action'].split('.')[1] + expected_identity = { + 'document': document, + 'active_config_sha256': kwargs['active_config_sha256'], + 'active_secrets_sha256': kwargs['active_secrets_sha256'], + 'candidate_config_sha256': kwargs['candidate_config_sha256'], + 'candidate_secrets_sha256': kwargs['candidate_secrets_sha256'], + 'candidate_after_sha256': kwargs['candidate_after_sha256'], + 'candidate_before_bytes': kwargs['candidate_before_bytes'], + 'candidate_after_bytes': kwargs['candidate_after_bytes'], + 'candidate_before_present': kwargs['candidate_before_present'], + } + existing = self.document_operations.get(operation_id) + if existing is not None: + if not ( + existing['actor'] == kwargs['actor'] + and existing['action'] == kwargs['action'] + and existing['target_ref'] == document + and existing['expected_identity'] == expected_identity + ): + raise RuntimeOperationIdentityConflictError('fixture identity conflict') + return {**existing, 'replayed': True} + if operation_id in self.apply_operations: + raise RuntimeOperationIdentityConflictError('fixture identity conflict') + operation = { + 'operation_id': operation_id, 'status': 'running', 'replayed': False, + 'actor': kwargs['actor'], 'action': kwargs['action'], + 'target_kind': 'runtime-document', 'target_ref': document, + 'expected_identity': expected_identity, + } + self.document_operations[operation_id] = operation + return dict(operation) + + def complete_runtime_document_operation( + self, operation_id, *, succeeded, candidate_sha256=None, written=None, + ): + self.calls.append(( + 'document_operation_complete', operation_id, succeeded, + candidate_sha256, written, + )) + if type(self).document_completion_failures: + type(self).document_completion_failures -= 1 + raise RuntimeError('fixture document completion failure') + operation = { + **self.document_operations[operation_id], + 'status': 'succeeded' if succeeded else 'failed', + } + self.document_operations[operation_id] = operation + return dict(operation) + + def create_runtime_operation(self, **kwargs): + self.calls.append(('runtime_operation_create', kwargs)) + operation_id = kwargs['operation_id'] + existing = self.apply_operations.get(operation_id) + if existing is not None: + if not ( + existing['actor'] == kwargs['actor'] + and existing['action'] == kwargs['action'] + and existing['expected_identity'] == kwargs['expected_identity'] + ): + raise RuntimeOperationIdentityConflictError('fixture identity conflict') + return {**existing, 'replayed': True} + if operation_id in self.document_operations: + raise RuntimeOperationIdentityConflictError('fixture identity conflict') + operation = { + 'operation_id': operation_id, 'status': 'requested', 'replayed': False, + 'actor': kwargs['actor'], 'action': kwargs['action'], + 'expected_identity': dict(kwargs['expected_identity']), + } + self.apply_operations[operation_id] = operation + return dict(operation) + + def create_runtime_managed_file_operation(self, **kwargs): + self.calls.append(('managed_file_operation_create', dict(kwargs))) + operation_id = kwargs['operation_id'] + expected_identity = { + key: kwargs[key] for key in ( + 'root_id', 'relative_path', 'expected_sha256', + 'proposed_sha256', 'proposed_byte_count', + ) + } + existing = self.managed_file_operations.get(operation_id) + if existing is not None: + if not ( + existing['actor'] == kwargs['actor'] + and existing['action'] == kwargs['action'] + and existing['target_ref'] == kwargs['root_id'] + and existing['expected_identity'] == expected_identity + ): + raise RuntimeOperationIdentityConflictError('fixture identity conflict') + return {**existing, 'replayed': True} + operation = { + 'operation_id': operation_id, 'status': 'running', + 'replayed': False, 'actor': kwargs['actor'], + 'action': kwargs['action'], 'target_kind': 'managed-file', + 'target_ref': kwargs['root_id'], + 'expected_identity': expected_identity, + 'resulting_identity': None, + } + self.managed_file_operations[operation_id] = operation + return dict(operation) + + def complete_runtime_managed_file_operation(self, operation_id, **kwargs): + self.calls.append(( + 'managed_file_operation_complete', operation_id, dict(kwargs), + )) + if type(self).managed_file_completion_failures: + type(self).managed_file_completion_failures -= 1 + raise RuntimeError('fixture managed file completion failure') + operation = self.managed_file_operations[operation_id] + resulting = None + if kwargs['succeeded']: + expected = operation['expected_identity'] + resulting = { + 'root_id': expected['root_id'], + 'relative_path': expected['relative_path'], + 'outcome': 'completed', + 'before_sha256': kwargs.get('before_sha256'), + 'before_byte_count': kwargs.get('before_byte_count'), + 'after_sha256': kwargs.get('after_sha256'), + 'after_byte_count': kwargs.get('after_byte_count'), + 'written': kwargs.get('written'), + } + operation = { + **operation, + 'status': 'succeeded' if kwargs['succeeded'] else 'failed', + 'resulting_identity': resulting, + } + self.managed_file_operations[operation_id] = operation + return dict(operation) + + def runtime_operation(self, operation_id): + self.calls.append(('runtime_operation', operation_id)) + operation = ( + self.document_operations.get(operation_id) + or self.apply_operations.get(operation_id) + or self.managed_file_operations.get(operation_id) + ) + return None if operation is None else dict(operation) + + def runtime_audit_events(self, *, before_event_id=None, limit=50): + self.calls.append(('runtime_audit_events', before_event_id, limit)) + return { + 'events': [dict(event) for event in type(self).audit_events], + 'next_before_event_id': type(self).audit_next_before_event_id, + } + + def create_remote_worker_user(self, user_key, cap): + self.calls.append(('create_user', user_key, cap)) + return {'user_key': user_key} + + def set_remote_worker_user_cap(self, user_key, cap): + self.calls.append(('set_cap', user_key, cap)) + return True + + def set_remote_worker_user_disabled(self, user_key, disabled): + self.calls.append(('set_disabled', user_key, disabled)) + return True + + def issue_remote_worker_device(self, user_key, device_key, token_sha256, *, rotate=False): + self.calls.append(('issue_device', user_key, device_key, token_sha256, rotate)) + return {'user_key': user_key, 'device_key': device_key} + + def set_remote_worker_device_revoked(self, device_key, revoked=True): + self.calls.append(('set_revoked', device_key, revoked)) + return True + + def admin_requeue_deferred_targets(self, queue_ids, *, max_items): + self.calls.append(('requeue', list(queue_ids), max_items)) + return len(queue_ids) + + def admin_discard_queued_source(self, source): + self.calls.append(('discard_source_queue', source)) + return 17 + + +class AdminAPITests(unittest.TestCase): + def setUp(self): + RecordingDB.calls = [] + RecordingDB.fail_queue = False + RecordingDB.source_operations = {} + RecordingDB.worker_admin_operations = {} + RecordingDB.document_operations = {} + RecordingDB.apply_operations = {} + RecordingDB.managed_file_operations = {} + RecordingDB.audit_events = [] + RecordingDB.audit_next_before_event_id = None + RecordingDB.source_identity_conflict = False + RecordingDB.completion_failures = 0 + RecordingDB.control_identity_conflict = False + RecordingDB.control_revision_conflict = False + RecordingDB.document_completion_failures = 0 + RecordingDB.managed_file_completion_failures = 0 + self.temp_dir = tempfile.TemporaryDirectory() + root = ensure_private_directory( + os.path.join(self.temp_dir.name, 'bundles'), reject_reparse=True, + ) + self.worker = FakeWorkerService(root) + self.runtime_provider = mock.Mock(return_value={ + 'snapshot_schema': 2, + 'runtime': { + 'pid': 123, 'phase': 'ACTIVE', 'manages_postgres': True, + 'start_gate_open': True, 'shutdown_requested': False, + 'runtime_failed': False, + }, + 'postgres': { + 'state': 'READY', 'ready': True, 'failures': 0, + 'safe_error_category': '', + }, + 'dashboard': { + 'status': 'running', 'desired_state': 'running', 'healthy': True, + 'pid': 456, 'safe_error_category': '', + }, + 'sources': [ + { + 'id': f'discovery-producer:{source}', 'source': source, + 'role': 'discovery-producer', 'lifecycle_state': 'waiting', + 'process_state': 'stopped', 'desired_state': 'running', + 'safe_error_category': '', + 'interval_seconds': 3600, 'restart_enabled': True, + 'restart_count': 2, + 'restart_delay_seconds': 5, 'restart_streak': 0, + 'mode': 'repeat', 'pid': None, + 'allowed_actions': [ + 'start', 'stop', 'restart', 'pause', 'resume', 'set-interval', + ], + 'last_cycle_result': { + 'status': 'completed', 'fetched_count': 4, + 'queued_new_count': 2, 'queued_updated_count': 1, + }, + 'last_successful_discovery_at': '2026-09-20T00:00:00+00:00', + 'next_scheduled_run_at': '2026-09-20T01:00:00+00:00', + } + for source in ('gitlab', 'dockerhub', 'huggingface') + ] + [ + { + 'id': source, 'source': source, 'role': source, + 'lifecycle_state': 'running', 'process_state': 'running', + 'desired_state': 'running', 'safe_error_category': '', + 'mode': 'singleton', 'pid': 789, 'interval_seconds': 0, + 'restart_enabled': True, 'restart_delay_seconds': 5, + 'restart_count': 0, 'restart_streak': 0, + 'allowed_actions': [ + 'start', 'stop', 'restart', 'pause', 'resume', + 'set-restart', 'set-restart-delay', + ], + } + for source in ('result-ingester', 'jsonl-projector', 'janitor', 'worker-api') + ] + [{ + 'id': 'keychecks', 'source': 'keychecks', 'role': 'keycheck', + 'lifecycle_state': 'waiting', 'process_state': 'stopped', + 'desired_state': 'running', 'safe_error_category': '', 'pid': None, + 'mode': 'repeat', 'interval_seconds': 3600, 'restart_enabled': True, + 'restart_delay_seconds': 5, 'restart_count': 0, 'restart_streak': 0, + 'allowed_actions': [ + 'start', 'stop', 'restart', 'pause', 'resume', 'once', + 'set-mode', 'set-interval', 'set-restart', 'set-restart-delay', + ], + }, { + 'id': 'docker-shadow', 'source': 'docker-shadow', + 'role': 'docker-shadow', 'lifecycle_state': 'stopped', + 'process_state': 'stopped', 'desired_state': 'stopped', + 'safe_error_category': '', 'pid': None, 'mode': 'manual', + 'interval_seconds': 0, 'restart_enabled': False, + 'restart_delay_seconds': 0, 'restart_count': 0, + 'restart_streak': 0, 'allowed_actions': ['start', 'stop'], + }], + 'pipeline': { + 'ingester_ready': True, 'projector_ready': True, + 'cutover_ready': True, + 'ingester_state': 'ready', 'projector_state': 'ready', + 'bundle_items': 1, 'bundle_bytes': 4096, + 'projection_items': 0, 'projection_bytes': 0, + 'keycheck_items': 0, 'keycheck_bytes': 0, + 'quarantine_items': 0, 'quarantine_bytes': 0, + }, + 'scan_workers': { + 'active': 0, 'limit': 2, 'base_active': 0, 'base_limit': 2, + 'bonus_active': 0, 'bonus_limit': 0, 'trufflehog': 0, + 'sources': {}, + }, + }) + self.source_action_provider = mock.Mock(return_value={ + 'source_action': 'restart', 'outcome': 'completed', 'source': {}, + }) + self.dashboard_action_provider = mock.Mock(return_value={ + 'dashboard_action': 'restart', 'outcome': 'completed', 'dashboard': {}, + }) + self.source_log_provider = mock.Mock(return_value={ + 'source_id': 'result-ingester', 'line_count': 2, + 'lines': ['line ', 'line two'], 'response_truncated': False, + }) + self.package_compatibility_provider = mock.Mock(return_value={ + 'profiles': [{ + 'profile_name': 'linux-x86_64', 'protocol_version': 2, + 'bundle_format_version': 2, 'platform_tag': 'linux-x86_64', + 'code_manifest_sha256': '1' * 64, + 'detector_policy_sha256': '2' * 64, + 'sources': ['dockerhub', 'gitlab', 'huggingface'], + 'capabilities': [ + {'source': 'gitlab', 'platform': 'gitlab', 'planning_kind': 'exact_git_v1'}, + {'source': 'dockerhub', 'platform': 'docker', 'planning_kind': 'docker_direct_v1'}, + {'source': 'huggingface', 'platform': 'huggingface', 'planning_kind': 'huggingface_space_v1'}, + ], + }], + 'required_capabilities': [ + {'source': 'gitlab', 'platform': 'gitlab', 'planning_kind': 'exact_git_v1'}, + {'source': 'dockerhub', 'platform': 'docker', 'planning_kind': 'docker_direct_v1'}, + {'source': 'huggingface', 'platform': 'huggingface', 'planning_kind': 'huggingface_space_v1'}, + ], + }) + self.document_state = SimpleNamespace( + active_config=SimpleNamespace(sha256='a' * 64, byte_count=10, present=True), + active_secrets=SimpleNamespace(sha256='b' * 64, byte_count=10, present=True), + candidate_config=SimpleNamespace(sha256='c' * 64, byte_count=10, present=True), + candidate_secrets=SimpleNamespace(sha256='d' * 64, byte_count=10, present=True), + ) + self.runtime_config = { + 'global': {'project_dir': '/opt/truf/app'}, + 'sources': { + 'gitlab': {'timeout': 600}, + 'dockerhub': {'timeout': 900}, + 'huggingface': {'timeout': 300}, + }, + 'supervisor': {'worker_api': { + 'sources': ['gitlab', 'dockerhub', 'huggingface'], + 'assignment_ttl_seconds': 86400, + 'assignment_ttl_seconds_by_source': {'dockerhub': 90000}, + 'bundle_body_timeout_seconds': 1800, + }}, + } + self.runtime_config_provider = mock.Mock(return_value=SimpleNamespace( + config=self.runtime_config, + )) + self.document_loader = mock.Mock(side_effect=lambda _path, document: SimpleNamespace( + document=document, source='candidate', + text=( + 'global:\n project_dir: /opt/truf/app\n' + 'sources:\n gitlab:\n timeout: 600\n' + ' dockerhub:\n timeout: 900\n' + ' huggingface:\n timeout: 300\n' + 'supervisor:\n worker_api:\n' + ' sources: [gitlab, dockerhub, huggingface]\n' + ' assignment_ttl_seconds: 86400\n' + ' assignment_ttl_seconds_by_source:\n dockerhub: 90000\n' + ' bundle_body_timeout_seconds: 1800\n' + if document == 'config' else + 'auth_pools:\n fixture:\n - name: fixture\n token: editor-secret\n' + ), + state=self.document_state, + selected=( + self.document_state.candidate_config + if document == 'config' else self.document_state.candidate_secrets + ), + )) + self.candidate_preview_provider = mock.Mock(side_effect=lambda _path, document, payload: SimpleNamespace( + document=document, state=self.document_state, + proposed=SimpleNamespace( + sha256=hashlib.sha256(payload).hexdigest(), + byte_count=len(payload), present=True, + ), + diff=( + SimpleNamespace(entries=(), truncated=False, format_only_changed=False) + if document == 'config' else + SimpleNamespace(**{ + name: value for name, value in ( + ('document_changed', True), ('semantic_changed', True), + ('pools_before', 1), ('pools_after', 1), ('pools_added', 0), + ('pools_removed', 0), ('pools_changed', 1), + ('entries_before', 1), ('entries_after', 1), + ('entries_added', 0), ('entries_removed', 0), + ('pools_reordered', 0), ('usernames_added', 0), + ('usernames_removed', 0), ('usernames_changed', 0), + ('tokens_changed', 1), + ) + }) + ), + )) + def save_candidate(_path, document, payload, **_kwargs): + proposed = SimpleNamespace( + sha256=hashlib.sha256(payload).hexdigest(), + byte_count=len(payload), present=True, + ) + if document == 'config': + self.document_state.candidate_config = proposed + else: + self.document_state.candidate_secrets = proposed + return SimpleNamespace( + document=document, proposed=proposed, written=True, + ) + + self.candidate_save_provider = mock.Mock(side_effect=save_candidate) + self.candidate_verify_provider = mock.Mock(return_value=SimpleNamespace(action='apply-config')) + self.runtime_apply_provider = mock.Mock() + self.managed_file_traversal = FakeManagedFileTraversal() + managed_limits = ManagedFileLimits(1024, 255, 16, 100, 65536, 65536) + self.managed_file_registry = ManagedFileRootRegistry(( + ManagedFileRoot( + 'exports', '/data/managed-files/private-host-path', + ManagedFilePermissions(True, True, True, True), managed_limits, + ), + ManagedFileRoot( + 'readonly', '/data/managed-files/private-readonly-path', + ManagedFilePermissions(True, True, False, False), managed_limits, + ), + )) + self.admin = AdminService( + 'postgresql://fixture', ORIGIN, MARKER, db_factory=RecordingDB, + supervisor_metadata={ + 'instance_id': 'fixture', 'token': 'supervisor-token-must-not-render', + }, + runtime_snapshot_provider=self.runtime_provider, + source_action_provider=self.source_action_provider, + dashboard_action_provider=self.dashboard_action_provider, + source_log_provider=self.source_log_provider, + package_compatibility_provider=self.package_compatibility_provider, + runtime_config_path='/data/config/runtime.yaml', + runtime_config_provider=self.runtime_config_provider, + document_loader=self.document_loader, + candidate_preview_provider=self.candidate_preview_provider, + candidate_save_provider=self.candidate_save_provider, + candidate_verify_provider=self.candidate_verify_provider, + runtime_apply_provider=self.runtime_apply_provider, + managed_file_roots=self.managed_file_registry, + ) + self.traversal_patcher = mock.patch( + 'worker_api.ManagedFileTraversal', + return_value=self.managed_file_traversal, + ) + self.traversal_patcher.start() + + def tearDown(self): + self.traversal_patcher.stop() + self.temp_dir.cleanup() + + def _client(self, enabled=True): + return TestClient(create_worker_app( + self.worker, reaper_interval_seconds=3600, + admin_service=self.admin if enabled else None, + )) + + def _headers(self, **extra): + return { + EDGE_MARKER_HEADER: MARKER, + OPERATOR_HEADER: OPERATOR, + 'Origin': ORIGIN, + **extra, + } + + def _post(self, client, path, fields): + fields = dict(fields) + if path.startswith(('/users/', '/devices/', '/queue/')): + fields.setdefault('operation_id', str(uuid.uuid4())) + return client.post( + ADMIN_PREFIX + path, + data={'csrf_token': self.admin.csrf_token, **fields}, + headers=self._headers(), + ) + + def test_managed_file_registry_pages_and_download_are_logical_and_bounded(self): + self.assertIs(self.admin.managed_file_roots, self.managed_file_registry) + registry = ManagedFileRootRegistry((object(),)) + service = AdminService( + 'postgresql://fixture', ORIGIN, MARKER, db_factory=RecordingDB, + managed_file_roots=registry, + ) + self.assertIs(service.managed_file_roots, registry) + with self.assertRaisesRegex(ValueError, 'registry is invalid'): + AdminService( + 'postgresql://fixture', ORIGIN, MARKER, db_factory=RecordingDB, + managed_file_roots={}, + ) + with self._client() as client: + index = client.get(ADMIN_PREFIX + '/files', headers=self._headers()) + listing = client.get( + ADMIN_PREFIX + '/files?root_id=exports', headers=self._headers(), + ) + nested = client.get( + ADMIN_PREFIX + '/files?root_id=exports&relative_path=nested', + headers=self._headers(), + ) + download = client.get( + ADMIN_PREFIX + + '/files/download?root_id=exports&relative_path=nested%2Freport.bin', + headers=self._headers(), + ) + for response in (index, listing, nested, download): + self.assertEqual(response.status_code, 200, response.text) + self.assertSecurityHeaders(response) + self.assertNotIn('/data/managed-files', response.text) + self.assertIn('>Files<', index.text) + self.assertIn('exports', index.text) + self.assertIn('nested/', listing.text) + self.assertIn('report.bin', nested.text) + self.assertEqual(download.content, b'\x00\xffreport') + self.assertEqual(download.headers['content-type'], 'application/octet-stream') + self.assertEqual(download.headers['content-length'], '8') + self.assertEqual( + download.headers['etag'], + '"' + hashlib.sha256(b'\x00\xffreport').hexdigest() + '"', + ) + self.assertIn("filename*=UTF-8''report.bin", download.headers['content-disposition']) + + def test_managed_file_snapshot_is_streamed_and_closed(self): + self.managed_file_traversal.stream_downloads = True + with self._client() as client: + download = client.get( + ADMIN_PREFIX + + '/files/download?root_id=exports&relative_path=nested%2Freport.bin', + headers=self._headers(), + ) + self.assertEqual(download.status_code, 200, download.text) + self.assertEqual(download.content, b'\x00\xffreport') + self.assertEqual(download.headers['content-length'], '8') + self.assertEqual(self.managed_file_traversal.last_snapshot.closed, 1) + + def test_cancelled_managed_file_download_closes_completed_snapshot(self): + snapshot = FakeManagedFileSnapshot(b'bounded snapshot') + started = threading.Event() + release = threading.Event() + + class BlockingService: + @staticmethod + def download_managed_file(*_args): + started.set() + if not release.wait(5): + raise RuntimeError('test download timed out') + return ManagedFileDownload( + ManagedFileIdentity('0' * 64, len(snapshot.content)), + snapshot=snapshot, + ) + + async def scenario(): + task = asyncio.create_task(_download_managed_file( + BlockingService(), object(), 'runtime-results', + 'scan_results.jsonl', + )) + self.assertTrue(await asyncio.to_thread(started.wait, 2)) + task.cancel() + release.set() + with self.assertRaises(asyncio.CancelledError): + await task + for _ in range(20): + if snapshot.closed: + break + await asyncio.sleep(0.01) + + asyncio.run(scenario()) + self.assertEqual(snapshot.closed, 1) + + def test_streaming_response_construction_failure_closes_snapshot(self): + snapshot = FakeManagedFileSnapshot(b'bounded snapshot') + download = ManagedFileDownload( + ManagedFileIdentity('0' * 64, len(snapshot.content)), + snapshot=snapshot, + ) + with mock.patch( + 'admin_api._ManagedFileStreamingResponse', + side_effect=RuntimeError('failed')): + with self.assertRaisesRegex(RuntimeError, 'failed'): + _managed_file_download_response( + download, 'scan_results.jsonl', + ) + self.assertEqual(snapshot.closed, 1) + + def test_streaming_send_failure_closes_snapshot(self): + snapshot = FakeManagedFileSnapshot(b'bounded snapshot') + response = _ManagedFileStreamingResponse( + snapshot, snapshot.chunks(), media_type='application/octet-stream', + ) + + async def scenario(): + async def receive(): + return {'type': 'http.disconnect'} + + async def send(_message): + raise RuntimeError('send failed') + + with self.assertRaisesRegex(RuntimeError, 'send failed'): + await response( + { + 'type': 'http', + 'method': 'GET', + 'path': '/', + 'headers': [], + 'asgi': {'version': '3.0', 'spec_version': '2.4'}, + }, + receive, + send, + ) + + asyncio.run(scenario()) + self.assertEqual(snapshot.closed, 1) + + def test_managed_file_mutations_are_typed_audited_and_content_free(self): + created_payload = b'created-binary-\x00-credential-sentinel' + replacement_payload = b'replaced-binary-\xff-credential-sentinel' + with self._client() as client: + create_id = str(uuid.uuid4()) + created = self._post(client, '/files/create', { + 'operation_id': create_id, 'root_id': 'exports', + 'relative_path': 'new.bin', + 'content_base64': __import__('base64').urlsafe_b64encode( + created_payload, + ).decode('ascii'), + }) + created_hash = hashlib.sha256(created_payload).hexdigest() + replace_id = str(uuid.uuid4()) + replaced = self._post(client, '/files/replace', { + 'operation_id': replace_id, 'root_id': 'exports', + 'relative_path': 'new.bin', 'expected_sha256': created_hash, + 'content_base64': __import__('base64').urlsafe_b64encode( + replacement_payload, + ).decode('ascii'), + }) + replacement_hash = hashlib.sha256(replacement_payload).hexdigest() + delete_id = str(uuid.uuid4()) + deleted = self._post(client, '/files/delete', { + 'operation_id': delete_id, 'root_id': 'exports', + 'relative_path': 'new.bin', 'expected_sha256': replacement_hash, + }) + + for response in (created, replaced, deleted): + self.assertEqual(response.status_code, 200, response.text) + self.assertSecurityHeaders(response) + self.assertNotIn('new.bin', self.managed_file_traversal.files) + operations = [ + call for call in RecordingDB.calls + if call[0] == 'managed_file_operation_create' + ] + self.assertEqual( + [call[1]['action'] for call in operations], + ['files.create', 'files.replace', 'files.delete'], + ) + self.assertEqual([call[1]['actor'] for call in operations], [OPERATOR] * 3) + self.assertEqual( + [call[1]['operation_id'] for call in operations], + [create_id, replace_id, delete_id], + ) + recorded = repr(RecordingDB.calls) + self.assertNotIn('credential-sentinel', recorded) + self.assertNotIn(__import__('base64').urlsafe_b64encode(created_payload).decode(), recorded) + + def test_managed_file_create_replay_recovers_after_completion_loss(self): + payload = b'durable replay payload' + encoded = __import__('base64').urlsafe_b64encode(payload).decode('ascii') + operation_id = str(uuid.uuid4()) + fields = { + 'csrf_token': self.admin.csrf_token, + 'operation_id': operation_id, 'root_id': 'exports', + 'relative_path': 'replayed.bin', 'content_base64': encoded, + } + RecordingDB.managed_file_completion_failures = 3 + with self._client() as client: + first = client.post( + ADMIN_PREFIX + '/files/create', data=fields, + headers=self._headers(), follow_redirects=False, + ) + second = client.post( + ADMIN_PREFIX + '/files/create', data=fields, + headers=self._headers(), follow_redirects=False, + ) + terminal = client.post( + ADMIN_PREFIX + '/files/create', data=fields, + headers=self._headers(), follow_redirects=False, + ) + self.assertEqual(first.status_code, 503) + self.assertEqual((second.status_code, terminal.status_code), (303, 303)) + self.assertSecurityHeaders(first) + self.assertEqual(self.managed_file_traversal.files['replayed.bin'], payload) + mutations = [ + call for call in self.managed_file_traversal.calls + if call[0] == 'create_replace' and call[2] == 'replayed.bin' + ] + self.assertEqual(len(mutations), 1) + self.assertEqual( + RecordingDB.managed_file_operations[operation_id]['status'], + 'succeeded', + ) + replay = self.admin.create_managed_file( + None, 'exports', 'replayed.bin', payload, OPERATOR, operation_id, + ) + self.assertEqual(replay['status'], 'succeeded') + + def test_managed_file_acceptance_precedes_one_concurrent_physical_mutation(self): + payload = b'concurrent managed payload' + operation_id = str(uuid.uuid4()) + original = self.managed_file_traversal.create_replace_file + other_admin = AdminService( + 'postgresql://fixture', ORIGIN, MARKER, db_factory=RecordingDB, + managed_file_roots=self.managed_file_registry, + ) + + def accepted_provider(*args, **kwargs): + self.assertEqual( + RecordingDB.managed_file_operations[operation_id]['status'], + 'running', + ) + return original(*args, **kwargs) + + barrier = threading.Barrier(3) + results = [] + + def submit(service): + barrier.wait() + results.append(service.create_managed_file( + self.managed_file_traversal, 'exports', 'concurrent.bin', + payload, OPERATOR, operation_id, + )) + + with mock.patch.object( + self.managed_file_traversal, 'create_replace_file', + side_effect=accepted_provider): + threads = [ + threading.Thread(target=submit, args=(service,)) + for service in (self.admin, other_admin) + ] + for thread in threads: + thread.start() + barrier.wait() + for thread in threads: + thread.join(timeout=5) + + self.assertTrue(all(not thread.is_alive() for thread in threads)) + self.assertEqual([result['status'] for result in results], [ + 'succeeded', 'succeeded', + ]) + mutations = [ + call for call in self.managed_file_traversal.calls + if call[0] == 'create_replace' and call[2] == 'concurrent.bin' + ] + self.assertEqual(len(mutations), 1) + self.assertEqual( + RecordingDB.managed_file_operations[operation_id]['status'], + 'succeeded', + ) + + def test_fresh_managed_file_cas_loser_is_not_promoted_to_success(self): + payload = b'external matching payload' + operation_id = str(uuid.uuid4()) + + def lose_cas(_root_id, relative_path, value, *, expected_sha256): + self.managed_file_traversal.files[relative_path] = value + raise ManagedFileAccessError('hash_conflict') + + with mock.patch.object( + self.managed_file_traversal, 'create_replace_file', + side_effect=lose_cas): + with self.assertRaises(AdminAPIError) as raised: + self.admin.create_managed_file( + self.managed_file_traversal, 'exports', 'external.bin', + payload, OPERATOR, operation_id, + ) + + self.assertEqual(raised.exception.status_code, 409) + self.assertEqual( + RecordingDB.managed_file_operations[operation_id]['status'], + 'failed', + ) + + def assertSecurityHeaders(self, response): + for name, expected in SECURITY_HEADERS.items(): + self.assertEqual(response.headers.get(name), expected) + + def _document_fields(self, text, operation_id=None): + return { + 'document_text': text, + 'operation_id': operation_id or str(uuid.uuid4()), + 'expected_active_config_sha256': 'a' * 64, + 'expected_active_secrets_sha256': 'b' * 64, + 'expected_candidate_config_sha256': 'c' * 64, + 'expected_candidate_secrets_sha256': 'd' * 64, + } + + def test_runtime_document_pages_preview_save_and_apply_are_exact_and_no_store(self): + with self._client() as client: + config = client.get(ADMIN_PREFIX + '/config', headers=self._headers()) + secrets = client.get(ADMIN_PREFIX + '/secrets', headers=self._headers()) + self.assertEqual(config.status_code, 200) + self.assertEqual(secrets.status_code, 200) + self.assertSecurityHeaders(config) + self.assertSecurityHeaders(secrets) + self.assertIn('Config YAML', config.text) + self.assertNotIn('editor-secret', config.text) + self.assertIn('editor-secret', secrets.text) + self.assertNotIn('', + 'action': 'apply-secrets', + 'target_kind': 'runtime-deployment', + 'target_ref': 'secrets', + 'status': 'failed', + 'safe_category': 'startup_failed', + 'safe_detail': 'startup_failed', + 'expected_revision': None, + 'resulting_revision': None, + 'expected_identity': expected_identity, + 'resulting_identity': None, + 'agent_state': 'failed', + 'agent_result_sha256': 'e' * 64, + 'requested_at': '2026-09-20T00:00:00+00:00', + 'started_at': '2026-09-20T00:00:01+00:00', + 'completed_at': '2026-09-20T00:00:02+00:00', + 'agent_reconciled_at': '2026-09-20T00:00:02+00:00', + 'updated_at': '2026-09-20T00:00:02+00:00', + } + RecordingDB.audit_events = [{ + 'id': 41, + 'operation_id': operation_id, + 'actor': '', + 'action': 'apply-secrets', + 'target_kind': 'runtime-deployment', + 'target_ref': 'secrets', + 'result': 'accepted', + 'safe_category': None, + 'before_identity': expected_identity, + 'after_identity': None, + 'before_bytes': None, + 'after_bytes': None, + 'previous_event_id': None, + 'previous_event_sha256': None, + 'event_sha256': 'f' * 64, + 'created_at': '2026-09-20T00:00:00+00:00', + }] + RecordingDB.audit_next_before_event_id = 41 + + with self._client() as client: + index = client.get( + ADMIN_PREFIX + '/operations', headers=self._headers(), + ) + status = client.get( + ADMIN_PREFIX + f'/operations/{operation_id}', + headers=self._headers(), + ) + audit = client.get( + ADMIN_PREFIX + '/audit', headers=self._headers(), + ) + older = client.get( + ADMIN_PREFIX + '/audit?before=41', headers=self._headers(), + ) + + for response in (index, status, audit, older): + self.assertEqual(response.status_code, 200, response.text) + self.assertSecurityHeaders(response) + self.assertNotIn('editor-secret', response.text) + self.assertNotIn('supervisor-token-must-not-render', response.text) + self.assertNotIn(']*>).*?()', r'\1\2', secrets, + flags=re.DOTALL | re.IGNORECASE, + ) + self.assertNotIn('editor-secret', outside_textarea) + self.assertIn('
    Workers / Dispatch', response.text) + self.assertIn('Dispatch and drain', response.text) + self.assertIn('Explicit pause', response.text) + self.assertIn('Effective pause', response.text) + self.assertIn('Drain blockers', response.text) + self.assertIn('value="7"', response.text) + for action in ( + './dispatch/pause', './dispatch/resume', + './dispatch/drain/start', './dispatch/drain/cancel', + ): + self.assertIn(f'action="{action}"', response.text) + self.assertEqual(response.text.count('name="operation_id"'), 14) + self.assertIn('Authenticated status, terminal reports, uploads', response.text) + self.assertIn('Protocol-1 workers are completion-only', response.text) + self.assertIn('linux-x86_64', response.text) + self.assertIn('exact_git_v1', response.text) + self.assertIn('docker_direct_v1', response.text) + self.assertIn('huggingface_space_v1', response.text) + self.assertIn('fixture-user', response.text) + self.assertIn('fixture-device', response.text) + self.assertIn('Accepted', response.text) + self.assertIn('Ingestion / projection', response.text) + snapshot_calls = [call for call in RecordingDB.calls if call[0] == 'snapshot'] + self.assertEqual(len(snapshot_calls), 1) + self.assertEqual(snapshot_calls[0][1], 25) + self.assertFalse(any( + call[0] in {'diagnostic_groups', 'duration_metrics'} + for call in RecordingDB.calls + )) + self.assertIn('value="assignments" selected', response.text) + self.assertIn('Not loaded. Select Assignments + diagnostics', response.text) + self.assertRegex(snapshot_calls[0][2]['since'], r'^\d{4}-\d{2}-\d{2}T') + self.assertIn(('drain_progress',), RecordingDB.calls) + self.package_compatibility_provider.assert_called_once_with() + self.assertNotIn('supervisor-token-must-not-render', response.text) + + def test_worker_observability_list_separates_final_models_and_keeps_occurrences(self): + with self._client() as client: + response = client.get( + ADMIN_PREFIX + '?details=all', headers=self._headers(), + ) + + self.assertEqual(response.status_code, 200, response.text) + for label in ( + 'Assignment outcome', 'Scan outcome', 'Diagnostics', + 'Phase / progress', 'Deadlines', 'Slot / cap', 'Package identity', + 'Ingestion / projection', + ): + self.assertIn(label, response.text) + self.assertIn('active progress', response.text) + self.assertIn('Observability scope:', response.text) + self.assertEqual(response.text.count('