Initial server source import
This commit is contained in:
+237
@@ -0,0 +1,237 @@
|
||||
# Deny by default. File-only exceptions keep credentials and caches out of the context.
|
||||
**
|
||||
!app/app.py
|
||||
!app/audit_github_tokens.py
|
||||
!app/capacity_model.py
|
||||
!app/child_bootstrap.py
|
||||
!app/console_runner.py
|
||||
!app/container_import.py
|
||||
!app/container_import_config.py
|
||||
!app/container_projection_recovery.py
|
||||
!app/container_runtime.py
|
||||
!app/dashboard.py
|
||||
!app/db_backend.py
|
||||
!app/admin_api.py
|
||||
!app/docker_depth_experiment.py
|
||||
!app/docker_depth_operator.py
|
||||
!app/docker_depth_report.py
|
||||
!app/docker_shadow.py
|
||||
!app/host_agent_client.py
|
||||
!app/host_agent_apply.py
|
||||
!app/host_agent_lifecycle.py
|
||||
!app/host_agent_protocol.py
|
||||
!app/host_agent_reconcile.py
|
||||
!app/host_agent_runtime.py
|
||||
!app/host_agent_server.py
|
||||
!app/host_agent_state.py
|
||||
!app/janitor.py
|
||||
!app/jsonl_projector.py
|
||||
!app/keycheck_accounting_smoke.py
|
||||
!app/keycheck_candidates.py
|
||||
!app/keycheck_runner.py
|
||||
!app/keycheckers/__init__.py
|
||||
!app/keycheckers/keycheck_common.py
|
||||
!app/keycheckers/provider_resolution.py
|
||||
!app/keycheckers/anthropic/anthropicKeycheck.py
|
||||
!app/keycheckers/aws/awsKeycheck.py
|
||||
!app/keycheckers/azure/azureKeycheck.py
|
||||
!app/keycheckers/deepseek/deepseekKeycheck.py
|
||||
!app/keycheckers/dockerhub/dockerhubKeycheck.py
|
||||
!app/keycheckers/gcp/gcpKeycheck.py
|
||||
!app/keycheckers/gemini/geminiKeycheck.py
|
||||
!app/keycheckers/github/githubKeycheck.py
|
||||
!app/keycheckers/gitlab/gitlabKeycheck.py
|
||||
!app/keycheckers/groq/groqKeycheck.py
|
||||
!app/keycheckers/huggingface/huggingfaceKeycheck.py
|
||||
!app/keycheckers/kimi/kimiKeycheck.py
|
||||
!app/keycheckers/openai/Keycheck.py
|
||||
!app/keycheckers/openrouter/OpenrouterKeycheck.py
|
||||
!app/keycheckers/provider_resolver/providerResolverKeycheck.py
|
||||
!app/keycheckers/qwen/qwenKeycheck.py
|
||||
!app/keycheckers/replicate/replicateKeycheck.py
|
||||
!app/keycheckers/xai/xaiKeycheck.py
|
||||
!app/keycheckers/zai/zaiKeycheck.py
|
||||
!app/lifecycle_authority.py
|
||||
!app/managed_files.py
|
||||
!app/migrate_layout.py
|
||||
!app/migrate_observability_db.py
|
||||
!app/migrate_runtime_safety.py
|
||||
!app/optimize_dashboard_db.py
|
||||
!app/owned_process.py
|
||||
!app/paths.py
|
||||
!app/postgres_runtime.py
|
||||
!app/process_identity.py
|
||||
!app/query_policy.py
|
||||
!app/result_bundle.py
|
||||
!app/result_ingester.py
|
||||
!app/result_spool.py
|
||||
!app/runtime_bootstrap.py
|
||||
!app/runtime_document.py
|
||||
!app/runtime_document_io.py
|
||||
!app/runtime_security.py
|
||||
!app/scan_manager.py
|
||||
!app/scanner.py
|
||||
!app/scan_execution.py
|
||||
!app/scanner_db.py
|
||||
!app/scanner_error_policy_smoke.py
|
||||
!app/supervisor.py
|
||||
!app/supervisor_instance.py
|
||||
!app/sync_alive_github_tokens.py
|
||||
!app/target_identity.py
|
||||
!app/ui_components.py
|
||||
!app/worker_api.py
|
||||
!app/worker_assignment.py
|
||||
!app/worker_assignment_runner.py
|
||||
!app/worker_cli.py
|
||||
!app/worker_contracts.py
|
||||
!app/worker_local_state.py
|
||||
!app/worker_package.py
|
||||
!app/worker_package_builder.py
|
||||
!app/worker_supervisor.py
|
||||
!app/remote_worker_bootstrap.py
|
||||
!app/remote_worker_client.py
|
||||
!app/requirements.txt
|
||||
!app/requirements-keycheckers.txt
|
||||
!app/config.linux.yaml
|
||||
!app/trufflehog-custom-detectors.yaml
|
||||
!app/.streamlit/config.toml
|
||||
!start_runtime.ps1
|
||||
!start_core_runtime.ps1
|
||||
!stop_runtime.ps1
|
||||
!Dockerfile
|
||||
!.dockerignore
|
||||
!compose.yaml
|
||||
!compose.edge.yaml
|
||||
!compose.shared-host.yaml
|
||||
!docs/remote-worker-quickstart-ru.md
|
||||
!docs/remote-worker-cheatsheet-windows-ru.md
|
||||
!docs/remote-worker-cheatsheet-linux-ru.md
|
||||
!docs/remote-worker-cheatsheet-docker-ru.md
|
||||
!deploy/edge/Dockerfile
|
||||
!deploy/edge/Caddyfile
|
||||
!deploy/edge/Caddyfile.shared-host
|
||||
!deploy/edge/host-caddy-shared.caddy
|
||||
!deploy/edge/entrypoint.sh
|
||||
!deploy/edge/README.md
|
||||
!deploy/edge/admin-denylist.caddy
|
||||
!deploy/edge/automatic-tls.caddy
|
||||
!deploy/fail2ban/truf_caddy_admin_denylist.py
|
||||
!deploy/fail2ban/Dockerfile.edge-e2e
|
||||
!deploy/fail2ban/edge_e2e_docker_shim.py
|
||||
!deploy/fail2ban/fail2ban.d-edge-e2e.local
|
||||
!deploy/fail2ban/filter.d-truf-admin-auth.conf
|
||||
!deploy/fail2ban/jail.d-truf-admin-auth.local
|
||||
!deploy/fail2ban/action.d-truf-caddy-admin-denylist.conf
|
||||
!deploy/fail2ban/fail2ban.d-truf-persistence.local
|
||||
!deploy/host-agent/truf_host_agent.py
|
||||
!deploy/host-agent/truf_host_agent_install.py
|
||||
!deploy/host-agent/truf-host-agent.conf
|
||||
!deploy/host-agent/truf-host-agent.service
|
||||
!deploy/host-agent/truf-host-agent.socket
|
||||
!deploy/systemd/truf-caddy-admin-denylist-expire.service
|
||||
!deploy/systemd/truf-caddy-admin-denylist-expire.timer
|
||||
!docker/requirements.in
|
||||
!docker/requirements.lock
|
||||
!docker/Dockerfile.edge-e2e
|
||||
!docker/requirements-test.in
|
||||
!docker/requirements-test.lock
|
||||
!docker/requirements-worker.in
|
||||
!docker/requirements-worker.lock
|
||||
!docker/worker-package-pins.json
|
||||
!docker/verify.py
|
||||
!docker/build-dependencies/requirements.in
|
||||
!docker/build-dependencies/requirements.lock
|
||||
!tests/container_unit.py
|
||||
!tests/container_e2e.py
|
||||
!tests/container_import_stop_e2e.py
|
||||
!tests/container_projection_recovery_e2e.py
|
||||
!tests/owned_process_helper.py
|
||||
!tests/parity_helpers.py
|
||||
!tests/packaged_worker_e2e_server.py
|
||||
!tests/edge_e2e_backend.py
|
||||
!tests/edge_e2e_client.py
|
||||
!tests/test_synthetic_llm_pipeline.py
|
||||
!tests/test_docker_foundation.py
|
||||
!tests/test_db_backend_safety.py
|
||||
!tests/test_owned_process.py
|
||||
!tests/test_owned_process_linux.py
|
||||
!tests/test_owned_process_boundary.py
|
||||
!tests/test_runtime_bootstrap_authority.py
|
||||
!tests/test_runtime_document.py
|
||||
!tests/test_runtime_document_io.py
|
||||
!tests/test_managed_files.py
|
||||
!tests/test_operations_schema.py
|
||||
!tests/test_operations_control.py
|
||||
!tests/test_host_agent_protocol.py
|
||||
!tests/test_host_agent_linux.py
|
||||
!tests/test_host_agent_apply.py
|
||||
!tests/test_host_agent_deploy.py
|
||||
!tests/test_host_agent_lifecycle.py
|
||||
!tests/test_host_agent_reconcile.py
|
||||
!tests/test_host_agent_runtime.py
|
||||
!tests/test_host_agent_state.py
|
||||
!tests/test_supervisor_foreground_shutdown.py
|
||||
!tests/test_supervisor_startup_rollback.py
|
||||
!tests/test_supervisor_managed_postgres_gate.py
|
||||
!tests/test_observer_only_coordinated_shutdown.py
|
||||
!tests/test_supervisor_safety.py
|
||||
!tests/test_discovery_producer_supervisor.py
|
||||
!tests/test_distributed_core_profile.py
|
||||
!tests/test_discovery_only_cycle.py
|
||||
!tests/test_discovery_request_budgets.py
|
||||
!tests/test_operations_service.py
|
||||
!tests/test_postgres_runtime.py
|
||||
!tests/test_container_security.py
|
||||
!tests/test_runtime_security.py
|
||||
!tests/test_postgres_empty_initialization.py
|
||||
!tests/test_container_migration_paths.py
|
||||
!tests/test_container_provider_portability.py
|
||||
!tests/test_container_e2e_helpers.py
|
||||
!tests/test_container_runtime.py
|
||||
!tests/test_container_import.py
|
||||
!tests/test_container_import_config.py
|
||||
!tests/test_container_projection_recovery.py
|
||||
!tests/test_result_bundle_v2.py
|
||||
!tests/test_pipeline_cutover_invariants.py
|
||||
!tests/test_custom_provider_detector_compatibility.py
|
||||
!tests/test_scan_execution.py
|
||||
!tests/test_worker_api.py
|
||||
!tests/test_worker_api_runtime.py
|
||||
!tests/test_worker_assignment.py
|
||||
!tests/test_worker_cli.py
|
||||
!tests/test_worker_contracts.py
|
||||
!tests/test_worker_local_state.py
|
||||
!tests/test_worker_package.py
|
||||
!tests/test_worker_supervisor.py
|
||||
!tests/test_multisource_execution_snapshot.py
|
||||
!tests/test_remote_direct_credentials.py
|
||||
!tests/test_docker_staging_bounds.py
|
||||
!tests/test_remote_worker_db.py
|
||||
!tests/test_admin_api.py
|
||||
!tests/test_edge_deployment.py
|
||||
!tests/fixtures/worker_tls_cert.pem
|
||||
!tests/fixtures/worker_tls_key.pem
|
||||
|
||||
# Defense in depth if the allowlist is expanded later.
|
||||
**/.env*
|
||||
**/secrets*
|
||||
**/credentials*
|
||||
**/*service-account*
|
||||
**/__pycache__/
|
||||
**/.venv/
|
||||
**/venv/
|
||||
**/node_modules/
|
||||
**/*.py[cod]
|
||||
**/*.db*
|
||||
**/*.sqlite*
|
||||
**/*.jsonl*
|
||||
**/*.log*
|
||||
.git/
|
||||
.opencode/
|
||||
runtime/
|
||||
tmp/
|
||||
state/
|
||||
logs/
|
||||
data/
|
||||
docker/imports/
|
||||
truf-cluster-authority-*/
|
||||
@@ -0,0 +1,14 @@
|
||||
TRUF_POSTGRES_DB=truf
|
||||
TRUF_POSTGRES_USER=truf
|
||||
TRUF_POSTGRES_PASSWORD=change-me-long-random-password
|
||||
TRUF_POSTGRES_PORT=5432
|
||||
|
||||
# Optional tuning overrides.
|
||||
TRUF_POSTGRES_MAX_CONNECTIONS=100
|
||||
TRUF_POSTGRES_SHARED_BUFFERS=512MB
|
||||
TRUF_POSTGRES_EFFECTIVE_CACHE_SIZE=2GB
|
||||
TRUF_POSTGRES_CHECKPOINT_TIMEOUT=15min
|
||||
TRUF_POSTGRES_LOG_MIN_DURATION_STATEMENT=2000
|
||||
|
||||
# App DSN template. Keep the real password out of config snapshots/logs.
|
||||
SCANNER_DB_URL=postgresql://truf:change-me-long-random-password@127.0.0.1:5432/truf
|
||||
@@ -0,0 +1,12 @@
|
||||
* text=auto
|
||||
*.py text eol=lf
|
||||
*.sh text eol=lf
|
||||
*.yaml text eol=lf
|
||||
*.yml text eol=lf
|
||||
*.toml text eol=lf
|
||||
*.md text eol=lf
|
||||
*.txt text eol=lf
|
||||
Dockerfile text eol=lf
|
||||
.dockerignore text eol=lf
|
||||
.gitignore text eol=lf
|
||||
.gitattributes text eol=lf
|
||||
+44
@@ -0,0 +1,44 @@
|
||||
# Local credentials and provider input pools.
|
||||
.env*
|
||||
!.env*.example
|
||||
secrets.yaml*
|
||||
secrets.*.yaml
|
||||
!secrets.example.yaml
|
||||
credentials*.json
|
||||
*-service-account*.json
|
||||
/app/keycheckers/**/*.txt
|
||||
|
||||
# Runtime data, findings, queues, logs, and cluster authority.
|
||||
/runtime/
|
||||
/tmp/
|
||||
/state/
|
||||
/logs/
|
||||
/docker/test-results/
|
||||
/docker/imports/
|
||||
/data/
|
||||
/truf-cluster-authority-*/
|
||||
/checked_*.txt
|
||||
/todo_*.txt
|
||||
/runner_state.json
|
||||
*.db
|
||||
*.db-*
|
||||
*.sqlite
|
||||
*.sqlite-*
|
||||
*.sqlite3
|
||||
*.sqlite3-*
|
||||
*.jsonl
|
||||
*.jsonl.*
|
||||
*.log
|
||||
*.log.*
|
||||
|
||||
# Reproducible dependencies and generated files.
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
.pytest_cache/
|
||||
.venv/
|
||||
venv/
|
||||
node_modules/
|
||||
build/
|
||||
dist/
|
||||
.coverage
|
||||
htmlcov/
|
||||
@@ -0,0 +1,15 @@
|
||||
# Engineering Decisions
|
||||
|
||||
## Provider Source Execution
|
||||
|
||||
- Keep source adapters minimal. The server validates assignment shape, canonical target identity, and the immutable identity required by the protocol.
|
||||
- Git planning may bind an exact commit. Docker planning may resolve a mutable tag to an immutable digest. These are identity operations, not provider-access proofs.
|
||||
- The worker is the final authority for real provider access. It reports success, a permanent target failure, or a retryable provider failure; the server settles or retries from that result.
|
||||
- Discovery credentials are not assignment fields unless a separately approved capability explicitly defines credential delivery.
|
||||
- Do not add per-target server preflight requests, durable public-access proofs, proof TTL/freshness columns, access-evidence migrations, broad child-environment credential scrubbing, credential sandboxes, or post-hoc redaction pipelines by default.
|
||||
- Before adding any such defensive or security-specific mechanism, stop and obtain explicit user approval. Record the approved behavior in an OpenSpec requirement and task before implementation.
|
||||
- Do not treat existing defensive code as precedent for duplicating the same machinery for another source.
|
||||
- The worker machine's ambient environment belongs to its operator. Assignment code must not silently rewrite HOME, XDG, Git, Docker, or provider environments merely to enforce a nominally tokenless assignment.
|
||||
- Prefer direct, bounded provider-error classification over preventive infrastructure: authentication/access/not-found failures are permanent when target-scoped; rate limits, network failures, and provider 5xx responses are retryable.
|
||||
|
||||
These rules apply to future source adapters and to changes in `worker_assignment.py`, `scan_execution.py`, `remote_worker_client.py`, `scanner.py`, `scanner_db.py`, and discovery producers.
|
||||
@@ -0,0 +1,418 @@
|
||||
# Docker Development Copy
|
||||
|
||||
Status: runnable Linux container deployment with a passed fresh offline E2E.
|
||||
Production source/provider parity and original-data migration are not proven.
|
||||
Original Windows installation: `D:\truf`. Development copy: `D:\truf-docker`.
|
||||
Only the development copy is changed; the original `D:\truf` and storage on `S:`
|
||||
remain untouched. No original STOP or START was performed for this migration.
|
||||
Docker/Compose installation in WSL was user-approved and has occurred. Earlier
|
||||
statements about unavailable Docker, no installations, and no images describe
|
||||
historical stages, not the current deployment.
|
||||
|
||||
The current acceptance evidence is from the isolated Docker runs on 2026-09-15.
|
||||
The historical original-runtime lineage proof was not rerun.
|
||||
|
||||
## Copy And Local History
|
||||
|
||||
These are historical source-copy/baseline records, not a fresh Git inventory.
|
||||
|
||||
- The user changed the initial full-snapshot request to a source-only copy.
|
||||
- The retained source selection was 295 files, approximately 6.82 MiB before Git and migration edits.
|
||||
- Application Python, tests, configuration, documentation, OpenSpec artifacts, and project scripts were retained.
|
||||
- Databases, WAL/SHM files, results, queues, logs, runtime state, provider input pools, real secrets, native Windows bundles, and dependency caches were omitted or removed from the copy.
|
||||
- The partial external-data copy `D:\truf-docker-data` was removed. Original external storage on `S:` was not modified.
|
||||
- SHA-256 comparison verified the selected source files before migration edits. The clone's `.gitignore` was strengthened before staging.
|
||||
- Initial local commit: `1b3c7fc`, `chore: snapshot source for Docker migration`, 292 tracked files on `main`.
|
||||
- Three copied OpenCode package metadata files remain ignored by their original nested ignore policy.
|
||||
- No remote, push, Git configuration change, or second commit was made. Migration edits remain separate from the baseline.
|
||||
|
||||
## Safety Boundary
|
||||
|
||||
Copied root PowerShell tools and direct canonical Python runtime/control CLI
|
||||
launches retain their staging refusals. The supported container path is the
|
||||
image's `tini -> python3 -u -I -S -B app/container_runtime.py` entrypoint, which
|
||||
uses authenticated bootstrap dispatch rather than removing host safeguards.
|
||||
Do not use copied Windows maintenance/import scripts to operate this deployment.
|
||||
|
||||
The original `app/config.yaml` is preserved in the initial Git commit and in the
|
||||
unchanged original installation. In this working tree it was renamed to
|
||||
`app/config.linux.yaml`. The initial slice changed filesystem/deployment paths
|
||||
only; subsequent container contracts fix private storage and control paths.
|
||||
There is no automatically selected `app/config.yaml`; the container entrypoint
|
||||
defaults explicitly to `/opt/truf/app/config.linux.yaml`. This production profile
|
||||
is distinct from the verifier's deliberately narrowed `/data/config/e2e.yaml`.
|
||||
|
||||
The image requires read-only application storage, UID/GID 10001 for runtime
|
||||
commands, a private native Linux named volume at `/data`, and private tmpfs at
|
||||
`/run/truf`. Provisioning alone runs as root, with only CHOWN, DAC_OVERRIDE, and
|
||||
FOWNER added to the dropped capability set. Private application/data directories
|
||||
are mode 0700 and files 0600. System executables remain root-owned and immutable;
|
||||
they must not be chowned to the application user to satisfy private-file checks.
|
||||
|
||||
PostgreSQL 16 runs under the supervisor in the same container, with generated
|
||||
private credentials, loopback connectivity, and exact cluster identity binding.
|
||||
External PostgreSQL authority is not implemented; changing a DSN or starting a
|
||||
separate PostgreSQL service does not implement that backend. Never reuse physical
|
||||
Windows PGDATA on Linux. Any original-data migration requires separately
|
||||
approved logical export/import and validation. No host data bind mount, Docker
|
||||
socket, privileged mode, or Docker-in-Docker is required for DockerHub scanning.
|
||||
|
||||
## Implemented Foundation
|
||||
|
||||
- Path defaults derive from this checkout instead of `D:\truf` or the current working directory.
|
||||
- Generated filesystem templates use portable separators. Explicit YAML still overrides environment defaults.
|
||||
- POSIX path resolution rejects Windows drive, UNC, device, and backslash syntax rather than silently joining it to a Linux directory.
|
||||
- Generic path defaults retain the runtime-relative result-bundle directory and PostgreSQL's `runtime/postgres/data` suffix; the container profile explicitly selects the separate `/data` paths listed below.
|
||||
- The Windows TruffleHog fallback is checked only on Windows. Managed PostgreSQL DSN precedence is unchanged.
|
||||
- `.gitignore` excludes credentials and consumables. `.dockerignore` denies everything except reviewed, explicitly named build/source files, including nested provider modules.
|
||||
- `.gitattributes` specifies LF for Linux-facing source/configuration files without changing global Git settings.
|
||||
- Offline tests cover path behavior, the Linux profile, context allowlist, and refusal of copied control entrypoints.
|
||||
|
||||
## Implemented Lifecycle Changes
|
||||
|
||||
The lifecycle changes in `app/owned_process.py` and `app/supervisor.py` implement
|
||||
Linux ownership and coordinated foreground shutdown. Native Linux ownership
|
||||
tests and the fresh container E2E now pass within their selected scope; this is
|
||||
not acceptance of every production source/provider or unsafe failure scenario.
|
||||
|
||||
- Linux startup now reports complete kernel-derived identities, verifies the host and payload sessions, and requires a verified child subreaper. Unreadable or already-exited executables fail startup rather than receiving a PID-only identity.
|
||||
- Nested observers have independent sessions. Cleanup signals the still-pinned payload group before reaping its leader, then drains adopted children. Completed adopted observers are also reaped while the root remains alive.
|
||||
- Payload signal status is reproduced only after cleanup, including SIGKILL as a real negative subprocess return code. Administrative stop remains a distinct nonzero result.
|
||||
- Linux failed-start cleanup retains its observer through repeated interruptions until exit is confirmed; it does not hard-kill that observer on a timer.
|
||||
- A configured owner sends an explicit stop byte over its retained pipe, so another inherited writer cannot suppress cancellation. Forked proxy finalization only detaches its local descriptor, without signalling the owner's job or taking an inherited mutex.
|
||||
- Startup status descriptors have a single atomic owner. An unclaimed reader is cancelled before waiting for host cleanup, avoiding double close, reused-FD reads, and a blocked status writer during interrupted thread startup.
|
||||
- Windows keeps its Job Object containment and resource checks. Its isolated host now uses the base interpreter rather than a virtual-environment redirecting launcher, so the retained process and acknowledged identity have the same PID.
|
||||
- The supervisor's POSIX SIGTERM callback only latches a shutdown request. Locked checkpoints close admission before activation or further starts, and preserve coordinated child-before-PostgreSQL teardown.
|
||||
- Metadata publication and shutdown interruptions retain closed gates and unsafe authority in `FAILED_HOLD`. Partial activation uses full coordinated cleanup; failures remain nonzero even if a later cleanup retry succeeds.
|
||||
- Foreground and background shutdown publish the same instance-bound receipt after fallible control/log cleanup. The waiting authenticated stopper or locked stale reconciliation removes metadata; foreground no longer deletes it before the stopper can verify completion.
|
||||
- PostgreSQL stop and close share one remaining timeout. This is not a bound on the complete shutdown sequence: unsafe authority is retained indefinitely rather than released when a timer expires.
|
||||
|
||||
Additional container work is implemented, rather than still pending:
|
||||
|
||||
- Source-start rollback retains uncertain owners and closes admission; locked stale-metadata reconciliation uses exact identity rather than PID alone.
|
||||
- Durable state stays on `/data`; recoverable control metadata is under `/run/truf/control` and does not survive recreation.
|
||||
- Linux manifests distinguish private application files from root-owned native binaries; the obsolete Windows OpenRouter PowerShell dependency is not the Linux provider entrypoint.
|
||||
- `Dockerfile` packages Python 3.12.14, PostgreSQL 16.15, Git, tini, and hash-locked Python dependencies in the production server. TruffleHog 3.97.4 is pinned only in worker and test targets; the remote-only production server does not contain it.
|
||||
- Fresh provisioning, empty-cluster initialization, 27 schema migrations, final cutover, and initialization/identity markers are implemented. Partial initialization fails closed and is not automatically repaired or adopted.
|
||||
- Compose applies a read-only root filesystem, dropped capabilities, no-new-privileges, 2 CPUs, 6 GiB memory, 512 PIDs, 256 MiB shared memory, and bounded log rotation. Tini forwards SIGTERM to the foreground runtime/supervisor, not indiscriminately to its process group.
|
||||
- Dependency-aware health checks require authenticated ACTIVE control, READY PostgreSQL, required workers and durable pipeline leases, schema/cutover validity, the expected PG16 data directory, and writable nonfull persistent storage.
|
||||
|
||||
Containment covers managed nested `OwnedProcess` trees, not arbitrary session
|
||||
escapes or an observer independently killed by SIGKILL/OOM. The application
|
||||
retains unsafe authority indefinitely in `FAILED_HOLD`, but Compose's stop grace
|
||||
is only 10 minutes. Docker can then force termination; that is unsafe shutdown,
|
||||
not successful coordinated cleanup. Two clean E2E stops do not prove safe
|
||||
termination of `FAILED_HOLD` at that deadline.
|
||||
|
||||
## Current Linux Paths
|
||||
|
||||
These are the implemented image, volume, and tmpfs contracts.
|
||||
|
||||
| Purpose | Path |
|
||||
| --- | --- |
|
||||
| Image application root | `/opt/truf` |
|
||||
| Application code | `/opt/truf/app` |
|
||||
| New Linux runtime state | `/data/runtime-linux` |
|
||||
| Separately initialized Linux PostgreSQL cluster | `/data/postgres-linux` |
|
||||
| Durable result bundles | `/data/scanner-result-bundles` |
|
||||
| Scanner scratch | `/data/scanner-work` |
|
||||
| Private imported provider credentials | `/data/config/secrets.yaml` |
|
||||
| Generated PostgreSQL password | `/data/postgres-password` |
|
||||
| Ephemeral control and authority state | `/run/truf/control`, `/run/truf/authority` |
|
||||
| Worker/test TruffleHog executable | `/usr/local/bin/trufflehog` (absent from the production server) |
|
||||
|
||||
The original physical cluster was PostgreSQL 16, but it was not retained in this
|
||||
source-only copy. Do not mount Windows PostgreSQL data into a Linux server.
|
||||
Initialize isolated development data; any later production-data migration needs
|
||||
a separately approved logical export/import and validation procedure.
|
||||
|
||||
## Development Linux/WSL Procedure
|
||||
|
||||
This volume-only procedure is retained for isolated development and migration verification;
|
||||
it is not the production edge installation procedure. Production operators must use
|
||||
`deploy/edge/README.md`, including the fixed host-agent installer, active documents under
|
||||
`/etc/truf/runtime`, immutable package manifests under `/etc/truf/worker-packages`, the
|
||||
host-agent socket, and the combined base plus edge Compose invocation.
|
||||
|
||||
The following are operator commands, not commands executed by this documentation
|
||||
update. Use a Linux shell or WSL with Python 3, Git, Docker's Linux daemon, and
|
||||
the Compose plugin (`docker compose`). Verified host versions were Docker 29.8.0
|
||||
and Compose 5.5.1. Keep Docker-managed named volumes on native Linux storage, not NTFS
|
||||
or an original-runtime directory. Build steps need package/download network
|
||||
access; the offline test runs below do not. Do not run the full legacy suite.
|
||||
|
||||
### Build
|
||||
|
||||
From the development checkout, define an explicitly scoped Compose helper. The
|
||||
example uses the passwordless sudo Docker access used by the recorded verifier;
|
||||
omit `sudo -n` if your account already has direct daemon access. Do not change
|
||||
daemon permissions or install/reset WSL as part of these instructions.
|
||||
|
||||
```sh
|
||||
cd /mnt/d/truf-docker
|
||||
dc() { sudo -n docker compose --project-name truf-docker --project-directory "$PWD" --env-file /dev/null --file compose.yaml "$@"; }
|
||||
dc --profile test build runtime test
|
||||
```
|
||||
|
||||
On native Linux, substitute the development checkout path for `/mnt/d/truf-docker`.
|
||||
This creates `truf-local:runtime` and `truf-local:test`. All lifecycle commands
|
||||
below must retain this project name and checkout so they use the same volume.
|
||||
The explicit env file avoids implicitly loading a checkout `.env`; do not supply
|
||||
unreviewed Docker/Compose environment overrides or proxy credentials.
|
||||
|
||||
### Provision And Initialize
|
||||
|
||||
Use these one-off commands only while the runtime is stopped. `--no-deps` avoids
|
||||
implicitly starting other services. Provision is network-disabled, generates a
|
||||
new private database password, and seeds an empty provider-secret mapping. A
|
||||
valid already-provisioned layout is checked without regenerating credentials;
|
||||
nonempty or partially provisioned layouts are refused.
|
||||
|
||||
```sh
|
||||
dc run --rm --no-deps --pull never -T provision
|
||||
```
|
||||
|
||||
For an authorized production-profile run, import an existing private YAML mapping
|
||||
from stdin. Replace the placeholder filename with an approved Linux-side secret
|
||||
file, not a file in the original installation. Do not put secret values in command
|
||||
arguments, the image, the checkout, or this document. This is not a Compose secret
|
||||
mount: the locked, atomic import writes `/data/config/secrets.yaml` with private
|
||||
ownership/mode and refuses an active runtime or PostgreSQL PID file. Imports can
|
||||
also be repeated after confirmed shutdown for credential rotation.
|
||||
|
||||
```sh
|
||||
dc run --rm --no-deps --pull never -T runtime import-secrets < /absolute/private/provider-secrets.yaml
|
||||
dc run --rm --no-deps --pull never -T runtime initialize
|
||||
```
|
||||
|
||||
Initialization creates an independent PG16 cluster, starts maintenance mode,
|
||||
applies base/schema migrations and final cutover, confirms PostgreSQL stopped,
|
||||
then publishes the initialization marker. A matching initialized volume is not
|
||||
reinitialized. On partial initialization, stop and inspect offline; do not delete
|
||||
markers, change identity, or rerun repair scripts to force admission.
|
||||
|
||||
### Start, Status, And Health
|
||||
|
||||
Starting `compose.yaml` uses the production configuration and can launch enabled
|
||||
sources and provider workers with real network access. It is NOT the offline E2E
|
||||
procedure and must only be used with separately authorized targets/credentials.
|
||||
`run` initializes if needed before execing the noninteractive autostart supervisor;
|
||||
the explicit initialization step above makes that first-install phase visible.
|
||||
|
||||
```sh
|
||||
dc up --detach --no-deps --no-build --pull never runtime
|
||||
dc ps runtime
|
||||
dc exec -T runtime /usr/local/bin/python3 -I -S -B /opt/truf/app/container_runtime.py status
|
||||
dc exec -T runtime /usr/local/bin/python3 -I -S -B /opt/truf/app/container_runtime.py health
|
||||
```
|
||||
|
||||
`status` and `health` both call the same readiness function and return JSON on
|
||||
success, nonzero on failure. They are not general stopped-runtime inventory
|
||||
commands. Run them with `exec` in the existing runtime, not `compose run`, because
|
||||
a new container has a different control tmpfs. During initial startup, readiness
|
||||
can fail until activation and worker leases complete; Compose checks every 30
|
||||
seconds with a 15-second timeout, 240-second start period, and three retries.
|
||||
Readiness is not proof that every provider works or that production load is safe.
|
||||
|
||||
### Stop And Recreate
|
||||
|
||||
```sh
|
||||
dc stop --timeout 600 runtime
|
||||
cid=$(dc ps --all --quiet runtime)
|
||||
sudo -n docker container inspect --format 'status={{.State.Status}} exit={{.State.ExitCode}} oom={{.State.OOMKilled}} restarts={{.RestartCount}}' "$cid"
|
||||
```
|
||||
|
||||
Require an exited container, exit code 0, and OOM false; investigate unexpected
|
||||
restarts. A successful `compose stop` invocation alone is not graceful-exit proof.
|
||||
Do not shorten the timeout, force-kill, remove the data volume, or treat a
|
||||
10-minute forced termination as safe. Runtime restart policy is `on-failure:3`;
|
||||
it is not a substitute for investigating uncertain ownership or partial state.
|
||||
|
||||
To replace a confirmed-stopped container while retaining its existing data:
|
||||
|
||||
```sh
|
||||
dc up --detach --no-deps --no-build --pull never --force-recreate runtime
|
||||
dc exec -T runtime /usr/local/bin/python3 -I -S -B /opt/truf/app/container_runtime.py health
|
||||
```
|
||||
|
||||
Allow readiness to complete again. Never use `down --volumes` on data you intend
|
||||
to retain. The old `docker-compose.postgres.yml` is a noncanonical manual-recovery
|
||||
fixture, not a deployment or an external-authority implementation.
|
||||
|
||||
### Offline Verification
|
||||
|
||||
Use the reviewed container selection, not unrestricted pytest discovery:
|
||||
|
||||
```sh
|
||||
dc --profile test run --rm --no-deps --pull never -T test
|
||||
python3 -I -S -B docker/test_verify.py
|
||||
python3 -I -S -B docker/verify.py
|
||||
```
|
||||
|
||||
The selected container suite runs without `/data` or provider credentials, with
|
||||
network disabled and temporary fixtures in `/tmp`; native local child processes
|
||||
and loopback control sockets are intentional. Only this unit-test service allows
|
||||
execution from its 512 MiB `/tmp` for temporary venv and askpass fixtures. Both
|
||||
the production and E2E services use the same 128 MiB `noexec,nosuid,nodev` `/tmp`.
|
||||
`docker/test_verify.py` is a separate six-test stdlib regression suite, passing
|
||||
on both Windows and WSL Linux;
|
||||
its total is not silently added to the selected container-suite count.
|
||||
|
||||
`docker/verify.py` requires already-built local runtime/test images and an already
|
||||
Git-ignored `docker/test-results/latest.json`. It performs no builds or pulls and
|
||||
does not edit ignore files. Run the verifier itself as the normal Linux user; it
|
||||
tries direct Docker access, then `sudo -n docker`. Git is used for the read-only
|
||||
evidence ignore guard, not for mutations.
|
||||
|
||||
The verifier invokes only `compose.e2e.yaml`, generates a unique `truf-e2e-*`
|
||||
project, pins the local images by ID, and uses fresh private `data` and `tools`
|
||||
named volumes. All services have network disabled, proxy settings cleared, no
|
||||
host data binds or published ports, and no automatic restarts. It preserves the
|
||||
production runtime image/entrypoint but prepares a narrowed offline configuration:
|
||||
a local Git fixture and real TruffleHog with verification disabled, then the real
|
||||
scan/bundle/ingestion/projection pipeline and OpenAI worker. ONLY that worker's
|
||||
HTTP transport is substituted with a synthetic 401 response; no live provider
|
||||
request is made. Production source selection/provider parity is not tested by
|
||||
this profile. Do not manually merge it into production Compose or rerun prepare
|
||||
after recreation.
|
||||
|
||||
The verifier checks healthy activation, pipeline lineage, one synthetic HTTP
|
||||
request, graceful stop, recreation on the same volume, unchanged persisted
|
||||
counts/hashes, no duplicates and no second HTTP request, health, and a second
|
||||
graceful stop. Its aggregate check budget is 3600 seconds, each health wait at
|
||||
most 240 seconds, each stop 600 seconds, and failure handling has a separate
|
||||
720-second budget. Nonzero, forced, or OOM exits cannot pass.
|
||||
|
||||
On success it removes only its ownership-verified containers and volumes; use
|
||||
`python3 -I -S -B docker/verify.py --keep` instead to retain stopped test artifacts.
|
||||
Failures retain artifacts after a guarded stop attempt, possibly with running
|
||||
containers if ownership or stopping cannot be proven. Record the printed project
|
||||
name for investigation; do not use broad prune/down/kill commands. Evidence is
|
||||
written to `docker/test-results/latest.json` as counts, hashes, image IDs, statuses,
|
||||
and durations without raw command logs or secret values.
|
||||
|
||||
## Current Verified Evidence
|
||||
|
||||
The fresh recorded E2E in `docker/test-results/latest.json` has `status.result`
|
||||
and `status.checks` both `passed`, `status.cleanup` equal to `removed`, and zero
|
||||
owned containers, networks, or volumes remaining. The final run on 2026-09-15
|
||||
took 54.374 seconds and repeated the earlier successful fresh-volume run.
|
||||
|
||||
- Fresh PG16 initialization and all 27 migrations completed with identity/cutover checks; both authenticated health checks passed.
|
||||
- Real local Git and native TruffleHog produced one finding, one scan, one result bundle/reservation, and one queue attempt through ingestion and projection, with zero pipeline quarantine/errors.
|
||||
- The real OpenAI worker used only the offline synthetic 401 transport: one first HTTP request, one linked keycheck result/current state, two projection jobs, and three projection appends.
|
||||
- Recreation used a new container on the same volume without rerunning prepare. Persisted artifact IDs, migration/cutover hashes, SQL summary, output bytes and projection ledger hashes matched; repeated keycheck made zero HTTP requests and introduced no duplicate rows/appends.
|
||||
- Both graceful stops recorded container exit 0, OOM false, and zero restarts. Both separate read-only stopped-volume checks confirmed private PG storage, initialization, and absence of `postmaster.pid`.
|
||||
- Shutdown receipt publication is inferred ONLY from the foreground zero-exit contract. The receipt itself was not read after stop because `/run/truf` tmpfs had disappeared; evidence labels this `exit_contract_only_tmpfs_removed`.
|
||||
- Final selected container regression suite: 578 passed, seven Windows-only tests skipped on Linux, in 22.25 seconds. This includes all six fresh-volume regressions. Separately, `docker/test_verify.py` passed all six tests on both Windows and WSL Linux. These are scope-specific counts, not counts stored in the E2E JSON.
|
||||
- All 157 Python files under `app`, `tests`, and `docker` parsed successfully. `git diff --check` passed; existing PowerShell LF/CRLF notices were not whitespace failures. No changes were staged or committed and no Git remote was added.
|
||||
|
||||
Recorded image IDs (not a promise that mutable local tags still point to them):
|
||||
|
||||
- Runtime: `sha256:92502a2581ebcabe79ddc28744dabcfb3000b99b7c0c5262c4c09aff9dc0267b`.
|
||||
- Test: `sha256:4ce11325728ba3e58e6643c1c8e800f317179d5c7c50e7e80568b58f62dbdfd0`.
|
||||
|
||||
## Remaining Limitations
|
||||
|
||||
`DOCKER_READINESS_AUDIT.md` records the original Windows audit. Its historical
|
||||
line references and pending foundation tasks are not a current implementation
|
||||
checklist. The remaining acceptance boundaries are:
|
||||
|
||||
1. `FAILED_HOLD` can outlive Docker's 10-minute grace and be terminated unsafely. No general crash/OOM/forced-stop recovery proof is claimed.
|
||||
2. Production source/provider parity, live providers, real credentials, broad discovery, sustained load, throughput, and resource sizing have not been validated by the offline E2E.
|
||||
3. There is no arm64 build/runtime proof, even though TruffleHog has an arm64 checksum entry.
|
||||
4. External PostgreSQL authority is not implemented. Only a fresh supervisor-owned native Linux PG16 cluster is supported here.
|
||||
5. There is no original Windows database migration, original-data equivalence, or production cutover proof. The historical read-only lineage record below is not such a migration proof and was not executed by this update.
|
||||
|
||||
## Historical Verification (Superseded Status)
|
||||
|
||||
The sections below preserve earlier scoped verification records. Their test
|
||||
counts, file inventories, no-install/no-image statements, staging status, and
|
||||
then-open container checks apply only to those stages and are superseded by the
|
||||
current evidence above. At the earlier lifecycle stage Docker CLI was unavailable
|
||||
and no image build/container/database migration had yet been performed. That is
|
||||
no longer the current state. No historical original-runtime operation below is
|
||||
claimed as an execution by this documentation update or by the fresh offline E2E.
|
||||
|
||||
### WSL Ownership Verification
|
||||
|
||||
Verified on 2026-09-14 in `Ubuntu-24.04`, WSL version 2, Linux kernel
|
||||
`6.18.33.2-microsoft-standard-WSL2`, CPython 3.12.3, as unprivileged UID 1000:
|
||||
|
||||
- The initial successful startup took approximately 44 seconds, exceeding the earlier 10-second probe limit. WSL reported automatic NAT-to-VirtioProxy networking fallback. No network setting was changed, and the tests needed no network access.
|
||||
- All nine previously skipped `LinuxOwnedProcessIntegrationTests` first passed on the real kernel. They were then included in the expanded run: 38 tests passed, zero failures/errors/skips, in 3.682 seconds excluding WSL startup.
|
||||
- The expanded selection also covers mocked failure paths, ordinary output/timeout behavior, static process-safety checks, isolated-host startup, an offline temporary virtual environment, and synthetic credential filtering.
|
||||
- The first expanded run exposed a test-fixture issue: Ubuntu's standard-library `sitecustomize.py` shadowed the virtual environment's fixture in the positive control. The test now puts its own module first through a temporary `PYTHONPATH`, supplied to both control and isolated-host scenarios. Both startup-hook markers must appear in the control and remain absent for the isolated host. No application code was changed for this correction.
|
||||
- The existing Windows selection was rerun after that test-only fix: 159 passed, nine Linux-only skips. Those nine skips are covered by the successful WSL run, not left untested.
|
||||
- The Linux runner used only the standard-library `unittest`, `python3 -I -S -B`, an empty inherited environment via `env -i`, and explicit safe locale/path/temp settings. No packages were installed and no pytest plugins were loaded. It asserted the effective `tempfile` directory before collection; a 90-second test watchdog was separate from the longer WSL startup allowance.
|
||||
- Code was read from `/mnt/d/truf-docker`; `HOME`, `TMPDIR`, `TEMP`, and `TMP` were confined to `/mnt/c/Users/PRO100~1/AppData/Local/Temp/opencode`. Fixtures used mounted Windows storage, not a new native Linux data volume. This does not validate Linux storage ownership, permissions, or container mounts.
|
||||
- No canonical runtime CLI, original database, provider, Docker service, or copied recovery Compose fixture was launched. Staging refusals remain unchanged.
|
||||
|
||||
Exact expanded `unittest` selection, with the clone's `tests` directory explicitly
|
||||
added to the isolated runner's module search path:
|
||||
|
||||
- `test_owned_process_linux`
|
||||
- `test_owned_process.OwnedProcessTests`
|
||||
- `test_owned_process.StaticProcessSafetyTests`
|
||||
- `test_owned_process_boundary.OwnedProcessHostBoundaryTests`
|
||||
- `test_owned_process_boundary.CredentialBoundaryTests.test_host_environment_strips_mixed_case_database_credentials_only`
|
||||
|
||||
### Earlier Windows Lifecycle Verification
|
||||
|
||||
Verified on 2026-09-14 with Windows CPython 3.12.3 after the lifecycle changes:
|
||||
|
||||
- 159 selected tests passed; nine native Linux tests were skipped because they require a real Linux kernel and `/proc`. The passing selection includes native Windows Job/virtual-environment checks, mocked Linux kernel operations, mocked supervisor lifecycle scenarios, and the previous foundation checks.
|
||||
- Regressions cover interrupted status-reader startup, cancellation before host reaping, duplicate control writers, foreign-proxy detachment, sticky failure results, receipt verification, shutdown ordering, and graceful TERM between activation/start checkpoints.
|
||||
- The nine Linux-only fixtures cover live kernel identities and sessions, transitive cleanup on normal exit and actual pipe EOF, explicit stop with another writer retained, forked-proxy finalization, live adopted-child reaping, SIGKILL/SIGTERM status, and rejected startup after nested children are ready. Test signals use pidfds bound to the acknowledged payload identity.
|
||||
- All 145 application/test Python files parsed successfully. The source-only artifact check found 301 files excluding `.git`, with no credential pools, databases, results, runtime/cache directories, or links.
|
||||
- `git diff --check` passed. Seven existing PowerShell LF/CRLF warnings are not whitespace failures. The index and remote list remain empty; the only commit is still `1b3c7fc`.
|
||||
- Independent scoped static reviews were followed by regression fixes and reruns. The last ownership review found no remaining concrete issue in the reviewed corrections; this is not a Linux conformance result.
|
||||
- At that stage, a bounded WSL probe timed out without output; the subsequent WSL investigation and successful tests are recorded above. No WSL reset/install, image build, application launch, database access, or live provider test was performed by that lifecycle continuation.
|
||||
|
||||
The combined selection was limited to these modules/node IDs:
|
||||
|
||||
- `tests/test_owned_process_linux.py`
|
||||
- `tests/test_owned_process.py`
|
||||
- `tests/test_owned_process_boundary.py::OwnedProcessHostBoundaryTests`
|
||||
- `tests/test_owned_process_boundary.py::CredentialBoundaryTests::test_host_environment_strips_mixed_case_database_credentials_only`
|
||||
- `tests/test_supervisor_foreground_shutdown.py`
|
||||
- `tests/test_supervisor_managed_postgres_gate.py`
|
||||
- `tests/test_observer_only_coordinated_shutdown.py`
|
||||
- `tests/test_docker_foundation.py`
|
||||
- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_default_data_directory_is_unchanged`
|
||||
- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_external_data_directory_expands_without_moving_runtime_assets`
|
||||
- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_external_data_identity_mismatch_remains_fail_closed`
|
||||
- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_external_data_identity_verifies_only_when_exactly_bound`
|
||||
- `tests/test_supervisor_safety.py::ManagedConfigurationAuthorityTests::test_managed_dsn_overrides_config_database_urls`
|
||||
|
||||
The test child used `python -X utf8 -B`, pytest `-q --tb=short -rs`,
|
||||
`-p no:cacheprovider -o addopts= --confcutdir=tests`, disabled plugin autoload,
|
||||
and empty `PYTHONPATH`, `PYTEST_ADDOPTS`, and `PYTEST_PLUGINS`. Runtime/DSN
|
||||
overrides were removed only from that child's environment. All of `TMPDIR`,
|
||||
`TEMP`, and `TMP` pointed at the approved temporary work area, and the runner
|
||||
asserted the actual `tempfile` directory before collecting tests.
|
||||
|
||||
### Previously Recorded Verification
|
||||
|
||||
The following earlier results are retained as historical records. They are not
|
||||
new original-runtime operations performed by this lifecycle continuation.
|
||||
|
||||
Recorded on 2026-09-14 with Windows CPython 3.12.3:
|
||||
|
||||
- 19 targeted offline tests passed: all 14 foundation tests, four existing PostgreSQL path/identity tests, and the existing managed-DSN precedence test.
|
||||
- The existing external-data path fixture now uses portable separators too; intentional Windows-path rejection remains separately covered.
|
||||
- Python CLI tests verify the complete literal refusal AST before invoking an isolated interpreter. PowerShell safety checks parse scripts without executing them, so a broken refusal cannot make the test control the host.
|
||||
- All 143 Python application/test files parsed successfully; no syntax errors.
|
||||
- Parsed YAML comparison against the initial Git commit found exactly 31 changed deployment-path fields; all other settings were identical.
|
||||
- The final source tree contained 299 files, approximately 6.85 MiB excluding `.git`, with no real credentials/pools, databases, results, runtime/cache directories, or links found by the artifact check.
|
||||
- The original supervisor and PostgreSQL were still running; no copy writers remained. There were no staged changes or Git remotes, and the only commit remained the initial baseline.
|
||||
- A read-only original-runtime lineage proof joined one completed queue reservation through its bundle, scan, finding, keycheck candidate/result, projection jobs, appends, and stream generations. Exact event/hash relationships held, and both scan projections plus the keycheck projection matched their recorded byte offsets, lengths, record counts, and SHA-256 digests; no target or credential value was emitted.
|
||||
|
||||
Tests ran with bytecode writes, pytest plugin autoload, and pytest cache disabled;
|
||||
application/database environment overrides were removed from the test process.
|
||||
Temporary test files were confined to the approved temporary work area. The
|
||||
PowerShell review and Windows POSIX-path emulation do not prove Linux container
|
||||
behavior, and the context allowlist check is not an actual Docker build.
|
||||
|
||||
Do not run the entire existing test suite against this machine: it includes
|
||||
native-process, socket, database, and live integration scenarios.
|
||||
@@ -0,0 +1,364 @@
|
||||
# Docker Readiness Audit
|
||||
|
||||
Date: 2026-09-14. Scope: the application and runtime launch chain in `D:\truf`.
|
||||
|
||||
This is an inspection report, not an implementation. Application code, configuration, secrets, databases and runtime data were not changed. The target assumed here is a Linux container. Windows containers, target CPU architecture, deployment host and resource budget have not been specified.
|
||||
|
||||
## Verdict
|
||||
|
||||
**The project is not ready to containerize unchanged. Adding a Dockerfile around the PowerShell launchers is insufficient.** There are both packaging gaps and concrete defects in the POSIX process/security paths. PostgreSQL is also part of a local process-ownership protocol, not just a replaceable connection URL.
|
||||
|
||||
The first deployment should retain **one supervisor and its authenticated children in one container, with one runtime replica**. PostgreSQL can initially remain locally managed in that container, or become a separate service after an explicit external-database authority mode is implemented. Neither option is currently a configuration-only change.
|
||||
|
||||
No deployment Dockerfile or `.dockerignore` was found in the inspected application/root. `docker-compose.postgres.yml` is explicitly a **noncanonical manual recovery fixture**, not the production runtime definition. DockerHub scanning in the application is unrelated to deployment packaging.
|
||||
|
||||
Priority definitions:
|
||||
|
||||
- **P0 / B01-B14:** resolve before a working, safely restartable Linux deployment. Some items need packaging/provisioning rather than application changes.
|
||||
- **P1 / R01-R07:** resolve before unattended operation with persistent data.
|
||||
- **Conditional / C01-C07:** required only for the stated feature or deployment choice. These are not all prerequisites for a headless, single-runtime deployment.
|
||||
|
||||
## Runtime Map
|
||||
|
||||
| Component | Actual entry points and role |
|
||||
| --- | --- |
|
||||
| Canonical startup | `app/runtime_bootstrap.py`, `app/child_bootstrap.py`; isolated interpreter startup and authenticated imports |
|
||||
| Lifecycle | `app/supervisor.py`, `app/supervisor_instance.py`, `app/lifecycle_authority.py`; admission, ownership, control, manifests and shutdown |
|
||||
| Subprocess containment | `app/owned_process.py`, `app/process_identity.py` |
|
||||
| Scanning | `app/console_runner.py`, `app/scanner.py`; native TruffleHog and Git |
|
||||
| Key checking | `app/keycheck_runner.py`, `app/keycheckers/`; Python provider processes, HTTP and AWS SDK |
|
||||
| Database | `app/postgres_runtime.py`, `app/db_backend.py`, `app/scanner_db.py`; managed PostgreSQL and normalized runtime schema |
|
||||
| Result pipeline | `app/result_bundle.py`, `app/result_ingester.py`, `app/jsonl_projector.py`, `app/janitor.py` |
|
||||
| Optional UI | `app/dashboard.py`; supervised read-only Streamlit dashboard |
|
||||
| Retired entry points | `app/app.py:1-13` and `app/scan_manager.py:8`; do not use as the container application |
|
||||
|
||||
## P0: Deployment Blockers
|
||||
|
||||
### B01. Replace Windows Path Assumptions, Not Just Environment Variables
|
||||
|
||||
**Evidence:** `app/config.yaml:8-12,30-37,70-89,109-120,140-150`; `app/paths.py:7-8,17-18,46-59,70-110`.
|
||||
|
||||
The configuration contains `D:\truf`, `S:\postgres-data`, `S:\scanner-result-bundles`, `S:\scanner-work`, `C:\Tools\trufflehog.exe` and backslash-based derived paths. POSIX treats a Windows drive path as relative and a backslash as an ordinary filename character. `os.path.normpath()` does not translate them.
|
||||
|
||||
YAML `root_dir` wins over `SCANNER_ROOT_DIR`/`SCANNER_PROJECT_ROOT`; YAML `trufflehog_path` wins over `TRUFFLEHOG_PATH`. Several defaults remain Windows-specific even if those YAML values are removed. The managed PostgreSQL DSN has its own precedence and must remain consistent with the selected authority mode.
|
||||
|
||||
**Isolated reproduction:** executing the actual path functions with POSIX path semantics, a config location under `/opt/truf/app`, and Linux environment overrides produced `/opt/truf/app/D:\truf` for the root and `/opt/truf/app/C:\Tools\trufflehog.exe` for TruffleHog. With empty YAML, the root became portable but the default log path still became `/srv/truf/runtime\logs`. No application was imported or started.
|
||||
|
||||
**Correction:** supply a complete Linux config/profile and make path defaults platform-aware with joins or portable separators. Cover global paths, supervisor instance/status/lock paths, policy assets, caches and maintenance paths. Define and document precedence rather than assuming environment variables override YAML. Preserve the working Windows profile; do not replace backslashes indiscriminately in arbitrary settings or stored data.
|
||||
|
||||
### B02. Use the Canonical Foreground Entrypoint
|
||||
|
||||
**Evidence:** `start_runtime.ps1:18`; `start_core_runtime.ps1:23`; `app/runtime_bootstrap.py:24-31,115-129`; `app/supervisor.py:4468-4481,4878-4882,5000-5001,5277-5278`.
|
||||
|
||||
The launchers use `--background`, return after starting a child, and therefore have the wrong lifetime for a container entrypoint. Interactive mode is not automatically disabled without a TTY; EOF can end its loop. Noninteractive mode without autostart is not sufficient either.
|
||||
|
||||
**Correction:** use exec-style startup of the canonical supervisor, without daemonization, with explicit `--non-interactive --autostart`. Keep `-I -S -B` and the bootstrap entrypoint binding. Use `--no-dashboard` for the initial headless deployment. Do not launch workers directly or substitute `streamlit run app.py`.
|
||||
|
||||
Illustrative command contract for the embedded-PostgreSQL option, **only after the other fixes and offline provisioning**; not executed during this audit:
|
||||
|
||||
```text
|
||||
python3 -u -I -S -B /opt/truf/app/runtime_bootstrap.py supervisor -- --runtime-bootstrap-entrypoint /opt/truf/app/supervisor.py --config /opt/truf/app/config.yaml --non-interactive --autostart --no-dashboard --no-clear --with-postgres
|
||||
```
|
||||
|
||||
Select sources explicitly. `start_core_runtime.ps1` and `app/config.linux.yaml` use the exact distributed producer set `gitlab,dockerhub,huggingface`; operational workers such as `keychecks` are configured independently. `--once` is not a guarantee that the whole supervised pipeline is a terminating batch job (`app/supervisor.py:974`).
|
||||
|
||||
### B03. Fix the POSIX OwnedProcess Identity Handshake
|
||||
|
||||
**Evidence:** `app/owned_process.py:1006-1014`; `app/scanner.py:11479-11495,1825-1842`.
|
||||
|
||||
The POSIX handshake returns payload/host identities without `creation_time`; the payload executable is taken from command text. The scanner passes that identity into its ownership marker, which requires `pid`, `creation_time` and `executable`. Access to the missing field can raise `KeyError` on the real native-scanner launch path.
|
||||
|
||||
**Correction:** return complete, canonical, verified process identities before startup acknowledgement. Preserve the stdlib-only containment bootstrap and exact identity checks; a PID alone is insufficient. Add a real POSIX handshake-to-marker test, not a mock that supplies the missing field.
|
||||
|
||||
### B04. Make Nested POSIX Process Containment Actually Contain the Tree
|
||||
|
||||
**Evidence:** `app/owned_process.py:442-445,917-921,989-993`; `app/supervisor.py:1180-1189`; `app/keycheck_runner.py:2226-2233`; `app/scanner.py:11479`.
|
||||
|
||||
An outer payload owns process group A. Its inner containment host inherits A, but the inner provider/native payload creates session/group B. Killing A can kill the inner host while leaving B running. The dead host can no longer reliably process its parent pipe and stop B. This affects source restarts and dependency-loss handling inside a still-running container, not only final container termination.
|
||||
|
||||
**Correction:** implement nested containment with verified tree termination, such as isolated observer hosts with cascading stop/acknowledgement, or an appropriately designed cgroup mechanism. Do not replace identity-based ownership with name-based process killing. An init/reaper alone does not fix this defect.
|
||||
|
||||
### B05. Integrate Container Signals and a Real Shutdown Budget
|
||||
|
||||
**Evidence:** `app/supervisor.py:2338-2361,3447,3551,5293-5305`; `app/postgres_runtime.py:824`; `app/config.yaml:168-170,185`.
|
||||
|
||||
The supervisor handles `KeyboardInterrupt` but does not register a SIGTERM handler. Docker's normal stop signal therefore is not wired into coordinated shutdown; PID 1 also has special Linux signal semantics. The shutdown path includes admission closure, pipeline draining, sequential child stops and PostgreSQL shutdown. A short container grace period can interrupt that protocol. Locally managed PostgreSQL is daemonized through `pg_ctl`, so reaping also needs attention.
|
||||
|
||||
**Correction:** connect SIGTERM to the existing STOPPING/shutdown event flow, provide an init/reaper, and forward signals to the supervisor rather than indiscriminately to its whole process tree. Size `stop_grace_period` from the total measured shutdown deadline, not only the PostgreSQL timeout. Current configuration includes a 120-second PostgreSQL shutdown timeout and a 180-second background-shutdown timeout; neither proves that a particular total container grace is sufficient.
|
||||
|
||||
An explicit SIGINT stop signal could be an interim tested workaround, not a substitute for the complete TERM/PID1/nested-process fix. An unconfirmed stop intentionally enters `FAILED_HOLD`; no finite grace period can guarantee a clean outcome there. Preserve authority, expose failure and require an escalation procedure instead of releasing locks optimistically.
|
||||
|
||||
### B06. Make Restart Safe Across Container PID Reuse
|
||||
|
||||
**Evidence:** `app/supervisor.py:5067`; related identity and instance handling in `app/supervisor_instance.py` and `app/process_identity.py`.
|
||||
|
||||
Persisted, otherwise valid instance metadata combined with a reused PID can block a new foreground supervisor. PID reuse is particularly predictable across fresh container PID namespaces.
|
||||
|
||||
**Correction:** reconcile stale instance state using exact identity under the correct authority lock, and/or place runtime-instance/control metadata in deliberately ephemeral storage. Keep persistent business data separate from per-instance PID, control, shutdown-receipt and session state. Never delete a lock or metadata merely because it is old. Verify that the previous runtime cannot still own the database before recovery.
|
||||
|
||||
### B07. Provision Non-Root Ownership, Private Paths and a Usable Lock Root
|
||||
|
||||
**Evidence:** `app/runtime_security.py:391-400,651-685,842-889,989-998`; `app/postgres_runtime.py:258-271`.
|
||||
|
||||
POSIX private-file policy requires ownership by the effective UID and no group/other permissions. Lifecycle preflight is read-only and requires configured directories to exist already, including application/root paths, runtime data, bundle subdirectories and PostgreSQL paths. A fresh named volume or a default root-owned Docker secret does not automatically meet this contract. The PostgreSQL process inherits the runtime UID; Linux PostgreSQL cannot run as root.
|
||||
|
||||
The authority lock root is hardcoded to `/var/lock/truf`. `/var/lock` is a symlink on many Linux images and conflicts with the no-symlink policy; creating it as an unprivileged user is another problem.
|
||||
|
||||
**Correction:** choose a stable non-root UID/GID, provision all required paths and file ownership before normal startup, and use a real prepared private authority directory, for example `/run/truf/authority`. Make its location explicit instead of depending on a distribution's `/var/lock` layout. Data/config modes normally need owner-only access. Verify the actual behavior of named volumes, secret mounts and Docker Desktop mounts; do not solve this with `chmod 777` or privileged mode.
|
||||
|
||||
Provisioning may need a separate controlled initialization step. The steady-state application should not require root, and startup should retain its fail-closed validation rather than silently repairing arbitrary mounted data.
|
||||
|
||||
### B08. Separate Executable Trust Policy From Data-File Hardening
|
||||
|
||||
**Evidence:** `app/runtime_security.py:651-685`; `app/lifecycle_authority.py:432-443,636-639`; `app/migrate_runtime_safety.py:2502-2503,2508-2521`.
|
||||
|
||||
The current hardener sets every POSIX file to `0600`, removing native executable bits. Conversely, ordinary system-installed `0755`, root-owned Git/TruffleHog executables do not satisfy the current exact-private manifest policy. The offline hardener also hardens each file's parent and runtime trees; pointing it at a system binary can attempt to harden a shared system directory.
|
||||
|
||||
**Correction:** define and verify executable permissions separately, preserving `x` and a trusted owner. Choose deliberately between private executable copies and an explicit immutable-system-binary trust policy. Account for the complete Git installation and PostgreSQL libraries/helpers, not just one binary. Do not run the existing recursive hardener over `/usr/bin` or blindly apply `0600` to a native runtime. Test hardening idempotency without breaking execution.
|
||||
|
||||
### B09. Package the Code Authority and Isolated Import Layout Correctly
|
||||
|
||||
**Evidence:** `app/lifecycle_authority.py:21,38-70,378-443`; `app/runtime_bootstrap.py:47-83`; `app/child_bootstrap.py:177-195`.
|
||||
|
||||
The manifest unconditionally includes `../runtime/check-openrouter-keys.ps1`, `../start_runtime.ps1` and `../stop_runtime.ps1`. Excluding all PowerShell or all `runtime/` content before changing this contract can break authentication even on Linux. Application-tree symlinks and cached application bytecode are rejected. Creating an ordinary virtualenv under the application tree can introduce both.
|
||||
|
||||
**Correction:** make the external authority-file set OS-aware, or retain these inert first-party files at the required relative locations until that change is made. Ship all required application modules and detector/policy assets, without application `.pyc`/`__pycache__` artifacts or symlinked application paths. Keep the dependency environment outside the inspected `app/` tree. Install dependencies for the exact interpreter used by isolated bootstrap; arbitrary `PYTHONPATH` and user-site packages are not a substitute.
|
||||
|
||||
Use immutable releases with a full controlled restart. Live edits to mounted code/config are incompatible with manifest drift detection (`app/supervisor.py:3045`).
|
||||
|
||||
### B10. Decide and Implement the PostgreSQL Authority Topology
|
||||
|
||||
**Evidence:** `app/postgres_runtime.py:112-130,249-255,642-671,1335`; `app/db_backend.py:42-101`; `docker-compose.postgres.yml:1-20`.
|
||||
|
||||
Current managed mode expects loopback, local PostgreSQL executables, a local data directory, bound cluster identity and an inspectable local postmaster process. Changing the DSN host to a Compose service name does not implement external PostgreSQL support. The URL parser also rejects query parameters, so appending `?sslmode=...` is not currently a supported TLS configuration route.
|
||||
|
||||
| Option | Required work |
|
||||
| --- | --- |
|
||||
| Locally managed PostgreSQL in the runtime container | Preserve a shared PID/network namespace and lifecycle owner. Supply Linux PostgreSQL executables at the currently fixed `runtime/postgres/pgsql/bin` layout, or make the binary paths configurable. Keep PGDATA separate from binaries and use the compatible non-root UID. Include init/reaping and coordinated database shutdown. |
|
||||
| Separate PostgreSQL container/service | Add an explicit external authority/backend mode across `postgres_runtime.py`, DSN validation, supervisor lifecycle and readiness. It must not require local postmaster PIDs/data paths/binaries or attempt local start/stop. Preserve authenticated endpoint/cluster identity checks, fencing and schema readiness. Define explicit TLS settings if required. |
|
||||
|
||||
Simply disabling authority, process or endpoint checks is not an acceptable implementation. A pre-existing PostgreSQL instance can be observed without being owned; do not assume the supervisor will stop it. `maintenance-start` returns after startup and is not a PostgreSQL container service entrypoint (`app/postgres_runtime.py:1463`).
|
||||
|
||||
Keep `docker-compose.postgres.yml` separate: it is profile-gated, uses `restart: no`, Windows bind paths and a deliberately noncanonical endpoint. Its `postgres:16` image is not evidence of the version required by the authoritative cluster. Do not silently promote this recovery database to production authority.
|
||||
|
||||
### B11. Add an Explicit Offline Provisioning and Data-Migration Procedure
|
||||
|
||||
**Evidence:** `app/postgres_runtime.py:321-356,400`; `app/scanner_db.py:238,5145-5177,20520-20530`; `app/migrate_runtime_safety.py:2971-3008`.
|
||||
|
||||
`bootstrap_cluster_identity()` does not run `initdb`; it expects an existing cluster and executable set. It rejects supervisor metadata, `postmaster.pid` and a listening endpoint. The identity binds paths, binaries and cluster identity, so an old Windows identity file must not be reused as a Linux authority binding.
|
||||
|
||||
Workers also require the runtime safety schema and a valid `postgres-normalized-v2-authority` final-cutover marker with evidence. A fresh PostgreSQL service reporting `pg_isready` is not an application-ready database.
|
||||
|
||||
**Correction:** distinguish two offline phases. First, initialize/restore and bind the embedded cluster identity while the target PostgreSQL server is stopped. Then run database schema/cutover work with PostgreSQL available in controlled maintenance mode but all normal sources/pipeline workers stopped. Use `--initialize-base` for a genuinely fresh installation, not as a substitute for understanding an existing dataset. Complete the applicable normalization, projection reconciliation and final-cutover checks before admitting workers.
|
||||
|
||||
Determine the real source/target PostgreSQL versions before transfer. Prefer a planned logical dump/restore for the Windows-to-Linux move unless another backup method is explicitly validated as compatible; do not assume copying Windows PGDATA works. Back up and transfer matching result bundles/projection data as well. Review legacy absolute Windows locators and use the applicable migration/reconciliation paths, not blanket database string replacement.
|
||||
|
||||
`app/migrate_layout.py:18,280` and the legacy spool default in `app/migrate_runtime_safety.py:74` also contain host-specific paths; do not use their defaults as Linux provisioning instructions. Optional `pg_trgm` creation is attempted defensively, not a proven unconditional startup prerequisite. No live migration should be run until restore/rollback and exclusive ownership are established.
|
||||
|
||||
### B12. Build a Complete, Reproducible Linux Dependency Set
|
||||
|
||||
**Evidence:** `app/requirements.txt:1-9`; `app/requirements-keycheckers.txt:1-4`; `app/child_bootstrap.py:24-34,177-195`; `app/keycheck_runner.py:2168-2175`; `app/lifecycle_authority.py:242-291,378-390`; `app/scanner.py:11168,11403-11405,11932,12993,13012,13834`.
|
||||
|
||||
- Installing only `requirements.txt` misses `boto3`/`botocore`; isolated bootstrap requires them for every keycheck-provider. Installing only `requirements-keycheckers.txt` misses `PyYAML`. Install the union or define complete, tested profiles. `zstandard` is currently required for every scanner bootstrap, not just an enabled Docker source.
|
||||
- Pin a tested, patched CPython minor and dependency resolution, including a compatible boto3/botocore pair. The host has Python 3.12.3, but that is neither a recommended security patch level nor a Linux compatibility result. Archive extraction uses version-sensitive tarfile APIs; a strict minimum of 3.12 was not established because some security APIs were backported.
|
||||
- Supply Linux TruffleHog and full Git for the target architecture, with release/checksum verification. The resolver currently prefers a present private `runtime/git/cmd/git.exe` without an OS check. Exclude Windows vendor binaries and make the Linux resolution explicit. Git and TruffleHog are required by the current global manifest even for a restricted source set.
|
||||
- Ensure TruffleHog's subprocess `PATH` resolves the same intended Git installation as the manifest, including its HTTPS transport helper. Copying a lone `git` executable is insufficient.
|
||||
- Validate native wheels/ABI and stdlib `ssl`, `sqlite3`, `zlib`, `bz2`, `lzma`, plus CA certificates. Check shared-library requirements of the selected TruffleHog and, if embedded, PostgreSQL build. Windows wheels and extensions cannot be reused. A glibc-based image is a simpler first target than assuming Alpine/musl compatibility.
|
||||
- `psycopg[binary]` with a supported wheel does not automatically require `libpq-dev`/`pg_config`; source-build requirements depend on wheel availability. The zstandard CLI does not replace the Python package. Go/CGO are build dependencies only if the selected TruffleHog is compiled from source.
|
||||
- Validate the actual TruffleHog CLI and output contract: the wrapper uses Git/Docker/filesystem/HuggingFace paths, archive flags and branch/SHA/local-development options. A successful version probe alone does not validate these. The parser also uses the `finished scanning` diagnostic to distinguish complete work from an incomplete command.
|
||||
|
||||
### B13. Prevent Secrets and Host State From Entering the Image
|
||||
|
||||
**Evidence:** root `.gitignore`; root/application file layout; `app/runtime_security.py:872-889,989-998`; manifest exceptions in `app/lifecycle_authority.py:66-70`.
|
||||
|
||||
The working directory contains credential files, credential backups, databases, scan output, keycheck output, state, logs, temporary data and bundled Windows tools. Their contents were not read for this audit. `.gitignore` does not protect a Docker build context.
|
||||
|
||||
**Correction:** create `.dockerignore` plus an allowlisted `COPY` strategy. Exclude real `.env*`, secret/backup/lock variants, databases including WAL/SHM, findings/results, queues, logs, state, scratch data, caches, local tool state and Windows vendor distributions. Account explicitly for the currently manifested first-party runtime scripts instead of blindly excluding them. Do not bake credentials into layers, build arguments or a committed Compose file.
|
||||
|
||||
Provide non-secret example configuration and inject secrets at runtime. Test owner and mode compatibility under B07: a root-owned `0444` secret mount does not satisfy the current effective-UID private policy. Keep code/config read-only after provisioning where feasible. The optional credential-writeback workflow is covered separately in C04.
|
||||
|
||||
### B14. Persist the Whole Data Pipeline and Preserve Filesystem Semantics
|
||||
|
||||
**Evidence:** `app/config.yaml:11-19,34-37,79-81,109-120,140,148`; `app/result_bundle.py:67-85,331-354,388-403`; `app/jsonl_projector.py:109-153`; `app/keycheck_runner.py:2201-2203`.
|
||||
|
||||
PostgreSQL is not the only durable store. Result reservations refer to payload bundles on disk. Bundles are flushed/fsynced and atomically published from `tmp` to `ready` under one root; the commit reference is relative to that root. Losing the bundle volume while keeping PostgreSQL can lose pending ingestion inputs. Output publication also has file-level state and locks.
|
||||
|
||||
| Data class | Deployment treatment |
|
||||
| --- | --- |
|
||||
| PostgreSQL data | Durable volume; version-compatible backup/restore; one authority |
|
||||
| Result bundles | Durable volume with `tmp`, `ready` and `quarantine` together; preserve atomic rename/fsync behavior |
|
||||
| Results and keycheck outputs | Preserve JSONL, rotation/publication state and needed replay inputs; maintain owner-only access |
|
||||
| Queue/state/resolver and SQLite caches | Classify individually; persist required resume state, distinguish rebuildable caches from authoritative PostgreSQL data |
|
||||
| Work clones/download/extraction scratch | Separate bounded writable storage; do not assume it fits memory-backed tmpfs |
|
||||
| Control/PID/session metadata | Deliberately per-instance storage or exact-identity reconciliation; do not restore stale runtime identity as business data |
|
||||
| Logs | Bounded retention or a secure collector; do not grow the container writable layer indefinitely |
|
||||
| Config and secrets | Separately provisioned/injected; not bundled into data/image backups indiscriminately |
|
||||
|
||||
Do not mount bundle `tmp` on a different filesystem from `ready`. Validate ownership, no-symlink policy, locks and durable atomic publication on the actual storage driver. Do not assume Windows binds, SMB or NFS have the required POSIX behavior. Named volumes backed by a suitable native Linux filesystem are the safer first choice, but still require testing.
|
||||
|
||||
Mounting only the old `runtime/` directory misses the configured `S:` locations. Root-level `scanner.db` and old output files are not automatically the authoritative deployment dataset. Establish the transfer inventory before copying.
|
||||
|
||||
Current capacity settings include a 3 GiB bundle budget, a 192 MiB per-event cap, a 2 GiB projection backlog budget and a 20 GiB free-space floor. Provision space for concurrent work, PostgreSQL/WAL, bundles and outputs, or deliberately retune those policies. A small default container disk can refuse scans even while the process is healthy. Test a coordinated database-plus-bundle restore, not just `pg_dump` in isolation.
|
||||
|
||||
## P1: Unattended Operation
|
||||
|
||||
### R01. Retain Ownership When Failed Startup Cleanup Cannot Confirm Exit
|
||||
|
||||
`app/supervisor.py:1200-1208` swallows errors from terminate/wait and clears the retained process reference. Preserve the owner/identity and enter the existing failed-hold path if rollback cannot prove that a child stopped. Otherwise a failed start can leave an untracked process. Verify this after the POSIX containment correction.
|
||||
|
||||
### R02. Preserve Signal/OOM Exit Status
|
||||
|
||||
`app/owned_process.py:930` attempts to reproduce a signalled payload exit through signal handling, but installing a handler for SIGKILL is invalid. A payload terminated with `-9` can be reported as host exit 127. Correct the signal-exit reproduction and test OOM/SIGKILL separately from ordinary program failures; do not label this as a container memory-policy fix by itself.
|
||||
|
||||
### R03. Unify Foreground Shutdown Completion
|
||||
|
||||
`app/supervisor.py:5327-5339` writes shutdown receipts only for background children, while POSIX inspection of a non-child process cannot retrieve its exit code. This can make the existing stop workflow report failure after a foreground container runtime has actually exited. Define one completion protocol for both launch modes. Until then, authenticated `--cmd shutdown` plus independently waiting for the supervisor/container to exit is different from trusting the shutdown acknowledgement alone.
|
||||
|
||||
### R04. Add Dependency-Aware Health and Recovery Semantics
|
||||
|
||||
`app/supervisor.py:5225,1285,3361-3367,3551` distinguishes activation, held workers, initial ingester readiness and failed-hold state. A live PID, ACTIVE handshake, Streamlit health response or PostgreSQL TCP response is not enough to certify the pipeline.
|
||||
|
||||
Expose a read-only machine health result covering supervisor phase, authenticated database/cluster identity, schema/cutover readiness, ingester/projector heartbeat or singleton lease, configured required workers, storage/backlog health and any unrecoverable hold. Allow an honest startup period without admitting work prematurely. The initial source gate opens once; explicitly decide whether later dependency loss should close it or allow bounded asynchronous intake, and test that policy through outage, backlog exhaustion and recovery.
|
||||
|
||||
Distinguish degraded readiness from a dead process. A Docker healthcheck alone does not restart an unhealthy container; a restart policy normally responds to process exit. Do not configure blind health-triggered replacement that discards a `FAILED_HOLD` ownership dispute.
|
||||
|
||||
### R05. Replace Host Resource Assumptions With Container Budgets
|
||||
|
||||
`app/scanner.py:291,1221,1279-1290,11460-11472`; `app/owned_process.py:965`; `app/config.yaml:90-95,145`.
|
||||
|
||||
Windows Job memory/CPU/priority settings are not enforced by the POSIX branch. The configured TruffleHog Job memory limit is 4 GiB; putting that value in YAML does not create a Linux limit. CPU counts may describe the host rather than the effective quota. The optional bonus scan slot uses Windows resource counters and fails closed on Linux.
|
||||
|
||||
Set explicit workload concurrency, cgroup CPU/memory/PID budgets and storage limits. Account for PostgreSQL, Python, native payloads, one containment-host process per owned job and threads. A whole-container memory cap is not equivalent to the old per-tree Windows Job cap. Explicitly disable the bonus slot initially or implement quota-aware Linux telemetry without weakening admission safety. Derive limits from representative tests rather than multiplying configured maxima into an asserted minimum RAM requirement.
|
||||
|
||||
### R06. Make Logs Observable Without Depending on Ignored Python Variables
|
||||
|
||||
`app/supervisor.py:240` and `app/keycheck_runner.py:2168-2175` launch isolated interpreters. `-I` ignores `PYTHONUNBUFFERED` and `PYTHONIOENCODING`; adding those variables to Compose is not a reliable buffering/encoding fix. Use explicit interpreter flags such as `-u` or configure streams, and verify child output under the container locale. Retain necessary file logs with rotation/collection and keep credential-bearing output private and redacted from generic health messages.
|
||||
|
||||
### R07. Enforce Provider Process Deadlines
|
||||
|
||||
`app/keycheck_runner.py:2263` waits for the provider process without a process-level timeout. A provider can outlive a scheduler deadline even when individual HTTP operations have timeouts. Add a bounded process deadline/watchdog using corrected owned-tree termination; preserve partial durable results and lease recovery. A liveness check must not silently treat a stuck provider as productive work.
|
||||
|
||||
## Conditional Requirements
|
||||
|
||||
### C01. Authenticated Git Needs a POSIX Askpass Helper
|
||||
|
||||
**Applies when:** Git requests credentials, including relevant Git/HuggingFace/package paths.
|
||||
|
||||
`app/scanner.py:11426-11433` unconditionally creates a Windows `git-askpass.cmd` using batch syntax when `TRUF_GIT_TOKEN` is set. Callers include `app/scanner.py:12384-12388,12600-12605`. Anonymous clones can hide the defect.
|
||||
|
||||
Provide a POSIX helper with a correct interpreter/shebang, LF and private executable permissions; retain the Windows branch. Read the token from the controlled environment, not a credential-bearing URL or argv. If the work volume is `noexec`, a prepackaged trusted helper outside that scratch volume is preferable to weakening the whole volume. Coordinate this with executable hardening and immutable code policy. Test using fake credentials and a local/mocked Git interaction.
|
||||
|
||||
### C02. Dashboard Publication Requires an Explicit Security Design
|
||||
|
||||
**Applies when:** the UI must be accessed from outside the runtime container.
|
||||
|
||||
`app/.streamlit/config.toml:1-4` sets `127.0.0.1:5000`; `app/supervisor.py:3608-3624` and `app/dashboard.py:2528-2538` independently reject non-loopback hosts. Dashboard launch also requires authenticated supervisor-child context. Changing only Streamlit configuration or publishing a Docker port will not make the in-container loopback listener reachable.
|
||||
|
||||
Either add an explicit secured container-bind mode in both guards, or use a proxy/tunnel in the **same network namespace** that can reach the existing loopback listener. A normal separate bridge-network proxy cannot reach it. Add access control and TLS at the appropriate boundary; read-only database access does not make scan/credential observability safe for public exposure. Preserve the supervised launch contract.
|
||||
|
||||
Keep the control interface `127.0.0.1:8765` private (`app/supervisor.py:3726-3733`; `app/config.yaml:182-183`). Use authenticated bootstrap commands such as `--cmd status`, `--cmd shutdown` and `--attach` through `docker exec` in the same container and UID. Do not publish port 8765 or broadly remove loopback restrictions. This entire UI exposure change can be deferred by using `--no-dashboard`.
|
||||
|
||||
### C03. Restricted Egress, Proxies, Custom CA and IPv6 Need Explicit Support
|
||||
|
||||
**Applies when:** deployment cannot use the existing direct outbound network behavior.
|
||||
|
||||
- `app/scanner.py:374-375,560,11397` uses different routing for discovery and downloads/native Git/TruffleHog. Some paths deliberately remove proxy environment variables or use direct clients. Configured download-proxy flags do not themselves implement that routing. `HTTP_PROXY` alone is not enough for a proxy-only deployment.
|
||||
- `app/scanner.py:480-484` accepts proxy formats that differ from `app/keycheckers/keycheck_common.py:2200`; OpenAI/Gemini/OpenRouter also have duplicated parsers. Unify or explicitly constrain all formats and fallback behavior, including escaped credentials and ambient environment proxies. Add PySocks/`requests[socks]` only if SOCKS is required; it is not currently declared.
|
||||
- `app/scanner.py:13932` ignores ambient CA settings on the downloader path. Plumb the trusted CA explicitly for corporate interception/custom trust instead of disabling verification. `app/scanner.py:105` forces IPv4 by default; consider `SCANNER_FORCE_IPV4=0` only if the target network needs IPv6 and the path is tested.
|
||||
- Allow the selected sources' API, registry/CDN, download and redirect destinations, plus configured provider/resolver endpoints. DNS/private-address protections can reject destinations (`app/scanner.py:9340-9389,9416`). Preserve SSRF safeguards while making any intended private-network exception explicit. Provider resolution may contact DeepSeek/Z.ai/Qwen/Kimi depending on configured order (`app/keycheckers/provider_resolution.py:64-156`).
|
||||
|
||||
Use mocks/local fixtures for proxy, TLS, redirect and DNS tests. Do not use recovered credentials to test network readiness.
|
||||
|
||||
### C04. Offline Credential Writeback Needs a Different Mount Contract
|
||||
|
||||
**Applies when:** `sync_alive_github_tokens.py` will update the canonical secrets file.
|
||||
|
||||
`app/sync_alive_github_tokens.py:120-138,162-175,257-262` requires verified stopped authority, an adjacent lock, a same-directory private temporary file and atomic replacement. A read-only secret can be suitable for normal runtime but not for this maintenance operation. A single-file bind mount also cannot be assumed to support replacement of its mountpoint.
|
||||
|
||||
Choose a separate external/offline rotation workflow or a private writable containing directory for this maintenance mode. Preserve atomic publication and exact canonical path checks. Do not make all application code/secrets permanently writable merely to support an optional operation.
|
||||
|
||||
### C05. Multiple Replicas or Split Workers Require New Coordination
|
||||
|
||||
**Applies when:** scaling the runtime or moving authenticated workers into separate containers.
|
||||
|
||||
`app/runtime_security.py:502` uses filesystem-scoped locks. `app/lifecycle_authority.py:641` relies on local process verification, metadata and control reachability. `app/jsonl_projector.py:140-153` has both a singleton database lease and a file lock; ingester/projector are not arbitrary scalable workers.
|
||||
|
||||
Local lock paths in separate container filesystems do not provide a cross-container exclusion guarantee. Worker authentication also does not become remote authentication simply because a directory is mounted. A scale-out design needs explicit shared/distributed fencing, control/identity transport, data ownership and volume semantics. Until then, use one owner and one runtime replica, prevent the old host runtime from remaining active, and do not suggest `docker compose --scale` as an operational option.
|
||||
|
||||
### C06. Decide Whether Windows Archive-Name Rules Remain Policy
|
||||
|
||||
**Applies when:** Linux deployments should accept archive entries valid on POSIX but invalid on Windows.
|
||||
|
||||
`app/scanner.py:13760-13775` still rejects Windows reserved names, colons and trailing dot/space on Linux. This is a policy limitation, not an unconditional container boot defect. Either document it unchanged or separate OS-specific name restrictions. Keep traversal, entry-type, size and expansion-budget protections intact.
|
||||
|
||||
### C07. Extend the Manifest if First-Party Linux Native Modules Are Added
|
||||
|
||||
**Applies when:** native `.so` application modules are introduced inside the authenticated application tree.
|
||||
|
||||
`app/lifecycle_authority.py:21` includes `.pyd` but not Linux `.so` in application import suffixes. Add the appropriate native extension suffix policy and tests when such first-party modules exist. This is not a reason to add every installed dependency to the current application-code manifest or to block the present pure-Python application solely on this basis.
|
||||
|
||||
## What Is Not Required
|
||||
|
||||
- No Docker daemon, Docker socket mount, Docker CLI, DinD, privileged container or image-architecture emulation is needed for the inspected DockerHub scanning path. It reads image content rather than executing the image (`app/scanner.py:12980,13645`).
|
||||
- `DOCKER_CONFIG` is an authentication input, not a need for Docker Desktop. Recovery intentionally rejects implicit keychain/helper assumptions; preserve the managed credential pool (`app/scanner.py:4790,13433-13456`).
|
||||
- npm/PyPI content is scan data. Node.js and a browser are not runtime dependencies of these scanner/keychecker paths. AWS CLI, `gcloud` and `az` are not required merely because those providers are checked.
|
||||
- Go is not needed in the final image when supplying a compatible prebuilt TruffleHog. GCP keychecker RSA handling does not establish a dependency on Google SDK/cryptography/openssl CLI (`app/keycheckers/gcp/gcpKeycheck.py:288`).
|
||||
- Existing POSIX `/proc` identity support, `flock`, PostgreSQL command branches, activation/STOPPING states, leases and owned-versus-observed database semantics should be preserved and completed, not rewritten wholesale.
|
||||
- A general path-case rename or repository-wide CRLF rewrite was not justified. Fix genuinely platform-specific helpers and configured paths instead.
|
||||
|
||||
## Documentation and Operational Corrections
|
||||
|
||||
Create a deployment Compose definition separate from the recovery fixture, an allowlisted image build, a non-secret Linux configuration example, an ownership/volume provisioning procedure, and a backup/restore/upgrade runbook. These are missing deployment deliverables, not files generated by this audit.
|
||||
|
||||
Document the exact source profile and foreground lifecycle, explicit health semantics, stop/restart deadlines, singleton restriction, volume classes, secret maintenance, pinned versions and supported architecture. Remove Windows freeze-counter/diagnostic scripts from the Linux launch chain (`start_freeze_counters.ps1:27`); keeping a script as inert manifest data is different from executing it.
|
||||
|
||||
Correct any assumption that provider checks are free/read-only readiness probes. `app/KEYCHECKERS.md:44,87,109` must be reconciled with actual provider defaults. Qwen and several other providers can perform generation by default (`app/keycheckers/qwen/qwenKeycheck.py:544`); AWS/Replicate/Azure paths can probe IAM, resources or RBAC, with additional optional model requests. TruffleHog `no-verification` does not disable the separate keychecker subsystem (`app/config.yaml:136`). Healthchecks and image smoke tests must not invoke those real credential checks.
|
||||
|
||||
## Recommended Implementation Order
|
||||
|
||||
1. Choose Linux distribution/CPU architecture, PostgreSQL topology, UI requirement, source set, egress policy, non-root UID and storage/resource budgets. Confirm which existing data is authoritative and define rollback.
|
||||
2. Fix portable paths, POSIX identity/containment, signal/restart behavior and executable/ACL policy. Add focused offline Linux regression tests while preserving the current Windows contracts.
|
||||
3. Assemble the pinned dependency/native-tool image and code-authority layout. Add `.dockerignore`, a foreground entrypoint contract, private provisioning and a separate deployment Compose definition. Keep one runtime replica.
|
||||
4. Build and exercise a disposable Linux environment with fake credentials and local fixtures. Validate process lifetime, shutdown, restart, permissions, imports, native CLI contracts, health and resource limits before touching real data.
|
||||
5. Implement the chosen PostgreSQL mode. Rehearse fresh initialization and a restored dataset, schema/cutover migration, bundle/projection reconciliation and coordinated recovery. Embedded identity binding needs a stopped target PostgreSQL; SQL migration needs PostgreSQL available with normal workers stopped.
|
||||
6. Complete the conditional features actually needed: authenticated Git, secured UI, restricted-network support or credential maintenance. Defer unrelated scale-out work.
|
||||
7. Perform a controlled real-data cutover only after backup/restore rehearsal, exclusive ownership and rollback checks. Start with conservative concurrency and verify health/backlog behavior before increasing load.
|
||||
|
||||
## Acceptance Tests
|
||||
|
||||
| Area | Required evidence before claiming support |
|
||||
| --- | --- |
|
||||
| Build and ABI | Build on each supported target architecture; resolved dependency check; stdlib/native imports through the intended isolated interpreter; TruffleHog/Git help/version and library compatibility |
|
||||
| Authority image layout | No rejected application bytecode/symlinks; required manifest files/assets present; stable code/config hashes; dependencies outside the application tree |
|
||||
| Paths and permissions | Linux path resolution for every configured directory/file; non-root fresh-volume provisioning; correct private owners/modes; executable bits survive hardening; usable real lock root |
|
||||
| Process identity | Real POSIX host/payload handshake can create the scanner owner marker; exact identity survives normal lifecycle checks |
|
||||
| Process containment | Nested provider/native grandchildren terminate on source restart, parent death and failed startup; no orphan payloads or unreaped zombies |
|
||||
| Container lifecycle | Foreground no-TTY autostart; SIGTERM during active work; confirmed drain/stop; bounded ordinary shutdown; explicit failed-hold escalation; restart after reused PID/stale metadata |
|
||||
| Database | Fresh schema and valid cutover; wrong cluster rejected; offline identity rebinding; authenticated outage/recovery; external mode, if chosen, has no local PG start/stop dependency |
|
||||
| Durability | Container recreation preserves reservations/bundles/publications; interrupted atomic publication recovers safely; coordinated PG-plus-bundle backup can actually be restored |
|
||||
| Limits | Full disk, low free-space floor, bounded backlog, constrained CPU/RAM/PIDs, OOM exit status, bonus-slot denial and provider process deadline |
|
||||
| Network | Mocked direct/proxy/SOCKS-as-needed, parser formats, CA, IPv4/IPv6-as-needed, redirects and DNS/SSRF behavior |
|
||||
| Git | POSIX authenticated askpass with fake credentials, private executable permissions and the selected `noexec` work-volume arrangement |
|
||||
| Archives | gzip/zstd/PAX, malformed archives/missing decoders, entry-name policy and traversal/size protections on the chosen patched CPython |
|
||||
| Optional UI | Reachable only by the intended secured path; supervised authentication intact; health distinguished from full pipeline readiness; control port not published |
|
||||
| Safe probes | No provider generation, credential validation or production endpoint activity from build/health tests |
|
||||
|
||||
Existing tests to extend/select carefully:
|
||||
|
||||
- `tests/test_owned_process.py:79-82,98`: the identity assertion covers PID, and tree cleanup coverage is Windows-specific.
|
||||
- `tests/test_temp_owner_child_safety.py:49-59`: a mock supplies complete identity and can hide the real POSIX handshake defect.
|
||||
- `tests/test_supervisor_safety.py:978`: a mocked zero exit code can hide the foreground completion gap.
|
||||
- `tests/test_pipeline_postgres_integration.py:69`: adapt `.exe` assumptions to a disposable Linux PostgreSQL setup, not the authoritative host database.
|
||||
- `tests/test_runtime_security.py:227`: add actual container UID/mount/symlink/executable-policy cases.
|
||||
- `tests/test_api_proxy_routing.py:62,92,215,248`: extend mocked routing, proxy-parser and trust behavior.
|
||||
- `tests/test_docker_codec_recovery.py:72,130,149,175` and `tests/test_resource_lifecycle_fixes.py:191`: extend codec and cgroup/admission coverage.
|
||||
- Some tests exercise installed/native/live paths, including `tests/test_docker_codec_recovery.py:666` and `tests/test_huggingface_long_paths.py:324`. Separate offline tests from explicitly opted-in integration tests; do not run the entire suite against existing data or credentials by default.
|
||||
|
||||
## Verification Performed and Limits
|
||||
|
||||
- Inspected the application, configuration, entrypoints, dependency manifests, security/process/DB/result-pipeline code, recovery Compose and relevant tests. Findings cite inspected file/line locations; line numbers may move with later edits.
|
||||
- Reproduced the Windows-path/config-precedence failure using the actual pure path functions under POSIX path semantics, without application startup or filesystem mutation.
|
||||
- Parsed all 61 application Python files with Python 3.12.3 using AST-only analysis: zero syntax errors. This is not an import, dependency, Linux execution or behavior test.
|
||||
- Docker CLI was unavailable in this session: `docker version --format '{{json .Server}}'` failed because the command was not found. This does not prove that the host has no Docker installation or can never run containers.
|
||||
- No Docker build, Compose deployment, Linux process integration test or full pytest suite was run. No supervisor, scanner, provider checker or PostgreSQL server was started. Secret contents, real credential checks and database migrations were not used for verification.
|
||||
- Only this report was added. The findings identify the correction surface visible from repository inspection; target-platform tests may reveal additional issues. No claim of Docker readiness is made until the acceptance checks pass.
|
||||
+368
@@ -0,0 +1,368 @@
|
||||
FROM python:3.12-slim-bookworm@sha256:782412e85d0f0984994c290652577d4018aff08145c85b262bb63dc0c7522254 AS python-base
|
||||
|
||||
ENV PATH=/usr/local/bin:/usr/bin:/bin:/usr/lib/postgresql/16/bin \
|
||||
HOME=/data/home \
|
||||
LANG=C.UTF-8 \
|
||||
LC_ALL=C.UTF-8 \
|
||||
PYTHONDONTWRITEBYTECODE=1
|
||||
|
||||
RUN /usr/local/bin/python3 -I -S -B -c "import sys; assert sys.version_info[:3] == (3, 12, 14), sys.version"
|
||||
|
||||
FROM python-base AS lock-generator
|
||||
|
||||
COPY docker/build-dependencies/requirements.lock /tmp/compiler.lock
|
||||
RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \
|
||||
--require-hashes --only-binary=:all: --no-compile --no-cache-dir \
|
||||
-r /tmp/compiler.lock \
|
||||
&& rm /tmp/compiler.lock
|
||||
WORKDIR /src
|
||||
COPY app/requirements.txt app/requirements-keycheckers.txt ./app/
|
||||
COPY docker/requirements.in docker/requirements.lock docker/requirements-test.in docker/requirements-test.lock docker/requirements-worker.in docker/requirements-worker.lock ./docker/
|
||||
COPY docker/build-dependencies/requirements.in docker/build-dependencies/requirements.lock ./docker/build-dependencies/
|
||||
ENV CUSTOM_COMPILE_COMMAND="See docker/build-dependencies/README.md for the pinned Python 3.12.14 pip-tools generation command."
|
||||
CMD ["python3", "-m", "piptools", "compile", "--generate-hashes", "--allow-unsafe", "--resolver=backtracking", "--strip-extras", "--no-emit-index-url", "--no-emit-trusted-host", "--index-url=https://pypi.org/simple", "--pip-args=--only-binary=:all:", "--output-file=docker/requirements.lock", "docker/requirements.in"]
|
||||
|
||||
FROM python-base AS trufflehog-download
|
||||
|
||||
ARG TARGETARCH
|
||||
RUN python3 -I -S -B - "$TARGETARCH" <<'PY'
|
||||
import hashlib
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
import tarfile
|
||||
import urllib.request
|
||||
|
||||
checksums = {
|
||||
"amd64": "dc24007c2f233bd61c05beabeb44aa27ea9b43288166279209abe0458c5ce76b",
|
||||
"arm64": "7e65e771d2a247964056aa5edba0f8ae3945895e5dce867fe0ffbc7b0128239a",
|
||||
}
|
||||
architecture = sys.argv[1]
|
||||
if architecture not in checksums:
|
||||
raise SystemExit("TruffleHog is pinned only for linux/amd64 and linux/arm64")
|
||||
name = f"trufflehog_3.97.4_linux_{architecture}.tar.gz"
|
||||
url = "https://github.com/trufflesecurity/trufflehog/releases/download/v3.97.4/" + name
|
||||
digest = hashlib.sha256()
|
||||
size = 0
|
||||
with urllib.request.urlopen(url, timeout=60) as response, open("/tmp/trufflehog.tar.gz", "wb") as output:
|
||||
while chunk := response.read(1024 * 1024):
|
||||
digest.update(chunk)
|
||||
size += len(chunk)
|
||||
output.write(chunk)
|
||||
if digest.hexdigest() != checksums[architecture]:
|
||||
raise SystemExit("TruffleHog archive SHA-256 mismatch")
|
||||
if architecture == "amd64" and size != 34970205:
|
||||
raise SystemExit("TruffleHog amd64 archive length mismatch")
|
||||
with tarfile.open("/tmp/trufflehog.tar.gz", "r:gz") as archive:
|
||||
member = archive.getmember("trufflehog")
|
||||
if not member.isfile():
|
||||
raise SystemExit("TruffleHog archive executable must be a regular file")
|
||||
with archive.extractfile(member) as source, open("/trufflehog", "wb") as output:
|
||||
shutil.copyfileobj(source, output)
|
||||
os.chmod("/trufflehog", 0o755)
|
||||
os.unlink("/tmp/trufflehog.tar.gz")
|
||||
print(f"Verified {name}: {size} bytes, sha256:{digest.hexdigest()}")
|
||||
PY
|
||||
|
||||
FROM python-base AS worker-dependencies
|
||||
|
||||
COPY docker/requirements-worker.lock /tmp/requirements-worker.lock
|
||||
RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \
|
||||
--require-hashes --only-binary=:all: --no-compile --no-cache-dir \
|
||||
--target /worker-dependencies -r /tmp/requirements-worker.lock \
|
||||
&& PYTHONPATH=/worker-dependencies python3 -I -S -B - <<'PY'
|
||||
import sys
|
||||
sys.path.insert(0, "/worker-dependencies")
|
||||
import requests
|
||||
import yaml
|
||||
import zstandard
|
||||
PY
|
||||
RUN rm /tmp/requirements-worker.lock
|
||||
|
||||
FROM python-base AS worker-native-dependencies
|
||||
|
||||
RUN <<'SH'
|
||||
set -eu
|
||||
rm -f /etc/apt/sources.list /etc/apt/sources.list.d/debian.sources
|
||||
printf '%s\n' \
|
||||
'Types: deb' \
|
||||
'URIs: https://snapshot.debian.org/archive/debian/20260914T000000Z/' \
|
||||
'Suites: bookworm bookworm-updates' \
|
||||
'Components: main' \
|
||||
'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \
|
||||
'Check-Valid-Until: no' \
|
||||
'' \
|
||||
'Types: deb' \
|
||||
'URIs: https://snapshot.debian.org/archive/debian-security/20260914T000000Z/' \
|
||||
'Suites: bookworm-security' \
|
||||
'Components: main' \
|
||||
'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \
|
||||
'Check-Valid-Until: no' \
|
||||
> /etc/apt/sources.list.d/debian.sources
|
||||
export DEBIAN_FRONTEND=noninteractive
|
||||
apt-get -o Acquire::Retries=3 -o Acquire::https::Timeout=30 -o APT::Update::Error-Mode=any update
|
||||
apt-get install -y --no-install-recommends \
|
||||
ca-certificates=20250419~deb12u1 \
|
||||
git=1:2.39.5-0+deb12u3 \
|
||||
tini=0.19.0-1+b3
|
||||
install -d -o 10001 -g 10001 -m 0700 /data /data/home
|
||||
/usr/sbin/groupadd --gid 10001 truf
|
||||
/usr/sbin/useradd --uid 10001 --gid 10001 --no-create-home --home-dir /data/home --shell /usr/sbin/nologin truf
|
||||
install -d -o 0 -g 0 -m 0755 /worker-git/bin /worker-git/libexec /worker-git/share
|
||||
cp -aL /usr/bin/git /worker-git/bin/git
|
||||
cp -aL /usr/lib/git-core /worker-git/libexec/git-core
|
||||
cp -aL /usr/share/git-core /worker-git/share/git-core
|
||||
find /worker-git -type d -exec chmod 0755 {} +
|
||||
find /worker-git -type f -exec chmod go-w {} +
|
||||
test -x /worker-git/bin/git
|
||||
test -x /worker-git/libexec/git-core/git-remote-https
|
||||
rm -rf /var/lib/apt/lists/* /var/cache/apt/archives/* /var/log/apt/*
|
||||
SH
|
||||
|
||||
COPY --from=trufflehog-download --chown=0:0 --chmod=0755 /trufflehog /usr/local/bin/trufflehog
|
||||
|
||||
FROM worker-native-dependencies AS worker-package-build
|
||||
|
||||
ARG TARGETARCH
|
||||
COPY --from=worker-dependencies --chown=0:0 /worker-dependencies /build/app/dependencies
|
||||
COPY app/ /build/app/
|
||||
COPY docs/remote-worker-quickstart-ru.md /build/README_RU.md
|
||||
COPY docs/remote-worker-cheatsheet-windows-ru.md /build/
|
||||
COPY docs/remote-worker-cheatsheet-linux-ru.md /build/
|
||||
COPY docs/remote-worker-cheatsheet-docker-ru.md /build/
|
||||
COPY docker/worker-package-pins.json /build/worker-package-pins.json
|
||||
RUN case "$TARGETARCH" in \
|
||||
amd64) platform_tag=linux-x86_64 ;; \
|
||||
arm64) platform_tag=linux-aarch64 ;; \
|
||||
*) echo "unsupported worker architecture" >&2; exit 1 ;; \
|
||||
esac \
|
||||
&& python3 -u -I -S -B /build/app/worker_package_builder.py assemble \
|
||||
--output /opt/truf-worker \
|
||||
--source-app /build/app \
|
||||
--dependencies /build/app/dependencies \
|
||||
--detector-policy /build/app/trufflehog-custom-detectors.yaml \
|
||||
--trufflehog /usr/local/bin/trufflehog \
|
||||
--git-root /worker-git \
|
||||
--git-executable bin/git \
|
||||
--platform-tag "$platform_tag" \
|
||||
--operator-readme /build/README_RU.md \
|
||||
--operator-cheatsheet /build/remote-worker-cheatsheet-windows-ru.md \
|
||||
--operator-cheatsheet /build/remote-worker-cheatsheet-linux-ru.md \
|
||||
--operator-cheatsheet /build/remote-worker-cheatsheet-docker-ru.md \
|
||||
--build-inputs /build/worker-package-pins.json
|
||||
|
||||
FROM worker-native-dependencies AS worker
|
||||
|
||||
ENV GIT_EXEC_PATH=/opt/truf-worker/runtime/git/libexec/git-core \
|
||||
GIT_TEMPLATE_DIR=/opt/truf-worker/runtime/git/share/git-core/templates
|
||||
COPY --from=worker-package-build --chown=0:0 /opt/truf-worker /opt/truf-worker
|
||||
RUN chown -R 10001:10001 /opt/truf-worker/app \
|
||||
&& find /opt/truf-worker/app -type d -exec chmod 0700 {} + \
|
||||
&& find /opt/truf-worker/app -type f -exec chmod 0600 {} + \
|
||||
&& chown 10001:10001 /opt/truf-worker/worker-package.json \
|
||||
&& chmod 0600 /opt/truf-worker/worker-package.json \
|
||||
&& find /opt/truf-worker/bin /opt/truf-worker/runtime -type d -exec chmod 0755 {} + \
|
||||
&& find /opt/truf-worker/bin /opt/truf-worker/runtime -type f -exec chmod go-w {} + \
|
||||
&& test ! -e /opt/truf-worker/app/keycheck_runner.py \
|
||||
&& test ! -d /opt/truf-worker/app/keycheckers \
|
||||
&& test ! -e /usr/lib/postgresql \
|
||||
&& test ! -e /usr/bin/psql
|
||||
USER 10001:10001
|
||||
RUN /opt/truf-worker/bin/trufflehog --version >/dev/null \
|
||||
&& /opt/truf-worker/runtime/git/bin/git --version >/dev/null \
|
||||
&& /usr/local/bin/python3 -I -S -B - <<'PY'
|
||||
import sys
|
||||
sys.path[:0] = ['/opt/truf-worker/app', '/opt/truf-worker/app/dependencies']
|
||||
from importlib.util import find_spec
|
||||
from worker_cli import parse_args
|
||||
from worker_package import verify_worker_package
|
||||
|
||||
package = verify_worker_package('/opt/truf-worker/worker-package.json')
|
||||
assert package['manifest']['schema'] == 3
|
||||
assert package['manifest']['protocol_version'] == 2
|
||||
assert {
|
||||
(item['source'], item['platform'], item['planning_kind'])
|
||||
for item in package['manifest']['capabilities']
|
||||
} == {
|
||||
('gitlab', 'gitlab', 'exact_git_v1'),
|
||||
('dockerhub', 'docker', 'docker_direct_v1'),
|
||||
('huggingface', 'huggingface', 'huggingface_space_v1'),
|
||||
}
|
||||
assert set(package['runtime_trees']) == {'git'}
|
||||
assert parse_args(['run', '--server', 'https://worker.example', '--token', 'x' * 32]).command == 'run'
|
||||
assert all(find_spec(name) is None for name in ('httpx', 'psycopg', 'starlette', 'streamlit'))
|
||||
PY
|
||||
WORKDIR /data
|
||||
ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/python3", "-u", "-I", "-S", "-B", "/opt/truf-worker/app/remote_worker_bootstrap.py", "--"]
|
||||
CMD ["run"]
|
||||
|
||||
FROM python-base AS native-dependencies
|
||||
|
||||
ADD --checksum=sha256:0144068502a1eddd2a0280ede10ef607d1ec592ce819940991203941564e8e76 https://www.postgresql.org/media/keys/ACCC4CF8.asc /usr/share/keyrings/postgresql.asc
|
||||
|
||||
RUN <<'SH'
|
||||
set -eu
|
||||
chmod 0644 /usr/share/keyrings/postgresql.asc
|
||||
rm -f /etc/apt/sources.list /etc/apt/sources.list.d/debian.sources
|
||||
printf '%s\n' \
|
||||
'Types: deb' \
|
||||
'URIs: https://snapshot.debian.org/archive/debian/20260914T000000Z/' \
|
||||
'Suites: bookworm bookworm-updates' \
|
||||
'Components: main' \
|
||||
'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \
|
||||
'Check-Valid-Until: no' \
|
||||
'' \
|
||||
'Types: deb' \
|
||||
'URIs: https://snapshot.debian.org/archive/debian-security/20260914T000000Z/' \
|
||||
'Suites: bookworm-security' \
|
||||
'Components: main' \
|
||||
'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \
|
||||
'Check-Valid-Until: no' \
|
||||
> /etc/apt/sources.list.d/debian.sources
|
||||
printf '%s\n' \
|
||||
'deb [signed-by=/usr/share/keyrings/postgresql.asc] https://apt-archive.postgresql.org/pub/repos/apt bookworm-pgdg-archive main' \
|
||||
> /etc/apt/sources.list.d/postgresql.list
|
||||
# Only these exact PGDG packages may supplement the immutable Debian snapshot.
|
||||
printf '%s\n' \
|
||||
'Package: postgresql-16 postgresql-client-16 libpq5' \
|
||||
'Pin: version 16.15-1.pgdg12+2' \
|
||||
'Pin-Priority: 1001' \
|
||||
'' \
|
||||
'Package: postgresql-common postgresql-client-common' \
|
||||
'Pin: version 293.pgdg12+1' \
|
||||
'Pin-Priority: 1001' \
|
||||
'' \
|
||||
'Package: *' \
|
||||
'Pin: origin apt-archive.postgresql.org' \
|
||||
'Pin-Priority: -1' \
|
||||
> /etc/apt/preferences.d/postgresql
|
||||
printf '#!/bin/sh\nexit 101\n' > /usr/sbin/policy-rc.d
|
||||
chmod 0755 /usr/sbin/policy-rc.d
|
||||
export DEBIAN_FRONTEND=noninteractive
|
||||
apt-get -o Acquire::Retries=3 -o Acquire::https::Timeout=30 -o APT::Update::Error-Mode=any update
|
||||
apt-get install -y --no-install-recommends \
|
||||
postgresql-common=293.pgdg12+1 \
|
||||
postgresql-client-common=293.pgdg12+1
|
||||
# Set this after common is installed but before installing any server package.
|
||||
printf '\ncreate_main_cluster = false\n' >> /etc/postgresql-common/createcluster.conf
|
||||
apt-get install -y --no-install-recommends \
|
||||
ca-certificates=20250419~deb12u1 \
|
||||
git=1:2.39.5-0+deb12u3 \
|
||||
tini=0.19.0-1+b3 \
|
||||
postgresql-16=16.15-1.pgdg12+2 \
|
||||
postgresql-client-16=16.15-1.pgdg12+2 \
|
||||
libpq5=16.15-1.pgdg12+2
|
||||
test ! -d /var/lib/postgresql/16/main
|
||||
rm -f /etc/ssl/private/ssl-cert-snakeoil.key /etc/ssl/certs/ssl-cert-snakeoil.pem
|
||||
rm -rf /var/lib/apt/lists/* /var/cache/apt/archives/* /var/log/apt/*
|
||||
/usr/sbin/groupadd --gid 10001 truf
|
||||
/usr/sbin/useradd --uid 10001 --gid 10001 --no-create-home --home-dir /data/home --shell /usr/sbin/nologin truf
|
||||
install -d -o 10001 -g 10001 -m 0700 /data /data/home
|
||||
SH
|
||||
|
||||
RUN python3 -I -S -B - <<'PY'
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
# Keep the package's complete Git helper tree; regular hard links preserve argv[0].
|
||||
for directory in (Path("/usr/local/bin"), Path("/usr/lib/git-core")):
|
||||
for path in directory.iterdir():
|
||||
if path.is_symlink() and os.access(path, os.X_OK):
|
||||
target = path.resolve(strict=True)
|
||||
if not target.is_file() or target.stat().st_uid != 0:
|
||||
raise SystemExit(f"Untrusted executable target: {path}")
|
||||
path.unlink()
|
||||
os.link(target, path)
|
||||
# Prefer native PG16 clients over the distribution's symlinked version wrappers.
|
||||
for target in Path("/usr/lib/postgresql/16/bin").iterdir():
|
||||
path = Path("/usr/bin") / target.name
|
||||
if path.is_symlink():
|
||||
path.unlink()
|
||||
os.link(target, path)
|
||||
for name in ("/usr/local/bin/python3", "/usr/bin/git", "/usr/lib/git-core/git-remote-https", "/usr/bin/tini"):
|
||||
path = Path(name)
|
||||
details = path.lstat()
|
||||
if path.is_symlink() or not path.is_file() or details.st_uid != 0 or details.st_mode & 0o022:
|
||||
raise SystemExit(f"Untrusted native executable: {path}")
|
||||
PY
|
||||
|
||||
FROM native-dependencies AS dependencies
|
||||
|
||||
COPY docker/requirements.lock /tmp/requirements.lock
|
||||
RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \
|
||||
--require-hashes --only-binary=:all: --no-compile --no-cache-dir \
|
||||
-r /tmp/requirements.lock \
|
||||
&& python3 -m pip --isolated check \
|
||||
&& rm /tmp/requirements.lock
|
||||
USER 10001:10001
|
||||
|
||||
# Both public targets inherit these exact runtime contents; the default stays runtime.
|
||||
FROM dependencies AS runtime-base
|
||||
|
||||
USER 0:0
|
||||
RUN install -d -o 10001 -g 10001 -m 0700 /opt/truf /opt/truf/app /opt/truf/tests
|
||||
USER 10001:10001
|
||||
COPY --chown=10001:10001 app/ /opt/truf/app/
|
||||
RUN python3 -I -S -B - <<'PY'
|
||||
import os
|
||||
import stat
|
||||
|
||||
for directory, directories, files in os.walk("/opt/truf/app", followlinks=False):
|
||||
for path in [directory, *(os.path.join(directory, name) for name in directories + files)]:
|
||||
details = os.lstat(path)
|
||||
if not (stat.S_ISDIR(details.st_mode) or stat.S_ISREG(details.st_mode)):
|
||||
raise SystemExit(f"Application links/special files are forbidden: {path}")
|
||||
if os.path.basename(path) == "__pycache__" or path.endswith((".pyc", ".pyo")):
|
||||
raise SystemExit(f"Application bytecode is forbidden: {path}")
|
||||
if (details.st_uid, details.st_gid) != (10001, 10001):
|
||||
raise SystemExit(f"Application ownership mismatch: {path}")
|
||||
os.chmod(path, 0o700 if stat.S_ISDIR(details.st_mode) else 0o600)
|
||||
PY
|
||||
WORKDIR /opt/truf/app
|
||||
ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/python3", "-u", "-I", "-S", "-B", "/opt/truf/app/container_runtime.py"]
|
||||
CMD ["run"]
|
||||
|
||||
FROM runtime-base AS test
|
||||
|
||||
USER 0:0
|
||||
COPY --from=trufflehog-download --chown=0:0 --chmod=0755 /trufflehog /usr/local/bin/trufflehog
|
||||
COPY docker/requirements-test.lock /tmp/requirements-test.lock
|
||||
RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \
|
||||
--require-hashes --only-binary=:all: --no-compile --no-cache-dir \
|
||||
-r /tmp/requirements-test.lock \
|
||||
&& python3 -m pip --isolated check \
|
||||
&& rm /tmp/requirements-test.lock
|
||||
USER 10001:10001
|
||||
COPY --chown=10001:10001 tests/ /opt/truf/tests/
|
||||
COPY --chown=10001:10001 --chmod=0600 .dockerignore /opt/truf/.dockerignore
|
||||
COPY --chown=10001:10001 --chmod=0600 Dockerfile /opt/truf/Dockerfile
|
||||
COPY --chown=10001:10001 --chmod=0600 start_runtime.ps1 start_core_runtime.ps1 stop_runtime.ps1 /opt/truf/
|
||||
COPY --chown=10001:10001 --chmod=0600 compose.yaml compose.edge.yaml /opt/truf/
|
||||
COPY --chown=10001:10001 deploy/ /opt/truf/deploy/
|
||||
COPY --chown=10001:10001 docker/ /opt/truf/docker/
|
||||
RUN python3 -I -S -B - <<'PY'
|
||||
import os
|
||||
from pathlib import Path
|
||||
import stat
|
||||
|
||||
edge_e2e = {
|
||||
path.name for path in Path('/opt/truf/tests').glob('edge_e2e_*.py')
|
||||
}
|
||||
if edge_e2e != {'edge_e2e_backend.py', 'edge_e2e_client.py'}:
|
||||
raise SystemExit(f'Unexpected edge E2E test-stage inputs: {sorted(edge_e2e)}')
|
||||
for directory, directories, files in os.walk("/opt/truf/tests", followlinks=False):
|
||||
for path in [directory, *(os.path.join(directory, name) for name in directories + files)]:
|
||||
details = os.lstat(path)
|
||||
if not (stat.S_ISDIR(details.st_mode) or stat.S_ISREG(details.st_mode)):
|
||||
raise SystemExit(f"Test links/special files are forbidden: {path}")
|
||||
if os.path.basename(path) == "__pycache__" or path.endswith((".pyc", ".pyo")):
|
||||
raise SystemExit(f"Test bytecode is forbidden: {path}")
|
||||
if (details.st_uid, details.st_gid) != (10001, 10001):
|
||||
raise SystemExit(f"Test ownership mismatch: {path}")
|
||||
os.chmod(path, 0o700 if stat.S_ISDIR(details.st_mode) else 0o600)
|
||||
PY
|
||||
WORKDIR /opt/truf
|
||||
ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/python3", "-u", "-I", "-S", "-B", "/opt/truf/tests/container_unit.py"]
|
||||
CMD []
|
||||
|
||||
FROM runtime-base AS runtime
|
||||
@@ -0,0 +1,102 @@
|
||||
# Pipeline Quarantine Audit
|
||||
|
||||
Initial snapshot and remediation: `2026-08-15`
|
||||
|
||||
## Current Impact
|
||||
|
||||
- PostgreSQL quarantine rows: `177`
|
||||
- Accounted capacity: `4354 items / 1,817,503,389 bytes`
|
||||
- Configured admission limit: `10000 items / 1,073,741,824 bytes`
|
||||
- New scan admission is closed because the byte limit is exceeded.
|
||||
- `49,215` admission intents have already ended with `quarantine_admission_closed`.
|
||||
|
||||
`pipeline: ready` means that PostgreSQL and workers are healthy. It does not mean that new scan admission is open.
|
||||
|
||||
## Result Bundles
|
||||
|
||||
Two rows account for `4004 items / 1,212,153,856 bytes`.
|
||||
|
||||
| Quarantine ID | Source | Original error | Physical size | Read-only validation now |
|
||||
|---|---|---|---:|---|
|
||||
| `29` | Hugging Face `spaces` | Transient Windows `Permission denied` | `2,870 B` | Valid; 3 frames, no findings/errors/candidates |
|
||||
| `91` | Docker Hub query `tokenizer` | Transient Windows `Permission denied` | `5,354 B` | Valid; 8 frames, 1 finding, 4 errors, no candidates |
|
||||
|
||||
The bundle contents are valid. Their large capacity cost comes from worst-case reservations transferred into quarantine, not their physical file sizes.
|
||||
|
||||
Current code now reports a temporarily unavailable private bundle as an availability error. Result ingester defers it instead of classifying it as invalid content.
|
||||
|
||||
Recommendation: recover both bundles through an audited offline path rather than discard them. ID `91` contains one finding and must not be deleted without an explicit decision.
|
||||
|
||||
## Keycheck Quarantine
|
||||
|
||||
There are `175` pending keycheck quarantine rows.
|
||||
|
||||
| Class | Rows | Distinct credentials | Assessment |
|
||||
|---|---:|---:|---|
|
||||
| Repeated DeepSeek unconsumed rechecks | `104` | `1` | Duplicate hourly retries; current state is now `NO_CONTEXT` |
|
||||
| DeepSeek unconsumed findings | `15` | `6` | Legitimate historical candidates filtered by routing |
|
||||
| Azure Foundry unconsumed | `15` | `12` | Historical provider-consumption issue |
|
||||
| Hugging Face unconsumed | `7` | `5` | Historical provider-consumption issue |
|
||||
| Replicate unconsumed | `6` | `3` | Historical provider-consumption issue |
|
||||
| xAI unconsumed | `3` | `3` | Historical provider-consumption issue |
|
||||
| DeepSeek route mismatch | `24` | `14` | Misrouted non-DeepSeek detectors; source findings remain in PostgreSQL |
|
||||
| Azure route mismatch | `1` | `1` | Historical Azure Foundry route mismatch; current state is `UNKNOWN` |
|
||||
|
||||
The active growth came from one DeepSeek credential whose current state was `NETWORK`. Hourly network retry created a fresh candidate, provider routing silently rejected it, and the candidate entered quarantine after three unconsumed attempts.
|
||||
|
||||
## Preventive Changes
|
||||
|
||||
- Non-DeepSeek routing decisions now complete as `NO_CONTEXT` instead of leaving a leased candidate unconsumed.
|
||||
- DeepSeek has a canonical `deepseekNoContext.txt` status projection.
|
||||
- Recheck generation now refuses to enqueue a credential while it has an unresolved `provider_candidate_unconsumed` or `candidate_provider_route_mismatch` quarantine row.
|
||||
- Temporarily unavailable result bundle files are deferred instead of quarantined as validation failures.
|
||||
|
||||
Verification:
|
||||
|
||||
- Targeted tests: `60 passed`.
|
||||
- Manual `recheck deepseek network`: code `0`, no candidates processed.
|
||||
- DeepSeek unconsumed quarantine remained exactly `119`; no new row was created.
|
||||
|
||||
## Approved Remediation
|
||||
|
||||
1. Stop sources and acquire the offline migration guard.
|
||||
2. Recover bundle IDs `29` and `91` through deterministic re-ingestion.
|
||||
3. Discard the `104` duplicate DeepSeek retry rows after exact manifest review.
|
||||
4. Discard the `24` confirmed DeepSeek route-mismatch rows after exact manifest review.
|
||||
5. Requeue one current candidate per distinct credential from the remaining legitimate unconsumed groups; keep source attribution.
|
||||
6. Review the single Azure route mismatch separately.
|
||||
7. Resolve superseded duplicate candidate rows with an audited discard manifest.
|
||||
8. Verify physical artifacts, capacity accounting, projections and reopened scan admission before restarting sources.
|
||||
|
||||
All destructive decisions must use an exact private `truf-pipeline-quarantine-review-v1` manifest containing quarantine ID, reason code, payload hash and action. The offline review command requires both `--apply` and `--sources-stopped`.
|
||||
|
||||
## Applied Result
|
||||
|
||||
The user approved recovery plus selective cleanup.
|
||||
|
||||
- Manifest: `runtime/control/quarantine-remediation-20260815.json`
|
||||
- Manifest SHA-256: `c3143e6b0e15e07b6afacffd50b449444b9932e75001f633ac82b5b79e21fe3f`
|
||||
- Reviewed: `177`
|
||||
- Approved retry: `32`
|
||||
- Audited discard: `145`
|
||||
- Duplicate/conflicting reviews: `0`
|
||||
- Final quarantine capacity: `0 items / 0 bytes`
|
||||
|
||||
Bundle outcomes:
|
||||
|
||||
| Quarantine ID | Final reservation | Final bundle | Target disposition | Preserved contents |
|
||||
|---|---|---|---|---|
|
||||
| `29` | `acknowledged` | `acknowledged` | `done` | Clean empty result |
|
||||
| `91` | `acknowledged` | `acknowledged` | `deferred` | `1 finding`, `4 errors` |
|
||||
|
||||
Keycheck outcomes from the retried unique credentials:
|
||||
|
||||
| Service | Final current statuses |
|
||||
|---|---|
|
||||
| Azure | `12 FOUNDRY_UNRESOLVED`, `1 UNKNOWN` |
|
||||
| DeepSeek | `6 NO_CONTEXT` |
|
||||
| Hugging Face | `5 NO_CONTEXT` |
|
||||
| Replicate | `3 NO_CONTEXT` |
|
||||
| xAI | `3 NO_CONTEXT` |
|
||||
|
||||
The malformed legacy provider candidates are now terminal current-state records instead of repeatedly deferred/quarantined candidates. Scan admission reopened and scanner workers returned to `3/3` active operation.
|
||||
@@ -0,0 +1,273 @@
|
||||
# Truf Runtime Cheatsheet
|
||||
|
||||
## Быстрый старт
|
||||
|
||||
Команды выполняются из `D:\truf` в PowerShell.
|
||||
|
||||
```powershell
|
||||
# Запустить весь canonical runtime
|
||||
.\start_runtime.ps1
|
||||
|
||||
# Запустить explicit core set с keychecks, без dashboard
|
||||
.\start_core_runtime.ps1
|
||||
|
||||
# Подключиться к интерактивной консоли supervisor
|
||||
.\attach_runtime.ps1
|
||||
|
||||
# Координированно остановить весь runtime и PostgreSQL
|
||||
.\stop_runtime.ps1
|
||||
```
|
||||
|
||||
Если PowerShell блокирует запуск скриптов:
|
||||
|
||||
```powershell
|
||||
powershell.exe -NoProfile -ExecutionPolicy Bypass -File .\start_runtime.ps1
|
||||
```
|
||||
|
||||
Для полного рестарта используй именно:
|
||||
|
||||
```powershell
|
||||
.\stop_runtime.ps1
|
||||
.\start_runtime.ps1
|
||||
```
|
||||
|
||||
Для полного рестарта в core-only режиме:
|
||||
|
||||
```powershell
|
||||
.\stop_runtime.ps1
|
||||
.\start_core_runtime.ps1
|
||||
```
|
||||
|
||||
Не используй `restart all` как замену полному рестарту: pipeline workers защищены от ручного рестарта, пока scanner sources работают.
|
||||
|
||||
## Что запускается
|
||||
|
||||
| Компонент | Назначение |
|
||||
|---|---|
|
||||
| PostgreSQL | Единственный authoritative storage |
|
||||
| `result-ingester` | Переносит scan bundles в PostgreSQL |
|
||||
| `jsonl-projector` | Создаёт compatibility JSONL projections |
|
||||
| `janitor` | Обслуживает runtime queues и временные данные |
|
||||
| `github` | GitHub scanner loop |
|
||||
| `gitlab` | GitLab scanner loop |
|
||||
| `huggingface` | Hugging Face scanner loop |
|
||||
| `dockerhub` | Docker Hub scanner loop |
|
||||
| `package_git` | Package/repository scanner loop |
|
||||
| `keychecks` | Почасовой provider checker scheduler |
|
||||
|
||||
Одновременно выполняется максимум `3` scan workers. Dashboard при обычном запуске выключен.
|
||||
|
||||
## PowerShell-скрипты
|
||||
|
||||
| Скрипт | Что делает |
|
||||
|---|---|
|
||||
| `start_runtime.ps1` | Запускает freeze diagnostics, проверяет identity PostgreSQL и поднимает background supervisor |
|
||||
| `start_core_runtime.ps1` | Поднимает PostgreSQL, pipeline, janitor и три discovery-only producer; dashboard выключен |
|
||||
| `stop_runtime.ps1` | Выполняет authenticated coordinated shutdown supervisor, children и PostgreSQL |
|
||||
| `attach_runtime.ps1` | Открывает интерактивную supervisor-консоль; `quit` только отключает консоль |
|
||||
| `start_freeze_counters.ps1` | Запускает Windows performance counters в `H:\truf-diagnostics` |
|
||||
| `monitor_runtime_lag.ps1` | Пишет CPU/RAM/disk/runtime lag в CSV |
|
||||
| `cleanup_stale_agentui_vite.ps1` | Отдельная уборка старых AgentUI/Vite процессов; без `-Apply` только dry run |
|
||||
| `runtime\check-openrouter-keys.ps1` | Retired; намеренно завершается ошибкой |
|
||||
|
||||
В проекте нет собственных `.bat`/`.cmd`. Найденные BAT внутри `runtime\postgres\pgsql\pgAdmin 4` принадлежат pgAdmin и для Truf не используются.
|
||||
|
||||
## Full и Core-only режимы
|
||||
|
||||
`start_runtime.ps1` использует allowlist `supervisor.enabled_sources` из `config.linux.yaml`. Сейчас этот allowlist уже равен distributed core set, поэтому оба start-скрипта запускают одинаковые discovery producer.
|
||||
|
||||
`start_core_runtime.ps1` фиксирует core set прямо в wrapper и не зависит от будущего расширения default allowlist:
|
||||
|
||||
```text
|
||||
gitlab,dockerhub,huggingface
|
||||
```
|
||||
|
||||
PostgreSQL, `result-ingester`, `jsonl-projector`, `janitor` и независимо включённый `keychecks` также запускаются. Dashboard не запускается.
|
||||
|
||||
Чтобы сменить режим, сначала останови текущий supervisor через `.\stop_runtime.ps1`, затем запусти нужный start-скрипт. `stop_runtime.ps1` одинаков для обоих режимов.
|
||||
|
||||
## Core Sources
|
||||
|
||||
| Source | Что производит на сервере |
|
||||
|---|---|
|
||||
| `gitlab` | Ищет недавно активные GitLab projects и ставит их в очередь remote workers |
|
||||
| `dockerhub` | Ищет Docker Hub images и ставит в очередь только immutable `repo@sha256:...` targets |
|
||||
| `huggingface` | Ищет новейшие Hugging Face Spaces и ставит их в очередь remote workers |
|
||||
|
||||
`result-ingester`, `jsonl-projector`, `janitor` и `keychecks` отображаются как отдельные system workers, но не являются discovery sources. GitHub и `package_git` остаются доступными legacy/manual source, однако в distributed core profile не входят.
|
||||
|
||||
## Janitor
|
||||
|
||||
Janitor обслуживает только scanner work area (`S:\scanner-work`), а не PostgreSQL и не provider status files.
|
||||
|
||||
- Каждые `60` секунд ищет временные каталоги разрешённых типов.
|
||||
- Рассматривает только каталоги старше `7200` секунд.
|
||||
- Требует приватный `.scanner-owner.json` с точным process identity.
|
||||
- Удаляет каталог только если owner и parent гарантированно мертвы.
|
||||
- Не следует по symlink, junction или другим reparse points.
|
||||
- Один проход ограничен `50` каталогами, `10000` entries, `1 GiB`, `30` секундами и depth `64`.
|
||||
- Не имеет PostgreSQL credentials и не удаляет findings, keycheck history, current state или найденные секреты.
|
||||
|
||||
Примеры диагностики:
|
||||
|
||||
```powershell
|
||||
# Один диагностический замер
|
||||
.\monitor_runtime_lag.ps1 -Once
|
||||
|
||||
# Свой файл и интервал
|
||||
.\monitor_runtime_lag.ps1 -OutputPath H:\truf-diagnostics\lag.csv -IntervalSeconds 10
|
||||
|
||||
# Безопасный просмотр кандидатов на очистку Vite
|
||||
.\cleanup_stale_agentui_vite.ps1
|
||||
|
||||
# Реальная очистка найденного точного набора
|
||||
.\cleanup_stale_agentui_vite.ps1 -Apply
|
||||
```
|
||||
|
||||
## Supervisor-команды
|
||||
|
||||
Сначала запусти `.\attach_runtime.ps1`, затем используй команды ниже.
|
||||
|
||||
| Команда | Назначение |
|
||||
|---|---|
|
||||
| `help` | Полная встроенная справка |
|
||||
| `status` | Свежий status table |
|
||||
| `watch` | Live status; `q` возвращает в prompt |
|
||||
| `auth <source|all>` | Состояние auth pools |
|
||||
| `logs <source> [N]` | Последние `N` строк bounded-лога |
|
||||
| `command <source|all>` | Фактическая child-команда, log и state paths |
|
||||
| `start <source|all>` | Запустить остановленный source |
|
||||
| `stop <source|all>` | Остановить и оставить остановленным |
|
||||
| `restart <source|all>` | Перезапустить отдельный source |
|
||||
| `pause <source|all>` | Остановить и отметить paused |
|
||||
| `resume <source|all>` | Снять pause и запустить |
|
||||
| `once <source|all>` | Один проход source с `--once` |
|
||||
| `mode <source|all> loop|once|repeat` | Изменить режим source |
|
||||
| `set <source|all> interval <sec>` | Интервал repeat mode |
|
||||
| `set <source|all> restart on|off` | Автоматический restart после сбоя |
|
||||
| `set <source|all> restart_delay <sec>` | Начальная задержка restart |
|
||||
| `dashboard status|start|stop|restart` | Управление dashboard |
|
||||
| `shutdown` | Полный coordinated shutdown |
|
||||
| `quit` | В attach-режиме только отсоединиться |
|
||||
|
||||
`reload` намеренно отключён. После изменения `config.yaml` или runtime-кода нужен полный `stop_runtime.ps1` + `start_runtime.ps1`.
|
||||
|
||||
Source alias: `docker` означает `dockerhub`.
|
||||
|
||||
## Статусы
|
||||
|
||||
| Статус | Значение |
|
||||
|---|---|
|
||||
| `running` | Child сейчас работает |
|
||||
| `waiting` | Ожидает следующего запуска/retry |
|
||||
| `blocked` | Ждёт стабильной готовности PostgreSQL |
|
||||
| `paused` | Остановлен командой `pause` |
|
||||
| `done` | Успешный one-shot завершён |
|
||||
| `failed` | Child завершился с ошибкой, restart выключен |
|
||||
|
||||
`desired=running` показывает желаемое состояние. `rs` означает текущую серию ошибок / общее число automatic restarts. Старый `exit=1` рядом с уже `running` source относится к предыдущей попытке запуска.
|
||||
|
||||
## Keycheck Recheck
|
||||
|
||||
Формат:
|
||||
|
||||
```text
|
||||
recheck <service|all> [type ...] [options]
|
||||
```
|
||||
|
||||
Если type не указан, выполняется полный `--recheck-all` выбранного service.
|
||||
|
||||
### Типы
|
||||
|
||||
| Type | Что ставится в очередь |
|
||||
|---|---|
|
||||
| `network` | Текущие transient network statuses |
|
||||
| `ratelimited` | Limited/rate-limited и связанные no-balance statuses |
|
||||
| `unknown` | Unknown и no-context |
|
||||
| `restricted` | Restricted |
|
||||
| `nobalance` | No-balance/no-quota |
|
||||
| `valid` или `alive` | Текущие alive credentials |
|
||||
| `all` | Все известные credentials |
|
||||
| `legacy-vertex` | Только GCP: импортировать и проверить отсутствующие legacy Vertex TXT credentials |
|
||||
|
||||
### Опции
|
||||
|
||||
| Опция | Значение |
|
||||
|---|---|
|
||||
| `--force` | Остановить уже работающий keycheck batch и начать этот |
|
||||
| `--max-keys N` | Ограничить число credentials |
|
||||
| `--proxy-file PATH` | Временно переопределить proxy file |
|
||||
| `--no-resource-probe` | Отключить resource probe; сейчас используется Replicate |
|
||||
| `--no-summary` | Не пересобирать summary/status projections после batch |
|
||||
|
||||
`--input PATH` является legacy/offline compatibility option и в canonical PostgreSQL runtime не используется.
|
||||
|
||||
### Примеры
|
||||
|
||||
```text
|
||||
# Повторить только network failures у всех providers
|
||||
recheck all network
|
||||
|
||||
# Перепроверить все текущие alive GCP credentials
|
||||
recheck gcp valid
|
||||
|
||||
# Полностью перепроверить Qwen
|
||||
recheck qwen all
|
||||
|
||||
# Проверить максимум 5 alive Replicate без resource probe
|
||||
recheck replicate valid --max-keys 5 --no-resource-probe
|
||||
|
||||
# Импортировать/дедуплицировать старые GCP Vertex TXT записи и проверить только их
|
||||
recheck gcp legacy-vertex
|
||||
|
||||
# Прервать текущий keycheck batch и запустить новый
|
||||
recheck gcp valid --force
|
||||
```
|
||||
|
||||
Scheduled keychecks запускаются раз в `3600` секунд. По умолчанию проверяются новые candidates и повторяются только `NETWORK`; alive/limited/unknown/restricted/no-balance автоматически каждый час не перепроверяются.
|
||||
|
||||
## Текущие GCP Vertex Probes
|
||||
|
||||
| Provider | Модели | Locations | Проверка |
|
||||
|---|---|---|---|
|
||||
| Google | `gemini-3.6-flash`, `gemini-3.1-pro-preview` | `global`, `us`, `eu` | `countTokens`, без генерации |
|
||||
| Anthropic | `claude-opus-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-fable-5` | `global`, `us`, `eu`, `us-east5`, `europe-west1` | `rawPredict`, до 1 output token |
|
||||
|
||||
Anthropic probe является реальным минимальным inference-вызовом и может иметь небольшой расход.
|
||||
|
||||
## Куда идут данные
|
||||
|
||||
1. Scanner sources создают result bundles.
|
||||
2. `result-ingester` пишет findings и keycheck candidates в PostgreSQL.
|
||||
3. Provider checker арендует candidate и выполняет API probe через `runtime\proxy.txt`.
|
||||
4. Новый результат добавляется в append-only `keycheck_results`.
|
||||
5. `keycheck_current_state` переключается на последний результат.
|
||||
6. `jsonl-projector` создаёт compatibility JSONL.
|
||||
7. Summary projection атомарно обновляет status TXT.
|
||||
|
||||
PostgreSQL является source of truth. TXT/JSONL в `runtime\keychecks` являются compatibility projections, а не входом для обычного recheck.
|
||||
|
||||
## Полезные пути
|
||||
|
||||
| Путь | Назначение |
|
||||
|---|---|
|
||||
| `app\config.yaml` | Основная конфигурация runtime, sources и probes |
|
||||
| `runtime\proxy.txt` | Proxy для provider checks |
|
||||
| `runtime\logs\supervisor.status.txt` | Последний status snapshot |
|
||||
| `runtime\logs\supervisor.log` | Supervisor log |
|
||||
| `runtime\logs\keychecks.log` | Общий keycheck log |
|
||||
| `runtime\keychecks\summary.tsv` | Текущий provider summary |
|
||||
| `runtime\keychecks\alive_summary.tsv` | Краткий alive summary |
|
||||
| `runtime\keychecks\<service>` | Compatibility status/results files provider-а |
|
||||
| `runtime\control\supervisor.instance.json` | Private control metadata; вручную не редактировать |
|
||||
| `S:\postgres-data` | Canonical PostgreSQL cluster |
|
||||
| `S:\scanner-work` | Scanner scratch/work area |
|
||||
|
||||
## Безопасность
|
||||
|
||||
- Не запускай provider scripts напрямую.
|
||||
- Не передавай raw credentials через CLI.
|
||||
- Не редактируй `supervisor.instance.json`.
|
||||
- Для управления используй только authenticated supervisor.
|
||||
- Для полного рестарта используй canonical start/stop scripts.
|
||||
- Не удаляй PostgreSQL cluster или runtime queues вручную.
|
||||
@@ -0,0 +1,285 @@
|
||||
# Windows Snapshot Import
|
||||
|
||||
## Current Status
|
||||
|
||||
As of 2026-09-16, source capture succeeded, but the target import failed during
|
||||
maintenance cleanup and was manually interrupted. The destination remains
|
||||
**failed, unmarked and stopped**, not verified-stopped or ready for normal run.
|
||||
The user then explicitly authorized removal of copied SQLite databases,
|
||||
backups, `found_secrets` outputs and archival files, preserving Windows originals.
|
||||
|
||||
**The staged snapshot is now intentionally incomplete: `files.tar` was deleted.**
|
||||
Its retained manifest describes the original capture, not the reduced target.
|
||||
Do not rewrite the manifest, retry import, restart the retained container, or
|
||||
recapture/repopulate the removed copies automatically. The procedures below
|
||||
describe the original full-snapshot workflow, not a resume procedure for this
|
||||
pruned destination. Any future recovery needs a separately reviewed plan.
|
||||
|
||||
| Execution evidence | Result |
|
||||
| --- | --- |
|
||||
| Source snapshot publication | Manifest published 2026-09-15T19:36:28.010915+00:00; source supervisor/PG stopped flags true |
|
||||
| Approved manifest SHA-256 | `08344147133c37d4b6f404cf4fac3e59d58f94917f1fa58a77cbb68c36db7e8a` |
|
||||
| Original capture inventory | 50,501 files, 34,851,776,467 bytes; 49,897 active and 604 archival files; 61 tables and 38 sequences |
|
||||
| Retained PostgreSQL dump | `database.dump`, 2,619,119,892 bytes; not deleted or modified by cleanup |
|
||||
| Failed import container | `63286fd554f832fd3a1f073e7c923977e209149e940b684cdbf35e4479ebd5ba`; exited 129, PID 0, restarts 0, restart policy `no` |
|
||||
| Pinned runtime/cleanup image | `sha256:ecf1ee044fd6e936359a5955e0a42b452b8098a3f9d822272ab373b697761de2` |
|
||||
| Cleanup verification | 635 original Windows files checked for presence/size and unchanged metadata; original PG control hashes unchanged; Windows `postgres.exe` count 0 |
|
||||
| Retained destination verification | Metadata of 49,866 remaining inventory files and 1,883 PG files unchanged; PG control/config hashes unchanged; initialized marker absent; application remains stopped |
|
||||
| Final verified-stopped acceptance | NOT ACHIEVED; cleanup does not repair the failed import |
|
||||
|
||||
### Authorized Copy Cleanup
|
||||
|
||||
Only these copied locations were removed on 2026-09-16:
|
||||
|
||||
| Copied location/family | Files | Bytes removed |
|
||||
| --- | ---: | ---: |
|
||||
| `/data/windows-archive` including old SQLite backups and archived configurations | 604 | 10,827,425,254 |
|
||||
| `/data/runtime-linux/results/scanner*.db` and associated WAL/SHM/journal files | 11 | 10,281,779,360 |
|
||||
| `/data/runtime-linux/results/found_secrets.*`, including generations, manifest and publication ledger | 20 | 5,584,368,562 |
|
||||
| Total from native `truf-docker_data` volume | 635 | 26,693,573,176 |
|
||||
| Completed staging directory's `files.tar` on Windows D: | 1 | 34,917,959,680 |
|
||||
|
||||
Original `D:\truf`, `S:\postgres-data` and source bundles were not deleted or
|
||||
modified. Target PostgreSQL, its dump, translated configuration, credentials,
|
||||
proxies, queues, other result streams, bundles, caches and their required
|
||||
publication ledgers were retained. The one-off cleanup used a network-disabled
|
||||
utility container with only the verified native target volume writable; it did
|
||||
not run PostgreSQL, the importer, scanners, providers or application services.
|
||||
|
||||
The volume gained approximately 26.69 GB of filesystem free space. Approximately
|
||||
34.92 GB was freed on Windows D:. This did not compact the WSL VHDX on S: or
|
||||
return all newly free ext4 blocks to the Windows host; S: reported
|
||||
30,467,690,496 bytes free after cleanup. No WSL/storage reconfiguration was done.
|
||||
|
||||
Removing `found_secrets` files does not reset PostgreSQL projector cursors or
|
||||
pending append/rotation proofs. A future authorized startup must first address
|
||||
that projection state explicitly; removing files or their SQLite ledger alone
|
||||
is not a safe live-cursor reset. Do not erase PostgreSQL findings, counters or
|
||||
pipeline evidence to make the removed files appear consistent.
|
||||
|
||||
Reported regression evidence, not rerun by this documentation change: selected
|
||||
Docker suite **663 passed, 7 Windows-only skipped**; synthetic snapshot tests
|
||||
**58 passed**; pure config tests **12 passed**; host importer tests **66 passed,
|
||||
1 POSIX-only skipped**. None is proof of this snapshot's capture or import E2E.
|
||||
|
||||
## Scope And Paths
|
||||
|
||||
The authorized operation copies the original logical database, proxies, secrets
|
||||
and reviewed durable files. It does not move/delete the originals, migrate the
|
||||
source schema, or execute providers, scanners, keycheckers or archived scripts.
|
||||
Only exclusive PostgreSQL maintenance is allowed during capture/import. Leave
|
||||
both the original supervisor and original PostgreSQL stopped after capture,
|
||||
and the destination stopped after import.
|
||||
|
||||
Current private staging directory:
|
||||
|
||||
- Windows: `D:\truf-docker\docker\imports\windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3`
|
||||
- WSL: `/mnt/d/truf-docker/docker/imports/windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3`
|
||||
- Container: the same directory bound read-only at `/import`.
|
||||
|
||||
A completed snapshot contains exactly `manifest.json`, `files.tar` and
|
||||
`database.dump`. Do not add reports or other files inside it. Windows staging
|
||||
remains private to the capturing account and SYSTEM; preserve its ACLs rather
|
||||
than making it world-readable for Docker. `docker/imports/` is excluded from
|
||||
Git and the image build context. Never put dump/tar contents, credentials,
|
||||
proxy values, application data or raw logs in Git, images or terminal output.
|
||||
|
||||
The directory listed above currently retains only `manifest.json` and
|
||||
`database.dump` after the authorized cleanup. It is not a completed import input.
|
||||
|
||||
The destination is the base Compose native Linux volume `truf-docker_data`,
|
||||
mounted at `/data`, not a Windows bind mount. Paths in braces below are reviewed
|
||||
families; optional archival inputs are copied only when present.
|
||||
|
||||
| Original source | Destination within `/data` |
|
||||
| --- | --- |
|
||||
| `S:\postgres-data` via a full PG16 logical dump | `/data/postgres-linux`, independently initialized native Linux PG16 |
|
||||
| `D:\truf\runtime\{results,queues,state,keychecks,postman_cache,result_spool}` | `/data/runtime-linux/{results,queues,state,keychecks,postman_cache,result_spool}` |
|
||||
| `D:\truf\runtime\proxy.txt` | `/data/runtime-linux/proxy.txt` |
|
||||
| `D:\truf\app\{secrets.yaml,trufflehog-custom-detectors.yaml}` | `/data/config/{secrets.yaml,trufflehog-custom-detectors.yaml}` |
|
||||
| `D:\truf\app\config.yaml` | `/data/windows-archive/app/config.yaml`; translated profile at `/data/config/windows-import.yaml` |
|
||||
| `S:\scanner-result-bundles\{tmp,ready,quarantine}` | `/data/scanner-result-bundles/{tmp,ready,quarantine}` |
|
||||
| Reviewed archival inputs under `D:\truf` | `/data/windows-archive/` with their original relative paths |
|
||||
|
||||
Archival scope includes `D:\truf\state`, `runtime\backups`, `runtime\imports`,
|
||||
non-authority JSON reports from `runtime\control`, `app\.streamlit\config.toml`,
|
||||
`.env.postgres`, `docker-compose.postgres.yml`, `runner_state.json`,
|
||||
`runtime\keychecks.7z`, `runtime\orkey.txt`, `runtime\check-openrouter-keys.ps1`,
|
||||
`runtime\*.md`, root `checked_*.txt`/`todo_*.txt`, root/app `scanner.db*`,
|
||||
app `config.yaml.*`/`secrets.yaml.*`, root result projection families and
|
||||
`*.publication-ledger.sqlite3*`. The legacy copy tree is also archival:
|
||||
`D:\truf\runtime\keychecks \u2014 \u043a\u043e\u043f\u0438\u044f`
|
||||
(Unicode escapes describe the actual folder name, not a literal shell path).
|
||||
Within `runtime\state`, `scan_limiter*.db*`, `*.tmp*` and
|
||||
`janitor.cursor.json` are archival only, never active Linux authority/state.
|
||||
|
||||
Excluded: physical PGDATA/WAL, Windows PostgreSQL binaries/logs, live control
|
||||
authority, locks/PIDs, `S:\scanner-work`, `runtime\downloads`, runtime git/traces/
|
||||
freeze-diagnostics, `gharchive_cache`, `.git`, `.opencode`, tests and code caches.
|
||||
Ordinary logs are excluded outside retained result/keycheck projection families;
|
||||
`scan_errors.log*` is deliberately durable data, not a diagnostic to display.
|
||||
The manifest records the actual selected inventory and exclusion counts.
|
||||
|
||||
The archived `.env.postgres` is never sourced or used for the target connection.
|
||||
Provision generates `/data/postgres-password`; Linux uses this new
|
||||
password and a different PG16 system identifier, not the source password or
|
||||
physical cluster. Archived Windows configs are not executable runtime profiles.
|
||||
|
||||
## Capture And Capacity Gates
|
||||
|
||||
Main runs `D:\truf-docker\docker\windows_snapshot.py` using native PowerShell,
|
||||
not context-mode or another Windows Job wrapper: original PostgreSQL correctly
|
||||
refuses Job membership. The exporter temporarily starts only source maintenance
|
||||
PostgreSQL and must confirm its stop before publication. Do not rerun capture
|
||||
into the current attempt directory, reuse partial files, or kill a process that
|
||||
is retaining authority while stop remains unconfirmed.
|
||||
|
||||
After capture exits 0 and publishes its final manifest, main records its digest
|
||||
from the trusted Windows path, then checks that the WSL-visible manifest has
|
||||
the same digest. Do not replace the approved pin with a newly computed digest
|
||||
merely to bypass a mismatch. Native PowerShell digest command:
|
||||
|
||||
```powershell
|
||||
(Get-FileHash -LiteralPath 'D:\truf-docker\docker\imports\windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3\manifest.json' -Algorithm SHA256).Hash.ToLowerInvariant()
|
||||
```
|
||||
|
||||
Check Windows `D:` staging capacity and, independently, physical free space on
|
||||
`S:`, which backs the Docker/WSL VHDX, and free space on native Linux `/data`.
|
||||
Record the actual VHDX/daemon storage location; a large Linux `df` result does
|
||||
not prove the Windows host can grow the VHDX. Budget its anticipated growth
|
||||
while retaining at least **20 GiB physical free on S:**. The importer cannot
|
||||
measure or enforce this host-side reserve.
|
||||
|
||||
The Linux preflight requires `file_bytes + database_bytes + 20 GiB` free, using
|
||||
source physical database size from manifest metadata. Older v1 metadata without
|
||||
that size uses `max(24 GiB, 4 * dump_bytes)` as the database estimate. At least
|
||||
**20 GiB must still be free on Linux after import**. Check both host and guest
|
||||
capacity during and after restoration; compressed dump size alone is not a
|
||||
capacity estimate. Do not delete original data to make space.
|
||||
|
||||
## Offline Procedure
|
||||
|
||||
Run these steps separately in WSL Bash only after main approves the capture and
|
||||
capacity evidence. Use the existing Linux Docker daemon and already-built,
|
||||
importer-integrated `truf-local:runtime` image; no builds or pulls here. Keep the
|
||||
fixed project/directory below. Do not use the generic initialize/start procedure
|
||||
in `DOCKER_MIGRATION.md` for this full-schema snapshot.
|
||||
|
||||
```bash
|
||||
TRUF_WINDOWS_SNAPSHOT='/mnt/d/truf-docker/docker/imports/windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3'
|
||||
TRUF_WINDOWS_SNAPSHOT_SHA256='PENDING'
|
||||
|
||||
dci() {
|
||||
if [[ ! "${TRUF_WINDOWS_SNAPSHOT_SHA256:-}" =~ ^[0-9a-f]{64}$ ]]; then
|
||||
printf '%s\n' 'STOP: set the approved lowercase manifest SHA-256.' >&2
|
||||
return 1
|
||||
fi
|
||||
sudo -n env \
|
||||
TRUF_WINDOWS_SNAPSHOT="${TRUF_WINDOWS_SNAPSHOT:?Set the completed snapshot directory}" \
|
||||
TRUF_WINDOWS_SNAPSHOT_SHA256="$TRUF_WINDOWS_SNAPSHOT_SHA256" \
|
||||
docker compose \
|
||||
--project-name truf-docker \
|
||||
--project-directory /mnt/d/truf-docker \
|
||||
--env-file /dev/null \
|
||||
--file /mnt/d/truf-docker/compose.yaml \
|
||||
--file /mnt/d/truf-docker/compose.snapshot-import.yaml \
|
||||
"$@"
|
||||
}
|
||||
|
||||
sha256sum "$TRUF_WINDOWS_SNAPSHOT/manifest.json"
|
||||
dci config --quiet
|
||||
sudo -n docker image inspect --format '{{.Id}}' truf-local:runtime
|
||||
sudo -n docker volume inspect --format '{{.Name}} {{.Driver}} {{.Mountpoint}}' truf-docker_data
|
||||
sudo -n docker container inspect --format '{{.State.Status}}' truf-docker-snapshot-import
|
||||
```
|
||||
|
||||
Replace `PENDING` with the previously approved pin, not a credential. The
|
||||
`sudo -n env NAME=value ... docker compose` form explicitly passes the two
|
||||
non-secret interpolation inputs even when sudo strips shell exports. Do not
|
||||
use `sudo -E` or pass source connection/provider variables. `--env-file /dev/null`
|
||||
prevents implicit checkout `.env` loading, but does not sanitize shell exports;
|
||||
use a clean operator shell and no unreviewed Docker/Compose overrides.
|
||||
|
||||
Main must separately confirm the image identity, **absence** of
|
||||
`truf-docker_data`, and absence of the retained import container name before
|
||||
provisioning. Only specific no-such-volume/no-such-container responses establish
|
||||
absence; daemon/permission errors do not. If either already exists, stop for
|
||||
review instead of adopting, overwriting or deleting it.
|
||||
|
||||
```bash
|
||||
dci run --rm --no-deps --pull never -T provision
|
||||
```
|
||||
|
||||
Proceed only after successful provision. This unchanged base service creates
|
||||
the private layout, generated password and empty placeholders, not an application
|
||||
schema. **Do not call `initialize`, `import-secrets`, or normal `run` first.**
|
||||
The importer uses `initialize-empty` internally and restores the entire custom
|
||||
dump into a virgin schema before raw table/sequence comparison and permitted
|
||||
target-only recovery/migrations.
|
||||
|
||||
```bash
|
||||
dci run --detach --no-deps --pull never -T \
|
||||
--name truf-docker-snapshot-import runtime
|
||||
```
|
||||
|
||||
This is the single retained maintenance container: no `--rm`, no automatic
|
||||
restart, no dependency startup, no healthcheck and `network_mode: none`.
|
||||
The override preserves the base image/entrypoint, non-root UID/GID, read-only
|
||||
rootfs, capabilities/security policy, native `/data` volume, tmpfs and resource
|
||||
limits. It adds only read-only `/import`; `create_host_path: false` rejects a
|
||||
missing source directory rather than silently creating one.
|
||||
|
||||
Import starts with `/opt/truf/app/config.linux.yaml` in both the environment
|
||||
and explicit `--config`. **Do not merge `compose.windows-import.yaml` here.**
|
||||
The translated private profile does not exist at initial preflight; the importer
|
||||
creates and selects it internally only after validating/extracting the snapshot.
|
||||
|
||||
## Stopped Verification
|
||||
|
||||
```bash
|
||||
sudo -n docker container wait truf-docker-snapshot-import
|
||||
sudo -n docker container inspect --format \
|
||||
'status={{.State.Status}} exit={{.State.ExitCode}} oom={{.State.OOMKilled}} restarts={{.RestartCount}} network={{.HostConfig.NetworkMode}} restart={{.HostConfig.RestartPolicy.Name}} auto_remove={{.HostConfig.AutoRemove}}' \
|
||||
truf-docker-snapshot-import
|
||||
```
|
||||
|
||||
Require `status=exited exit=0 oom=false restarts=0 network=none restart=no
|
||||
auto_remove=false`. `wait` prints the container exit code; the command's own
|
||||
exit status alone is not import success. Waiting may take hours or hold while
|
||||
maintenance stop is uncertain. Do not impose a timeout that kills the container.
|
||||
|
||||
Do not display raw `docker logs`, Compose logs, database logs, full environment
|
||||
dumps or application data. Only structured importer numeric phase/count/byte
|
||||
events and allowlisted aggregate/hash evidence are suitable for progress.
|
||||
Phase 13 is emitted before final publication and is not success proof.
|
||||
|
||||
Main must privately inspect the following evidence from the stopped retained
|
||||
container, for example with `docker cp` into a separate owner-only evidence
|
||||
directory outside `/import`. Do not start another runtime to inspect it; `health`
|
||||
and `status` are live readiness actions, not stopped-import verification.
|
||||
|
||||
- `/data/config/windows-import-manifest.json`: its byte SHA-256 equals the approved staging manifest pin; source stopped flags are true. The raw evidence, report and initialized marker all carry that same `manifest_sha256`. Report archive/dump hashes match this manifest and the verified staging files.
|
||||
- `/data/config/windows-import-raw.json`: its SHA-256 matches report `raw_evidence_sha256`; `table_counts` equals manifest `database.table_counts`, `sequences_provided` is true and `sequences_verified` equals manifest `database.sequence_count`. The importer checks actual sequence values before transformations; this evidence records their verified count, not their values. Record only aggregate tables/rows/sequences.
|
||||
- `/data/config/windows-import.yaml`: hash bytes without displaying values; SHA-256 matches report `config_sha256`.
|
||||
- `/data/config/windows-import-report.json`: `status` is `verified-stopped`, `maintenance_stopped` is true, all `pipeline_after` counts are zero, final Linux reserve is at least 20 GiB, and cutover/migration/preserved-evidence checks succeeded. Record any reported fenced recovery or Postman rebasing; these may legitimately change final target counts or move incoming tmp/ready bundles after raw comparison.
|
||||
- `/data/initialized.json`: exists as the last publication, has format `truf-container-data-v1`, `pg_major` 16 and the approved `manifest_sha256`; `import_report_sha256` matches the actual report bytes. Its system identifier equals report `linux_system_identifier` and differs from the source identifier.
|
||||
- Confirm destination `/data/postgres-linux/postmaster.pid` is absent, original supervisor/PostgreSQL remain stopped, and host/guest space reserves still hold. Record all results in the PENDING table before declaring verified-stopped.
|
||||
|
||||
## Failure And Release
|
||||
|
||||
Any nonzero exit, OOM, missing/mismatched evidence or uncertain stop is not
|
||||
verified-stopped. Retain the container, volume and private snapshot. A partial
|
||||
import is failed/unmarked, not automatically resumable; early rejection can
|
||||
leave no report. A report saying verified-stopped without a matching final
|
||||
initialized marker and clean container exit is still insufficient.
|
||||
|
||||
Do not automatically retry/restart, overwrite, delete, remove locks/markers,
|
||||
initialize, prune, run `down --volumes`, or force-kill an authority-holding
|
||||
importer. Review the retained state first; any new attempt needs an explicit
|
||||
decision and separately approved fresh destination, not cleanup by this runbook.
|
||||
|
||||
There is **no automatic normal run**. A future live run requires separate user
|
||||
authorization after verified-stopped acceptance. Only then use base
|
||||
`compose.yaml` plus `compose.windows-import.yaml`, without the snapshot override,
|
||||
so runtime, health and status all select `/data/config/windows-import.yaml`.
|
||||
Do not start either the original or destination supervisor as part of import.
|
||||
@@ -0,0 +1,398 @@
|
||||
# Worker Operator Experience Handoff
|
||||
|
||||
> Historical handoff. The current continuation entry point is
|
||||
> `docs/session-handoff/README.md`. This file retains implementation and artifact
|
||||
> provenance, but its stop point and immediate-next-actions section are obsolete.
|
||||
|
||||
Last updated: 2026-09-25
|
||||
|
||||
This is the historical implementation record for the OpenSpec change
|
||||
`add-worker-operator-experience`. For current continuation instructions, read
|
||||
`docs/session-handoff/README.md`. Do not repeat completed production validation
|
||||
or rebuild accepted artifacts unless a current verification fails.
|
||||
|
||||
## User intent and constraints
|
||||
|
||||
- Continue autonomously from this handoff and finish the change end to end.
|
||||
- The user explicitly requested a file handoff because invoking conversation
|
||||
compression appears to stop or destabilize all OpenCode sessions. Avoid
|
||||
proactively invoking the compression tool in the continuation session.
|
||||
- Workspace: `D:\truf-workers`.
|
||||
- Use the configured SSH server named `sec` only. Never call or connect through
|
||||
the configured server named `prod`.
|
||||
- Never print or record tokens, credentials, the private admin prefix, raw
|
||||
targets/findings, runtime YAML, or worker command lines containing auth data.
|
||||
- Do not add a new masking, redaction, credential-sandbox, or other security
|
||||
scope without explicit approval and an OpenSpec requirement.
|
||||
- Do not remove or revert unrelated workspace files. The repository baseline is
|
||||
entirely untracked (`git status --short` shows the whole tree as `??`), so Git
|
||||
cannot provide a meaningful task-specific diff.
|
||||
- Do not archive the OpenSpec change unless the user explicitly asks. Completing
|
||||
tasks and reporting "ready to archive" is expected.
|
||||
|
||||
## Current OpenSpec state
|
||||
|
||||
- Change: `add-worker-operator-experience`
|
||||
- Schema: `spec-driven`
|
||||
- Artifact status: proposal, design, specs, and tasks are complete.
|
||||
- Apply progress before final closure: 23/27 tasks complete.
|
||||
- File: `openspec/changes/add-worker-operator-experience/tasks.md`
|
||||
- Tasks 1.1 through 5.5 are checked.
|
||||
- Remaining unchecked tasks:
|
||||
- 6.1: validate the canonical from-zero operator guide.
|
||||
- 6.2: complete test matrix, reproducible packages, manifest registration,
|
||||
and documented identities.
|
||||
- 6.3: bounded Windows and WSL/Docker production validation and restoration.
|
||||
- 6.4: durable dated report with timings, watchdog evidence, snapshots,
|
||||
transcript, known limits, rollout, and rollback.
|
||||
|
||||
Do not check 6.1-6.4 until the remaining focused matrix, report, and strict
|
||||
OpenSpec validation have passed.
|
||||
|
||||
## Implemented scope
|
||||
|
||||
The change now includes:
|
||||
|
||||
- Versioned worker phase/events and canonical transitions.
|
||||
- Monotonic sequence handling and JSON/NDJSON contracts.
|
||||
- Unified bounded diagnostics with deterministic identities.
|
||||
- PostgreSQL progress/diagnostic persistence and admin queries.
|
||||
- Server-owned global/per-source assignment deadline policy.
|
||||
- Private local state, logs, history, diagnostic artifacts, and retention.
|
||||
- Status, attach, logs/history, JSON/NDJSON, bounded follow/tail CLI behavior.
|
||||
- Cross-platform worker supervisor, control protocol, drain/stop, shutdown
|
||||
receipts, stale instance handling, and recovery slots.
|
||||
- Per-assignment contained runner, controller protocol, watchdog, timeout bundle
|
||||
publication, restart adoption, and abandoned-root cleanup.
|
||||
- Authenticated progress endpoint and diagnostic ingestion.
|
||||
- Admin assignment/progress/diagnostic experience.
|
||||
- Windows portable and Linux image packaging for the supervisor runtime.
|
||||
- Canonical operator runbook in `docs/remote-worker-operations.md`.
|
||||
|
||||
Important implementation files include:
|
||||
|
||||
- `app/worker_contracts.py`
|
||||
- `app/worker_local_state.py`
|
||||
- `app/worker_supervisor.py`
|
||||
- `app/worker_cli.py`
|
||||
- `app/worker_assignment_runner.py`
|
||||
- `app/remote_worker_client.py`
|
||||
- `app/worker_api.py`
|
||||
- `app/scanner_db.py`
|
||||
- `app/admin_api.py`
|
||||
- `app/worker_package.py`
|
||||
- `app/worker_package_builder.py`
|
||||
- `docker/verify_packaged_workers.py`
|
||||
- Worker-related tests under `tests/`
|
||||
|
||||
## Final correctness fixes
|
||||
|
||||
### Recovered ready-bundle transition
|
||||
|
||||
`WorkerSlot` could recover a published ready bundle while its persisted event
|
||||
phase was still `assigned`. Upload code emitted `uploading` only from
|
||||
`bundling/backoff`, then attempted the invalid transition
|
||||
`assigned -> awaiting_receipt`.
|
||||
|
||||
Fix in `app/remote_worker_client.py`:
|
||||
|
||||
- Emit `UPLOADING` when the current event phase is `ASSIGNED`, as well as the
|
||||
existing bundling/backoff cases.
|
||||
- Regression in `tests/test_worker_api.py` validates the event sequence
|
||||
`assigned -> uploading -> awaiting_receipt`.
|
||||
|
||||
The focused worker API/local-state/supervisor suite passed 100 tests after this
|
||||
fix.
|
||||
|
||||
### Packaged E2E abandoned work invariant
|
||||
|
||||
Completed runner roots are intentionally retained under `work/abandoned` for at
|
||||
least 60 seconds; retention maintenance normally runs every 300 seconds. The E2E
|
||||
harness incorrectly required the total work file count to be zero, causing a
|
||||
false `linux_direct_claims_timeout` after Linux had correctly claimed both direct
|
||||
assignments.
|
||||
|
||||
Fix in `docker/verify_packaged_workers.py`:
|
||||
|
||||
- Linux and Windows work-tree identities now include `active_entries`.
|
||||
- Files/directories beneath top-level `abandoned` are retained but not active.
|
||||
- Direct-assignment and final-cleanup predicates require zero active entries,
|
||||
while preserving strict state and bundle identity checks.
|
||||
- Outage marker waits also check worker liveness, so an exited worker fails
|
||||
immediately rather than timing out after four minutes.
|
||||
|
||||
### Windows `prepare-worker.ps1` ACL defect
|
||||
|
||||
Testing a freshly extracted ZIP exposed a real release bug. The old generated
|
||||
script ran `icacls ... /grant:r ... /T`; on descendants this produced
|
||||
inheritance-only ACEs, returned success, and made packaged `python.exe`
|
||||
inaccessible.
|
||||
|
||||
Final fix in `app/worker_package_builder.py`:
|
||||
|
||||
1. Set private inheritable full-control ACEs for the current user, SYSTEM, and
|
||||
Administrators on the package root only.
|
||||
2. Run `icacls (Join-Path $root '*') /inheritance:d /T /C` so each descendant
|
||||
converts inherited ACLs to explicit protected ACLs with the correct file or
|
||||
directory flags.
|
||||
|
||||
Regression in `tests/test_worker_package.py` checks the generated script and, on
|
||||
Windows, executes it and verifies `private_directory_ready(root)` plus
|
||||
`private_file_ready(child)`. `tests/test_worker_package.py` passes 20 tests.
|
||||
|
||||
Do not use the earlier `/reset /T` idea: inherited ACLs are not accepted because
|
||||
runtime trust requires protected explicit ACLs.
|
||||
|
||||
### Watchdog test timing stabilization
|
||||
|
||||
The broad focused suite exposed two false failures because three tests created a
|
||||
100 ms absolute watchdog deadline before runner protocol-root/state setup. Under
|
||||
the complete Windows suite that setup could consume the deadline, exercising the
|
||||
startup-deadline branch instead of the intended blocked-operation watchdog.
|
||||
|
||||
Test-only changes in `tests/test_worker_assignment_runner.py`:
|
||||
|
||||
- Affected tests:
|
||||
- `test_watchdog_kills_while_state_persistence_is_blocked`
|
||||
- `test_blocked_startup_gate_write_enters_preparing_timeout_result_path`
|
||||
- `test_watchdog_kills_while_event_drain_is_blocked`
|
||||
- Scan deadline: 1 second -> 2 seconds.
|
||||
- Watchdog deadline: 0.1 second -> 1 second.
|
||||
- Injected block: 0.4 second -> 1.4 seconds.
|
||||
- Kill bound: 0.3 second -> 1.3 seconds.
|
||||
|
||||
This preserves the independent watchdog assertion and does not weaken product
|
||||
code. The exact three-test rerun passed: `3 passed in 5.17s`.
|
||||
|
||||
## Accepted reproducible artifacts
|
||||
|
||||
### Windows final pair: I and J
|
||||
|
||||
Paths:
|
||||
|
||||
- `build/operator-experience-validation/windows-i.zip`
|
||||
- `build/operator-experience-validation/windows-i.zip.json`
|
||||
- `build/operator-experience-validation/windows-j.zip`
|
||||
- `build/operator-experience-validation/windows-j.zip.json`
|
||||
|
||||
Both independently built archives are identical:
|
||||
|
||||
- Bytes: `134850988`
|
||||
- Archive SHA-256:
|
||||
`6ea9290736a059f1e17d8e89d9cf83506fa4abe2ba2f3731a7422a7b0f386e97`
|
||||
- Package manifest identity:
|
||||
`78a962b2bd3fa411413c79e9a8ffb021608a08ff020b1ad851f4505ea634b2b6`
|
||||
- Build-input identity:
|
||||
`6991ebbce6ae758c2bdd19a6ae934335aa585a50f86b18ccde8d88bca40ce436`
|
||||
- Raw `worker-package.json` SHA-256:
|
||||
`e0b17d70fcb868fe39fac45ab6e05a17c6d40852e6034010fb63b6cab31f8a3c`
|
||||
|
||||
Acceptance used a fresh extraction, not the builder output:
|
||||
|
||||
- `build/pwe-final-i-extracted`
|
||||
- The package's own corrected `prepare-worker.ps1` was run once.
|
||||
- Direct package verification then passed.
|
||||
|
||||
The older Windows G/H archives are obsolete for acceptance because they contain
|
||||
the broken preparation script. Their package manifest identity happens to be the
|
||||
same because the support script is outside that manifest, but their archive
|
||||
identity is not accepted. Do not publish or register G/H as final Windows ZIPs.
|
||||
|
||||
### Linux final pair: G and H
|
||||
|
||||
Tags:
|
||||
|
||||
- `truf-worker-test:operator-experience-final-3g`
|
||||
- `truf-worker-test:operator-experience-final-3h`
|
||||
|
||||
Both were built with provenance disabled and are reproducible:
|
||||
|
||||
- Worker package identity:
|
||||
`45588f2cf406b41b239cfa3b8a9dc83fe84b587229bc997b2729016e1f0dde42`
|
||||
- Image manifest / accepted image ID:
|
||||
`sha256:3a088f5743121d823aae132234a29730a84339cecbfda5fc601e8e942f9948c3`
|
||||
- Config:
|
||||
`sha256:687a1c4c51c1b962c7fa7ea0cc4b04d159e7ba4f94ef347940c9fb225f7cb87d`
|
||||
- Raw `worker-package.json` SHA-256:
|
||||
`ee926cce3c19e9e6094753f51fa902415bd7364c24fa649cd0c1b659c0aa4d60`
|
||||
|
||||
Extracted final manifest:
|
||||
|
||||
- `build/operator-experience-validation/linux-worker-package-g.json`
|
||||
|
||||
Test image:
|
||||
|
||||
- Tag: `truf-worker-test:operator-experience-final-3`
|
||||
- ID:
|
||||
`sha256:1a22c396dbf329e20caf77f88b7c7a310bda86befbcf3b917f10ded3720ee712`
|
||||
|
||||
## Final packaged E2E
|
||||
|
||||
Passed run:
|
||||
|
||||
- Run ID: `35f3f52e232067c1`
|
||||
- Safe summary: `build/pwe-35f3f52e232067c1/summary.json`
|
||||
- Windows input: freshly extracted and prepared Windows I.
|
||||
- Linux input: Linux G.
|
||||
- Status: passed.
|
||||
- Cleanup: complete.
|
||||
- Foreign Docker state: unchanged.
|
||||
- Windows and Linux normalized evidence matched.
|
||||
- Restart, outage, durable bundle, direct assignment, direct bundle, receipt,
|
||||
shutdown, local cleanup, and cross-platform evidence gates all passed.
|
||||
|
||||
Do not copy raw target values from the summary into reports or chat. Only the
|
||||
safe aggregate facts above are needed.
|
||||
|
||||
All Docker resources from final and diagnosed failed runs were cleaned by exact
|
||||
owned IDs/names. Some local `build/pwe-*` failure evidence directories remain and
|
||||
are safe to leave. `build/pwe-final-g` may still have unusable ACLs after running
|
||||
the old broken preparation script; do not use it. `build/pwe-final-i-extracted`
|
||||
is the accepted extracted Windows directory.
|
||||
|
||||
## Production validation evidence
|
||||
|
||||
Bounded production validation was completed before final package acceptance and
|
||||
production was restored afterward.
|
||||
|
||||
Safe aggregate results:
|
||||
|
||||
- Assignments issued: 34.
|
||||
- Accepted: 33.
|
||||
- One intentional expected expiry.
|
||||
- Accepted assignments ingested, settled, and projected: 33.
|
||||
- Unresolved, precommit, and quarantine counts: zero.
|
||||
- Natural timeout evidence reservation: 1453.
|
||||
- Full-stage progress/watchdog evidence reservation: 1455.
|
||||
|
||||
Evidence files:
|
||||
|
||||
- `build/operator-experience-validation/final-evidence.json`
|
||||
- `build/operator-experience-validation/progress-v3-evidence.json`
|
||||
- `build/operator-experience-validation/timeout-evidence.json`
|
||||
- `build/operator-experience-validation/server-baseline.json`
|
||||
|
||||
These files are the source for duration percentiles, phase/watchdog evidence,
|
||||
diagnostic/admin snapshots, and reconciled counts in the final report. Derive
|
||||
only aggregate/sanitized facts. Do not reproduce raw targets, findings, secrets,
|
||||
or private route names.
|
||||
|
||||
Final production state after restoration:
|
||||
|
||||
- Operations controls: normal/open, revision 126.
|
||||
- Standard WSL production worker user: enabled, assignment cap 1.
|
||||
- Standard production device: enabled and not revoked.
|
||||
- Temporary validation identities: disabled/revoked.
|
||||
- Runtime canonical health: healthy.
|
||||
- Edge remained up.
|
||||
|
||||
Do not repeat production assignments merely to write the report. Existing
|
||||
evidence is sufficient.
|
||||
|
||||
## Registered trusted manifests
|
||||
|
||||
Registration was completed only after the final packaged E2E passed, using SSH
|
||||
server `sec` only.
|
||||
|
||||
Remote paths:
|
||||
|
||||
- `/etc/truf/worker-packages/linux-worker-package-v2.json`
|
||||
- `/etc/truf/worker-packages/windows-worker-package-v3.json`
|
||||
|
||||
Final remote SHA-256 values match the accepted manifests:
|
||||
|
||||
- Linux: `ee926cce3c19e9e6094753f51fa902415bd7364c24fa649cd0c1b659c0aa4d60`
|
||||
- Windows: `e0b17d70fcb868fe39fac45ab6e05a17c6d40852e6034010fb63b6cab31f8a3c`
|
||||
|
||||
Both are `root:root` mode `0644`. Existing
|
||||
`.pre-operator-experience` backups were preserved unchanged. Upload temp files
|
||||
were removed. After registration, canonical runtime health succeeded and Docker
|
||||
reported `truf-docker-runtime-1` healthy. No restart or config mutation was
|
||||
needed.
|
||||
|
||||
## Test state
|
||||
|
||||
Completed checks:
|
||||
|
||||
- Worker API/local-state/supervisor focused suite: 100 passed.
|
||||
- Worker package tests after ACL fix: 20 passed.
|
||||
- Exact three watchdog timing tests after stabilization: 3 passed.
|
||||
- Full packaged Windows/Linux E2E: passed, run `35f3f52e232067c1`.
|
||||
- Production health after final manifest registration: passed.
|
||||
|
||||
The broad focused matrix was run before the watchdog test timing patch:
|
||||
|
||||
```powershell
|
||||
python -B -m pytest tests/test_worker_api.py tests/test_worker_api_runtime.py tests/test_worker_assignment.py tests/test_worker_assignment_runner.py tests/test_worker_cli.py tests/test_worker_contracts.py tests/test_worker_local_state.py tests/test_worker_observability_db.py tests/test_worker_package.py tests/test_worker_runner_handoff_linux.py tests/test_worker_supervisor.py tests/test_remote_worker_db.py tests/test_scan_execution.py tests/test_admin_api.py -q
|
||||
```
|
||||
|
||||
Result before the timing-only patch:
|
||||
|
||||
- 353 passed.
|
||||
- 3 skipped.
|
||||
- 2 false timing failures described above.
|
||||
|
||||
The two failures and the nearby equivalent test pass after the patch, but the
|
||||
complete 14-file command has not yet been rerun. This is the exact current stop
|
||||
point.
|
||||
|
||||
An unrestricted repository-wide pytest run is not a useful release gate in this
|
||||
checkout because unrelated private/generated assets and platform assumptions are
|
||||
absent. Its known baseline was `3031 passed, 134 skipped, 68 failed`. Do not try
|
||||
to fix unrelated failures as part of this change. The focused change matrix,
|
||||
packaged E2E, production proof, and strict OpenSpec validation are the gates.
|
||||
|
||||
## Immediate next actions
|
||||
|
||||
1. Rerun the exact 14-file focused matrix shown above. Expected result after the
|
||||
timing patch is 355 passed and 3 skipped. If it fails, diagnose only genuine
|
||||
worker-operator regressions; do not broaden scope.
|
||||
2. Create the durable report:
|
||||
`docs/worker-operator-experience-validation-2026-09-24.md`.
|
||||
3. In the report, include only sanitized aggregate evidence:
|
||||
- Scope and acceptance criteria.
|
||||
- Final Windows I/J and Linux G/H identities from this handoff.
|
||||
- Packaged E2E run `35f3f52e232067c1` and cleanup/foreign-state result.
|
||||
- Production issued/accepted/reconciled counts.
|
||||
- Duration percentiles derived from `final-evidence.json`.
|
||||
- Watchdog/full-stage evidence from `progress-v3-evidence.json`.
|
||||
- Natural timeout evidence from `timeout-evidence.json`.
|
||||
- Diagnostic/admin snapshot facts without private content.
|
||||
- Sanitized operator command transcript.
|
||||
- Known limits, especially no public registry/auto-updater and intentional
|
||||
abandoned-root retention.
|
||||
- Rollout and rollback/restoration facts, controls revision 126, and final
|
||||
healthy state.
|
||||
4. Re-read `docs/remote-worker-operations.md` against task 6.1. It already covers
|
||||
package acquisition/build, Windows preparation, install/first run, lifecycle,
|
||||
status/attach/logs/history, phases/deadlines, diagnostics, drain/stop,
|
||||
recovery, update, and removal. Make only a minimal correction if the final
|
||||
artifact/report facts expose an actual gap.
|
||||
5. Run strict validation:
|
||||
|
||||
```powershell
|
||||
openspec validate add-worker-operator-experience --strict
|
||||
```
|
||||
|
||||
6. If the focused matrix, report, runbook review, and strict validation pass,
|
||||
change only task checkboxes 6.1-6.4 in
|
||||
`openspec/changes/add-worker-operator-experience/tasks.md` from `[ ]` to `[x]`.
|
||||
7. Re-run `openspec instructions apply --change "add-worker-operator-experience" --json`
|
||||
and confirm progress 27/27 with state `all_done`.
|
||||
8. Give the user a concise completion result and say the change is ready to
|
||||
archive. Do not archive it without an explicit request.
|
||||
|
||||
## Report safety checklist
|
||||
|
||||
Before saving or quoting the final report, verify it contains none of:
|
||||
|
||||
- Tokens or credentials.
|
||||
- Raw worker targets or findings.
|
||||
- Runtime YAML or secret environment values.
|
||||
- The private admin route prefix.
|
||||
- Worker argv/auth command lines.
|
||||
- Unbounded log or diagnostic bodies.
|
||||
|
||||
Allowed report content includes hashes, aggregate counts, reservation numeric
|
||||
IDs used as evidence references, phase names, durations/percentiles, safe test
|
||||
counts, generic command names, and public artifact paths within this workspace.
|
||||
@@ -0,0 +1,40 @@
|
||||
# Remote Worker Development Workspace
|
||||
|
||||
This is an independent source-only snapshot of the current `D:\truf-docker` working tree, not a copy of its running system.
|
||||
|
||||
## Snapshot
|
||||
|
||||
- Source HEAD for provenance: `1b3c7fc4948c5cf2fc389065db3693a27b65300c`.
|
||||
- Current modified and selected untracked source files are included; this snapshot is not equivalent to that commit alone.
|
||||
- Copied 325 files, 8,150,338 bytes (about 7.8 MiB), with SHA-256 equality checked for every copied file.
|
||||
- Source-side deletions are preserved, including the absence of `app/config.yaml`.
|
||||
- No database, PGDATA, runtime directory, finding/keycheck output, logs, imports, caches, dependencies, or executable binaries were copied.
|
||||
- No actual `.env`, secrets file, provider credential pool, or source Git history was copied. The tracked `.env.postgres.example` is only a template.
|
||||
- Git was initialized independently. No commit, remote, runtime container, or Docker data volume was created for this workspace.
|
||||
- Source `.opencode` skills, the old session handoff, and the loose operator note were not copied. Existing OpenSpec change artifacts remain unchanged and unarchived.
|
||||
|
||||
## Active Plan
|
||||
|
||||
`openspec/changes/add-minimal-remote-scan-workers/` contains the completed proposal, design, requirements, implementation checklist, and isolated worker implementation. It preserves existing scan and server-side keycheck logic.
|
||||
|
||||
## Safe Local Checks
|
||||
|
||||
Run from this directory:
|
||||
|
||||
```powershell
|
||||
python -I -S -B docker/test_verify.py -v
|
||||
python -I -S -B tests/container_unit.py --check-selection
|
||||
openspec validate add-minimal-remote-scan-workers --strict --no-interactive
|
||||
```
|
||||
|
||||
The first two commands use the standard library only, do not import the application, and do not start Docker, PostgreSQL, a scanner, or provider checks. The selection check validates test declarations, not their execution.
|
||||
|
||||
The original planning-stage verification passed. Current implementation evidence is recorded by the change checklist and isolated test outputs, including cross-platform packaged-client scans and the empty-database end-to-end gates.
|
||||
|
||||
## Before Runtime Testing
|
||||
|
||||
The inherited deployment files are SOURCE REFERENCES, not an isolated test setup. In particular, `compose.yaml` still names the production-style `truf-docker` project and shared `truf-local:*` image tags; `docker-compose.postgres.yml` and import overrides must not be used here. Do not run plain `docker compose up`, import a snapshot, invoke old native launchers, or run unrestricted pytest.
|
||||
|
||||
The first implementation tasks must establish unique test project/image/volume names, neutral configuration, scrubbed inherited credentials/DSNs/proxies, disabled live discovery, and synthetic source/provider transports. Review existing `compose.e2e.yaml`, `docker/verify.py`, and `tests/container_unit.py` for reuse before adding any new test infrastructure. Never mount `D:\truf`, `D:\truf-docker`, their runtime directories, or existing Docker data volumes. A future test database must initialize empty and contain only synthetic fixtures.
|
||||
|
||||
Local test implementation and worker packaging must use this workspace, not the active source or runtime. Production deployment/import and archiving unrelated changes require separate authorization.
|
||||
@@ -0,0 +1,10 @@
|
||||
[server]
|
||||
headless = true
|
||||
address = "127.0.0.1"
|
||||
port = 5000
|
||||
|
||||
[theme]
|
||||
base = "light"
|
||||
|
||||
[browser]
|
||||
gatherUsageStats = false
|
||||
@@ -0,0 +1,289 @@
|
||||
# Truf Runtime Operations
|
||||
|
||||
All scanner, keycheck, dashboard, and PostgreSQL lifecycle mutation is owned by `supervisor.py`. Direct `console_runner.py`, mutating `keycheck_runner.py`, and legacy `app.py` controls are retired.
|
||||
|
||||
PostgreSQL is the sole authority for scan results, queue completion, keycheck results, and keycheck current state. Scanner sources publish private version-2 bundles to `S:\scanner-result-bundles`; the singleton result ingester commits them transactionally. JSONL and status files are asynchronous, rebuildable compatibility projections and may lag without rolling back a committed scan.
|
||||
|
||||
The scanner has `max_active_scans=3` guaranteed fair permits plus at most one memory-gated non-Docker bonus permit. A permit covers staging, TruffleHog, normalization, bundle fsync, and the atomic ready rename only. PostgreSQL ingestion, JSONL projection, and keychecks do not hold scan permits.
|
||||
|
||||
Run commands from `D:\truf\app` unless a full path is shown.
|
||||
|
||||
## Start And Stop
|
||||
|
||||
Canonical production start:
|
||||
|
||||
```powershell
|
||||
..\start_runtime.ps1
|
||||
```
|
||||
|
||||
Canonical full coordinated shutdown:
|
||||
|
||||
```powershell
|
||||
..\stop_runtime.ps1
|
||||
```
|
||||
|
||||
Equivalent authenticated launch after cluster identity has been verified:
|
||||
|
||||
```powershell
|
||||
python -I -S -B runtime_bootstrap.py supervisor -- --runtime-bootstrap-entrypoint D:\truf\app\supervisor.py --config config.yaml --background --no-dashboard --with-postgres
|
||||
```
|
||||
|
||||
Direct read/control operations remain supported by `supervisor.py`:
|
||||
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --background-status
|
||||
..\attach_runtime.ps1
|
||||
python supervisor.py --config config.yaml --cmd "status"
|
||||
python supervisor.py --config config.yaml --stop-background --with-postgres
|
||||
```
|
||||
|
||||
In an attached prompt, `q` only detaches; `shutdown` requests full coordinated shutdown. Prefer `..\stop_runtime.ps1` for canonical full shutdown.
|
||||
|
||||
The dashboard is currently disabled. To use the read-only dashboard, enable it in `config.yaml` and perform a coordinated runtime restart. The legacy scanner UI is intentionally retired.
|
||||
|
||||
## Source Commands
|
||||
|
||||
Use the foreground supervisor prompt or authenticated `--cmd` requests:
|
||||
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "status"
|
||||
python supervisor.py --config config.yaml --cmd "start github"
|
||||
python supervisor.py --config config.yaml --cmd "once gitlab"
|
||||
python supervisor.py --config config.yaml --cmd "restart dockerhub"
|
||||
python supervisor.py --config config.yaml --cmd "pause npm"
|
||||
python supervisor.py --config config.yaml --cmd "resume npm"
|
||||
python supervisor.py --config config.yaml --cmd "stop package_git"
|
||||
python supervisor.py --config config.yaml --cmd "stop all"
|
||||
python supervisor.py --config config.yaml --cmd "start all"
|
||||
python supervisor.py --config config.yaml --cmd "logs pypi 80"
|
||||
python supervisor.py --config config.yaml --cmd "command github"
|
||||
```
|
||||
|
||||
`stop all` stops managed children while the supervisor and PostgreSQL remain running. `start all` includes `pypi`; do not use it when `pypi` must remain stopped.
|
||||
|
||||
Configure discovery mode, queries, custom target files, timeouts, workers, and source-specific arguments in `config.yaml` before starting or restarting a source. Do not pass tokens or mutable scan options through a direct runner command.
|
||||
|
||||
## Keychecks
|
||||
|
||||
The supervisor manages keychecks as the `keychecks` pseudo-source:
|
||||
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "start keychecks"
|
||||
python supervisor.py --config config.yaml --cmd "recheck all network"
|
||||
python supervisor.py --config config.yaml --cmd "recheck gemini all --max-keys 100"
|
||||
python supervisor.py --config config.yaml --cmd "recheck replicate valid --max-keys 25"
|
||||
python supervisor.py --config config.yaml --cmd "recheck all --summary-only"
|
||||
python supervisor.py --config config.yaml --cmd "logs keychecks 80"
|
||||
```
|
||||
|
||||
Provider probe arguments belong under `keychecks.service_args` in `config.yaml`. Normal providers claim fenced PostgreSQL `keycheck_candidates` and commit `keycheck_results` plus `keycheck_current_state` directly. Files under `D:\truf\runtime\keychecks` are compatibility projections, not current-state authority. At most four provider children run concurrently, each with a bounded candidate slice so later services cannot starve.
|
||||
|
||||
## Configuration And Secrets
|
||||
|
||||
Primary files:
|
||||
|
||||
```text
|
||||
D:\truf\app\config.yaml
|
||||
D:\truf\app\secrets.yaml
|
||||
D:\truf\.env.postgres
|
||||
D:\truf\runtime\proxy.txt
|
||||
```
|
||||
|
||||
`config.yaml` contains paths, source settings, and auth-pool names. Actual source tokens belong in `secrets.yaml`; PostgreSQL credentials belong in `.env.postgres`. Managed children receive one canonical loopback PostgreSQL DSN after all configurable environment overrides.
|
||||
|
||||
Important runtime paths:
|
||||
|
||||
```text
|
||||
D:\truf\runtime\results
|
||||
S:\scanner-result-bundles
|
||||
D:\truf\runtime\result_spool (legacy import compatibility only)
|
||||
D:\truf\runtime\queues
|
||||
D:\truf\runtime\state
|
||||
D:\truf\runtime\logs
|
||||
D:\truf\runtime\control
|
||||
D:\truf\runtime\keychecks
|
||||
D:\truf\runtime\postman_cache
|
||||
D:\truf\runtime\postgres\data
|
||||
D:\truf\tmp
|
||||
```
|
||||
|
||||
Runtime startup performs read-only ACL/owner/reparse preflight and never repairs paths.
|
||||
|
||||
### Remote Assignment Capacity
|
||||
|
||||
`global.result_bundle_max_event_bytes` and `supervisor.worker_api.max_bundle_bytes` are hard per-bundle limits and remain 64 MiB. They are not admission reservations. Each unresolved remote assignment instead charges the persisted `global.remote_assignment_reserve_bytes` baseline of 2 MiB on both the bundle and projection byte axes; local scans retain their existing worst-case reservation behavior.
|
||||
|
||||
Remote admission enforces `global.remote_assignment_max_active: 50` atomically in addition to each user's typed `active_assignment_cap`. The intended 50-assignment user must therefore have its cap set to 50 through the authenticated worker administration path. Lower either cap to reduce concurrency; do not raise the global cap above the validated maximum.
|
||||
|
||||
A valid remote bundle larger than 2 MiB atomically expands its persisted bundle charge to actual bytes before the server returns an acceptance receipt. Temporary aggregate bundle exhaustion returns retryable capacity backpressure, and the worker must retry the identical durable upload. Projection serialization similarly expands a leased job to exact aggregate bytes before any append. If projection capacity is unavailable, the untouched job returns to `pending`; capacity backpressure alone never quarantines it.
|
||||
|
||||
The production keycheck limits of 131,072 items and 128 MiB cover fifty baseline candidate reservations. Bundle and projection aggregate capacities and projection headroom remain independent safety bounds. Runtime-document validation rejects a hard bundle limit above 64 MiB, a remote baseline below 2 MiB or above the hard limit, a global cap above 50, and any aggregate axis that cannot hold all configured baselines.
|
||||
|
||||
## Authority Model
|
||||
|
||||
One cross-session lock is derived only from the canonical bundled PostgreSQL data directory. It is held by every lifecycle-owning foreground/background supervisor and by PostgreSQL bootstrap/verification, migration, reconciliation, and offline hardening. Changing control directory, instance file, or port cannot split authority; different data directories have independent locks.
|
||||
|
||||
The configurable control-directory lock remains a secondary per-instance safety layer. Duplicate launch failure never sends coordinated shutdown to a different owner.
|
||||
|
||||
Before spawn, the launcher captures exact config, supervisor, and code-manifest authority. The manifest also covers the result bundle, ingester, projector, keycheck candidate, and isolated janitor modules.
|
||||
|
||||
The child completes canonical DSN validation and all controller/backend/source/keycheck/dashboard construction before publishing `ACTIVATING`. PostgreSQL, dashboard, and source ticks are forbidden until authenticated parent activation and exact post-activation recheck publish `ACTIVE`.
|
||||
|
||||
If activation becomes uncertain, rollback uses authenticated shutdown and the full configured deadline. It never force-terminates an exact published candidate that may be active. An unconfirmed exact candidate is left running and reported rather than risking an orphaned PostgreSQL tree.
|
||||
|
||||
Authenticated shutdown atomically publishes `STOPPING` and closes all start gates under the control lock before acknowledgement. Mutating commands and snapshot-triggered polls cannot start children after this transition.
|
||||
|
||||
Code/config drift closes start gates and triggers safe shutdown. Authenticated shutdown remains available through private instance credentials, endpoint binding, and retained process identity even when on-disk code changed. Tokens, raw DSNs, and secret-derived hashes are not written to logs.
|
||||
|
||||
## Child Authentication
|
||||
|
||||
Managed scanner, result-ingester, JSONL-projector, keycheck, janitor, and dashboard children must prove all of the following before mutation. The janitor receives no database capability and persists only an exact-private bounded local cursor:
|
||||
|
||||
1. Private per-instance metadata matches inherited instance credentials.
|
||||
2. The retained supervisor process and config command line match metadata.
|
||||
3. The supervisor handshake reports `ACTIVE`.
|
||||
4. Config, supervisor, and complete code-manifest hashes match.
|
||||
5. `SCANNER_DB_URL`, `DATABASE_URL`, and the inherited managed DSN name the same canonical loopback PostgreSQL authority.
|
||||
|
||||
An environment marker by itself grants no authority. Empty or foreign DSNs fail before application files, `ScannerDB`, dependency checks, provider checks, or scanning.
|
||||
|
||||
## PostgreSQL Maintenance
|
||||
|
||||
One-time offline cluster identity binding, with all runtime processes stopped:
|
||||
|
||||
```powershell
|
||||
python postgres_runtime.py bootstrap --config config.yaml
|
||||
```
|
||||
|
||||
Read-only identity verification:
|
||||
|
||||
```powershell
|
||||
python postgres_runtime.py verify --config config.yaml
|
||||
```
|
||||
|
||||
Identity-verified maintenance start/stop, without scanner or keycheck children:
|
||||
|
||||
```powershell
|
||||
python postgres_runtime.py maintenance-start --config config.yaml
|
||||
python postgres_runtime.py maintenance-stop --config config.yaml
|
||||
```
|
||||
|
||||
### Docker Depth Experiment Operator
|
||||
|
||||
Keep `sources.dockerhub.docker_depth_experiment.enabled: false` while reviewing and applying the cohort and hold. From `D:\truf\app`, use the existing private `runtime\state` directory and always stop maintenance PostgreSQL in a `finally` step if an operator command fails:
|
||||
|
||||
```powershell
|
||||
..\stop_runtime.ps1
|
||||
python postgres_runtime.py maintenance-start --config config.yaml
|
||||
python docker_depth_operator.py --config config.yaml --status
|
||||
python docker_depth_operator.py --config config.yaml --generate-cohort-manifest D:\truf\runtime\state\docker-depth-cohort-review.json
|
||||
$cohortSha256 = Read-Host 'Reviewed cohort SHA-256'
|
||||
python docker_depth_operator.py --config config.yaml --apply-cohort-manifest D:\truf\runtime\state\docker-depth-cohort-review.json --approve-sha256 $cohortSha256 --apply --sources-stopped
|
||||
python docker_depth_operator.py --config config.yaml --generate-hold-manifest D:\truf\runtime\state\docker-depth-hold-review.json
|
||||
$holdSha256 = Read-Host 'Reviewed hold SHA-256'
|
||||
python docker_depth_operator.py --config config.yaml --apply-hold-manifest D:\truf\runtime\state\docker-depth-hold-review.json --approve-sha256 $holdSha256 --apply --sources-stopped
|
||||
python postgres_runtime.py maintenance-stop --config config.yaml
|
||||
```
|
||||
|
||||
Review the private files out of band and approve exactly the SHA-256 printed by their generation commands. The operator has no DSN option, never prints queries or targets, and never starts or stops PostgreSQL. After the hold apply and maintenance stop, changing only `enabled` from `false` to `true` is the separate activation decision; its semantic `config_sha256` must remain unchanged. Run `--status` between another maintenance start/stop pair to verify that hash before a separately approved `..\start_runtime.ps1`.
|
||||
|
||||
An attempt-limit hold caused by the retired zero-graph resolver defect has a separate one-time reviewed recovery. Keep enabled config, stop sources, start maintenance PostgreSQL, review the private manifest, and approve only its exact printed SHA-256:
|
||||
|
||||
```powershell
|
||||
python docker_depth_operator.py --config config.yaml --generate-resolver-refund-manifest D:\truf\runtime\state\docker-depth-resolver-refund-review.json
|
||||
$refundSha256 = Read-Host 'Reviewed resolver refund SHA-256'
|
||||
python docker_depth_operator.py --config config.yaml --apply-resolver-refund-manifest D:\truf\runtime\state\docker-depth-resolver-refund-review.json --approve-sha256 $refundSha256 --apply --sources-stopped
|
||||
```
|
||||
|
||||
A genuine remote `resolver_attempt_limit` hold uses a separate reviewed disposition. The manifest deterministically selects the next fresh repository or records terminal remote-unavailable scarcity when none remains:
|
||||
|
||||
```powershell
|
||||
python docker_depth_operator.py --config config.yaml --generate-resolver-disposition-manifest D:\truf\runtime\state\docker-depth-resolver-disposition-review.json
|
||||
$dispositionSha256 = Read-Host 'Reviewed resolver disposition SHA-256'
|
||||
python docker_depth_operator.py --config config.yaml --apply-resolver-disposition-manifest D:\truf\runtime\state\docker-depth-resolver-disposition-review.json --approve-sha256 $dispositionSha256 --apply --sources-stopped
|
||||
```
|
||||
|
||||
Release is reviewed only after the experiment reaches `completed` under enabled config:
|
||||
|
||||
```powershell
|
||||
..\stop_runtime.ps1
|
||||
python postgres_runtime.py maintenance-start --config config.yaml
|
||||
python docker_depth_operator.py --config config.yaml --generate-reactivation-manifest D:\truf\runtime\state\docker-depth-reactivation-review.json
|
||||
$reactivationSha256 = Read-Host 'Reviewed reactivation SHA-256'
|
||||
python docker_depth_operator.py --config config.yaml --apply-reactivation-manifest D:\truf\runtime\state\docker-depth-reactivation-review.json --approve-sha256 $reactivationSha256 --apply --sources-stopped
|
||||
python postgres_runtime.py maintenance-stop --config config.yaml
|
||||
```
|
||||
|
||||
Runtime-safety schema migration, with supervisor, dashboard, scanner, and keycheck sessions stopped:
|
||||
|
||||
```powershell
|
||||
python migrate_runtime_safety.py --config config.yaml --apply --sources-stopped
|
||||
```
|
||||
|
||||
Final cutover refuses a nonempty legacy result spool, any `scan_publication_outbox` row, any legacy `raw_result_json` row, or a prepared legacy JSONL-ledger append. Runtime workers refuse to start until the migration records the singleton PostgreSQL cutover marker.
|
||||
|
||||
Import bounded batches from the retired durable spool until `remaining=0`:
|
||||
|
||||
```powershell
|
||||
python migrate_runtime_safety.py --config config.yaml --import-legacy-spool --max-rows 1000 --apply --sources-stopped
|
||||
```
|
||||
|
||||
Convert bounded legacy raw rows to normalized-v2 data, then rerun the normal migration to restore the cutover marker:
|
||||
|
||||
```powershell
|
||||
python migrate_runtime_safety.py --config config.yaml --backfill-normalized-results --max-rows 1000 --max-bytes 201326592 --max-seconds 30 --apply --sources-stopped
|
||||
```
|
||||
|
||||
If the cutover gate reports legacy outbox rows, project only that bounded backlog offline before retrying migration:
|
||||
|
||||
```powershell
|
||||
python migrate_runtime_safety.py --config config.yaml --drain-legacy-outbox --max-rows 1000 --apply --sources-stopped
|
||||
```
|
||||
|
||||
If a legacy event is too expensive for the per-finding ledger drain, first apply the additive schema (the command remains nonzero while the gate is closed), then transfer exact outbox references into the singleton projector queue without rewriting historical payloads:
|
||||
|
||||
```powershell
|
||||
python migrate_runtime_safety.py --config config.yaml --import-legacy-outbox-to-projection --max-rows 1000 --apply --sources-stopped
|
||||
```
|
||||
|
||||
Build a full bounded, resumable PostgreSQL-derived projection in a dedicated empty directory. Repeat until `completed=true`; live projection files are never overwritten:
|
||||
|
||||
```powershell
|
||||
python migrate_runtime_safety.py --config config.yaml --rebuild-jsonl-output "D:\truf\runtime\rebuild" --max-rows 1000 --max-bytes 201326592 --max-seconds 30 --apply --sources-stopped
|
||||
```
|
||||
|
||||
Quarantine review accepts only a private `truf-pipeline-quarantine-review-v1` manifest with exact `id`, `reason_code`, `payload_sha256`, and `action` (`discard`, deterministic `retry`, or Docker layer `rescan`) entries. `rescan` retires the stale bundle and returns its immutable target to the normal fresh-claim path:
|
||||
|
||||
```powershell
|
||||
python migrate_runtime_safety.py --config config.yaml --review-pipeline-quarantine "D:\review\quarantine.json" --max-rows 1000 --apply --sources-stopped
|
||||
```
|
||||
|
||||
Bounded todo reconciliation:
|
||||
|
||||
```powershell
|
||||
python migrate_runtime_safety.py --config config.yaml --apply --sources-stopped --todo "D:\truf\runtime\queues\todo_github.txt" --source github --platform github --max-rows 1000 --max-bytes 4194304 --max-seconds 5
|
||||
```
|
||||
|
||||
Offline layout hardening creates required directories parent-first, recursively hardens runtime-owned trees, and hardens existing config, secrets, proxy, detector config, and PostgreSQL environment files:
|
||||
|
||||
```powershell
|
||||
python migrate_runtime_safety.py --config config.yaml --harden-runtime --apply --sources-stopped
|
||||
```
|
||||
|
||||
Maintenance and runtime exclude each other through the same cluster authority lock. PostgreSQL sessions use `search_path=public`; `pg_catalog` keeps implicit precedence, authority built-ins are explicitly qualified, and PUBLIC `CREATE` on `public` fails closed.
|
||||
|
||||
`D:\truf\docker-compose.postgres.yml` is a disabled, noncanonical manual-recovery fixture. It is behind the `noncanonical-manual-recovery` profile, has no restart policy, requires an explicit unused `TRUF_DOCKER_POSTGRES_PORT`, and binds data under `D:\truf`; never use it to start or replace the supervisor-owned cluster. Any legacy Docker named volume is intentionally left untouched.
|
||||
|
||||
## Dashboard Secrecy
|
||||
|
||||
Default dashboard queries and frames do not contain raw credentials. Raw scanner and validation values are available only behind explicit default-false per-session reveal controls with a local warning. Treat the database itself as sensitive because persisted findings still contain raw values.
|
||||
|
||||
The dashboard binds to loopback only. Do not expose it through `0.0.0.0`, a reverse proxy, screenshots, or shared logs.
|
||||
|
||||
## Cleanup
|
||||
|
||||
The authenticated `janitor` child is the only stale-tree recovery worker. It does not import `scanner`, and it enforces exact PID/creation-time/executable identities plus enumeration, candidate, entry, byte, time, and depth budgets. Its exact-private local cursor rotates layouts after every inspected entry so a huge first layout cannot starve later trees. Unknown identity retains the tree. Sources may only attempt bounded cleanup of a directory they just used; there is no startup sweep, low-space sweep, periodic supervisor scanner import, or `atexit` cleanup.
|
||||
|
||||
Definitively aborted unreferenced admission intents and deleted artifacts are retired in bounded keyset batches after 30 days. Retirement updates an aggregate SHA-256 chain and count before deleting exact rows; pending, committed/referenced, active, and open-quarantine authority is never eligible.
|
||||
|
||||
Do not manually delete results, result spool events, queue state, PostgreSQL data, or control metadata. Use authenticated coordinated shutdown before offline maintenance.
|
||||
@@ -0,0 +1,351 @@
|
||||
# Detector Notes
|
||||
|
||||
Working notes about TruffleHog detector behavior and local post-processing ideas.
|
||||
|
||||
## GitHub / GitLab Noise
|
||||
|
||||
Current TruffleHog source contains both modern and legacy detectors.
|
||||
|
||||
GitHub v2 detects modern prefixed PATs:
|
||||
|
||||
```text
|
||||
(ghp|gho|ghu|ghs|ghr|github_pat)_[a-zA-Z0-9_]{36,255}
|
||||
```
|
||||
|
||||
GitHub v1 detects legacy 40-character hex tokens near words such as `github`, `gh`, `pat`, or `token`:
|
||||
|
||||
```text
|
||||
(?:github|gh|pat|token).{0,40}([a-f0-9]{40})
|
||||
```
|
||||
|
||||
GitHubOauth2 detects a 20-character client id and a 40-character client secret near `github`:
|
||||
|
||||
```text
|
||||
client_id: [a-zA-Z0-9]{20}
|
||||
client_secret: [a-f0-9]{40}
|
||||
Raw = client_id
|
||||
RawV2 = client_id + client_secret
|
||||
```
|
||||
|
||||
GitLab v2 detects modern PATs:
|
||||
|
||||
```text
|
||||
glpat-[a-zA-Z0-9\-=_]{20,22}
|
||||
```
|
||||
|
||||
GitLab v1 detects any 20-22 character token-like value near `gitlab` and skips `glpat-` so v2 can handle it:
|
||||
|
||||
```text
|
||||
gitlab ... ([a-zA-Z0-9\-=_]{20,22})
|
||||
```
|
||||
|
||||
Observed local results show high false-positive volume for unverified GitHub v1, GitHubOauth2, and GitLab v1 detections. The current scanner post-filter drops unverified GitHub/GitLab findings that do not match known modern token prefixes. This reduces noise but can hide real legacy/OAuth credentials if verification cannot run.
|
||||
|
||||
TODO: Prefer a confidence model over hard dropping:
|
||||
|
||||
```text
|
||||
verified -> high confidence
|
||||
modern prefix shape -> high/medium confidence
|
||||
legacy GitHub v1 / GitHubOauth2 / GitLab v1 -> low confidence unless verified
|
||||
```
|
||||
|
||||
Dashboard should hide low-confidence findings by default, but allow explicit review.
|
||||
|
||||
## GCP
|
||||
|
||||
### GCP service account JSON
|
||||
|
||||
Detector: `GCP`
|
||||
|
||||
TruffleHog detects JSON blobs containing `auth_provider_x509_cert_url` and parses service-account style credentials.
|
||||
|
||||
Useful fields already present in the JSON:
|
||||
|
||||
```text
|
||||
type
|
||||
project_id
|
||||
private_key_id
|
||||
private_key
|
||||
client_email
|
||||
client_id
|
||||
auth_uri
|
||||
token_uri
|
||||
auth_provider_x509_cert_url
|
||||
client_x509_cert_url
|
||||
```
|
||||
|
||||
TruffleHog output behavior:
|
||||
|
||||
```text
|
||||
Raw = client_email, or full key JSON if client_email is missing
|
||||
RawV2 = full cleaned credential JSON
|
||||
Redacted = client_email
|
||||
ExtraData.project = project_id
|
||||
AnalysisInfo.principal = client_email
|
||||
AnalysisInfo.type = type
|
||||
```
|
||||
|
||||
Practical enrichment fields:
|
||||
|
||||
```text
|
||||
gcp_project_id
|
||||
gcp_client_email
|
||||
gcp_client_id
|
||||
gcp_private_key_id
|
||||
gcp_credential_type
|
||||
```
|
||||
|
||||
This detector has enough context to verify/function without extra source-code lookup if `RawV2` is preserved.
|
||||
|
||||
### GCP Application Default Credentials
|
||||
|
||||
Detector: `GCPApplicationDefaultCredentials`
|
||||
|
||||
TruffleHog detects ADC JSON containing `client_secret` and `.apps.googleusercontent.com` client IDs.
|
||||
|
||||
Useful fields:
|
||||
|
||||
```text
|
||||
client_id
|
||||
client_secret
|
||||
refresh_token
|
||||
type
|
||||
```
|
||||
|
||||
TruffleHog output behavior:
|
||||
|
||||
```text
|
||||
Raw = client_id without .apps.googleusercontent.com suffix
|
||||
RawV2 = client_id_without_suffix + refresh_token
|
||||
Redacted = shortened refresh_token
|
||||
ExtraData may contain verification details when verified
|
||||
```
|
||||
|
||||
Risk: `RawV2` is concatenated and does not retain `client_secret` cleanly. The raw finding JSON may not be enough to reconstruct the original ADC JSON unless the source line/file is available.
|
||||
|
||||
Practical enrichment fields:
|
||||
|
||||
```text
|
||||
gcp_client_id
|
||||
gcp_refresh_token_redacted
|
||||
gcp_credential_type
|
||||
```
|
||||
|
||||
TODO: For ADC findings, use source context around the finding to parse the whole JSON and preserve `client_secret`/`refresh_token` as structured fields.
|
||||
|
||||
### Google AQ authentication keys
|
||||
|
||||
`AQ.` credentials are currently classified through the Gemini Developer API at
|
||||
`generativelanguage.googleapis.com`. That result does not establish Vertex AI access.
|
||||
|
||||
TODO: Add a separate Vertex AI Express probe for `AQ.` credentials against the supported
|
||||
`aiplatform.googleapis.com` key-authenticated methods. Keep Gemini Developer API and Vertex
|
||||
results independent, and do not infer access to the full project/location-scoped Vertex API
|
||||
from the key prefix or from a successful Gemini Developer API check.
|
||||
|
||||
## Azure
|
||||
|
||||
### Azure Container Registry
|
||||
|
||||
Detector: `AzureContainerRegistry`
|
||||
|
||||
Detector finds registry hosts and ACR password-like values.
|
||||
|
||||
Patterns:
|
||||
|
||||
```text
|
||||
registry: <name>.azurecr.io
|
||||
password: [a-zA-Z0-9+/]{42}+ACR[a-zA-Z0-9]{6}
|
||||
```
|
||||
|
||||
TruffleHog output behavior:
|
||||
|
||||
```text
|
||||
Raw = password
|
||||
RawV2 = {"username":"<registry>","password":"<password>"}
|
||||
Redacted = registry name
|
||||
```
|
||||
|
||||
Verification uses:
|
||||
|
||||
```text
|
||||
https://<registry>.azurecr.io/v2/
|
||||
BasicAuth(username=<registry>, password=<password>)
|
||||
```
|
||||
|
||||
Practical enrichment fields:
|
||||
|
||||
```text
|
||||
azure_acr_registry
|
||||
azure_acr_login_server = <registry>.azurecr.io
|
||||
```
|
||||
|
||||
This detector has enough context in `RawV2` to be useful.
|
||||
|
||||
### Azure OpenAI
|
||||
|
||||
Detector: `AzureOpenAI`
|
||||
|
||||
Detector finds API keys and Azure OpenAI endpoints.
|
||||
|
||||
Patterns:
|
||||
|
||||
```text
|
||||
endpoint: <service>.openai.azure.com
|
||||
key: 32 lowercase hex chars near api_key/openai_key keywords
|
||||
```
|
||||
|
||||
TruffleHog output behavior:
|
||||
|
||||
```text
|
||||
Raw = api key
|
||||
RawV2 = key:endpoint when endpoint is paired during verification or when only one endpoint exists
|
||||
Redacted = shortened key
|
||||
```
|
||||
|
||||
Verification calls:
|
||||
|
||||
```text
|
||||
https://<endpoint>/openai/deployments?api-version=2023-03-15-preview
|
||||
Header: Api-Key: <key>
|
||||
```
|
||||
|
||||
Practical enrichment fields:
|
||||
|
||||
```text
|
||||
azure_openai_endpoint
|
||||
azure_openai_resource_name
|
||||
```
|
||||
|
||||
TODO: If RawV2 is empty, scan nearby source context for `.openai.azure.com` to pair keys with endpoints.
|
||||
|
||||
### Azure DevOps PAT
|
||||
|
||||
Detector: `AzureDevopsPersonalAccessToken`
|
||||
|
||||
Detector finds a 52-character token and an organization-like string near `azure`.
|
||||
|
||||
TruffleHog output behavior:
|
||||
|
||||
```text
|
||||
Raw = PAT
|
||||
RawV2 = PAT + organization
|
||||
```
|
||||
|
||||
Verification calls:
|
||||
|
||||
```text
|
||||
https://dev.azure.com/<organization>/_apis/projects
|
||||
BasicAuth(username="", password=<PAT>)
|
||||
```
|
||||
|
||||
Risk: `RawV2` is concatenated without delimiter, so organization extraction from RawV2 is ambiguous unless the token length is known.
|
||||
|
||||
Practical enrichment fields:
|
||||
|
||||
```text
|
||||
azure_devops_org
|
||||
```
|
||||
|
||||
TODO: Parse organization from raw finding JSON/source context rather than relying on concatenated RawV2 alone.
|
||||
|
||||
## DockerHub
|
||||
|
||||
Detector: `Dockerhub`
|
||||
|
||||
DockerHub v2 detects modern PATs:
|
||||
|
||||
```text
|
||||
dckr_pat_[a-zA-Z0-9_-]{27}
|
||||
```
|
||||
|
||||
DockerHub v1 detects UUID-like legacy tokens near `docker`.
|
||||
|
||||
Both versions try to pair the token with nearby usernames or emails:
|
||||
|
||||
```text
|
||||
username: user/usr/-u/id nearby value or email address
|
||||
Raw = token
|
||||
RawV2 = username:token when username/email is found
|
||||
```
|
||||
|
||||
Verification calls:
|
||||
|
||||
```text
|
||||
POST https://hub.docker.com/v2/users/login
|
||||
{"username":"<username>","password":"<token>"}
|
||||
```
|
||||
|
||||
If verified, ExtraData can include:
|
||||
|
||||
```text
|
||||
hub_username
|
||||
hub_email
|
||||
hub_scope
|
||||
2fa_required
|
||||
```
|
||||
|
||||
Practical enrichment fields:
|
||||
|
||||
```text
|
||||
dockerhub_username
|
||||
dockerhub_email
|
||||
dockerhub_scope
|
||||
dockerhub_2fa_required
|
||||
```
|
||||
|
||||
Risk: A token without nearby username cannot be verified by this detector, but may still be useful if a username can be found elsewhere in the same package/repo.
|
||||
|
||||
TODO: For unverified DockerHub PATs with empty RawV2, scan nearby context and package/repo metadata for plausible DockerHub usernames.
|
||||
|
||||
## Proposed Enrichment Layer
|
||||
|
||||
Add a post-processing enrichment layer after TruffleHog result parsing and before DB insert.
|
||||
|
||||
Input:
|
||||
|
||||
```text
|
||||
finding JSON
|
||||
target metadata
|
||||
source file path/line if available
|
||||
optional nearby source context
|
||||
```
|
||||
|
||||
Output fields stored in DB/dashboard:
|
||||
|
||||
```text
|
||||
provider
|
||||
credential_kind
|
||||
credential_confidence
|
||||
required_context_missing
|
||||
principal
|
||||
project_id
|
||||
tenant_id
|
||||
organization
|
||||
registry
|
||||
endpoint
|
||||
username
|
||||
email
|
||||
scope
|
||||
resource
|
||||
```
|
||||
|
||||
Suggested confidence levels:
|
||||
|
||||
```text
|
||||
verified
|
||||
structured_complete
|
||||
prefix_shape_complete
|
||||
token_only_missing_context
|
||||
legacy_unverified
|
||||
noisy_unverified
|
||||
```
|
||||
|
||||
Priority implementation:
|
||||
|
||||
1. Parse GCP service-account JSON from `RawV2`.
|
||||
2. Parse Azure ACR `RawV2` JSON.
|
||||
3. Parse Azure OpenAI `RawV2` as `key:endpoint` when available.
|
||||
4. Parse DockerHub `RawV2` as `username:token` and ExtraData when verified.
|
||||
5. Add low-confidence classification for GitHub/GitLab legacy detectors instead of hard-dropping them.
|
||||
6. Optionally read nearby file context for detectors where RawV2 lacks required context.
|
||||
@@ -0,0 +1,13 @@
|
||||
# Offline JSONL Reconciliation
|
||||
|
||||
PostgreSQL is authoritative. JSONL and provider status files are asynchronous, bounded, rebuildable compatibility projections and may lag. Normal keychecks never consume `found_secrets.jsonl`; reconciliation is explicit offline compatibility work only.
|
||||
|
||||
Provider `*Checked.txt` files are rebuilt from PostgreSQL by the keycheck runner and are not append streams owned by the JSONL projector.
|
||||
|
||||
1. Stop the supervisor and acquire the same cluster authority used by `migrate_runtime_safety.py`.
|
||||
2. Make an immutable backup of the current file, every numbered segment, the manifest, the publication ledger, and any `*.torn-tail.bin` file.
|
||||
3. Validate every retained segment as newline-terminated UTF-8 JSON. Quarantine, rather than concatenate, any final partial record.
|
||||
4. Do not use keycheck `input_state.json` as a retention checkpoint. PostgreSQL candidates and current state are authoritative; immutable compatibility generations may be retired by the configured generation limit.
|
||||
5. For pre-v2 multi-GiB history, do not raise online bounds or backfill it during startup. Preserve it and use a separately reviewed, bounded offline rebuild/import operation.
|
||||
6. The singleton projector recovers prepared appends by exact generation, offset, length, and SHA-256; partial tails are quarantined and truncated to the prepared offset before retry.
|
||||
7. Rotation renames the active generation atomically and never copies full history. A deterministic poison job is quarantined individually and later jobs continue.
|
||||
@@ -0,0 +1,192 @@
|
||||
# Keychecker Layout
|
||||
|
||||
Normal input and authority:
|
||||
```text
|
||||
PostgreSQL keycheck_candidates fenced provider work queue
|
||||
PostgreSQL keycheck_results authoritative history
|
||||
PostgreSQL keycheck_current_state authoritative current classification
|
||||
D:\truf\runtime\proxy.txt optional provider proxy input
|
||||
```
|
||||
|
||||
`found_secrets.jsonl`, `*Results.jsonl`, and `*Checked.txt`/status files are projector-owned compatibility outputs. They may lag and normal providers do not read or write them.
|
||||
|
||||
Shared helper:
|
||||
```text
|
||||
D:\truf\app\keycheckers\keycheck_common.py
|
||||
```
|
||||
|
||||
`keycheck_runner.py --input-mode postgres` is the default managed mode. `--input` is accepted only with explicit `--input-mode jsonl` for reviewed offline/import compatibility. Provider completion inserts the result, updates current state, completes the exact candidate lease, releases candidate capacity, and creates its projection job in one PostgreSQL transaction.
|
||||
|
||||
## DeepSeek
|
||||
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "recheck deepseek all --max-keys 100"
|
||||
```
|
||||
|
||||
Output folder:
|
||||
```text
|
||||
deepseek\deepseekAlive.txt
|
||||
deepseek\deepseekNoBalance.txt
|
||||
deepseek\deepseekDead.txt
|
||||
deepseek\deepseekLimited.txt
|
||||
deepseek\deepseekNetwork.txt
|
||||
deepseek\deepseekUnknown.txt
|
||||
deepseek\deepseekChecked.txt
|
||||
deepseek\deepseekResults.jsonl
|
||||
```
|
||||
|
||||
## Qwen / DashScope
|
||||
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "recheck qwen all --max-keys 100"
|
||||
```
|
||||
|
||||
The checker validates `QwenDashScope` findings with `GET /models` against the public DashScope OpenAI-compatible region endpoints. Coding Plan keys (`sk-sp-...`) use `https://coding-intl.dashscope.aliyuncs.com/v1` by default. Workspace-specific endpoints can be added with `--base-url` via `keychecks.service_args.qwen` or `QWEN_BASE_URLS`.
|
||||
|
||||
Output folder:
|
||||
```text
|
||||
qwen\qwenAlive.txt
|
||||
qwen\qwenNoBalance.txt
|
||||
qwen\qwenNoContext.txt
|
||||
qwen\qwenDead.txt
|
||||
qwen\qwenLimited.txt
|
||||
qwen\qwenRestricted.txt
|
||||
qwen\qwenNetwork.txt
|
||||
qwen\qwenUnknown.txt
|
||||
qwen\qwenChecked.txt
|
||||
qwen\qwenResults.jsonl
|
||||
```
|
||||
|
||||
## Kimi / Moonshot AI
|
||||
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "recheck kimi all --max-keys 100"
|
||||
```
|
||||
|
||||
The explicit `MOONSHOT_API_KEY` / `KIMI_API_KEY` detector is validated without generation by calling `GET /v1/users/me/balance` on the independent global and China Moonshot endpoints.
|
||||
|
||||
Output folder:
|
||||
```text
|
||||
kimi\kimiAlive.txt
|
||||
kimi\kimiNoBalance.txt
|
||||
kimi\kimiDead.txt
|
||||
kimi\kimiLimited.txt
|
||||
kimi\kimiRestricted.txt
|
||||
kimi\kimiNetwork.txt
|
||||
kimi\kimiUnknown.txt
|
||||
kimi\kimiChecked.txt
|
||||
kimi\kimiResults.jsonl
|
||||
```
|
||||
|
||||
## Groq
|
||||
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "recheck groq all --max-keys 100"
|
||||
```
|
||||
|
||||
The checker validates TruffleHog `Groq` findings with `GET https://api.groq.com/openai/v1/models` and does not run generation probes.
|
||||
|
||||
Output folder:
|
||||
```text
|
||||
groq\groqAlive.txt
|
||||
groq\groqDead.txt
|
||||
groq\groqLimited.txt
|
||||
groq\groqRestricted.txt
|
||||
groq\groqNetwork.txt
|
||||
groq\groqUnknown.txt
|
||||
groq\groqChecked.txt
|
||||
groq\groqResults.jsonl
|
||||
```
|
||||
|
||||
## Replicate / xAI / HuggingFace
|
||||
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "recheck replicate all --max-keys 100"
|
||||
python supervisor.py --config config.yaml --cmd "recheck xai all --max-keys 100"
|
||||
python supervisor.py --config config.yaml --cmd "recheck huggingface all --max-keys 100"
|
||||
```
|
||||
|
||||
These checkers validate built-in TruffleHog findings through non-generating endpoints: Replicate account lookup, xAI model list, and HuggingFace whoami.
|
||||
|
||||
## Anthropic
|
||||
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "recheck anthropic all --max-keys 100"
|
||||
```
|
||||
|
||||
Output folder:
|
||||
```text
|
||||
anthropic\anthropicAlive.txt
|
||||
anthropic\anthropicNoQuota.txt
|
||||
anthropic\anthropicDead.txt
|
||||
anthropic\anthropicLimited.txt
|
||||
anthropic\anthropicRestricted.txt
|
||||
anthropic\anthropicNetwork.txt
|
||||
anthropic\anthropicUnknown.txt
|
||||
anthropic\anthropicChecked.txt
|
||||
anthropic\anthropicResults.jsonl
|
||||
```
|
||||
|
||||
## AWS
|
||||
|
||||
Default mode only checks STS identity:
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "recheck aws all --max-keys 100"
|
||||
```
|
||||
|
||||
Optional Bedrock probing is configured under `keychecks.service_args.aws`, then run:
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "recheck aws all --max-keys 100"
|
||||
```
|
||||
|
||||
Output folder:
|
||||
```text
|
||||
aws\awsAlive.txt
|
||||
aws\awsBedrock.txt
|
||||
aws\awsAdmin.txt
|
||||
aws\awsCanary.txt
|
||||
aws\awsQuarantined.txt
|
||||
aws\awsAccessDenied.txt
|
||||
aws\awsDead.txt
|
||||
aws\awsNetwork.txt
|
||||
aws\awsUnknown.txt
|
||||
aws\awsChecked.txt
|
||||
aws\awsResults.jsonl
|
||||
```
|
||||
|
||||
Canary AWS credentials are detected before active AWS probes when TruffleHog provides `ExtraData.is_canary` / canary message. If metadata is absent, STS ARN containing `canarytokens` is also classified as `awsCanary.txt` and IAM/Bedrock probes are skipped.
|
||||
|
||||
## Azure
|
||||
|
||||
```powershell
|
||||
python supervisor.py --config config.yaml --cmd "recheck azure all --max-keys 100"
|
||||
```
|
||||
|
||||
This checks Azure service-principal findings from `DetectorName=Azure` using `tenantId`, `clientId`, `clientSecret` from `RawV2`.
|
||||
|
||||
`DetectorName=AzureOpenAI` is placed into `azureOpenAIUnresolved.txt` unless an endpoint/resource name is available.
|
||||
|
||||
Output folder:
|
||||
```text
|
||||
azure\azureAlive.txt
|
||||
azure\azureDead.txt
|
||||
azure\azureRestricted.txt
|
||||
azure\azureNetwork.txt
|
||||
azure\azureUnknown.txt
|
||||
azure\azureOpenAIUnresolved.txt
|
||||
azure\azureChecked.txt
|
||||
azure\azureResults.jsonl
|
||||
```
|
||||
|
||||
## Retry Flags
|
||||
|
||||
Common flags:
|
||||
```text
|
||||
--retry-network
|
||||
--retry-limited
|
||||
--retry-unknown
|
||||
--recheck-all
|
||||
--max-keys N
|
||||
```
|
||||
|
||||
Network/proxy failures are committed with the `network` status group and can be selected for a bounded PostgreSQL recheck. `*Network.txt` is only its asynchronous compatibility projection.
|
||||
+4555
File diff suppressed because it is too large
Load Diff
+13
@@ -0,0 +1,13 @@
|
||||
"""Retired legacy mutation UI.
|
||||
|
||||
Scanner lifecycle control is intentionally available only through supervisor.py.
|
||||
The read-only observability UI remains dashboard.py.
|
||||
"""
|
||||
|
||||
import streamlit as st
|
||||
|
||||
|
||||
st.set_page_config(page_title='Scanner UI Retired', page_icon='LOCK', layout='centered')
|
||||
st.title('Legacy scanner controls are retired')
|
||||
st.error('This UI cannot start, pause, resume, cancel, or configure scans.')
|
||||
st.info('Use the authenticated supervisor commands for lifecycle control and dashboard.py for read-only observability.')
|
||||
@@ -0,0 +1,15 @@
|
||||
"""Retired direct GitHub credential audit entrypoint."""
|
||||
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
|
||||
def main():
|
||||
raise SystemExit(
|
||||
'This direct credential audit is retired. Use authenticated supervisor-managed GitHub keychecks.'
|
||||
)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,55 @@
|
||||
MAX_RESULT_BUNDLE_BYTES = 64 * 1024 * 1024
|
||||
REMOTE_ASSIGNMENT_BASELINE_BYTES = 2 * 1024 * 1024
|
||||
REMOTE_ASSIGNMENT_MAX_ACTIVE = 50
|
||||
|
||||
|
||||
def validate_remote_assignment_capacity(config):
|
||||
values = {}
|
||||
fields = (
|
||||
'result_bundle_max_event_bytes',
|
||||
'remote_assignment_reserve_bytes',
|
||||
'remote_assignment_max_active',
|
||||
'result_bundle_max_total_bytes',
|
||||
'projection_backlog_max_bytes',
|
||||
'projection_backlog_headroom_bytes',
|
||||
'keycheck_queue_max_items',
|
||||
'keycheck_queue_max_bytes',
|
||||
'keycheck_candidates_per_event',
|
||||
'keycheck_candidate_bytes_per_event',
|
||||
)
|
||||
for name in fields:
|
||||
value = config.get(name)
|
||||
if type(value) is not int or value < 0:
|
||||
raise ValueError(name)
|
||||
values[name] = value
|
||||
|
||||
hard_limit = values['result_bundle_max_event_bytes']
|
||||
reserve = values['remote_assignment_reserve_bytes']
|
||||
active = values['remote_assignment_max_active']
|
||||
if not REMOTE_ASSIGNMENT_BASELINE_BYTES <= reserve <= hard_limit:
|
||||
raise ValueError('remote_assignment_reserve_bytes')
|
||||
if not REMOTE_ASSIGNMENT_BASELINE_BYTES <= hard_limit <= MAX_RESULT_BUNDLE_BYTES:
|
||||
raise ValueError('result_bundle_max_event_bytes')
|
||||
if not 1 <= active <= REMOTE_ASSIGNMENT_MAX_ACTIVE:
|
||||
raise ValueError('remote_assignment_max_active')
|
||||
|
||||
required_bytes = active * reserve
|
||||
if values['result_bundle_max_total_bytes'] < required_bytes:
|
||||
raise ValueError('result_bundle_max_total_bytes')
|
||||
projection_admission_bytes = (
|
||||
values['projection_backlog_max_bytes']
|
||||
- values['projection_backlog_headroom_bytes']
|
||||
)
|
||||
if projection_admission_bytes < required_bytes:
|
||||
raise ValueError('projection_backlog_max_bytes')
|
||||
if (
|
||||
values['keycheck_queue_max_items']
|
||||
< active * values['keycheck_candidates_per_event']
|
||||
):
|
||||
raise ValueError('keycheck_queue_max_items')
|
||||
if (
|
||||
values['keycheck_queue_max_bytes']
|
||||
< active * values['keycheck_candidate_bytes_per_event']
|
||||
):
|
||||
raise ValueError('keycheck_queue_max_bytes')
|
||||
return values
|
||||
@@ -0,0 +1,486 @@
|
||||
"""Stdlib-only authentication boundary for supervised application children."""
|
||||
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
if not sys.dont_write_bytecode:
|
||||
raise RuntimeError('supervised child bootstrap could not disable bytecode writes')
|
||||
|
||||
import hashlib
|
||||
import hmac
|
||||
import json
|
||||
import os
|
||||
import runpy
|
||||
import socket
|
||||
import stat
|
||||
|
||||
|
||||
MAX_METADATA_BYTES = 256 * 1024
|
||||
MAX_HANDSHAKE_BYTES = 1024 * 1024
|
||||
MAX_PYVENV_BYTES = 64 * 1024
|
||||
MANIFEST_SCHEMA = 5
|
||||
APPLICATION_IMPORT_SUFFIXES = ('.py', '.pyw', '.pyc', '.pyd')
|
||||
CONTROL_SCHEMA = 1
|
||||
REQUIRED_DEPENDENCIES = {
|
||||
'supervisor': ('psycopg', 'yaml'),
|
||||
'postgres-runtime': ('psycopg', 'yaml'),
|
||||
'migrate-runtime-safety': ('psycopg', 'yaml'),
|
||||
'scanner': ('psycopg', 'requests', 'urllib3', 'yaml', 'zstandard'),
|
||||
'discovery-producer': ('psycopg', 'requests', 'urllib3', 'yaml', 'zstandard'),
|
||||
'docker-shadow': ('psycopg', 'requests', 'urllib3', 'yaml', 'zstandard'),
|
||||
'keycheck': ('psycopg', 'requests', 'yaml'),
|
||||
'dashboard': ('pandas', 'plotly', 'psycopg', 'streamlit', 'yaml'),
|
||||
'keycheck-provider': ('boto3', 'botocore', 'psycopg', 'requests', 'yaml'),
|
||||
'janitor': ('yaml',),
|
||||
'result-ingester': ('psycopg', 'yaml'),
|
||||
'jsonl-projector': ('psycopg', 'yaml'),
|
||||
'worker-api': ('psycopg', 'requests', 'starlette', 'urllib3', 'uvicorn', 'yaml', 'zstandard'),
|
||||
}
|
||||
|
||||
|
||||
class ChildRuntimeError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
ENV = {
|
||||
'instance_file': 'TRUF_SUPERVISOR_INSTANCE_FILE',
|
||||
'instance_id': 'TRUF_SUPERVISOR_INSTANCE_ID',
|
||||
'token': 'TRUF_SUPERVISOR_TOKEN',
|
||||
'config_sha256': 'TRUF_SUPERVISOR_CONFIG_SHA256',
|
||||
'supervisor_sha256': 'TRUF_SUPERVISOR_SHA256',
|
||||
'code_manifest_sha256': 'TRUF_SUPERVISOR_CODE_MANIFEST_SHA256',
|
||||
'dsn_sha256': 'TRUF_SUPERVISOR_DSN_SHA256',
|
||||
'kind': 'TRUF_SUPERVISOR_CHILD_KIND',
|
||||
}
|
||||
|
||||
|
||||
def _canonical(path):
|
||||
return os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(path))))
|
||||
|
||||
|
||||
def _is_reparse_point(path):
|
||||
details = os.lstat(path)
|
||||
if stat.S_ISLNK(details.st_mode):
|
||||
return True
|
||||
attributes = getattr(details, 'st_file_attributes', 0)
|
||||
reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0)
|
||||
return bool(attributes & reparse_attribute) or getattr(os.path, 'isjunction', lambda _path: False)(path)
|
||||
|
||||
|
||||
def _contained(path, roots):
|
||||
for root in roots:
|
||||
try:
|
||||
if path != root and os.path.commonpath((root, path)) == root:
|
||||
return True
|
||||
except ValueError:
|
||||
continue
|
||||
return False
|
||||
|
||||
|
||||
def _validated_site_directory(path, roots):
|
||||
if not path or not os.path.isdir(path):
|
||||
return ''
|
||||
candidate = _canonical(path)
|
||||
trusted_roots = tuple(_canonical(root) for root in roots if root)
|
||||
if os.path.basename(candidate).lower() not in ('site-packages', 'dist-packages'):
|
||||
raise RuntimeError(f'interpreter dependency path is not a site-packages directory: {candidate}')
|
||||
if not _contained(candidate, trusted_roots):
|
||||
raise RuntimeError(f'interpreter dependency path escapes its trusted root: {candidate}')
|
||||
return candidate
|
||||
|
||||
|
||||
def _append_site_directories(paths, roots):
|
||||
existing = {_canonical(path) for path in sys.path if path}
|
||||
for path in paths:
|
||||
candidate = _validated_site_directory(path, roots)
|
||||
if candidate and candidate not in existing:
|
||||
# Direct insertion intentionally does not evaluate .pth hook lines.
|
||||
sys.path.append(candidate)
|
||||
existing.add(candidate)
|
||||
|
||||
|
||||
def _venv_configuration():
|
||||
# Preserve a venv's bin/python symlink location while locating pyvenv.cfg.
|
||||
executable_dir = os.path.dirname(os.path.normcase(os.path.abspath(sys.executable)))
|
||||
roots = [executable_dir]
|
||||
if os.path.basename(executable_dir).lower() in ('bin', 'scripts'):
|
||||
roots.insert(0, os.path.dirname(executable_dir))
|
||||
for root in roots:
|
||||
config_path = os.path.join(root, 'pyvenv.cfg')
|
||||
if not os.path.isfile(config_path):
|
||||
continue
|
||||
details = os.stat(config_path, follow_symlinks=False)
|
||||
if not stat.S_ISREG(details.st_mode) or details.st_size > MAX_PYVENV_BYTES:
|
||||
raise RuntimeError('interpreter pyvenv.cfg is not a bounded regular file')
|
||||
with open(config_path, 'rb') as handle:
|
||||
payload = handle.read(MAX_PYVENV_BYTES + 1)
|
||||
if len(payload) > MAX_PYVENV_BYTES:
|
||||
raise RuntimeError('interpreter pyvenv.cfg exceeds its byte bound')
|
||||
include_system = False
|
||||
for raw_line in payload.decode('utf-8', errors='strict').splitlines():
|
||||
key, separator, value = raw_line.partition('=')
|
||||
if separator and key.strip().lower() == 'include-system-site-packages':
|
||||
include_system = value.strip().lower() in ('1', 'true', 'yes')
|
||||
return _canonical(root), include_system
|
||||
return '', True
|
||||
|
||||
|
||||
def _venv_site_directories(root):
|
||||
if os.name == 'nt':
|
||||
return [os.path.join(root, 'Lib', 'site-packages')]
|
||||
version = f'python{sys.version_info.major}.{sys.version_info.minor}'
|
||||
return [
|
||||
os.path.join(root, library, version, name)
|
||||
for library in ('lib', 'lib64')
|
||||
for name in ('site-packages', 'dist-packages')
|
||||
]
|
||||
|
||||
|
||||
def _system_site_directories():
|
||||
import sysconfig
|
||||
|
||||
roots = tuple(dict.fromkeys((_canonical(sys.base_prefix), _canonical(sys.base_exec_prefix))))
|
||||
paths = sysconfig.get_paths(vars={
|
||||
'base': sys.base_prefix,
|
||||
'platbase': sys.base_exec_prefix,
|
||||
})
|
||||
return [paths.get('purelib'), paths.get('platlib')], roots
|
||||
|
||||
|
||||
def _user_site_directories():
|
||||
if os.name == 'nt':
|
||||
try:
|
||||
import ctypes
|
||||
|
||||
appdata = ctypes.create_unicode_buffer(32768)
|
||||
if ctypes.windll.shell32.SHGetFolderPathW(None, 0x001A, None, 0, appdata) != 0:
|
||||
return [], ()
|
||||
root = _canonical(os.path.join(appdata.value, 'Python'))
|
||||
version = f'Python{sys.version_info.major}{sys.version_info.minor}'
|
||||
return [os.path.join(root, version, 'site-packages')], (root,)
|
||||
except (AttributeError, OSError, ValueError):
|
||||
return [], ()
|
||||
try:
|
||||
import pwd
|
||||
|
||||
home = _canonical(pwd.getpwuid(os.getuid()).pw_dir)
|
||||
except (ImportError, KeyError, OSError):
|
||||
return [], ()
|
||||
version = f'python{sys.version_info.major}.{sys.version_info.minor}'
|
||||
if sys.platform == 'darwin':
|
||||
root = _canonical(os.path.join(home, 'Library', 'Python', f'{sys.version_info.major}.{sys.version_info.minor}'))
|
||||
return [os.path.join(root, 'lib', 'python', 'site-packages')], (root,)
|
||||
root = _canonical(os.path.join(home, '.local'))
|
||||
return [
|
||||
os.path.join(root, 'lib', version, 'site-packages'),
|
||||
os.path.join(root, 'lib', version, 'dist-packages'),
|
||||
], (root,)
|
||||
|
||||
|
||||
def _missing_dependencies(kind):
|
||||
import importlib.util
|
||||
|
||||
return [name for name in REQUIRED_DEPENDENCIES[kind] if importlib.util.find_spec(name) is None]
|
||||
|
||||
|
||||
def _enable_dependency_paths(kind):
|
||||
venv_root, include_system = _venv_configuration()
|
||||
if venv_root:
|
||||
_append_site_directories(_venv_site_directories(venv_root), (venv_root,))
|
||||
if include_system:
|
||||
system_paths, system_roots = _system_site_directories()
|
||||
_append_site_directories(system_paths, system_roots)
|
||||
if _missing_dependencies(kind):
|
||||
user_paths, user_roots = _user_site_directories()
|
||||
_append_site_directories(user_paths, user_roots)
|
||||
missing = _missing_dependencies(kind)
|
||||
if missing:
|
||||
raise RuntimeError('required authenticated child dependencies are unavailable: ' + ', '.join(missing))
|
||||
|
||||
|
||||
def _sha256_file(path):
|
||||
digest = hashlib.sha256()
|
||||
with open(path, 'rb') as handle:
|
||||
while True:
|
||||
block = handle.read(1024 * 1024)
|
||||
if not block:
|
||||
return digest.hexdigest()
|
||||
digest.update(block)
|
||||
|
||||
|
||||
def _read_object(path):
|
||||
details = os.stat(path, follow_symlinks=False)
|
||||
if not stat.S_ISREG(details.st_mode) or details.st_size <= 0 or details.st_size > MAX_METADATA_BYTES:
|
||||
raise RuntimeError('supervisor metadata is not a bounded regular file')
|
||||
with open(path, 'rb') as handle:
|
||||
payload = handle.read(MAX_METADATA_BYTES + 1)
|
||||
if len(payload) > MAX_METADATA_BYTES:
|
||||
raise RuntimeError('supervisor metadata exceeds its byte bound')
|
||||
value = json.loads(payload.decode('utf-8'))
|
||||
if not isinstance(value, dict):
|
||||
raise RuntimeError('supervisor metadata root is invalid')
|
||||
return value
|
||||
|
||||
|
||||
def _manifest_digest(manifest):
|
||||
payload = json.dumps(manifest, ensure_ascii=True, sort_keys=True, separators=(',', ':')).encode('utf-8')
|
||||
return hashlib.sha256(payload).hexdigest()
|
||||
|
||||
|
||||
def _application_code_files(root):
|
||||
names = set()
|
||||
|
||||
def raise_walk_error(exc):
|
||||
raise RuntimeError(f'unable to inspect the application root: {exc}') from exc
|
||||
|
||||
for current, directories, files in os.walk(root, followlinks=False, onerror=raise_walk_error):
|
||||
for name in directories:
|
||||
candidate = os.path.join(current, name)
|
||||
if _is_reparse_point(candidate):
|
||||
relative = os.path.relpath(candidate, root).replace(os.sep, '/')
|
||||
raise RuntimeError(f'application directory reparse point is forbidden: {relative}')
|
||||
relative_current = os.path.relpath(current, root)
|
||||
in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep))
|
||||
suffixes = ('.pyc',) if in_cache else APPLICATION_IMPORT_SUFFIXES
|
||||
for name in files:
|
||||
source_path = os.path.abspath(os.path.join(current, name))
|
||||
if _is_reparse_point(source_path):
|
||||
relative = os.path.relpath(source_path, root).replace(os.sep, '/')
|
||||
raise RuntimeError(f'application file reparse point is forbidden: {relative}')
|
||||
if not name.lower().endswith(suffixes):
|
||||
continue
|
||||
path = _canonical(source_path)
|
||||
try:
|
||||
contained = os.path.commonpath((root, path)) == root
|
||||
except ValueError:
|
||||
contained = False
|
||||
if not contained:
|
||||
raise RuntimeError('application Python authority escapes its root')
|
||||
names.add(os.path.relpath(source_path, root).replace(os.sep, '/'))
|
||||
return names
|
||||
|
||||
|
||||
def _reject_cached_bytecode(root):
|
||||
def raise_walk_error(exc):
|
||||
raise RuntimeError(f'unable to inspect the application root: {exc}') from exc
|
||||
|
||||
try:
|
||||
root_details = os.lstat(root)
|
||||
except OSError as exc:
|
||||
raise RuntimeError(f'application root is unavailable: {root}') from exc
|
||||
if _is_reparse_point(root):
|
||||
raise RuntimeError(f'application root reparse point is forbidden: {root}')
|
||||
if not stat.S_ISDIR(root_details.st_mode):
|
||||
raise RuntimeError(f'application root is not a directory: {root}')
|
||||
for current, directories, files in os.walk(root, followlinks=False, onerror=raise_walk_error):
|
||||
for name in directories:
|
||||
candidate = os.path.join(current, name)
|
||||
if _is_reparse_point(candidate):
|
||||
relative = os.path.relpath(candidate, root).replace(os.sep, '/')
|
||||
if name.lower() == '__pycache__':
|
||||
raise RuntimeError(f'application __pycache__ link is forbidden: {relative}')
|
||||
raise RuntimeError(f'application directory reparse point is forbidden: {relative}')
|
||||
relative_current = os.path.relpath(current, root)
|
||||
in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep))
|
||||
for name in files:
|
||||
candidate = os.path.join(current, name)
|
||||
relative = os.path.relpath(candidate, root).replace(os.sep, '/')
|
||||
if _is_reparse_point(candidate):
|
||||
raise RuntimeError(f'application file reparse point is forbidden: {relative}')
|
||||
if in_cache and name.lower().endswith('.pyc'):
|
||||
raise RuntimeError(f'application __pycache__ bytecode is forbidden: {relative}')
|
||||
|
||||
|
||||
def _verify_manifest(metadata, inherited):
|
||||
manifest = metadata.get('code_manifest')
|
||||
if not isinstance(manifest, dict) or manifest.get('schema') != MANIFEST_SCHEMA:
|
||||
raise RuntimeError('unsupported child code manifest')
|
||||
expected_digest = str(metadata.get('code_manifest_sha256') or '')
|
||||
if not hmac.compare_digest(_manifest_digest(manifest), expected_digest):
|
||||
raise RuntimeError('child code manifest digest mismatch')
|
||||
if not hmac.compare_digest(expected_digest, inherited['code_manifest_sha256']):
|
||||
raise RuntimeError('inherited child code manifest mismatch')
|
||||
root_value = manifest.get('root') or ''
|
||||
files = manifest.get('files')
|
||||
executables = manifest.get('executables')
|
||||
assets = manifest.get('assets')
|
||||
if not root_value or not isinstance(files, dict) or not isinstance(executables, dict) or not isinstance(assets, dict):
|
||||
raise RuntimeError('child code manifest is incomplete')
|
||||
raw_root = os.path.abspath(os.fspath(root_value))
|
||||
_reject_cached_bytecode(raw_root)
|
||||
root = _canonical(raw_root)
|
||||
manifested_code = set()
|
||||
for name, value in files.items():
|
||||
if not isinstance(value, dict):
|
||||
raise RuntimeError('child code manifest file entry is invalid')
|
||||
expected_path = _canonical(os.path.join(root, *str(name).split('/')))
|
||||
path = _canonical(value.get('path') or '')
|
||||
if path != expected_path or not hmac.compare_digest(_sha256_file(path), str(value.get('sha256') or '')):
|
||||
raise RuntimeError(f'child code authority drifted: {name}')
|
||||
try:
|
||||
contained = os.path.commonpath((root, path)) == root
|
||||
except ValueError:
|
||||
contained = False
|
||||
if contained and str(name).lower().endswith(APPLICATION_IMPORT_SUFFIXES):
|
||||
manifested_code.add(str(name).replace('\\', '/'))
|
||||
current_code = _application_code_files(root)
|
||||
if current_code != manifested_code:
|
||||
added = sorted(current_code - manifested_code)
|
||||
removed = sorted(manifested_code - current_code)
|
||||
detail = added[0] if added else removed[0] if removed else 'unknown'
|
||||
raise RuntimeError(f'application code authority file set drifted: {detail}')
|
||||
for group_name, values in (('executable', executables), ('asset', assets)):
|
||||
for name, value in values.items():
|
||||
if not isinstance(value, dict):
|
||||
raise RuntimeError(f'child {group_name} authority entry is invalid')
|
||||
path = _canonical(value.get('path') or '')
|
||||
if not os.path.isabs(path) or not hmac.compare_digest(_sha256_file(path), str(value.get('sha256') or '')):
|
||||
raise RuntimeError(f'child {group_name} authority drifted: {name}')
|
||||
return root
|
||||
|
||||
|
||||
def _handshake(metadata):
|
||||
control = metadata.get('control') or {}
|
||||
request = {
|
||||
'schema': CONTROL_SCHEMA,
|
||||
'instance_id': metadata['instance_id'],
|
||||
'token': metadata['token'],
|
||||
'action': 'handshake',
|
||||
}
|
||||
encoded = json.dumps(request, ensure_ascii=True, separators=(',', ':')).encode('utf-8') + b'\n'
|
||||
chunks = []
|
||||
total = 0
|
||||
with socket.create_connection((control.get('host'), int(control.get('port') or 0)), timeout=3) as client:
|
||||
client.settimeout(3)
|
||||
client.sendall(encoded)
|
||||
client.shutdown(socket.SHUT_WR)
|
||||
while True:
|
||||
chunk = client.recv(65536)
|
||||
if not chunk:
|
||||
break
|
||||
total += len(chunk)
|
||||
if total > MAX_HANDSHAKE_BYTES:
|
||||
raise RuntimeError('supervisor handshake exceeds its byte bound')
|
||||
chunks.append(chunk)
|
||||
response = json.loads(b''.join(chunks).decode('utf-8'))
|
||||
if (
|
||||
not isinstance(response, dict)
|
||||
or response.get('schema') != CONTROL_SCHEMA
|
||||
or response.get('instance_id') != metadata['instance_id']
|
||||
or response.get('ok') is not True
|
||||
or not isinstance(response.get('result'), dict)
|
||||
):
|
||||
raise RuntimeError('authenticated supervisor handshake failed')
|
||||
return response['result']
|
||||
|
||||
|
||||
def _authenticate(kind):
|
||||
inherited = {name: str(os.getenv(variable) or '') for name, variable in ENV.items()}
|
||||
if not all(inherited.values()):
|
||||
raise RuntimeError('direct mutation is retired; use an authenticated active supervisor command')
|
||||
if inherited['kind'] != kind:
|
||||
raise RuntimeError('supervised child kind does not match the bootstrap entrypoint')
|
||||
instance_file = _canonical(inherited['instance_file'])
|
||||
metadata = _read_object(instance_file)
|
||||
if metadata.get('schema') != 2 or _canonical(metadata.get('instance_file') or '') != instance_file:
|
||||
raise RuntimeError('supervisor child instance metadata is invalid')
|
||||
for key in ('instance_id', 'token', 'config_sha256', 'supervisor_sha256', 'code_manifest_sha256'):
|
||||
if not hmac.compare_digest(str(metadata.get(key) or ''), inherited[key]):
|
||||
raise RuntimeError(f'supervisor child {key} authority mismatch')
|
||||
if str(metadata.get('activation_state') or '').upper() != 'ACTIVE':
|
||||
raise RuntimeError('supervisor is not ACTIVE; child launch is refused')
|
||||
if not hmac.compare_digest(_sha256_file(metadata['config_path']), inherited['config_sha256']):
|
||||
raise RuntimeError('supervisor config authority drifted')
|
||||
if not hmac.compare_digest(_sha256_file(metadata['supervisor_path']), inherited['supervisor_sha256']):
|
||||
raise RuntimeError('supervisor script authority drifted')
|
||||
root = _verify_manifest(metadata, inherited)
|
||||
dsn = str(os.getenv('TRUF_MANAGED_POSTGRES_DSN') or '')
|
||||
if kind == 'janitor':
|
||||
if dsn or os.getenv('SCANNER_DB_URL') or os.getenv('DATABASE_URL'):
|
||||
raise RuntimeError('janitor child must not receive database mutation capability')
|
||||
else:
|
||||
dsn_digest = hashlib.sha256(dsn.encode('utf-8')).hexdigest() if dsn else ''
|
||||
if not dsn or not hmac.compare_digest(dsn_digest, inherited['dsn_sha256']):
|
||||
raise RuntimeError('managed PostgreSQL DSN authority mismatch')
|
||||
for variable in ('SCANNER_DB_URL', 'DATABASE_URL'):
|
||||
if not hmac.compare_digest(str(os.getenv(variable) or ''), dsn):
|
||||
raise RuntimeError(f'{variable} does not match managed PostgreSQL authority')
|
||||
handshake = _handshake(metadata)
|
||||
if handshake.get('activation_state') != 'ACTIVE' or handshake.get('instance_id') != metadata['instance_id']:
|
||||
raise RuntimeError('supervisor handshake did not confirm ACTIVE authority')
|
||||
for key in ('config_sha256', 'supervisor_sha256', 'code_manifest_sha256'):
|
||||
if not hmac.compare_digest(str(handshake.get(key) or ''), inherited[key]):
|
||||
raise RuntimeError(f'supervisor handshake {key} mismatch')
|
||||
if not hmac.compare_digest(str(handshake.get('canonical_dsn_sha256') or ''), inherited['dsn_sha256']):
|
||||
raise RuntimeError('supervisor handshake PostgreSQL authority mismatch')
|
||||
return root, metadata
|
||||
|
||||
|
||||
def main():
|
||||
if not sys.flags.isolated or not sys.flags.no_site or not sys.flags.dont_write_bytecode:
|
||||
raise RuntimeError('supervised child bootstrap requires isolated no-site bytecode-free startup (-I -S -B)')
|
||||
if len(sys.argv) < 2:
|
||||
raise SystemExit('supervised child bootstrap kind is required')
|
||||
kind = str(sys.argv[1]).strip().lower()
|
||||
root, metadata = _authenticate(kind)
|
||||
_enable_dependency_paths(kind)
|
||||
arguments = list(sys.argv[2:])
|
||||
if kind != 'keycheck-provider' and arguments[:1] == ['--']:
|
||||
arguments.pop(0)
|
||||
if kind in ('scanner', 'discovery-producer'):
|
||||
entrypoint = os.path.join(root, 'console_runner.py')
|
||||
elif kind == 'docker-shadow':
|
||||
entrypoint = os.path.join(root, 'docker_shadow.py')
|
||||
elif kind == 'keycheck':
|
||||
entrypoint = os.path.join(root, 'keycheck_runner.py')
|
||||
elif kind == 'dashboard':
|
||||
entrypoint = os.path.join(root, 'dashboard.py')
|
||||
elif kind == 'janitor':
|
||||
entrypoint = os.path.join(root, 'janitor.py')
|
||||
elif kind == 'result-ingester':
|
||||
entrypoint = os.path.join(root, 'result_ingester.py')
|
||||
elif kind == 'jsonl-projector':
|
||||
entrypoint = os.path.join(root, 'jsonl_projector.py')
|
||||
elif kind == 'worker-api':
|
||||
entrypoint = os.path.join(root, 'worker_api.py')
|
||||
elif kind == 'keycheck-provider':
|
||||
if not arguments:
|
||||
raise RuntimeError('keycheck provider bootstrap entrypoint is required')
|
||||
relative = arguments.pop(0).replace('\\', '/')
|
||||
if not arguments or arguments.pop(0) != '--':
|
||||
raise RuntimeError('keycheck provider bootstrap separator is required')
|
||||
if '--' in arguments:
|
||||
raise RuntimeError('duplicate keycheck provider bootstrap separator')
|
||||
entrypoint = _canonical(os.path.join(root, *relative.split('/')))
|
||||
provider_root = _canonical(os.path.join(root, 'keycheckers'))
|
||||
try:
|
||||
allowed = os.path.commonpath((provider_root, entrypoint)) == provider_root
|
||||
except ValueError:
|
||||
allowed = False
|
||||
if not allowed or not relative.lower().endswith('.py'):
|
||||
raise RuntimeError('keycheck provider bootstrap entrypoint is outside authority')
|
||||
else:
|
||||
raise RuntimeError('unsupported supervised child bootstrap kind')
|
||||
entrypoint = _canonical(entrypoint)
|
||||
files = (metadata.get('code_manifest') or {}).get('files') or {}
|
||||
if not any(_canonical(value.get('path') or '') == entrypoint for value in files.values() if isinstance(value, dict)):
|
||||
raise RuntimeError('child entrypoint is absent from immutable authority')
|
||||
sys.path.insert(0, root)
|
||||
try:
|
||||
if kind == 'dashboard':
|
||||
sys.argv = ['streamlit', 'run', entrypoint, *arguments]
|
||||
runpy.run_module('streamlit', run_name='__main__', alter_sys=True)
|
||||
else:
|
||||
sys.argv = [entrypoint, *arguments]
|
||||
runpy.run_path(entrypoint, run_name='__main__')
|
||||
except Exception as exc:
|
||||
raise ChildRuntimeError(str(exc)) from exc
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
main()
|
||||
except ChildRuntimeError as exc:
|
||||
raise SystemExit(f'supervised child runtime failed: {exc}') from exc
|
||||
except Exception as exc:
|
||||
raise SystemExit(f'supervised child bootstrap rejected launch: {exc}') from exc
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,127 @@
|
||||
"""Pure translation of the reviewed Windows config; YAML and file copies are caller-owned."""
|
||||
|
||||
from copy import deepcopy
|
||||
import ntpath
|
||||
|
||||
|
||||
# Only these reviewed absolute Windows paths have known container replacements.
|
||||
_FIXED_GLOBAL_PATHS = {
|
||||
'root_dir': (r'D:\truf', '/opt/truf'),
|
||||
'project_dir': (r'D:\truf\app', '/opt/truf/app'),
|
||||
'runtime_dir': (r'D:\truf\runtime', '/data/runtime-linux'),
|
||||
'postgres_data_dir': (r'S:\postgres-data', '/data/postgres-linux'),
|
||||
'postgres_bin_dir': (r'D:\truf\runtime\postgres\pgsql\bin', '/usr/lib/postgresql/16/bin'),
|
||||
'result_bundle_dir': (r'S:\scanner-result-bundles', '/data/scanner-result-bundles'),
|
||||
'work_dir': (r'S:\scanner-work', '/data/scanner-work'),
|
||||
'control_dir': (r'D:\truf\runtime\control', '/run/truf/control'),
|
||||
'trufflehog_path': (r'C:\Tools\trufflehog.exe', '/usr/local/bin/trufflehog'),
|
||||
'proxy_file': (r'D:\truf\runtime\proxy.txt', '/data/runtime-linux/proxy.txt'),
|
||||
'secrets_file': (r'D:\truf\app\secrets.yaml', '/data/config/secrets.yaml'),
|
||||
'trufflehog_config': (
|
||||
r'D:\truf\app\trufflehog-custom-detectors.yaml',
|
||||
'/data/config/trufflehog-custom-detectors.yaml',
|
||||
),
|
||||
}
|
||||
_PATH_FIELDS = {
|
||||
'global': (
|
||||
'result_spool_dir', 'legacy_result_spool_dir', 'results_dir', 'queue_dir',
|
||||
'state_dir', 'log_dir', 'keycheck_dir', 'postman_cache_dir', 'gharchive_cache_dir',
|
||||
'database_path', 'state_file', 'api_proxy_file', 'download_proxy_file',
|
||||
'dashboard_db_path', 'scan_limiter_db', 'dockerhub_tag_cache_path',
|
||||
),
|
||||
'supervisor': ('log_dir', 'supervisor_log', 'status_file', 'dashboard_log', 'state_dir'),
|
||||
'keychecks': ('input', 'proxy_file', 'keycheck_dir', 'summary_tsv', 'summary_json', 'alive_summary_tsv'),
|
||||
}
|
||||
_SOURCE_PATH_FIELDS = ('target_file', 'trufflehog_config', 'postman_cache_dir', 'gharchive_cache_dir')
|
||||
_WINDOWS_KNOBS = (
|
||||
'trufflehog_job_memory_limit_bytes',
|
||||
'trufflehog_windows_job_cpu_weight',
|
||||
'trufflehog_windows_memory_priority',
|
||||
)
|
||||
|
||||
|
||||
def translate_windows_config(original, baseline):
|
||||
"""Return an independent config and sorted, changed dotted key paths (never values).
|
||||
|
||||
``baseline`` is the parsed config.linux.yaml, not a general merge source.
|
||||
Fixed paths must match the container contract. Other known path fields keep
|
||||
relative paths/templates with Linux separators; unreviewed Windows absolute
|
||||
paths raise ValueError naming only the key. No environment, filesystem,
|
||||
runtime, database, or YAML operations are performed.
|
||||
"""
|
||||
if not isinstance(original, dict) or not isinstance(baseline, dict):
|
||||
raise TypeError('original and baseline must be dictionaries')
|
||||
config = deepcopy(original)
|
||||
adjusted = set()
|
||||
|
||||
def assign(mapping, key, value, prefix):
|
||||
if key not in mapping or mapping[key] != value:
|
||||
mapping[key] = deepcopy(value)
|
||||
adjusted.add(prefix + '.' + key)
|
||||
|
||||
def path_value(value, key_path, approved=None):
|
||||
if value is None:
|
||||
return value
|
||||
if not isinstance(value, str):
|
||||
raise ValueError('Expected path string at ' + key_path)
|
||||
if ntpath.splitdrive(value)[0] or value.startswith('\\'):
|
||||
if approved is None or ntpath.normcase(ntpath.normpath(value)) != ntpath.normcase(ntpath.normpath(approved[0])):
|
||||
raise ValueError('Unsupported Windows path at ' + key_path)
|
||||
return approved[1]
|
||||
return value.replace('\\', '/')
|
||||
|
||||
linux_paths = {key: pair[1] for key, pair in _FIXED_GLOBAL_PATHS.items()}
|
||||
control = _FIXED_GLOBAL_PATHS['control_dir']
|
||||
supervisor_paths = {'control_dir': control}
|
||||
for key, filename in (('instance_file', 'supervisor.instance.json'), ('lock_file', 'supervisor.lock')):
|
||||
supervisor_paths[key] = (control[0] + '\\' + filename, control[1] + '/' + filename)
|
||||
|
||||
for section, fixed_paths in (('global', _FIXED_GLOBAL_PATHS), ('supervisor', supervisor_paths)):
|
||||
mapping = config.setdefault(section, {})
|
||||
for key, pair in fixed_paths.items():
|
||||
key_path = section + '.' + key
|
||||
value = path_value(mapping.get(key), key_path, pair)
|
||||
if key == 'trufflehog_path' and value not in (None, '', 'trufflehog', 'trufflehog.exe', pair[1]):
|
||||
raise ValueError('Unsupported executable path at ' + key_path)
|
||||
# The copied original policy intentionally replaces the image policy.
|
||||
if key != 'trufflehog_config':
|
||||
try:
|
||||
baseline_path = baseline[section][key].format_map(linux_paths)
|
||||
except (KeyError, AttributeError, ValueError):
|
||||
raise ValueError('Invalid Linux baseline path at ' + key_path) from None
|
||||
if baseline_path != pair[1]:
|
||||
raise ValueError('Invalid Linux baseline path at ' + key_path)
|
||||
assign(mapping, key, pair[1], section)
|
||||
|
||||
groups = [(section, config.get(section, {}), fields) for section, fields in _PATH_FIELDS.items()]
|
||||
groups.extend(('sources.' + name, source, _SOURCE_PATH_FIELDS)
|
||||
for name, source in config.get('sources', {}).items())
|
||||
for prefix, mapping, fields in groups:
|
||||
for key in fields:
|
||||
if key not in mapping:
|
||||
continue
|
||||
approved = None
|
||||
if key in ('proxy_file', 'api_proxy_file', 'download_proxy_file'):
|
||||
approved = _FIXED_GLOBAL_PATHS['proxy_file']
|
||||
elif key == 'trufflehog_config':
|
||||
approved = _FIXED_GLOBAL_PATHS[key]
|
||||
value = path_value(mapping[key], prefix + '.' + key, approved)
|
||||
if key == 'trufflehog_config' and value in (
|
||||
'{project_dir}/trufflehog-custom-detectors.yaml', 'trufflehog-custom-detectors.yaml',
|
||||
):
|
||||
value = linux_paths[key]
|
||||
assign(mapping, key, value, prefix)
|
||||
|
||||
for key in ('max_active_scans', 'opportunistic_scan_slots') + _WINDOWS_KNOBS:
|
||||
assign(config['global'], key, baseline['global'][key], 'global')
|
||||
for key in ('interactive', 'autostart', 'control_host', 'control_port'):
|
||||
assign(config['supervisor'], key, baseline['supervisor'][key], 'supervisor')
|
||||
assign(config['supervisor'].setdefault('dashboard', {}), 'enabled',
|
||||
baseline['supervisor']['dashboard']['enabled'], 'supervisor.dashboard')
|
||||
for name, source in config.get('sources', {}).items():
|
||||
source_baseline = baseline.get('sources', {}).get(name, {})
|
||||
for key in _WINDOWS_KNOBS:
|
||||
if key in source or key in source_baseline:
|
||||
assign(source, key, source_baseline.get(key, baseline['global'][key]), 'sources.' + name)
|
||||
|
||||
return config, sorted(adjusted)
|
||||
@@ -0,0 +1,416 @@
|
||||
"""Explicit maintenance-only loss acknowledgement for the reviewed copied output.
|
||||
|
||||
No CLI, startup hook, connection creation, PostgreSQL lifecycle, or output rebuild.
|
||||
The caller must retain initialize.lock and ClusterAuthorityLock through this call
|
||||
AND subsequent positively verified maintenance stop, including every exception or
|
||||
uncertain commit. It must keep the target isolated with no workers/network clients.
|
||||
Use an idle, writable, autocommit=True psycopg connection with dict_row and quiet server log
|
||||
settings. The supplied runtime is the prepared container_runtime module, not its
|
||||
main()/initialize() entrypoint. Never call this during normal initialized startup.
|
||||
|
||||
The immutable PREPARED journal describes intent, not a fabricated completed rename.
|
||||
An exact before state can be applied; an exact after state is a read-only retry.
|
||||
Partial journals and later output require review, never automatic cleanup/rewind.
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import re
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
|
||||
|
||||
DATA = Path('/data')
|
||||
RUN = Path('/run/truf')
|
||||
APPROVED_MANIFEST_SHA256 = '08344147133c37d4b6f404cf4fac3e59d58f94917f1fa58a77cbb68c36db7e8a'
|
||||
JOURNAL_NAME = 'found-secrets-loss-g13-g14.prepared.json'
|
||||
FORMAT = 'truf-found-secrets-loss-g13-g14-v1'
|
||||
LOSS = 'Previously published copied output intentionally lost; all PostgreSQL history retained. No rotation or rename occurred.'
|
||||
OLD_GENERATION, NEW_GENERATION, OLD_OFFSET = 13, 14, 97783145
|
||||
APPEND_COUNT, ROTATION_COUNT = 38024, 13
|
||||
MAX_JOURNAL = 1024 * 1024
|
||||
STREAM_COLUMNS = {'stream_name', 'base_relative_path', 'current_generation', 'rotation_bytes',
|
||||
'max_generations', 'created_at', 'updated_at'}
|
||||
CURSOR_COLUMNS = {'stream_name', 'generation', 'committed_offset', 'last_append_id', 'last_job_id',
|
||||
'last_event_id', 'last_event_hash', 'updated_at'}
|
||||
|
||||
|
||||
class ProjectionRecoveryError(RuntimeError):
|
||||
"""Safe diagnostic only; caller still owns maintenance/stop authority."""
|
||||
|
||||
|
||||
def _encoded(value):
|
||||
return (json.dumps(value, ensure_ascii=True, sort_keys=True,
|
||||
separators=(',', ':'), allow_nan=False) + '\n').encode('ascii')
|
||||
|
||||
|
||||
def recover_found_secrets_projection(runtime, connection, *, system_identifier, manifest_sha256,
|
||||
initialize_lock, authority_lock):
|
||||
"""Apply only the pinned g13 loss transition, or recognize its exact retry.
|
||||
|
||||
Caller-owned locks must be acquired runtime_security lock objects for the fixed
|
||||
target. This function never closes the connection or releases those locks.
|
||||
Its own file lock and transaction-scoped advisory/table locks exclude writers.
|
||||
Returned 'committed'/'already-committed' is not target readiness or stop proof.
|
||||
journal_sha256 hashes the complete immutable file, not just its inner record.
|
||||
"""
|
||||
stage = 'preflight'
|
||||
deadline = time.monotonic() + 10800
|
||||
try:
|
||||
from container_import import _fsync_dir, _identifier, _input, _integer, _json, _manifest, _regular, _write, MAX_MANIFEST
|
||||
from runtime_security import ClusterAuthorityLock, PrivateFileLock
|
||||
|
||||
runtime.require_container()
|
||||
if (runtime.DATA != DATA or runtime.RUN != RUN
|
||||
or runtime.INITIALIZED != DATA / 'initialized.json'
|
||||
or runtime.INITIALIZE_LOCK != DATA / 'initialize.lock'
|
||||
or manifest_sha256 != APPROVED_MANIFEST_SHA256
|
||||
or not isinstance(system_identifier, str)
|
||||
or re.fullmatch(r'[1-9][0-9]{0,19}', system_identifier) is None
|
||||
or connection.closed or connection.broken or connection.autocommit is not True
|
||||
or int(connection.info.transaction_status) != 0):
|
||||
raise ValueError()
|
||||
config_dir, results = DATA / 'config', DATA / 'runtime-linux/results'
|
||||
identity_path = DATA / 'runtime-linux/postgres/cluster_identity.json'
|
||||
manifest_path = config_dir / 'windows-import-manifest.json'
|
||||
journal_path = config_dir / JOURNAL_NAME
|
||||
projector_lock = results / '.jsonl-projector.lock'
|
||||
endpoint = _encoded({'database': 'truf', 'host': '127.0.0.1', 'port': 5432,
|
||||
'schema': 'public'}).decode('ascii').strip()
|
||||
data_hash = hashlib.sha256(str(DATA / 'postgres-linux').encode('utf-8')).hexdigest()
|
||||
endpoint_hash = hashlib.sha256(endpoint.encode('ascii')).hexdigest()
|
||||
|
||||
def files_stopped():
|
||||
if (time.monotonic() >= deadline or getattr(runtime, '_shutdown_requested', True) is not False
|
||||
or not isinstance(initialize_lock, PrivateFileLock) or initialize_lock.acquired is not True
|
||||
or initialize_lock.path != os.path.normcase(str(runtime.INITIALIZE_LOCK))
|
||||
or not isinstance(authority_lock, ClusterAuthorityLock) or authority_lock.acquired is not True
|
||||
or authority_lock.data_directory != str(DATA / 'postgres-linux')
|
||||
or authority_lock.endpoint_identity != endpoint
|
||||
or authority_lock.path != str(DATA / 'runtime-linux/postgres' / f'.cluster-authority-{data_hash}.lock')
|
||||
or authority_lock.endpoint_path != str(RUN / 'authority' / f'endpoint-{endpoint_hash}.lock')):
|
||||
raise ValueError()
|
||||
for directory in (DATA, config_dir, results, identity_path.parent,
|
||||
DATA / 'runtime-linux/logs', RUN, RUN / 'control'):
|
||||
runtime.private_path(directory, directory=True)
|
||||
_regular(runtime.INITIALIZE_LOCK, runtime)
|
||||
for path in (runtime.INITIALIZED, RUN / 'control/supervisor.instance.json',
|
||||
RUN / 'control/supervisor.pid', DATA / 'runtime-linux/logs/supervisor.instance.json',
|
||||
DATA / 'runtime-linux/logs/supervisor.pid'):
|
||||
try:
|
||||
path.lstat()
|
||||
except FileNotFoundError:
|
||||
continue
|
||||
raise ValueError()
|
||||
with os.scandir(results) as entries:
|
||||
for index, entry in enumerate(entries):
|
||||
if index >= 100000 or entry.name.casefold() == 'found_secrets' or entry.name.casefold().startswith('found_secrets.'):
|
||||
raise ValueError()
|
||||
|
||||
def read_private(path, limit):
|
||||
with _input(path, runtime) as (handle, before):
|
||||
if not 0 < before[4] <= limit:
|
||||
raise ValueError()
|
||||
raw = handle.read(limit + 1)
|
||||
if len(raw) != before[4]:
|
||||
raise ValueError()
|
||||
return raw, _json(raw)
|
||||
|
||||
files_stopped()
|
||||
identity_raw, identity = read_private(identity_path, MAX_JOURNAL)
|
||||
manifest_raw, _ = read_private(manifest_path, MAX_MANIFEST)
|
||||
manifest, _ = _manifest(manifest_raw, manifest_sha256)
|
||||
expected_identity = {'pg_major': 16, 'system_identifier': system_identifier,
|
||||
'data_directory': str(DATA / 'postgres-linux'), 'database': 'truf',
|
||||
'user': 'truf', 'port': 5432}
|
||||
if (not isinstance(identity, dict) or any(identity.get(k) != v for k, v in expected_identity.items())
|
||||
or manifest['database']['system_identifier'] == system_identifier
|
||||
or 'sequence_states' not in manifest['database']):
|
||||
raise ValueError()
|
||||
binding = {'system_identifier': system_identifier, 'manifest_sha256': manifest_sha256,
|
||||
'identity_sha256': hashlib.sha256(identity_raw).hexdigest()}
|
||||
|
||||
def online():
|
||||
if time.monotonic() >= deadline or runtime._shutdown_requested:
|
||||
raise ValueError()
|
||||
connection.execute('SELECT pg_catalog.pg_stat_clear_snapshot()')
|
||||
row = connection.execute("""SELECT pg_catalog.current_database() AS database,
|
||||
current_user AS user_name, pg_catalog.current_setting('data_directory') AS data_directory,
|
||||
pg_catalog.current_setting('port')::int AS port,
|
||||
pg_catalog.current_setting('server_version_num')::int AS version_num,
|
||||
pg_catalog.pg_is_in_recovery() AS in_recovery,
|
||||
(SELECT system_identifier::text FROM pg_catalog.pg_control_system()) AS system_identifier,
|
||||
(SELECT rolsuper FROM pg_catalog.pg_roles WHERE rolname = current_user) AS superuser,
|
||||
(SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE backend_type = 'client backend'
|
||||
AND pid <> pg_catalog.pg_backend_pid()) AS other_clients,
|
||||
pg_catalog.current_schema() AS schema_name,
|
||||
pg_catalog.current_setting('search_path') AS search_path,
|
||||
pg_catalog.current_setting('transaction_read_only') AS read_only,
|
||||
pg_catalog.current_setting('fsync') AS fsync,
|
||||
pg_catalog.current_setting('full_page_writes') AS full_page_writes,
|
||||
EXISTS (SELECT 1 FROM pg_catalog.pg_namespace n CROSS JOIN LATERAL pg_catalog.aclexplode(
|
||||
COALESCE(n.nspacl, pg_catalog.acldefault('n', n.nspowner))) acl
|
||||
WHERE n.nspname = 'public' AND acl.grantee = 0 AND acl.privilege_type = 'CREATE') AS public_create""").fetchone()
|
||||
expected = {'database': 'truf', 'user_name': 'truf', 'data_directory': str(DATA / 'postgres-linux'),
|
||||
'port': 5432, 'system_identifier': system_identifier, 'in_recovery': False,
|
||||
'superuser': True, 'other_clients': 0, 'schema_name': 'public',
|
||||
'search_path': 'public', 'read_only': 'off', 'public_create': False,
|
||||
'fsync': 'on', 'full_page_writes': 'on'}
|
||||
if (not isinstance(row, dict) or any(row.get(k) != v for k, v in expected.items())
|
||||
or _integer(row.get('version_num')) // 10000 != 16):
|
||||
raise ValueError()
|
||||
|
||||
def metadata():
|
||||
rows = connection.execute("""SELECT s.stream_name, c.stream_name AS cursor_stream_name,
|
||||
pg_catalog.row_to_json(s) AS stream, pg_catalog.row_to_json(c) AS cursor
|
||||
FROM public.projection_streams s FULL JOIN public.projection_cursors c
|
||||
ON c.stream_name = s.stream_name""").fetchall()
|
||||
seen, found = set(), None
|
||||
scans = {'scan_results': 'scan_results.jsonl', 'found_secrets': 'found_secrets.jsonl',
|
||||
'scan_errors': 'scan_errors.log'}
|
||||
for row in rows:
|
||||
name, stream, cursor = row['stream_name'], row['stream'], row['cursor']
|
||||
if (not isinstance(name, str) or name in seen or row['cursor_stream_name'] != name
|
||||
or not isinstance(stream, dict) or set(stream) != STREAM_COLUMNS
|
||||
or not isinstance(cursor, dict) or set(cursor) != CURSOR_COLUMNS
|
||||
or stream['stream_name'] != name or cursor['stream_name'] != name
|
||||
or _integer(stream['current_generation']) != _integer(cursor['generation'])):
|
||||
raise ValueError()
|
||||
seen.add(name)
|
||||
_integer(cursor['committed_offset'])
|
||||
_integer(stream['rotation_bytes'], 1)
|
||||
_integer(stream['max_generations'])
|
||||
expected_path = scans.get(name)
|
||||
if expected_path is None:
|
||||
match = re.fullmatch(r'keycheck:([a-z0-9][a-z0-9_.-]{0,63}):(results|status)', name)
|
||||
if not match:
|
||||
raise ValueError()
|
||||
suffix = 'Results.jsonl' if match[2] == 'results' else 'Checked.txt'
|
||||
expected_path = f'{match[1]}/{match[1]}{suffix}'
|
||||
if stream['base_relative_path'] != expected_path:
|
||||
raise ValueError()
|
||||
if name == 'found_secrets':
|
||||
found = {'stream': stream, 'cursor': cursor}
|
||||
if len(seen) != 34 or not scans.keys() <= seen:
|
||||
raise ValueError()
|
||||
return found
|
||||
|
||||
tables = sorted(manifest['database']['table_counts'])
|
||||
sequences = manifest['database']['sequence_states']
|
||||
if set(sequences) != {'public'}:
|
||||
raise ValueError()
|
||||
|
||||
def preserved():
|
||||
proof = {}
|
||||
for table in tables:
|
||||
if time.monotonic() >= deadline or runtime._shutdown_requested:
|
||||
raise ValueError()
|
||||
where = " WHERE t.stream_name <> 'found_secrets'" if table in ('projection_streams', 'projection_cursors') else ''
|
||||
digest, count = hashlib.sha256(), 0
|
||||
with connection.cursor(name='found_loss_digest') as cursor:
|
||||
cursor.itersize = 1000
|
||||
cursor.execute("SELECT pg_catalog.encode(pg_catalog.sha256(pg_catalog.convert_to("
|
||||
"pg_catalog.row_to_json(t)::text, 'UTF8')), 'hex') COLLATE \"C\" AS digest FROM public."
|
||||
+ _identifier(table) + ' AS t' + where + ' ORDER BY digest')
|
||||
for row in cursor:
|
||||
value = row['digest']
|
||||
if not isinstance(value, str) or re.fullmatch(r'[0-9a-f]{64}', value) is None:
|
||||
raise ValueError()
|
||||
digest.update(value.encode('ascii'))
|
||||
count += 1
|
||||
if count % 1000 == 0 and (time.monotonic() >= deadline or runtime._shutdown_requested):
|
||||
raise ValueError()
|
||||
if count != manifest['database']['table_counts'][table] - int(bool(where)):
|
||||
raise ValueError()
|
||||
proof[table] = {'rows': count, 'sha256': digest.hexdigest()}
|
||||
rows = connection.execute("""SELECT c.relname AS name FROM pg_catalog.pg_class c
|
||||
JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
|
||||
WHERE n.nspname = 'public' AND c.relkind = 'S' ORDER BY c.relname""").fetchall()
|
||||
if {row['name'] for row in rows} != set(sequences['public']):
|
||||
raise ValueError()
|
||||
for row in rows:
|
||||
state = connection.execute('SELECT last_value, is_called FROM public.' + _identifier(row['name'])).fetchone()
|
||||
if _encoded(state) != _encoded(sequences['public'][row['name']]):
|
||||
raise ValueError()
|
||||
return {'tables': proof, 'sequences': {'count': len(rows),
|
||||
'sha256': hashlib.sha256(_encoded(sequences)).hexdigest()}}
|
||||
|
||||
with PrivateFileLock(str(projector_lock)):
|
||||
_regular(projector_lock, runtime)
|
||||
with connection.transaction():
|
||||
stage = 'database-fences'
|
||||
connection.execute('SET TRANSACTION ISOLATION LEVEL READ COMMITTED')
|
||||
online()
|
||||
connection.execute("SET LOCAL lock_timeout = '5s'")
|
||||
connection.execute("SET LOCAL statement_timeout = '10800s'")
|
||||
connection.execute("SET LOCAL temp_file_limit = '4GB'")
|
||||
connection.execute("SET LOCAL work_mem = '128MB'")
|
||||
connection.execute('SET LOCAL max_parallel_workers_per_gather = 0')
|
||||
connection.execute("SET LOCAL row_security = off")
|
||||
connection.execute('SET LOCAL synchronous_commit = on')
|
||||
for setting, value in (('log_min_error_statement', 'panic'), ('log_min_messages', 'panic'),
|
||||
('log_statement', 'none'), ('log_min_duration_statement', '-1')):
|
||||
connection.execute('SET LOCAL ' + setting + " = '" + value + "'")
|
||||
for arguments in ((1414681926, 1785753445), (1414681926, 1768842867), (781273968142991337,)):
|
||||
lock = connection.execute('SELECT pg_catalog.pg_try_advisory_xact_lock('
|
||||
+ ','.join(['%s'] * len(arguments)) + ') AS locked', arguments).fetchone()
|
||||
if not lock or lock['locked'] is not True:
|
||||
raise ValueError()
|
||||
catalog_sql = """SELECT c.relname AS name, c.relkind AS kind FROM pg_catalog.pg_class c
|
||||
JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
|
||||
WHERE n.nspname = 'public' AND c.relkind IN ('r','p','f') ORDER BY c.relname"""
|
||||
catalog = connection.execute(catalog_sql).fetchall()
|
||||
if len(catalog) != len(tables) or {row['name'] for row in catalog} != set(tables) or any(row['kind'] != 'r' for row in catalog):
|
||||
raise ValueError()
|
||||
connection.execute('LOCK TABLE ' + ','.join('public.' + _identifier(table) for table in tables)
|
||||
+ ' IN EXCLUSIVE MODE NOWAIT')
|
||||
online()
|
||||
stage = 'history-gates'
|
||||
gates = connection.execute("""WITH a AS (
|
||||
SELECT count(*) AS appends, max(a.generation) AS append_highwater,
|
||||
count(*) FILTER (WHERE a.state IS DISTINCT FROM 'appended' OR j.id IS NULL
|
||||
OR j.status IS DISTINCT FROM 'completed' OR j.capacity_released IS DISTINCT FROM 1
|
||||
OR j.job_kind IS DISTINCT FROM 'scan_event' OR (j.required_stream_mask & 2) IS DISTINCT FROM 2
|
||||
OR a.event_id IS DISTINCT FROM j.event_id OR a.event_hash IS DISTINCT FROM j.event_hash
|
||||
OR a.generation IS NULL OR a.generation NOT BETWEEN 0 AND 13
|
||||
OR a.byte_offset IS NULL OR a.byte_offset < 0 OR a.byte_length IS NULL OR a.byte_length < 0
|
||||
OR a.record_count IS NULL OR a.record_count < 0
|
||||
OR (a.generation = 13 AND a.byte_offset::numeric + a.byte_length::numeric > 97783145)) AS bad_appends
|
||||
FROM public.projection_appends a LEFT JOIN public.projection_jobs j ON j.id = a.job_id
|
||||
WHERE a.stream_name = 'found_secrets'), r AS (
|
||||
SELECT count(*) AS rotations, max(to_generation) AS rotation_highwater,
|
||||
count(*) FILTER (WHERE state IS DISTINCT FROM 'completed' OR from_generation IS NULL OR from_generation < 0
|
||||
OR to_generation IS DISTINCT FROM from_generation + 1 OR to_generation > 13
|
||||
OR source_bytes IS NULL OR source_bytes < 0
|
||||
OR segment_relative_path IS DISTINCT FROM
|
||||
'found_secrets.g' || lpad(from_generation::text, 6, '0') || '.jsonl') AS bad_rotations
|
||||
FROM public.projection_rotations WHERE stream_name = 'found_secrets')
|
||||
SELECT a.*, r.*, (SELECT count(*) FROM public.projection_append_audit) AS audits,
|
||||
(SELECT count(*) FROM public.projection_jobs WHERE (required_stream_mask & 2) <> 0
|
||||
AND status <> 'completed') AS pending,
|
||||
(SELECT count(*) FROM public.result_reservations
|
||||
WHERE state IN ('scanning','ready','ingesting','db_committed')) AS reservations,
|
||||
(SELECT count(*) FROM public.target_queue q LEFT JOIN public.result_reservations r
|
||||
ON r.id = q.current_result_reservation_id WHERE q.status = 'in_progress'
|
||||
OR r.state IN ('scanning','ready','ingesting','db_committed')) AS queue_leases,
|
||||
(SELECT count(*) FROM public.docker_content_blobs
|
||||
WHERE state IN ('leased','submitted') OR lease_reservation_id IS NOT NULL) AS blob_leases,
|
||||
(SELECT count(*) FROM pg_catalog.pg_trigger WHERE NOT tgisinternal
|
||||
AND tgrelid IN ('public.projection_streams'::regclass,'public.projection_cursors'::regclass)) AS triggers,
|
||||
(SELECT count(*) FROM pg_catalog.pg_rewrite
|
||||
WHERE ev_class IN ('public.projection_streams'::regclass,'public.projection_cursors'::regclass)) AS rules
|
||||
FROM a CROSS JOIN r""").fetchone()
|
||||
expected_gates = {'appends': APPEND_COUNT, 'append_highwater': 13, 'bad_appends': 0,
|
||||
'rotations': ROTATION_COUNT, 'rotation_highwater': 13, 'bad_rotations': 0,
|
||||
'audits': 0, 'pending': 0, 'reservations': 0, 'queue_leases': 0,
|
||||
'blob_leases': 0, 'triggers': 0, 'rules': 0}
|
||||
if _encoded(gates) != _encoded(expected_gates):
|
||||
raise ValueError()
|
||||
current = metadata()
|
||||
proof = preserved()
|
||||
stage = 'journal'
|
||||
try:
|
||||
journal_path.lstat()
|
||||
except FileNotFoundError:
|
||||
journal = None
|
||||
else:
|
||||
raw, journal = read_private(journal_path, MAX_JOURNAL)
|
||||
if (not isinstance(journal, dict) or set(journal) != {'record', 'sha256'}
|
||||
or raw != _encoded(journal)
|
||||
or journal['sha256'] != hashlib.sha256(_encoded(journal['record'])).hexdigest()):
|
||||
raise ValueError()
|
||||
if journal is None:
|
||||
stamp = datetime.now(timezone.utc).isoformat(timespec='seconds')
|
||||
before = current
|
||||
record = {'format': FORMAT, 'state': 'PREPARED', 'loss': LOSS, 'binding': binding,
|
||||
'prepared_at': stamp, 'before': before, 'preserved': proof,
|
||||
'digest_algorithm': 'sha256-concatenated-sorted-pg-row-sha256-hex-v1'}
|
||||
else:
|
||||
record = journal['record']
|
||||
if (not isinstance(record, dict) or set(record) != {'format', 'state', 'loss', 'binding',
|
||||
'prepared_at', 'before', 'after', 'preserved', 'digest_algorithm'}
|
||||
or record['format'] != FORMAT or record['state'] != 'PREPARED' or record['loss'] != LOSS
|
||||
or _encoded(record['binding']) != _encoded(binding)
|
||||
or _encoded(record['preserved']) != _encoded(proof)
|
||||
or record['digest_algorithm'] != 'sha256-concatenated-sorted-pg-row-sha256-hex-v1'):
|
||||
raise ValueError()
|
||||
stamp, before = record['prepared_at'], record['before']
|
||||
if (not isinstance(stamp, str) or datetime.fromisoformat(stamp).isoformat(timespec='seconds') != stamp
|
||||
or not stamp.endswith('+00:00') or set(before) != {'stream', 'cursor'}
|
||||
or set(before['stream']) != STREAM_COLUMNS or set(before['cursor']) != CURSOR_COLUMNS
|
||||
or before['stream']['stream_name'] != 'found_secrets'
|
||||
or before['stream']['base_relative_path'] != 'found_secrets.jsonl'
|
||||
or before['cursor']['stream_name'] != 'found_secrets'
|
||||
or _integer(before['stream']['current_generation']) != OLD_GENERATION
|
||||
or _integer(before['cursor']['generation']) != OLD_GENERATION
|
||||
or _integer(before['cursor']['committed_offset']) != OLD_OFFSET):
|
||||
raise ValueError()
|
||||
_integer(before['cursor']['last_append_id'], 1)
|
||||
_integer(before['cursor']['last_job_id'], 1)
|
||||
last = connection.execute("""SELECT id, job_id, stream_name, generation, byte_offset,
|
||||
byte_length, event_id, event_hash, state FROM public.projection_appends WHERE id = %s""",
|
||||
(before['cursor']['last_append_id'],)).fetchone()
|
||||
if (not last or last['id'] != before['cursor']['last_append_id']
|
||||
or last['job_id'] != before['cursor']['last_job_id'] or last['stream_name'] != 'found_secrets'
|
||||
or last['generation'] != OLD_GENERATION or last['state'] != 'appended'
|
||||
or _integer(last['byte_offset']) + _integer(last['byte_length']) != OLD_OFFSET
|
||||
or last['event_id'] != before['cursor']['last_event_id']
|
||||
or last['event_hash'] != before['cursor']['last_event_hash']
|
||||
or not isinstance(last['event_id'], str) or not last['event_id']
|
||||
or re.fullmatch(r'[0-9a-f]{64}', last['event_hash'] or '') is None):
|
||||
raise ValueError()
|
||||
after = {key: dict(value) for key, value in before.items()}
|
||||
after['stream'].update(current_generation=NEW_GENERATION, updated_at=stamp)
|
||||
after['cursor'].update(generation=NEW_GENERATION, committed_offset=0, last_append_id=None, updated_at=stamp)
|
||||
if journal is not None and _encoded(record['after']) != _encoded(after):
|
||||
raise ValueError()
|
||||
already = _encoded(current) == _encoded(after)
|
||||
if (not already and _encoded(current) != _encoded(before)) or (already and journal is None):
|
||||
raise ValueError()
|
||||
if journal is None:
|
||||
record['after'] = after
|
||||
journal = {'record': record, 'sha256': hashlib.sha256(_encoded(record)).hexdigest()}
|
||||
encoded = _encoded(journal)
|
||||
if len(encoded) > MAX_JOURNAL:
|
||||
raise ValueError()
|
||||
_write(runtime, journal_path, encoded)
|
||||
if read_private(journal_path, MAX_JOURNAL)[0] != encoded:
|
||||
raise ValueError()
|
||||
# A prior failure may have left complete bytes without a confirmed fsync.
|
||||
with _input(journal_path, runtime) as (handle, _):
|
||||
if handle.read(MAX_JOURNAL + 1) != _encoded(journal):
|
||||
raise ValueError()
|
||||
os.fsync(handle.fileno())
|
||||
_fsync_dir(config_dir)
|
||||
if not already:
|
||||
stage = 'compare-and-swap'
|
||||
result = connection.execute("""UPDATE public.projection_streams AS s
|
||||
SET current_generation = 14, updated_at = %s
|
||||
WHERE s.stream_name = 'found_secrets' AND pg_catalog.to_jsonb(s) = %s::jsonb""",
|
||||
(stamp, _encoded(before['stream']).decode('ascii')))
|
||||
if result.rowcount != 1:
|
||||
raise ValueError()
|
||||
result = connection.execute("""UPDATE public.projection_cursors AS c
|
||||
SET generation = 14, committed_offset = 0, last_append_id = NULL, updated_at = %s
|
||||
WHERE c.stream_name = 'found_secrets' AND pg_catalog.to_jsonb(c) = %s::jsonb""",
|
||||
(stamp, _encoded(before['cursor']).decode('ascii')))
|
||||
if result.rowcount != 1:
|
||||
raise ValueError()
|
||||
if _encoded(metadata()) != _encoded(after) or _encoded(preserved()) != _encoded(proof):
|
||||
raise ValueError()
|
||||
stage = 'precommit'
|
||||
files_stopped()
|
||||
if (read_private(identity_path, MAX_JOURNAL)[0] != identity_raw
|
||||
or read_private(manifest_path, MAX_MANIFEST)[0] != manifest_raw
|
||||
or read_private(journal_path, MAX_JOURNAL)[0] != _encoded(journal)
|
||||
or _encoded(connection.execute(catalog_sql).fetchall()) != _encoded(catalog)):
|
||||
raise ValueError()
|
||||
online()
|
||||
stage = 'commit'
|
||||
return {'status': 'already-committed' if already else 'committed', 'journal_path': str(journal_path),
|
||||
'journal_sha256': hashlib.sha256(_encoded(journal)).hexdigest(), **binding, 'generation': NEW_GENERATION}
|
||||
except BaseException:
|
||||
raise ProjectionRecoveryError('Projection recovery refused at ' + stage
|
||||
+ '; retain caller maintenance authority and verify stop.') from None
|
||||
@@ -0,0 +1,594 @@
|
||||
"""Isolated entrypoint for the private, read-only Docker deployment."""
|
||||
|
||||
import argparse
|
||||
import http.client
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import re
|
||||
import runpy
|
||||
import secrets
|
||||
import signal
|
||||
import stat
|
||||
import subprocess
|
||||
import sys
|
||||
from urllib.parse import quote
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
if not sys.dont_write_bytecode:
|
||||
raise RuntimeError('container runtime could not disable bytecode writes')
|
||||
|
||||
|
||||
APP = Path('/opt/truf/app')
|
||||
DATA = Path('/data')
|
||||
RUN = Path('/run/truf')
|
||||
UID = GID = 10001
|
||||
DEFAULT_CONFIG = APP / 'config.linux.yaml'
|
||||
PROVISIONED = DATA / '.provisioned.json'
|
||||
INITIALIZED = DATA / 'initialized.json'
|
||||
INITIALIZE_LOCK = DATA / 'initialize.lock'
|
||||
PASSWORD = DATA / 'postgres-password'
|
||||
PROVIDER_SECRETS = DATA / 'config/secrets.yaml'
|
||||
FORMAT = 'truf-container-data-v1'
|
||||
WORKER_HEALTH_TOKEN = '0' * 64
|
||||
DIRECTORIES = (
|
||||
'home', 'config', 'managed-files', 'runtime-linux', 'runtime-linux/results',
|
||||
'runtime-linux/queues', 'runtime-linux/state',
|
||||
'runtime-linux/state/gharchive_cache', 'runtime-linux/logs',
|
||||
'runtime-linux/keychecks', 'runtime-linux/postman_cache',
|
||||
'runtime-linux/result_spool', 'runtime-linux/postgres',
|
||||
'runtime-linux/postgres/logs', 'postgres-linux', 'scanner-work',
|
||||
'scanner-result-bundles', 'scanner-result-bundles/tmp',
|
||||
'scanner-result-bundles/ready', 'scanner-result-bundles/quarantine',
|
||||
)
|
||||
_shutdown_requested = False
|
||||
|
||||
|
||||
def private_path(path, *, directory=False):
|
||||
path = Path(path)
|
||||
if not path.is_absolute() or '..' in path.parts:
|
||||
raise RuntimeError('private path must be absolute and normalized')
|
||||
for component in (*reversed(path.parents), path):
|
||||
if stat.S_ISLNK(component.lstat().st_mode):
|
||||
raise RuntimeError('private paths must not contain symlinks')
|
||||
details = path.lstat()
|
||||
expected_type = stat.S_ISDIR if directory else stat.S_ISREG
|
||||
if (not expected_type(details.st_mode) or details.st_uid != UID
|
||||
or details.st_gid != GID
|
||||
or stat.S_IMODE(details.st_mode) != (0o700 if directory else 0o600)):
|
||||
raise RuntimeError('private path ownership, type, or mode is invalid: ' + str(path))
|
||||
return path
|
||||
|
||||
|
||||
def require_container(*, provisioning=False):
|
||||
if (sys.platform != 'linux' or Path(__file__) != APP / 'container_runtime.py'
|
||||
or not Path('/.dockerenv').is_file()):
|
||||
raise RuntimeError('runtime commands are restricted to the prepared Docker image')
|
||||
if not (sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode):
|
||||
raise RuntimeError('container entrypoint requires python -I -S -B')
|
||||
if os.getuid() != os.geteuid() or os.geteuid() != (0 if provisioning else UID):
|
||||
raise RuntimeError('unexpected container runtime UID')
|
||||
if not provisioning and os.getgid() != GID:
|
||||
raise RuntimeError('unexpected container runtime GID')
|
||||
if not os.statvfs(APP).f_flag & os.ST_RDONLY:
|
||||
raise RuntimeError('the application image must be mounted read-only')
|
||||
private_path(APP.parent, directory=True)
|
||||
private_path(APP, directory=True)
|
||||
private_path(APP / 'container_runtime.py')
|
||||
with open('/proc/self/mountinfo', 'rb') as handle:
|
||||
payload = handle.read(1024 * 1024 + 1)
|
||||
if len(payload) > 1024 * 1024:
|
||||
raise RuntimeError('mount inventory exceeds its bound')
|
||||
mounts = {}
|
||||
for line in payload.splitlines():
|
||||
fields = line.split()
|
||||
if len(fields) > 6 and b'-' in fields:
|
||||
mounts[fields[4]] = fields[fields.index(b'-') + 1]
|
||||
if mounts.get(b'/data') not in (b'ext4', b'xfs', b'btrfs', b'zfs'):
|
||||
raise RuntimeError('/data must be an independent native Linux data volume')
|
||||
if mounts.get(b'/run/truf') != b'tmpfs':
|
||||
raise RuntimeError('/run/truf must be an independent private tmpfs')
|
||||
private_path(DATA, directory=True)
|
||||
private_path(RUN, directory=True)
|
||||
os.umask(0o077)
|
||||
|
||||
|
||||
def _read_json(path):
|
||||
path = private_path(path)
|
||||
if path.stat().st_size > 4096:
|
||||
raise RuntimeError('container marker exceeds its bound')
|
||||
value = json.loads(path.read_text(encoding='utf-8'))
|
||||
if not isinstance(value, dict) or value.get('format') != FORMAT:
|
||||
raise RuntimeError('unrecognized container data marker')
|
||||
return value
|
||||
|
||||
|
||||
def _write_new(path, payload, *, provisioning=False):
|
||||
try:
|
||||
private_path(path.parent, directory=True)
|
||||
descriptor = os.open(
|
||||
path,
|
||||
os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW,
|
||||
0o600,
|
||||
)
|
||||
with os.fdopen(descriptor, 'wb') as handle:
|
||||
if provisioning:
|
||||
os.fchown(handle.fileno(), UID, GID)
|
||||
handle.write(payload)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
descriptor = os.open(path.parent, os.O_RDONLY | os.O_DIRECTORY)
|
||||
try:
|
||||
os.fsync(descriptor)
|
||||
finally:
|
||||
os.close(descriptor)
|
||||
private_path(path)
|
||||
finally:
|
||||
payload = None
|
||||
|
||||
|
||||
def provision():
|
||||
import fcntl
|
||||
|
||||
lock = DATA / '.provision.lock'
|
||||
try:
|
||||
descriptor = os.open(lock, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600)
|
||||
os.fchown(descriptor, UID, GID)
|
||||
except FileExistsError:
|
||||
private_path(lock)
|
||||
descriptor = os.open(lock, os.O_WRONLY | os.O_NOFOLLOW)
|
||||
with os.fdopen(descriptor, 'wb') as handle:
|
||||
fcntl.flock(handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||
if PROVISIONED.exists():
|
||||
_read_json(PROVISIONED)
|
||||
managed_files = DATA / 'managed-files'
|
||||
try:
|
||||
os.mkdir(managed_files, 0o700)
|
||||
os.chown(managed_files, UID, GID)
|
||||
except FileExistsError:
|
||||
pass
|
||||
for name in DIRECTORIES:
|
||||
private_path(DATA / name, directory=True)
|
||||
private_path(PASSWORD)
|
||||
private_path(PROVIDER_SECRETS)
|
||||
print('Container data layout is already provisioned; nothing was changed.', flush=True)
|
||||
return
|
||||
if set(os.listdir(DATA)) - {'home', '.provision.lock'}:
|
||||
raise RuntimeError('refusing to provision nonempty or partially initialized data')
|
||||
home = DATA / 'home'
|
||||
if home.exists() and any(home.iterdir()):
|
||||
raise RuntimeError('refusing to provision a nonempty home directory')
|
||||
for name in DIRECTORIES:
|
||||
path = DATA / name
|
||||
if not path.exists():
|
||||
os.mkdir(path, 0o700)
|
||||
os.chown(path, UID, GID)
|
||||
private_path(path, directory=True)
|
||||
_write_new(PASSWORD, (secrets.token_urlsafe(48) + '\n').encode('ascii'), provisioning=True)
|
||||
_write_new(PROVIDER_SECRETS, b'{}\n', provisioning=True)
|
||||
_write_new(DATA / 'runtime-linux/proxy.txt', b'', provisioning=True)
|
||||
_write_new(PROVISIONED, (json.dumps({'format': FORMAT, 'uid': UID, 'gid': GID}) + '\n').encode('ascii'), provisioning=True)
|
||||
print('Fresh private container data layout provisioned; PostgreSQL is not initialized yet.', flush=True)
|
||||
|
||||
|
||||
def prepare_environment(
|
||||
config_path, *, validate_documents=True, return_config_sha256=False,
|
||||
):
|
||||
_read_json(PROVISIONED)
|
||||
for name in ('authority', 'control', 'tmp'):
|
||||
path = RUN / name
|
||||
try:
|
||||
path.mkdir(mode=0o700)
|
||||
except FileExistsError:
|
||||
pass
|
||||
private_path(path, directory=True)
|
||||
config_path = Path(config_path)
|
||||
if config_path != DEFAULT_CONFIG and config_path.parent != DATA / 'config':
|
||||
raise RuntimeError('configuration must be image-owned or in /data/config')
|
||||
private_path(config_path)
|
||||
password = private_path(PASSWORD).read_text(encoding='ascii').rstrip('\n')
|
||||
if re.fullmatch(r'[A-Za-z0-9_-]{32,128}', password) is None:
|
||||
raise RuntimeError('the generated PostgreSQL password is invalid')
|
||||
for name in tuple(os.environ):
|
||||
upper = name.upper()
|
||||
if (upper.startswith(('PG', 'TRUF_', 'SCANNER_', 'SCAN_', 'TRUFFLEHOG_', 'KEYCHECK_'))
|
||||
or upper in ('DATABASE_URL', 'PYTHONPATH', 'PYTHONHOME')):
|
||||
del os.environ[name]
|
||||
url = 'postgresql://truf:' + quote(password, safe='') + '@127.0.0.1:5432/truf'
|
||||
os.environ.update({
|
||||
'PATH': '/usr/local/bin:/usr/bin:/bin:/usr/lib/postgresql/16/bin',
|
||||
'HOME': '/data/home', 'TMPDIR': str(RUN / 'tmp'),
|
||||
'TMP': str(RUN / 'tmp'), 'TEMP': str(RUN / 'tmp'),
|
||||
'TRUF_CONTAINER_CONFIG': str(config_path),
|
||||
'TRUF_POSTGRES_DB': 'truf', 'TRUF_POSTGRES_USER': 'truf',
|
||||
'TRUF_POSTGRES_PORT': '5432', 'TRUF_POSTGRES_PASSWORD': password,
|
||||
'SCANNER_DB_URL': url, 'DATABASE_URL': url, 'TRUF_MANAGED_POSTGRES_DSN': url,
|
||||
'TRUF_DB_CONNECT_TIMEOUT_SEC': '3', 'TRUF_DB_STATEMENT_TIMEOUT_MS': '5000',
|
||||
'TRUF_DB_LOCK_TIMEOUT_MS': '2000', 'TRUF_DB_IDLE_TRANSACTION_TIMEOUT_MS': '10000',
|
||||
})
|
||||
bootstrap = runpy.run_path(str(APP / 'child_bootstrap.py'))
|
||||
bootstrap['_enable_dependency_paths']('supervisor')
|
||||
sys.path.insert(0, str(APP))
|
||||
from paths import apply_path_config
|
||||
from runtime_document_io import (
|
||||
load_managed_runtime_config,
|
||||
validate_managed_runtime_files,
|
||||
)
|
||||
from runtime_security import preflight_lifecycle_paths
|
||||
|
||||
def check_resolved(candidate):
|
||||
expected = {
|
||||
'root_dir': '/opt/truf', 'project_dir': str(APP),
|
||||
'runtime_dir': '/data/runtime-linux', 'postgres_data_dir': '/data/postgres-linux',
|
||||
'postgres_bin_dir': '/usr/lib/postgresql/16/bin',
|
||||
'result_bundle_dir': '/data/scanner-result-bundles', 'work_dir': '/data/scanner-work',
|
||||
'control_dir': str(RUN / 'control'), 'secrets_file': str(PROVIDER_SECRETS),
|
||||
}
|
||||
if any(candidate['global'].get(name) != value for name, value in expected.items()):
|
||||
raise RuntimeError('configuration escapes the fixed container storage contract')
|
||||
if candidate['supervisor'].get('control_dir') != str(RUN / 'control'):
|
||||
raise RuntimeError('supervisor control must remain on private ephemeral storage')
|
||||
preflight_lifecycle_paths(
|
||||
str(config_path), candidate, authority_profile='server',
|
||||
)
|
||||
|
||||
documents = (
|
||||
validate_managed_runtime_files(str(config_path))
|
||||
if validate_documents
|
||||
else load_managed_runtime_config(str(config_path))
|
||||
)
|
||||
config = documents.config
|
||||
config_sha256 = documents.config_sha256
|
||||
documents = None
|
||||
config = apply_path_config(config, str(config_path))
|
||||
check_resolved(config)
|
||||
if return_config_sha256:
|
||||
return config, config_sha256
|
||||
return config
|
||||
|
||||
|
||||
def _bootstrap_command(target, *arguments):
|
||||
return [sys.executable, '-u', '-I', '-S', '-B', str(APP / 'runtime_bootstrap.py'), target, '--', *arguments]
|
||||
|
||||
|
||||
def initialize(config_path, config, *, expected_config_sha256=None):
|
||||
from postgres_runtime import postgres_runtime_paths
|
||||
from runtime_document_io import load_managed_runtime_config
|
||||
from runtime_security import PrivateFileLock, read_private_json, write_private_json_exclusive
|
||||
|
||||
def require_stable_config():
|
||||
if expected_config_sha256 is None:
|
||||
return
|
||||
current = load_managed_runtime_config(str(config_path))
|
||||
current_sha256 = current.config_sha256
|
||||
current = None
|
||||
if current_sha256 != expected_config_sha256:
|
||||
raise RuntimeError('configuration changed after managed runtime validation')
|
||||
|
||||
require_stable_config()
|
||||
paths = postgres_runtime_paths(config)
|
||||
|
||||
def migrate(*, initialize_base):
|
||||
if _shutdown_requested:
|
||||
return False
|
||||
try:
|
||||
# Even an uncertain maintenance start must enter the verified stop path.
|
||||
require_stable_config()
|
||||
subprocess.run(_bootstrap_command(
|
||||
'postgres-runtime', 'maintenance-start', '--config', str(config_path),
|
||||
), check=True)
|
||||
if _shutdown_requested:
|
||||
return False
|
||||
require_stable_config()
|
||||
arguments = [
|
||||
'migrate-runtime-safety', '--config', str(config_path),
|
||||
]
|
||||
if initialize_base:
|
||||
arguments.append('--initialize-base')
|
||||
arguments.extend(('--apply', '--sources-stopped'))
|
||||
subprocess.run(_bootstrap_command(*arguments), check=True)
|
||||
return True
|
||||
finally:
|
||||
subprocess.run(_bootstrap_command(
|
||||
'postgres-runtime', 'maintenance-stop', '--config', str(config_path),
|
||||
), check=True)
|
||||
|
||||
with PrivateFileLock(str(INITIALIZE_LOCK)):
|
||||
if INITIALIZED.exists():
|
||||
marker = _read_json(INITIALIZED)
|
||||
identity = read_private_json(paths['identity_path'])
|
||||
if (not marker.get('system_identifier')
|
||||
or marker.get('system_identifier') != identity.get('system_identifier')
|
||||
or marker.get('pg_major') != 16 or identity.get('pg_major') != 16):
|
||||
raise RuntimeError('initialization marker does not match the bound cluster')
|
||||
migrated = migrate(initialize_base=False)
|
||||
if migrated:
|
||||
print('Existing PostgreSQL migrated and confirmed stopped.', flush=True)
|
||||
return
|
||||
if Path(paths['identity_path']).exists() or any(Path(paths['data_dir']).iterdir()):
|
||||
raise RuntimeError('partial initialization requires offline inspection; no automatic repair is allowed')
|
||||
if _shutdown_requested:
|
||||
return
|
||||
require_stable_config()
|
||||
subprocess.run(_bootstrap_command('postgres-runtime', 'initialize-empty', '--config', str(config_path)), check=True)
|
||||
if _shutdown_requested:
|
||||
return
|
||||
if not migrate(initialize_base=True):
|
||||
return
|
||||
identity = read_private_json(paths['identity_path'])
|
||||
require_stable_config()
|
||||
write_private_json_exclusive(str(INITIALIZED), {
|
||||
'format': FORMAT, 'system_identifier': identity['system_identifier'],
|
||||
'pg_major': identity['pg_major'],
|
||||
})
|
||||
print('Independent PostgreSQL initialized, migrated, cut over, and confirmed stopped.', flush=True)
|
||||
|
||||
|
||||
def _probe_worker_api(config):
|
||||
worker = config['supervisor']['worker_api']
|
||||
connection = None
|
||||
try:
|
||||
connection = http.client.HTTPConnection(
|
||||
worker['address'], int(worker['port']), timeout=2,
|
||||
)
|
||||
connection.request(
|
||||
'POST', '/api/v1/worker/claim', body=b'',
|
||||
headers={
|
||||
'Authorization': 'Bearer ' + WORKER_HEALTH_TOKEN,
|
||||
'Content-Length': '0',
|
||||
},
|
||||
)
|
||||
response = connection.getresponse()
|
||||
payload = response.read(4097)
|
||||
if (
|
||||
len(payload) > 4096
|
||||
or response.status != 401
|
||||
or response.getheader('WWW-Authenticate') != 'Bearer'
|
||||
or json.loads(payload) != {
|
||||
'error': {
|
||||
'code': 'unauthorized',
|
||||
'message': 'worker credentials are invalid',
|
||||
},
|
||||
}
|
||||
):
|
||||
raise RuntimeError('worker API is not ready')
|
||||
except Exception:
|
||||
raise RuntimeError('worker API is not ready') from None
|
||||
finally:
|
||||
if connection is not None:
|
||||
connection.close()
|
||||
|
||||
|
||||
def health(config, *, require_worker_api=False, require_discovery_producers=False):
|
||||
# Never create a competing SQL session while first-install migration is exclusive.
|
||||
_read_json(INITIALIZED)
|
||||
from scanner_db import ScannerDB
|
||||
from lifecycle_authority import DISCOVERY_PRODUCER_SOURCES
|
||||
from supervisor import get_control_snapshot
|
||||
from supervisor_instance import load_instance_metadata
|
||||
|
||||
metadata = load_instance_metadata(config['supervisor']['instance_file'])
|
||||
snapshot = get_control_snapshot(metadata)
|
||||
if (snapshot.get('activation_state') != 'ACTIVE'
|
||||
or snapshot.get('postgres', {}).get('state') != 'READY'
|
||||
or snapshot.get('postgres', {}).get('ready') is not True):
|
||||
raise RuntimeError('supervisor and PostgreSQL are not ready')
|
||||
signatures = {row[0]: row for row in snapshot.get('signature', ())}
|
||||
required = ['result-ingester', 'jsonl-projector']
|
||||
if config['supervisor'].get('janitor', {}).get('enabled', True):
|
||||
required.append('janitor')
|
||||
worker_api_enabled = config['supervisor'].get('worker_api', {}).get(
|
||||
'enabled', False,
|
||||
)
|
||||
if require_worker_api and not worker_api_enabled:
|
||||
raise RuntimeError('worker API is required but disabled')
|
||||
if worker_api_enabled:
|
||||
required.append('worker-api')
|
||||
for name in required:
|
||||
row = signatures.get(name)
|
||||
if not row or tuple(row[1:3]) != ('running', 'running') or not row[3] or row[7]:
|
||||
raise RuntimeError('a required pipeline worker is not running')
|
||||
if any(row[1] == 'failed' or row[8] for row in signatures.values()):
|
||||
raise RuntimeError('a managed source is failed or retains uncertain ownership')
|
||||
if require_discovery_producers:
|
||||
source_config = config.get('sources') or {}
|
||||
supervisor_sources = config['supervisor'].get('sources') or {}
|
||||
enabled_producers = [
|
||||
name for name in DISCOVERY_PRODUCER_SOURCES
|
||||
if (
|
||||
(supervisor_sources.get(name) or {}).get('enabled')
|
||||
if 'enabled' in (supervisor_sources.get(name) or {})
|
||||
else (source_config.get(name) or {}).get('enabled', False)
|
||||
)
|
||||
]
|
||||
for name in enabled_producers:
|
||||
row = signatures.get(name)
|
||||
ready = bool(
|
||||
row
|
||||
and row[2] == 'running'
|
||||
and not row[7]
|
||||
and not row[8]
|
||||
and (
|
||||
(row[1] == 'running' and row[3])
|
||||
or (row[1] == 'waiting' and row[4] == 0)
|
||||
)
|
||||
)
|
||||
if not ready:
|
||||
raise RuntimeError('a required discovery producer is not ready')
|
||||
if require_worker_api:
|
||||
_probe_worker_api(config)
|
||||
db = ScannerDB(db_url=os.environ['SCANNER_DB_URL'], initialize=False)
|
||||
try:
|
||||
if not db.enabled or not db.conn.is_postgres:
|
||||
raise RuntimeError('PostgreSQL application connection is unavailable')
|
||||
db.set_application_name('truf-container-health')
|
||||
db.conn.execute('SET default_transaction_read_only = on')
|
||||
db.conn.commit()
|
||||
db.require_runtime_safety_schema()
|
||||
db.require_final_cutover()
|
||||
for name in ('result_ingester', 'jsonl_projector'):
|
||||
if not db.pipeline_worker_health(name, metadata['instance_id'])['healthy']:
|
||||
raise RuntimeError('a required durable worker lease is not ready')
|
||||
row = db.conn.execute("SELECT current_setting('data_directory') AS data_directory, current_setting('server_version_num') AS version").fetchone()
|
||||
db.conn.commit()
|
||||
if row['data_directory'] != '/data/postgres-linux' or int(row['version']) // 10000 != 16:
|
||||
raise RuntimeError('PostgreSQL identity does not match the container')
|
||||
finally:
|
||||
db.close()
|
||||
for name in ('work_dir', 'result_bundle_dir', 'results_dir'):
|
||||
path = config['global'][name]
|
||||
if not os.access(path, os.W_OK | os.X_OK) or os.statvfs(path).f_bavail == 0:
|
||||
raise RuntimeError('required persistent storage is not writable or is full')
|
||||
return {'healthy': True, 'activation_state': 'ACTIVE', 'postgres': 'READY', 'workers': sorted(signatures)}
|
||||
|
||||
|
||||
def import_secrets(config_path, config, *, expected_config_sha256=None):
|
||||
from postgres_runtime import postgres_runtime_paths
|
||||
from runtime_document_io import (
|
||||
load_managed_runtime_config,
|
||||
validate_managed_runtime_files,
|
||||
)
|
||||
from runtime_security import ClusterAuthorityLock, PrivateFileLock, durable_replace, fsync_directory
|
||||
import yaml
|
||||
|
||||
with PrivateFileLock(str(INITIALIZE_LOCK)), ClusterAuthorityLock(
|
||||
config, create_parent=False, endpoint_dsn=os.environ['SCANNER_DB_URL'],
|
||||
):
|
||||
if (Path(config['supervisor']['instance_file']).exists()
|
||||
or (Path(postgres_runtime_paths(config)['data_dir']) / 'postmaster.pid').exists()):
|
||||
raise RuntimeError('stop the runtime before replacing provider credentials')
|
||||
payload = sys.stdin.buffer.read(1024 * 1024 + 1)
|
||||
try:
|
||||
if not payload or len(payload) > 1024 * 1024:
|
||||
raise RuntimeError('credential input must be a nonempty YAML document below 1 MiB')
|
||||
validated = validate_managed_runtime_files(
|
||||
str(config_path), secrets_bytes=payload,
|
||||
)
|
||||
except BaseException:
|
||||
payload = None
|
||||
raise
|
||||
payload = None
|
||||
validated_config_sha256 = validated.config_sha256
|
||||
if (
|
||||
expected_config_sha256 is not None
|
||||
and validated_config_sha256 != expected_config_sha256
|
||||
):
|
||||
validated = None
|
||||
raise RuntimeError('configuration changed while provider credentials were being validated')
|
||||
try:
|
||||
serialized = yaml.safe_dump(
|
||||
validated.secrets, allow_unicode=True,
|
||||
).encode('utf-8')
|
||||
except BaseException:
|
||||
validated = None
|
||||
raise
|
||||
validated = None
|
||||
temporary = DATA / ('config/secrets-import-' + secrets.token_hex(16))
|
||||
try:
|
||||
try:
|
||||
_write_new(temporary, serialized)
|
||||
finally:
|
||||
serialized = None
|
||||
private_path(PROVIDER_SECRETS)
|
||||
current = load_managed_runtime_config(str(config_path))
|
||||
current_sha256 = current.config_sha256
|
||||
current = None
|
||||
if current_sha256 != validated_config_sha256:
|
||||
raise RuntimeError(
|
||||
'configuration changed while provider credentials were being validated'
|
||||
)
|
||||
durable_replace(str(temporary), str(PROVIDER_SECRETS))
|
||||
fsync_directory(str(PROVIDER_SECRETS.parent))
|
||||
finally:
|
||||
temporary.unlink(missing_ok=True)
|
||||
print('Private provider credentials replaced; no credentials were printed.', flush=True)
|
||||
|
||||
|
||||
def main(argv=None):
|
||||
global _shutdown_requested
|
||||
|
||||
parser = argparse.ArgumentParser(description=__doc__, allow_abbrev=False)
|
||||
parser.add_argument('action', choices=('provision', 'initialize', 'run', 'health', 'status', 'import-secrets', 'import-snapshot'))
|
||||
parser.add_argument('--config', default=os.environ.get('TRUF_CONTAINER_CONFIG', str(DEFAULT_CONFIG)))
|
||||
parser.add_argument('--manifest-sha256')
|
||||
parser.add_argument('--require-worker-api', action='store_true')
|
||||
parser.add_argument('--require-discovery-producers', action='store_true')
|
||||
args = parser.parse_args(argv)
|
||||
require_container(provisioning=args.action == 'provision')
|
||||
if (
|
||||
(args.require_worker_api or args.require_discovery_producers)
|
||||
and args.action not in ('health', 'status')
|
||||
):
|
||||
raise RuntimeError('strict health requirements are restricted to health and status')
|
||||
if args.action == 'import-snapshot':
|
||||
if re.fullmatch(r'[0-9a-f]{64}', args.manifest_sha256 or '') is None:
|
||||
raise RuntimeError('snapshot import requires --manifest-sha256 with the approved manifest digest')
|
||||
if Path(args.config) != DEFAULT_CONFIG:
|
||||
raise RuntimeError('snapshot import must begin with the image-owned default configuration')
|
||||
elif args.manifest_sha256 is not None:
|
||||
raise RuntimeError('--manifest-sha256 is restricted to snapshot import')
|
||||
if args.action == 'provision':
|
||||
provision()
|
||||
return 0
|
||||
bind_config = args.action in ('initialize', 'run', 'import-secrets')
|
||||
prepared = prepare_environment(
|
||||
args.config,
|
||||
validate_documents=args.action != 'import-secrets',
|
||||
return_config_sha256=bind_config,
|
||||
)
|
||||
if bind_config:
|
||||
config, config_sha256 = prepared
|
||||
else:
|
||||
config = prepared
|
||||
if args.action in ('health', 'status'):
|
||||
print(json.dumps(health(
|
||||
config, require_worker_api=args.require_worker_api,
|
||||
require_discovery_producers=args.require_discovery_producers,
|
||||
), sort_keys=True), flush=True)
|
||||
return 0
|
||||
if args.action == 'import-secrets':
|
||||
import_secrets(
|
||||
args.config, config,
|
||||
expected_config_sha256=config_sha256,
|
||||
)
|
||||
return 0
|
||||
|
||||
def request_shutdown(_signum, _frame):
|
||||
global _shutdown_requested
|
||||
_shutdown_requested = True
|
||||
|
||||
previous = signal.signal(signal.SIGTERM, request_shutdown)
|
||||
try:
|
||||
if args.action == 'import-snapshot':
|
||||
from container_import import import_snapshot
|
||||
|
||||
return import_snapshot(sys.modules[__name__], args.manifest_sha256)
|
||||
initialize(
|
||||
Path(args.config), config,
|
||||
expected_config_sha256=config_sha256,
|
||||
)
|
||||
if _shutdown_requested or args.action == 'initialize':
|
||||
return 0
|
||||
from runtime_document_io import load_managed_runtime_config
|
||||
|
||||
current = load_managed_runtime_config(str(args.config))
|
||||
current_sha256 = current.config_sha256
|
||||
current = None
|
||||
if current_sha256 != config_sha256:
|
||||
raise RuntimeError('configuration changed after managed runtime validation')
|
||||
command = _bootstrap_command(
|
||||
'supervisor', '--runtime-bootstrap-entrypoint', str(APP / 'supervisor.py'),
|
||||
'--config', args.config, '--with-postgres', '--non-interactive',
|
||||
'--autostart', '--no-dashboard',
|
||||
)
|
||||
os.execv(sys.executable, command)
|
||||
finally:
|
||||
signal.signal(signal.SIGTERM, previous)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
raise SystemExit(main())
|
||||
except Exception as exc:
|
||||
print('Container runtime rejected: ' + str(exc), file=sys.stderr, flush=True)
|
||||
raise SystemExit(1) from None
|
||||
+2581
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,836 @@
|
||||
import os
|
||||
import re
|
||||
import sqlite3
|
||||
import time
|
||||
from urllib.parse import parse_qsl, quote, unquote, urlencode, urlsplit, urlunsplit
|
||||
|
||||
|
||||
POSTGRES_SCHEMES = ('postgresql://', 'postgres://')
|
||||
DEFAULT_POSTGRES_CONNECT_TIMEOUT_SEC = 10
|
||||
DEFAULT_POSTGRES_STATEMENT_TIMEOUT_MS = 30000
|
||||
DEFAULT_POSTGRES_LOCK_TIMEOUT_MS = 10000
|
||||
DEFAULT_POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS = 30000
|
||||
DEFAULT_POSTGRES_TCP_USER_TIMEOUT_MS = 30000
|
||||
POSTGRES_APPLICATION_SCHEMA = 'public'
|
||||
POSTGRES_CHILD_START_RETRY_ATTEMPTS = 3
|
||||
POSTGRES_CHILD_START_RETRY_MARKERS = (
|
||||
'server closed the connection unexpectedly',
|
||||
'connection reset by peer',
|
||||
'connection was forcibly closed by the remote host',
|
||||
)
|
||||
HOST_AGENT_POSTGRES_SOCKET_DIRECTORY = '/run/truf-postgres'
|
||||
HOST_AGENT_POSTGRES_DATABASE = 'truf'
|
||||
HOST_AGENT_POSTGRES_USER = 'truf'
|
||||
HOST_AGENT_POSTGRES_PORT = 5432
|
||||
|
||||
|
||||
class DatabaseUrlError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def is_postgres_url(value):
|
||||
return str(value or '').strip().lower().startswith(POSTGRES_SCHEMES)
|
||||
|
||||
|
||||
def database_url_from_env():
|
||||
# Supervised processes receive TRUF_MANAGED_POSTGRES_DSN as the canonical
|
||||
# connection authority. Use that same precedence everywhere that derives
|
||||
# an endpoint identity or opens a managed connection.
|
||||
return (
|
||||
os.getenv('TRUF_MANAGED_POSTGRES_DSN')
|
||||
or os.getenv('SCANNER_DB_URL')
|
||||
or os.getenv('DATABASE_URL')
|
||||
)
|
||||
|
||||
|
||||
def parse_postgres_url(value):
|
||||
"""Parse one URL-form libpq DSN without allowing alternate authorities."""
|
||||
text = str(value or '').strip()
|
||||
if not text or any(character in text for character in ('\x00', '\r', '\n')):
|
||||
raise DatabaseUrlError('invalid PostgreSQL database URL')
|
||||
try:
|
||||
parsed = urlsplit(text)
|
||||
port = parsed.port or 5432
|
||||
host = parsed.hostname or ''
|
||||
username = unquote(parsed.username or '')
|
||||
password = unquote(parsed.password or '')
|
||||
database = unquote((parsed.path or '')[1:]) if (parsed.path or '').startswith('/') else ''
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise DatabaseUrlError('invalid PostgreSQL database URL') from exc
|
||||
if parsed.scheme.lower() not in ('postgresql', 'postgres'):
|
||||
raise DatabaseUrlError('database URL must use the PostgreSQL scheme')
|
||||
if parsed.query:
|
||||
raise DatabaseUrlError('PostgreSQL database URL query parameters are forbidden')
|
||||
if parsed.fragment:
|
||||
raise DatabaseUrlError('PostgreSQL database URL fragments are forbidden')
|
||||
if not parsed.netloc or not host or not username or not database:
|
||||
raise DatabaseUrlError('PostgreSQL database URL must include one host, user, and database')
|
||||
authority = parsed.netloc.rsplit('@', 1)[-1]
|
||||
decoded_authority = unquote(authority)
|
||||
if (
|
||||
',' in decoded_authority
|
||||
or any(character.isspace() for character in decoded_authority)
|
||||
or '%' in authority
|
||||
or any(character in host for character in (',', '/', '\\', '\x00'))
|
||||
):
|
||||
raise DatabaseUrlError('PostgreSQL database URL must contain exactly one literal host authority')
|
||||
if not 0 < int(port) <= 65535:
|
||||
raise DatabaseUrlError('PostgreSQL database URL port is invalid')
|
||||
if any(character in database for character in ('/', '\\', '?', '#', '\x00')):
|
||||
raise DatabaseUrlError('PostgreSQL database URL database name contains encoded authority syntax')
|
||||
if parsed.path.count('/') != 1:
|
||||
raise DatabaseUrlError('PostgreSQL database URL must contain exactly one database path segment')
|
||||
return {
|
||||
'parsed': parsed,
|
||||
'host': host.lower(),
|
||||
'port': int(port),
|
||||
'database': database,
|
||||
'user': username,
|
||||
'password': password,
|
||||
}
|
||||
|
||||
|
||||
def canonical_postgres_url(value, database, user, port, host='127.0.0.1'):
|
||||
"""Return one libpq URL whose endpoint cannot be redirected by DSN options."""
|
||||
values = parse_postgres_url(value)
|
||||
expected_host = str(host).lower()
|
||||
if values['host'] != expected_host or values['port'] != int(port):
|
||||
raise DatabaseUrlError('PostgreSQL database URL endpoint does not match managed cluster authority')
|
||||
if values['database'] != str(database) or values['user'] != str(user):
|
||||
raise DatabaseUrlError('PostgreSQL database URL identity does not match managed cluster authority')
|
||||
credentials = quote(str(user), safe='')
|
||||
if values['parsed'].password is not None:
|
||||
credentials += ':' + quote(values['password'], safe='')
|
||||
netloc = f'{credentials}@{expected_host}:{int(port)}'
|
||||
return urlunsplit(('postgresql', netloc, '/' + quote(str(database), safe=''), '', ''))
|
||||
|
||||
|
||||
def redact_database_url(value):
|
||||
text = str(value or '')
|
||||
if not is_postgres_url(text):
|
||||
if text.strip().lower().startswith(('postgres', 'postgre')):
|
||||
return 'postgresql://***'
|
||||
if '://' in text:
|
||||
return text.split('://', 1)[0] + '://***'
|
||||
return text
|
||||
try:
|
||||
parsed = urlsplit(text)
|
||||
username = parsed.username or ''
|
||||
host = parsed.hostname or ''
|
||||
port = f':{parsed.port}' if parsed.port else ''
|
||||
netloc = parsed.netloc
|
||||
if parsed.password:
|
||||
netloc = f'{username}:***@{host}{port}' if username else f'***@{host}{port}'
|
||||
query = []
|
||||
for key, value in parse_qsl(parsed.query, keep_blank_values=True):
|
||||
key_lower = key.lower()
|
||||
sensitive = any(part in key_lower for part in ('password', 'passwd', 'pwd', 'token', 'secret', 'credential'))
|
||||
query.append((key, '***' if sensitive else value))
|
||||
return urlunsplit((parsed.scheme, netloc, parsed.path, urlencode(query), parsed.fragment))
|
||||
except Exception:
|
||||
return 'postgresql://***'
|
||||
|
||||
|
||||
def _split_sql_script(script):
|
||||
statements = []
|
||||
current = []
|
||||
quote = None
|
||||
escape = False
|
||||
for char in str(script or ''):
|
||||
current.append(char)
|
||||
if escape:
|
||||
escape = False
|
||||
continue
|
||||
if char == '\\':
|
||||
escape = True
|
||||
continue
|
||||
if quote:
|
||||
if char == quote:
|
||||
quote = None
|
||||
continue
|
||||
if char in ("'", '"'):
|
||||
quote = char
|
||||
continue
|
||||
if char == ';':
|
||||
statement = ''.join(current).strip()
|
||||
if statement:
|
||||
statements.append(statement[:-1].strip())
|
||||
current = []
|
||||
tail = ''.join(current).strip()
|
||||
if tail:
|
||||
statements.append(tail)
|
||||
return [statement for statement in statements if statement]
|
||||
|
||||
|
||||
def _convert_qmark_to_psycopg(sql):
|
||||
out = []
|
||||
quote = None
|
||||
escape = False
|
||||
for char in str(sql or ''):
|
||||
if escape:
|
||||
out.append(char)
|
||||
escape = False
|
||||
continue
|
||||
if char == '\\':
|
||||
out.append(char)
|
||||
escape = True
|
||||
continue
|
||||
if quote:
|
||||
out.append('%%' if char == '%' else char)
|
||||
if char == quote:
|
||||
quote = None
|
||||
continue
|
||||
if char in ("'", '"'):
|
||||
out.append(char)
|
||||
quote = char
|
||||
continue
|
||||
if char == '?':
|
||||
out.append('%s')
|
||||
elif char == '%':
|
||||
out.append('%%')
|
||||
else:
|
||||
out.append(char)
|
||||
return ''.join(out)
|
||||
|
||||
|
||||
def _postgres_schema_sql(script):
|
||||
converted = str(script or '').replace(
|
||||
'INTEGER PRIMARY KEY AUTOINCREMENT',
|
||||
'BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY',
|
||||
)
|
||||
# PostgreSQL cannot create either side of the target_queue/target_scans
|
||||
# cycle with both inline FKs. The offline migration adds this edge after
|
||||
# both tables exist; SQLite can retain it in the base schema.
|
||||
converted = converted.replace(
|
||||
',\n FOREIGN KEY(queue_id) REFERENCES target_queue(id)',
|
||||
'',
|
||||
)
|
||||
for future_foreign_key in (
|
||||
',\n FOREIGN KEY(result_reservation_id) REFERENCES result_reservations(id)',
|
||||
',\n FOREIGN KEY(reservation_id) REFERENCES result_reservations(id)',
|
||||
',\n FOREIGN KEY(current_result_reservation_id) REFERENCES result_reservations(id)',
|
||||
',\n FOREIGN KEY(candidate_id) REFERENCES keycheck_candidates(id)',
|
||||
',\n FOREIGN KEY(credential_id) REFERENCES keycheck_credentials(id)',
|
||||
',\n FOREIGN KEY(last_append_id) REFERENCES projection_appends(id)',
|
||||
',\n FOREIGN KEY(experiment_id) REFERENCES docker_depth_experiments(id)',
|
||||
):
|
||||
converted = converted.replace(future_foreign_key, '')
|
||||
for column in (
|
||||
'run_id', 'cycle_id', 'target_scan_id', 'finding_id', 'keycheck_result_id',
|
||||
'last_run_id', 'last_cycle_id', 'queue_id', 'byte_offset', 'line_number',
|
||||
'current_result_reservation_id', 'result_reservation_id', 'reservation_id',
|
||||
'projection_job_id', 'keycheck_candidate_id', 'candidate_id', 'credential_id',
|
||||
'keycheck_result_id', 'target_scan_id', 'job_id', 'last_append_id',
|
||||
'last_job_id', 'last_result_id', 'object_id', 'lease_reservation_id',
|
||||
'covered_reservation_id', 'declared_bytes', 'verified_bytes',
|
||||
'experiment_id', 'pass_id', 'page_id', 'source_cycle_id', 'retry_work_id',
|
||||
'repository_queue_id', 'first_cycle_id', 'last_cycle_id', 'first_page_id',
|
||||
'last_page_id', 'target_queue_id', 'manifest_id', 'manifest_layer_id',
|
||||
'eligibility_page_id', 'experiment_repository_id', 'experiment_target_id',
|
||||
'scan_binding_id', 'manifest_size_bytes', 'layer_size_bytes',
|
||||
'fence_generation', 'resolver_generation', 'dispatch_order', 'total_count',
|
||||
'user_id', 'remote_user_id', 'remote_device_id',
|
||||
'expected_revision', 'resulting_revision', 'revision',
|
||||
'before_bytes', 'after_bytes', 'previous_event_id',
|
||||
):
|
||||
converted = converted.replace(f'{column} INTEGER', f'{column} BIGINT')
|
||||
converted = converted.replace(
|
||||
'CREATE VIEW IF NOT EXISTS keycheck_latest_state AS',
|
||||
'CREATE OR REPLACE VIEW keycheck_latest_state AS',
|
||||
)
|
||||
return converted
|
||||
|
||||
|
||||
def _sqlite_check_constraints(sql):
|
||||
text = str(sql or '')
|
||||
constraints = {}
|
||||
index = 0
|
||||
position = 0
|
||||
quote = None
|
||||
while position < len(text):
|
||||
char = text[position]
|
||||
if quote:
|
||||
if char == quote:
|
||||
if position + 1 < len(text) and text[position + 1] == quote:
|
||||
position += 2
|
||||
continue
|
||||
quote = None
|
||||
position += 1
|
||||
continue
|
||||
if char in ("'", '"', '`'):
|
||||
quote = char
|
||||
position += 1
|
||||
continue
|
||||
if (
|
||||
text[position:position + 5].lower() != 'check'
|
||||
or (position and (text[position - 1].isalnum() or text[position - 1] == '_'))
|
||||
or (
|
||||
position + 5 < len(text)
|
||||
and (text[position + 5].isalnum() or text[position + 5] == '_')
|
||||
)
|
||||
):
|
||||
position += 1
|
||||
continue
|
||||
opening = position + 5
|
||||
while opening < len(text) and text[opening].isspace():
|
||||
opening += 1
|
||||
if opening >= len(text) or text[opening] != '(':
|
||||
position += 5
|
||||
continue
|
||||
depth = 1
|
||||
closing = opening + 1
|
||||
expression_quote = None
|
||||
while closing < len(text) and depth:
|
||||
current = text[closing]
|
||||
if expression_quote:
|
||||
if current == expression_quote:
|
||||
if closing + 1 < len(text) and text[closing + 1] == expression_quote:
|
||||
closing += 2
|
||||
continue
|
||||
expression_quote = None
|
||||
elif current in ("'", '"', '`'):
|
||||
expression_quote = current
|
||||
elif current == '(':
|
||||
depth += 1
|
||||
elif current == ')':
|
||||
depth -= 1
|
||||
closing += 1
|
||||
if depth:
|
||||
break
|
||||
prefix = text[:position]
|
||||
named = re.search(
|
||||
r'\bCONSTRAINT\s+(?:"([A-Za-z_][A-Za-z0-9_$]*)"|'
|
||||
r'([A-Za-z_][A-Za-z0-9_$]*))\s*$',
|
||||
prefix,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
name = (named.group(1) or named.group(2)) if named else f'__unnamed_check_{index}'
|
||||
expression = text[opening + 1:closing - 1]
|
||||
constraint = {
|
||||
'expression': expression,
|
||||
'definition': f'CHECK ({expression})',
|
||||
'valid': True,
|
||||
}
|
||||
if name in constraints:
|
||||
constraints[name]['valid'] = False
|
||||
name = f'__duplicate_check_{index}_{name}'
|
||||
constraint['valid'] = False
|
||||
constraints[name] = constraint
|
||||
index += 1
|
||||
position = closing
|
||||
return constraints
|
||||
|
||||
|
||||
class DatabaseConnection:
|
||||
def __init__(self, dialect, conn, application_schema=None):
|
||||
self.dialect = dialect
|
||||
self._conn = conn
|
||||
self.application_schema = application_schema if dialect == 'postgres' else None
|
||||
|
||||
@property
|
||||
def is_postgres(self):
|
||||
return self.dialect == 'postgres'
|
||||
|
||||
@property
|
||||
def is_sqlite(self):
|
||||
return self.dialect == 'sqlite'
|
||||
|
||||
def execute(self, sql, params=None):
|
||||
params = tuple(params or ())
|
||||
if self.is_postgres:
|
||||
params = tuple(value.replace('\x00', '') if isinstance(value, str) else value for value in params)
|
||||
cur = self._conn.cursor()
|
||||
cur.execute(_convert_qmark_to_psycopg(sql), params)
|
||||
return cur
|
||||
return self._conn.execute(sql, params)
|
||||
|
||||
def executescript(self, script):
|
||||
if self.is_sqlite:
|
||||
return self._conn.executescript(script)
|
||||
for statement in _split_sql_script(_postgres_schema_sql(script)):
|
||||
self.execute(statement)
|
||||
return None
|
||||
|
||||
def commit(self):
|
||||
return self._conn.commit()
|
||||
|
||||
def rollback(self):
|
||||
return self._conn.rollback()
|
||||
|
||||
def close(self):
|
||||
return self._conn.close()
|
||||
|
||||
def table_columns(self, table):
|
||||
if self.is_postgres:
|
||||
rows = self.execute(
|
||||
'''SELECT a.attname AS name
|
||||
FROM pg_catalog.pg_attribute a
|
||||
JOIN pg_catalog.pg_class c ON c.oid = a.attrelid
|
||||
JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
|
||||
WHERE n.nspname = ?
|
||||
AND c.relname = ?
|
||||
AND a.attnum > 0
|
||||
AND NOT a.attisdropped''',
|
||||
(self.application_schema, table),
|
||||
).fetchall()
|
||||
return {row['name'] for row in rows}
|
||||
return {row['name'] for row in self.execute(f'PRAGMA table_info({table})').fetchall()}
|
||||
|
||||
def table_column_details(self, table):
|
||||
if self.is_postgres:
|
||||
rows = self.execute(
|
||||
'''SELECT a.attname AS name,
|
||||
pg_catalog.format_type(a.atttypid, a.atttypmod) AS type,
|
||||
a.attnotnull AS not_null,
|
||||
a.attidentity AS identity_generation,
|
||||
a.attgenerated AS generated_kind,
|
||||
pg_catalog.pg_get_expr(d.adbin, d.adrelid) AS default_sql,
|
||||
(d.oid IS NOT NULL) AS has_default,
|
||||
pg_catalog.pg_get_serial_sequence(
|
||||
pg_catalog.quote_ident(n.nspname) || '.' || pg_catalog.quote_ident(c.relname),
|
||||
a.attname
|
||||
) AS sequence_name,
|
||||
COALESCE(i.indisprimary, false) AS primary_key
|
||||
FROM pg_catalog.pg_attribute a
|
||||
JOIN pg_catalog.pg_class c ON c.oid = a.attrelid
|
||||
JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
|
||||
LEFT JOIN pg_catalog.pg_attrdef d ON d.adrelid = a.attrelid AND d.adnum = a.attnum
|
||||
LEFT JOIN pg_catalog.pg_index i ON i.indrelid = a.attrelid
|
||||
AND i.indisprimary AND a.attnum = ANY(i.indkey)
|
||||
WHERE n.nspname = ?
|
||||
AND c.relname = ?
|
||||
AND a.attnum > 0
|
||||
AND NOT a.attisdropped
|
||||
ORDER BY a.attnum''',
|
||||
(self.application_schema, table),
|
||||
).fetchall()
|
||||
return {
|
||||
row['name']: {
|
||||
'type': str(row['type'] or '').lower(),
|
||||
'not_null': bool(row['not_null']),
|
||||
'default': str(row['default_sql'] or ''),
|
||||
'has_default': bool(row['has_default']),
|
||||
'primary_key': bool(row['primary_key']),
|
||||
'identity': str(row['identity_generation'] or ''),
|
||||
'generated': str(row['generated_kind'] or ''),
|
||||
'sequence': str(row['sequence_name'] or ''),
|
||||
}
|
||||
for row in rows
|
||||
}
|
||||
rows = self.execute(f'PRAGMA table_info({table})').fetchall()
|
||||
return {
|
||||
row['name']: {
|
||||
'type': str(row['type'] or '').lower(),
|
||||
'not_null': bool(row['notnull']) or bool(row['pk']),
|
||||
'default': str(row['dflt_value'] or ''),
|
||||
'has_default': row['dflt_value'] is not None,
|
||||
'primary_key': bool(row['pk']),
|
||||
'identity': '',
|
||||
'generated': '',
|
||||
'sequence': '',
|
||||
}
|
||||
for row in rows
|
||||
}
|
||||
|
||||
def table_indexes(self, table):
|
||||
if self.is_postgres:
|
||||
rows = self.execute(
|
||||
'''SELECT idx.relname AS name,
|
||||
i.indisunique AS is_unique,
|
||||
i.indisprimary AS is_primary,
|
||||
i.indisvalid AS is_valid,
|
||||
i.indisready AS is_ready,
|
||||
i.indislive AS is_live,
|
||||
pg_catalog.pg_get_expr(i.indpred, i.indrelid) AS predicate,
|
||||
ARRAY(
|
||||
SELECT pg_catalog.pg_get_indexdef(i.indexrelid, position, true)
|
||||
FROM pg_catalog.generate_series(1, i.indnkeyatts) AS position
|
||||
ORDER BY position
|
||||
) AS columns,
|
||||
pg_catalog.pg_get_indexdef(i.indexrelid) AS sql
|
||||
FROM pg_catalog.pg_index i
|
||||
JOIN pg_catalog.pg_class tbl ON tbl.oid = i.indrelid
|
||||
JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace
|
||||
JOIN pg_catalog.pg_class idx ON idx.oid = i.indexrelid
|
||||
WHERE n.nspname = ? AND tbl.relname = ?''',
|
||||
(self.application_schema, table),
|
||||
).fetchall()
|
||||
return {
|
||||
row['name']: {
|
||||
'unique': bool(row['is_unique']),
|
||||
'primary': bool(row['is_primary']),
|
||||
'valid': bool(row['is_valid']),
|
||||
'ready': bool(row['is_ready']),
|
||||
'live': bool(row['is_live']),
|
||||
'predicate': str(row['predicate'] or ''),
|
||||
'columns': [str(value).strip('"') for value in (row['columns'] or [])],
|
||||
'sql': str(row['sql'] or ''),
|
||||
}
|
||||
for row in rows
|
||||
}
|
||||
output = {}
|
||||
for row in self.execute(f'PRAGMA index_list({table})').fetchall():
|
||||
name = row['name']
|
||||
columns = [item['name'] for item in self.execute(f'PRAGMA index_info({name})').fetchall()]
|
||||
sql_row = self.execute(
|
||||
"SELECT sql FROM sqlite_master WHERE type = 'index' AND name = ?",
|
||||
(name,),
|
||||
).fetchone()
|
||||
sql = str(sql_row['sql'] if sql_row else '')
|
||||
predicate = sql.split(' WHERE ', 1)[1] if ' WHERE ' in sql.upper() else ''
|
||||
if ' WHERE ' in sql.upper():
|
||||
position = sql.upper().index(' WHERE ')
|
||||
predicate = sql[position + 7:]
|
||||
output[name] = {
|
||||
'unique': bool(row['unique']),
|
||||
'primary': str(row['origin'] or '') == 'pk',
|
||||
'valid': True,
|
||||
'ready': True,
|
||||
'live': True,
|
||||
'predicate': predicate,
|
||||
'columns': columns,
|
||||
'sql': sql,
|
||||
}
|
||||
return output
|
||||
|
||||
def table_foreign_keys(self, table):
|
||||
if self.is_postgres:
|
||||
rows = self.execute(
|
||||
'''SELECT con.conname AS name,
|
||||
ARRAY(
|
||||
SELECT src.attname
|
||||
FROM pg_catalog.unnest(con.conkey) WITH ORDINALITY AS keys(attnum, position)
|
||||
JOIN pg_catalog.pg_attribute src
|
||||
ON src.attrelid = con.conrelid AND src.attnum = keys.attnum
|
||||
ORDER BY keys.position
|
||||
) AS columns,
|
||||
ref_n.nspname AS referenced_schema,
|
||||
ref.relname AS referenced_table,
|
||||
ARRAY(
|
||||
SELECT dst.attname
|
||||
FROM pg_catalog.unnest(con.confkey) WITH ORDINALITY AS keys(attnum, position)
|
||||
JOIN pg_catalog.pg_attribute dst
|
||||
ON dst.attrelid = con.confrelid AND dst.attnum = keys.attnum
|
||||
ORDER BY keys.position
|
||||
) AS referenced_columns,
|
||||
con.confupdtype::text AS update_action,
|
||||
con.confdeltype::text AS delete_action,
|
||||
con.convalidated AS is_valid
|
||||
FROM pg_catalog.pg_constraint con
|
||||
JOIN pg_catalog.pg_class tbl ON tbl.oid = con.conrelid
|
||||
JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace
|
||||
JOIN pg_catalog.pg_class ref ON ref.oid = con.confrelid
|
||||
JOIN pg_catalog.pg_namespace ref_n ON ref_n.oid = ref.relnamespace
|
||||
WHERE con.contype = 'f' AND n.nspname = ? AND tbl.relname = ?''',
|
||||
(self.application_schema, table),
|
||||
).fetchall()
|
||||
action_names = {
|
||||
'a': 'NO ACTION', 'r': 'RESTRICT', 'c': 'CASCADE',
|
||||
'n': 'SET NULL', 'd': 'SET DEFAULT',
|
||||
}
|
||||
return {
|
||||
row['name']: {
|
||||
'columns': [str(value) for value in (row['columns'] or [])],
|
||||
'referenced_schema': str(row['referenced_schema'] or ''),
|
||||
'referenced_table': str(row['referenced_table'] or ''),
|
||||
'referenced_columns': [str(value) for value in (row['referenced_columns'] or [])],
|
||||
'update_action': action_names.get(str(row['update_action'] or ''), str(row['update_action'] or '')),
|
||||
'delete_action': action_names.get(str(row['delete_action'] or ''), str(row['delete_action'] or '')),
|
||||
'valid': bool(row['is_valid']),
|
||||
}
|
||||
for row in rows
|
||||
}
|
||||
|
||||
output = {}
|
||||
for row in self.execute(f'PRAGMA foreign_key_list({table})').fetchall():
|
||||
name = f'fk_{row["id"]}'
|
||||
current = output.setdefault(name, {
|
||||
'columns': [],
|
||||
'referenced_schema': 'main',
|
||||
'referenced_table': str(row['table'] or ''),
|
||||
'referenced_columns': [],
|
||||
'update_action': str(row['on_update'] or '').upper(),
|
||||
'delete_action': str(row['on_delete'] or '').upper(),
|
||||
'valid': True,
|
||||
})
|
||||
current['columns'].append(str(row['from'] or ''))
|
||||
current['referenced_columns'].append(str(row['to'] or ''))
|
||||
return output
|
||||
|
||||
def table_check_constraints(self, table):
|
||||
if self.is_postgres:
|
||||
rows = self.execute(
|
||||
'''SELECT con.conname AS name,
|
||||
pg_catalog.pg_get_expr(con.conbin, con.conrelid, true) AS expression,
|
||||
pg_catalog.pg_get_constraintdef(con.oid, true) AS definition,
|
||||
con.convalidated AS is_valid
|
||||
FROM pg_catalog.pg_constraint con
|
||||
JOIN pg_catalog.pg_class tbl ON tbl.oid = con.conrelid
|
||||
JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace
|
||||
WHERE con.contype = 'c' AND n.nspname = ? AND tbl.relname = ?''',
|
||||
(self.application_schema, table),
|
||||
).fetchall()
|
||||
return {
|
||||
str(row['name']): {
|
||||
'expression': str(row['expression'] or ''),
|
||||
'definition': str(row['definition'] or ''),
|
||||
'valid': bool(row['is_valid']),
|
||||
}
|
||||
for row in rows
|
||||
}
|
||||
row = self.execute(
|
||||
"SELECT sql FROM sqlite_master WHERE type = 'table' AND name = ?",
|
||||
(table,),
|
||||
).fetchone()
|
||||
return _sqlite_check_constraints(row['sql'] if row else '')
|
||||
|
||||
def table_triggers(self, table):
|
||||
if self.is_postgres:
|
||||
rows = self.execute(
|
||||
'''SELECT trg.tgname AS name,
|
||||
trg.tgenabled <> 'D' AS enabled,
|
||||
pg_catalog.pg_get_triggerdef(trg.oid, true) AS sql,
|
||||
pg_catalog.pg_get_functiondef(trg.tgfoid) AS function_sql
|
||||
FROM pg_catalog.pg_trigger trg
|
||||
JOIN pg_catalog.pg_class tbl ON tbl.oid = trg.tgrelid
|
||||
JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace
|
||||
WHERE n.nspname = ? AND tbl.relname = ?
|
||||
AND NOT trg.tgisinternal''',
|
||||
(self.application_schema, table),
|
||||
).fetchall()
|
||||
return {
|
||||
str(row['name']): {
|
||||
'enabled': bool(row['enabled']),
|
||||
'sql': str(row['sql'] or ''),
|
||||
'function_sql': str(row['function_sql'] or ''),
|
||||
}
|
||||
for row in rows
|
||||
}
|
||||
rows = self.execute(
|
||||
"SELECT name, sql FROM sqlite_master WHERE type = 'trigger' AND tbl_name = ?",
|
||||
(table,),
|
||||
).fetchall()
|
||||
return {
|
||||
str(row['name']): {
|
||||
'enabled': True,
|
||||
'sql': str(row['sql'] or ''),
|
||||
'function_sql': '',
|
||||
}
|
||||
for row in rows
|
||||
}
|
||||
|
||||
def table_exists(self, table):
|
||||
if self.is_postgres:
|
||||
row = self.execute(
|
||||
'''SELECT c.oid AS name FROM pg_catalog.pg_class c
|
||||
JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
|
||||
WHERE n.nspname = ? AND c.relname = ? AND c.relkind IN ('r', 'p')''',
|
||||
(self.application_schema, table),
|
||||
).fetchone()
|
||||
return bool(row and row['name'])
|
||||
row = self.execute("SELECT name FROM sqlite_master WHERE type = 'table' AND name = ?", (table,)).fetchone()
|
||||
return bool(row)
|
||||
|
||||
def insert_returning_id(self, sql, params=None):
|
||||
if self.is_postgres:
|
||||
cur = self.execute(f'{sql.rstrip()} RETURNING id', params)
|
||||
row = cur.fetchone()
|
||||
return row['id'] if row else None
|
||||
cur = self.execute(sql, params)
|
||||
return cur.lastrowid if int(getattr(cur, 'rowcount', 0) or 0) != 0 else None
|
||||
|
||||
def json_extract(self, column, path):
|
||||
if self.is_sqlite:
|
||||
return f"json_extract({column}, '{path}')"
|
||||
parts = str(path or '').lstrip('$.').split('.')
|
||||
pg_path = ','.join(part for part in parts if part)
|
||||
return f"(NULLIF({column}, '')::jsonb #>> '{{{pg_path}}}')"
|
||||
|
||||
|
||||
def connect_sqlite(path, timeout_sec=30, read_only=False, immutable=False, check_same_thread=True):
|
||||
if read_only:
|
||||
params = 'mode=ro&immutable=1' if immutable else 'mode=ro'
|
||||
uri = 'file:' + str(path).replace('\\', '/') + '?' + params
|
||||
conn = sqlite3.connect(uri, uri=True, timeout=max(1, int(timeout_sec or 30)), check_same_thread=check_same_thread)
|
||||
else:
|
||||
conn = sqlite3.connect(path, timeout=max(1, int(timeout_sec or 30)), check_same_thread=check_same_thread)
|
||||
conn.row_factory = sqlite3.Row
|
||||
return DatabaseConnection('sqlite', conn)
|
||||
|
||||
|
||||
def _bounded_int(value, default, minimum=1):
|
||||
try:
|
||||
return max(minimum, int(value))
|
||||
except (TypeError, ValueError):
|
||||
return max(minimum, int(default))
|
||||
|
||||
|
||||
def connect_postgres(
|
||||
url,
|
||||
connect_timeout_sec=None,
|
||||
statement_timeout_ms=None,
|
||||
lock_timeout_ms=None,
|
||||
idle_in_transaction_timeout_ms=None,
|
||||
tcp_user_timeout_ms=None,
|
||||
):
|
||||
parse_postgres_url(url)
|
||||
try:
|
||||
import psycopg
|
||||
from psycopg.rows import dict_row
|
||||
except ImportError as exc:
|
||||
raise RuntimeError('PostgreSQL backend requires psycopg[binary]. Install app requirements first.') from exc
|
||||
connect_timeout_sec = _bounded_int(
|
||||
connect_timeout_sec if connect_timeout_sec is not None else os.getenv('TRUF_DB_CONNECT_TIMEOUT_SEC'),
|
||||
DEFAULT_POSTGRES_CONNECT_TIMEOUT_SEC,
|
||||
)
|
||||
statement_timeout_ms = _bounded_int(
|
||||
statement_timeout_ms if statement_timeout_ms is not None else os.getenv('TRUF_DB_STATEMENT_TIMEOUT_MS'),
|
||||
DEFAULT_POSTGRES_STATEMENT_TIMEOUT_MS,
|
||||
)
|
||||
lock_timeout_ms = _bounded_int(
|
||||
lock_timeout_ms if lock_timeout_ms is not None else os.getenv('TRUF_DB_LOCK_TIMEOUT_MS'),
|
||||
DEFAULT_POSTGRES_LOCK_TIMEOUT_MS,
|
||||
)
|
||||
idle_in_transaction_timeout_ms = _bounded_int(
|
||||
idle_in_transaction_timeout_ms if idle_in_transaction_timeout_ms is not None else os.getenv('TRUF_DB_IDLE_TRANSACTION_TIMEOUT_MS'),
|
||||
DEFAULT_POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS,
|
||||
)
|
||||
tcp_user_timeout_ms = _bounded_int(
|
||||
tcp_user_timeout_ms if tcp_user_timeout_ms is not None else os.getenv('TRUF_DB_TCP_USER_TIMEOUT_MS'),
|
||||
DEFAULT_POSTGRES_TCP_USER_TIMEOUT_MS,
|
||||
minimum=1000,
|
||||
)
|
||||
options = ' '.join((
|
||||
f'-c search_path={POSTGRES_APPLICATION_SCHEMA}',
|
||||
f'-c statement_timeout={statement_timeout_ms}',
|
||||
f'-c lock_timeout={lock_timeout_ms}',
|
||||
f'-c idle_in_transaction_session_timeout={idle_in_transaction_timeout_ms}',
|
||||
))
|
||||
for attempt in range(POSTGRES_CHILD_START_RETRY_ATTEMPTS):
|
||||
try:
|
||||
conn = psycopg.connect(
|
||||
url,
|
||||
row_factory=dict_row,
|
||||
connect_timeout=connect_timeout_sec,
|
||||
options=options,
|
||||
tcp_user_timeout=tcp_user_timeout_ms,
|
||||
keepalives=1,
|
||||
keepalives_idle=5,
|
||||
keepalives_interval=5,
|
||||
keepalives_count=2,
|
||||
)
|
||||
break
|
||||
except Exception as exc:
|
||||
transient_child_start = any(
|
||||
marker in str(exc).lower() for marker in POSTGRES_CHILD_START_RETRY_MARKERS
|
||||
)
|
||||
if not transient_child_start or attempt + 1 >= POSTGRES_CHILD_START_RETRY_ATTEMPTS:
|
||||
raise
|
||||
time.sleep(0.05 * (attempt + 1))
|
||||
try:
|
||||
cursor = conn.cursor()
|
||||
cursor.execute(
|
||||
"""SELECT pg_catalog.current_schema() AS schema_name,
|
||||
pg_catalog.current_setting('search_path') AS search_path,
|
||||
EXISTS (
|
||||
SELECT 1
|
||||
FROM pg_catalog.pg_namespace n
|
||||
CROSS JOIN LATERAL pg_catalog.aclexplode(
|
||||
COALESCE(n.nspacl, pg_catalog.acldefault('n', n.nspowner))
|
||||
) acl
|
||||
WHERE n.nspname = 'public'
|
||||
AND acl.grantee = 0
|
||||
AND acl.privilege_type = 'CREATE'
|
||||
) AS public_create"""
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
cursor.close()
|
||||
schema_name = row.get('schema_name') if isinstance(row, dict) else row[0] if row else None
|
||||
search_path = row.get('search_path') if isinstance(row, dict) else row[1] if row else None
|
||||
public_create = row.get('public_create', False) if isinstance(row, dict) else row[2] if row and len(row) > 2 else False
|
||||
normalized_path = re.sub(r'[\s\"]', '', str(search_path or '').lower())
|
||||
if schema_name != POSTGRES_APPLICATION_SCHEMA or normalized_path != 'public' or bool(public_create):
|
||||
raise RuntimeError('PostgreSQL application schema/search_path validation failed')
|
||||
conn.rollback()
|
||||
except Exception:
|
||||
try:
|
||||
conn.close()
|
||||
except Exception:
|
||||
pass
|
||||
raise
|
||||
return DatabaseConnection('postgres', conn, application_schema=POSTGRES_APPLICATION_SCHEMA)
|
||||
|
||||
|
||||
def connect_host_agent_postgres():
|
||||
"""Open the one fixed peer-authenticated host-agent authority."""
|
||||
if os.name != 'posix' or not hasattr(os, 'geteuid') or os.geteuid() != 0:
|
||||
raise RuntimeError('PostgreSQL host-agent authority requires root on POSIX')
|
||||
try:
|
||||
import psycopg
|
||||
from psycopg.rows import dict_row
|
||||
except ImportError as exc:
|
||||
raise RuntimeError(
|
||||
'PostgreSQL backend requires psycopg[binary]. Install app requirements first.'
|
||||
) from exc
|
||||
options = ' '.join((
|
||||
f'-c search_path={POSTGRES_APPLICATION_SCHEMA}',
|
||||
f'-c statement_timeout={DEFAULT_POSTGRES_STATEMENT_TIMEOUT_MS}',
|
||||
f'-c lock_timeout={DEFAULT_POSTGRES_LOCK_TIMEOUT_MS}',
|
||||
f'-c idle_in_transaction_session_timeout={DEFAULT_POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS}',
|
||||
))
|
||||
conn = psycopg.connect(
|
||||
dbname=HOST_AGENT_POSTGRES_DATABASE,
|
||||
user=HOST_AGENT_POSTGRES_USER,
|
||||
host=HOST_AGENT_POSTGRES_SOCKET_DIRECTORY,
|
||||
port=HOST_AGENT_POSTGRES_PORT,
|
||||
row_factory=dict_row,
|
||||
connect_timeout=DEFAULT_POSTGRES_CONNECT_TIMEOUT_SEC,
|
||||
options=options,
|
||||
sslmode='disable',
|
||||
)
|
||||
try:
|
||||
cursor = conn.cursor()
|
||||
cursor.execute(
|
||||
"""SELECT pg_catalog.current_schema() AS schema_name,
|
||||
pg_catalog.current_setting('search_path') AS search_path,
|
||||
CURRENT_USER AS current_user,
|
||||
current_database() AS database_name,
|
||||
pg_catalog.inet_server_addr() IS NULL AS unix_socket,
|
||||
pg_catalog.current_setting('port')::integer AS port,
|
||||
EXISTS (
|
||||
SELECT 1
|
||||
FROM pg_catalog.pg_namespace n
|
||||
CROSS JOIN LATERAL pg_catalog.aclexplode(
|
||||
COALESCE(n.nspacl, pg_catalog.acldefault('n', n.nspowner))
|
||||
) acl
|
||||
WHERE n.nspname = 'public'
|
||||
AND acl.grantee = 0
|
||||
AND acl.privilege_type = 'CREATE'
|
||||
) AS public_create"""
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
cursor.close()
|
||||
normalized_path = re.sub(
|
||||
r'[\s\"]', '', str((row or {}).get('search_path') or '').lower()
|
||||
)
|
||||
if (
|
||||
not isinstance(row, dict)
|
||||
or row.get('schema_name') != POSTGRES_APPLICATION_SCHEMA
|
||||
or normalized_path != POSTGRES_APPLICATION_SCHEMA
|
||||
or row.get('current_user') != HOST_AGENT_POSTGRES_USER
|
||||
or row.get('database_name') != HOST_AGENT_POSTGRES_DATABASE
|
||||
or row.get('unix_socket') is not True
|
||||
or row.get('port') != HOST_AGENT_POSTGRES_PORT
|
||||
or bool(row.get('public_create'))
|
||||
):
|
||||
raise RuntimeError('PostgreSQL host-agent authority validation failed')
|
||||
conn.rollback()
|
||||
except Exception:
|
||||
try:
|
||||
conn.close()
|
||||
except Exception:
|
||||
pass
|
||||
raise
|
||||
return DatabaseConnection(
|
||||
'postgres', conn, application_schema=POSTGRES_APPLICATION_SCHEMA,
|
||||
)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,546 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import contextlib
|
||||
import hmac
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
|
||||
from db_backend import canonical_postgres_url, is_postgres_url
|
||||
from docker_depth_experiment import (
|
||||
DOCKER_DEPTH_QUERY_COUNT,
|
||||
_docker_depth_authority,
|
||||
_experiment_identity_matches,
|
||||
_stored_cohort_plan,
|
||||
apply_docker_depth_cohort_manifest,
|
||||
apply_docker_depth_hold_manifest,
|
||||
apply_docker_depth_reactivation_manifest,
|
||||
apply_docker_depth_resolver_disposition_manifest,
|
||||
apply_docker_depth_resolver_refund_manifest,
|
||||
canonical_docker_depth_plan_hash,
|
||||
generate_docker_depth_cohort_manifest,
|
||||
generate_docker_depth_hold_manifest,
|
||||
generate_docker_depth_reactivation_manifest,
|
||||
generate_docker_depth_resolver_disposition_manifest,
|
||||
generate_docker_depth_resolver_refund_manifest,
|
||||
summarize_docker_depth_fresh_coverage,
|
||||
validate_docker_depth_cohort_manifest,
|
||||
validate_docker_depth_config,
|
||||
validate_docker_depth_hold_manifest,
|
||||
validate_docker_depth_reactivation_manifest,
|
||||
validate_docker_depth_resolver_disposition_manifest,
|
||||
validate_docker_depth_resolver_refund_manifest,
|
||||
validate_dockerhub_discovery_policies,
|
||||
)
|
||||
from migrate_runtime_safety import (
|
||||
postgres_migration_guard,
|
||||
require_local_sources_stopped,
|
||||
)
|
||||
from paths import apply_path_config
|
||||
from postgres_runtime import (
|
||||
canonical_database_url,
|
||||
load_postgres_environment,
|
||||
verify_cluster_identity,
|
||||
)
|
||||
from runtime_security import (
|
||||
ClusterAuthorityLock,
|
||||
MAX_EXTENDED_PRIVATE_JSON_BYTES,
|
||||
read_private_json,
|
||||
reject_reparse_components,
|
||||
require_private_directory,
|
||||
require_private_file,
|
||||
preflight_lifecycle_paths,
|
||||
write_private_json_exclusive,
|
||||
)
|
||||
from scanner_db import ScannerDB
|
||||
|
||||
|
||||
APPLICATION_NAME = 'truf-docker-depth-operator'
|
||||
MANIFEST_MAX_BYTES = MAX_EXTENDED_PRIVATE_JSON_BYTES
|
||||
_SHA256_RE = re.compile(r'^[a-f0-9]{64}$')
|
||||
_DISABLED_ACTIONS = frozenset({
|
||||
'generate-cohort', 'apply-cohort', 'generate-hold', 'apply-hold',
|
||||
})
|
||||
_ENABLED_ACTIONS = frozenset({
|
||||
'generate-reactivation', 'apply-reactivation',
|
||||
'generate-resolver-disposition', 'apply-resolver-disposition',
|
||||
'generate-resolver-refund', 'apply-resolver-refund',
|
||||
})
|
||||
|
||||
|
||||
def load_config(path):
|
||||
"""Load path-expanded YAML only; validation and runtime access are separate."""
|
||||
try:
|
||||
import yaml
|
||||
except ImportError as exc:
|
||||
raise RuntimeError('PyYAML is required') from exc
|
||||
with open(path, 'r', encoding='utf-8') as handle:
|
||||
return apply_path_config(yaml.safe_load(handle) or {}, path)
|
||||
|
||||
|
||||
def provenance_policy_sha256(validated):
|
||||
source = validated.normalized_config['sources']['dockerhub']
|
||||
policies = validate_dockerhub_discovery_policies(
|
||||
source, validated.experiment.queries,
|
||||
)
|
||||
hashes = {policy['policy_sha256'] for policy in policies}
|
||||
if len(policies) != DOCKER_DEPTH_QUERY_COUNT or len(hashes) != 1:
|
||||
raise RuntimeError('Docker depth provenance policy authority is ambiguous')
|
||||
return next(iter(hashes))
|
||||
|
||||
|
||||
def _action_name(args):
|
||||
for attribute, name in (
|
||||
('status', 'status'),
|
||||
('generate_cohort_manifest', 'generate-cohort'),
|
||||
('apply_cohort_manifest', 'apply-cohort'),
|
||||
('generate_hold_manifest', 'generate-hold'),
|
||||
('apply_hold_manifest', 'apply-hold'),
|
||||
('generate_reactivation_manifest', 'generate-reactivation'),
|
||||
('apply_reactivation_manifest', 'apply-reactivation'),
|
||||
('generate_resolver_disposition_manifest', 'generate-resolver-disposition'),
|
||||
('apply_resolver_disposition_manifest', 'apply-resolver-disposition'),
|
||||
('generate_resolver_refund_manifest', 'generate-resolver-refund'),
|
||||
('apply_resolver_refund_manifest', 'apply-resolver-refund'),
|
||||
):
|
||||
if getattr(args, attribute, None):
|
||||
return name
|
||||
raise RuntimeError('Docker depth operator action is unavailable')
|
||||
|
||||
|
||||
def _require_action_arguments(parser, args, action):
|
||||
applying = action.startswith('apply-')
|
||||
supplied_apply_option = bool(
|
||||
args.confirm_apply or args.sources_stopped or args.approve_sha256
|
||||
)
|
||||
if applying:
|
||||
if not args.confirm_apply or not args.sources_stopped or not args.approve_sha256:
|
||||
parser.error(
|
||||
'apply actions require --approve-sha256, --apply, and --sources-stopped'
|
||||
)
|
||||
if not _SHA256_RE.fullmatch(args.approve_sha256):
|
||||
parser.error('--approve-sha256 must be one lowercase SHA-256 value')
|
||||
elif supplied_apply_option:
|
||||
parser.error('approval options are valid only for apply actions')
|
||||
|
||||
|
||||
def _require_action_config_state(experiment, action):
|
||||
if action in _DISABLED_ACTIONS and experiment.enabled:
|
||||
raise RuntimeError('Docker depth reviewed preparation requires disabled config')
|
||||
if action in _ENABLED_ACTIONS and not experiment.enabled:
|
||||
raise RuntimeError('Docker depth reviewed release requires enabled config')
|
||||
|
||||
|
||||
def _manifest_path(path, *, existing):
|
||||
absolute = reject_reparse_components(os.path.abspath(os.fspath(path)))
|
||||
require_private_directory(os.path.dirname(absolute), create=False)
|
||||
if existing:
|
||||
require_private_file(absolute)
|
||||
elif os.path.lexists(absolute):
|
||||
require_private_file(absolute)
|
||||
return absolute
|
||||
|
||||
|
||||
def _publish_manifest(path, manifest):
|
||||
absolute = _manifest_path(path, existing=False)
|
||||
if os.path.lexists(absolute):
|
||||
if read_private_json(absolute, max_bytes=MANIFEST_MAX_BYTES) != manifest:
|
||||
raise RuntimeError('A different reviewed manifest already exists')
|
||||
return absolute, False
|
||||
try:
|
||||
write_private_json_exclusive(
|
||||
absolute, manifest, max_bytes=MANIFEST_MAX_BYTES,
|
||||
)
|
||||
except FileExistsError:
|
||||
if read_private_json(absolute, max_bytes=MANIFEST_MAX_BYTES) != manifest:
|
||||
raise RuntimeError('Reviewed manifest publication raced a different file')
|
||||
return absolute, False
|
||||
require_private_file(absolute)
|
||||
return absolute, True
|
||||
|
||||
|
||||
def _read_approved_manifest(path, validator, experiment, policy_sha256, approved):
|
||||
absolute = _manifest_path(path, existing=True)
|
||||
manifest = read_private_json(absolute, max_bytes=MANIFEST_MAX_BYTES)
|
||||
normalized, manifest_sha256 = validator(
|
||||
manifest, experiment, policy_sha256,
|
||||
)
|
||||
if not hmac.compare_digest(manifest_sha256, approved):
|
||||
raise ValueError('Reviewed manifest approval hash conflicts')
|
||||
return absolute, normalized, manifest_sha256
|
||||
|
||||
|
||||
def _prepare_action(args, action, experiment, policy_sha256):
|
||||
if action.startswith('generate-'):
|
||||
attribute = action.replace('-', '_') + '_manifest'
|
||||
path = _manifest_path(getattr(args, attribute), existing=False)
|
||||
return {'path': path}
|
||||
if not action.startswith('apply-'):
|
||||
return {}
|
||||
kind = action.removeprefix('apply-')
|
||||
validator = {
|
||||
'cohort': validate_docker_depth_cohort_manifest,
|
||||
'hold': validate_docker_depth_hold_manifest,
|
||||
'reactivation': validate_docker_depth_reactivation_manifest,
|
||||
'resolver-disposition': validate_docker_depth_resolver_disposition_manifest,
|
||||
'resolver-refund': validate_docker_depth_resolver_refund_manifest,
|
||||
}[kind]
|
||||
path = getattr(args, f'apply_{kind.replace("-", "_")}_manifest')
|
||||
absolute, manifest, manifest_sha256 = _read_approved_manifest(
|
||||
path, validator, experiment, policy_sha256, args.approve_sha256,
|
||||
)
|
||||
return {
|
||||
'path': absolute,
|
||||
'manifest': manifest,
|
||||
'manifest_sha256': manifest_sha256,
|
||||
}
|
||||
|
||||
|
||||
def _verify_online_cluster_identity(db, dsn, identity):
|
||||
canonical = canonical_postgres_url(
|
||||
dsn, identity['database'], identity['user'], identity['port'],
|
||||
)
|
||||
if canonical != dsn:
|
||||
raise RuntimeError('Managed PostgreSQL DSN is not canonical')
|
||||
row = db.conn.execute(
|
||||
'''SELECT pg_catalog.current_database() AS database,
|
||||
CURRENT_USER AS user_name,
|
||||
pg_catalog.current_setting('data_directory') AS data_directory,
|
||||
pg_catalog.current_setting('port')::integer AS port,
|
||||
(SELECT system_identifier::text
|
||||
FROM pg_catalog.pg_control_system()) AS system_identifier'''
|
||||
).fetchone()
|
||||
checks = {
|
||||
'database': (str(row['database']), str(identity['database'])),
|
||||
'user': (str(row['user_name']), str(identity['user'])),
|
||||
'data_directory': (
|
||||
os.path.normcase(os.path.realpath(os.path.abspath(row['data_directory']))),
|
||||
os.path.normcase(os.path.realpath(os.path.abspath(identity['data_directory']))),
|
||||
),
|
||||
'port': (int(row['port']), int(identity['port'])),
|
||||
'system_identifier': (
|
||||
str(row['system_identifier']), str(identity['system_identifier']),
|
||||
),
|
||||
}
|
||||
if any(actual != expected for actual, expected in checks.values()):
|
||||
raise RuntimeError('Online PostgreSQL identity does not match private authority')
|
||||
db.conn.commit()
|
||||
|
||||
|
||||
@contextlib.contextmanager
|
||||
def operator_database(config_path, config, *, read_only):
|
||||
preflight_lifecycle_paths(config_path, config)
|
||||
load_postgres_environment(config_path, config)
|
||||
dsn = canonical_database_url()
|
||||
if not dsn or not is_postgres_url(dsn):
|
||||
raise RuntimeError('Canonical managed PostgreSQL DSN is unavailable')
|
||||
with ClusterAuthorityLock(config, endpoint_dsn=dsn):
|
||||
require_local_sources_stopped(config)
|
||||
identity = verify_cluster_identity(config)
|
||||
dsn = canonical_postgres_url(
|
||||
dsn, identity['database'], identity['user'], identity['port'],
|
||||
)
|
||||
db = ScannerDB(db_url=dsn, initialize=False)
|
||||
try:
|
||||
if not db.enabled:
|
||||
raise RuntimeError('Managed PostgreSQL connection is unavailable')
|
||||
_verify_online_cluster_identity(db, dsn, identity)
|
||||
db.set_application_name(APPLICATION_NAME)
|
||||
with postgres_migration_guard(db):
|
||||
db.require_runtime_safety_schema()
|
||||
db.require_final_cutover()
|
||||
if read_only:
|
||||
db.conn.execute('SET default_transaction_read_only = on')
|
||||
db.conn.commit()
|
||||
yield db
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def _status(db, experiment, policy_sha256):
|
||||
authority = _docker_depth_authority(experiment, policy_sha256)
|
||||
coverage = summarize_docker_depth_fresh_coverage(
|
||||
db, experiment, policy_sha256,
|
||||
)
|
||||
row = db.conn.execute(
|
||||
'SELECT * FROM docker_depth_experiments WHERE experiment_key = ?',
|
||||
(authority['experiment_key'],),
|
||||
).fetchone()
|
||||
state = 'absent'
|
||||
plan_sha256 = ''
|
||||
hold_manifest_sha256 = ''
|
||||
fence_active = 0
|
||||
counts = {
|
||||
'owned_policy_events': 0,
|
||||
'planned_queries': 0,
|
||||
'planned_repositories': 0,
|
||||
'targets': 0,
|
||||
'unreleased_holds': 0,
|
||||
}
|
||||
if row:
|
||||
if not _experiment_identity_matches(row, authority):
|
||||
raise RuntimeError('Docker depth persisted authority drifted')
|
||||
state = str(row['state'])
|
||||
plan_sha256 = str(row['plan_sha256'] or '')
|
||||
hold_manifest_sha256 = str(row['hold_manifest_sha256'] or '')
|
||||
fence_active = int(any(
|
||||
row[name] is not None
|
||||
for name in ('fence_owner', 'fence_token', 'fence_expires_at')
|
||||
))
|
||||
if plan_sha256:
|
||||
stored = _stored_cohort_plan(db.conn, row, authority)
|
||||
if canonical_docker_depth_plan_hash(stored) != plan_sha256:
|
||||
raise RuntimeError('Docker depth persisted cohort hash drifted')
|
||||
count_row = db.conn.execute(
|
||||
'''SELECT
|
||||
(SELECT COUNT(*) FROM docker_depth_experiment_queries
|
||||
WHERE experiment_id = ?) AS planned_queries,
|
||||
(SELECT COUNT(*) FROM docker_depth_experiment_repositories
|
||||
WHERE experiment_id = ?) AS planned_repositories,
|
||||
(SELECT COUNT(*) FROM docker_depth_experiment_targets
|
||||
WHERE experiment_id = ?) AS targets,
|
||||
(SELECT COUNT(*) FROM target_queue_policy_events
|
||||
WHERE experiment_id = ?) AS owned_policy_events,
|
||||
(SELECT COUNT(*)
|
||||
FROM target_queue_policy_events cold_event
|
||||
LEFT JOIN target_queue_policy_events reverse_event
|
||||
ON reverse_event.reverses_event_id = cold_event.id
|
||||
WHERE cold_event.experiment_id = ?
|
||||
AND cold_event.action = 'cold'
|
||||
AND reverse_event.id IS NULL) AS unreleased_holds''',
|
||||
(row['id'], row['id'], row['id'], row['id'], row['id']),
|
||||
).fetchone()
|
||||
counts = {name: int(count_row[name]) for name in counts}
|
||||
db.conn.commit()
|
||||
return {
|
||||
'action': 'status',
|
||||
'config_enabled': bool(experiment.enabled),
|
||||
'config_sha256': authority['config_sha256'],
|
||||
'counts': counts,
|
||||
'experiment_present': bool(row),
|
||||
'fence_active': fence_active,
|
||||
'fresh_coverage': coverage,
|
||||
'hold_manifest_sha256': hold_manifest_sha256,
|
||||
'ordered_queries_sha256': authority['ordered_queries_sha256'],
|
||||
'plan_sha256': plan_sha256,
|
||||
'provenance_policy_sha256': authority['provenance_policy_sha256'],
|
||||
'selector_sha256': authority['selector_sha256'],
|
||||
'state': state,
|
||||
}
|
||||
|
||||
|
||||
def _execute_action(db, action, prepared, experiment, policy_sha256):
|
||||
if action == 'status':
|
||||
return _status(db, experiment, policy_sha256)
|
||||
if action == 'generate-cohort':
|
||||
manifest, manifest_sha256 = generate_docker_depth_cohort_manifest(
|
||||
db, experiment, policy_sha256,
|
||||
)
|
||||
path, created = _publish_manifest(prepared['path'], manifest)
|
||||
return {
|
||||
'action': action,
|
||||
'files_created': int(created),
|
||||
'manifest_sha256': manifest_sha256,
|
||||
'path': os.path.basename(path),
|
||||
'plan_sha256': manifest['plan_sha256'],
|
||||
'queries': len(manifest['plan']['queries']),
|
||||
'repositories': sum(
|
||||
len(item['repositories']) for item in manifest['plan']['queries']
|
||||
),
|
||||
}
|
||||
if action == 'apply-cohort':
|
||||
result = apply_docker_depth_cohort_manifest(
|
||||
db, experiment, policy_sha256, prepared['manifest'],
|
||||
prepared['manifest_sha256'],
|
||||
)
|
||||
return {
|
||||
'action': action,
|
||||
'manifest_sha256': prepared['manifest_sha256'],
|
||||
'path': os.path.basename(prepared['path']),
|
||||
'plan_sha256': result['plan_sha256'],
|
||||
'plans_persisted': int(bool(result['planned'])),
|
||||
'queries': len(result['plan']['queries']),
|
||||
'repositories': sum(
|
||||
len(item['repositories']) for item in result['plan']['queries']
|
||||
),
|
||||
}
|
||||
if action == 'generate-hold':
|
||||
manifest, manifest_sha256 = generate_docker_depth_hold_manifest(
|
||||
db, experiment, policy_sha256,
|
||||
)
|
||||
path, created = _publish_manifest(prepared['path'], manifest)
|
||||
return {
|
||||
'action': action,
|
||||
'conflicts': manifest['conflict_count'],
|
||||
'entries': manifest['entry_count'],
|
||||
'files_created': int(created),
|
||||
'manifest_sha256': manifest_sha256,
|
||||
'path': os.path.basename(path),
|
||||
'plan_sha256': manifest['plan_sha256'],
|
||||
}
|
||||
if action == 'apply-hold':
|
||||
result = apply_docker_depth_hold_manifest(
|
||||
db, experiment, policy_sha256, prepared['manifest'],
|
||||
prepared['manifest_sha256'],
|
||||
)
|
||||
return {
|
||||
'action': action,
|
||||
'conflicts': int(result['conflicts']),
|
||||
'duplicates': int(result['duplicates']),
|
||||
'manifest_sha256': prepared['manifest_sha256'],
|
||||
'path': os.path.basename(prepared['path']),
|
||||
'transitioned': int(result['transitioned']),
|
||||
}
|
||||
if action == 'generate-reactivation':
|
||||
manifest, manifest_sha256 = generate_docker_depth_reactivation_manifest(
|
||||
db, experiment, policy_sha256,
|
||||
)
|
||||
path, created = _publish_manifest(prepared['path'], manifest)
|
||||
return {
|
||||
'action': action,
|
||||
'entries': manifest['entry_count'],
|
||||
'files_created': int(created),
|
||||
'hold_manifest_sha256': manifest['hold_manifest_sha256'],
|
||||
'manifest_sha256': manifest_sha256,
|
||||
'path': os.path.basename(path),
|
||||
}
|
||||
if action == 'apply-reactivation':
|
||||
result = apply_docker_depth_reactivation_manifest(
|
||||
db, experiment, policy_sha256, prepared['manifest'],
|
||||
prepared['manifest_sha256'],
|
||||
)
|
||||
return {
|
||||
'action': action,
|
||||
'duplicates': int(result['duplicates']),
|
||||
'manifest_sha256': prepared['manifest_sha256'],
|
||||
'path': os.path.basename(prepared['path']),
|
||||
'transitioned': int(result['transitioned']),
|
||||
}
|
||||
if action == 'generate-resolver-disposition':
|
||||
manifest, manifest_sha256 = (
|
||||
generate_docker_depth_resolver_disposition_manifest(
|
||||
db, experiment, policy_sha256,
|
||||
)
|
||||
)
|
||||
path, created = _publish_manifest(prepared['path'], manifest)
|
||||
return {
|
||||
'action': action,
|
||||
'files_created': int(created),
|
||||
'manifest_sha256': manifest_sha256,
|
||||
'outcome': manifest['entries'][0]['outcome'],
|
||||
'path': os.path.basename(path),
|
||||
}
|
||||
if action == 'apply-resolver-disposition':
|
||||
result = apply_docker_depth_resolver_disposition_manifest(
|
||||
db, experiment, policy_sha256, prepared['manifest'],
|
||||
prepared['manifest_sha256'],
|
||||
)
|
||||
return {
|
||||
'action': action,
|
||||
'applied': int(result['applied']),
|
||||
'duplicates': int(result['duplicates']),
|
||||
'manifest_sha256': prepared['manifest_sha256'],
|
||||
'outcome': result['outcome'],
|
||||
'path': os.path.basename(prepared['path']),
|
||||
'state': result['state'],
|
||||
}
|
||||
if action == 'generate-resolver-refund':
|
||||
manifest, manifest_sha256 = generate_docker_depth_resolver_refund_manifest(
|
||||
db, experiment, policy_sha256, prepared['evidence_log_path'],
|
||||
)
|
||||
path, created = _publish_manifest(prepared['path'], manifest)
|
||||
return {
|
||||
'action': action,
|
||||
'entries': manifest['entry_count'],
|
||||
'files_created': int(created),
|
||||
'held_entries': manifest['held_entry_count'],
|
||||
'manifest_sha256': manifest_sha256,
|
||||
'path': os.path.basename(path),
|
||||
'refund_attempts': manifest['refund_attempts'],
|
||||
}
|
||||
if action == 'apply-resolver-refund':
|
||||
result = apply_docker_depth_resolver_refund_manifest(
|
||||
db, experiment, policy_sha256, prepared['manifest'],
|
||||
prepared['manifest_sha256'], prepared['evidence_log_path'],
|
||||
)
|
||||
return {
|
||||
'action': action,
|
||||
'duplicates': int(result['duplicates']),
|
||||
'manifest_sha256': prepared['manifest_sha256'],
|
||||
'path': os.path.basename(prepared['path']),
|
||||
'refunded': int(result['refunded']),
|
||||
'state': result['state'],
|
||||
}
|
||||
raise RuntimeError('Docker depth operator action is unsupported')
|
||||
|
||||
|
||||
def parse_args(argv=None):
|
||||
parser = argparse.ArgumentParser(
|
||||
description='Offline reviewed operator for the bounded Docker depth experiment.',
|
||||
)
|
||||
parser.add_argument(
|
||||
'--config', default=os.path.join(os.path.dirname(__file__), 'config.yaml'),
|
||||
)
|
||||
actions = parser.add_mutually_exclusive_group(required=True)
|
||||
actions.add_argument('--status', action='store_true')
|
||||
actions.add_argument('--generate-cohort-manifest')
|
||||
actions.add_argument('--apply-cohort-manifest')
|
||||
actions.add_argument('--generate-hold-manifest')
|
||||
actions.add_argument('--apply-hold-manifest')
|
||||
actions.add_argument('--generate-reactivation-manifest')
|
||||
actions.add_argument('--apply-reactivation-manifest')
|
||||
actions.add_argument('--generate-resolver-disposition-manifest')
|
||||
actions.add_argument('--apply-resolver-disposition-manifest')
|
||||
actions.add_argument('--generate-resolver-refund-manifest')
|
||||
actions.add_argument('--apply-resolver-refund-manifest')
|
||||
parser.add_argument('--approve-sha256')
|
||||
parser.add_argument('--apply', dest='confirm_apply', action='store_true')
|
||||
parser.add_argument('--sources-stopped', action='store_true')
|
||||
args = parser.parse_args(argv)
|
||||
action = _action_name(args)
|
||||
_require_action_arguments(parser, args, action)
|
||||
return args, action
|
||||
|
||||
|
||||
def main(argv=None):
|
||||
try:
|
||||
args, action = parse_args(argv)
|
||||
config_path = os.path.abspath(args.config)
|
||||
config = load_config(config_path)
|
||||
validated = validate_docker_depth_config(
|
||||
config, managed_postgres=True, final_cutover=True,
|
||||
)
|
||||
if validated.experiment is None:
|
||||
raise RuntimeError('Docker depth experiment configuration is unavailable')
|
||||
experiment = validated.experiment
|
||||
_require_action_config_state(experiment, action)
|
||||
policy_sha256 = provenance_policy_sha256(validated)
|
||||
prepared = _prepare_action(
|
||||
args, action, experiment, policy_sha256,
|
||||
)
|
||||
if action in ('generate-resolver-refund', 'apply-resolver-refund'):
|
||||
log_path = reject_reparse_components(os.path.abspath(os.path.join(
|
||||
validated.normalized_config['global']['log_dir'], 'dockerhub.log',
|
||||
)))
|
||||
require_private_file(log_path)
|
||||
prepared['evidence_log_path'] = log_path
|
||||
with operator_database(
|
||||
config_path, validated.normalized_config,
|
||||
read_only=not action.startswith('apply-'),
|
||||
) as db:
|
||||
report = _execute_action(
|
||||
db, action, prepared, experiment, policy_sha256,
|
||||
)
|
||||
print(json.dumps(report, ensure_ascii=True, sort_keys=True))
|
||||
return 0
|
||||
except Exception as exc:
|
||||
raise SystemExit(
|
||||
f'Docker depth operator failed closed: {type(exc).__name__}'
|
||||
) from None
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,849 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
if not sys.dont_write_bytecode:
|
||||
raise RuntimeError('Docker shadow runner could not disable bytecode writes')
|
||||
|
||||
import argparse
|
||||
import copy
|
||||
import hashlib
|
||||
import math
|
||||
import os
|
||||
import re
|
||||
import secrets
|
||||
import socket
|
||||
import time
|
||||
|
||||
import scanner
|
||||
import console_runner
|
||||
from keycheck_candidates import extract_candidates
|
||||
from lifecycle_authority import require_active_supervisor_child
|
||||
from scanner_db import (
|
||||
DOCKER_ADAPTIVE_GATE_MAX_CONTROLS,
|
||||
DOCKER_ADAPTIVE_GATE_MIN_CONTROLS,
|
||||
DOCKER_ADAPTIVE_LAYER_CLASS_ORDER,
|
||||
DOCKER_ADAPTIVE_SELECTOR_VERSION,
|
||||
DOCKER_ADAPTIVE_SHADOW_SELECTION_METRIC_KEYS,
|
||||
ScannerDB,
|
||||
canonical_docker_layer_plan_bytes,
|
||||
docker_content_media_class,
|
||||
docker_layer_execution_policy_sha256,
|
||||
docker_layer_selection_policy_sha256,
|
||||
finding_identity,
|
||||
select_docker_adaptive_payload,
|
||||
validate_docker_adaptive_checkpoint,
|
||||
validate_docker_adaptive_shadow_selection_metrics,
|
||||
validate_docker_layer_execution,
|
||||
validate_docker_layer_limits,
|
||||
validate_docker_layer_plan,
|
||||
validate_docker_layer_resolution,
|
||||
)
|
||||
|
||||
|
||||
_SHA256_RE = re.compile(r'[a-f0-9]{64}')
|
||||
_ROUTED_SERVICE_RE = re.compile(r'[a-z0-9][a-z0-9_.-]{0,63}')
|
||||
SHADOW_FAILURE_METRIC_KEYS = (
|
||||
'diagnostic_full_incomplete',
|
||||
'diagnostic_adaptive_incomplete',
|
||||
'diagnostic_blob_chunk_processing',
|
||||
'diagnostic_blob_detector_timeout',
|
||||
'diagnostic_blob_network',
|
||||
'diagnostic_blob_timeout',
|
||||
'diagnostic_blob_mixed',
|
||||
'diagnostic_blob_other',
|
||||
)
|
||||
_SHADOW_BLOB_FAILURE_METRICS = {
|
||||
'chunk_processing': 'diagnostic_blob_chunk_processing',
|
||||
'detector_timeout': 'diagnostic_blob_detector_timeout',
|
||||
'network': 'diagnostic_blob_network',
|
||||
'transfer_timeout': 'diagnostic_blob_timeout',
|
||||
'timeout': 'diagnostic_blob_timeout',
|
||||
'mixed': 'diagnostic_blob_mixed',
|
||||
}
|
||||
|
||||
|
||||
class DockerShadowPrivacyError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
def empty_selection_metrics():
|
||||
return {name: 0 for name in DOCKER_ADAPTIVE_SHADOW_SELECTION_METRIC_KEYS}
|
||||
|
||||
|
||||
def empty_failure_metrics():
|
||||
return {name: 0 for name in SHADOW_FAILURE_METRIC_KEYS}
|
||||
|
||||
|
||||
def _blob_failure_metric(error_code):
|
||||
return _SHADOW_BLOB_FAILURE_METRICS.get(
|
||||
str(error_code or ''), 'diagnostic_blob_other',
|
||||
)
|
||||
|
||||
|
||||
def _canonical_private_plan(plan):
|
||||
plan = validate_docker_layer_plan(plan)
|
||||
payload = canonical_docker_layer_plan_bytes(plan)
|
||||
return plan, hashlib.sha256(payload).hexdigest()
|
||||
|
||||
|
||||
def _validate_duplicate_descriptors(descriptors):
|
||||
identities = {}
|
||||
for descriptor in descriptors:
|
||||
identity = (
|
||||
descriptor['kind'], descriptor['size'],
|
||||
docker_content_media_class(descriptor['kind'], descriptor['media_type']),
|
||||
)
|
||||
previous = identities.setdefault(descriptor['digest'], identity)
|
||||
if previous != identity:
|
||||
raise ValueError('Docker shadow duplicate digest metadata conflicts')
|
||||
|
||||
|
||||
def build_private_adaptive_plan(
|
||||
resolved, payload_classes, limits, checkpoint, scan_policy_sha256,
|
||||
):
|
||||
resolved = validate_docker_layer_resolution(resolved)
|
||||
limits = validate_docker_layer_limits(limits)
|
||||
checkpoint = validate_docker_adaptive_checkpoint(checkpoint)
|
||||
scan_policy_sha256 = str(scan_policy_sha256 or '').lower()
|
||||
if not _SHA256_RE.fullmatch(scan_policy_sha256):
|
||||
raise ValueError('Docker shadow scan policy must be a lowercase SHA-256')
|
||||
|
||||
descriptors = [resolved['config'], *resolved['layers']]
|
||||
_validate_duplicate_descriptors(descriptors)
|
||||
decisions = select_docker_adaptive_payload(
|
||||
descriptors, payload_classes, limits, covered_digests=(),
|
||||
)
|
||||
metrics = empty_selection_metrics()
|
||||
planned = []
|
||||
omitted = 0
|
||||
for descriptor, selected, reason, payload_class in decisions:
|
||||
if reason == 'config_selected':
|
||||
metrics['selected_config'] += 1
|
||||
elif reason.startswith('selected_'):
|
||||
metrics[reason] += 1
|
||||
elif reason == 'already_covered':
|
||||
metrics['reuse_already_covered'] += 1
|
||||
elif reason == 'duplicate_digest':
|
||||
metrics['reuse_duplicate_digest'] += 1
|
||||
elif not selected:
|
||||
metric_name = f'omitted_{reason}'
|
||||
if metric_name not in metrics:
|
||||
raise ValueError('Docker shadow selection reason is not aggregate-safe')
|
||||
metrics[metric_name] += 1
|
||||
omitted += 1
|
||||
planned.append({
|
||||
**descriptor,
|
||||
'payload_class': payload_class,
|
||||
'selected': bool(selected),
|
||||
'selection_reason': reason,
|
||||
'coverage_state': 'selected' if selected else 'skipped',
|
||||
'lease_token': None,
|
||||
'attempt': 0,
|
||||
'max_attempts': limits['blob_max_attempts'],
|
||||
})
|
||||
|
||||
plan, plan_sha256 = _canonical_private_plan({
|
||||
'version': 2,
|
||||
'image': resolved['image'],
|
||||
'repository': resolved['repository'],
|
||||
'manifest_digest': resolved['manifest_digest'],
|
||||
'platform_os': resolved['platform_os'],
|
||||
'platform_arch': resolved['platform_arch'],
|
||||
'manifest_media_type': resolved['manifest_media_type'],
|
||||
'limits': limits,
|
||||
'selector_version': DOCKER_ADAPTIVE_SELECTOR_VERSION,
|
||||
'selection_policy_sha256': docker_layer_selection_policy_sha256(limits),
|
||||
'scan_policy_sha256': scan_policy_sha256,
|
||||
'execution_policy_sha256': docker_layer_execution_policy_sha256(
|
||||
scan_policy_sha256, limits,
|
||||
),
|
||||
'checkpoint': checkpoint,
|
||||
'descriptors': planned,
|
||||
})
|
||||
return {
|
||||
'plan': plan,
|
||||
'plan_sha256': plan_sha256,
|
||||
'selection_metrics': validate_docker_adaptive_shadow_selection_metrics(metrics),
|
||||
'omitted_descriptor_count': omitted,
|
||||
}
|
||||
|
||||
|
||||
def _checkpoint_order(descriptor):
|
||||
if descriptor['kind'] == 'config':
|
||||
return (0, 0, -descriptor['position'], descriptor['size'], descriptor['digest'])
|
||||
try:
|
||||
class_rank = DOCKER_ADAPTIVE_LAYER_CLASS_ORDER.index(descriptor['payload_class'])
|
||||
except ValueError as exc:
|
||||
raise ValueError('Docker shadow descriptor class is not schedulable') from exc
|
||||
return (1, class_rank, -descriptor['position'], descriptor['size'], descriptor['digest'])
|
||||
|
||||
|
||||
def lease_private_adaptive_checkpoint(plan, token_factory=None):
|
||||
plan = validate_docker_layer_plan(plan)
|
||||
if plan['version'] != 2:
|
||||
raise ValueError('Docker shadow checkpoints require a version-two plan')
|
||||
if any(
|
||||
descriptor['coverage_state'] in ('leased', 'shared_pending')
|
||||
for descriptor in plan['descriptors']
|
||||
):
|
||||
raise ValueError('Docker shadow plan already contains active leases')
|
||||
|
||||
pending = {}
|
||||
for descriptor in plan['descriptors']:
|
||||
if descriptor['coverage_state'] == 'selected':
|
||||
pending.setdefault(descriptor['digest'], []).append(descriptor)
|
||||
if not pending:
|
||||
return None
|
||||
|
||||
groups = []
|
||||
for digest, descriptors in pending.items():
|
||||
attempts = {item['attempt'] for item in descriptors}
|
||||
maximums = {item['max_attempts'] for item in descriptors}
|
||||
if len(attempts) != 1 or len(maximums) != 1:
|
||||
raise ValueError('Docker shadow duplicate attempts conflict')
|
||||
if next(iter(attempts)) >= next(iter(maximums)):
|
||||
raise ValueError('Docker shadow plan exceeded its private attempt budget')
|
||||
representative = min(descriptors, key=_checkpoint_order)
|
||||
groups.append((representative, digest))
|
||||
groups.sort(key=lambda item: _checkpoint_order(item[0]))
|
||||
|
||||
leased_digests = set()
|
||||
leased_bytes = 0
|
||||
max_blobs = plan['checkpoint']['max_blobs']
|
||||
max_bytes = plan['checkpoint']['max_bytes']
|
||||
for descriptor, digest in groups:
|
||||
if len(leased_digests) >= max_blobs:
|
||||
continue
|
||||
if leased_digests and leased_bytes + descriptor['size'] > max_bytes:
|
||||
continue
|
||||
leased_digests.add(digest)
|
||||
leased_bytes += descriptor['size']
|
||||
if not leased_digests:
|
||||
raise ValueError('Docker shadow checkpoint made no bounded progress')
|
||||
|
||||
token_factory = token_factory or (lambda: secrets.token_urlsafe(32))
|
||||
tokens = {digest: str(token_factory()) for digest in leased_digests}
|
||||
leased_plan = copy.deepcopy(plan)
|
||||
for descriptor in leased_plan['descriptors']:
|
||||
if descriptor['digest'] not in leased_digests:
|
||||
continue
|
||||
descriptor['coverage_state'] = 'leased'
|
||||
descriptor['lease_token'] = tokens[descriptor['digest']]
|
||||
descriptor['attempt'] += 1
|
||||
leased_plan, plan_sha256 = _canonical_private_plan(leased_plan)
|
||||
return {'plan': leased_plan, 'plan_sha256': plan_sha256}
|
||||
|
||||
|
||||
def apply_private_adaptive_execution(plan, execution):
|
||||
plan, plan_sha256 = _canonical_private_plan(plan)
|
||||
execution = validate_docker_layer_execution(execution, plan, plan_sha256)
|
||||
records = {record['digest']: record for record in execution['blobs']}
|
||||
next_plan = copy.deepcopy(plan)
|
||||
for descriptor in next_plan['descriptors']:
|
||||
if descriptor['coverage_state'] != 'leased':
|
||||
continue
|
||||
status = records[descriptor['digest']]['status']
|
||||
if status == 'covered':
|
||||
next_state = 'covered'
|
||||
elif status == 'retryable_failed' and descriptor['attempt'] < descriptor['max_attempts']:
|
||||
next_state = 'selected'
|
||||
else:
|
||||
next_state = 'terminal_failed'
|
||||
descriptor['coverage_state'] = next_state
|
||||
descriptor['lease_token'] = None
|
||||
return _canonical_private_plan(next_plan)[0]
|
||||
|
||||
|
||||
def private_result_identities(result, normalized_target):
|
||||
if not isinstance(result, dict):
|
||||
raise ValueError('Docker shadow private result must be an object')
|
||||
routed = set()
|
||||
detectors = set()
|
||||
try:
|
||||
scanner.strip_nearby_context_for_persistence(result)
|
||||
findings = result.get('findings') or []
|
||||
if not isinstance(findings, list):
|
||||
raise ValueError('Docker shadow private findings must be a list')
|
||||
for finding in findings:
|
||||
if not isinstance(finding, dict):
|
||||
raise ValueError('Docker shadow private finding must be an object')
|
||||
detector_hash = str(
|
||||
finding_identity('dockerhub', normalized_target, finding)[2] or ''
|
||||
).lower()
|
||||
if not _SHA256_RE.fullmatch(detector_hash):
|
||||
raise DockerShadowPrivacyError('Docker shadow detector identity is invalid')
|
||||
detectors.add(detector_hash)
|
||||
for candidate in extract_candidates(finding):
|
||||
service = str(candidate.service or '')
|
||||
provider_key_hash = str(candidate.provider_key_hash or '').lower()
|
||||
if (
|
||||
not _ROUTED_SERVICE_RE.fullmatch(service)
|
||||
or not _SHA256_RE.fullmatch(provider_key_hash)
|
||||
):
|
||||
raise DockerShadowPrivacyError('Docker shadow routed identity is invalid')
|
||||
routed.add((service, provider_key_hash))
|
||||
finally:
|
||||
findings = result.get('findings') if isinstance(result, dict) else None
|
||||
if isinstance(findings, list):
|
||||
for finding in findings:
|
||||
if isinstance(finding, dict):
|
||||
finding.clear()
|
||||
findings.clear()
|
||||
result.clear()
|
||||
return frozenset(routed), frozenset(detectors)
|
||||
|
||||
|
||||
def run_timed_private_side(
|
||||
side, operation, durable_checkpoint, *, timeout_sec,
|
||||
monotonic_ns=time.monotonic_ns,
|
||||
):
|
||||
if side not in ('full', 'adaptive'):
|
||||
raise ValueError('Docker shadow side is invalid')
|
||||
with scanner.scan_slot_scope(['docker-shadow', side], timeout_sec=timeout_sec):
|
||||
started_ns = monotonic_ns()
|
||||
metrics = operation()
|
||||
if not isinstance(metrics, dict) or any(
|
||||
not isinstance(name, str)
|
||||
or isinstance(value, bool)
|
||||
or not isinstance(value, int)
|
||||
or value < 0
|
||||
for name, value in metrics.items()
|
||||
):
|
||||
raise ValueError('Docker shadow private sink accepts only non-negative aggregates')
|
||||
durable_checkpoint()
|
||||
elapsed_ns = max(1, monotonic_ns() - started_ns)
|
||||
return metrics, max(1, math.ceil(elapsed_ns / 1_000_000))
|
||||
|
||||
|
||||
def _clear_private_result(result):
|
||||
if not isinstance(result, dict):
|
||||
return
|
||||
findings = result.get('findings')
|
||||
if isinstance(findings, list):
|
||||
for finding in findings:
|
||||
if isinstance(finding, dict):
|
||||
finding.clear()
|
||||
findings.clear()
|
||||
result.clear()
|
||||
|
||||
|
||||
def _full_scan_incomplete_reason(result):
|
||||
if result.get('source_failure'):
|
||||
return 'full_scan_source_failure'
|
||||
if result.get('skipped'):
|
||||
return 'full_scan_skipped'
|
||||
return 'full_scan_error'
|
||||
|
||||
|
||||
def _adaptive_scan_incomplete_reason(exc):
|
||||
if isinstance(exc, scanner.DockerLayerInfrastructureError):
|
||||
return 'adaptive_scan_infrastructure'
|
||||
if isinstance(exc, scanner.DockerContentTransferError):
|
||||
return 'adaptive_scan_transfer'
|
||||
if isinstance(exc, TimeoutError):
|
||||
return 'adaptive_scan_timeout'
|
||||
if isinstance(exc, scanner.DockerRemoteAccessError):
|
||||
return 'adaptive_scan_remote_access'
|
||||
if isinstance(exc, ValueError):
|
||||
return 'adaptive_scan_contract'
|
||||
return 'adaptive_scan_error'
|
||||
|
||||
|
||||
def shadow_failure_reason_code(exc):
|
||||
if isinstance(exc, DockerShadowPrivacyError):
|
||||
return 'privacy_violation'
|
||||
if isinstance(exc, (KeyboardInterrupt, SystemExit)):
|
||||
return 'operator_interrupt'
|
||||
if exc.__class__.__name__ == 'ScanSlotFatalError':
|
||||
return 'scan_slot_fatal'
|
||||
if isinstance(exc, TimeoutError):
|
||||
return 'timeout'
|
||||
if isinstance(exc, ValueError):
|
||||
return 'invalid_contract'
|
||||
if isinstance(exc, RuntimeError):
|
||||
message = str(exc).lower()
|
||||
if 'checkpoint' in message:
|
||||
return 'checkpoint_failure'
|
||||
if 'fence' in message or 'lease' in message:
|
||||
return 'report_fence_failure'
|
||||
if 'cohort' in message or 'control' in message:
|
||||
return 'control_failure'
|
||||
return 'runtime_failure'
|
||||
return 'unexpected_failure'
|
||||
|
||||
|
||||
def execute_private_adaptive_scan(
|
||||
normalized_target, *, limits, checkpoint, scan_policy_sha256, timeout_sec,
|
||||
platform_os='linux', platform_arch='amd64', min_free_bytes=0,
|
||||
scan_kwargs=None,
|
||||
):
|
||||
timeout_sec = max(1, int(timeout_sec))
|
||||
deadline = time.monotonic() + timeout_sec
|
||||
scan_kwargs = dict(scan_kwargs or {})
|
||||
resolved, bearer_auth = scanner.resolve_docker_content_manifest(
|
||||
normalized_target,
|
||||
platform_os=str(platform_os or 'linux'),
|
||||
platform_arch=str(platform_arch or 'amd64'),
|
||||
deadline=deadline,
|
||||
)
|
||||
payload_classes, bearer_auth = scanner.fetch_docker_config_payload_classes(
|
||||
resolved, bearer_auth, deadline=deadline,
|
||||
min_free_bytes=max(0, int(min_free_bytes or 0)),
|
||||
)
|
||||
built = build_private_adaptive_plan(
|
||||
resolved, payload_classes, limits, checkpoint, scan_policy_sha256,
|
||||
)
|
||||
plan = built['plan']
|
||||
selection_metrics = dict(built['selection_metrics'])
|
||||
failure_metrics = empty_failure_metrics()
|
||||
routed = set()
|
||||
detectors = set()
|
||||
selected_unique = {
|
||||
item['digest']: item['max_attempts']
|
||||
for item in plan['descriptors'] if item['selected']
|
||||
}
|
||||
max_checkpoints = max(1, sum(selected_unique.values()))
|
||||
checkpoint_count = 0
|
||||
while True:
|
||||
leased = lease_private_adaptive_checkpoint(plan)
|
||||
if leased is None:
|
||||
break
|
||||
checkpoint_count += 1
|
||||
if checkpoint_count > max_checkpoints or time.monotonic() >= deadline:
|
||||
raise RuntimeError('Docker shadow adaptive checkpoint budget was exhausted')
|
||||
work = {
|
||||
'plan': leased['plan'],
|
||||
'plan_sha256': leased['plan_sha256'],
|
||||
'bearer_auth': bearer_auth,
|
||||
'min_free_bytes': max(0, int(min_free_bytes or 0)),
|
||||
'deadline': deadline,
|
||||
}
|
||||
result = scanner.scan_docker_layer_plan(
|
||||
normalized_target, work,
|
||||
timeout_sec=max(1, math.ceil(deadline - time.monotonic())),
|
||||
detectors=scan_kwargs.get('detectors'),
|
||||
exclude_detectors=scan_kwargs.get('exclude_detectors'),
|
||||
no_verification=bool(scan_kwargs.get('no_verification', False)),
|
||||
trufflehog_config=scan_kwargs.get('trufflehog_config'),
|
||||
log_target=False,
|
||||
)
|
||||
try:
|
||||
execution = result.get('docker_layer_execution')
|
||||
next_plan = apply_private_adaptive_execution(
|
||||
leased['plan'], result.get('docker_layer_execution'),
|
||||
)
|
||||
records = {
|
||||
record['digest']: record for record in execution['blobs']
|
||||
}
|
||||
newly_terminal = {
|
||||
item['digest'] for item in next_plan['descriptors']
|
||||
if item['coverage_state'] == 'terminal_failed'
|
||||
and item['digest'] in records
|
||||
}
|
||||
for digest in newly_terminal:
|
||||
failure_metrics[
|
||||
_blob_failure_metric(records[digest]['error_code'])
|
||||
] += 1
|
||||
plan = next_plan
|
||||
checkpoint_routed, checkpoint_detectors = private_result_identities(
|
||||
result, normalized_target,
|
||||
)
|
||||
except Exception:
|
||||
_clear_private_result(result)
|
||||
raise
|
||||
routed.update(checkpoint_routed)
|
||||
detectors.update(checkpoint_detectors)
|
||||
del checkpoint_routed, checkpoint_detectors
|
||||
|
||||
if any(
|
||||
item['coverage_state'] in ('selected', 'leased', 'shared_pending')
|
||||
for item in plan['descriptors']
|
||||
):
|
||||
raise RuntimeError('Docker shadow adaptive plan did not reach a terminal state')
|
||||
selection_metrics['adaptive_checkpoints'] += checkpoint_count
|
||||
terminal_digests = {
|
||||
item['digest'] for item in plan['descriptors']
|
||||
if item['coverage_state'] == 'terminal_failed'
|
||||
}
|
||||
if sum(failure_metrics.values()) != len(terminal_digests):
|
||||
raise RuntimeError('Docker shadow terminal failure accounting is inconsistent')
|
||||
return {
|
||||
'routed': frozenset(routed),
|
||||
'detectors': frozenset(detectors),
|
||||
'selection_metrics': validate_docker_adaptive_shadow_selection_metrics(
|
||||
selection_metrics,
|
||||
),
|
||||
'omitted_descriptor_count': int(built['omitted_descriptor_count']),
|
||||
'failure_count': len(terminal_digests),
|
||||
'failure_metrics': failure_metrics,
|
||||
}
|
||||
|
||||
|
||||
def private_full_side_metrics(db, control, scan_kwargs):
|
||||
reference_routed, reference_detectors = db.docker_adaptive_shadow_control_identities(
|
||||
control['target_scan_id'],
|
||||
)
|
||||
reference_routed = set(reference_routed)
|
||||
reference_detectors = set(reference_detectors)
|
||||
rerun_routed = set()
|
||||
rerun_detectors = set()
|
||||
result = None
|
||||
try:
|
||||
max_attempts = min(10, max(1, int(scan_kwargs.get('max_attempts', 3) or 3)))
|
||||
incomplete_reason = 'full_scan_error'
|
||||
attempts = 0
|
||||
for attempts in range(1, max_attempts + 1):
|
||||
result = scanner.scan_docker_image(
|
||||
control['normalized_target'],
|
||||
timeout_sec=int(scan_kwargs['timeout_sec']),
|
||||
detectors=scan_kwargs.get('detectors'),
|
||||
exclude_detectors=scan_kwargs.get('exclude_detectors'),
|
||||
no_verification=bool(scan_kwargs.get('no_verification', False)),
|
||||
trufflehog_config=scan_kwargs.get('trufflehog_config'),
|
||||
config_dir=scanner.docker_token_manager.get_next_config(),
|
||||
trufflehog_concurrency=int(scan_kwargs.get('trufflehog_concurrency', 0) or 0),
|
||||
docker_recovery_limits=scan_kwargs.get('docker_recovery_limits'),
|
||||
docker_recovery_min_free_bytes=int(scan_kwargs.get('docker_recovery_min_free_bytes', 20 << 30)),
|
||||
log_target=False,
|
||||
)
|
||||
if not isinstance(result, dict):
|
||||
raise ValueError('Docker shadow full result must be an object')
|
||||
if not (
|
||||
result.get('errors') or result.get('skipped')
|
||||
or result.get('source_failure')
|
||||
or result.get('warnings') or result.get('degraded')
|
||||
or ((result.get('scan_meta') or {}).get('docker_layer_scope') or {}).get('coverage_complete') is False
|
||||
or ((result.get('scan_meta') or {}).get('docker_full_recovery') or {}).get('coverage_complete') is False
|
||||
):
|
||||
rerun_routed, rerun_detectors = map(
|
||||
set, private_result_identities(result, control['normalized_target']),
|
||||
)
|
||||
result = None
|
||||
return {
|
||||
'full_routed_count': len(reference_routed),
|
||||
'full_detector_count': len(reference_detectors),
|
||||
'failure_count': 0,
|
||||
'safety_regression_count': int(
|
||||
rerun_routed != reference_routed
|
||||
or rerun_detectors != reference_detectors
|
||||
),
|
||||
**empty_failure_metrics(),
|
||||
}
|
||||
incomplete_reason = _full_scan_incomplete_reason(result)
|
||||
retryable = bool(result.get('retryable', True))
|
||||
_clear_private_result(result)
|
||||
result = None
|
||||
if not retryable:
|
||||
break
|
||||
print(
|
||||
'Docker adaptive shadow full side incomplete: '
|
||||
f'reason_code={incomplete_reason} attempts={attempts}'
|
||||
)
|
||||
metrics = empty_failure_metrics()
|
||||
metrics['diagnostic_full_incomplete'] = 1
|
||||
metrics.update({
|
||||
'full_routed_count': len(reference_routed),
|
||||
'full_detector_count': len(reference_detectors),
|
||||
'failure_count': 1,
|
||||
'safety_regression_count': 0,
|
||||
})
|
||||
return metrics
|
||||
finally:
|
||||
_clear_private_result(result)
|
||||
reference_routed.clear()
|
||||
reference_detectors.clear()
|
||||
rerun_routed.clear()
|
||||
rerun_detectors.clear()
|
||||
|
||||
|
||||
def private_adaptive_side_metrics(
|
||||
db, control, *, limits, checkpoint, scan_policy_sha256, scan_kwargs,
|
||||
platform_os='linux', platform_arch='amd64', min_free_bytes=0,
|
||||
):
|
||||
reference_routed, reference_detectors = db.docker_adaptive_shadow_control_identities(
|
||||
control['target_scan_id'],
|
||||
)
|
||||
reference_routed = set(reference_routed)
|
||||
reference_detectors = set(reference_detectors)
|
||||
adaptive_routed = set()
|
||||
adaptive_detectors = set()
|
||||
outcome = None
|
||||
try:
|
||||
max_attempts = min(10, max(1, int(scan_kwargs.get('max_attempts', 3) or 3)))
|
||||
incomplete_reason = 'adaptive_scan_error'
|
||||
attempts = 0
|
||||
for attempts in range(1, max_attempts + 1):
|
||||
try:
|
||||
outcome = execute_private_adaptive_scan(
|
||||
control['normalized_target'], limits=limits, checkpoint=checkpoint,
|
||||
scan_policy_sha256=scan_policy_sha256,
|
||||
timeout_sec=int(scan_kwargs['timeout_sec']),
|
||||
platform_os=platform_os, platform_arch=platform_arch,
|
||||
min_free_bytes=min_free_bytes, scan_kwargs=scan_kwargs,
|
||||
)
|
||||
adaptive_routed = set(outcome.pop('routed'))
|
||||
adaptive_detectors = set(outcome.pop('detectors'))
|
||||
metrics = {
|
||||
'adaptive_routed_count': len(adaptive_routed),
|
||||
'routed_intersection_count': len(reference_routed & adaptive_routed),
|
||||
'adaptive_detector_count': len(adaptive_detectors),
|
||||
'detector_intersection_count': len(reference_detectors & adaptive_detectors),
|
||||
'omitted_descriptor_count': int(outcome['omitted_descriptor_count']),
|
||||
'failure_count': int(outcome['failure_count']),
|
||||
}
|
||||
metrics.update(outcome['selection_metrics'])
|
||||
metrics.update(outcome['failure_metrics'])
|
||||
return metrics
|
||||
except DockerShadowPrivacyError:
|
||||
raise
|
||||
except scanner.ScanSlotFatalError:
|
||||
raise
|
||||
except MemoryError:
|
||||
raise
|
||||
except Exception as exc:
|
||||
incomplete_reason = _adaptive_scan_incomplete_reason(exc)
|
||||
retryable = bool(getattr(exc, 'retryable', not isinstance(exc, ValueError)))
|
||||
if not retryable:
|
||||
break
|
||||
finally:
|
||||
if isinstance(outcome, dict):
|
||||
outcome.clear()
|
||||
outcome = None
|
||||
adaptive_routed.clear()
|
||||
adaptive_detectors.clear()
|
||||
print(
|
||||
'Docker adaptive shadow adaptive side incomplete: '
|
||||
f'reason_code={incomplete_reason} attempts={attempts}'
|
||||
)
|
||||
metrics = empty_selection_metrics()
|
||||
metrics.update(empty_failure_metrics())
|
||||
metrics['diagnostic_adaptive_incomplete'] = 1
|
||||
metrics.update({
|
||||
'adaptive_routed_count': 0,
|
||||
'routed_intersection_count': 0,
|
||||
'adaptive_detector_count': 0,
|
||||
'detector_intersection_count': 0,
|
||||
'omitted_descriptor_count': 0,
|
||||
'failure_count': 1,
|
||||
})
|
||||
return metrics
|
||||
finally:
|
||||
if isinstance(outcome, dict):
|
||||
outcome.clear()
|
||||
reference_routed.clear()
|
||||
reference_detectors.clear()
|
||||
adaptive_routed.clear()
|
||||
adaptive_detectors.clear()
|
||||
|
||||
|
||||
def _shadow_config(config):
|
||||
supervisor = config.get('supervisor') if isinstance(config, dict) else None
|
||||
supervisor = supervisor if isinstance(supervisor, dict) else {}
|
||||
value = supervisor.get('docker_shadow')
|
||||
if not isinstance(value, dict):
|
||||
raise ValueError('Docker shadow supervisor configuration is required')
|
||||
allowed = {'enabled', 'cohort_size', 'lease_seconds'}
|
||||
if set(value) - allowed:
|
||||
raise ValueError('Docker shadow supervisor configuration has unsupported fields')
|
||||
if not console_runner.bool_config(value.get('enabled'), False):
|
||||
raise ValueError('Docker shadow operator command is disabled')
|
||||
cohort_size = int(value.get('cohort_size', DOCKER_ADAPTIVE_GATE_MIN_CONTROLS))
|
||||
lease_seconds = int(value.get('lease_seconds', 3600))
|
||||
if not DOCKER_ADAPTIVE_GATE_MIN_CONTROLS <= cohort_size <= DOCKER_ADAPTIVE_GATE_MAX_CONTROLS:
|
||||
raise ValueError('Docker shadow cohort size must be between 50 and 100')
|
||||
if not 60 <= lease_seconds <= 86400:
|
||||
raise ValueError('Docker shadow lease duration is outside the supported range')
|
||||
return {'cohort_size': cohort_size, 'lease_seconds': lease_seconds}
|
||||
|
||||
|
||||
def _scan_kwargs(source_args):
|
||||
return {
|
||||
'timeout_sec': max(1, int(source_args.timeout)),
|
||||
'detectors': source_args.detectors,
|
||||
'exclude_detectors': source_args.exclude_detectors,
|
||||
'no_verification': bool(source_args.no_verification),
|
||||
'trufflehog_config': source_args.trufflehog_config,
|
||||
'trufflehog_concurrency': max(0, int(source_args.trufflehog_concurrency or 0)),
|
||||
'max_attempts': max(
|
||||
1, int(getattr(source_args, 'target_retry_max_attempts', 3) or 3),
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def run_shadow(config_path):
|
||||
require_active_supervisor_child(
|
||||
config_path, child_kind='docker-shadow', require_dsn=True,
|
||||
)
|
||||
config = console_runner.load_config(config_path)
|
||||
settings = _shadow_config(config)
|
||||
source_config = (config.get('sources') or {}).get('dockerhub')
|
||||
if not isinstance(source_config, dict):
|
||||
raise ValueError('Docker shadow requires the DockerHub source configuration')
|
||||
global_config = config.get('global') or {}
|
||||
if not isinstance(global_config, dict):
|
||||
raise ValueError('Docker shadow global configuration must be a mapping')
|
||||
|
||||
console_runner.apply_global_config(global_config)
|
||||
scanner.initialize_scanner_runtime(preflight_complete=True)
|
||||
secrets_config = console_runner.load_secrets(config, config_path)
|
||||
state = {
|
||||
'version': 1,
|
||||
'sources': {'dockerhub': console_runner.default_source_state()},
|
||||
}
|
||||
console_runner.configure_source_auth(
|
||||
'dockerhub', source_config, state=state, secrets=secrets_config,
|
||||
)
|
||||
source_args = console_runner.build_args_from_source_config(
|
||||
'dockerhub', source_config, global_config, '', auth_entry=None,
|
||||
)
|
||||
scanner.scan_config.drop_detectors = scanner.csv_items(source_args.drop_detectors)
|
||||
scanner.scan_config.trufflehog_job_memory_limit_bytes = int(
|
||||
source_args.trufflehog_job_memory_limit_bytes
|
||||
)
|
||||
scan_kwargs = _scan_kwargs(source_args)
|
||||
limits = console_runner.docker_layer_limits(source_args)
|
||||
scan_kwargs['docker_recovery_limits'] = limits
|
||||
scan_kwargs['docker_recovery_min_free_bytes'] = int(getattr(source_args, 'docker_layer_min_free_bytes', 20 << 30))
|
||||
checkpoint = console_runner.docker_adaptive_checkpoint(source_args)
|
||||
scan_policy_sha256 = console_runner.docker_layer_scan_policy_sha256(
|
||||
source_args, scan_kwargs,
|
||||
)
|
||||
execution_policy_sha256 = docker_layer_execution_policy_sha256(
|
||||
scan_policy_sha256, limits,
|
||||
)
|
||||
selection_policy_sha256 = docker_layer_selection_policy_sha256(limits)
|
||||
selection_salt = hashlib.sha256(
|
||||
('docker-shadow-controls-v1:' + scan_policy_sha256 + ':'
|
||||
+ execution_policy_sha256 + ':' + selection_policy_sha256).encode('ascii')
|
||||
).hexdigest()
|
||||
|
||||
db_url = str(os.getenv('TRUF_MANAGED_POSTGRES_DSN') or '')
|
||||
if not db_url:
|
||||
raise RuntimeError('Docker shadow canonical PostgreSQL authority is unavailable')
|
||||
db = ScannerDB(db_url=db_url, initialize=False)
|
||||
controls = []
|
||||
report = None
|
||||
owner = 'docker-shadow:{}:{}'.format(
|
||||
os.getpid(), hashlib.sha256(socket.gethostname().encode('utf-8')).hexdigest()[:16],
|
||||
)
|
||||
try:
|
||||
if not db.enabled:
|
||||
raise RuntimeError('Docker shadow PostgreSQL connection is unavailable')
|
||||
db.set_application_name('truf-docker-adaptive-shadow')
|
||||
controls = db.docker_adaptive_shadow_controls(
|
||||
scan_policy_sha256, settings['cohort_size'], selection_salt,
|
||||
)
|
||||
report = db.start_docker_adaptive_shadow_report(
|
||||
scan_policy_sha256, execution_policy_sha256, selection_policy_sha256,
|
||||
settings['cohort_size'], owner, lease_seconds=settings['lease_seconds'],
|
||||
)
|
||||
aggregate = {
|
||||
'completed_pairs': 0,
|
||||
'full_routed_count': 0,
|
||||
'adaptive_routed_count': 0,
|
||||
'routed_intersection_count': 0,
|
||||
'full_detector_count': 0,
|
||||
'adaptive_detector_count': 0,
|
||||
'detector_intersection_count': 0,
|
||||
'full_slot_ms': 0,
|
||||
'adaptive_slot_ms': 0,
|
||||
'omitted_descriptor_count': 0,
|
||||
'failure_count': 0,
|
||||
'privacy_violation_count': 0,
|
||||
'safety_regression_count': 0,
|
||||
}
|
||||
selection_metrics = empty_selection_metrics()
|
||||
failure_metrics = empty_failure_metrics()
|
||||
|
||||
def durable_checkpoint():
|
||||
db.checkpoint_docker_adaptive_shadow_report(
|
||||
report['report_token'], owner, report['lease_token'],
|
||||
lease_seconds=settings['lease_seconds'],
|
||||
)
|
||||
|
||||
for index, control in enumerate(controls):
|
||||
sides = ('full', 'adaptive') if index % 2 == 0 else ('adaptive', 'full')
|
||||
for side in sides:
|
||||
if side == 'full':
|
||||
operation = lambda control=control: private_full_side_metrics(
|
||||
db, control, scan_kwargs,
|
||||
)
|
||||
else:
|
||||
operation = lambda control=control: private_adaptive_side_metrics(
|
||||
db, control, limits=limits, checkpoint=checkpoint,
|
||||
scan_policy_sha256=scan_policy_sha256,
|
||||
scan_kwargs=scan_kwargs,
|
||||
platform_os=source_args.docker_platform_os,
|
||||
platform_arch=source_args.docker_platform_arch,
|
||||
min_free_bytes=source_args.docker_layer_min_free_bytes,
|
||||
)
|
||||
side_metrics, elapsed_ms = run_timed_private_side(
|
||||
side, operation, durable_checkpoint,
|
||||
timeout_sec=scan_kwargs['timeout_sec'],
|
||||
)
|
||||
aggregate[f'{side}_slot_ms'] += elapsed_ms
|
||||
for name, value in side_metrics.items():
|
||||
if name in selection_metrics:
|
||||
selection_metrics[name] += value
|
||||
elif name in failure_metrics:
|
||||
failure_metrics[name] += value
|
||||
else:
|
||||
aggregate[name] += value
|
||||
side_metrics.clear()
|
||||
aggregate['completed_pairs'] += 1
|
||||
control.clear()
|
||||
|
||||
completed = db.finish_docker_adaptive_shadow_report(
|
||||
report['report_token'], owner, report['lease_token'],
|
||||
selection_metrics=selection_metrics, **aggregate,
|
||||
)
|
||||
print(
|
||||
'Docker adaptive shadow report: '
|
||||
f'id={completed["report_id"]} passed={str(completed["passed"]).lower()} '
|
||||
f'controls={completed["completed_pairs"]} '
|
||||
f'routed_recall_ppm={completed["routed_recall_ppm"]} '
|
||||
f'slot_ratio_ppm={completed["slot_ratio_ppm"]} '
|
||||
'failure_categories=' + ','.join(
|
||||
f'{name.removeprefix("diagnostic_")}:{failure_metrics[name]}'
|
||||
for name in SHADOW_FAILURE_METRIC_KEYS
|
||||
if failure_metrics[name]
|
||||
)
|
||||
)
|
||||
return 0 if completed['passed'] else 2
|
||||
except BaseException as exc:
|
||||
if report is not None:
|
||||
try:
|
||||
db.fail_docker_adaptive_shadow_report(
|
||||
report['report_token'], owner, report['lease_token'],
|
||||
privacy_violation_count=int(isinstance(exc, DockerShadowPrivacyError)),
|
||||
safety_regression_count=0,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
print(
|
||||
'Docker adaptive shadow report failed safely: '
|
||||
f'reason_code={shadow_failure_reason_code(exc)}'
|
||||
)
|
||||
return 1
|
||||
finally:
|
||||
for control in controls:
|
||||
if isinstance(control, dict):
|
||||
control.clear()
|
||||
controls.clear()
|
||||
db.close()
|
||||
scanner.docker_token_manager.cleanup()
|
||||
|
||||
|
||||
def parse_args(argv=None):
|
||||
parser = argparse.ArgumentParser(description='Run one private Docker adaptive shadow report')
|
||||
parser.add_argument('--config', required=True)
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def main(argv=None):
|
||||
args = parse_args(argv)
|
||||
return run_shadow(args.config)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main())
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,131 @@
|
||||
import os
|
||||
import socket
|
||||
import stat
|
||||
import struct
|
||||
import time
|
||||
|
||||
from host_agent_protocol import (
|
||||
CLIENT_CONNECT_TIMEOUT_SECONDS,
|
||||
EXCHANGE_TIMEOUT_SECONDS,
|
||||
HOST_AGENT_SOCKET_PATH,
|
||||
MAX_RESPONSE_PAYLOAD_BYTES,
|
||||
HostAgentAction,
|
||||
HostAgentProtocolError,
|
||||
HostAgentRequest,
|
||||
HostAgentStatus,
|
||||
decode_response_frame,
|
||||
encode_request_frame,
|
||||
receive_frame,
|
||||
require_eof,
|
||||
send_frame,
|
||||
)
|
||||
|
||||
|
||||
class HostAgentClientError(RuntimeError):
|
||||
def __init__(self, category):
|
||||
self.category = category
|
||||
super().__init__('host operations agent request failed')
|
||||
|
||||
|
||||
class HostAgentUnavailableError(HostAgentClientError):
|
||||
pass
|
||||
|
||||
|
||||
class HostAgentRejectedError(HostAgentClientError):
|
||||
pass
|
||||
|
||||
|
||||
def _fixed_socket_is_safe():
|
||||
try:
|
||||
details = os.lstat(HOST_AGENT_SOCKET_PATH)
|
||||
except OSError:
|
||||
return False
|
||||
return stat.S_ISSOCK(details.st_mode) and details.st_uid == 0
|
||||
|
||||
|
||||
def _peer_credentials(sock):
|
||||
if not hasattr(socket, 'SO_PEERCRED'):
|
||||
raise HostAgentUnavailableError('peer_credentials_unavailable')
|
||||
try:
|
||||
raw = sock.getsockopt(
|
||||
socket.SOL_SOCKET, socket.SO_PEERCRED, struct.calcsize('3i'),
|
||||
)
|
||||
pid, uid, gid = struct.unpack('3i', raw)
|
||||
except (OSError, struct.error) as exc:
|
||||
raise HostAgentUnavailableError('peer_credentials_unavailable') from exc
|
||||
# A peer outside the client's PID namespace is reported as PID 0 even
|
||||
# though its UID/GID remain authoritative through SO_PEERCRED.
|
||||
if pid < 0:
|
||||
raise HostAgentUnavailableError('peer_identity_invalid')
|
||||
return pid, uid, gid
|
||||
|
||||
|
||||
class HostAgentClient:
|
||||
def __init__(self):
|
||||
pass
|
||||
|
||||
@staticmethod
|
||||
def is_available():
|
||||
return _fixed_socket_is_safe()
|
||||
|
||||
def dispatch(
|
||||
self, *, operation_id, action, active_config_sha256,
|
||||
active_secrets_sha256, candidate_config_sha256,
|
||||
candidate_secrets_sha256,
|
||||
):
|
||||
try:
|
||||
request = HostAgentRequest(
|
||||
operation_id=operation_id,
|
||||
action=HostAgentAction(action),
|
||||
active_config_sha256=active_config_sha256,
|
||||
active_secrets_sha256=active_secrets_sha256,
|
||||
candidate_config_sha256=candidate_config_sha256,
|
||||
candidate_secrets_sha256=candidate_secrets_sha256,
|
||||
)
|
||||
frame = encode_request_frame(request)
|
||||
except (HostAgentProtocolError, TypeError, ValueError) as exc:
|
||||
raise HostAgentRejectedError('request_invalid') from exc
|
||||
if not _fixed_socket_is_safe():
|
||||
raise HostAgentUnavailableError('socket_unavailable')
|
||||
deadline = time.monotonic() + EXCHANGE_TIMEOUT_SECONDS
|
||||
connection = None
|
||||
response = None
|
||||
try:
|
||||
family = getattr(socket, 'AF_UNIX', None)
|
||||
if family is None:
|
||||
raise HostAgentUnavailableError('unix_socket_unavailable')
|
||||
connection = socket.socket(family, socket.SOCK_STREAM)
|
||||
connection.settimeout(min(
|
||||
CLIENT_CONNECT_TIMEOUT_SECONDS,
|
||||
max(0.001, deadline - time.monotonic()),
|
||||
))
|
||||
connection.connect(HOST_AGENT_SOCKET_PATH)
|
||||
_pid, peer_uid, _gid = _peer_credentials(connection)
|
||||
if peer_uid != 0:
|
||||
raise HostAgentUnavailableError('server_identity_invalid')
|
||||
send_frame(connection, frame, deadline=deadline)
|
||||
connection.shutdown(socket.SHUT_WR)
|
||||
response = decode_response_frame(receive_frame(
|
||||
connection, maximum=MAX_RESPONSE_PAYLOAD_BYTES, deadline=deadline,
|
||||
))
|
||||
require_eof(connection, deadline=deadline)
|
||||
except HostAgentClientError:
|
||||
raise
|
||||
except HostAgentProtocolError as exc:
|
||||
raise HostAgentUnavailableError('response_invalid') from exc
|
||||
except (OSError, TimeoutError) as exc:
|
||||
raise HostAgentUnavailableError('transport_unavailable') from exc
|
||||
finally:
|
||||
if connection is not None:
|
||||
try:
|
||||
connection.close()
|
||||
except OSError:
|
||||
pass
|
||||
connection = frame = request = family = None
|
||||
if response.operation_id != operation_id:
|
||||
raise HostAgentUnavailableError('response_identity_invalid')
|
||||
if response.status is HostAgentStatus.ACCEPTED:
|
||||
return response
|
||||
if response.status is HostAgentStatus.REJECTED:
|
||||
raise HostAgentRejectedError('request_rejected')
|
||||
raise HostAgentUnavailableError('request_unavailable')
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,315 @@
|
||||
import hmac
|
||||
import json
|
||||
import re
|
||||
import struct
|
||||
import time
|
||||
import uuid
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
|
||||
|
||||
HOST_AGENT_SOCKET_PATH = '/run/truf/host-agent.sock'
|
||||
HOST_AGENT_RUNTIME_UID = 10001
|
||||
MAX_REQUEST_PAYLOAD_BYTES = 1024
|
||||
MAX_RESPONSE_PAYLOAD_BYTES = 256
|
||||
CLIENT_CONNECT_TIMEOUT_SECONDS = 1.0
|
||||
SERVER_READ_TIMEOUT_SECONDS = 2.0
|
||||
EXCHANGE_TIMEOUT_SECONDS = 5.0
|
||||
|
||||
_FRAME_HEADER_BYTES = 4
|
||||
_SHA256_RE = re.compile(r'^[0-9a-f]{64}$')
|
||||
_REQUEST_FIELDS = frozenset((
|
||||
'operation_id', 'action',
|
||||
'active_config_sha256', 'active_secrets_sha256',
|
||||
'candidate_config_sha256', 'candidate_secrets_sha256',
|
||||
))
|
||||
_RESPONSE_FIELDS = frozenset(('operation_id', 'status'))
|
||||
|
||||
|
||||
class HostAgentProtocolError(ValueError):
|
||||
def __init__(self, category):
|
||||
self.category = category
|
||||
super().__init__('host agent protocol message is invalid')
|
||||
|
||||
|
||||
class HostAgentAction(str, Enum):
|
||||
APPLY_CONFIG = 'apply-config'
|
||||
APPLY_SECRETS = 'apply-secrets'
|
||||
APPLY_BOTH = 'apply-both'
|
||||
RESTART = 'restart'
|
||||
|
||||
|
||||
class HostAgentStatus(str, Enum):
|
||||
ACCEPTED = 'accepted'
|
||||
UNAVAILABLE = 'unavailable'
|
||||
REJECTED = 'rejected'
|
||||
INVALID = 'invalid'
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class HostAgentRequest:
|
||||
operation_id: str
|
||||
action: HostAgentAction
|
||||
active_config_sha256: str
|
||||
active_secrets_sha256: str
|
||||
candidate_config_sha256: str | None
|
||||
candidate_secrets_sha256: str | None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class HostAgentResponse:
|
||||
operation_id: str | None
|
||||
status: HostAgentStatus
|
||||
|
||||
|
||||
def _canonical_uuid(value):
|
||||
if not isinstance(value, str) or not value:
|
||||
raise HostAgentProtocolError('operation_id')
|
||||
try:
|
||||
parsed = uuid.UUID(value)
|
||||
except (ValueError, AttributeError) as exc:
|
||||
raise HostAgentProtocolError('operation_id') from exc
|
||||
if parsed.int == 0 or str(parsed) != value:
|
||||
raise HostAgentProtocolError('operation_id')
|
||||
return value
|
||||
|
||||
|
||||
def _sha256(value, field, *, optional=False):
|
||||
if optional and value is None:
|
||||
return None
|
||||
if not isinstance(value, str) or _SHA256_RE.fullmatch(value) is None:
|
||||
raise HostAgentProtocolError(field)
|
||||
return value
|
||||
|
||||
|
||||
def _canonical_json(value):
|
||||
try:
|
||||
return json.dumps(
|
||||
value, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
|
||||
allow_nan=False,
|
||||
).encode('ascii')
|
||||
except (TypeError, ValueError, UnicodeError) as exc:
|
||||
raise HostAgentProtocolError('json') from exc
|
||||
|
||||
|
||||
def _strict_json(payload, *, maximum):
|
||||
if type(payload) is not bytes or not 1 <= len(payload) <= maximum:
|
||||
raise HostAgentProtocolError('bounds')
|
||||
|
||||
def reject_duplicate(pairs):
|
||||
result = {}
|
||||
for key, value in pairs:
|
||||
if key in result:
|
||||
raise HostAgentProtocolError('duplicate_field')
|
||||
result[key] = value
|
||||
return result
|
||||
|
||||
try:
|
||||
text = payload.decode('utf-8', errors='strict')
|
||||
value = json.loads(
|
||||
text, object_pairs_hook=reject_duplicate,
|
||||
parse_constant=lambda _value: (_ for _ in ()).throw(
|
||||
HostAgentProtocolError('constant')
|
||||
),
|
||||
)
|
||||
except HostAgentProtocolError:
|
||||
raise
|
||||
except (UnicodeDecodeError, json.JSONDecodeError, TypeError, ValueError) as exc:
|
||||
raise HostAgentProtocolError('json') from exc
|
||||
finally:
|
||||
text = None
|
||||
if not isinstance(value, dict):
|
||||
raise HostAgentProtocolError('shape')
|
||||
if not hmac.compare_digest(_canonical_json(value), payload):
|
||||
raise HostAgentProtocolError('canonical')
|
||||
return value
|
||||
|
||||
|
||||
def _normalize_request(value):
|
||||
if not isinstance(value, dict) or set(value) != _REQUEST_FIELDS:
|
||||
raise HostAgentProtocolError('shape')
|
||||
try:
|
||||
action = HostAgentAction(value.get('action'))
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise HostAgentProtocolError('action') from exc
|
||||
active_config = _sha256(value.get('active_config_sha256'), 'active_config_sha256')
|
||||
active_secrets = _sha256(value.get('active_secrets_sha256'), 'active_secrets_sha256')
|
||||
candidate_config = _sha256(
|
||||
value.get('candidate_config_sha256'), 'candidate_config_sha256', optional=True,
|
||||
)
|
||||
candidate_secrets = _sha256(
|
||||
value.get('candidate_secrets_sha256'), 'candidate_secrets_sha256', optional=True,
|
||||
)
|
||||
required = {
|
||||
HostAgentAction.APPLY_CONFIG: (True, False),
|
||||
HostAgentAction.APPLY_SECRETS: (False, True),
|
||||
HostAgentAction.APPLY_BOTH: (True, True),
|
||||
HostAgentAction.RESTART: (False, False),
|
||||
}[action]
|
||||
if (candidate_config is not None, candidate_secrets is not None) != required:
|
||||
raise HostAgentProtocolError('candidate_identity')
|
||||
return HostAgentRequest(
|
||||
operation_id=_canonical_uuid(value.get('operation_id')),
|
||||
action=action,
|
||||
active_config_sha256=active_config,
|
||||
active_secrets_sha256=active_secrets,
|
||||
candidate_config_sha256=candidate_config,
|
||||
candidate_secrets_sha256=candidate_secrets,
|
||||
)
|
||||
|
||||
|
||||
def _request_value(request):
|
||||
if not isinstance(request, HostAgentRequest):
|
||||
raise HostAgentProtocolError('request_type')
|
||||
return {
|
||||
'operation_id': request.operation_id,
|
||||
'action': request.action.value if isinstance(request.action, HostAgentAction) else request.action,
|
||||
'active_config_sha256': request.active_config_sha256,
|
||||
'active_secrets_sha256': request.active_secrets_sha256,
|
||||
'candidate_config_sha256': request.candidate_config_sha256,
|
||||
'candidate_secrets_sha256': request.candidate_secrets_sha256,
|
||||
}
|
||||
|
||||
|
||||
def encode_request_payload(request):
|
||||
normalized = _normalize_request(_request_value(request))
|
||||
payload = _canonical_json(_request_value(normalized))
|
||||
if len(payload) > MAX_REQUEST_PAYLOAD_BYTES:
|
||||
raise HostAgentProtocolError('bounds')
|
||||
return payload
|
||||
|
||||
|
||||
def decode_request_payload(payload):
|
||||
return _normalize_request(_strict_json(payload, maximum=MAX_REQUEST_PAYLOAD_BYTES))
|
||||
|
||||
|
||||
def _normalize_response(value):
|
||||
if not isinstance(value, dict) or set(value) != _RESPONSE_FIELDS:
|
||||
raise HostAgentProtocolError('shape')
|
||||
try:
|
||||
status = HostAgentStatus(value.get('status'))
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise HostAgentProtocolError('status') from exc
|
||||
operation_id = value.get('operation_id')
|
||||
if status is HostAgentStatus.INVALID:
|
||||
if operation_id is not None:
|
||||
raise HostAgentProtocolError('operation_id')
|
||||
else:
|
||||
operation_id = _canonical_uuid(operation_id)
|
||||
return HostAgentResponse(operation_id=operation_id, status=status)
|
||||
|
||||
|
||||
def _response_value(response):
|
||||
if not isinstance(response, HostAgentResponse):
|
||||
raise HostAgentProtocolError('response_type')
|
||||
return {
|
||||
'operation_id': response.operation_id,
|
||||
'status': response.status.value if isinstance(response.status, HostAgentStatus) else response.status,
|
||||
}
|
||||
|
||||
|
||||
def encode_response_payload(response):
|
||||
normalized = _normalize_response(_response_value(response))
|
||||
payload = _canonical_json(_response_value(normalized))
|
||||
if len(payload) > MAX_RESPONSE_PAYLOAD_BYTES:
|
||||
raise HostAgentProtocolError('bounds')
|
||||
return payload
|
||||
|
||||
|
||||
def decode_response_payload(payload):
|
||||
return _normalize_response(_strict_json(payload, maximum=MAX_RESPONSE_PAYLOAD_BYTES))
|
||||
|
||||
|
||||
def _encode_frame(payload, maximum):
|
||||
if type(payload) is not bytes or not 1 <= len(payload) <= maximum:
|
||||
raise HostAgentProtocolError('bounds')
|
||||
return struct.pack('!I', len(payload)) + payload
|
||||
|
||||
|
||||
def _decode_frame(frame, maximum):
|
||||
if type(frame) is not bytes or len(frame) < _FRAME_HEADER_BYTES:
|
||||
raise HostAgentProtocolError('frame')
|
||||
length = struct.unpack('!I', frame[:_FRAME_HEADER_BYTES])[0]
|
||||
if not 1 <= length <= maximum or len(frame) != _FRAME_HEADER_BYTES + length:
|
||||
raise HostAgentProtocolError('frame')
|
||||
return frame[_FRAME_HEADER_BYTES:]
|
||||
|
||||
|
||||
def encode_request_frame(request):
|
||||
return _encode_frame(encode_request_payload(request), MAX_REQUEST_PAYLOAD_BYTES)
|
||||
|
||||
|
||||
def decode_request_frame(frame):
|
||||
return decode_request_payload(_decode_frame(frame, MAX_REQUEST_PAYLOAD_BYTES))
|
||||
|
||||
|
||||
def encode_response_frame(response):
|
||||
return _encode_frame(encode_response_payload(response), MAX_RESPONSE_PAYLOAD_BYTES)
|
||||
|
||||
|
||||
def decode_response_frame(frame):
|
||||
return decode_response_payload(_decode_frame(frame, MAX_RESPONSE_PAYLOAD_BYTES))
|
||||
|
||||
|
||||
def _remaining(deadline):
|
||||
remaining = deadline - time.monotonic()
|
||||
if remaining <= 0:
|
||||
raise HostAgentProtocolError('timeout')
|
||||
return remaining
|
||||
|
||||
|
||||
def receive_frame(sock, *, maximum, deadline):
|
||||
header = _receive_exact(sock, _FRAME_HEADER_BYTES, deadline)
|
||||
length = struct.unpack('!I', header)[0]
|
||||
if not 1 <= length <= maximum:
|
||||
raise HostAgentProtocolError('bounds')
|
||||
return header + _receive_exact(sock, length, deadline)
|
||||
|
||||
|
||||
def _receive_exact(sock, length, deadline):
|
||||
chunks = bytearray()
|
||||
try:
|
||||
while len(chunks) < length:
|
||||
sock.settimeout(_remaining(deadline))
|
||||
chunk = sock.recv(length - len(chunks))
|
||||
if not chunk:
|
||||
raise HostAgentProtocolError('truncated')
|
||||
chunks.extend(chunk)
|
||||
return bytes(chunks)
|
||||
except HostAgentProtocolError:
|
||||
raise
|
||||
except (OSError, TimeoutError) as exc:
|
||||
raise HostAgentProtocolError('transport') from exc
|
||||
finally:
|
||||
chunks.clear()
|
||||
chunk = None
|
||||
|
||||
|
||||
def require_eof(sock, *, deadline):
|
||||
try:
|
||||
sock.settimeout(_remaining(deadline))
|
||||
if sock.recv(1):
|
||||
raise HostAgentProtocolError('trailing_data')
|
||||
except HostAgentProtocolError:
|
||||
raise
|
||||
except (OSError, TimeoutError) as exc:
|
||||
raise HostAgentProtocolError('transport') from exc
|
||||
|
||||
|
||||
def send_frame(sock, frame, *, deadline):
|
||||
if type(frame) is not bytes:
|
||||
raise HostAgentProtocolError('frame')
|
||||
view = memoryview(frame)
|
||||
try:
|
||||
while view:
|
||||
sock.settimeout(_remaining(deadline))
|
||||
sent = sock.send(view)
|
||||
if sent <= 0:
|
||||
raise HostAgentProtocolError('transport')
|
||||
view = view[sent:]
|
||||
except HostAgentProtocolError:
|
||||
raise
|
||||
except (OSError, TimeoutError) as exc:
|
||||
raise HostAgentProtocolError('transport') from exc
|
||||
finally:
|
||||
view.release()
|
||||
@@ -0,0 +1,94 @@
|
||||
"""Bounded runtime reconciliation of fixed host-agent result evidence."""
|
||||
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import stat
|
||||
|
||||
from host_agent_state import MAX_STATE_BYTES
|
||||
from runtime_security import reject_reparse_components
|
||||
|
||||
|
||||
HOST_RESULT_DIRECTORY = Path('/data/host-agent-results')
|
||||
HOST_ROOT_UID = 0
|
||||
HOST_RUNTIME_GID = 10001
|
||||
|
||||
|
||||
class HostResultError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
def fixed_result_directory_is_safe():
|
||||
try:
|
||||
reject_reparse_components(HOST_RESULT_DIRECTORY)
|
||||
details = os.stat(HOST_RESULT_DIRECTORY, follow_symlinks=False)
|
||||
return (
|
||||
stat.S_ISDIR(details.st_mode)
|
||||
and details.st_uid == HOST_ROOT_UID
|
||||
and details.st_gid == HOST_RUNTIME_GID
|
||||
and stat.S_IMODE(details.st_mode) == 0o750
|
||||
)
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _read_result(operation_id):
|
||||
path = HOST_RESULT_DIRECTORY / f'{operation_id}.json'
|
||||
descriptor = None
|
||||
try:
|
||||
flags = os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) | getattr(os, 'O_NOFOLLOW', 0)
|
||||
descriptor = os.open(path, flags)
|
||||
before = os.fstat(descriptor)
|
||||
if (
|
||||
not stat.S_ISREG(before.st_mode)
|
||||
or before.st_nlink != 1
|
||||
or before.st_uid != HOST_ROOT_UID
|
||||
or before.st_gid != HOST_RUNTIME_GID
|
||||
or stat.S_IMODE(before.st_mode) != 0o640
|
||||
):
|
||||
raise HostResultError('host result metadata is invalid')
|
||||
with os.fdopen(descriptor, 'rb') as handle:
|
||||
descriptor = None
|
||||
payload = handle.read(MAX_STATE_BYTES + 1)
|
||||
after = os.fstat(handle.fileno())
|
||||
current = os.stat(path, follow_symlinks=False)
|
||||
identity = lambda item: (
|
||||
item.st_dev, item.st_ino, item.st_size,
|
||||
getattr(item, 'st_mtime_ns', None), getattr(item, 'st_ctime_ns', None),
|
||||
)
|
||||
if (
|
||||
len(payload) > MAX_STATE_BYTES
|
||||
or identity(before) != identity(after)
|
||||
or identity(after) != identity(current)
|
||||
):
|
||||
raise HostResultError('host result changed during read')
|
||||
value = json.loads(payload.decode('ascii'))
|
||||
canonical = json.dumps(
|
||||
value, sort_keys=True, separators=(',', ':'), ensure_ascii=True,
|
||||
allow_nan=False,
|
||||
).encode('ascii')
|
||||
if canonical != payload:
|
||||
raise HostResultError('host result is not canonical')
|
||||
return payload
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
except HostResultError:
|
||||
raise
|
||||
except Exception:
|
||||
raise HostResultError('host result is invalid') from None
|
||||
finally:
|
||||
if descriptor is not None:
|
||||
os.close(descriptor)
|
||||
|
||||
|
||||
def reconcile_pending_host_results(database, *, limit=32):
|
||||
if not fixed_result_directory_is_safe():
|
||||
return 0
|
||||
reconciled = 0
|
||||
for operation in database.pending_runtime_agent_operations(limit=limit):
|
||||
envelope = _read_result(operation['operation_id'])
|
||||
if envelope is None:
|
||||
continue
|
||||
if database.reconcile_runtime_operation_result(envelope) is not None:
|
||||
reconciled += 1
|
||||
return reconciled
|
||||
@@ -0,0 +1,214 @@
|
||||
"""Fixed production authority and asynchronous host-operation dispatch."""
|
||||
|
||||
import hmac
|
||||
from pathlib import Path, PurePosixPath
|
||||
import threading
|
||||
|
||||
from host_agent_apply import HostApplyError, HostApplySession
|
||||
from host_agent_lifecycle import execute_fixed_operation
|
||||
from host_agent_protocol import HostAgentStatus, encode_request_payload
|
||||
from host_agent_state import HostOperationState
|
||||
from runtime_document import (
|
||||
MAX_CONFIG_DOCUMENT_BYTES,
|
||||
_resolve_package_manifest_path,
|
||||
load_yaml_document,
|
||||
)
|
||||
from runtime_security import read_stable_root_file
|
||||
from scanner_db import ScannerDB
|
||||
from worker_package import (
|
||||
MAX_WORKER_PACKAGE_MANIFEST_BYTES,
|
||||
load_worker_package_manifest_bytes,
|
||||
)
|
||||
|
||||
|
||||
HOST_RUNTIME_ACTIVE_ROOT = Path('/etc/truf/runtime')
|
||||
HOST_WORKER_PACKAGE_ROOT = Path('/etc/truf/worker-packages')
|
||||
|
||||
|
||||
class HostRuntimeError(RuntimeError):
|
||||
def __init__(self, category):
|
||||
self.category = str(category)
|
||||
super().__init__('host operation runtime failed')
|
||||
|
||||
|
||||
def _stable_root_file(path, maximum):
|
||||
try:
|
||||
return read_stable_root_file(path, maximum, HOST_WORKER_PACKAGE_ROOT)
|
||||
except Exception:
|
||||
raise HostRuntimeError('package_evidence') from None
|
||||
|
||||
|
||||
def _host_manifest_path(resolved):
|
||||
value = PurePosixPath(resolved)
|
||||
try:
|
||||
relative = value.relative_to(PurePosixPath('/data/worker-packages'))
|
||||
except ValueError:
|
||||
raise HostRuntimeError('package_evidence') from None
|
||||
if not relative.parts or any(part in ('', '.', '..') for part in relative.parts):
|
||||
raise HostRuntimeError('package_evidence')
|
||||
return HOST_WORKER_PACKAGE_ROOT.joinpath(*relative.parts)
|
||||
|
||||
|
||||
def load_fixed_package_capabilities(config_payload):
|
||||
config = load_yaml_document(
|
||||
config_payload, max_bytes=MAX_CONFIG_DOCUMENT_BYTES,
|
||||
)
|
||||
try:
|
||||
profiles = config['supervisor']['worker_api']['compatibility_profiles']
|
||||
except (KeyError, TypeError):
|
||||
raise HostRuntimeError('package_evidence') from None
|
||||
if type(profiles) is not dict:
|
||||
raise HostRuntimeError('package_evidence')
|
||||
evidence = {}
|
||||
try:
|
||||
for profile_name, profile in profiles.items():
|
||||
if type(profile_name) is not str or type(profile) is not dict:
|
||||
raise HostRuntimeError('package_evidence')
|
||||
reference = profile.get('package_manifest')
|
||||
resolved = _resolve_package_manifest_path(config, reference)
|
||||
if resolved is None:
|
||||
raise HostRuntimeError('package_evidence')
|
||||
payload = _stable_root_file(
|
||||
_host_manifest_path(resolved), MAX_WORKER_PACKAGE_MANIFEST_BYTES,
|
||||
)
|
||||
manifest = load_worker_package_manifest_bytes(payload)
|
||||
evidence[profile_name] = {
|
||||
'package_manifest': reference,
|
||||
'capabilities': manifest['capabilities'],
|
||||
}
|
||||
return evidence
|
||||
except HostRuntimeError:
|
||||
raise
|
||||
except Exception:
|
||||
raise HostRuntimeError('package_evidence') from None
|
||||
finally:
|
||||
config = profiles = profile_name = profile = reference = None
|
||||
resolved = payload = manifest = None
|
||||
|
||||
|
||||
class FixedHostOperationDispatcher:
|
||||
def __init__(self):
|
||||
self._guard = threading.Lock()
|
||||
self._active_request = None
|
||||
self._worker = None
|
||||
self._closing = False
|
||||
|
||||
def _execute(self, session, database, state):
|
||||
try:
|
||||
try:
|
||||
execute_fixed_operation(session, state=state)
|
||||
except BaseException:
|
||||
# The durable executor owns safety/result handling. Do not let
|
||||
# thread tracebacks disclose host details at this outer boundary.
|
||||
pass
|
||||
finally:
|
||||
try:
|
||||
session.close()
|
||||
except BaseException:
|
||||
pass
|
||||
try:
|
||||
database.close()
|
||||
except BaseException:
|
||||
pass
|
||||
finally:
|
||||
with self._guard:
|
||||
self._active_request = None
|
||||
self._worker = None
|
||||
|
||||
@staticmethod
|
||||
def _record_validation_failure(state, request):
|
||||
try:
|
||||
phase = state.initialize('original')
|
||||
if phase.get('phase') == 'prepared':
|
||||
if phase.get('publication_state') != 'original':
|
||||
return False
|
||||
phase = state.advance(
|
||||
'prepared', 'failed', 'original',
|
||||
forward_category='validation_failed',
|
||||
safe_detail='validation_failed',
|
||||
)
|
||||
if (
|
||||
phase.get('phase') != 'failed'
|
||||
or phase.get('publication_state') != 'original'
|
||||
or phase.get('forward_category') != 'validation_failed'
|
||||
or phase.get('safe_detail') != 'validation_failed'
|
||||
):
|
||||
return False
|
||||
state.publish_result(
|
||||
'failed',
|
||||
safe_category='validation_failed',
|
||||
safe_detail='validation_failed',
|
||||
resulting_identity={
|
||||
'active_config_sha256': request.active_config_sha256,
|
||||
'active_secrets_sha256': request.active_secrets_sha256,
|
||||
},
|
||||
)
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def handle(self, request):
|
||||
encoded = encode_request_payload(request)
|
||||
with self._guard:
|
||||
if self._closing:
|
||||
return HostAgentStatus.UNAVAILABLE
|
||||
if self._active_request is not None:
|
||||
return (
|
||||
HostAgentStatus.ACCEPTED
|
||||
if hmac.compare_digest(encoded, self._active_request)
|
||||
else HostAgentStatus.REJECTED
|
||||
)
|
||||
state = HostOperationState(request)
|
||||
if state.terminal_result() is not None:
|
||||
return HostAgentStatus.ACCEPTED
|
||||
database = None
|
||||
session = None
|
||||
try:
|
||||
database = ScannerDB.host_agent_authority()
|
||||
session = HostApplySession(
|
||||
request, database,
|
||||
package_capability_provider=load_fixed_package_capabilities,
|
||||
)
|
||||
session.__enter__()
|
||||
state.initialize(session.publication_state)
|
||||
worker = threading.Thread(
|
||||
target=self._execute,
|
||||
args=(session, database, state),
|
||||
name='truf-host-operation',
|
||||
daemon=False,
|
||||
)
|
||||
self._active_request = encoded
|
||||
self._worker = worker
|
||||
worker.start()
|
||||
except Exception as error:
|
||||
accepted = (
|
||||
isinstance(error, HostApplyError)
|
||||
and error.category == 'validation'
|
||||
and session is not None
|
||||
and session.claim is not None
|
||||
and self._record_validation_failure(state, request)
|
||||
)
|
||||
self._active_request = None
|
||||
self._worker = None
|
||||
if session is not None:
|
||||
try:
|
||||
session.close()
|
||||
except BaseException:
|
||||
pass
|
||||
if database is not None:
|
||||
try:
|
||||
database.close()
|
||||
except BaseException:
|
||||
pass
|
||||
return (
|
||||
HostAgentStatus.ACCEPTED
|
||||
if accepted else HostAgentStatus.UNAVAILABLE
|
||||
)
|
||||
return HostAgentStatus.ACCEPTED
|
||||
|
||||
def close(self):
|
||||
with self._guard:
|
||||
self._closing = True
|
||||
worker = self._worker
|
||||
if worker is not None:
|
||||
worker.join()
|
||||
@@ -0,0 +1,167 @@
|
||||
import os
|
||||
import socket
|
||||
import stat
|
||||
import struct
|
||||
import time
|
||||
|
||||
from host_agent_protocol import (
|
||||
EXCHANGE_TIMEOUT_SECONDS,
|
||||
HOST_AGENT_RUNTIME_UID,
|
||||
HOST_AGENT_SOCKET_PATH,
|
||||
MAX_REQUEST_PAYLOAD_BYTES,
|
||||
SERVER_READ_TIMEOUT_SECONDS,
|
||||
HostAgentProtocolError,
|
||||
HostAgentResponse,
|
||||
HostAgentStatus,
|
||||
decode_request_frame,
|
||||
encode_response_frame,
|
||||
receive_frame,
|
||||
require_eof,
|
||||
send_frame,
|
||||
)
|
||||
|
||||
|
||||
SYSTEMD_LISTEN_FD = 3
|
||||
ACCEPT_POLL_SECONDS = 1.0
|
||||
|
||||
|
||||
class HostAgentServerError(RuntimeError):
|
||||
def __init__(self, category):
|
||||
self.category = category
|
||||
super().__init__('host operations agent server failed')
|
||||
|
||||
|
||||
def _peer_credentials(connection):
|
||||
if not hasattr(socket, 'SO_PEERCRED'):
|
||||
raise HostAgentServerError('peer_credentials_unavailable')
|
||||
try:
|
||||
raw = connection.getsockopt(
|
||||
socket.SOL_SOCKET, socket.SO_PEERCRED, struct.calcsize('3i'),
|
||||
)
|
||||
pid, uid, gid = struct.unpack('3i', raw)
|
||||
except (OSError, struct.error) as exc:
|
||||
raise HostAgentServerError('peer_credentials_unavailable') from exc
|
||||
return pid, uid, gid
|
||||
|
||||
|
||||
def unavailable_handler(_request):
|
||||
return HostAgentStatus.UNAVAILABLE
|
||||
|
||||
|
||||
def serve_connection(connection, *, handler=unavailable_handler, accepted_at=None):
|
||||
if not isinstance(connection, socket.socket):
|
||||
raise HostAgentServerError('connection_invalid')
|
||||
started = time.monotonic() if accepted_at is None else accepted_at
|
||||
try:
|
||||
peer_pid, peer_uid, _peer_gid = _peer_credentials(connection)
|
||||
except HostAgentServerError:
|
||||
return False
|
||||
if peer_pid <= 0 or peer_uid != HOST_AGENT_RUNTIME_UID:
|
||||
return False
|
||||
|
||||
request = None
|
||||
response = None
|
||||
try:
|
||||
read_deadline = min(
|
||||
started + SERVER_READ_TIMEOUT_SECONDS,
|
||||
started + EXCHANGE_TIMEOUT_SECONDS,
|
||||
)
|
||||
frame = receive_frame(
|
||||
connection, maximum=MAX_REQUEST_PAYLOAD_BYTES,
|
||||
deadline=read_deadline,
|
||||
)
|
||||
require_eof(connection, deadline=read_deadline)
|
||||
request = decode_request_frame(frame)
|
||||
except HostAgentProtocolError:
|
||||
response = HostAgentResponse(None, HostAgentStatus.INVALID)
|
||||
else:
|
||||
try:
|
||||
outcome = handler(request)
|
||||
if isinstance(outcome, HostAgentResponse):
|
||||
response = outcome
|
||||
else:
|
||||
response = HostAgentResponse(
|
||||
request.operation_id, HostAgentStatus(outcome),
|
||||
)
|
||||
if response.operation_id != request.operation_id:
|
||||
raise HostAgentServerError('handler_identity_invalid')
|
||||
except BaseException as exc:
|
||||
if not isinstance(exc, Exception):
|
||||
raise
|
||||
response = HostAgentResponse(
|
||||
request.operation_id, HostAgentStatus.UNAVAILABLE,
|
||||
)
|
||||
try:
|
||||
send_frame(
|
||||
connection, encode_response_frame(response),
|
||||
deadline=started + EXCHANGE_TIMEOUT_SECONDS,
|
||||
)
|
||||
except HostAgentProtocolError:
|
||||
return False
|
||||
finally:
|
||||
frame = request = response = outcome = None
|
||||
return True
|
||||
|
||||
|
||||
def _validate_listener(listener):
|
||||
if not isinstance(listener, socket.socket):
|
||||
raise HostAgentServerError('listener_invalid')
|
||||
unix_family = getattr(socket, 'AF_UNIX', None)
|
||||
if unix_family is None:
|
||||
raise HostAgentServerError('listener_invalid')
|
||||
try:
|
||||
socket_type = listener.getsockopt(socket.SOL_SOCKET, socket.SO_TYPE)
|
||||
accepting = listener.getsockopt(socket.SOL_SOCKET, socket.SO_ACCEPTCONN)
|
||||
except OSError as exc:
|
||||
raise HostAgentServerError('listener_invalid') from exc
|
||||
if (
|
||||
listener.family != unix_family
|
||||
or socket_type != socket.SOCK_STREAM
|
||||
or accepting != 1
|
||||
):
|
||||
raise HostAgentServerError('listener_invalid')
|
||||
try:
|
||||
if listener.getsockname() != HOST_AGENT_SOCKET_PATH:
|
||||
raise HostAgentServerError('listener_path_invalid')
|
||||
details = os.lstat(HOST_AGENT_SOCKET_PATH)
|
||||
except HostAgentServerError:
|
||||
raise
|
||||
except OSError as exc:
|
||||
raise HostAgentServerError('listener_unavailable') from exc
|
||||
if not stat.S_ISSOCK(details.st_mode) or details.st_uid != 0:
|
||||
raise HostAgentServerError('listener_owner_invalid')
|
||||
|
||||
|
||||
def inherited_systemd_listener():
|
||||
geteuid = getattr(os, 'geteuid', None)
|
||||
if geteuid is None or geteuid() != 0:
|
||||
raise HostAgentServerError('root_required')
|
||||
if os.environ.get('LISTEN_PID') != str(os.getpid()):
|
||||
raise HostAgentServerError('socket_activation_invalid')
|
||||
if os.environ.get('LISTEN_FDS') != '1':
|
||||
raise HostAgentServerError('socket_activation_invalid')
|
||||
try:
|
||||
listener = socket.socket(fileno=SYSTEMD_LISTEN_FD)
|
||||
_validate_listener(listener)
|
||||
except BaseException:
|
||||
try:
|
||||
listener.close()
|
||||
except (OSError, UnboundLocalError):
|
||||
pass
|
||||
raise
|
||||
return listener
|
||||
|
||||
|
||||
def serve_forever(listener, *, handler=unavailable_handler, stop_event=None):
|
||||
_validate_listener(listener)
|
||||
if stop_event is not None:
|
||||
listener.settimeout(ACCEPT_POLL_SECONDS)
|
||||
while stop_event is None or not stop_event.is_set():
|
||||
try:
|
||||
connection, _address = listener.accept()
|
||||
except InterruptedError:
|
||||
continue
|
||||
except TimeoutError:
|
||||
continue
|
||||
with connection:
|
||||
serve_connection(connection, handler=handler)
|
||||
@@ -0,0 +1,412 @@
|
||||
"""Durable fixed-path evidence for privileged runtime operations."""
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import stat
|
||||
|
||||
from host_agent_protocol import decode_request_payload, encode_request_payload
|
||||
from runtime_security import fsync_directory, reject_reparse_components
|
||||
|
||||
|
||||
HOST_ROOT_UID = 0
|
||||
HOST_ROOT_GID = 0
|
||||
HOST_RUNTIME_GID = 10001
|
||||
|
||||
HOST_STATE_ROOT = Path('/var/lib/truf/host-agent')
|
||||
HOST_STATE_ROOT_MODE = 0o700
|
||||
HOST_OPERATION_DIRECTORY = HOST_STATE_ROOT / 'operations'
|
||||
HOST_OPERATION_DIRECTORY_MODE = 0o700
|
||||
HOST_RESULT_DIRECTORY = HOST_STATE_ROOT / 'results'
|
||||
HOST_RESULT_DIRECTORY_MODE = 0o750
|
||||
HOST_FAILED_HOLD_PATH = HOST_STATE_ROOT / 'failed-hold.json'
|
||||
|
||||
MAX_STATE_BYTES = 16 * 1024
|
||||
_PHASES = {
|
||||
'prepared', 'forward_started', 'rollback_started',
|
||||
'succeeded', 'failed', 'rolled_back', 'failed_hold',
|
||||
}
|
||||
_PUBLICATION_STATES = {'original', 'partial', 'candidate'}
|
||||
_TERMINAL_RESULTS = {'succeeded', 'failed', 'rolled_back', 'failed_hold'}
|
||||
_TRANSITIONS = {
|
||||
'prepared': {'forward_started', 'rollback_started', 'failed'},
|
||||
'forward_started': {'rollback_started', 'succeeded'},
|
||||
'rollback_started': {'rolled_back', 'failed_hold'},
|
||||
}
|
||||
|
||||
|
||||
class HostStateError(RuntimeError):
|
||||
def __init__(self, category, *, cancellation=None):
|
||||
self.category = str(category)
|
||||
self.cancellation = cancellation
|
||||
super().__init__('host runtime state failed')
|
||||
|
||||
|
||||
def _canonical(value):
|
||||
try:
|
||||
payload = json.dumps(
|
||||
value, sort_keys=True, separators=(',', ':'), ensure_ascii=True,
|
||||
allow_nan=False,
|
||||
).encode('ascii')
|
||||
except (TypeError, ValueError):
|
||||
raise HostStateError('evidence') from None
|
||||
if not payload or len(payload) > MAX_STATE_BYTES:
|
||||
raise HostStateError('evidence')
|
||||
return payload
|
||||
|
||||
|
||||
def _require_directory(path, *, gid, mode):
|
||||
try:
|
||||
reject_reparse_components(path)
|
||||
details = os.stat(path, follow_symlinks=False)
|
||||
if not stat.S_ISDIR(details.st_mode):
|
||||
raise OSError('not a directory')
|
||||
if os.name != 'nt' and (
|
||||
details.st_uid != HOST_ROOT_UID
|
||||
or details.st_gid != gid
|
||||
or stat.S_IMODE(details.st_mode) != mode
|
||||
):
|
||||
raise OSError('directory metadata')
|
||||
except Exception:
|
||||
raise HostStateError('filesystem') from None
|
||||
|
||||
|
||||
def _read_file(path, *, gid, mode):
|
||||
descriptor = None
|
||||
try:
|
||||
reject_reparse_components(Path(path).parent)
|
||||
flags = os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0)
|
||||
if hasattr(os, 'O_BINARY'):
|
||||
flags |= os.O_BINARY
|
||||
if hasattr(os, 'O_NOFOLLOW'):
|
||||
flags |= os.O_NOFOLLOW
|
||||
descriptor = os.open(path, flags)
|
||||
before = os.fstat(descriptor)
|
||||
if (
|
||||
not stat.S_ISREG(before.st_mode) or before.st_nlink != 1
|
||||
or (
|
||||
os.name != 'nt'
|
||||
and (
|
||||
before.st_uid != HOST_ROOT_UID or before.st_gid != gid
|
||||
or stat.S_IMODE(before.st_mode) != mode
|
||||
)
|
||||
)
|
||||
):
|
||||
raise OSError('file metadata')
|
||||
with os.fdopen(descriptor, 'rb') as handle:
|
||||
descriptor = None
|
||||
payload = handle.read(MAX_STATE_BYTES + 1)
|
||||
after = os.fstat(handle.fileno())
|
||||
current = os.stat(path, follow_symlinks=False)
|
||||
identity = lambda item: (
|
||||
item.st_dev, item.st_ino, item.st_size,
|
||||
getattr(item, 'st_mtime_ns', None),
|
||||
None if os.name == 'nt' else getattr(item, 'st_ctime_ns', None),
|
||||
)
|
||||
if (
|
||||
identity(before) != identity(after)
|
||||
or identity(after) != identity(current)
|
||||
or len(payload) > MAX_STATE_BYTES
|
||||
):
|
||||
raise OSError('file changed')
|
||||
return payload
|
||||
except FileNotFoundError:
|
||||
raise
|
||||
except Exception:
|
||||
raise HostStateError('filesystem') from None
|
||||
finally:
|
||||
if descriptor is not None:
|
||||
os.close(descriptor)
|
||||
|
||||
|
||||
def _decode_canonical(payload):
|
||||
try:
|
||||
value = json.loads(payload.decode('ascii'))
|
||||
except (UnicodeDecodeError, json.JSONDecodeError):
|
||||
raise HostStateError('evidence') from None
|
||||
if not isinstance(value, dict) or _canonical(value) != payload:
|
||||
raise HostStateError('evidence')
|
||||
return value
|
||||
|
||||
|
||||
def _write_stage(path, payload, *, gid, mode):
|
||||
stage = Path(path).parent / f'.{Path(path).name}.stage'
|
||||
descriptor = None
|
||||
created = False
|
||||
published = False
|
||||
try:
|
||||
try:
|
||||
details = os.stat(stage, follow_symlinks=False)
|
||||
if (
|
||||
not stat.S_ISREG(details.st_mode) or details.st_nlink != 1
|
||||
or (os.name != 'nt' and details.st_uid != HOST_ROOT_UID)
|
||||
):
|
||||
raise OSError('unsafe stage')
|
||||
os.unlink(stage)
|
||||
fsync_directory(stage.parent)
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_CLOEXEC', 0)
|
||||
if hasattr(os, 'O_BINARY'):
|
||||
flags |= os.O_BINARY
|
||||
if hasattr(os, 'O_NOFOLLOW'):
|
||||
flags |= os.O_NOFOLLOW
|
||||
descriptor = os.open(stage, flags, mode)
|
||||
created = True
|
||||
if os.name != 'nt':
|
||||
os.fchmod(descriptor, mode)
|
||||
details = os.fstat(descriptor)
|
||||
if details.st_uid != HOST_ROOT_UID or details.st_gid != gid:
|
||||
os.fchown(descriptor, HOST_ROOT_UID, gid)
|
||||
view = memoryview(payload)
|
||||
written = 0
|
||||
while written < len(view):
|
||||
count = os.write(descriptor, view[written:])
|
||||
if count <= 0:
|
||||
raise OSError('short write')
|
||||
written += count
|
||||
os.fsync(descriptor)
|
||||
os.close(descriptor)
|
||||
descriptor = None
|
||||
os.replace(stage, path)
|
||||
created = False
|
||||
published = True
|
||||
fsync_directory(Path(path).parent)
|
||||
stored = _read_file(path, gid=gid, mode=mode)
|
||||
if not hashlib.sha256(stored).digest() == hashlib.sha256(payload).digest():
|
||||
raise HostStateError('evidence')
|
||||
except HostStateError:
|
||||
if published:
|
||||
raise HostStateError('uncertain') from None
|
||||
raise
|
||||
except BaseException as error:
|
||||
if published:
|
||||
cancellation = error if not isinstance(error, Exception) else None
|
||||
raise HostStateError(
|
||||
'uncertain', cancellation=cancellation,
|
||||
) from None
|
||||
if not isinstance(error, Exception):
|
||||
raise
|
||||
raise HostStateError('filesystem') from None
|
||||
finally:
|
||||
if descriptor is not None:
|
||||
os.close(descriptor)
|
||||
if created:
|
||||
try:
|
||||
os.unlink(stage)
|
||||
fsync_directory(stage.parent)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _publish_exact(path, payload, *, gid, mode):
|
||||
try:
|
||||
existing = _read_file(path, gid=gid, mode=mode)
|
||||
except FileNotFoundError:
|
||||
_write_stage(path, payload, gid=gid, mode=mode)
|
||||
return payload
|
||||
if existing != payload:
|
||||
raise HostStateError('conflict')
|
||||
return existing
|
||||
|
||||
|
||||
def failed_hold_operation():
|
||||
try:
|
||||
payload = _read_file(
|
||||
HOST_FAILED_HOLD_PATH, gid=HOST_ROOT_GID, mode=0o600,
|
||||
)
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
value = _decode_canonical(payload)
|
||||
if (
|
||||
set(value) != {
|
||||
'schema', 'operation_id', 'action', 'forward_category',
|
||||
'publication_state', 'containment_confirmed',
|
||||
}
|
||||
or value.get('schema') != 1
|
||||
or not isinstance(value.get('operation_id'), str)
|
||||
or value.get('publication_state') not in _PUBLICATION_STATES
|
||||
or type(value.get('containment_confirmed')) is not bool
|
||||
or not isinstance(value.get('forward_category'), str)
|
||||
):
|
||||
raise HostStateError('evidence')
|
||||
return value['operation_id']
|
||||
|
||||
|
||||
class HostOperationState:
|
||||
def __init__(self, request):
|
||||
self.request = decode_request_payload(encode_request_payload(request))
|
||||
self.operation_path = (
|
||||
HOST_OPERATION_DIRECTORY / f'{self.request.operation_id}.json'
|
||||
)
|
||||
self.result_path = HOST_RESULT_DIRECTORY / f'{self.request.operation_id}.json'
|
||||
|
||||
def _phase_record(
|
||||
self, phase, publication_state, *, forward_category=None,
|
||||
safe_detail=None, containment_confirmed=None,
|
||||
):
|
||||
if (
|
||||
phase not in _PHASES
|
||||
or publication_state not in _PUBLICATION_STATES
|
||||
or forward_category is not None
|
||||
and not isinstance(forward_category, str)
|
||||
or safe_detail is not None
|
||||
and not isinstance(safe_detail, str)
|
||||
or containment_confirmed is not None
|
||||
and type(containment_confirmed) is not bool
|
||||
):
|
||||
raise HostStateError('evidence')
|
||||
return {
|
||||
'schema': 1,
|
||||
'operation_id': self.request.operation_id,
|
||||
'action': self.request.action.value,
|
||||
'active_config_sha256': self.request.active_config_sha256,
|
||||
'active_secrets_sha256': self.request.active_secrets_sha256,
|
||||
'candidate_config_sha256': self.request.candidate_config_sha256,
|
||||
'candidate_secrets_sha256': self.request.candidate_secrets_sha256,
|
||||
'phase': phase,
|
||||
'publication_state': publication_state,
|
||||
'forward_category': forward_category,
|
||||
'safe_detail': safe_detail,
|
||||
'containment_confirmed': containment_confirmed,
|
||||
}
|
||||
|
||||
def _read_phase(self):
|
||||
payload = _read_file(self.operation_path, gid=HOST_ROOT_GID, mode=0o600)
|
||||
value = _decode_canonical(payload)
|
||||
if set(value) != set(self._phase_record('prepared', 'original')):
|
||||
raise HostStateError('evidence')
|
||||
expected = self._phase_record(
|
||||
value.get('phase'), value.get('publication_state'),
|
||||
forward_category=value.get('forward_category'),
|
||||
safe_detail=value.get('safe_detail'),
|
||||
containment_confirmed=value.get('containment_confirmed'),
|
||||
)
|
||||
if value != expected:
|
||||
raise HostStateError('evidence')
|
||||
return value
|
||||
|
||||
def initialize(self, publication_state='original'):
|
||||
_require_directory(
|
||||
HOST_STATE_ROOT, gid=HOST_ROOT_GID, mode=HOST_STATE_ROOT_MODE,
|
||||
)
|
||||
_require_directory(
|
||||
HOST_OPERATION_DIRECTORY,
|
||||
gid=HOST_ROOT_GID,
|
||||
mode=HOST_OPERATION_DIRECTORY_MODE,
|
||||
)
|
||||
hold = failed_hold_operation()
|
||||
if hold is not None and hold != self.request.operation_id:
|
||||
raise HostStateError('failed_hold')
|
||||
try:
|
||||
return self._read_phase()
|
||||
except FileNotFoundError:
|
||||
value = self._phase_record('prepared', publication_state)
|
||||
_write_stage(
|
||||
self.operation_path, _canonical(value),
|
||||
gid=HOST_ROOT_GID, mode=0o600,
|
||||
)
|
||||
return value
|
||||
|
||||
def advance(
|
||||
self, expected_phase, next_phase, publication_state, *,
|
||||
forward_category=None, safe_detail=None, containment_confirmed=None,
|
||||
):
|
||||
current = self._read_phase()
|
||||
value = self._phase_record(
|
||||
next_phase, publication_state,
|
||||
forward_category=forward_category,
|
||||
safe_detail=safe_detail,
|
||||
containment_confirmed=containment_confirmed,
|
||||
)
|
||||
if current == value:
|
||||
return value
|
||||
if (
|
||||
current['phase'] != expected_phase
|
||||
or next_phase not in _TRANSITIONS.get(expected_phase, set())
|
||||
):
|
||||
raise HostStateError('state')
|
||||
_write_stage(
|
||||
self.operation_path, _canonical(value),
|
||||
gid=HOST_ROOT_GID, mode=0o600,
|
||||
)
|
||||
return value
|
||||
|
||||
def terminal_result(self):
|
||||
try:
|
||||
payload = _read_file(
|
||||
self.result_path, gid=HOST_RUNTIME_GID, mode=0o640,
|
||||
)
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
value = _decode_canonical(payload)
|
||||
if (
|
||||
set(value) != {
|
||||
'schema', 'operation_id', 'action', 'result', 'safe_category',
|
||||
'safe_detail', 'resulting_identity',
|
||||
}
|
||||
or value.get('schema') != 1
|
||||
or value.get('operation_id') != self.request.operation_id
|
||||
or value.get('action') != self.request.action.value
|
||||
or value.get('result') not in _TERMINAL_RESULTS
|
||||
):
|
||||
raise HostStateError('evidence')
|
||||
return value
|
||||
|
||||
def publish_result(
|
||||
self, result, *, safe_category, safe_detail, resulting_identity,
|
||||
):
|
||||
if result not in _TERMINAL_RESULTS:
|
||||
raise HostStateError('evidence')
|
||||
if result == 'succeeded':
|
||||
if safe_category is not None or safe_detail is not None:
|
||||
raise HostStateError('evidence')
|
||||
elif not isinstance(safe_category, str) or not isinstance(safe_detail, str):
|
||||
raise HostStateError('evidence')
|
||||
if resulting_identity is not None and (
|
||||
not isinstance(resulting_identity, dict)
|
||||
or set(resulting_identity) != {
|
||||
'active_config_sha256', 'active_secrets_sha256',
|
||||
}
|
||||
or any(
|
||||
not isinstance(value, str) or len(value) != 64
|
||||
for value in resulting_identity.values()
|
||||
)
|
||||
):
|
||||
raise HostStateError('evidence')
|
||||
value = {
|
||||
'schema': 1,
|
||||
'operation_id': self.request.operation_id,
|
||||
'action': self.request.action.value,
|
||||
'result': result,
|
||||
'safe_category': safe_category,
|
||||
'safe_detail': safe_detail,
|
||||
'resulting_identity': resulting_identity,
|
||||
}
|
||||
_require_directory(
|
||||
HOST_RESULT_DIRECTORY,
|
||||
gid=HOST_RUNTIME_GID,
|
||||
mode=HOST_RESULT_DIRECTORY_MODE,
|
||||
)
|
||||
_publish_exact(
|
||||
self.result_path, _canonical(value), gid=HOST_RUNTIME_GID, mode=0o640,
|
||||
)
|
||||
return value
|
||||
|
||||
def publish_failed_hold(
|
||||
self, *, forward_category, publication_state, containment_confirmed,
|
||||
):
|
||||
marker = {
|
||||
'schema': 1,
|
||||
'operation_id': self.request.operation_id,
|
||||
'action': self.request.action.value,
|
||||
'forward_category': str(forward_category),
|
||||
'publication_state': publication_state,
|
||||
'containment_confirmed': bool(containment_confirmed),
|
||||
}
|
||||
_publish_exact(
|
||||
HOST_FAILED_HOLD_PATH, _canonical(marker),
|
||||
gid=HOST_ROOT_GID, mode=0o600,
|
||||
)
|
||||
return marker
|
||||
+455
@@ -0,0 +1,455 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
if not sys.dont_write_bytecode:
|
||||
raise RuntimeError('janitor could not disable bytecode writes')
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import stat
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from lifecycle_authority import require_active_supervisor_child
|
||||
from paths import apply_path_config
|
||||
from process_identity import exact_process_identity_state
|
||||
from runtime_security import (
|
||||
atomic_write_private_json,
|
||||
canonical_path,
|
||||
fsync_directory,
|
||||
is_reparse_point,
|
||||
private_directory_ready,
|
||||
private_file_ready,
|
||||
read_private_json,
|
||||
reject_reparse_components,
|
||||
require_private_directory,
|
||||
)
|
||||
|
||||
|
||||
MARKER_NAME = '.scanner-owner.json'
|
||||
MARKER_SCHEMA = 2
|
||||
APPROVED_LAYOUTS = (
|
||||
('work', '', ('trufflehog-', 'trufflehog-run-', 'trufflehog-probe-', 'hg-run-', 'docker-config-', 'docker-layer-', 'tmp-', 'worker-assignment-')),
|
||||
('work', 'docker-config', ('docker-config-',)),
|
||||
('work', 'hg', ('hg-run-',)),
|
||||
('work', 'tmp', ('trufflehog-', 'trufflehog-run-', 'hg-run-', 'docker-config-', 'docker-layer-', 'tmp-')),
|
||||
('work', os.path.join('tmp', 'docker-config'), ('docker-config-',)),
|
||||
('work', 'abandoned', ('worker-assignment-',)),
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class JanitorBudget:
|
||||
max_candidates: int = 50
|
||||
max_entries: int = 10000
|
||||
max_bytes: int = 1024 * 1024 * 1024
|
||||
max_seconds: float = 30.0
|
||||
max_depth: int = 64
|
||||
max_enumerated: int = 1000
|
||||
candidates: int = 0
|
||||
entries: int = 0
|
||||
bytes: int = 0
|
||||
started_at: float = 0.0
|
||||
exhausted: bool = False
|
||||
enumerated: int = 0
|
||||
|
||||
def __post_init__(self):
|
||||
self.started_at = self.started_at or time.monotonic()
|
||||
|
||||
def consume(self, size=0, candidate=False, depth=0, allow_oversized=False):
|
||||
if candidate:
|
||||
self.candidates += 1
|
||||
else:
|
||||
self.entries += 1
|
||||
self.bytes += max(0, int(size or 0))
|
||||
candidate_limit = self.candidates > self.max_candidates
|
||||
entry_limit = self.entries > self.max_entries
|
||||
byte_limit = self.bytes > self.max_bytes
|
||||
depth_limit = depth > self.max_depth
|
||||
time_limit = time.monotonic() - self.started_at >= self.max_seconds
|
||||
self.exhausted = bool(
|
||||
candidate_limit or entry_limit or byte_limit or depth_limit or time_limit
|
||||
)
|
||||
# Unlinking one regular file is bounded metadata work regardless of its
|
||||
# payload size. Directory traversal remains bounded by the other limits.
|
||||
oversized_progress = bool(
|
||||
allow_oversized and byte_limit
|
||||
and not (candidate_limit or entry_limit or depth_limit or time_limit)
|
||||
)
|
||||
return not self.exhausted or oversized_progress
|
||||
|
||||
def consume_enumerated(self):
|
||||
self.enumerated += 1
|
||||
self.exhausted = bool(
|
||||
self.enumerated > self.max_enumerated
|
||||
or time.monotonic() - self.started_at >= self.max_seconds
|
||||
)
|
||||
return not self.exhausted
|
||||
|
||||
|
||||
def _marker_relative_path(root, path):
|
||||
relative = os.path.relpath(path, root)
|
||||
if relative == '.' or relative.startswith('..' + os.sep) or os.path.isabs(relative):
|
||||
raise ValueError('candidate escapes the approved janitor root')
|
||||
return relative.replace(os.sep, '/')
|
||||
|
||||
|
||||
def validate_marker(root, path, marker, allowed_executables, minimum_age_sec, now=None):
|
||||
if marker.get('schema') != MARKER_SCHEMA or marker.get('root_kind') != 'work':
|
||||
return False, 'unsupported_marker'
|
||||
try:
|
||||
if marker.get('relative_path') != _marker_relative_path(root, path):
|
||||
return False, 'path_mismatch'
|
||||
created = datetime.fromisoformat(str(marker.get('created_at') or '').replace('Z', '+00:00'))
|
||||
if created.tzinfo is None:
|
||||
created = created.replace(tzinfo=timezone.utc)
|
||||
now_value = now or datetime.now(timezone.utc)
|
||||
if (now_value - created).total_seconds() < max(0, float(minimum_age_sec)):
|
||||
return False, 'too_young'
|
||||
except (TypeError, ValueError):
|
||||
return False, 'invalid_time_or_path'
|
||||
|
||||
allowed = {canonical_path(value) for value in allowed_executables if value}
|
||||
identities = {}
|
||||
states = {}
|
||||
prefixes = ('owner', 'parent')
|
||||
if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')):
|
||||
prefixes += ('child',)
|
||||
for prefix in prefixes:
|
||||
if prefix == 'child' and any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')):
|
||||
return False, 'child_identity_invalid'
|
||||
identity = {
|
||||
'pid': marker.get(f'{prefix}_pid'),
|
||||
'creation_time': marker.get(f'{prefix}_creation_time'),
|
||||
'executable': marker.get(f'{prefix}_executable'),
|
||||
}
|
||||
try:
|
||||
executable = canonical_path(identity['executable'])
|
||||
except (OSError, TypeError, ValueError):
|
||||
return False, f'{prefix}_identity_invalid'
|
||||
if executable not in allowed:
|
||||
return False, f'{prefix}_executable_unapproved'
|
||||
identity['executable'] = executable
|
||||
identities[prefix] = identity
|
||||
states[prefix] = exact_process_identity_state(
|
||||
identity['pid'], identity['creation_time'], identity['executable'],
|
||||
)
|
||||
if 'child' in states and states['child'] != 'dead':
|
||||
return False, 'child_live_or_unknown'
|
||||
if states['owner'] != 'dead':
|
||||
return False, 'owner_live_or_unknown'
|
||||
same_identity = all(
|
||||
identities['owner'].get(field) == identities['parent'].get(field)
|
||||
for field in ('pid', 'creation_time', 'executable')
|
||||
)
|
||||
if same_identity and states['parent'] != 'dead':
|
||||
return False, 'parent_owned_live_or_unknown'
|
||||
return True, 'eligible'
|
||||
|
||||
|
||||
def bounded_remove_tree(path, budget, marker_name=MARKER_NAME):
|
||||
"""Delete without recursion or reparse traversal; leave the root marker last."""
|
||||
path = os.path.abspath(path)
|
||||
reject_reparse_components(path)
|
||||
if is_reparse_point(path) or not private_directory_ready(path):
|
||||
raise OSError(f'janitor candidate is not an exact private directory: {path}')
|
||||
marker_path = os.path.join(path, marker_name)
|
||||
stack = []
|
||||
root_iterator = os.scandir(path)
|
||||
stack.append((path, root_iterator, 0))
|
||||
try:
|
||||
while stack:
|
||||
if time.monotonic() - budget.started_at >= budget.max_seconds:
|
||||
budget.exhausted = True
|
||||
return False
|
||||
directory, iterator, depth = stack[-1]
|
||||
try:
|
||||
entry = next(iterator)
|
||||
except StopIteration:
|
||||
iterator.close()
|
||||
stack.pop()
|
||||
if directory == path:
|
||||
if os.path.lexists(marker_path):
|
||||
details = os.stat(marker_path, follow_symlinks=False)
|
||||
# The verified owner marker is removed only after every payload
|
||||
# entry is gone, so finishing the empty root must make progress
|
||||
# even when one oversized payload exhausted this pass's budget.
|
||||
budget.consume(details.st_size, depth=depth + 1)
|
||||
if not private_file_ready(marker_path):
|
||||
raise OSError('janitor owner marker lost its private identity')
|
||||
os.remove(marker_path)
|
||||
os.rmdir(directory)
|
||||
fsync_directory(os.path.dirname(directory))
|
||||
return True
|
||||
os.rmdir(directory)
|
||||
continue
|
||||
if entry.path == marker_path:
|
||||
continue
|
||||
details = entry.stat(follow_symlinks=False)
|
||||
is_regular = stat.S_ISREG(details.st_mode)
|
||||
if not budget.consume(
|
||||
details.st_size, depth=depth + 1, allow_oversized=is_regular,
|
||||
):
|
||||
return False
|
||||
if entry.is_symlink() or is_reparse_point(entry.path):
|
||||
raise OSError(f'janitor candidate contains a link or reparse point: {entry.path}')
|
||||
if stat.S_ISDIR(details.st_mode):
|
||||
child_iterator = os.scandir(entry.path)
|
||||
stack.append((entry.path, child_iterator, depth + 1))
|
||||
elif is_regular:
|
||||
os.chmod(entry.path, stat.S_IWRITE | stat.S_IREAD)
|
||||
os.remove(entry.path)
|
||||
else:
|
||||
raise OSError(f'janitor candidate contains an unsupported entry: {entry.path}')
|
||||
finally:
|
||||
for _, iterator, _ in stack:
|
||||
iterator.close()
|
||||
return False
|
||||
|
||||
|
||||
def _layout_cursor_name(root, relative_parent, prefixes):
|
||||
identity = '|'.join((
|
||||
canonical_path(root), str(relative_parent).replace(os.sep, '/'), ','.join(prefixes),
|
||||
))
|
||||
return 'layout:' + hashlib.sha256(identity.encode('utf-8')).hexdigest()
|
||||
|
||||
|
||||
class JanitorCursorStore:
|
||||
SCHEMA = 1
|
||||
|
||||
def __init__(self, path, root):
|
||||
self.path = os.path.abspath(path)
|
||||
self.root_hash = hashlib.sha256(canonical_path(root).encode('utf-8')).hexdigest()
|
||||
self.dirty = False
|
||||
self.state = {
|
||||
'schema': self.SCHEMA,
|
||||
'root_sha256': self.root_hash,
|
||||
'next_layout': 0,
|
||||
'layouts': {},
|
||||
}
|
||||
self.iterators = {}
|
||||
self.seeking = {}
|
||||
if os.path.lexists(self.path):
|
||||
loaded = read_private_json(self.path, max_bytes=256 * 1024)
|
||||
if (
|
||||
not isinstance(loaded, dict)
|
||||
or loaded.get('schema') != self.SCHEMA
|
||||
or loaded.get('root_sha256') != self.root_hash
|
||||
or not isinstance(loaded.get('layouts'), dict)
|
||||
):
|
||||
raise RuntimeError('janitor cursor authority is invalid')
|
||||
self.state = loaded
|
||||
else:
|
||||
require_private_directory(os.path.dirname(self.path), create=True)
|
||||
atomic_write_private_json(self.path, self.state)
|
||||
|
||||
def _save(self):
|
||||
atomic_write_private_json(self.path, self.state)
|
||||
self.dirty = False
|
||||
|
||||
def _mark_dirty(self):
|
||||
self.dirty = True
|
||||
|
||||
def flush(self):
|
||||
if self.dirty:
|
||||
self._save()
|
||||
|
||||
def next_layout(self, count):
|
||||
index = int(self.state.get('next_layout') or 0) % max(1, int(count))
|
||||
self.state['next_layout'] = (index + 1) % max(1, int(count))
|
||||
self._mark_dirty()
|
||||
return index
|
||||
|
||||
def next_entry(self, layout_name, parent):
|
||||
iterator = self.iterators.get(layout_name)
|
||||
if iterator is None:
|
||||
iterator = os.scandir(parent)
|
||||
self.iterators[layout_name] = iterator
|
||||
last_name = str((self.state['layouts'].get(layout_name) or {}).get('last_name') or '')
|
||||
self.seeking[layout_name] = bool(last_name)
|
||||
try:
|
||||
entry = next(iterator)
|
||||
except StopIteration:
|
||||
iterator.close()
|
||||
self.iterators.pop(layout_name, None)
|
||||
self.seeking.pop(layout_name, None)
|
||||
current = self.state['layouts'].setdefault(layout_name, {})
|
||||
current['last_name'] = ''
|
||||
current['wrap_count'] = int(current.get('wrap_count') or 0) + 1
|
||||
self._mark_dirty()
|
||||
return None, False
|
||||
|
||||
current = self.state['layouts'].setdefault(layout_name, {'last_name': '', 'wrap_count': 0})
|
||||
target = str(current.get('last_name') or '')
|
||||
if self.seeking.get(layout_name):
|
||||
if entry.name == target:
|
||||
self.seeking[layout_name] = False
|
||||
return entry, True
|
||||
current['last_name'] = entry.name
|
||||
self._mark_dirty()
|
||||
return entry, False
|
||||
|
||||
def close(self):
|
||||
for iterator in self.iterators.values():
|
||||
iterator.close()
|
||||
self.iterators.clear()
|
||||
|
||||
|
||||
class _MemoryCursorStore(JanitorCursorStore):
|
||||
def __init__(self):
|
||||
self.path = ''
|
||||
self.root_hash = ''
|
||||
self.dirty = False
|
||||
self.state = {'schema': 1, 'root_sha256': '', 'next_layout': 0, 'layouts': {}}
|
||||
self.iterators = {}
|
||||
self.seeking = {}
|
||||
|
||||
def _save(self):
|
||||
self.dirty = False
|
||||
return None
|
||||
|
||||
|
||||
def iter_candidates(root, budget, cursor_store):
|
||||
layouts = list(APPROVED_LAYOUTS)
|
||||
completed_layouts = set()
|
||||
while not budget.exhausted and len(completed_layouts) < len(layouts):
|
||||
if (
|
||||
budget.enumerated >= budget.max_enumerated
|
||||
or time.monotonic() - budget.started_at >= budget.max_seconds
|
||||
):
|
||||
budget.exhausted = True
|
||||
return
|
||||
index = cursor_store.next_layout(len(layouts))
|
||||
root_kind, relative_parent, prefixes = layouts[index]
|
||||
layout_name = _layout_cursor_name(root, relative_parent, prefixes)
|
||||
if layout_name in completed_layouts:
|
||||
continue
|
||||
parent = os.path.join(root, relative_parent) if relative_parent else root
|
||||
try:
|
||||
if not os.path.isdir(parent) or is_reparse_point(parent):
|
||||
completed_layouts.add(layout_name)
|
||||
continue
|
||||
entry, seeking = cursor_store.next_entry(layout_name, parent)
|
||||
except OSError:
|
||||
completed_layouts.add(layout_name)
|
||||
continue
|
||||
if entry is None:
|
||||
completed_layouts.add(layout_name)
|
||||
continue
|
||||
if not budget.consume_enumerated():
|
||||
return
|
||||
if seeking:
|
||||
continue
|
||||
if not entry.name.startswith(prefixes) or not entry.is_dir(follow_symlinks=False):
|
||||
continue
|
||||
if not budget.consume(candidate=True):
|
||||
return
|
||||
yield root_kind, layout_name, entry.name, entry.path
|
||||
|
||||
|
||||
def run_janitor_pass(
|
||||
root, allowed_executables, minimum_age_sec=7200, budget=None,
|
||||
cursor_store=None, excluded_relative_paths=(),
|
||||
):
|
||||
budget = budget or JanitorBudget()
|
||||
root = require_private_directory(root, create=False)
|
||||
cursor_store = cursor_store or _MemoryCursorStore()
|
||||
excluded = set()
|
||||
for value in excluded_relative_paths:
|
||||
relative = str(value or '').replace('\\', '/')
|
||||
if (
|
||||
not relative or relative.startswith('/') or relative.endswith('/')
|
||||
or any(part in ('', '.', '..') for part in relative.split('/'))
|
||||
):
|
||||
raise ValueError('janitor exclusion path is invalid')
|
||||
excluded.add(relative)
|
||||
if len(excluded) > 4096:
|
||||
raise ValueError('janitor exclusion set exceeds its bound')
|
||||
report = {'considered': 0, 'removed': 0, 'retained': 0, 'errors': 0, 'exhausted': False}
|
||||
try:
|
||||
for _, _, _, path in iter_candidates(root, budget, cursor_store):
|
||||
report['considered'] += 1
|
||||
if _marker_relative_path(root, path) in excluded:
|
||||
report['retained'] += 1
|
||||
continue
|
||||
marker_path = os.path.join(path, MARKER_NAME)
|
||||
try:
|
||||
if not private_file_ready(marker_path):
|
||||
report['retained'] += 1
|
||||
continue
|
||||
marker = read_private_json(marker_path)
|
||||
eligible, _ = validate_marker(
|
||||
root, path, marker, allowed_executables, minimum_age_sec,
|
||||
)
|
||||
if not eligible:
|
||||
report['retained'] += 1
|
||||
continue
|
||||
if bounded_remove_tree(path, budget):
|
||||
report['removed'] += 1
|
||||
else:
|
||||
report['retained'] += 1
|
||||
except (OSError, ValueError):
|
||||
report['errors'] += 1
|
||||
if budget.exhausted:
|
||||
break
|
||||
finally:
|
||||
cursor_store.flush()
|
||||
report['exhausted'] = budget.exhausted
|
||||
report['entries'] = budget.entries
|
||||
report['bytes'] = budget.bytes
|
||||
report['enumerated'] = budget.enumerated
|
||||
return report
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description='Bounded scanner work-directory janitor')
|
||||
parser.add_argument('--config', required=True)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
metadata = require_active_supervisor_child(child_kind='janitor', require_dsn=False)
|
||||
args = parse_args()
|
||||
import yaml
|
||||
|
||||
with open(args.config, 'r', encoding='utf-8') as handle:
|
||||
config = apply_path_config(yaml.safe_load(handle) or {}, args.config)
|
||||
global_config = config.get('global') or {}
|
||||
janitor_config = ((config.get('supervisor') or {}).get('janitor') or {})
|
||||
manifest = metadata.get('code_manifest') or {}
|
||||
allowed = [sys.executable]
|
||||
allowed.extend(
|
||||
item.get('path') for item in (manifest.get('executables') or {}).values()
|
||||
if isinstance(item, dict) and item.get('path')
|
||||
)
|
||||
interval = max(5.0, float(janitor_config.get('interval_sec', 60) or 60))
|
||||
cursor_store = JanitorCursorStore(
|
||||
os.path.join(global_config['state_dir'], 'janitor.cursor.json'),
|
||||
global_config['work_dir'],
|
||||
)
|
||||
try:
|
||||
while True:
|
||||
budget = JanitorBudget(
|
||||
max_candidates=max(1, int(janitor_config.get('max_candidates', 50) or 50)),
|
||||
max_entries=max(1, int(janitor_config.get('max_entries', 10000) or 10000)),
|
||||
max_bytes=max(1, int(janitor_config.get('max_bytes', 1024 * 1024 * 1024) or 1)),
|
||||
max_seconds=max(0.1, float(janitor_config.get('max_seconds', 30) or 30)),
|
||||
max_depth=max(1, int(janitor_config.get('max_depth', 64) or 64)),
|
||||
max_enumerated=max(1, int(janitor_config.get('max_enumerated', 1000) or 1000)),
|
||||
)
|
||||
report = run_janitor_pass(
|
||||
global_config['work_dir'], allowed,
|
||||
minimum_age_sec=max(0, int(janitor_config.get('minimum_age_sec', 7200) or 0)),
|
||||
budget=budget, cursor_store=cursor_store,
|
||||
)
|
||||
print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True)
|
||||
time.sleep(interval)
|
||||
finally:
|
||||
cursor_store.close()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,736 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
if not sys.dont_write_bytecode:
|
||||
raise RuntimeError('JSONL projector could not disable bytecode writes')
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
|
||||
from lifecycle_authority import require_active_supervisor_child
|
||||
from paths import apply_path_config
|
||||
from process_identity import current_process_identity
|
||||
from runtime_security import (
|
||||
PrivateFileLock,
|
||||
PrivatePathState,
|
||||
durable_publish,
|
||||
durable_unlink,
|
||||
harden_private_file,
|
||||
inspect_private_relative_path,
|
||||
private_file_ready,
|
||||
reject_reparse_components,
|
||||
require_private_directory,
|
||||
)
|
||||
from scanner_db import ScannerDB
|
||||
|
||||
|
||||
STREAM_MASKS = {'scan_results': 1, 'found_secrets': 2, 'scan_errors': 4}
|
||||
MAX_SERIALIZED_EVENT_BYTES = 192 * 1024 * 1024
|
||||
MAX_TAIL_QUARANTINE_BYTES = MAX_SERIALIZED_EVENT_BYTES
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SerializedStream:
|
||||
stream_name: str
|
||||
path: str
|
||||
byte_length: int
|
||||
payload_sha256: str
|
||||
record_count: int
|
||||
artifact_id: int = 0
|
||||
|
||||
|
||||
class _HashedWriter:
|
||||
def __init__(self, handle, max_bytes):
|
||||
self.handle = handle
|
||||
self.max_bytes = int(max_bytes)
|
||||
self.digest = hashlib.sha256()
|
||||
self.bytes_written = 0
|
||||
|
||||
def write(self, payload):
|
||||
payload = payload.encode('utf-8') if isinstance(payload, str) else bytes(payload)
|
||||
if self.bytes_written + len(payload) > self.max_bytes:
|
||||
raise ValueError('projection serialization exceeds its event byte bound')
|
||||
self.handle.write(payload)
|
||||
self.digest.update(payload)
|
||||
self.bytes_written += len(payload)
|
||||
|
||||
|
||||
def _write_json_line(writer, value):
|
||||
encoder = json.JSONEncoder(
|
||||
ensure_ascii=False, sort_keys=True, separators=(',', ':'), default=str,
|
||||
)
|
||||
for chunk in encoder.iterencode(value):
|
||||
writer.write(chunk)
|
||||
writer.write(b'\n')
|
||||
|
||||
|
||||
def _write_json_value(writer, value):
|
||||
encoder = json.JSONEncoder(
|
||||
ensure_ascii=False, sort_keys=True, separators=(',', ':'), default=str,
|
||||
)
|
||||
for chunk in encoder.iterencode(value):
|
||||
writer.write(chunk)
|
||||
|
||||
|
||||
def _write_json_array(writer, values):
|
||||
writer.write(b'[')
|
||||
first = True
|
||||
for value in values:
|
||||
if not first:
|
||||
writer.write(b',')
|
||||
_write_json_value(writer, value)
|
||||
first = False
|
||||
writer.write(b']')
|
||||
|
||||
|
||||
def _write_scan_result(writer, header, findings, errors):
|
||||
values = dict(header)
|
||||
keys = sorted(set(values) | {'findings', 'errors'})
|
||||
writer.write(b'{')
|
||||
for index, key in enumerate(keys):
|
||||
if index:
|
||||
writer.write(b',')
|
||||
_write_json_value(writer, key)
|
||||
writer.write(b':')
|
||||
if key == 'findings':
|
||||
_write_json_array(writer, findings)
|
||||
elif key == 'errors':
|
||||
_write_json_array(writer, errors)
|
||||
else:
|
||||
_write_json_value(writer, values[key])
|
||||
writer.write(b'}\n')
|
||||
|
||||
|
||||
class JsonlProjector:
|
||||
def __init__(
|
||||
self, db, results_dir, supervisor_instance_id, lease_seconds=300,
|
||||
fault=None, keycheck_dir=None, quarantine_max_items=10000,
|
||||
quarantine_max_bytes=1024 * 1024 * 1024, artifact_tracking=True,
|
||||
projection_max_bytes=2 * 1024 * 1024 * 1024,
|
||||
):
|
||||
self.db = db
|
||||
self.results_dir = require_private_directory(results_dir, create=False)
|
||||
self.keycheck_dir = require_private_directory(
|
||||
keycheck_dir or os.path.join(os.path.dirname(self.results_dir), 'keychecks'),
|
||||
create=False,
|
||||
)
|
||||
self.supervisor_instance_id = str(supervisor_instance_id)
|
||||
self.lease_seconds = max(30, int(lease_seconds))
|
||||
self.fault = fault
|
||||
self.lease = None
|
||||
self.file_lock = None
|
||||
self.temp_dir = require_private_directory(
|
||||
os.path.join(self.results_dir, '.projection-tmp'), create=True,
|
||||
)
|
||||
self.quarantine_dir = require_private_directory(
|
||||
os.path.join(self.results_dir, '.projection-quarantine'), create=True,
|
||||
)
|
||||
self.quarantine_max_items = max(0, int(quarantine_max_items))
|
||||
self.quarantine_max_bytes = max(0, int(quarantine_max_bytes))
|
||||
self.projection_max_bytes = max(0, int(projection_max_bytes))
|
||||
self.artifact_tracking = bool(artifact_tracking)
|
||||
|
||||
def _inject(self, stage, value=None):
|
||||
if self.fault is not None:
|
||||
self.fault(stage, value)
|
||||
|
||||
def start(self):
|
||||
self.db.require_runtime_safety_schema()
|
||||
self.db.require_final_cutover()
|
||||
self.file_lock = PrivateFileLock(
|
||||
os.path.join(self.results_dir, '.jsonl-projector.lock')
|
||||
).acquire()
|
||||
self.lease = self.db.acquire_pipeline_lease(
|
||||
'jsonl_projector', self.supervisor_instance_id, current_process_identity(),
|
||||
lease_seconds=self.lease_seconds, initial_state='starting',
|
||||
)
|
||||
if not self.lease:
|
||||
self.file_lock.release()
|
||||
self.file_lock = None
|
||||
raise RuntimeError('another JSONL projector owns the singleton advisory lock')
|
||||
if self.artifact_tracking:
|
||||
self.reconcile_terminal_temps()
|
||||
self.recover_rotations()
|
||||
if not self.heartbeat():
|
||||
raise RuntimeError('JSONL projector ready lease publication failed')
|
||||
return self
|
||||
|
||||
def heartbeat(self, error=''):
|
||||
return self.db.heartbeat_pipeline_lease(
|
||||
'jsonl_projector', self.lease['generation'], self.lease['lease_token'],
|
||||
lease_seconds=self.lease_seconds, state='ready', error=error,
|
||||
)
|
||||
|
||||
def _rollback_database(self):
|
||||
connection = getattr(self.db, 'conn', None)
|
||||
if connection is not None:
|
||||
try:
|
||||
connection.rollback()
|
||||
except BaseException:
|
||||
pass
|
||||
|
||||
def stop(self, error=''):
|
||||
try:
|
||||
if self.lease:
|
||||
self._rollback_database()
|
||||
try:
|
||||
self.db.release_pipeline_lease(
|
||||
'jsonl_projector', self.lease['generation'], self.lease['lease_token'],
|
||||
state='failed' if error else 'released', error=error,
|
||||
)
|
||||
except BaseException:
|
||||
self._rollback_database()
|
||||
raise
|
||||
self.lease = None
|
||||
finally:
|
||||
if self.file_lock:
|
||||
self.file_lock.release()
|
||||
self.file_lock = None
|
||||
|
||||
def _results_path(self, relative, stream_name=''):
|
||||
root = self.keycheck_dir if str(stream_name).startswith('keycheck:') else self.results_dir
|
||||
path = os.path.abspath(os.path.join(root, str(relative).replace('/', os.sep)))
|
||||
if os.path.commonpath((root, path)) != root or path == root:
|
||||
raise ValueError('projection path escapes results_dir')
|
||||
reject_reparse_components(os.path.dirname(path))
|
||||
return path
|
||||
|
||||
@staticmethod
|
||||
def _segment_relative(base_relative, generation):
|
||||
base, extension = os.path.splitext(base_relative)
|
||||
return f'{base}.g{int(generation):06d}{extension}'
|
||||
|
||||
def _create_private_empty(self, path):
|
||||
if os.path.exists(path):
|
||||
if not private_file_ready(path):
|
||||
raise OSError(f'projection active file is not private: {path}')
|
||||
return
|
||||
descriptor = os.open(
|
||||
path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), 0o600,
|
||||
)
|
||||
os.close(descriptor)
|
||||
harden_private_file(path)
|
||||
|
||||
def _prepared_path(self, job, stream_name):
|
||||
safe_stream_name = re.sub(r'[^A-Za-z0-9_.-]+', '_', stream_name)
|
||||
event_hash = str(job['event_hash'] or '').lower()
|
||||
if not re.fullmatch(r'[a-f0-9]{64}', event_hash):
|
||||
raise ValueError('projection job event hash is not a canonical SHA-256 identity')
|
||||
path = os.path.join(
|
||||
self.temp_dir,
|
||||
f'job-{int(job["id"])}-{safe_stream_name}-{event_hash}.prepared',
|
||||
)
|
||||
relative = os.path.relpath(path, self.results_dir).replace(os.sep, '/')
|
||||
artifact_id = 0
|
||||
if self.artifact_tracking:
|
||||
artifact_id = self.db.register_pipeline_artifact(
|
||||
'jsonl_projector', 'prepared_stream', job['id'], stream_name,
|
||||
relative, state='expected', byte_count=int(job['capacity_bytes']),
|
||||
)
|
||||
if os.path.lexists(path):
|
||||
if not private_file_ready(path):
|
||||
raise OSError(f'projection prepared file is not private: {path}')
|
||||
else:
|
||||
descriptor = os.open(
|
||||
path,
|
||||
os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0),
|
||||
0o600,
|
||||
)
|
||||
os.close(descriptor)
|
||||
harden_private_file(path)
|
||||
return path, artifact_id
|
||||
|
||||
def _delete_registered_artifact(self, path, artifact_id):
|
||||
relative = os.path.relpath(path, self.results_dir).replace(os.sep, '/')
|
||||
inspection = inspect_private_relative_path(self.results_dir, relative)
|
||||
if inspection.state == PrivatePathState.UNKNOWN:
|
||||
raise OSError('projection artifact state is unknown during cleanup')
|
||||
if inspection.state == PrivatePathState.PRESENT:
|
||||
durable_unlink(inspection.path)
|
||||
inspection = inspect_private_relative_path(self.results_dir, relative)
|
||||
if inspection.state != PrivatePathState.ABSENT:
|
||||
raise OSError('projection artifact unlink was not confirmed')
|
||||
if artifact_id:
|
||||
self.db.mark_pipeline_artifact_deleted(artifact_id)
|
||||
|
||||
def reconcile_terminal_temps(self, max_pages=100):
|
||||
if not self.artifact_tracking or not hasattr(self.db, 'projection_terminal_temp_artifacts'):
|
||||
return
|
||||
for _ in range(max(1, int(max_pages))):
|
||||
rows = self.db.projection_terminal_temp_artifacts(100)
|
||||
if not rows:
|
||||
return
|
||||
for row in rows:
|
||||
path = self._results_path(row['relative_path'])
|
||||
self._delete_registered_artifact(path, row['id'])
|
||||
if len(rows) < 100:
|
||||
return
|
||||
return
|
||||
|
||||
def recover_rotations(self):
|
||||
for rotation in self.db.pending_projection_rotations(100):
|
||||
active = self._results_path(rotation['base_relative_path'], rotation['stream_name'])
|
||||
segment = self._results_path(rotation['segment_relative_path'], rotation['stream_name'])
|
||||
require_private_directory(os.path.dirname(active), create=True)
|
||||
active_exists = os.path.isfile(active)
|
||||
segment_exists = os.path.isfile(segment)
|
||||
if active_exists and segment_exists:
|
||||
if (
|
||||
os.path.getsize(active) != 0
|
||||
or os.path.getsize(segment) != int(rotation['source_bytes'])
|
||||
):
|
||||
raise RuntimeError('projection rotation has conflicting active and immutable names')
|
||||
elif active_exists:
|
||||
if os.path.getsize(active) != int(rotation['source_bytes']):
|
||||
raise RuntimeError('projection rotation active size changed')
|
||||
durable_publish(active, segment)
|
||||
elif not segment_exists:
|
||||
raise RuntimeError('projection rotation lost both exact source names')
|
||||
self._create_private_empty(active)
|
||||
if not self.db.complete_projection_rotation(rotation['id']):
|
||||
raise RuntimeError('projection rotation completion fence failed')
|
||||
|
||||
def _serialize(self, job):
|
||||
if job['job_kind'] == 'scan_event':
|
||||
scan = self.db.projection_scan_header(
|
||||
job['target_scan_id'], max_bytes=self.projection_max_bytes,
|
||||
)
|
||||
if not isinstance(scan, dict) or not isinstance(scan.get('result'), dict):
|
||||
raise ValueError('authoritative scan result cannot be reconstructed')
|
||||
result = scan['result']
|
||||
normalized = scan['storage'] == 'normalized_v2'
|
||||
stream_specs = [
|
||||
(name, mask) for name, mask in STREAM_MASKS.items()
|
||||
if int(job['required_stream_mask']) & mask
|
||||
]
|
||||
elif job['job_kind'] == 'keycheck_event':
|
||||
result = self.db.keycheck_result_for_projection(job['keycheck_result_id'])
|
||||
if not isinstance(result, dict):
|
||||
raise ValueError('authoritative keycheck result cannot be reconstructed')
|
||||
service = str(result['service'])
|
||||
stream_specs = [
|
||||
(name, mask) for name, mask in (
|
||||
(f'keycheck:{service}:results', 8),
|
||||
(f'keycheck:{service}:status', 16),
|
||||
) if int(job['required_stream_mask']) & mask
|
||||
]
|
||||
else:
|
||||
raise ValueError(f'unsupported projection job kind: {job["job_kind"]}')
|
||||
streams = []
|
||||
prepared_paths = []
|
||||
try:
|
||||
for stream_name, mask in stream_specs:
|
||||
path, artifact_id = self._prepared_path(job, stream_name)
|
||||
prepared_paths.append((path, artifact_id))
|
||||
count = 0
|
||||
with open(path, 'wb', buffering=0) as handle:
|
||||
writer = _HashedWriter(handle, self.projection_max_bytes)
|
||||
if stream_name == 'scan_results':
|
||||
findings = (
|
||||
self.db.iter_projection_findings(job['target_scan_id'])
|
||||
if normalized else iter(result.get('findings') or ())
|
||||
)
|
||||
errors = (
|
||||
self.db.iter_projection_errors(job['target_scan_id'])
|
||||
if normalized else iter(result.get('errors') or ())
|
||||
)
|
||||
_write_scan_result(writer, result, findings, errors)
|
||||
count = 1
|
||||
elif stream_name == 'found_secrets':
|
||||
findings = (
|
||||
self.db.iter_projection_findings(job['target_scan_id'])
|
||||
if normalized else iter(result.get('findings') or ())
|
||||
)
|
||||
for finding in findings:
|
||||
_write_json_line(writer, finding)
|
||||
count += 1
|
||||
elif stream_name == 'scan_errors':
|
||||
timestamp = result.get('timestamp') or ''
|
||||
scan_type = result.get('scan_type') or ''
|
||||
target = result.get('target') or ''
|
||||
event_id = result.get('scan_event_id') or job['event_id']
|
||||
errors = (
|
||||
self.db.iter_projection_errors(job['target_scan_id'])
|
||||
if normalized else iter(result.get('errors') or ())
|
||||
)
|
||||
for index, error in enumerate(errors, 1):
|
||||
row_id = hashlib.sha256(
|
||||
f'{event_id}|{index}'.encode('utf-8')
|
||||
).hexdigest()
|
||||
writer.write(
|
||||
f'{row_id}\t{timestamp}\t{scan_type}\t{target}\t{error}\n'
|
||||
)
|
||||
count += 1
|
||||
elif stream_name.endswith(':results'):
|
||||
payload = {
|
||||
'event_id': result['event_id'],
|
||||
'service': result['service'],
|
||||
'status': result['status'],
|
||||
'status_group': result['status_group'],
|
||||
'checked_at': result['checked_at'],
|
||||
'key_hash': result['key_hash'],
|
||||
'secret_hash': result['secret_hash'],
|
||||
'key_masked': result['key_masked'],
|
||||
'finding_uid': result['finding_uid'],
|
||||
'detector': result['detector_name'],
|
||||
'source': result['source'],
|
||||
'message': result['message'],
|
||||
'metadata': json.loads(result['metadata_json'] or '{}'),
|
||||
'result_source': result['result_source'],
|
||||
}
|
||||
_write_json_line(writer, payload)
|
||||
count = 1
|
||||
else:
|
||||
secret = result.get('credential_secret_text') or result.get('credential_secret_json') or ''
|
||||
message = str(result.get('message') or '').replace('\r', ' ').replace('\n', ' ')[:1000]
|
||||
writer.write(
|
||||
f'{secret}\t{result["status"]}\t{result["checked_at"]}\t{message}\n'
|
||||
)
|
||||
count = 1
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
streams.append(SerializedStream(
|
||||
stream_name, path, writer.bytes_written, writer.digest.hexdigest(), count,
|
||||
artifact_id,
|
||||
))
|
||||
if self.artifact_tracking:
|
||||
self.db.register_pipeline_artifact(
|
||||
'jsonl_projector', 'prepared_stream', job['id'], stream_name,
|
||||
os.path.relpath(path, self.results_dir).replace(os.sep, '/'),
|
||||
state='present', payload_sha256=writer.digest.hexdigest(),
|
||||
byte_count=writer.bytes_written,
|
||||
)
|
||||
return streams
|
||||
except BaseException:
|
||||
self._rollback_database()
|
||||
for path, artifact_id in prepared_paths:
|
||||
try:
|
||||
self._delete_registered_artifact(path, artifact_id)
|
||||
except BaseException:
|
||||
pass
|
||||
raise
|
||||
|
||||
def _hash_region(self, path, offset, length):
|
||||
digest = hashlib.sha256()
|
||||
remaining = int(length)
|
||||
with open(path, 'rb', buffering=0) as handle:
|
||||
handle.seek(int(offset))
|
||||
while remaining:
|
||||
block = handle.read(min(1024 * 1024, remaining))
|
||||
if not block:
|
||||
raise OSError('projection append proof is truncated')
|
||||
digest.update(block)
|
||||
remaining -= len(block)
|
||||
return digest.hexdigest()
|
||||
|
||||
def _quarantine_tail(self, active, offset, job, stream_name, append):
|
||||
size = os.path.getsize(active)
|
||||
length = max(0, size - int(offset))
|
||||
safe_stream_name = re.sub(r'[^A-Za-z0-9_.-]+', '_', stream_name)
|
||||
tail_hash = self._hash_region(active, offset, length) if length else hashlib.sha256(b'').hexdigest()
|
||||
path = os.path.join(
|
||||
self.quarantine_dir,
|
||||
f'append-{append["id"]}-o{int(offset)}-l{length}-{tail_hash}.tail',
|
||||
)
|
||||
evidence_error = None
|
||||
evidence_registered = False
|
||||
try:
|
||||
if length:
|
||||
if length > MAX_TAIL_QUARANTINE_BYTES:
|
||||
raise ValueError('projection partial tail exceeds quarantine byte bound')
|
||||
relative = os.path.relpath(path, self.results_dir).replace(os.sep, '/')
|
||||
registration = self.db.register_projection_tail_quarantine(
|
||||
job['id'], append['id'], stream_name, relative, tail_hash, length,
|
||||
self.quarantine_max_items, self.quarantine_max_bytes,
|
||||
)
|
||||
evidence_registered = True
|
||||
if not os.path.lexists(path):
|
||||
temporary = path + '.partial'
|
||||
temp_relative = os.path.relpath(temporary, self.results_dir).replace(os.sep, '/')
|
||||
temp_artifact = self.db.register_pipeline_artifact(
|
||||
'jsonl_projector', 'projection_tail_temp', job['id'],
|
||||
f'{append["id"]}:{int(offset)}:{length}:{tail_hash}',
|
||||
temp_relative, state='expected', byte_count=length,
|
||||
)
|
||||
if os.path.lexists(temporary):
|
||||
if not private_file_ready(temporary):
|
||||
raise OSError('existing projection tail temporary is not private')
|
||||
self._delete_registered_artifact(temporary, temp_artifact)
|
||||
temp_artifact = self.db.register_pipeline_artifact(
|
||||
'jsonl_projector', 'projection_tail_temp', job['id'],
|
||||
f'{append["id"]}:{int(offset)}:{length}:{tail_hash}',
|
||||
temp_relative, state='expected', byte_count=length,
|
||||
)
|
||||
descriptor = os.open(
|
||||
temporary,
|
||||
os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0),
|
||||
0o600,
|
||||
)
|
||||
try:
|
||||
os.close(descriptor)
|
||||
descriptor = None
|
||||
harden_private_file(temporary)
|
||||
with open(active, 'rb', buffering=0) as source, open(temporary, 'wb', buffering=0) as target:
|
||||
source.seek(int(offset))
|
||||
remaining = length
|
||||
while remaining:
|
||||
block = source.read(min(1024 * 1024, remaining))
|
||||
if not block:
|
||||
raise OSError('projection partial tail changed while quarantining')
|
||||
target.write(block)
|
||||
remaining -= len(block)
|
||||
target.flush()
|
||||
os.fsync(target.fileno())
|
||||
if (
|
||||
os.path.getsize(temporary) != length
|
||||
or self._hash_region(temporary, 0, length) != tail_hash
|
||||
):
|
||||
raise OSError('projection tail quarantine proof failed before publication')
|
||||
self.db.register_pipeline_artifact(
|
||||
'jsonl_projector', 'projection_tail_temp', job['id'],
|
||||
f'{append["id"]}:{int(offset)}:{length}:{tail_hash}',
|
||||
temp_relative, state='present', payload_sha256=tail_hash,
|
||||
byte_count=length,
|
||||
)
|
||||
durable_publish(temporary, path)
|
||||
temporary_state = inspect_private_relative_path(
|
||||
self.results_dir, temp_relative,
|
||||
)
|
||||
if temporary_state.state != PrivatePathState.ABSENT:
|
||||
raise OSError('projection tail temporary retirement was not confirmed')
|
||||
self.db.mark_pipeline_artifact_deleted(temp_artifact)
|
||||
finally:
|
||||
if descriptor is not None:
|
||||
os.close(descriptor)
|
||||
elif (
|
||||
not private_file_ready(path)
|
||||
or os.path.getsize(path) != length
|
||||
or self._hash_region(path, 0, length) != tail_hash
|
||||
):
|
||||
raise OSError('immutable projection tail evidence is invalid')
|
||||
if not self.db.confirm_projection_tail_artifact(
|
||||
registration['artifact_id'], tail_hash, length,
|
||||
):
|
||||
raise RuntimeError('projection tail artifact confirmation lost its fence')
|
||||
except BaseException as exc:
|
||||
evidence_error = exc
|
||||
finally:
|
||||
with open(active, 'r+b', buffering=0) as handle:
|
||||
handle.truncate(int(offset))
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
if os.path.getsize(active) != int(offset):
|
||||
raise OSError('projection partial tail corrective truncation was not confirmed')
|
||||
if evidence_error is not None:
|
||||
raise evidence_error
|
||||
|
||||
def _rotate_if_needed(self, state, payload_bytes):
|
||||
active = self._results_path(state['base_relative_path'], state['stream_name'])
|
||||
require_private_directory(os.path.dirname(active), create=True)
|
||||
self._create_private_empty(active)
|
||||
current_size = os.path.getsize(active)
|
||||
if current_size != int(state['committed_offset']):
|
||||
state = self.db.initialize_projection_stream_offset(
|
||||
state['stream_name'], current_size,
|
||||
)
|
||||
if current_size != int(state['committed_offset']):
|
||||
raise RuntimeError('projection active size does not match its committed cursor')
|
||||
if not current_size or current_size + payload_bytes <= int(state['rotation_bytes']):
|
||||
return state
|
||||
segment_relative = self._segment_relative(
|
||||
state['base_relative_path'], state['current_generation'],
|
||||
)
|
||||
rotation = self.db.prepare_projection_rotation(
|
||||
state['stream_name'], current_size, segment_relative,
|
||||
)
|
||||
segment = self._results_path(segment_relative, state['stream_name'])
|
||||
self._inject('before_rotation_rename', rotation)
|
||||
durable_publish(active, segment)
|
||||
self._inject('after_rotation_rename', rotation)
|
||||
self._create_private_empty(active)
|
||||
if not self.db.complete_projection_rotation(rotation['id']):
|
||||
raise RuntimeError('projection rotation completion fence failed')
|
||||
updated = self.db.projection_stream_state(state['stream_name'])
|
||||
oldest = int(updated['current_generation']) - int(updated['max_generations'])
|
||||
if oldest >= 0:
|
||||
old_relative = self._segment_relative(updated['base_relative_path'], oldest)
|
||||
old_path = self._results_path(old_relative, updated['stream_name'])
|
||||
if os.path.isfile(old_path):
|
||||
durable_unlink(old_path)
|
||||
return updated
|
||||
|
||||
def _append_stream(self, job, serialized):
|
||||
state = self.db.projection_stream_state(serialized.stream_name)
|
||||
if not state:
|
||||
raise ValueError(f'projection stream is absent: {serialized.stream_name}')
|
||||
existing = self.db.projection_append_for_job(job['id'], serialized.stream_name)
|
||||
if existing is None:
|
||||
state = self._rotate_if_needed(state, serialized.byte_length)
|
||||
append = self.db.prepare_projection_append(
|
||||
job['id'], job['lease_token'], serialized.stream_name,
|
||||
state['generation'], serialized.byte_length,
|
||||
serialized.payload_sha256, serialized.record_count,
|
||||
)
|
||||
else:
|
||||
append = existing
|
||||
if (
|
||||
int(append['byte_length']) != serialized.byte_length
|
||||
or str(append['payload_sha256']) != serialized.payload_sha256
|
||||
or int(append['record_count']) != serialized.record_count
|
||||
):
|
||||
raise ValueError('prepared projection append conflicts with deterministic serialization')
|
||||
active = self._results_path(state['base_relative_path'], serialized.stream_name)
|
||||
require_private_directory(os.path.dirname(active), create=True)
|
||||
self._create_private_empty(active)
|
||||
offset = int(append['byte_offset'])
|
||||
end = offset + int(append['byte_length'])
|
||||
size = os.path.getsize(active)
|
||||
if append['state'] == 'appended':
|
||||
if size < end or self._hash_region(active, offset, append['byte_length']) != append['payload_sha256']:
|
||||
raise ValueError('acknowledged projection append proof is invalid')
|
||||
return
|
||||
if size >= end:
|
||||
if size == end and self._hash_region(active, offset, append['byte_length']) == append['payload_sha256']:
|
||||
if not self.db.complete_projection_append(append['id'], job['id'], job['lease_token']):
|
||||
raise RuntimeError('projection append replay acknowledgement failed')
|
||||
return
|
||||
self._quarantine_tail(active, offset, job, serialized.stream_name, append)
|
||||
self._inject('after_tail_recovery', append)
|
||||
elif size > offset:
|
||||
self._quarantine_tail(active, offset, job, serialized.stream_name, append)
|
||||
self._inject('after_tail_recovery', append)
|
||||
elif size < offset:
|
||||
raise ValueError('projection active file is shorter than its prepared offset')
|
||||
self._inject('before_append', append)
|
||||
with open(active, 'ab', buffering=0) as target, open(serialized.path, 'rb', buffering=0) as source:
|
||||
while True:
|
||||
block = source.read(1024 * 1024)
|
||||
if not block:
|
||||
break
|
||||
target.write(block)
|
||||
target.flush()
|
||||
os.fsync(target.fileno())
|
||||
self._inject('after_append_fsync', append)
|
||||
if os.path.getsize(active) != end or self._hash_region(active, offset, append['byte_length']) != append['payload_sha256']:
|
||||
raise RuntimeError('projection append proof failed after fsync')
|
||||
if not self.db.complete_projection_append(append['id'], job['id'], job['lease_token']):
|
||||
raise RuntimeError('projection append completion fence failed')
|
||||
|
||||
def process_one(self):
|
||||
self.reconcile_terminal_temps(max_pages=1)
|
||||
job = self.db.claim_projection_job(
|
||||
self.lease['generation'], self.lease['lease_token'], self.lease_seconds,
|
||||
)
|
||||
if not job:
|
||||
return False
|
||||
streams = []
|
||||
primary_failure = False
|
||||
try:
|
||||
streams = self._serialize(job)
|
||||
actual_bytes = sum(item.byte_length for item in streams)
|
||||
expanded = self.db.expand_projection_job_capacity(
|
||||
job['id'], job['lease_token'], actual_bytes,
|
||||
self.projection_max_bytes,
|
||||
)
|
||||
if expanded is False:
|
||||
return True
|
||||
if expanded is None:
|
||||
raise RuntimeError('projection capacity expansion lost its lease fence')
|
||||
job = expanded
|
||||
for serialized in streams:
|
||||
self._append_stream(job, serialized)
|
||||
if not self.db.complete_projection_job(job['id'], job['lease_token']):
|
||||
raise RuntimeError('projection job completion fence failed')
|
||||
return True
|
||||
except (ValueError, TypeError, UnicodeError, json.JSONDecodeError) as exc:
|
||||
self._rollback_database()
|
||||
try:
|
||||
self.db.quarantine_projection_job(
|
||||
job['id'], job['lease_token'], 'deterministic_projection_error', str(exc),
|
||||
quarantine_max_items=self.quarantine_max_items,
|
||||
quarantine_max_bytes=self.quarantine_max_bytes,
|
||||
)
|
||||
except BaseException:
|
||||
primary_failure = True
|
||||
self._rollback_database()
|
||||
raise
|
||||
return True
|
||||
except BaseException:
|
||||
primary_failure = True
|
||||
self._rollback_database()
|
||||
raise
|
||||
finally:
|
||||
for serialized in streams:
|
||||
try:
|
||||
self._delete_registered_artifact(
|
||||
serialized.path, serialized.artifact_id,
|
||||
)
|
||||
except OSError:
|
||||
pass
|
||||
except BaseException:
|
||||
self._rollback_database()
|
||||
if not primary_failure:
|
||||
raise
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description='Singleton PostgreSQL-backed JSONL projector')
|
||||
parser.add_argument('--config', required=True)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
metadata = require_active_supervisor_child(child_kind='jsonl-projector', require_dsn=True)
|
||||
args = parse_args()
|
||||
import yaml
|
||||
|
||||
with open(args.config, 'r', encoding='utf-8') as handle:
|
||||
config = apply_path_config(yaml.safe_load(handle) or {}, args.config)
|
||||
global_config = config.get('global') or {}
|
||||
settings = ((config.get('supervisor') or {}).get('jsonl_projector') or {})
|
||||
db = ScannerDB(db_url=global_config['database_url'], initialize=False)
|
||||
if not db.enabled:
|
||||
raise SystemExit('JSONL projector PostgreSQL connection is unavailable')
|
||||
db.set_application_name('truf-jsonl-projector')
|
||||
worker = JsonlProjector(
|
||||
db, global_config['results_dir'], metadata['instance_id'],
|
||||
lease_seconds=int(settings.get('lease_seconds', 300)),
|
||||
keycheck_dir=global_config['keycheck_dir'],
|
||||
quarantine_max_items=int(global_config.get('pipeline_quarantine_max_items', 10000)),
|
||||
quarantine_max_bytes=int(global_config.get('pipeline_quarantine_max_bytes', 1024 * 1024 * 1024)),
|
||||
projection_max_bytes=int(global_config.get(
|
||||
'projection_backlog_max_bytes', 2 * 1024 * 1024 * 1024,
|
||||
)),
|
||||
)
|
||||
error = ''
|
||||
try:
|
||||
worker.start()
|
||||
idle = max(0.05, float(settings.get('poll_sec', 0.2)))
|
||||
next_heartbeat = time.monotonic() + worker.lease_seconds / 3
|
||||
while True:
|
||||
processed = worker.process_one()
|
||||
if time.monotonic() >= next_heartbeat:
|
||||
if not worker.heartbeat():
|
||||
raise RuntimeError('JSONL projector heartbeat fence was lost')
|
||||
next_heartbeat = time.monotonic() + worker.lease_seconds / 3
|
||||
if not processed:
|
||||
time.sleep(idle)
|
||||
except KeyboardInterrupt:
|
||||
pass
|
||||
except BaseException as exc:
|
||||
error = f'{type(exc).__name__}: {exc}'
|
||||
raise
|
||||
finally:
|
||||
try:
|
||||
worker.stop(error)
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,71 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import os
|
||||
import sqlite3
|
||||
import tempfile
|
||||
|
||||
from keycheckers.keycheck_common import (
|
||||
append_checked,
|
||||
append_status,
|
||||
record_cached_keycheck_occurrence,
|
||||
)
|
||||
from keycheck_runner import ingest_keycheck_results_to_db
|
||||
from scanner_db import ScannerDB
|
||||
from runtime_security import ensure_private_directory
|
||||
|
||||
|
||||
def main():
|
||||
for key in ('SCANNER_DB_URL', 'DATABASE_URL', 'TRUF_MANAGED_POSTGRES_DSN', 'KEYCHECK_DB_URL'):
|
||||
os.environ.pop(key, None)
|
||||
with tempfile.TemporaryDirectory(prefix="keycheck-accounting-") as tmp:
|
||||
db_path = os.path.join(tmp, "scanner.db")
|
||||
output_dir = os.path.join(tmp, "keychecks", "openai")
|
||||
ensure_private_directory(output_dir, reject_reparse=True)
|
||||
|
||||
db = ScannerDB(db_path=db_path)
|
||||
db.close()
|
||||
|
||||
alive_file = os.path.join(output_dir, "openaiAlive.txt")
|
||||
checked_file = os.path.join(output_dir, "openaiChecked.txt")
|
||||
key = "sk-test-keycheck-accounting-1234567890"
|
||||
append_status(alive_file, key, "ALIVE", "fixture", "smoke")
|
||||
append_checked(checked_file, key, "ALIVE")
|
||||
|
||||
os.environ["KEYCHECK_DB_PATH"] = db_path
|
||||
os.environ["KEYCHECK_OUTPUT_DIR"] = output_dir
|
||||
os.environ["KEYCHECK_SERVICE"] = "openai"
|
||||
|
||||
finding = {"DetectorName": "OpenAI", "Raw": key}
|
||||
ok = record_cached_keycheck_occurrence("openai", key, "ALIVE", "fixture.jsonl:1", finding, "OpenAI")
|
||||
if not ok:
|
||||
raise SystemExit("cached occurrence write returned false")
|
||||
|
||||
inserted = ingest_keycheck_results_to_db(
|
||||
{"database_path": db_path, "keycheck_dir": os.path.join(tmp, "keychecks")},
|
||||
["openai"],
|
||||
max_rows=10,
|
||||
)
|
||||
if inserted != 1:
|
||||
raise SystemExit(f"unexpected ingest count: {inserted}")
|
||||
|
||||
conn = sqlite3.connect(db_path)
|
||||
try:
|
||||
row = conn.execute(
|
||||
"SELECT service, status, status_group, metadata_json FROM keycheck_results"
|
||||
).fetchone()
|
||||
finally:
|
||||
conn.close()
|
||||
if not row:
|
||||
raise SystemExit("missing keycheck_results row")
|
||||
service, status, status_group, metadata_json = row
|
||||
if (service, status, status_group) != ("openai", "ALIVE", "alive"):
|
||||
raise SystemExit(f"unexpected row status: {(service, status, status_group)}")
|
||||
if "cached_status" not in metadata_json:
|
||||
raise SystemExit("missing cached_status metadata")
|
||||
print("keycheck accounting smoke ok")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,524 @@
|
||||
import hashlib
|
||||
import json
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
|
||||
MAX_CANDIDATE_SECRET_BYTES = 1024 * 1024
|
||||
MAX_CANDIDATE_METADATA_BYTES = 64 * 1024
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CandidateSpec:
|
||||
service: str
|
||||
candidate_kind: str
|
||||
credential_hash: str
|
||||
provider_key_hash: str
|
||||
secret_hash: str
|
||||
secret_text: str | None = None
|
||||
secret_json: str | None = None
|
||||
key_masked: str = ''
|
||||
endpoint: str = ''
|
||||
principal: str = ''
|
||||
metadata: dict = field(default_factory=dict)
|
||||
|
||||
def as_frame(self, attribution=None):
|
||||
return {
|
||||
'service': self.service,
|
||||
'candidate_kind': self.candidate_kind,
|
||||
'credential_hash': self.credential_hash,
|
||||
'provider_key_hash': self.provider_key_hash,
|
||||
'secret_hash': self.secret_hash,
|
||||
'secret_text': self.secret_text,
|
||||
'secret_json': self.secret_json,
|
||||
'key_masked': self.key_masked,
|
||||
'endpoint': self.endpoint,
|
||||
'principal': self.principal,
|
||||
'metadata': self.metadata,
|
||||
'attribution': dict(attribution or {}),
|
||||
}
|
||||
|
||||
|
||||
DETECTOR_SERVICES = {
|
||||
'openai': 'openai',
|
||||
'anthropic': 'anthropic',
|
||||
'qwendashscope': 'qwen',
|
||||
'qwen_dashscope': 'qwen',
|
||||
'qwen': 'qwen',
|
||||
'dashscope': 'qwen',
|
||||
'deepseek': 'deepseek',
|
||||
'deepseekapikey': 'deepseek',
|
||||
'deepseek_api_key': 'deepseek',
|
||||
'zaiglm': 'zai',
|
||||
'kimimoonshot': 'kimi',
|
||||
'moonshotai': 'kimi',
|
||||
'moonshot': 'kimi',
|
||||
'kimi': 'kimi',
|
||||
'openrouter': 'openrouter',
|
||||
'groq': 'groq',
|
||||
'replicate': 'replicate',
|
||||
'xai': 'xai',
|
||||
'huggingface': 'huggingface',
|
||||
'github': 'github',
|
||||
'githuboauth2': 'github',
|
||||
'gitlab': 'gitlab',
|
||||
'aws': 'aws',
|
||||
'gcp': 'gcp',
|
||||
'gcpapplicationdefaultcredentials': 'gcp',
|
||||
'googleai': 'gemini',
|
||||
'googleaistudio': 'gemini',
|
||||
'azure': 'azure',
|
||||
'azureopenai': 'azure',
|
||||
'azurecontainerregistry': 'azure',
|
||||
'azurefoundryendpointbeforekey': 'azure',
|
||||
'azurefoundrykeybeforeendpoint': 'azure',
|
||||
'dockerhub': 'dockerhub',
|
||||
}
|
||||
GEMINI_KEY_RE = re.compile(
|
||||
r'AIza[0-9A-Za-z_-]{20,}|(?<![0-9A-Za-z_-])AQ\.[0-9A-Za-z_-]{50}(?![0-9A-Za-z_-])'
|
||||
)
|
||||
AZURE_OPENAI_KEY_RE = re.compile(r'\b[a-fA-F0-9]{32}\b')
|
||||
AZURE_OPENAI_ENDPOINT_RE = re.compile(r'([a-z0-9-]+\.openai\.azure\.com)', re.IGNORECASE)
|
||||
AZURE_FOUNDRY_ENDPOINT_RE = re.compile(
|
||||
r'([a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models|services|inference)\.ai\.azure\.com)',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
AZURE_FOUNDRY_KEY_ASSIGNMENT_RE = re.compile(
|
||||
r'(?is)(?:authorization|api[_-]?key|key|token|secret|credential|bearer)'
|
||||
r'[^\n:=]{0,80}[:=]\s*["\']?(?:bearer\s+)?([A-Za-z0-9_./+=\-]{20,512})'
|
||||
)
|
||||
NON_FOUNDRY_KEY_PREFIXES = (
|
||||
'sk-', 'sk_', 'sk-or-', 'xai-', 'ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_',
|
||||
'github_pat_', 'glpat-', 'glrt-', 'hf_', 'AIza', 'AQ.', 'zai-', 'gsk_',
|
||||
'r8_', 'nvapi-',
|
||||
)
|
||||
GITHUB_TOKEN_RE = re.compile(r'\b(?:gh[pousr]_[A-Za-z0-9_]{20,}|github_pat_[A-Za-z0-9_]{20,})\b')
|
||||
GITLAB_TOKEN_RE = re.compile(r'\b(?:glpat|gloas|glcbt|glimt|glrt|glft|glsoat)-[A-Za-z0-9_\-=]{20,}\b')
|
||||
DOCKER_PAT_RE = re.compile(r'\bdckr_pat_[A-Za-z0-9_-]{27}\b')
|
||||
QWEN_KEY_RE = re.compile(r'\b(?:sk-sp-[A-Za-z0-9_-]{16,}|sk-[A-Za-z0-9_-]{20,})\b')
|
||||
KIMI_KEY_RE = re.compile(r'\bsk-[A-Za-z0-9][A-Za-z0-9_-]{20,505}\b')
|
||||
ZAI_KEY_RE = re.compile(
|
||||
r'(?<![A-Za-z0-9_.-])(?:'
|
||||
r'(?:zai|sk)-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|'
|
||||
r'[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}'
|
||||
r')(?![A-Za-z0-9_.-])'
|
||||
)
|
||||
GENERIC_SK_PROVIDERS = {'qwen', 'deepseek', 'kimi', 'zai'}
|
||||
AMBIGUOUS_PROVIDER_HINTS = {'ambiguous_qwen_deepseek', 'ambiguous_generic_sk'}
|
||||
|
||||
|
||||
def _json(value):
|
||||
return json.dumps(value, ensure_ascii=True, sort_keys=True, separators=(',', ':'))
|
||||
|
||||
|
||||
def _provider_json(value):
|
||||
return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(',', ':'))
|
||||
|
||||
|
||||
def _bounded(value, max_bytes):
|
||||
text = str(value or '')
|
||||
encoded = text.encode('utf-8', errors='strict')
|
||||
if len(encoded) > max_bytes:
|
||||
raise ValueError('keycheck candidate field exceeds its byte bound')
|
||||
return text
|
||||
|
||||
|
||||
def _mask(value):
|
||||
text = str(value or '')
|
||||
if len(text) <= 8:
|
||||
return '*' * len(text)
|
||||
return text[:4] + ('*' * min(24, len(text) - 8)) + text[-4:]
|
||||
|
||||
|
||||
def _service_for_finding(finding):
|
||||
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
|
||||
hint = str(context.get('provider_hint') or '').lower()
|
||||
if hint in AMBIGUOUS_PROVIDER_HINTS:
|
||||
return 'provider_resolver'
|
||||
if hint in GENERIC_SK_PROVIDERS:
|
||||
return hint
|
||||
detector = re.sub(r'[^a-z0-9_]', '', str(
|
||||
finding.get('DetectorName') or finding.get('DetectorType') or ''
|
||||
).lower())
|
||||
service = DETECTOR_SERVICES.get(detector, '')
|
||||
if service:
|
||||
return service
|
||||
extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {}
|
||||
name = re.sub(r'[^a-z0-9_]', '', str(extra.get('name') or '').lower())
|
||||
return DETECTOR_SERVICES.get(name, '')
|
||||
|
||||
|
||||
def _make_candidate(
|
||||
service, candidate_kind, probe_material, raw_material, *, secret_text=None,
|
||||
secret_json=None, endpoint='', principal='', metadata=None,
|
||||
):
|
||||
probe_material = _bounded(probe_material, MAX_CANDIDATE_SECRET_BYTES)
|
||||
raw_material = _bounded(raw_material, MAX_CANDIDATE_SECRET_BYTES)
|
||||
provider_key_hash = hashlib.sha256(probe_material.encode('utf-8')).hexdigest()
|
||||
secret_hash = hashlib.sha256(raw_material.encode('utf-8')).hexdigest()
|
||||
credential_hash = hashlib.sha256('|'.join((
|
||||
'truf-credential-v2', service, probe_material,
|
||||
)).encode('utf-8')).hexdigest()
|
||||
metadata = dict(metadata or {})
|
||||
if len(_json(metadata).encode('utf-8')) > MAX_CANDIDATE_METADATA_BYTES:
|
||||
raise ValueError('keycheck candidate metadata exceeds its byte bound')
|
||||
return CandidateSpec(
|
||||
service=service,
|
||||
candidate_kind=candidate_kind,
|
||||
credential_hash=credential_hash,
|
||||
provider_key_hash=provider_key_hash,
|
||||
secret_hash=secret_hash,
|
||||
secret_text=secret_text,
|
||||
secret_json=secret_json,
|
||||
key_masked=_mask(probe_material),
|
||||
endpoint=endpoint,
|
||||
principal=principal,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
|
||||
def stored_provider_key_hash(service, candidate_kind, secret_text, secret_json, endpoint='', principal=''):
|
||||
service = str(service or '').lower()
|
||||
candidate_kind = str(candidate_kind or '')
|
||||
secret_text = str(secret_text or '')
|
||||
secret_json = str(secret_json or '')
|
||||
endpoint = str(endpoint or '').lower()
|
||||
principal = str(principal or '')
|
||||
if service == 'gcp':
|
||||
parsed = _json_object(secret_json)
|
||||
probe = _provider_json(parsed) if parsed is not None else secret_json
|
||||
elif service == 'azure' and candidate_kind == 'azure_service_principal':
|
||||
parsed = _json_object(secret_json) or {}
|
||||
probe = ':'.join(str(parsed.get(name) or '') for name in ('tenantId', 'clientId', 'clientSecret'))
|
||||
if probe == '::':
|
||||
probe = ':'.join(str(parsed.get(name) or '') for name in ('tenant_id', 'client_id', 'client_secret'))
|
||||
elif service == 'azure' and candidate_kind == 'azure_container_registry':
|
||||
parsed = _json_object(secret_json) or {}
|
||||
probe = f"{parsed.get('username') or principal}:{parsed.get('password') or ''}"
|
||||
elif service == 'azure' and endpoint:
|
||||
probe = f'{endpoint}:{secret_text}'
|
||||
elif service == 'dockerhub' and principal:
|
||||
probe = f'{principal}:{secret_text}'
|
||||
else:
|
||||
probe = secret_text or secret_json
|
||||
probe = _bounded(probe, MAX_CANDIDATE_SECRET_BYTES)
|
||||
if not probe:
|
||||
raise ValueError('stored keycheck credential has no provider probe material')
|
||||
return hashlib.sha256(probe.encode('utf-8')).hexdigest()
|
||||
|
||||
|
||||
def _json_object(value):
|
||||
try:
|
||||
parsed = json.loads(str(value or ''))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return parsed if isinstance(parsed, dict) else None
|
||||
|
||||
|
||||
def _raw_material(finding):
|
||||
value = finding.get('RawV2') or finding.get('Raw')
|
||||
if value:
|
||||
return str(value)
|
||||
structured = finding.get('StructuredData')
|
||||
return _json(structured) if isinstance(structured, dict) and structured else ''
|
||||
|
||||
|
||||
def _azure_foundry_keyish(value):
|
||||
text = re.sub(r'(?i)^bearer\s+', '', str(value or '').strip().strip('"\'`,;')).strip()
|
||||
lowered = text.lower()
|
||||
if not (20 <= len(text) <= 512) or any(character.isspace() for character in text):
|
||||
return False
|
||||
if any(marker in lowered for marker in ('http://', 'https://', '{{', '${', '<', 'azure.com')):
|
||||
return False
|
||||
if text.startswith(NON_FOUNDRY_KEY_PREFIXES):
|
||||
return False
|
||||
return bool(re.search(r'[A-Za-z]', text) and re.search(r'[0-9]', text))
|
||||
|
||||
|
||||
def extract_azure_foundry_parts(raw, raw_v2=''):
|
||||
materials = [str(value or '') for value in (raw_v2, raw) if value]
|
||||
endpoint = next((
|
||||
match.group(1).lower()
|
||||
for value in materials
|
||||
for match in [AZURE_FOUNDRY_ENDPOINT_RE.search(value)]
|
||||
if match
|
||||
), '')
|
||||
if not endpoint:
|
||||
return None
|
||||
for value in materials:
|
||||
for key in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(value):
|
||||
key = str(key).strip().strip('"\'`,;')
|
||||
if _azure_foundry_keyish(key):
|
||||
return {'key': key, 'endpoint': endpoint}
|
||||
match = AZURE_FOUNDRY_ENDPOINT_RE.search(value)
|
||||
if not match:
|
||||
continue
|
||||
before = value[:match.start()].strip(' \t\r\n:=,;\'"/')
|
||||
after = value[match.end():].strip(' \t\r\n:=,;\'"/')
|
||||
for key in (after, before):
|
||||
if _azure_foundry_keyish(key):
|
||||
return {'key': key, 'endpoint': endpoint}
|
||||
return None
|
||||
|
||||
|
||||
def extract_candidates(finding, attribution=None):
|
||||
if not isinstance(finding, dict):
|
||||
return
|
||||
service = _service_for_finding(finding)
|
||||
if not service:
|
||||
return
|
||||
detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '')
|
||||
detector_key = re.sub(r'[^a-z0-9_]', '', detector.lower())
|
||||
postman = finding.get('PostmanContext') if isinstance(finding.get('PostmanContext'), dict) else {}
|
||||
raw = str(finding.get('Raw') or '')
|
||||
raw_v2 = str(finding.get('RawV2') or '')
|
||||
raw_material = _raw_material(finding)
|
||||
base_metadata = {
|
||||
'detector_name': detector,
|
||||
'finding_uid': str(finding.get('finding_uid') or ''),
|
||||
}
|
||||
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
|
||||
provider_candidates = [
|
||||
str(provider).lower() for provider in context.get('provider_candidates') or ()
|
||||
if str(provider).lower() in GENERIC_SK_PROVIDERS
|
||||
]
|
||||
provider_hint = str(context.get('provider_hint') or '')
|
||||
if provider_hint:
|
||||
base_metadata['provider_hint'] = provider_hint
|
||||
if provider_candidates:
|
||||
base_metadata['provider_candidates'] = list(dict.fromkeys(provider_candidates))
|
||||
endpoint = str(postman.get('endpoint') or '')
|
||||
principal = str(postman.get('principal') or postman.get('username') or '')
|
||||
|
||||
if service == 'aws':
|
||||
probe = raw_v2 or raw
|
||||
if ':' in probe:
|
||||
yield _make_candidate(
|
||||
service, 'aws_access_key_pair', probe, raw_material,
|
||||
secret_text=probe, metadata={**base_metadata, 'raw_v2': probe},
|
||||
)
|
||||
return
|
||||
if service == 'gcp':
|
||||
parsed = _json_object(raw_v2)
|
||||
if parsed is None:
|
||||
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
|
||||
parsed = _json_object(context.get('nearby'))
|
||||
if parsed is not None:
|
||||
probe = _provider_json(parsed)
|
||||
yield _make_candidate(
|
||||
service, 'gcp_json', probe, raw_material or probe,
|
||||
secret_json=probe, metadata={**base_metadata, 'raw_v2': probe},
|
||||
)
|
||||
return
|
||||
if service == 'azure':
|
||||
parsed = _json_object(raw_v2)
|
||||
if detector_key == 'azure':
|
||||
if parsed:
|
||||
tenant = parsed.get('tenantId') or parsed.get('tenant_id')
|
||||
client = parsed.get('clientId') or parsed.get('client_id')
|
||||
secret = parsed.get('clientSecret') or parsed.get('client_secret')
|
||||
if tenant and client and secret:
|
||||
probe = f'{tenant}:{client}:{secret}'
|
||||
canonical_json = _json(parsed)
|
||||
yield _make_candidate(
|
||||
service, 'azure_service_principal', probe, raw_material,
|
||||
secret_json=canonical_json,
|
||||
metadata={**base_metadata, 'raw_v2': canonical_json},
|
||||
)
|
||||
return
|
||||
if detector_key == 'azurecontainerregistry':
|
||||
if parsed:
|
||||
username = parsed.get('username')
|
||||
password = parsed.get('password')
|
||||
if username and password:
|
||||
probe = f'{username}:{password}'
|
||||
canonical_json = _json(parsed)
|
||||
yield _make_candidate(
|
||||
service, 'azure_container_registry', probe, raw_material,
|
||||
secret_json=canonical_json, principal=str(username),
|
||||
metadata={**base_metadata, 'raw_v2': canonical_json},
|
||||
)
|
||||
return
|
||||
if detector_key == 'azureopenai':
|
||||
match = re.match(r'^([a-fA-F0-9]{32}):(.+\.openai\.azure\.com)$', raw_v2)
|
||||
key = match.group(1) if match else raw
|
||||
found_endpoint = match.group(2).lower() if match else endpoint.lower()
|
||||
if key:
|
||||
probe = f'{found_endpoint}:{key}' if found_endpoint else key
|
||||
provider_raw_v2 = f'{key}:{found_endpoint}' if found_endpoint else raw_v2
|
||||
yield _make_candidate(
|
||||
service, 'azure_openai', probe, raw_material,
|
||||
secret_text=key, endpoint=found_endpoint,
|
||||
metadata={**base_metadata, 'raw_v2': provider_raw_v2},
|
||||
)
|
||||
return
|
||||
foundry = extract_azure_foundry_parts(raw, raw_v2)
|
||||
if foundry:
|
||||
foundry_endpoint = foundry['endpoint']
|
||||
foundry_key = foundry['key']
|
||||
probe = f'{foundry_endpoint}:{foundry_key}'
|
||||
yield _make_candidate(
|
||||
service, 'azure_foundry', probe, raw_material,
|
||||
secret_text=foundry_key, endpoint=foundry_endpoint,
|
||||
metadata={**base_metadata, 'raw_v2': probe},
|
||||
)
|
||||
return
|
||||
if service == 'dockerhub':
|
||||
token_match = DOCKER_PAT_RE.search(raw_v2) or DOCKER_PAT_RE.search(raw)
|
||||
if not token_match:
|
||||
return
|
||||
token = token_match.group(0)
|
||||
username = ''
|
||||
if ':' in raw_v2 and raw_v2.rsplit(':', 1)[-1] == token:
|
||||
username = raw_v2.rsplit(':', 1)[0]
|
||||
extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {}
|
||||
analysis = finding.get('AnalysisInfo') if isinstance(finding.get('AnalysisInfo'), dict) else {}
|
||||
username = username or str(extra.get('hub_username') or analysis.get('username') or '')
|
||||
probe = f'{username}:{token}' if username else token
|
||||
yield _make_candidate(
|
||||
service, 'dockerhub_pat', probe, raw_material,
|
||||
secret_text=token, principal=username,
|
||||
metadata={**base_metadata, 'raw_v2': probe},
|
||||
)
|
||||
return
|
||||
patterns = {
|
||||
'github': GITHUB_TOKEN_RE,
|
||||
'gitlab': GITLAB_TOKEN_RE,
|
||||
'gemini': GEMINI_KEY_RE,
|
||||
'qwen': QWEN_KEY_RE,
|
||||
'kimi': KIMI_KEY_RE,
|
||||
'zai': ZAI_KEY_RE,
|
||||
'provider_resolver': KIMI_KEY_RE,
|
||||
}
|
||||
pattern = patterns.get(service)
|
||||
values = pattern.findall(raw + '\n' + raw_v2) if pattern else [raw or raw_v2]
|
||||
seen = set()
|
||||
for value in values:
|
||||
value = str(value or '').strip()
|
||||
if not value or value in seen:
|
||||
continue
|
||||
seen.add(value)
|
||||
yield _make_candidate(
|
||||
service, 'provider_key', value, raw_material or value,
|
||||
secret_text=value, endpoint=endpoint, principal=principal,
|
||||
metadata={**base_metadata, 'raw_v2': raw_v2},
|
||||
)
|
||||
|
||||
|
||||
def extract_structured_candidates(artifact_context, attribution=None):
|
||||
if not isinstance(artifact_context, dict):
|
||||
return
|
||||
for finding in artifact_context.get('findings') or ():
|
||||
yield from extract_candidates(finding, attribution)
|
||||
contexts = artifact_context.get('contexts') or ()
|
||||
origin = str(artifact_context.get('origin') or 'structured-artifact')
|
||||
endpoint_values = []
|
||||
for context in contexts:
|
||||
if not isinstance(context, dict):
|
||||
continue
|
||||
text = ' '.join(str(context.get(key) or '') for key in ('value', 'endpoint', 'host', 'key'))
|
||||
endpoint_values.extend(AZURE_OPENAI_ENDPOINT_RE.findall(text))
|
||||
endpoint_values.extend(AZURE_FOUNDRY_ENDPOINT_RE.findall(text))
|
||||
seen = set()
|
||||
for index, context in enumerate(contexts):
|
||||
if not isinstance(context, dict):
|
||||
continue
|
||||
value = str(context.get('value') or '').strip()
|
||||
if not value or '{{' in value or '${' in value:
|
||||
continue
|
||||
path = str(context.get('path') or '')
|
||||
text = ' '.join((
|
||||
value, str(context.get('endpoint') or ''), str(context.get('host') or ''),
|
||||
str(context.get('key') or ''),
|
||||
))
|
||||
for key in GEMINI_KEY_RE.findall(value):
|
||||
identity = ('gemini', key)
|
||||
if identity in seen:
|
||||
continue
|
||||
seen.add(identity)
|
||||
yield _make_candidate(
|
||||
'gemini', 'structured_postman', key, key, secret_text=key,
|
||||
metadata={
|
||||
'detector_name': 'GoogleAIStudio', 'origin': origin,
|
||||
'structured_origin': f'{origin}:{path or index}',
|
||||
},
|
||||
)
|
||||
azure_key = AZURE_OPENAI_KEY_RE.search(value)
|
||||
if azure_key:
|
||||
endpoints = AZURE_OPENAI_ENDPOINT_RE.findall(text) or [
|
||||
endpoint for endpoint in endpoint_values
|
||||
if AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint)
|
||||
]
|
||||
for endpoint in endpoints[:5]:
|
||||
key = azure_key.group(0)
|
||||
secret = f'{key}:{str(endpoint).lower()}'
|
||||
identity = ('azure-openai', secret)
|
||||
if identity in seen:
|
||||
continue
|
||||
seen.add(identity)
|
||||
yield _make_candidate(
|
||||
'azure', 'structured_postman', f'{str(endpoint).lower()}:{key}',
|
||||
secret, secret_text=key, endpoint=str(endpoint).lower(), metadata={
|
||||
'detector_name': 'AzureOpenAI', 'origin': origin,
|
||||
'raw_v2': secret,
|
||||
'structured_origin': f'{origin}:{path or index}',
|
||||
},
|
||||
)
|
||||
key_label = str(context.get('key') or '').lower()
|
||||
normalized_key_label = re.sub(r'[^a-z0-9]+', '_', key_label).strip('_')
|
||||
structured_provider = None
|
||||
structured_detector = ''
|
||||
structured_pattern = None
|
||||
if normalized_key_label in ('dashscope_api_key', 'qwen_api_key'):
|
||||
structured_provider = 'qwen'
|
||||
structured_detector = 'QwenDashScope'
|
||||
structured_pattern = QWEN_KEY_RE
|
||||
elif normalized_key_label in ('moonshot_api_key', 'kimi_api_key'):
|
||||
structured_provider = 'kimi'
|
||||
structured_detector = 'KimiMoonshot'
|
||||
structured_pattern = KIMI_KEY_RE
|
||||
elif normalized_key_label in (
|
||||
'zai_api_key', 'z_ai_api_key', 'glm_api_key',
|
||||
'zhipuai_api_key', 'bigmodel_api_key',
|
||||
):
|
||||
structured_provider = 'zai'
|
||||
structured_detector = 'ZaiGLM'
|
||||
structured_pattern = ZAI_KEY_RE
|
||||
if structured_provider and structured_pattern.fullmatch(value):
|
||||
identity = (structured_provider, value)
|
||||
if identity not in seen:
|
||||
seen.add(identity)
|
||||
yield _make_candidate(
|
||||
structured_provider, 'structured_postman', value, value,
|
||||
secret_text=value, metadata={
|
||||
'detector_name': structured_detector, 'origin': origin,
|
||||
'structured_origin': f'{origin}:{path or index}',
|
||||
},
|
||||
)
|
||||
foundry_endpoints = AZURE_FOUNDRY_ENDPOINT_RE.findall(text)
|
||||
if (
|
||||
foundry_endpoints and 20 <= len(value) <= 512
|
||||
and not any(character.isspace() for character in value)
|
||||
and any(token in key_label for token in ('key', 'token', 'secret', 'authorization'))
|
||||
):
|
||||
for endpoint in foundry_endpoints[:5]:
|
||||
secret = f'{str(endpoint).lower()}:{value}'
|
||||
identity = ('azure-foundry', secret)
|
||||
if identity in seen:
|
||||
continue
|
||||
seen.add(identity)
|
||||
yield _make_candidate(
|
||||
'azure', 'structured_postman', secret, secret,
|
||||
secret_text=value, endpoint=str(endpoint).lower(), metadata={
|
||||
'detector_name': 'AzureFoundryEndpointBeforeKey', 'origin': origin,
|
||||
'raw_v2': secret,
|
||||
'structured_origin': f'{origin}:{path or index}',
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def candidate_uid(scan_event_id, finding_uid_or_origin, service, credential_hash):
|
||||
return hashlib.sha256('|'.join((
|
||||
'truf-keycheck-candidate-v1', str(scan_event_id), str(finding_uid_or_origin),
|
||||
str(service), str(credential_hash),
|
||||
)).encode('utf-8')).hexdigest()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1 @@
|
||||
"""Keychecker package for the unified scanner layout."""
|
||||
@@ -0,0 +1,274 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
read_plain_keys,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "anthropic"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "anthropicChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "anthropicResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "anthropicAlive.txt"),
|
||||
"NO_QUOTA": os.path.join(OUTPUT_DIR, "anthropicNoQuota.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "anthropicDead.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "anthropicLimited.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "anthropicRestricted.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "anthropicNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "anthropicUnknown.txt"),
|
||||
}
|
||||
|
||||
ANTHROPIC_REGEX = re.compile(r"sk-ant-(?:api03|admin01)-[A-Za-z0-9\-_]{93}AA|sk-ant-[A-Za-z0-9\-_]{86}")
|
||||
ANTHROPIC_ADMIN_PREFIX = "sk-ant-admin01-"
|
||||
ANTHROPIC_ADMIN_API_KEYS_URL = "https://api.anthropic.com/v1/organizations/api_keys"
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["Anthropic"]):
|
||||
key = item["raw"]
|
||||
if key and ANTHROPIC_REGEX.fullmatch(key):
|
||||
yield key, item["source"], item["finding"]
|
||||
for item in read_plain_keys(plain_files, ANTHROPIC_REGEX):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def tier_from_rpm(rpm):
|
||||
mapping = {5: "Free Tier", 50: "Tier 1", 1000: "Tier 2", 2000: "Tier 3", 4000: "Tier 4"}
|
||||
return mapping.get(rpm, "Scale/Unknown")
|
||||
|
||||
|
||||
def anthropic_rate_headers(response):
|
||||
output = {}
|
||||
for name, value in response.headers.items():
|
||||
lowered = name.lower()
|
||||
if lowered.startswith("anthropic-ratelimit-"):
|
||||
output[lowered.replace("anthropic-ratelimit-", "rate_").replace("-", "_")] = value
|
||||
return output
|
||||
|
||||
|
||||
def list_models(key, proxy, timeout):
|
||||
headers = {
|
||||
"anthropic-version": "2023-06-01",
|
||||
"x-api-key": key,
|
||||
}
|
||||
try:
|
||||
response = requests.get("https://api.anthropic.com/v1/models", headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"models_error": str(exc)[:500]}
|
||||
if response.status_code != 200:
|
||||
return {"models_status": response.status_code, "models_error": request_error_message(response)}
|
||||
try:
|
||||
data = response.json()
|
||||
except ValueError:
|
||||
return {"models_status": response.status_code, "models_error": "invalid JSON response"}
|
||||
models = []
|
||||
for item in data.get("data") or []:
|
||||
if isinstance(item, dict) and item.get("id"):
|
||||
models.append(item["id"])
|
||||
return {
|
||||
"models_status": response.status_code,
|
||||
"models_count": len(models),
|
||||
"models": models[:50],
|
||||
}
|
||||
|
||||
|
||||
def check_admin_key(key, proxy, timeout):
|
||||
headers = {
|
||||
"anthropic-version": "2023-06-01",
|
||||
"x-api-key": key,
|
||||
}
|
||||
try:
|
||||
response = requests.get(
|
||||
ANTHROPIC_ADMIN_API_KEYS_URL,
|
||||
headers=headers,
|
||||
params={"limit": 1},
|
||||
proxies=proxy,
|
||||
timeout=timeout,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "admin": True, "message": str(exc)[:500]}
|
||||
|
||||
if response.status_code == 200:
|
||||
return {
|
||||
"status": "VALID",
|
||||
"admin": True,
|
||||
"http_status": 200,
|
||||
"message": "Anthropic Admin API access confirmed",
|
||||
}
|
||||
|
||||
status = {
|
||||
401: "DEAD",
|
||||
403: "RESTRICTED",
|
||||
429: "LIMITED",
|
||||
}.get(response.status_code, "UNKNOWN")
|
||||
message = request_error_message(response).replace(key, "***REDACTED***")
|
||||
return {
|
||||
"status": status,
|
||||
"admin": True,
|
||||
"http_status": response.status_code,
|
||||
"message": message,
|
||||
}
|
||||
|
||||
|
||||
def check_key(key, proxy, timeout, model="claude-opus-4-6", include_models=False):
|
||||
if key.startswith(ANTHROPIC_ADMIN_PREFIX):
|
||||
return check_admin_key(key, proxy, timeout)
|
||||
|
||||
url = "https://api.anthropic.com/v1/messages"
|
||||
headers = {
|
||||
"content-type": "application/json",
|
||||
"anthropic-version": "2023-06-01",
|
||||
"x-api-key": key,
|
||||
}
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": 1,
|
||||
}
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
|
||||
if response.status_code == 200:
|
||||
rpm = 0
|
||||
try:
|
||||
rpm = int(response.headers.get("anthropic-ratelimit-requests-limit", "0"))
|
||||
except ValueError:
|
||||
rpm = 0
|
||||
rate_data = anthropic_rate_headers(response)
|
||||
result = {
|
||||
"status": "VALID",
|
||||
"model": model,
|
||||
"rpm": rpm,
|
||||
"tier": tier_from_rpm(rpm),
|
||||
**rate_data,
|
||||
}
|
||||
if include_models:
|
||||
result.update(list_models(key, proxy, timeout))
|
||||
token_limit = rate_data.get("rate_tokens_limit") or ""
|
||||
token_remaining = rate_data.get("rate_tokens_remaining") or ""
|
||||
token_part = f" tokens={token_remaining}/{token_limit}" if token_limit or token_remaining else ""
|
||||
result["message"] = f"model={model}; rpm={rpm}; tier={result['tier']}{token_part}"
|
||||
return result
|
||||
|
||||
if response.status_code == 429:
|
||||
return {"status": "LIMITED", "http_status": 429, "message": request_error_message(response)}
|
||||
|
||||
message = request_error_message(response)
|
||||
lower = message.lower()
|
||||
if "credit balance is too low" in lower or "usage limits" in lower:
|
||||
return {"status": "NO_QUOTA", "http_status": response.status_code, "message": message}
|
||||
if response.status_code in (401, 403):
|
||||
status = "RESTRICTED" if "disabled" in lower or response.status_code == 403 else "DEAD"
|
||||
return {"status": status, "http_status": response.status_code, "message": message}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "Anthropic")
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, "Anthropic")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Anthropic key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--model", default=os.getenv("ANTHROPIC_CHECK_MODEL", "claude-opus-4-6"))
|
||||
parser.add_argument("--list-models", action="store_true")
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.add("NO_QUOTA")
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in extract_candidates(args.input, args.plain):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="Anthropic"):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] Anthropic candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(key, proxy, args.timeout, args.model, args.list_models)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,607 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_known_statuses,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
read_plain_keys,
|
||||
recover_status_transaction,
|
||||
record_cached_keycheck_occurrence,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "aws"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "awsChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "awsResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "awsAlive.txt"),
|
||||
"BEDROCK": os.path.join(OUTPUT_DIR, "awsBedrock.txt"),
|
||||
"ADMIN": os.path.join(OUTPUT_DIR, "awsAdmin.txt"),
|
||||
"CANARY": os.path.join(OUTPUT_DIR, "awsCanary.txt"),
|
||||
"QUARANTINED": os.path.join(OUTPUT_DIR, "awsQuarantined.txt"),
|
||||
"ACCESS_DENIED": os.path.join(OUTPUT_DIR, "awsAccessDenied.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "awsDead.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "awsNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "awsUnknown.txt"),
|
||||
}
|
||||
|
||||
BEDROCK_REGIONS = ["us-east-1", "us-west-2", "eu-west-1", "eu-north-1", "ap-northeast-1", "ap-southeast-4"]
|
||||
ANTHROPIC_MESSAGES_PROBE = {
|
||||
"anthropic_version": "bedrock-2023-05-31",
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": -1,
|
||||
}
|
||||
ANTHROPIC_MESSAGES_LIVE_PING = {
|
||||
"anthropic_version": "bedrock-2023-05-31",
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": 1,
|
||||
}
|
||||
BEDROCK_MODEL_TESTS = {
|
||||
# Current Anthropic Bedrock runtime IDs. The default probe intentionally uses
|
||||
# invalid max_tokens to validate auth/model access without generating tokens.
|
||||
"anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"us.anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"global.anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-3-5-sonnet-20241022-v2:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-3-5-haiku-20241022-v1:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-3-haiku-20240307-v1:0": ANTHROPIC_MESSAGES_PROBE,
|
||||
"anthropic.claude-v2": {"prompt": "\n\nHuman:\n\nAssistant:", "max_tokens_to_sample": -1},
|
||||
"anthropic.claude-instant-v1": {"prompt": "\n\nHuman:\n\nAssistant:", "max_tokens_to_sample": -1},
|
||||
}
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["AWS"]):
|
||||
key = item["raw_v2"] or item["raw"]
|
||||
if key and ":" in key:
|
||||
yield key, item["source"], item["finding"]
|
||||
import re
|
||||
regex = re.compile(r"AKIA[0-9A-Z]{16}:[A-Za-z0-9+/]{40}")
|
||||
for item in read_plain_keys(plain_files, regex):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def is_dead_aws_error(code):
|
||||
return code in {"InvalidClientTokenId", "SignatureDoesNotMatch", "AuthFailure", "UnrecognizedClientException"}
|
||||
|
||||
|
||||
def is_canary_text(value):
|
||||
value = str(value or "").lower()
|
||||
return "canarytokens" in value or "canary token" in value or "is_canary" in value
|
||||
|
||||
|
||||
def is_canary_finding(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return False
|
||||
extra = finding.get("ExtraData") or {}
|
||||
if isinstance(extra, dict):
|
||||
if str(extra.get("is_canary", "")).lower() == "true":
|
||||
return True
|
||||
if any(is_canary_text(value) for value in extra.values()):
|
||||
return True
|
||||
return is_canary_text(finding.get("Raw")) or is_canary_text(finding.get("RawV2"))
|
||||
|
||||
|
||||
def is_canary_arn(arn):
|
||||
return is_canary_text(arn)
|
||||
|
||||
|
||||
def aws_client(session, service, proxy=None, region_name=None, timeout=20):
|
||||
kwargs = {}
|
||||
if region_name:
|
||||
kwargs["region_name"] = region_name
|
||||
from botocore.config import Config
|
||||
kwargs["config"] = Config(
|
||||
proxies=proxy or None,
|
||||
connect_timeout=timeout,
|
||||
read_timeout=timeout,
|
||||
retries={"max_attempts": 1},
|
||||
)
|
||||
return session.client(service, **kwargs)
|
||||
|
||||
|
||||
def bedrock_validation_allows_invoke(exc):
|
||||
text = str(exc or "").lower()
|
||||
if any(item in text for item in ("operation not allowed", "not authorized", "access denied")):
|
||||
return False
|
||||
# The default probe sends deliberately invalid token limits. If Bedrock only
|
||||
# rejects the payload shape after auth, InvokeModel reached the model path.
|
||||
return any(item in text for item in ("max_tokens", "max_tokens_to_sample", "malformed input", "schema"))
|
||||
|
||||
|
||||
def client_error_code(exc):
|
||||
try:
|
||||
return exc.response.get("Error", {}).get("Code", "ClientError")
|
||||
except Exception:
|
||||
return "ClientError"
|
||||
|
||||
|
||||
def client_error_message(exc):
|
||||
try:
|
||||
return exc.response.get("Error", {}).get("Message", str(exc))
|
||||
except Exception:
|
||||
return str(exc)
|
||||
|
||||
|
||||
def model_arn(region, model_id):
|
||||
# Cross-region inference profile IDs are not foundation-model ARNs.
|
||||
if model_id.startswith(("us.", "eu.", "jp.", "au.", "global.")):
|
||||
return "*"
|
||||
return f"arn:aws:bedrock:{region}::foundation-model/{model_id}"
|
||||
|
||||
|
||||
def iam_policy_source_arn(sts_arn, account):
|
||||
arn = str(sts_arn or "")
|
||||
if ":assumed-role/" in arn:
|
||||
role_part = arn.split(":assumed-role/", 1)[1].split("/", 1)[0]
|
||||
return f"arn:aws:iam::{account}:role/{role_part}"
|
||||
return arn if ":iam::" in arn else ""
|
||||
|
||||
|
||||
def simulate_bedrock_activation(session, arn, account, region, model_id, proxy=None, timeout=20):
|
||||
import botocore.exceptions
|
||||
|
||||
source_arn = iam_policy_source_arn(arn, account)
|
||||
if not source_arn:
|
||||
return {"status": "not_available", "message": "unsupported principal arn for IAM simulation"}
|
||||
actions = [
|
||||
"bedrock:GetFoundationModelAvailability",
|
||||
"bedrock:ListFoundationModelAgreementOffers",
|
||||
"bedrock:GetUseCaseForModelAccess",
|
||||
"bedrock:PutUseCaseForModelAccess",
|
||||
"bedrock:CreateFoundationModelAgreement",
|
||||
"bedrock:GetInferenceProfile",
|
||||
"bedrock:InvokeModel",
|
||||
]
|
||||
try:
|
||||
iam = aws_client(session, "iam", proxy, timeout=timeout)
|
||||
response = iam.simulate_principal_policy(
|
||||
PolicySourceArn=source_arn,
|
||||
ActionNames=actions,
|
||||
ResourceArns=[model_arn(region, model_id)],
|
||||
)
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
return {
|
||||
"status": "access_denied" if client_error_code(exc) == "AccessDenied" else "error",
|
||||
"code": client_error_code(exc),
|
||||
"message": client_error_message(exc)[:500],
|
||||
}
|
||||
decisions = {}
|
||||
for item in response.get("EvaluationResults", []):
|
||||
action = str(item.get("EvalActionName") or "")
|
||||
decisions[action] = str(item.get("EvalDecision") or "")
|
||||
activation_actions = ["bedrock:PutUseCaseForModelAccess", "bedrock:CreateFoundationModelAgreement"]
|
||||
can_activate = all(decisions.get(action) == "allowed" for action in activation_actions)
|
||||
return {"status": "ok", "source_arn": source_arn, "can_activate": can_activate, "decisions": decisions}
|
||||
|
||||
|
||||
def check_bedrock_management(session, arn, account, proxy=None, timeout=20, regions=None, models=None, max_attempts=12, debug=False):
|
||||
import botocore.exceptions
|
||||
|
||||
attempts = []
|
||||
findings = []
|
||||
tried = 0
|
||||
for region in (regions or BEDROCK_REGIONS):
|
||||
bedrock = aws_client(session, "bedrock", proxy, region, timeout)
|
||||
use_case = None
|
||||
try:
|
||||
use_case = bedrock.get_use_case_for_model_access()
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
use_case = {"error_code": client_error_code(exc), "message": client_error_message(exc)[:300]}
|
||||
try:
|
||||
profiles = bedrock.list_inference_profiles(typeEquals="SYSTEM_DEFINED", maxResults=20).get("inferenceProfileSummaries", [])
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
profiles = {"error_code": client_error_code(exc), "message": client_error_message(exc)[:300]}
|
||||
for model_id in (models or list(BEDROCK_MODEL_TESTS.keys())):
|
||||
if max_attempts and tried >= max_attempts:
|
||||
return {"enabled": bool(findings), "findings": findings, "message": "; ".join(attempts[:10])}
|
||||
tried += 1
|
||||
if debug:
|
||||
print(f" BEDROCK MGMT TRY: region={region}, model={model_id}")
|
||||
item = {"region": region, "model": model_id, "use_case": use_case, "profiles": profiles}
|
||||
try:
|
||||
item["foundation_model"] = bedrock.get_foundation_model(modelIdentifier=model_id).get("modelDetails", {})
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
item["foundation_model_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]}
|
||||
try:
|
||||
item["availability"] = bedrock.get_foundation_model_availability(modelId=model_id)
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
item["availability_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]}
|
||||
try:
|
||||
item["agreement_offers"] = bedrock.list_foundation_model_agreement_offers(modelId=model_id, offerType="ALL")
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
item["agreement_offers_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]}
|
||||
item["iam_simulation"] = simulate_bedrock_activation(session, arn, account, region, model_id, proxy, timeout)
|
||||
availability = item.get("availability") or {}
|
||||
simulation = item.get("iam_simulation") or {}
|
||||
can_activate = bool(simulation.get("can_activate"))
|
||||
authorized = str(availability.get("authorizationStatus") or "").lower() in ("authorized", "available")
|
||||
if can_activate or authorized:
|
||||
findings.append(item)
|
||||
else:
|
||||
code = (item.get("availability_error") or item.get("foundation_model_error") or {}).get("code") or "checked"
|
||||
attempts.append(f"{region}:{model_id}:can_activate={can_activate}:authorization={availability.get('authorizationStatus') or code}")
|
||||
return {"enabled": bool(findings), "findings": findings, "message": "; ".join(attempts[:10])}
|
||||
|
||||
|
||||
def check_bedrock(session, proxy=None, timeout=20, debug=False, regions=None, models=None, max_attempts=12, live_invoke=False):
|
||||
import json
|
||||
import botocore.exceptions
|
||||
|
||||
attempts = []
|
||||
accepted = []
|
||||
tried = 0
|
||||
model_ids = models or list(BEDROCK_MODEL_TESTS.keys())
|
||||
for region in (regions or BEDROCK_REGIONS):
|
||||
for model_id in model_ids:
|
||||
if max_attempts and tried >= max_attempts:
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"region": first.get("region", ""),
|
||||
"model": first.get("model", ""),
|
||||
"available_models": [f"{item['region']}/{item['model']}" for item in accepted],
|
||||
"message": "Bedrock InvokeModel accepted",
|
||||
}
|
||||
return {"enabled": False, "region": "", "model": "", "available_models": [], "message": "; ".join(attempts[:10]) or "Bedrock probe attempt limit reached"}
|
||||
tried += 1
|
||||
data = BEDROCK_MODEL_TESTS.get(model_id)
|
||||
if data is None:
|
||||
data = ANTHROPIC_MESSAGES_PROBE
|
||||
if live_invoke and data is ANTHROPIC_MESSAGES_PROBE:
|
||||
data = ANTHROPIC_MESSAGES_LIVE_PING
|
||||
client = aws_client(session, "bedrock-runtime", proxy, region, timeout)
|
||||
if debug:
|
||||
print(f" BEDROCK TRY: region={region}, model={model_id}")
|
||||
try:
|
||||
client.invoke_model(body=json.dumps(data), modelId=model_id)
|
||||
if debug:
|
||||
print(" BEDROCK RESULT: invoke_model succeeded")
|
||||
accepted.append({"region": region, "model": model_id, "message": "invoke_model succeeded"})
|
||||
continue
|
||||
except client.exceptions.ValidationException as exc:
|
||||
message = str(exc)
|
||||
if bedrock_validation_allows_invoke(exc):
|
||||
# ValidationException for the intentional bad payload means auth/model access passed.
|
||||
if debug:
|
||||
print(f" BEDROCK RESULT: validation_exception_after_auth: {message[:200]}")
|
||||
accepted.append({"region": region, "model": model_id, "message": message[:300]})
|
||||
else:
|
||||
if debug:
|
||||
print(f" BEDROCK RESULT: validation_rejected: {message[:200]}")
|
||||
attempts.append(f"{region}:{model_id}:validation:{message[:120]}")
|
||||
continue
|
||||
except client.exceptions.AccessDeniedException:
|
||||
if debug:
|
||||
print(" BEDROCK RESULT: access_denied")
|
||||
attempts.append(f"{region}:{model_id}:access_denied")
|
||||
continue
|
||||
except client.exceptions.ResourceNotFoundException:
|
||||
if debug:
|
||||
print(" BEDROCK RESULT: model_not_found")
|
||||
attempts.append(f"{region}:{model_id}:not_found")
|
||||
continue
|
||||
except botocore.exceptions.EndpointConnectionError as exc:
|
||||
if debug:
|
||||
print(f" BEDROCK RESULT: network_error: {str(exc)[:120]}")
|
||||
attempts.append(f"{region}:{model_id}:network:{str(exc)[:80]}")
|
||||
continue
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
code = exc.response.get("Error", {}).get("Code", "ClientError")
|
||||
if debug:
|
||||
print(f" BEDROCK RESULT: {code}: {str(exc)[:160]}")
|
||||
attempts.append(f"{region}:{model_id}:{code}")
|
||||
continue
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"region": first.get("region", ""),
|
||||
"model": first.get("model", ""),
|
||||
"available_models": [f"{item['region']}/{item['model']}" for item in accepted],
|
||||
"message": "Bedrock InvokeModel accepted",
|
||||
}
|
||||
return {"enabled": False, "region": "", "model": "", "available_models": [], "message": "; ".join(attempts[:10])}
|
||||
|
||||
|
||||
def inspect_iam(session, arn, proxy=None, timeout=20):
|
||||
import botocore.exceptions
|
||||
|
||||
output = {"admin": False, "quarantined": False, "policy_check": "not_checked", "message": ""}
|
||||
if ":user/" not in arn:
|
||||
output["policy_check"] = "not_user_arn"
|
||||
return output
|
||||
username = arn.rsplit("/", 1)[1]
|
||||
try:
|
||||
iam = aws_client(session, "iam", proxy, timeout=timeout)
|
||||
policies = iam.list_attached_user_policies(UserName=username).get("AttachedPolicies", [])
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
code = exc.response.get("Error", {}).get("Code", "")
|
||||
output["policy_check"] = "access_denied" if code == "AccessDenied" else "error"
|
||||
output["message"] = str(exc)
|
||||
return output
|
||||
output["policy_check"] = "ok"
|
||||
for policy in policies:
|
||||
name = policy.get("PolicyName", "")
|
||||
if "AWSCompromisedKeyQuarantine" in name:
|
||||
output["quarantined"] = True
|
||||
if name == "AdministratorAccess":
|
||||
output["admin"] = True
|
||||
return output
|
||||
|
||||
|
||||
def check_key(
|
||||
key,
|
||||
probe_bedrock=False,
|
||||
bedrock_debug=False,
|
||||
proxy=None,
|
||||
timeout=20,
|
||||
bedrock_regions=None,
|
||||
bedrock_models=None,
|
||||
bedrock_max_attempts=12,
|
||||
bedrock_live_invoke=False,
|
||||
probe_bedrock_management=False,
|
||||
):
|
||||
try:
|
||||
import boto3
|
||||
import botocore.exceptions
|
||||
except ImportError as exc:
|
||||
return {"status": "UNKNOWN", "message": f"boto3/botocore missing: {exc}"}
|
||||
|
||||
access_key, secret = key.split(":", 1)
|
||||
session = boto3.Session(aws_access_key_id=access_key, aws_secret_access_key=secret)
|
||||
try:
|
||||
identity = aws_client(session, "sts", proxy, timeout=timeout).get_caller_identity()
|
||||
except botocore.exceptions.EndpointConnectionError as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
except botocore.exceptions.ClientError as exc:
|
||||
code = exc.response.get("Error", {}).get("Code", "")
|
||||
status = "DEAD" if is_dead_aws_error(code) else "ACCESS_DENIED"
|
||||
return {"status": status, "code": code, "message": str(exc)}
|
||||
except Exception as exc:
|
||||
return {"status": "UNKNOWN", "message": str(exc)}
|
||||
|
||||
arn = identity.get("Arn", "")
|
||||
if is_canary_arn(arn):
|
||||
return {
|
||||
"status": "CANARY",
|
||||
"account": identity.get("Account", ""),
|
||||
"arn": arn,
|
||||
"admin": False,
|
||||
"quarantined": False,
|
||||
"iam_policy_check": "skipped_canary",
|
||||
"bedrock_enabled": False,
|
||||
"bedrock_region": "",
|
||||
"bedrock_model": "",
|
||||
"bedrock_message": "skipped_canary",
|
||||
"bedrock_management_enabled": False,
|
||||
"bedrock_management_message": "skipped_canary",
|
||||
"message": "canary credential detected from STS arn; skipped IAM/Bedrock probes",
|
||||
}
|
||||
|
||||
iam_info = inspect_iam(session, arn, proxy, timeout)
|
||||
bedrock_info = {"enabled": False, "region": "", "model": "", "message": "not_checked"}
|
||||
bedrock_management_info = {"enabled": False, "findings": [], "message": "not_checked"}
|
||||
if probe_bedrock:
|
||||
bedrock_info = check_bedrock(session, proxy, timeout, bedrock_debug, bedrock_regions, bedrock_models, bedrock_max_attempts, bedrock_live_invoke)
|
||||
if probe_bedrock_management:
|
||||
bedrock_management_info = check_bedrock_management(
|
||||
session,
|
||||
arn,
|
||||
identity.get("Account", ""),
|
||||
proxy,
|
||||
timeout,
|
||||
bedrock_regions,
|
||||
bedrock_models,
|
||||
bedrock_max_attempts,
|
||||
bedrock_debug,
|
||||
)
|
||||
|
||||
if iam_info.get("quarantined"):
|
||||
status = "QUARANTINED"
|
||||
elif bedrock_info.get("enabled"):
|
||||
status = "BEDROCK"
|
||||
else:
|
||||
status = "ADMIN" if iam_info.get("admin") else "VALID"
|
||||
|
||||
message_parts = [
|
||||
"sts_ok",
|
||||
f"iam_policy_check={iam_info.get('policy_check')}",
|
||||
]
|
||||
if probe_bedrock:
|
||||
message_parts.append(f"bedrock_enabled={bedrock_info.get('enabled')}")
|
||||
if bedrock_info.get("region"):
|
||||
message_parts.append(f"bedrock_region={bedrock_info.get('region')}")
|
||||
if bedrock_info.get("model"):
|
||||
message_parts.append(f"bedrock_model={bedrock_info.get('model')}")
|
||||
if probe_bedrock_management:
|
||||
findings = bedrock_management_info.get("findings") or []
|
||||
can_activate = any((item.get("iam_simulation") or {}).get("can_activate") for item in findings)
|
||||
message_parts.append(f"bedrock_mgmt_enabled={bedrock_management_info.get('enabled')}")
|
||||
message_parts.append(f"bedrock_can_activate={can_activate}")
|
||||
if iam_info.get("message") and iam_info.get("policy_check") != "access_denied":
|
||||
message_parts.append(iam_info.get("message")[:300])
|
||||
|
||||
return {
|
||||
"status": status,
|
||||
"account": identity.get("Account", ""),
|
||||
"arn": arn,
|
||||
"admin": iam_info.get("admin", False),
|
||||
"quarantined": iam_info.get("quarantined", False),
|
||||
"iam_policy_check": iam_info.get("policy_check"),
|
||||
"bedrock_enabled": bedrock_info.get("enabled"),
|
||||
"bedrock_region": bedrock_info.get("region"),
|
||||
"bedrock_model": bedrock_info.get("model"),
|
||||
"bedrock_available_models": bedrock_info.get("available_models") or [],
|
||||
"bedrock_message": bedrock_info.get("message"),
|
||||
"bedrock_management_enabled": bedrock_management_info.get("enabled"),
|
||||
"bedrock_management_findings": bedrock_management_info.get("findings") or [],
|
||||
"bedrock_management_message": bedrock_management_info.get("message"),
|
||||
"message": "; ".join(message_parts),
|
||||
}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "AWS")
|
||||
message = result.get("message", "")
|
||||
if result.get("status") == "BEDROCK":
|
||||
models = result.get("bedrock_available_models") or []
|
||||
model_text = ",".join(str(item) for item in models) or f"{result.get('bedrock_region', '')}/{result.get('bedrock_model', '')}".strip("/")
|
||||
message = f"{message}; models={model_text}"
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], message, result.get("arn", source),
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, "AWS")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="AWS key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--probe-bedrock", action="store_true")
|
||||
parser.add_argument("--bedrock-debug", action="store_true")
|
||||
parser.add_argument("--bedrock-regions", default=",".join(BEDROCK_REGIONS))
|
||||
parser.add_argument("--bedrock-models", default=",".join(BEDROCK_MODEL_TESTS))
|
||||
parser.add_argument("--bedrock-max-attempts", type=int, default=12)
|
||||
parser.add_argument("--bedrock-live-invoke", action="store_true")
|
||||
parser.add_argument("--probe-bedrock-management", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
|
||||
known = set(known_statuses)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_valid:
|
||||
retry_statuses.update({"VALID", "BEDROCK", "ADMIN"})
|
||||
processed = 0
|
||||
skipped = 0
|
||||
bedrock_regions = [item.strip() for item in str(args.bedrock_regions or "").split(",") if item.strip()]
|
||||
bedrock_models = [item.strip() for item in str(args.bedrock_models or "").split(",") if item.strip()]
|
||||
for key, source, finding in extract_candidates(args.input, args.plain):
|
||||
if should_skip_key(
|
||||
key, checked, known, args, retry_statuses,
|
||||
service=SERVICE, source=source, finding=finding, detector="AWS", known_statuses=known_statuses,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] AWS candidate {mask_secret(key)} from {source}")
|
||||
if is_canary_finding(finding):
|
||||
result = {
|
||||
"status": "CANARY",
|
||||
"message": "canary credential detected in TruffleHog ExtraData; skipped AWS API probes",
|
||||
}
|
||||
else:
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(
|
||||
key,
|
||||
args.probe_bedrock,
|
||||
args.bedrock_debug,
|
||||
proxy,
|
||||
args.timeout,
|
||||
bedrock_regions,
|
||||
bedrock_models,
|
||||
args.bedrock_max_attempts,
|
||||
args.bedrock_live_invoke,
|
||||
args.probe_bedrock_management,
|
||||
)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
if args.probe_bedrock:
|
||||
print(
|
||||
" BEDROCK PING: "
|
||||
f"enabled={result.get('bedrock_enabled')}, "
|
||||
f"region={result.get('bedrock_region') or '-'}, "
|
||||
f"model={result.get('bedrock_model') or '-'}"
|
||||
)
|
||||
if result.get('bedrock_message'):
|
||||
print(f" BEDROCK RESPONSE: {str(result.get('bedrock_message'))[:300]}")
|
||||
if args.probe_bedrock_management:
|
||||
findings = result.get("bedrock_management_findings") or []
|
||||
can_activate = any((item.get("iam_simulation") or {}).get("can_activate") for item in findings)
|
||||
print(
|
||||
" BEDROCK MGMT: "
|
||||
f"enabled={result.get('bedrock_management_enabled')}, "
|
||||
f"can_activate={can_activate}, "
|
||||
f"findings={len(findings)}"
|
||||
)
|
||||
if result.get("bedrock_management_message"):
|
||||
print(f" BEDROCK MGMT RESPONSE: {str(result.get('bedrock_management_message'))[:300]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,946 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
|
||||
|
||||
from keycheck_candidates import extract_azure_foundry_parts
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
append_status,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
keycheck_input_mode,
|
||||
iter_findings,
|
||||
iter_bounded_text_lines,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "azure"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "azureChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "azureResults.jsonl")
|
||||
AZURE_OPENAI_LLM_FILE = os.path.join(OUTPUT_DIR, "azureOpenAILLM.txt")
|
||||
AZURE_OPENAI_PLAIN_FILE = os.path.join(OUTPUT_DIR, "azureOpenAI.txt")
|
||||
AZURE_FOUNDRY_PLAIN_FILE = os.path.join(OUTPUT_DIR, "azureFoundry.txt")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "azureAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "azureDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "azureRestricted.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "azureNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "azureUnknown.txt"),
|
||||
"OPENAI_UNRESOLVED": os.path.join(OUTPUT_DIR, "azureOpenAIUnresolved.txt"),
|
||||
"OPENAI_BAD_ENDPOINT": os.path.join(OUTPUT_DIR, "azureOpenAIBadEndpoint.txt"),
|
||||
"FOUNDRY": os.path.join(OUTPUT_DIR, "azureFoundryLLM.txt"),
|
||||
"FOUNDRY_UNRESOLVED": os.path.join(OUTPUT_DIR, "azureFoundryUnresolved.txt"),
|
||||
"FOUNDRY_BAD_ENDPOINT": os.path.join(OUTPUT_DIR, "azureFoundryBadEndpoint.txt"),
|
||||
}
|
||||
|
||||
AZURE_OPENAI_ENDPOINT_RE = re.compile(r"([a-z0-9-]+\.openai\.azure\.com)", re.IGNORECASE)
|
||||
AZURE_FOUNDRY_HOST_RE = r"[a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models\.ai\.azure\.com|services\.ai\.azure\.com|inference\.ai\.azure\.com)"
|
||||
AZURE_FOUNDRY_ENDPOINT_RE = re.compile(r"((?:https?://)?" + AZURE_FOUNDRY_HOST_RE + r"(?:/[^\s:\"'<>\\]*)?)", re.IGNORECASE)
|
||||
AZURE_OPENAI_DEPLOYMENTS_API_VERSION = "2023-03-15-preview"
|
||||
AZURE_OPENAI_CHAT_API_VERSION = "2024-02-15-preview"
|
||||
AZURE_FOUNDRY_API_VERSION = "2024-05-01-preview"
|
||||
AZURE_FOUNDRY_KEY_ASSIGNMENT_RE = re.compile(
|
||||
r"(?is)(?:authorization|api[_-]?key|key|token|secret|credential|bearer)[^\n:=]{0,80}[:=]\s*[\"']?(?:bearer\s+)?([A-Za-z0-9_./+=\-]{20,512})"
|
||||
)
|
||||
NON_FOUNDRY_KEY_PREFIXES = (
|
||||
"sk-", "sk_", "sk-or-", "xai-", "ghp_", "gho_", "ghu_", "ghs_", "ghr_", "github_pat_",
|
||||
"glpat-", "glrt-", "hf_", "AIza", "AQ.", "zai-", "gsk_", "r8_", "nvapi-",
|
||||
)
|
||||
|
||||
|
||||
def transaction_status_files():
|
||||
return {**STATUS_FILES, "AUX_OPENAI_LLM": AZURE_OPENAI_LLM_FILE}
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, AZURE_OPENAI_LLM_FILE, AZURE_OPENAI_PLAIN_FILE, AZURE_FOUNDRY_PLAIN_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, transaction_status_files())
|
||||
|
||||
|
||||
def parse_azure_sp(raw_v2):
|
||||
try:
|
||||
data = json.loads(raw_v2)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
client_secret = data.get("clientSecret") or data.get("client_secret")
|
||||
client_id = data.get("clientId") or data.get("client_id")
|
||||
tenant_id = data.get("tenantId") or data.get("tenant_id")
|
||||
if not all([client_secret, client_id, tenant_id]):
|
||||
return None
|
||||
return {"client_secret": client_secret, "client_id": client_id, "tenant_id": tenant_id}
|
||||
|
||||
|
||||
def scanner_context_text(finding):
|
||||
context = finding.get("ScannerContext") if isinstance(finding, dict) else None
|
||||
if isinstance(context, dict):
|
||||
return str(context.get("nearby") or "")
|
||||
return ""
|
||||
|
||||
|
||||
def foundry_keyish(value):
|
||||
text = re.sub(r"(?i)^bearer\s+", "", str(value or "").strip().strip('"\'`,;')).strip()
|
||||
lower = text.lower()
|
||||
if not (20 <= len(text) <= 512):
|
||||
return False
|
||||
if any(marker in lower for marker in ("http://", "https://", "{{", "${", "<", "azure.com")):
|
||||
return False
|
||||
if any(ch.isspace() for ch in text):
|
||||
return False
|
||||
if re.match(r"(?i)^(?:authorization|api[_-]?key|key|token|secret|credential|bearer)\s*[:=]", text):
|
||||
return False
|
||||
if text.startswith(NON_FOUNDRY_KEY_PREFIXES):
|
||||
return False
|
||||
return bool(re.search(r"[A-Za-z]", text) and re.search(r"[0-9]", text))
|
||||
|
||||
|
||||
def normalize_foundry_endpoint(value):
|
||||
text = str(value or "").strip().strip('"\'`,;')
|
||||
if not text:
|
||||
return ""
|
||||
split_text = text if re.match(r"(?i)^https?://", text) else "https://" + text
|
||||
try:
|
||||
parsed = urlsplit(split_text)
|
||||
host = parsed.netloc or parsed.path.split("/", 1)[0]
|
||||
path = parsed.path if parsed.netloc else ("/" + parsed.path.split("/", 1)[1] if "/" in parsed.path else "")
|
||||
except Exception:
|
||||
host, path = re.sub(r"(?i)^https?://", "", text).split("/", 1)[0], ""
|
||||
path = path.rstrip(".,;:)]}/")
|
||||
terminal_routes = (
|
||||
("/models/chat/completions", ""),
|
||||
("/openai/v1/chat/completions", "/openai/v1"),
|
||||
("/v1/chat/completions", "/v1"),
|
||||
("/chat/completions", ""),
|
||||
("/v1/models", "/v1"),
|
||||
("/models", ""),
|
||||
)
|
||||
lower_path = path.lower()
|
||||
for suffix, replacement in terminal_routes:
|
||||
if lower_path.endswith(suffix):
|
||||
path = path[:-len(suffix)] + replacement
|
||||
break
|
||||
return (host + path).strip("/").lower()
|
||||
|
||||
|
||||
def split_foundry_endpoint_key(text):
|
||||
candidate = str(text or "").strip().split("\t", 1)[0].strip()
|
||||
if not candidate:
|
||||
return None
|
||||
endpoint_match = AZURE_FOUNDRY_ENDPOINT_RE.search(candidate)
|
||||
if not endpoint_match:
|
||||
return None
|
||||
endpoint = normalize_foundry_endpoint(endpoint_match.group(1))
|
||||
before = candidate[:endpoint_match.start()].replace("https://", " ").replace("http://", " ").strip(" \t:=,;'\"/")
|
||||
after = candidate[endpoint_match.end():].replace("https://", " ").replace("http://", " ").strip(" \t:=,;'\"/")
|
||||
for key in (after, before):
|
||||
if foundry_keyish(key):
|
||||
return {"key": key, "endpoint": endpoint}
|
||||
return None
|
||||
|
||||
|
||||
def foundry_context_values(finding):
|
||||
values = []
|
||||
for value in (finding.get("Raw"), finding.get("RawV2")) if isinstance(finding, dict) else ():
|
||||
if value:
|
||||
text = str(value)
|
||||
values.append(text)
|
||||
values.extend(str(item) for item in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(text))
|
||||
context = scanner_context_text(finding)
|
||||
if context:
|
||||
values.append(context)
|
||||
values.extend(str(item) for item in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(context or ""))
|
||||
extra = finding.get("ExtraData") if isinstance(finding, dict) else None
|
||||
if isinstance(extra, dict):
|
||||
values.extend(str(value) for value in extra.values() if isinstance(value, str))
|
||||
return values
|
||||
|
||||
|
||||
def parse_azure_openai(raw, raw_v2, finding):
|
||||
key = raw or ""
|
||||
endpoint = ""
|
||||
raw_v2 = raw_v2 or ""
|
||||
match = re.match(r"^([a-f0-9]{32}):(.+\.openai\.azure\.com)$", raw_v2, re.IGNORECASE)
|
||||
if match:
|
||||
key = match.group(1)
|
||||
endpoint = match.group(2)
|
||||
if not endpoint:
|
||||
context_match = AZURE_OPENAI_ENDPOINT_RE.search(scanner_context_text(finding))
|
||||
if context_match:
|
||||
endpoint = context_match.group(1)
|
||||
if not key:
|
||||
return None
|
||||
return {"key": key, "endpoint": endpoint}
|
||||
|
||||
|
||||
def parse_azure_openai_line(line):
|
||||
text = str(line or "").strip()
|
||||
if not text:
|
||||
return None
|
||||
text = text.split("\t", 1)[0].strip()
|
||||
if ":" in text:
|
||||
endpoint, key = text.split(":", 1)
|
||||
if AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint.strip()) and key.strip():
|
||||
return {"endpoint": endpoint.strip(), "key": key.strip()}
|
||||
endpoint_match = AZURE_OPENAI_ENDPOINT_RE.search(text)
|
||||
key_match = re.search(r"\b[a-f0-9]{32}\b", text, re.IGNORECASE)
|
||||
if endpoint_match and key_match:
|
||||
return {"endpoint": endpoint_match.group(1), "key": key_match.group(0)}
|
||||
if key_match:
|
||||
return {"endpoint": "", "key": key_match.group(0)}
|
||||
return None
|
||||
|
||||
|
||||
def parse_azure_foundry(raw, raw_v2, finding):
|
||||
key = raw or ""
|
||||
endpoint = ""
|
||||
raw_v2 = raw_v2 or ""
|
||||
split = split_foundry_endpoint_key(raw_v2) or split_foundry_endpoint_key(raw)
|
||||
if not split:
|
||||
split = extract_azure_foundry_parts(raw, raw_v2)
|
||||
if split:
|
||||
key = split["key"]
|
||||
endpoint = split["endpoint"]
|
||||
if not endpoint:
|
||||
for endpoint_text in (raw, raw_v2, scanner_context_text(finding)):
|
||||
context_match = AZURE_FOUNDRY_ENDPOINT_RE.search(str(endpoint_text or ""))
|
||||
if context_match:
|
||||
endpoint = normalize_foundry_endpoint(context_match.group(1))
|
||||
break
|
||||
if not foundry_keyish(key):
|
||||
for value in foundry_context_values(finding):
|
||||
if foundry_keyish(value):
|
||||
key = value.strip().strip('"\'`,;')
|
||||
break
|
||||
if not key or not foundry_keyish(key):
|
||||
return None
|
||||
return {"key": key.strip().strip('"\'`,;'), "endpoint": normalize_foundry_endpoint(endpoint)}
|
||||
|
||||
|
||||
def parse_azure_foundry_line(line):
|
||||
parts = str(line or "").strip().split("\t", 1)
|
||||
text = parts[0].strip()
|
||||
if not text:
|
||||
return None
|
||||
split = split_foundry_endpoint_key(text)
|
||||
parsed = split if split else ({"key": text, "endpoint": ""} if foundry_keyish(text) else None)
|
||||
if not parsed:
|
||||
return None
|
||||
if len(parts) > 1:
|
||||
try:
|
||||
metadata = json.loads(parts[1])
|
||||
except ValueError:
|
||||
metadata = {}
|
||||
if isinstance(metadata, dict):
|
||||
parsed["finding_uid"] = metadata.get("finding_uid") or ""
|
||||
parsed["origin"] = metadata.get("origin") or ""
|
||||
return parsed
|
||||
|
||||
|
||||
def parse_azure_acr(raw_v2):
|
||||
try:
|
||||
data = json.loads(raw_v2)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
username = data.get("username")
|
||||
password = data.get("password")
|
||||
if not username or not password:
|
||||
return None
|
||||
return {"username": username, "password": password}
|
||||
|
||||
|
||||
def azure_openai_key(parsed):
|
||||
endpoint = parsed.get("endpoint") or ""
|
||||
return f"{endpoint}:{parsed['key']}" if endpoint else parsed["key"]
|
||||
|
||||
|
||||
def azure_acr_key(parsed):
|
||||
return f"{parsed['username']}:{parsed['password']}"
|
||||
|
||||
|
||||
def azure_sp_key(parsed):
|
||||
return f"{parsed['tenant_id']}:{parsed['client_id']}:{parsed['client_secret']}"
|
||||
|
||||
|
||||
def azure_foundry_key(parsed):
|
||||
endpoint = parsed.get("endpoint") or ""
|
||||
return f"{endpoint}:{parsed['key']}" if endpoint else parsed["key"]
|
||||
|
||||
|
||||
def extract_candidates(input_file):
|
||||
seen_plain = set()
|
||||
foundry_detectors = {"AzureFoundryEndpointBeforeKey", "AzureFoundryKeyBeforeEndpoint"}
|
||||
detector_names = ["AzureOpenAI", "AzureContainerRegistry", "Azure", *sorted(foundry_detectors)]
|
||||
for item in iter_findings(input_file, detector_names):
|
||||
candidate_kind = item.get('candidate_kind') or ''
|
||||
if keycheck_input_mode() == 'postgres' and candidate_kind:
|
||||
secret_text = item.get('credential_secret_text') or ''
|
||||
secret_json = item.get('credential_secret_json') or ''
|
||||
endpoint = item.get('credential_endpoint') or ''
|
||||
parsed = None
|
||||
detector = ''
|
||||
if candidate_kind == 'azure_service_principal':
|
||||
parsed = parse_azure_sp(secret_json)
|
||||
detector = 'Azure'
|
||||
key = azure_sp_key(parsed) if parsed else ''
|
||||
elif candidate_kind == 'azure_container_registry':
|
||||
parsed = parse_azure_acr(secret_json)
|
||||
detector = 'AzureContainerRegistry'
|
||||
key = azure_acr_key(parsed) if parsed else ''
|
||||
elif candidate_kind == 'azure_openai':
|
||||
parsed = {'key': secret_text, 'endpoint': endpoint} if secret_text else None
|
||||
detector = 'AzureOpenAI'
|
||||
key = azure_openai_key(parsed) if parsed else ''
|
||||
elif candidate_kind == 'azure_foundry':
|
||||
parsed = (
|
||||
{'key': secret_text, 'endpoint': normalize_foundry_endpoint(endpoint)}
|
||||
if foundry_keyish(secret_text) and endpoint else
|
||||
parse_azure_foundry(secret_text, '', item.get('finding') or {})
|
||||
)
|
||||
if parsed and endpoint:
|
||||
parsed['endpoint'] = normalize_foundry_endpoint(endpoint)
|
||||
detector = 'AzureFoundry'
|
||||
key = azure_foundry_key(parsed) if parsed else ''
|
||||
else:
|
||||
key = ''
|
||||
if parsed and key:
|
||||
yield key, detector, item['source'], item['finding'], parsed
|
||||
continue
|
||||
unresolved_detector = {
|
||||
'azure_openai': 'AzureOpenAI',
|
||||
'azure_foundry': 'AzureFoundry',
|
||||
'azure_container_registry': 'AzureContainerRegistry',
|
||||
'azure_service_principal': 'Azure',
|
||||
}.get(candidate_kind, 'Azure')
|
||||
unresolved_key = secret_text or secret_json or candidate_kind
|
||||
if unresolved_key:
|
||||
yield unresolved_key, unresolved_detector, item['source'], item['finding'], {
|
||||
'_unresolved_candidate': True,
|
||||
'candidate_kind': candidate_kind,
|
||||
}
|
||||
continue
|
||||
if item["detector"] == "AzureOpenAI":
|
||||
foundry = parse_azure_foundry(item["raw"], item["raw_v2"], item["finding"])
|
||||
if foundry and foundry.get("endpoint"):
|
||||
key = azure_foundry_key(foundry)
|
||||
yield key, "AzureFoundry", item["source"], item["finding"], foundry
|
||||
parsed = parse_azure_openai(item["raw"], item["raw_v2"], item["finding"])
|
||||
if not parsed:
|
||||
continue
|
||||
key = azure_openai_key(parsed)
|
||||
yield key, "AzureOpenAI", item["source"], item["finding"], parsed
|
||||
continue
|
||||
if item["detector"] == "AzureContainerRegistry":
|
||||
parsed = parse_azure_acr(item["raw_v2"])
|
||||
if not parsed:
|
||||
continue
|
||||
key = azure_acr_key(parsed)
|
||||
yield key, "AzureContainerRegistry", item["source"], item["finding"], parsed
|
||||
continue
|
||||
if item["detector"] == "Azure":
|
||||
parsed = parse_azure_sp(item["raw_v2"])
|
||||
if not parsed:
|
||||
continue
|
||||
key = azure_sp_key(parsed)
|
||||
yield key, "Azure", item["source"], item["finding"], parsed
|
||||
continue
|
||||
|
||||
if item["detector"] in foundry_detectors:
|
||||
foundry = parse_azure_foundry(item.get("raw"), item.get("raw_v2"), item.get("finding"))
|
||||
if not foundry or not foundry.get("endpoint"):
|
||||
continue
|
||||
key = azure_foundry_key(foundry)
|
||||
yield key, "AzureFoundry", item["source"], item["finding"], foundry
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and os.path.exists(AZURE_FOUNDRY_PLAIN_FILE):
|
||||
for line_num, line in enumerate(iter_bounded_text_lines(AZURE_FOUNDRY_PLAIN_FILE), 1):
|
||||
parsed = parse_azure_foundry_line(line)
|
||||
if not parsed:
|
||||
continue
|
||||
key = azure_foundry_key(parsed)
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, "AzureFoundry", f"{AZURE_FOUNDRY_PLAIN_FILE}:{line_num}", {}, parsed
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and os.path.exists(AZURE_OPENAI_PLAIN_FILE):
|
||||
for line_num, line in enumerate(iter_bounded_text_lines(AZURE_OPENAI_PLAIN_FILE), 1):
|
||||
parsed = parse_azure_openai_line(line)
|
||||
if not parsed:
|
||||
continue
|
||||
key = azure_openai_key(parsed)
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, "AzureOpenAI", f"{AZURE_OPENAI_PLAIN_FILE}:{line_num}", {}, parsed
|
||||
|
||||
|
||||
def permission_matches(action, pattern):
|
||||
action = str(action or "").lower()
|
||||
pattern = str(pattern or "").lower()
|
||||
if pattern == "*":
|
||||
return True
|
||||
if pattern.endswith("/*"):
|
||||
return action.startswith(pattern[:-1])
|
||||
return action == pattern
|
||||
|
||||
|
||||
def has_action(actions, wanted):
|
||||
return any(permission_matches(wanted, action) for action in actions)
|
||||
|
||||
|
||||
def probe_azure_rbac(access_token, proxy, timeout, max_subscriptions=3):
|
||||
if not access_token:
|
||||
return {"azure_rbac_level": "unknown", "message": "rbac_probe=no_access_token"}
|
||||
headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"}
|
||||
try:
|
||||
response = requests.get(
|
||||
"https://management.azure.com/subscriptions?api-version=2020-01-01",
|
||||
headers=headers,
|
||||
proxies=proxy,
|
||||
timeout=timeout,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
return {"azure_rbac_level": "network", "message": f"rbac_probe_network={str(exc)[:200]}"}
|
||||
if response.status_code == 403:
|
||||
return {"azure_rbac_level": "token_only", "azure_subscription_count": 0, "message": "rbac_probe=subscriptions_forbidden"}
|
||||
if response.status_code >= 400:
|
||||
return {"azure_rbac_level": "unknown", "azure_rbac_http_status": response.status_code, "message": f"rbac_probe_http={response.status_code}:{request_error_message(response)[:200]}"}
|
||||
payload = response.json()
|
||||
subscriptions = payload.get("value") if isinstance(payload, dict) else []
|
||||
subscriptions = subscriptions or []
|
||||
sub_ids = [item.get("subscriptionId") for item in subscriptions if isinstance(item, dict) and item.get("subscriptionId")]
|
||||
if not sub_ids:
|
||||
return {"azure_rbac_level": "token_only", "azure_subscription_count": 0, "message": "rbac_probe=no_subscriptions"}
|
||||
|
||||
all_actions = set()
|
||||
all_not_actions = set()
|
||||
permission_errors = []
|
||||
for sub_id in sub_ids[:max_subscriptions]:
|
||||
url = f"https://management.azure.com/subscriptions/{sub_id}/providers/Microsoft.Authorization/permissions?api-version=2022-04-01"
|
||||
try:
|
||||
perms_response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
permission_errors.append(f"{sub_id}:network:{str(exc)[:120]}")
|
||||
continue
|
||||
if perms_response.status_code >= 400:
|
||||
permission_errors.append(f"{sub_id}:http_{perms_response.status_code}:{request_error_message(perms_response)[:120]}")
|
||||
continue
|
||||
data = perms_response.json().get("value") or []
|
||||
for item in data:
|
||||
for action in item.get("actions") or []:
|
||||
all_actions.add(str(action))
|
||||
for action in item.get("notActions") or []:
|
||||
all_not_actions.add(str(action))
|
||||
|
||||
can_all = has_action(all_actions, "*")
|
||||
can_assign_roles = has_action(all_actions, "Microsoft.Authorization/roleAssignments/write") and not has_action(all_not_actions, "Microsoft.Authorization/roleAssignments/write")
|
||||
can_manage_cognitive = any(
|
||||
has_action(all_actions, action) for action in (
|
||||
"Microsoft.CognitiveServices/accounts/write",
|
||||
"Microsoft.CognitiveServices/accounts/deployments/write",
|
||||
"Microsoft.CognitiveServices/*",
|
||||
)
|
||||
) or can_all
|
||||
can_manage_ml = any(
|
||||
has_action(all_actions, action) for action in (
|
||||
"Microsoft.MachineLearningServices/workspaces/write",
|
||||
"Microsoft.MachineLearningServices/*",
|
||||
)
|
||||
) or can_all
|
||||
can_deploy_resources = has_action(all_actions, "Microsoft.Resources/deployments/write") or can_all
|
||||
can_manage_ai = can_manage_cognitive or can_manage_ml
|
||||
|
||||
if can_all and can_assign_roles:
|
||||
level = "owner_like"
|
||||
elif can_all:
|
||||
level = "contributor_like"
|
||||
elif can_manage_ai:
|
||||
level = "ai_manager"
|
||||
elif all_actions:
|
||||
level = "limited"
|
||||
else:
|
||||
level = "subscriptions_visible"
|
||||
|
||||
message = (
|
||||
f"rbac_probe={level}; subscriptions={len(sub_ids)}; "
|
||||
f"can_manage_ai={can_manage_ai}; can_assign_roles={can_assign_roles}; can_deploy_resources={can_deploy_resources}"
|
||||
)
|
||||
if permission_errors and not all_actions:
|
||||
message += "; permission_errors=" + " | ".join(permission_errors[:3])
|
||||
return {
|
||||
"azure_rbac_level": level,
|
||||
"azure_subscription_count": len(sub_ids),
|
||||
"azure_subscription_ids": sub_ids[:10],
|
||||
"azure_can_manage_ai": can_manage_ai,
|
||||
"azure_can_manage_cognitive": can_manage_cognitive,
|
||||
"azure_can_manage_ml": can_manage_ml,
|
||||
"azure_can_assign_roles": can_assign_roles,
|
||||
"azure_can_deploy_resources": can_deploy_resources,
|
||||
"azure_permission_actions_sample": sorted(all_actions)[:40],
|
||||
"message": message,
|
||||
}
|
||||
|
||||
|
||||
def check_service_principal(parsed, proxy, timeout):
|
||||
url = f"https://login.microsoftonline.com/{parsed['tenant_id']}/oauth2/v2.0/token"
|
||||
payload = {
|
||||
"client_id": parsed["client_id"],
|
||||
"client_secret": parsed["client_secret"],
|
||||
"scope": "https://management.azure.com/.default",
|
||||
"grant_type": "client_credentials",
|
||||
}
|
||||
try:
|
||||
response = requests.post(url, data=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
rbac = probe_azure_rbac(data.get("access_token"), proxy, timeout)
|
||||
message = "token issued"
|
||||
if rbac.get("message"):
|
||||
message = f"{message}; {rbac.get('message')}"
|
||||
return {
|
||||
"status": "VALID",
|
||||
"tenant_id": parsed["tenant_id"],
|
||||
"client_id": parsed["client_id"],
|
||||
"expires_in": data.get("expires_in"),
|
||||
"message": message,
|
||||
**rbac,
|
||||
}
|
||||
message = request_error_message(response)
|
||||
lower = message.lower()
|
||||
if response.status_code in (400, 401) and ("invalid_client" in lower or "invalid_grant" in lower):
|
||||
return {"status": "DEAD", "http_status": response.status_code, "message": message}
|
||||
if response.status_code in (401, 403):
|
||||
return {"status": "RESTRICTED", "http_status": response.status_code, "message": message}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message}
|
||||
|
||||
|
||||
def azure_openai_deployment_ids(payload):
|
||||
data = payload.get("data") if isinstance(payload, dict) else None
|
||||
if data is None and isinstance(payload, dict):
|
||||
data = payload.get("value")
|
||||
deployments = []
|
||||
for item in data or []:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
deployment_id = item.get("id") or item.get("name")
|
||||
model = item.get("model") or item.get("modelName") or ""
|
||||
if deployment_id:
|
||||
deployments.append({"id": deployment_id, "model": model})
|
||||
return deployments
|
||||
|
||||
|
||||
def endpoint_url(endpoint, path):
|
||||
endpoint = str(endpoint or "").strip().rstrip("/")
|
||||
if not endpoint.startswith("http://") and not endpoint.startswith("https://"):
|
||||
endpoint = "https://" + endpoint
|
||||
path = "/" + str(path or "").lstrip("/")
|
||||
parsed = urlsplit(endpoint)
|
||||
endpoint_path = parsed.path.rstrip("/")
|
||||
if endpoint_path and path.lower().startswith(endpoint_path.lower() + "/"):
|
||||
path = path[len(endpoint_path):]
|
||||
return endpoint + path
|
||||
|
||||
|
||||
def azure_model_ids(payload):
|
||||
data = payload.get("data") if isinstance(payload, dict) else payload if isinstance(payload, list) else []
|
||||
if data is None and isinstance(payload, dict):
|
||||
data = payload.get("value") or payload.get("models")
|
||||
models = []
|
||||
for item in data or []:
|
||||
if isinstance(item, str):
|
||||
models.append(item)
|
||||
elif isinstance(item, dict):
|
||||
model_id = item.get("id") or item.get("name") or item.get("model") or item.get("modelName")
|
||||
if model_id:
|
||||
models.append(str(model_id))
|
||||
return models
|
||||
|
||||
|
||||
def endpoint_failure_status(error_text, status):
|
||||
lower = str(error_text or "").lower()
|
||||
if any(item in lower for item in (
|
||||
"name resolution", "no such host", "failed to resolve", "getaddrinfo",
|
||||
"unexpected_eof", "eof occurred in violation of protocol", "ssleoferror",
|
||||
)):
|
||||
return status
|
||||
return "NETWORK"
|
||||
|
||||
|
||||
def probe_azure_openai_chat_route(endpoint, key, deployments, proxy, timeout):
|
||||
if not deployments:
|
||||
return {"route_probe": "no_deployments"}
|
||||
preferred = None
|
||||
for item in deployments:
|
||||
text = f"{item.get('id', '')} {item.get('model', '')}".lower()
|
||||
if any(marker in text for marker in ("gpt", "chat", "turbo", "4o")):
|
||||
preferred = item
|
||||
break
|
||||
deployment = preferred or deployments[0]
|
||||
deployment_id = deployment["id"]
|
||||
url = f"https://{endpoint}/openai/deployments/{deployment_id}/chat/completions?api-version={AZURE_OPENAI_CHAT_API_VERSION}"
|
||||
headers = {"api-key": key, "Content-Type": "application/json"}
|
||||
# Empty messages should fail validation after auth/deployment routing, without generating content.
|
||||
payload = {"messages": [], "max_tokens": 1}
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"route_probe": "network", "route_deployment": deployment_id, "route_message": str(exc)[:500]}
|
||||
message = request_error_message(response)
|
||||
if response.status_code in (200, 400):
|
||||
return {
|
||||
"route_probe": "accepted_auth_route",
|
||||
"route_deployment": deployment_id,
|
||||
"route_model": deployment.get("model", ""),
|
||||
"route_http_status": response.status_code,
|
||||
"route_message": message,
|
||||
}
|
||||
if response.status_code in (401, 403):
|
||||
return {"route_probe": "auth_failed", "route_deployment": deployment_id, "route_http_status": response.status_code, "route_message": message}
|
||||
if response.status_code == 404:
|
||||
return {"route_probe": "not_found", "route_deployment": deployment_id, "route_http_status": 404, "route_message": message}
|
||||
return {"route_probe": "unknown", "route_deployment": deployment_id, "route_http_status": response.status_code, "route_message": message}
|
||||
|
||||
|
||||
def check_azure_openai(parsed, proxy, timeout, probe_openai_route=False):
|
||||
endpoint = (parsed.get("endpoint") or "").strip().strip("/")
|
||||
key = parsed.get("key")
|
||||
if not endpoint:
|
||||
return {"status": "OPENAI_UNRESOLVED", "message": "AzureOpenAI key found without endpoint/resource name"}
|
||||
url = f"https://{endpoint}/openai/deployments?api-version={AZURE_OPENAI_DEPLOYMENTS_API_VERSION}"
|
||||
headers = {"api-key": key, "Content-Type": "application/json"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
first_error = str(exc)
|
||||
def endpoint_failure_status(error_text):
|
||||
lower = str(error_text or "").lower()
|
||||
if any(item in lower for item in (
|
||||
"name resolution", "no such host", "failed to resolve", "getaddrinfo",
|
||||
"unexpected_eof", "eof occurred in violation of protocol", "ssleoferror",
|
||||
)):
|
||||
return {"status": "OPENAI_BAD_ENDPOINT", "endpoint": endpoint, "message": error_text}
|
||||
return None
|
||||
endpoint_status = endpoint_failure_status(first_error)
|
||||
if endpoint_status:
|
||||
return endpoint_status
|
||||
return {"status": "NETWORK", "endpoint": endpoint, "message": first_error}
|
||||
if response.status_code == 200:
|
||||
deployments = azure_openai_deployment_ids(response.json())
|
||||
result = {
|
||||
"status": "VALID",
|
||||
"endpoint": endpoint,
|
||||
"deployment_count": len(deployments),
|
||||
"deployments": [item.get("id") for item in deployments[:20]],
|
||||
"message": f"deployments endpoint accepted key; deployments={len(deployments)}",
|
||||
}
|
||||
if probe_openai_route:
|
||||
result.update(probe_azure_openai_chat_route(endpoint, key, deployments, proxy, timeout))
|
||||
return result
|
||||
message = request_error_message(response)
|
||||
if response.status_code in (401, 403):
|
||||
return {"status": "DEAD", "endpoint": endpoint, "http_status": response.status_code, "message": message}
|
||||
if response.status_code == 404:
|
||||
return {"status": "UNKNOWN", "endpoint": endpoint, "http_status": 404, "message": message}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "endpoint": endpoint, "http_status": response.status_code, "message": message}
|
||||
return {"status": "UNKNOWN", "endpoint": endpoint, "http_status": response.status_code, "message": message}
|
||||
|
||||
|
||||
def foundry_model_routes(endpoint):
|
||||
return [
|
||||
endpoint_url(endpoint, f"/models?api-version={AZURE_FOUNDRY_API_VERSION}"),
|
||||
endpoint_url(endpoint, "/models"),
|
||||
endpoint_url(endpoint, "/v1/models"),
|
||||
]
|
||||
|
||||
|
||||
def foundry_auth_headers(key):
|
||||
return [
|
||||
{"api-key": key, "Content-Type": "application/json"},
|
||||
{"Authorization": f"Bearer {key}", "Content-Type": "application/json"},
|
||||
]
|
||||
|
||||
|
||||
def check_foundry_models(endpoint, key, proxy, timeout):
|
||||
attempts = []
|
||||
auth_failures = 0
|
||||
attempted = 0
|
||||
for url in foundry_model_routes(endpoint):
|
||||
for headers in foundry_auth_headers(key):
|
||||
attempted += 1
|
||||
auth_kind = "bearer" if "Authorization" in headers else "api-key"
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
status = endpoint_failure_status(str(exc), "FOUNDRY_BAD_ENDPOINT")
|
||||
attempts.append(f"{url}:{auth_kind}:network:{str(exc)[:180]}")
|
||||
if status == "FOUNDRY_BAD_ENDPOINT":
|
||||
return {"status": status, "endpoint": endpoint, "message": str(exc)}
|
||||
continue
|
||||
message = request_error_message(response)
|
||||
if response.status_code == 200:
|
||||
models = azure_model_ids(response.json())
|
||||
return {
|
||||
"status": "FOUNDRY",
|
||||
"endpoint": endpoint,
|
||||
"auth_scheme": auth_kind,
|
||||
"model_count": len(models),
|
||||
"models": models[:50],
|
||||
"message": f"models endpoint accepted key; auth={auth_kind}; models={len(models)}",
|
||||
}
|
||||
if response.status_code == 429:
|
||||
return {
|
||||
"status": "FOUNDRY",
|
||||
"endpoint": endpoint,
|
||||
"auth_scheme": auth_kind,
|
||||
"model_count": 0,
|
||||
"models": [],
|
||||
"message": f"models endpoint rate limited after auth; auth={auth_kind}; {message[:200]}",
|
||||
}
|
||||
if response.status_code in (401, 403):
|
||||
auth_failures += 1
|
||||
attempts.append(f"{url}:{auth_kind}:auth_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
if response.status_code == 404:
|
||||
attempts.append(f"{url}:{auth_kind}:http_404:{message[:160]}")
|
||||
continue
|
||||
if response.status_code >= 500:
|
||||
attempts.append(f"{url}:{auth_kind}:server_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
attempts.append(f"{url}:{auth_kind}:http_{response.status_code}:{message[:160]}")
|
||||
if attempted and auth_failures == attempted:
|
||||
return {"status": "DEAD", "endpoint": endpoint, "message": "; ".join(attempts[:4])}
|
||||
return {"status": "UNKNOWN", "endpoint": endpoint, "message": "; ".join(attempts[:4])}
|
||||
|
||||
|
||||
def probe_foundry_route(endpoint, key, models, proxy, timeout):
|
||||
configured = [item.strip() for item in (models or []) if item.strip()]
|
||||
if not configured:
|
||||
return {"foundry_route_probe": "not_configured"}
|
||||
attempts = []
|
||||
accepted = []
|
||||
for model in configured:
|
||||
route_specs = [
|
||||
(endpoint_url(endpoint, f"/models/chat/completions?api-version={AZURE_FOUNDRY_API_VERSION}"), {"model": model, "messages": [], "max_tokens": 1}),
|
||||
(endpoint_url(endpoint, "/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}),
|
||||
(endpoint_url(endpoint, "/v1/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}),
|
||||
(endpoint_url(endpoint, "/openai/v1/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}),
|
||||
]
|
||||
for url, payload in route_specs:
|
||||
for headers in foundry_auth_headers(key):
|
||||
auth_kind = "bearer" if "Authorization" in headers else "api-key"
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
attempts.append(f"{model}:{auth_kind}:network:{str(exc)[:120]}")
|
||||
continue
|
||||
message = request_error_message(response)
|
||||
if response.status_code == 200:
|
||||
accepted.append(model)
|
||||
break
|
||||
if response.status_code == 400 and any(item in message.lower() for item in ("messages", "content", "validation", "empty")):
|
||||
accepted.append(model)
|
||||
break
|
||||
if response.status_code == 429:
|
||||
accepted.append(model)
|
||||
break
|
||||
if response.status_code in (401, 403, 404):
|
||||
attempts.append(f"{model}:{auth_kind}:http_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
attempts.append(f"{model}:{auth_kind}:http_{response.status_code}:{message[:160]}")
|
||||
if model in accepted:
|
||||
break
|
||||
if accepted:
|
||||
return {"foundry_route_probe": "accepted", "foundry_route_models": accepted, "foundry_route_message": "route accepted"}
|
||||
return {"foundry_route_probe": "not_accepted", "foundry_route_models": [], "foundry_route_message": "; ".join(attempts[:8])}
|
||||
|
||||
|
||||
def check_azure_foundry(parsed, proxy, timeout, probe_foundry_route_enabled=False, foundry_models=None):
|
||||
endpoint = (parsed.get("endpoint") or "").strip().strip("/")
|
||||
key = parsed.get("key")
|
||||
if not endpoint:
|
||||
return {"status": "FOUNDRY_UNRESOLVED", "message": "Azure Foundry key found without endpoint"}
|
||||
result = check_foundry_models(endpoint, key, proxy, timeout)
|
||||
if probe_foundry_route_enabled:
|
||||
route = probe_foundry_route(endpoint, key, foundry_models or [], proxy, timeout)
|
||||
if result.get("status") != "FOUNDRY" and route.get("foundry_route_probe") == "accepted":
|
||||
result = {"status": "FOUNDRY", "endpoint": endpoint, "model_count": 0, "models": [], "message": "route accepted without model-list support"}
|
||||
result.update(route)
|
||||
return result
|
||||
|
||||
|
||||
def is_azure_openai_llm(result):
|
||||
if result.get("status") != "VALID":
|
||||
return False
|
||||
if int(result.get("deployment_count") or 0) <= 0:
|
||||
return False
|
||||
route_probe = result.get("route_probe")
|
||||
if route_probe and route_probe != "accepted_auth_route":
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def append_azure_openai_llm(key, result, source):
|
||||
deployments = result.get("deployments") or []
|
||||
deployment_text = ",".join(str(item) for item in deployments[:20])
|
||||
details = " ".join(part for part in [
|
||||
f"deployments={int(result.get('deployment_count') or 0)}",
|
||||
f"route_probe={result.get('route_probe') or ''}" if result.get("route_probe") else "",
|
||||
f"route_deployment={result.get('route_deployment') or ''}" if result.get("route_deployment") else "",
|
||||
f"route_model={result.get('route_model') or ''}" if result.get("route_model") else "",
|
||||
f"deployment_ids={deployment_text}" if deployment_text else "",
|
||||
] if part)
|
||||
append_status(AZURE_OPENAI_LLM_FILE, key, result.get("status", "VALID"), details, source)
|
||||
|
||||
|
||||
def check_azure_acr(parsed, proxy, timeout):
|
||||
username = parsed["username"]
|
||||
password = parsed["password"]
|
||||
url = f"https://{username}.azurecr.io/v2/"
|
||||
try:
|
||||
response = requests.get(url, auth=(username, password), proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
text = str(exc)
|
||||
if "no such host" in text.lower():
|
||||
return {"status": "DEAD", "registry": username, "message": text}
|
||||
return {"status": "NETWORK", "registry": username, "message": text}
|
||||
if response.status_code == 200:
|
||||
return {"status": "VALID", "registry": username, "message": "ACR /v2 accepted basic auth"}
|
||||
message = request_error_message(response)
|
||||
if response.status_code == 401:
|
||||
return {"status": "DEAD", "registry": username, "http_status": 401, "message": message}
|
||||
if response.status_code == 403:
|
||||
return {"status": "RESTRICTED", "registry": username, "http_status": 403, "message": message}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "registry": username, "http_status": response.status_code, "message": message}
|
||||
return {"status": "UNKNOWN", "registry": username, "http_status": response.status_code, "message": message}
|
||||
|
||||
|
||||
def write_result(key, detector, result, source, finding):
|
||||
safe_finding = strip_finding_nearby_context(finding)
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, {"detector": detector, **result}, source, safe_finding, detector)
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE,
|
||||
transaction_status_files(),
|
||||
key,
|
||||
result["status"],
|
||||
result.get("message", ""),
|
||||
source,
|
||||
)
|
||||
if detector == "AzureOpenAI" and is_azure_openai_llm(result):
|
||||
append_azure_openai_llm(key, result, source)
|
||||
record_validation_result(SERVICE, key, {"detector": detector, **result}, source, safe_finding, detector)
|
||||
|
||||
|
||||
def strip_finding_nearby_context(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return finding
|
||||
output = dict(finding)
|
||||
context = output.get("ScannerContext")
|
||||
if isinstance(context, dict) and "nearby" in context:
|
||||
output["ScannerContext"] = {key: value for key, value in context.items() if key != "nearby"}
|
||||
return output
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Azure key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--probe-openai-route", action="store_true", help="Probe Azure OpenAI chat route with an invalid no-generation request after deployment listing succeeds")
|
||||
parser.add_argument("--probe-foundry-route", action="store_true", help="Probe Azure Foundry/MaaS chat route for configured models")
|
||||
parser.add_argument("--foundry-models", default="", help="Comma-separated Azure Foundry model IDs to route-probe")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.update({"NETWORK", "FOUNDRY_BAD_ENDPOINT", "OPENAI_BAD_ENDPOINT"})
|
||||
if args.retry_unknown:
|
||||
retry_statuses.update({"UNKNOWN", "FOUNDRY_UNRESOLVED", "OPENAI_UNRESOLVED"})
|
||||
if args.retry_valid:
|
||||
retry_statuses.update({"VALID", "FOUNDRY"})
|
||||
foundry_models = [item.strip() for item in str(args.foundry_models or "").split(",") if item.strip()]
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, detector, source, finding, parsed in extract_candidates(args.input):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=detector):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] {detector} candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
if parsed.get('_unresolved_candidate'):
|
||||
result = {
|
||||
'status': (
|
||||
'FOUNDRY_UNRESOLVED'
|
||||
if detector == 'AzureFoundry' else
|
||||
'OPENAI_UNRESOLVED'
|
||||
if detector == 'AzureOpenAI' else
|
||||
'UNKNOWN'
|
||||
),
|
||||
'message': f"normalized {parsed.get('candidate_kind') or 'azure'} candidate is incomplete",
|
||||
}
|
||||
elif detector == "AzureOpenAI":
|
||||
result = check_azure_openai(parsed, proxy, args.timeout, args.probe_openai_route)
|
||||
elif detector == "AzureFoundry":
|
||||
result = check_azure_foundry(parsed, proxy, args.timeout, args.probe_foundry_route, foundry_models)
|
||||
if parsed.get("finding_uid"):
|
||||
result["finding_uid"] = parsed.get("finding_uid")
|
||||
elif detector == "AzureContainerRegistry":
|
||||
result = check_azure_acr(parsed, proxy, args.timeout)
|
||||
else:
|
||||
result = check_service_principal(parsed, proxy, args.timeout)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, detector, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,292 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
classify_common_http_status,
|
||||
combined_provider_routing_hint,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
provider_routing_database_failed,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from keycheckers.provider_resolution import resolve_provider_key
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "deepseek"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "deepseekChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "deepseekResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "deepseekAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "deepseekNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "deepseekDead.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "deepseekLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "deepseekNetwork.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "deepseekNoContext.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "deepseekUnknown.txt"),
|
||||
}
|
||||
|
||||
DEEPSEEK_REGEX = re.compile(r"sk-[a-z0-9]{32}")
|
||||
DEEPSEEK_DETECTOR_NAMES = {"deepseek", "deepseekapikey", "deepseek_api_key"}
|
||||
DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"}
|
||||
QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"}
|
||||
KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"}
|
||||
QWEN_CONTEXT_REGEX = re.compile(
|
||||
r"(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
DEEPSEEK_CONTEXT_REGEX = re.compile(r"(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)", re.IGNORECASE)
|
||||
KIMI_CONTEXT_REGEX = re.compile(
|
||||
r"(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
AMBIGUOUS_PROVIDER_HINT = "ambiguous_qwen_deepseek"
|
||||
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
|
||||
GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"}
|
||||
EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment"
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def iter_candidate_decisions(input_file, plain_files):
|
||||
detector_names = ["DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key", "CustomRegex"]
|
||||
for item in iter_findings(input_file, detector_names):
|
||||
finding = item.get("finding") or {}
|
||||
if not finding_has_deepseek_detector(finding):
|
||||
continue
|
||||
key = item.get("credential_secret_text") or item["raw"]
|
||||
if key and DEEPSEEK_REGEX.fullmatch(key):
|
||||
hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding))
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
yield key, item["source"], finding, hint
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
for key, source, finding, hint in iter_candidate_decisions(input_file, plain_files):
|
||||
if hint == "deepseek":
|
||||
yield key, source, finding
|
||||
|
||||
|
||||
def route_rejection_result(hint):
|
||||
normalized = str(hint or "missing").strip().lower()
|
||||
return {
|
||||
"status": "NO_CONTEXT",
|
||||
"routing_hint": normalized,
|
||||
"message": f"candidate is not safely attributable to DeepSeek; routing_hint={normalized}",
|
||||
}
|
||||
|
||||
|
||||
def finding_detector_names(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return set()
|
||||
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
|
||||
names = {
|
||||
str(finding.get("DetectorName") or finding.get("detector") or "").strip().lower(),
|
||||
str(extra.get("name") or "").strip().lower(),
|
||||
}
|
||||
return {name for name in names if name}
|
||||
|
||||
|
||||
def finding_has_deepseek_detector(finding):
|
||||
return bool(finding_detector_names(finding) & DEEPSEEK_DETECTOR_NAMES)
|
||||
|
||||
|
||||
def finding_has_explicit_detector(finding, detector_names):
|
||||
return bool(finding_detector_names(finding) & set(detector_names))
|
||||
|
||||
|
||||
def finding_provider_routing_hint(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return ""
|
||||
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
|
||||
persisted_hint = context.get("provider_hint")
|
||||
if (
|
||||
context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE
|
||||
and persisted_hint in (*GENERIC_SK_PROVIDERS, AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
):
|
||||
return persisted_hint
|
||||
parts = []
|
||||
for key in ("nearby", "file"):
|
||||
if context.get(key):
|
||||
parts.append(str(context.get(key)))
|
||||
metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {}
|
||||
data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {}
|
||||
for details in data.values():
|
||||
if not isinstance(details, dict):
|
||||
continue
|
||||
for key in ("file", "repository", "repo", "link", "image"):
|
||||
if details.get(key):
|
||||
parts.append(str(details.get(key)))
|
||||
text = "\n".join(parts)
|
||||
evidence = set()
|
||||
if QWEN_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("qwen")
|
||||
if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("deepseek")
|
||||
if KIMI_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("kimi")
|
||||
if persisted_hint == AMBIGUOUS_PROVIDER_HINT:
|
||||
evidence.update(("qwen", "deepseek"))
|
||||
elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT:
|
||||
evidence.update(GENERIC_SK_PROVIDERS)
|
||||
elif persisted_hint in GENERIC_SK_PROVIDERS:
|
||||
evidence.add(persisted_hint)
|
||||
if len(evidence) > 1:
|
||||
return AMBIGUOUS_PROVIDER_HINT if evidence == {"qwen", "deepseek"} else AMBIGUOUS_GENERIC_SK_HINT
|
||||
return next(iter(evidence)) if evidence else ""
|
||||
|
||||
|
||||
def finding_has_ambiguous_provider_hint(finding):
|
||||
return finding_provider_routing_hint(finding) in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
|
||||
|
||||
def finding_looks_like_qwen_context(finding):
|
||||
return finding_provider_routing_hint(finding) == "qwen"
|
||||
|
||||
|
||||
def check_key(key, proxy, timeout):
|
||||
url = "https://api.deepseek.com/user/balance"
|
||||
headers = {"Authorization": f"Bearer {key}"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
balance_infos = data.get("balance_infos", [])
|
||||
total_usd = 0.0
|
||||
for balance in balance_infos:
|
||||
amount = float(balance.get("total_balance", "0") or 0)
|
||||
currency = balance.get("currency", "USD")
|
||||
if currency == "CNY":
|
||||
amount *= 0.14
|
||||
total_usd += amount
|
||||
available = bool(data.get("is_available", False))
|
||||
status = "VALID" if available and total_usd > 0 else "NO_BALANCE"
|
||||
return {
|
||||
"status": status,
|
||||
"authenticated": True,
|
||||
"available": available,
|
||||
"balance_usd": round(total_usd, 4),
|
||||
"message": f"available={available}; balance=${total_usd:.4f}",
|
||||
}
|
||||
|
||||
status = classify_common_http_status(response.status_code)
|
||||
return {"status": status, "http_status": response.status_code, "message": request_error_message(response)}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "DeepSeek")
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, "DeepSeek")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="DeepSeek key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.add("NO_BALANCE")
|
||||
if args.retry_valid:
|
||||
retry_statuses.add("VALID")
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
postgres_mode = keycheck_input_mode() == "postgres"
|
||||
for key, source, finding, routing_hint in iter_candidate_decisions(args.input, args.plain):
|
||||
route_rejected = routing_hint != "deepseek"
|
||||
ambiguous_route = routing_hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
if route_rejected and not postgres_mode:
|
||||
skipped += 1
|
||||
continue
|
||||
if not route_rejected and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="DeepSeek"):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] DeepSeek candidate {mask_secret(key)} from {source}")
|
||||
if route_rejected and ambiguous_route:
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = resolve_provider_key(
|
||||
key, finding, proxy, args.timeout,
|
||||
hint=routing_hint, origin_service=SERVICE,
|
||||
)
|
||||
elif route_rejected:
|
||||
result = route_rejection_result(routing_hint)
|
||||
else:
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(key, proxy, args.timeout)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,240 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "dockerhub"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
PLAIN_FILE = os.path.join(OUTPUT_DIR, "dockerhub.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "dockerhubChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "dockerhubResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "dockerhubAlive.txt"),
|
||||
"VALID_2FA": os.path.join(OUTPUT_DIR, "dockerhubAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "dockerhubDead.txt"),
|
||||
"NO_USERNAME": os.path.join(OUTPUT_DIR, "dockerhubNoUsername.txt"),
|
||||
"RATE_LIMITED": os.path.join(OUTPUT_DIR, "dockerhubRateLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "dockerhubNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "dockerhubUnknown.txt"),
|
||||
}
|
||||
|
||||
DOCKER_PAT_RE = re.compile(r"\bdckr_pat_[A-Za-z0-9_-]{27}\b")
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *set(STATUS_FILES.values()), PLAIN_FILE])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def candidate_key(username, token):
|
||||
return f"{username}:{token}" if username else token
|
||||
|
||||
|
||||
def parse_username_token(raw, raw_v2, finding):
|
||||
raw = raw or ""
|
||||
raw_v2 = raw_v2 or ""
|
||||
token = ""
|
||||
username = ""
|
||||
|
||||
if ":" in raw_v2:
|
||||
maybe_user, maybe_token = raw_v2.split(":", 1)
|
||||
if DOCKER_PAT_RE.fullmatch(maybe_token):
|
||||
username, token = maybe_user.strip(), maybe_token.strip()
|
||||
if not token:
|
||||
match = DOCKER_PAT_RE.search(raw_v2) or DOCKER_PAT_RE.search(raw)
|
||||
if match:
|
||||
token = match.group(0)
|
||||
|
||||
extra = finding.get("ExtraData") if isinstance(finding, dict) else {}
|
||||
analysis = finding.get("AnalysisInfo") if isinstance(finding, dict) else {}
|
||||
if isinstance(extra, dict):
|
||||
username = username or extra.get("hub_username") or ""
|
||||
if isinstance(analysis, dict):
|
||||
username = username or analysis.get("username") or ""
|
||||
return username, token
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_file):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["Dockerhub"]):
|
||||
username, token = parse_username_token(item.get("raw"), item.get("raw_v2"), item.get("finding") or {})
|
||||
if not token:
|
||||
continue
|
||||
key = candidate_key(username, token)
|
||||
yield key, username, token, item["source"], item["finding"]
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and os.path.exists(plain_file):
|
||||
with open(plain_file, "r", encoding="utf-8") as f:
|
||||
for line_num, line in enumerate(f, 1):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
username = ""
|
||||
token = ""
|
||||
if ":" in line:
|
||||
maybe_user, rest = line.split(":", 1)
|
||||
match = DOCKER_PAT_RE.search(rest)
|
||||
if match:
|
||||
username, token = maybe_user.strip(), match.group(0)
|
||||
else:
|
||||
match = DOCKER_PAT_RE.search(line)
|
||||
if match:
|
||||
token = match.group(0)
|
||||
if not token:
|
||||
continue
|
||||
key = candidate_key(username, token)
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, username, token, f"{plain_file}:{line_num}", {}
|
||||
|
||||
|
||||
def decode_jwt_payload(jwt_token):
|
||||
try:
|
||||
payload = jwt_token.split(".")[1]
|
||||
payload += "=" * (-len(payload) % 4)
|
||||
return json.loads(base64.urlsafe_b64decode(payload.encode()).decode())
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
|
||||
def check_token(username, token, proxy, timeout):
|
||||
if not username:
|
||||
return {"status": "NO_USERNAME", "message": "DockerHub PAT requires username/email for login check"}
|
||||
|
||||
url = "https://hub.docker.com/v2/users/login"
|
||||
payload = {"username": username, "password": token}
|
||||
try:
|
||||
response = requests.post(url, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "username": username}
|
||||
|
||||
message = request_error_message(response)
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
hub_token = data.get("token", "")
|
||||
claims = decode_jwt_payload(hub_token) if hub_token else {}
|
||||
hub_claims = claims.get("https://hub.docker.com", {}) if isinstance(claims, dict) else {}
|
||||
return {
|
||||
"status": "VALID",
|
||||
"message": "login accepted",
|
||||
"username": username,
|
||||
"hub_username": hub_claims.get("username", username),
|
||||
"hub_email": hub_claims.get("email", ""),
|
||||
"scope": claims.get("scope", "") if isinstance(claims, dict) else "",
|
||||
}
|
||||
if response.status_code == 401:
|
||||
try:
|
||||
data = response.json()
|
||||
except ValueError:
|
||||
data = {}
|
||||
if data.get("login_2fa_token"):
|
||||
return {"status": "VALID_2FA", "message": "credentials accepted; 2FA required", "username": username}
|
||||
return {"status": "DEAD", "http_status": 401, "message": message, "username": username}
|
||||
if response.status_code == 429:
|
||||
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "username": username}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "username": username}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "username": username}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "Dockerhub")
|
||||
extra = result.get("hub_username") or result.get("username") or source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, "Dockerhub")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="DockerHub PAT checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", default=PLAIN_FILE)
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-no-username", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("RATE_LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_no_username:
|
||||
retry_statuses.add("NO_USERNAME")
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, username, token, source, finding in extract_candidates(args.input, args.plain):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="Dockerhub"):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] DockerHub candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_token(username, token, proxy, args.timeout)
|
||||
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
time.sleep(0.1)
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,789 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import binascii
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
append_status,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_known_statuses,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_cached_keycheck_occurrence,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "gcp"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
PLAIN_FILE = os.path.join(OUTPUT_DIR, "gcp.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "gcpChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "gcpResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "gcpAlive.txt"),
|
||||
"VERTEX": os.path.join(OUTPUT_DIR, "gcpVertex.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "gcpDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "gcpRestricted.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "gcpNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "gcpUnknown.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "gcpNoContext.txt"),
|
||||
}
|
||||
VERTEX_GEMINI_FILE = os.path.join(OUTPUT_DIR, "gcpVertexGemini.txt")
|
||||
VERTEX_ANTHROPIC_FILE = os.path.join(OUTPUT_DIR, "gcpVertexAnthropic.txt")
|
||||
|
||||
GOOGLE_TOKEN_URL = "https://oauth2.googleapis.com/token"
|
||||
TRUSTED_GOOGLE_TOKEN_ENDPOINTS = frozenset({
|
||||
GOOGLE_TOKEN_URL,
|
||||
"https://accounts.google.com/o/oauth2/token",
|
||||
})
|
||||
TOKEN_REDIRECT_STATUSES = {301, 302, 303, 307, 308}
|
||||
SA_SCOPE = "https://www.googleapis.com/auth/cloud-platform"
|
||||
MAX_PEM_BYTES = 24 * 1024
|
||||
MAX_DER_BYTES = 16 * 1024
|
||||
MAX_DER_LENGTH_BYTES = 2
|
||||
MIN_RSA_BITS = 2048
|
||||
MAX_RSA_BITS = 8192
|
||||
MAX_RSA_INTEGER_BYTES = MAX_RSA_BITS // 8
|
||||
RSA_ENCRYPTION_OID = bytes.fromhex("2a864886f70d010101")
|
||||
VERTEX_LOCATIONS = ["global", "us", "eu"]
|
||||
VERTEX_MODELS = ["gemini-3.6-flash", "gemini-3.1-pro-preview"]
|
||||
VERTEX_ANTHROPIC_LOCATIONS = ["global", "us", "eu", "us-east5", "europe-west1"]
|
||||
VERTEX_ANTHROPIC_MODELS = ["claude-opus-5", "claude-opus-4-7", "claude-opus-4-6", "claude-fable-5"]
|
||||
|
||||
|
||||
def transaction_status_files():
|
||||
return {
|
||||
**STATUS_FILES,
|
||||
"RATE_LIMITED": STATUS_FILES["UNKNOWN"],
|
||||
"AUX_VERTEX_GEMINI": VERTEX_GEMINI_FILE,
|
||||
"AUX_VERTEX_ANTHROPIC": VERTEX_ANTHROPIC_FILE,
|
||||
}
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), VERTEX_GEMINI_FILE, VERTEX_ANTHROPIC_FILE, PLAIN_FILE])
|
||||
recover_status_transaction(CHECKED_FILE, transaction_status_files())
|
||||
|
||||
|
||||
def b64url(data):
|
||||
return base64.urlsafe_b64encode(data).rstrip(b"=").decode()
|
||||
|
||||
|
||||
def validate_google_token_uri(value):
|
||||
token_uri = str(value or GOOGLE_TOKEN_URL).strip()
|
||||
try:
|
||||
parsed = urlsplit(token_uri)
|
||||
port = parsed.port
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise ValueError("invalid Google OAuth token_uri") from exc
|
||||
if (
|
||||
parsed.scheme != "https"
|
||||
or parsed.username is not None
|
||||
or parsed.password is not None
|
||||
or port not in (None, 443)
|
||||
or parsed.query
|
||||
or parsed.fragment
|
||||
or token_uri not in TRUSTED_GOOGLE_TOKEN_ENDPOINTS
|
||||
):
|
||||
raise ValueError("untrusted Google OAuth token_uri")
|
||||
return token_uri
|
||||
|
||||
|
||||
class InvalidRSAPrivateKey(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
class DERReader:
|
||||
def __init__(self, data):
|
||||
if not isinstance(data, (bytes, bytearray, memoryview)):
|
||||
raise InvalidRSAPrivateKey("DER value is not binary")
|
||||
if len(data) > MAX_DER_BYTES:
|
||||
raise InvalidRSAPrivateKey("DER value exceeds size limit")
|
||||
self.data = data
|
||||
self.pos = 0
|
||||
|
||||
def read_tlv(self):
|
||||
if len(self.data) - self.pos < 2:
|
||||
raise InvalidRSAPrivateKey("truncated DER tag or length")
|
||||
tag = self.data[self.pos]
|
||||
self.pos += 1
|
||||
first_len = self.data[self.pos]
|
||||
self.pos += 1
|
||||
if first_len & 0x80:
|
||||
length_len = first_len & 0x7F
|
||||
if length_len == 0:
|
||||
raise InvalidRSAPrivateKey("indefinite DER length is not allowed")
|
||||
if length_len > MAX_DER_LENGTH_BYTES:
|
||||
raise InvalidRSAPrivateKey("DER length-of-length exceeds limit")
|
||||
if len(self.data) - self.pos < length_len:
|
||||
raise InvalidRSAPrivateKey("truncated DER length")
|
||||
length_bytes = self.data[self.pos:self.pos + length_len]
|
||||
if length_bytes[0] == 0:
|
||||
raise InvalidRSAPrivateKey("non-minimal DER length")
|
||||
length = int.from_bytes(length_bytes, "big")
|
||||
self.pos += length_len
|
||||
if length < 0x80:
|
||||
raise InvalidRSAPrivateKey("non-minimal DER length")
|
||||
else:
|
||||
length = first_len
|
||||
if length > MAX_DER_BYTES:
|
||||
raise InvalidRSAPrivateKey("DER value length exceeds limit")
|
||||
if length > len(self.data) - self.pos:
|
||||
raise InvalidRSAPrivateKey("truncated DER value")
|
||||
value = self.data[self.pos:self.pos + length]
|
||||
self.pos += length
|
||||
return tag, value
|
||||
|
||||
def expect(self, tag):
|
||||
actual, value = self.read_tlv()
|
||||
if actual != tag:
|
||||
raise InvalidRSAPrivateKey(f"expected DER tag {tag:#x}, got {actual:#x}")
|
||||
return value
|
||||
|
||||
def at_end(self):
|
||||
return self.pos == len(self.data)
|
||||
|
||||
def require_eof(self, context="DER structure"):
|
||||
if not self.at_end():
|
||||
raise InvalidRSAPrivateKey(f"trailing data in {context}")
|
||||
|
||||
|
||||
def der_int(value, name="integer", max_bytes=MAX_RSA_INTEGER_BYTES):
|
||||
if not value:
|
||||
raise InvalidRSAPrivateKey(f"empty RSA {name}")
|
||||
if len(value) > max_bytes + 1:
|
||||
raise InvalidRSAPrivateKey(f"RSA {name} exceeds size limit")
|
||||
if value[0] & 0x80:
|
||||
raise InvalidRSAPrivateKey(f"negative RSA {name}")
|
||||
if value[0] == 0:
|
||||
if len(value) > 1 and not value[1] & 0x80:
|
||||
raise InvalidRSAPrivateKey(f"non-minimal RSA {name}")
|
||||
unsigned = value[1:]
|
||||
else:
|
||||
unsigned = value
|
||||
if len(unsigned) > max_bytes:
|
||||
raise InvalidRSAPrivateKey(f"RSA {name} exceeds size limit")
|
||||
return int.from_bytes(unsigned, "big") if unsigned else 0
|
||||
|
||||
|
||||
def parse_pkcs1_rsa_private_key(data):
|
||||
rsa = DERReader(data)
|
||||
version = der_int(rsa.expect(0x02), "version", 1)
|
||||
if version != 0:
|
||||
raise InvalidRSAPrivateKey("unsupported RSA private key version")
|
||||
n = der_int(rsa.expect(0x02), "modulus")
|
||||
public_exponent = der_int(rsa.expect(0x02), "public exponent")
|
||||
d = der_int(rsa.expect(0x02), "private exponent")
|
||||
for name in ("prime1", "prime2", "exponent1", "exponent2", "coefficient"):
|
||||
der_int(rsa.expect(0x02), name)
|
||||
rsa.require_eof("RSA private key")
|
||||
|
||||
modulus_bits = n.bit_length()
|
||||
if not MIN_RSA_BITS <= modulus_bits <= MAX_RSA_BITS:
|
||||
raise InvalidRSAPrivateKey(
|
||||
f"RSA modulus must be between {MIN_RSA_BITS} and {MAX_RSA_BITS} bits"
|
||||
)
|
||||
if public_exponent == 0:
|
||||
raise InvalidRSAPrivateKey("RSA public exponent is zero")
|
||||
if d == 0 or d >= n:
|
||||
raise InvalidRSAPrivateKey("RSA private exponent is out of range")
|
||||
return n, d
|
||||
|
||||
|
||||
def validate_rsa_algorithm_identifier(data):
|
||||
algorithm = DERReader(data)
|
||||
if algorithm.expect(0x06) != RSA_ENCRYPTION_OID:
|
||||
raise InvalidRSAPrivateKey("PKCS#8 key does not use rsaEncryption")
|
||||
if not algorithm.at_end() and algorithm.expect(0x05):
|
||||
raise InvalidRSAPrivateKey("invalid rsaEncryption parameters")
|
||||
algorithm.require_eof("PKCS#8 algorithm identifier")
|
||||
|
||||
|
||||
def parse_rsa_private_key_from_pem(pem):
|
||||
pem = str(pem or "")
|
||||
if len(pem) > MAX_PEM_BYTES:
|
||||
raise InvalidRSAPrivateKey("PEM private key exceeds size limit")
|
||||
try:
|
||||
pem_bytes = pem.encode("utf-8")
|
||||
except UnicodeEncodeError as exc:
|
||||
raise InvalidRSAPrivateKey("PEM private key is not valid UTF-8") from exc
|
||||
if len(pem_bytes) > MAX_PEM_BYTES:
|
||||
raise InvalidRSAPrivateKey("PEM private key exceeds size limit")
|
||||
pem = pem.replace("\\n", "\n")
|
||||
match = re.fullmatch(
|
||||
r"\s*-----BEGIN (RSA PRIVATE KEY|PRIVATE KEY)-----\s*(.*?)\s*-----END \1-----\s*",
|
||||
pem,
|
||||
re.DOTALL,
|
||||
)
|
||||
if not match:
|
||||
raise InvalidRSAPrivateKey("missing complete PEM private key block")
|
||||
body = re.sub(r"\s+", "", match.group(2))
|
||||
if len(body) < 256:
|
||||
raise InvalidRSAPrivateKey("PEM private key body is too short")
|
||||
try:
|
||||
der = base64.b64decode(body + ("=" * (-len(body) % 4)), validate=True)
|
||||
except (binascii.Error, ValueError) as exc:
|
||||
raise InvalidRSAPrivateKey("invalid PEM base64") from exc
|
||||
if len(der) > MAX_DER_BYTES:
|
||||
raise InvalidRSAPrivateKey("DER private key exceeds size limit")
|
||||
|
||||
reader = DERReader(der)
|
||||
top_bytes = reader.expect(0x30)
|
||||
reader.require_eof("DER private key")
|
||||
top = DERReader(top_bytes)
|
||||
|
||||
# PKCS#8 PrivateKeyInfo: SEQUENCE(version, alg, OCTET STRING(RSAPrivateKey))
|
||||
first_tag, first_val = top.read_tlv()
|
||||
if first_tag != 0x02:
|
||||
raise InvalidRSAPrivateKey("unexpected private key structure")
|
||||
second_tag, second_val = top.read_tlv()
|
||||
if second_tag == 0x30:
|
||||
if der_int(first_val, "PKCS#8 version", 1) != 0:
|
||||
raise InvalidRSAPrivateKey("unsupported PKCS#8 version")
|
||||
validate_rsa_algorithm_identifier(second_val)
|
||||
private_octet = top.expect(0x04)
|
||||
top.require_eof("PKCS#8 private key")
|
||||
wrapped = DERReader(private_octet)
|
||||
rsa_bytes = wrapped.expect(0x30)
|
||||
wrapped.require_eof("PKCS#8 private key octets")
|
||||
return parse_pkcs1_rsa_private_key(rsa_bytes)
|
||||
if second_tag != 0x02:
|
||||
raise InvalidRSAPrivateKey("unexpected private key structure")
|
||||
return parse_pkcs1_rsa_private_key(top_bytes)
|
||||
|
||||
|
||||
def rsa_pkcs1v15_sha256_sign(message, pem):
|
||||
n, d = parse_rsa_private_key_from_pem(pem)
|
||||
digest = hashlib.sha256(message).digest()
|
||||
digest_info = bytes.fromhex("3031300d060960864801650304020105000420") + digest
|
||||
key_len = (n.bit_length() + 7) // 8
|
||||
if key_len < len(digest_info) + 11:
|
||||
raise InvalidRSAPrivateKey("RSA key too small")
|
||||
encoded = b"\x00\x01" + b"\xff" * (key_len - len(digest_info) - 3) + b"\x00" + digest_info
|
||||
sig = pow(int.from_bytes(encoded, "big"), d, n).to_bytes(key_len, "big")
|
||||
return sig
|
||||
|
||||
|
||||
def make_service_account_assertion(creds):
|
||||
now = int(time.time())
|
||||
token_uri = validate_google_token_uri(creds.get("token_uri"))
|
||||
header = {"alg": "RS256", "typ": "JWT", "kid": creds.get("private_key_id")}
|
||||
payload = {
|
||||
"iss": creds["client_email"],
|
||||
"scope": SA_SCOPE,
|
||||
"aud": token_uri,
|
||||
"iat": now,
|
||||
"exp": now + 3600,
|
||||
}
|
||||
signing_input = (b64url(json.dumps(header, separators=(",", ":")).encode()) + "." + b64url(json.dumps(payload, separators=(",", ":")).encode())).encode()
|
||||
signature = rsa_pkcs1v15_sha256_sign(signing_input, creds["private_key"])
|
||||
return signing_input.decode() + "." + b64url(signature), token_uri
|
||||
|
||||
|
||||
def compact_json(data):
|
||||
return json.dumps(data, ensure_ascii=False, separators=(",", ":"), sort_keys=True)
|
||||
|
||||
|
||||
def scanner_context_text(finding):
|
||||
context = finding.get("ScannerContext") if isinstance(finding, dict) else None
|
||||
if isinstance(context, dict):
|
||||
return str(context.get("nearby") or "")
|
||||
return ""
|
||||
|
||||
|
||||
def parse_json_object(text):
|
||||
try:
|
||||
return json.loads(text)
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
match = re.search(r"\{.*\}", str(text or ""), re.DOTALL)
|
||||
if not match:
|
||||
return None
|
||||
try:
|
||||
return json.loads(match.group(0))
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def parse_service_account(raw_v2):
|
||||
data = parse_json_object(raw_v2)
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
required = ["client_email", "private_key", "private_key_id"]
|
||||
if not all(data.get(item) for item in required):
|
||||
return None
|
||||
private_key = str(data.get("private_key") or "")
|
||||
if "-----BEGIN" not in private_key or "-----END" not in private_key or len(private_key) < 800:
|
||||
return None
|
||||
return data
|
||||
|
||||
|
||||
def parse_adc(raw_v2, finding):
|
||||
data = parse_json_object(scanner_context_text(finding)) or parse_json_object(raw_v2)
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
required = ["client_id", "client_secret", "refresh_token"]
|
||||
if not all(data.get(item) for item in required):
|
||||
return None
|
||||
return data
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_file):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["GCP", "GCPApplicationDefaultCredentials"]):
|
||||
if item["detector"] == "GCP":
|
||||
parsed = parse_service_account(item["raw_v2"])
|
||||
detector = "GCP"
|
||||
else:
|
||||
parsed = parse_adc(item["raw_v2"], item["finding"])
|
||||
detector = "GCPApplicationDefaultCredentials"
|
||||
if not parsed:
|
||||
key = f"{detector}:no_context:{item['source']}"
|
||||
yield key, detector, item["source"], item["finding"], None
|
||||
continue
|
||||
key = compact_json(parsed)
|
||||
yield key, detector, item["source"], item["finding"], parsed
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and os.path.exists(plain_file):
|
||||
with open(plain_file, "r", encoding="utf-8", errors="replace") as f:
|
||||
content = f.read()
|
||||
candidates = []
|
||||
whole_file = parse_json_object(content)
|
||||
if whole_file:
|
||||
candidates.append((plain_file, whole_file))
|
||||
for line_num, line in enumerate(content.splitlines(), 1):
|
||||
key_text = line.split("\t", 1)[0].strip()
|
||||
parsed = parse_json_object(key_text) or parse_json_object(line)
|
||||
if parsed:
|
||||
candidates.append((f"{plain_file}:{line_num}", parsed))
|
||||
for source, parsed in candidates:
|
||||
detector = "GCP" if parsed.get("private_key") else "GCPApplicationDefaultCredentials"
|
||||
key = compact_json(parsed)
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, detector, source, {}, parsed
|
||||
|
||||
|
||||
def vertex_api_host(location):
|
||||
location = str(location or "").strip().lower()
|
||||
if location == "global":
|
||||
return "aiplatform.googleapis.com"
|
||||
if location in ("us", "eu"):
|
||||
return f"aiplatform.{location}.rep.googleapis.com"
|
||||
return f"{location}-aiplatform.googleapis.com"
|
||||
|
||||
|
||||
def probe_vertex_llm(access_token, project_id, proxy, timeout, locations=None, models=None, max_attempts=2):
|
||||
if not project_id:
|
||||
return {"enabled": False, "message": "project_id unavailable"}
|
||||
headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"}
|
||||
payload = {"contents": [{"role": "user", "parts": [{"text": "ping"}]}]}
|
||||
attempts = []
|
||||
accepted = []
|
||||
tried = 0
|
||||
for location in (locations or VERTEX_LOCATIONS):
|
||||
for model in (models or VERTEX_MODELS):
|
||||
if max_attempts and tried >= max_attempts:
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"location": first.get("location", ""),
|
||||
"model": first.get("model", ""),
|
||||
"total_tokens": first.get("total_tokens"),
|
||||
"available_models": [f"google/{item['location']}/{item['model']}" for item in accepted],
|
||||
"message": "Vertex countTokens accepted",
|
||||
}
|
||||
return {"enabled": False, "message": "; ".join(attempts[:8]) or "Vertex probe attempt limit reached"}
|
||||
tried += 1
|
||||
url = f"https://{vertex_api_host(location)}/v1/projects/{project_id}/locations/{location}/publishers/google/models/{model}:countTokens"
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
attempts.append(f"{location}:{model}:network:{str(exc)[:120]}")
|
||||
continue
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
accepted.append({
|
||||
"location": location,
|
||||
"model": model,
|
||||
"total_tokens": data.get("totalTokens") or data.get("total_tokens"),
|
||||
})
|
||||
continue
|
||||
message = request_error_message(response)
|
||||
if response.status_code in (400, 401, 403, 404, 429):
|
||||
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
if response.status_code >= 500:
|
||||
attempts.append(f"{location}:{model}:server_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"location": first.get("location", ""),
|
||||
"model": first.get("model", ""),
|
||||
"total_tokens": first.get("total_tokens"),
|
||||
"available_models": [f"google/{item['location']}/{item['model']}" for item in accepted],
|
||||
"message": "Vertex countTokens accepted",
|
||||
}
|
||||
return {"enabled": False, "message": "; ".join(attempts[:8])}
|
||||
|
||||
|
||||
def probe_vertex_anthropic(access_token, project_id, proxy, timeout, locations=None, models=None, max_attempts=2):
|
||||
if not project_id:
|
||||
return {"enabled": False, "message": "project_id unavailable"}
|
||||
models = models or []
|
||||
if not models:
|
||||
return {"enabled": False, "message": "no Anthropic models configured"}
|
||||
headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"}
|
||||
payload = {
|
||||
"anthropic_version": "vertex-2023-10-16",
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": 1,
|
||||
}
|
||||
attempts = []
|
||||
accepted = []
|
||||
tried = 0
|
||||
for location in (locations or VERTEX_ANTHROPIC_LOCATIONS):
|
||||
for model in models:
|
||||
if max_attempts and tried >= max_attempts:
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"location": first.get("location", ""),
|
||||
"model": first.get("model", ""),
|
||||
"available_models": [f"anthropic/{item['location']}/{item['model']}" for item in accepted],
|
||||
"message": "Vertex Anthropic rawPredict accepted",
|
||||
}
|
||||
return {"enabled": False, "message": "; ".join(attempts[:8]) or "Vertex Anthropic probe attempt limit reached"}
|
||||
tried += 1
|
||||
url = f"https://{vertex_api_host(location)}/v1/projects/{project_id}/locations/{location}/publishers/anthropic/models/{model}:rawPredict"
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
attempts.append(f"{location}:{model}:network:{str(exc)[:120]}")
|
||||
continue
|
||||
if response.status_code == 200:
|
||||
accepted.append({"location": location, "model": model})
|
||||
continue
|
||||
message = request_error_message(response)
|
||||
if response.status_code in (400, 401, 403, 404, 429):
|
||||
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
if response.status_code >= 500:
|
||||
attempts.append(f"{location}:{model}:server_{response.status_code}:{message[:160]}")
|
||||
continue
|
||||
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
|
||||
if accepted:
|
||||
first = accepted[0]
|
||||
return {
|
||||
"enabled": True,
|
||||
"location": first.get("location", ""),
|
||||
"model": first.get("model", ""),
|
||||
"available_models": [f"anthropic/{item['location']}/{item['model']}" for item in accepted],
|
||||
"message": "Vertex Anthropic rawPredict accepted",
|
||||
}
|
||||
return {"enabled": False, "message": "; ".join(attempts[:8])}
|
||||
|
||||
|
||||
def merge_vertex_results(google_vertex, anthropic_vertex):
|
||||
google_vertex = google_vertex or {"enabled": False, "message": ""}
|
||||
anthropic_vertex = anthropic_vertex or {"enabled": False, "message": ""}
|
||||
available = []
|
||||
available.extend(google_vertex.get("available_models") or [])
|
||||
available.extend(anthropic_vertex.get("available_models") or [])
|
||||
first = google_vertex if google_vertex.get("enabled") else anthropic_vertex if anthropic_vertex.get("enabled") else {}
|
||||
messages = []
|
||||
if google_vertex.get("message"):
|
||||
messages.append(f"google: {google_vertex.get('message')}")
|
||||
if anthropic_vertex.get("message"):
|
||||
messages.append(f"anthropic: {anthropic_vertex.get('message')}")
|
||||
return {
|
||||
"enabled": bool(available),
|
||||
"location": first.get("location", ""),
|
||||
"model": first.get("model", ""),
|
||||
"total_tokens": first.get("total_tokens"),
|
||||
"available_models": available,
|
||||
"google_enabled": bool(google_vertex.get("enabled")),
|
||||
"anthropic_enabled": bool(anthropic_vertex.get("enabled")),
|
||||
"message": "; ".join(messages),
|
||||
}
|
||||
|
||||
|
||||
def check_service_account(creds, proxy, timeout, probe_vertex=False, vertex_timeout=6, vertex_locations=None, vertex_models=None, vertex_max_attempts=2, vertex_anthropic_locations=None, vertex_anthropic_models=None, vertex_anthropic_max_attempts=2):
|
||||
try:
|
||||
assertion, token_uri = make_service_account_assertion(creds)
|
||||
except InvalidRSAPrivateKey as exc:
|
||||
return {
|
||||
"status": "DEAD",
|
||||
"classification": "invalid_private_key",
|
||||
"message": f"invalid RSA private key: {exc}",
|
||||
"project_id": creds.get("project_id"),
|
||||
"client_email": creds.get("client_email"),
|
||||
}
|
||||
except Exception as exc:
|
||||
return {"status": "UNKNOWN", "message": f"failed to build JWT assertion: {exc}"}
|
||||
data = {"grant_type": "urn:ietf:params:oauth:grant-type:jwt-bearer", "assertion": assertion}
|
||||
try:
|
||||
response = requests.post(
|
||||
token_uri, data=data, proxies=proxy, timeout=timeout, allow_redirects=False,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
if response.status_code in TOKEN_REDIRECT_STATUSES:
|
||||
return {
|
||||
"status": "UNKNOWN",
|
||||
"http_status": response.status_code,
|
||||
"message": "Google OAuth token endpoint redirect refused",
|
||||
"project_id": creds.get("project_id"),
|
||||
"client_email": creds.get("client_email"),
|
||||
}
|
||||
if response.status_code == 200:
|
||||
payload = response.json()
|
||||
result = {
|
||||
"status": "VALID",
|
||||
"message": "OAuth token issued",
|
||||
"project_id": creds.get("project_id"),
|
||||
"client_email": creds.get("client_email"),
|
||||
"private_key_id": creds.get("private_key_id"),
|
||||
"expires_in": payload.get("expires_in"),
|
||||
}
|
||||
if probe_vertex:
|
||||
google_vertex = probe_vertex_llm(
|
||||
payload.get("access_token"), creds.get("project_id"), proxy,
|
||||
vertex_timeout, vertex_locations, vertex_models, vertex_max_attempts,
|
||||
)
|
||||
anthropic_vertex = probe_vertex_anthropic(
|
||||
payload.get("access_token"), creds.get("project_id"), proxy,
|
||||
vertex_timeout, vertex_anthropic_locations, vertex_anthropic_models, vertex_anthropic_max_attempts,
|
||||
) if vertex_anthropic_models else {"enabled": False, "message": ""}
|
||||
vertex = merge_vertex_results(google_vertex, anthropic_vertex)
|
||||
result.update({
|
||||
"vertex_enabled": vertex.get("enabled"),
|
||||
"vertex_location": vertex.get("location", ""),
|
||||
"vertex_model": vertex.get("model", ""),
|
||||
"vertex_available_models": vertex.get("available_models") or [],
|
||||
"vertex_google_enabled": vertex.get("google_enabled"),
|
||||
"vertex_anthropic_enabled": vertex.get("anthropic_enabled"),
|
||||
"vertex_total_tokens": vertex.get("total_tokens"),
|
||||
"vertex_message": vertex.get("message", ""),
|
||||
})
|
||||
if vertex.get("enabled"):
|
||||
result["status"] = "VERTEX"
|
||||
result["message"] = "OAuth token issued; Vertex countTokens accepted"
|
||||
return result
|
||||
message = request_error_message(response)
|
||||
lower = message.lower()
|
||||
if response.status_code in (400, 401) and any(item in lower for item in ("invalid_grant", "invalid_client", "invalid jwt", "invalid signature")):
|
||||
return {"status": "DEAD", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
if response.status_code in (401, 403):
|
||||
return {"status": "RESTRICTED", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
if response.status_code == 429:
|
||||
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
|
||||
|
||||
|
||||
def check_adc(creds, proxy, timeout, probe_vertex=False, vertex_timeout=6, vertex_locations=None, vertex_models=None, vertex_max_attempts=2, vertex_anthropic_locations=None, vertex_anthropic_models=None, vertex_anthropic_max_attempts=2):
|
||||
try:
|
||||
token_uri = validate_google_token_uri(creds.get("token_uri"))
|
||||
except ValueError as exc:
|
||||
return {"status": "UNKNOWN", "message": str(exc), "client_id": creds.get("client_id")}
|
||||
data = {
|
||||
"client_id": creds["client_id"],
|
||||
"client_secret": creds["client_secret"],
|
||||
"refresh_token": creds["refresh_token"],
|
||||
"grant_type": "refresh_token",
|
||||
}
|
||||
try:
|
||||
response = requests.post(
|
||||
token_uri, data=data, proxies=proxy, timeout=timeout, allow_redirects=False,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "client_id": creds.get("client_id")}
|
||||
if response.status_code in TOKEN_REDIRECT_STATUSES:
|
||||
return {
|
||||
"status": "UNKNOWN",
|
||||
"http_status": response.status_code,
|
||||
"message": "Google OAuth token endpoint redirect refused",
|
||||
"client_id": creds.get("client_id"),
|
||||
}
|
||||
if response.status_code == 200:
|
||||
payload = response.json()
|
||||
project_id = creds.get("quota_project_id") or creds.get("project_id")
|
||||
result = {"status": "VALID", "message": "refresh token accepted", "client_id": creds.get("client_id"), "project_id": project_id, "expires_in": payload.get("expires_in")}
|
||||
if probe_vertex:
|
||||
google_vertex = probe_vertex_llm(
|
||||
payload.get("access_token"), project_id, proxy,
|
||||
vertex_timeout, vertex_locations, vertex_models, vertex_max_attempts,
|
||||
)
|
||||
anthropic_vertex = probe_vertex_anthropic(
|
||||
payload.get("access_token"), project_id, proxy,
|
||||
vertex_timeout, vertex_anthropic_locations, vertex_anthropic_models, vertex_anthropic_max_attempts,
|
||||
) if vertex_anthropic_models else {"enabled": False, "message": ""}
|
||||
vertex = merge_vertex_results(google_vertex, anthropic_vertex)
|
||||
result.update({
|
||||
"vertex_enabled": vertex.get("enabled"),
|
||||
"vertex_location": vertex.get("location", ""),
|
||||
"vertex_model": vertex.get("model", ""),
|
||||
"vertex_available_models": vertex.get("available_models") or [],
|
||||
"vertex_google_enabled": vertex.get("google_enabled"),
|
||||
"vertex_anthropic_enabled": vertex.get("anthropic_enabled"),
|
||||
"vertex_total_tokens": vertex.get("total_tokens"),
|
||||
"vertex_message": vertex.get("message", ""),
|
||||
})
|
||||
if vertex.get("enabled"):
|
||||
result["status"] = "VERTEX"
|
||||
result["message"] = "refresh token accepted; Vertex countTokens accepted"
|
||||
return result
|
||||
message = request_error_message(response)
|
||||
lower = message.lower()
|
||||
if response.status_code in (400, 401) and any(item in lower for item in ("invalid_grant", "invalid_client", "unauthorized_client")):
|
||||
return {"status": "DEAD", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
|
||||
if response.status_code in (401, 403):
|
||||
return {"status": "RESTRICTED", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
|
||||
if response.status_code == 429:
|
||||
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "client_id": creds.get("client_id")}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
|
||||
|
||||
|
||||
def write_result(key, detector, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, {"detector": detector, **result}, source, finding, detector)
|
||||
extra = result.get("client_email") or result.get("client_id") or result.get("project_id") or source
|
||||
message = result.get("message", "")
|
||||
if result.get("status") == "VERTEX":
|
||||
models = result.get("vertex_available_models") or []
|
||||
model_text = ",".join(str(item) for item in models) or f"{result.get('vertex_location', '')}/{result.get('vertex_model', '')}".strip("/")
|
||||
message = f"{message}; models={model_text}"
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, transaction_status_files(), key, result["status"], message, extra,
|
||||
)
|
||||
if result.get("status") == "VERTEX" and result.get("vertex_google_enabled"):
|
||||
append_status(VERTEX_GEMINI_FILE, key, result["status"], message, extra)
|
||||
if result.get("status") == "VERTEX" and result.get("vertex_anthropic_enabled"):
|
||||
append_status(VERTEX_ANTHROPIC_FILE, key, result["status"], message, extra)
|
||||
record_validation_result(SERVICE, key, {"detector": detector, **result}, source, finding, detector)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="GCP credential checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", default=PLAIN_FILE)
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=25)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--probe-vertex", action="store_true", help="After OAuth succeeds, probe Vertex AI Gemini with countTokens through the configured proxy")
|
||||
parser.add_argument("--vertex-timeout", type=int, default=6, help="Seconds per Vertex countTokens request")
|
||||
parser.add_argument("--vertex-max-attempts", type=int, default=6, help="Maximum location/model countTokens attempts per credential")
|
||||
parser.add_argument("--vertex-locations", default=",".join(VERTEX_LOCATIONS), help="Comma-separated Vertex locations to probe")
|
||||
parser.add_argument("--vertex-models", default=",".join(VERTEX_MODELS), help="Comma-separated Vertex models to probe")
|
||||
parser.add_argument("--vertex-anthropic-locations", default=",".join(VERTEX_ANTHROPIC_LOCATIONS), help="Comma-separated Vertex Anthropic locations to probe")
|
||||
parser.add_argument("--vertex-anthropic-models", default=",".join(VERTEX_ANTHROPIC_MODELS), help="Comma-separated Vertex Anthropic model IDs to probe with rawPredict")
|
||||
parser.add_argument("--vertex-anthropic-max-attempts", type=int, default=20, help="Maximum Anthropic location/model attempts per credential")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
|
||||
known = set(known_statuses)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("RATE_LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_valid:
|
||||
retry_statuses.update({"VALID", "VERTEX"})
|
||||
vertex_locations = [item.strip() for item in str(args.vertex_locations or "").split(",") if item.strip()]
|
||||
vertex_models = [item.strip() for item in str(args.vertex_models or "").split(",") if item.strip()]
|
||||
vertex_anthropic_locations = [item.strip() for item in str(args.vertex_anthropic_locations or "").split(",") if item.strip()]
|
||||
vertex_anthropic_models = [item.strip() for item in str(args.vertex_anthropic_models or "").split(",") if item.strip()]
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, detector, source, finding, parsed in extract_candidates(args.input, args.plain):
|
||||
if should_skip_key(
|
||||
key, checked, known, args, retry_statuses,
|
||||
service=SERVICE, source=source, finding=finding, detector=detector, known_statuses=known_statuses,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] {detector} candidate {mask_secret(key)} from {source}", flush=True)
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
if not parsed:
|
||||
result = {"status": "NO_CONTEXT", "message": "credential JSON is incomplete or unavailable"}
|
||||
elif detector == "GCP":
|
||||
result = check_service_account(
|
||||
parsed, proxy, args.timeout, args.probe_vertex,
|
||||
args.vertex_timeout, vertex_locations, vertex_models, args.vertex_max_attempts,
|
||||
vertex_anthropic_locations, vertex_anthropic_models, args.vertex_anthropic_max_attempts,
|
||||
)
|
||||
else:
|
||||
result = check_adc(
|
||||
parsed, proxy, args.timeout, args.probe_vertex,
|
||||
args.vertex_timeout, vertex_locations, vertex_models, args.vertex_max_attempts,
|
||||
vertex_anthropic_locations, vertex_anthropic_models, args.vertex_anthropic_max_attempts,
|
||||
)
|
||||
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}", flush=True)
|
||||
write_result(key, detector, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
time.sleep(0.1)
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,738 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from itertools import cycle
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
acquire_file_lock,
|
||||
append_checked,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
env_int,
|
||||
ensure_output_files as ensure_private_output_files,
|
||||
iter_findings,
|
||||
iter_bounded_text_lines,
|
||||
keycheck_input_mode,
|
||||
load_known_statuses,
|
||||
private_atomic_writer,
|
||||
record_cached_keycheck_occurrence,
|
||||
record_validation_result,
|
||||
release_file_lock,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from runtime_security import durable_replace, reject_reparse_components, require_private_directory, require_private_file
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "gemini"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
|
||||
def here(*parts):
|
||||
return os.path.join(SCRIPT_DIR, *parts)
|
||||
|
||||
|
||||
def out(*parts):
|
||||
return os.path.join(OUTPUT_DIR, *parts)
|
||||
|
||||
|
||||
def parent(*parts):
|
||||
return os.path.join(PARENT_DIR, *parts)
|
||||
|
||||
|
||||
# --- Configuration ---
|
||||
DEFAULT_INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
DEFAULT_PLAIN_INPUT_FILES = [out("gem.txt")]
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = out("geminiChecked.txt")
|
||||
RESULTS_FILE = out("geminiResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": out("geminiAlive.txt"),
|
||||
"VALID_RATE_LIMITED": out("geminiAliveRateLimited.txt"),
|
||||
"INVALID": out("geminiDead.txt"),
|
||||
"EXPIRED": out("geminiExpired.txt"),
|
||||
"LEAKED_REVOKED": out("geminiLeaked.txt"),
|
||||
"API_DISABLED": out("geminiDisabled.txt"),
|
||||
"RESTRICTED": out("geminiRestricted.txt"),
|
||||
"RATE_LIMITED": out("geminiRateLimited.txt"),
|
||||
"NETWORK_ERROR": out("geminiNetwork.txt"),
|
||||
"UNKNOWN": out("geminiUnknown.txt"),
|
||||
}
|
||||
|
||||
GEMINI_KEY_REGEX = re.compile(r"(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})")
|
||||
GEMINI_DETECTOR_NAMES = {"googleai", "googleaistudio"}
|
||||
MODELS_URL = "https://generativelanguage.googleapis.com/v1beta/models"
|
||||
|
||||
PROBE_MODEL_PRIORITY = [
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3.7-flash",
|
||||
]
|
||||
|
||||
MODEL_PRIORITY = [
|
||||
"gemini-3",
|
||||
"gemini-2.5-pro",
|
||||
"gemini-2.5-flash",
|
||||
"gemini-2.0-flash",
|
||||
"gemini-1.5-pro",
|
||||
"gemini-1.5-flash",
|
||||
"imagen",
|
||||
"embedding",
|
||||
]
|
||||
|
||||
|
||||
def now_iso():
|
||||
return datetime.now(timezone.utc).isoformat(timespec="seconds")
|
||||
|
||||
|
||||
def mask_key(key):
|
||||
if not key or len(key) < 12:
|
||||
return key
|
||||
return f"{key[:8]}...{key[-4:]}"
|
||||
|
||||
|
||||
def redact_key_text(text, key):
|
||||
if not isinstance(text, str):
|
||||
return text
|
||||
redacted = text.replace(key, "***REDACTED***") if key else text
|
||||
return GEMINI_KEY_REGEX.sub("***REDACTED***", redacted)
|
||||
|
||||
|
||||
def redact_result_text(result, key):
|
||||
if isinstance(result, dict):
|
||||
return {k: redact_result_text(v, key) for k, v in result.items()}
|
||||
if isinstance(result, list):
|
||||
return [redact_result_text(v, key) for v in result]
|
||||
return redact_key_text(result, key)
|
||||
|
||||
|
||||
def key_from_line(line):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
return None
|
||||
if "\t" in line:
|
||||
return line.split("\t", 1)[0].strip()
|
||||
return line.split(":", 1)[0].strip()
|
||||
|
||||
|
||||
def load_keys_from_file(filepath):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return set()
|
||||
if not os.path.exists(filepath):
|
||||
return set()
|
||||
keys = set()
|
||||
for line in iter_bounded_text_lines(filepath):
|
||||
key = key_from_line(line)
|
||||
if key:
|
||||
keys.add(key)
|
||||
return keys
|
||||
|
||||
|
||||
def load_checked_statuses(filepath=CHECKED_FILE):
|
||||
statuses = {}
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return statuses
|
||||
if not os.path.exists(filepath):
|
||||
return statuses
|
||||
for line in iter_bounded_text_lines(filepath):
|
||||
parts = line.rstrip("\n").split("\t")
|
||||
if not parts or not parts[0]:
|
||||
continue
|
||||
key = parts[0]
|
||||
status = parts[1] if len(parts) > 1 else "UNKNOWN"
|
||||
statuses[key] = status
|
||||
return statuses
|
||||
|
||||
|
||||
def load_all_known_keys():
|
||||
known = set(load_checked_statuses().keys())
|
||||
for path in STATUS_FILES.values():
|
||||
known.update(load_keys_from_file(path))
|
||||
return known
|
||||
|
||||
|
||||
def ensure_output_files():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
require_private_directory(OUTPUT_DIR, create=True)
|
||||
|
||||
legacy_rate_limited = out("geminiLimited.txt")
|
||||
rate_limited = STATUS_FILES["RATE_LIMITED"]
|
||||
if os.path.exists(legacy_rate_limited) and not os.path.exists(rate_limited):
|
||||
require_private_file(legacy_rate_limited)
|
||||
durable_replace(legacy_rate_limited, rate_limited)
|
||||
require_private_file(rate_limited)
|
||||
|
||||
paths = {CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()}
|
||||
ensure_private_output_files(paths)
|
||||
migrate_legacy_alive_rate_limited()
|
||||
|
||||
|
||||
def effective_status(result):
|
||||
status = result.get("status")
|
||||
probe_status = (result.get("probe") or {}).get("status")
|
||||
if status == "VALID" and probe_status == "RATE_LIMITED":
|
||||
return "VALID_RATE_LIMITED"
|
||||
return status
|
||||
|
||||
|
||||
def _gemini_status_layout():
|
||||
paths_by_status = {
|
||||
status: os.path.abspath(os.fspath(path))
|
||||
for status, path in STATUS_FILES.items()
|
||||
}
|
||||
paths = list(dict.fromkeys(paths_by_status.values()))
|
||||
directories = {os.path.normcase(os.path.dirname(path)) for path in paths}
|
||||
if len(paths) != len(paths_by_status) or len(directories) != 1:
|
||||
raise RuntimeError("Gemini status files must be unique files in one directory")
|
||||
directory = os.path.dirname(paths[0])
|
||||
require_private_directory(directory, create=True)
|
||||
return paths_by_status, paths, os.path.join(directory, "geminiStatus.lock")
|
||||
|
||||
|
||||
def migrate_legacy_alive_rate_limited():
|
||||
paths_by_status, _, lock_path = _gemini_status_layout()
|
||||
alive_path = paths_by_status["VALID"]
|
||||
limited_path = paths_by_status["VALID_RATE_LIMITED"]
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
if not os.path.lexists(alive_path):
|
||||
return
|
||||
require_private_file(alive_path)
|
||||
|
||||
keep = []
|
||||
moved = {}
|
||||
for line in iter_bounded_text_lines(alive_path):
|
||||
key = key_from_line(line)
|
||||
if key and line.rstrip("\r\n").endswith(":RATE_LIMITED"):
|
||||
moved.setdefault(key, line if line.endswith("\n") else f"{line}\n")
|
||||
else:
|
||||
keep.append(line)
|
||||
if not moved:
|
||||
return
|
||||
|
||||
if os.path.lexists(limited_path):
|
||||
require_private_file(limited_path)
|
||||
existing = _normalized_status_lines(list(iter_bounded_text_lines(limited_path)))
|
||||
limited = []
|
||||
published = set()
|
||||
for line in existing:
|
||||
key = key_from_line(line)
|
||||
if key in moved:
|
||||
if key in published:
|
||||
continue
|
||||
published.add(key)
|
||||
limited.append(line)
|
||||
for key, line in moved.items():
|
||||
if key not in published:
|
||||
limited.append(line)
|
||||
published.add(key)
|
||||
|
||||
_validate_status_snapshot(limited_path, limited)
|
||||
_validate_status_snapshot(alive_path, keep)
|
||||
|
||||
# Make every moved key durable before publishing the source snapshot
|
||||
# that removes it. An interruption can therefore only leave duplicates.
|
||||
_replace_status_snapshot(limited_path, limited)
|
||||
confirmed = {key: 0 for key in moved}
|
||||
for line in iter_bounded_text_lines(limited_path):
|
||||
key = key_from_line(line)
|
||||
if key in confirmed:
|
||||
confirmed[key] += 1
|
||||
if any(count != 1 for count in confirmed.values()):
|
||||
raise RuntimeError("Gemini legacy rate-limited status publication was incomplete")
|
||||
_replace_status_snapshot(alive_path, keep)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def load_proxies(proxy_file):
|
||||
if not os.path.exists(proxy_file):
|
||||
print(f"Info: {proxy_file} not found. Requests will go directly.")
|
||||
return None
|
||||
|
||||
proxies = []
|
||||
with open(proxy_file, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
ip, port, login, password = line.split(":")
|
||||
proxy_url = f"http://{login}:{password}@{ip}:{port}"
|
||||
proxies.append({"http": proxy_url, "https": proxy_url})
|
||||
except ValueError:
|
||||
print(f"Warning: bad proxy format: {line}. Skipping.")
|
||||
|
||||
if not proxies:
|
||||
print(f"Warning: {proxy_file} is empty. Requests will go directly.")
|
||||
return None
|
||||
|
||||
print(f"Loaded proxies: {len(proxies)}")
|
||||
return cycle(proxies)
|
||||
|
||||
|
||||
def status_file_line(key, result, status):
|
||||
if status in ("VALID", "VALID_RATE_LIMITED"):
|
||||
models_str = ",".join(result.get("notable_models", [])) or "models-only"
|
||||
probe_status = result.get("probe", {}).get("status", "not_probed")
|
||||
return f"{key}:[{models_str}]:{result.get('model_class', 'unknown')}:{probe_status}\n"
|
||||
message = (result.get("error", {}).get("message") or "").replace("\n", " ")[:300]
|
||||
return f"{key}\t{status}\t{message}\n"
|
||||
|
||||
|
||||
def _normalized_status_lines(lines):
|
||||
return [line if line.endswith("\n") else f"{line}\n" for line in lines]
|
||||
|
||||
|
||||
def _validate_status_snapshot(path, lines):
|
||||
max_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_BYTES", 32 * 1024 * 1024))
|
||||
max_items = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_ITEMS", 100000))
|
||||
max_line_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_LINE_BYTES", 8192))
|
||||
if len(lines) > max_items:
|
||||
raise RuntimeError(f"Gemini status file exceeds its item bound: {path}")
|
||||
total = 0
|
||||
for index, line in enumerate(lines, 1):
|
||||
encoded = line.encode("utf-8")
|
||||
if len(encoded) > max_line_bytes:
|
||||
raise RuntimeError(f"Gemini status line exceeds its byte bound: {path}:{index}")
|
||||
total += len(encoded)
|
||||
if total > max_bytes:
|
||||
raise RuntimeError(f"Gemini status file exceeds its aggregate byte bound: {path}")
|
||||
|
||||
|
||||
def _replace_status_snapshot(path, lines):
|
||||
with private_atomic_writer(path, binary=True, suffix=".status.tmp") as handle:
|
||||
for line in lines:
|
||||
handle.write(line.encode("utf-8"))
|
||||
|
||||
|
||||
def append_status_file(key, result):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
paths_by_status, paths, lock_path = _gemini_status_layout()
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
status = effective_status(result)
|
||||
target_path = paths_by_status.get(status, paths_by_status["UNKNOWN"])
|
||||
new_line = status_file_line(key, result, status)
|
||||
snapshots = {}
|
||||
for path in paths:
|
||||
if os.path.lexists(path):
|
||||
reject_reparse_components(path)
|
||||
snapshots[path] = _normalized_status_lines(list(iter_bounded_text_lines(path)))
|
||||
|
||||
rewritten = {
|
||||
path: [line for line in lines if key_from_line(line) != key]
|
||||
for path, lines in snapshots.items()
|
||||
}
|
||||
rewritten[target_path].insert(0, new_line)
|
||||
for path, lines in rewritten.items():
|
||||
_validate_status_snapshot(path, lines)
|
||||
|
||||
# Publish the new classification before removing any old copies. A
|
||||
# failure after this point can leave duplicates, but never no status.
|
||||
_replace_status_snapshot(target_path, rewritten[target_path])
|
||||
for path in paths:
|
||||
if path == target_path or rewritten[path] == snapshots[path]:
|
||||
continue
|
||||
_replace_status_snapshot(path, rewritten[path])
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def append_checked_file(key, result):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
append_checked(CHECKED_FILE, key, effective_status(result))
|
||||
|
||||
|
||||
def is_gemini_detector(detector):
|
||||
return str(detector or "").lower() in GEMINI_DETECTOR_NAMES
|
||||
|
||||
|
||||
def custom_detector_name(data):
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
|
||||
name = extra.get("name") or ""
|
||||
if str(data.get("DetectorName") or "").lower() == "customregex" and is_gemini_detector(name):
|
||||
return name
|
||||
return ""
|
||||
|
||||
|
||||
def detector_name_from_finding(data):
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
if is_gemini_detector(data.get("DetectorName")):
|
||||
return data.get("DetectorName")
|
||||
custom_name = custom_detector_name(data)
|
||||
if custom_name:
|
||||
return custom_name
|
||||
|
||||
# Old wrapped format from earlier scanner versions.
|
||||
if is_gemini_detector(data.get("detector")):
|
||||
return data.get("detector")
|
||||
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")):
|
||||
return finding.get("DetectorName")
|
||||
custom_name = custom_detector_name(finding)
|
||||
if custom_name:
|
||||
return custom_name
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def extract_key_from_finding(data):
|
||||
if is_gemini_detector(data.get("DetectorName")):
|
||||
return data.get("Raw") or data.get("RawV2")
|
||||
if custom_detector_name(data):
|
||||
return data.get("Raw") or data.get("RawV2")
|
||||
|
||||
# Old wrapped format from earlier scanner versions.
|
||||
if is_gemini_detector(data.get("detector")):
|
||||
return data.get("raw") or data.get("raw_v2")
|
||||
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")):
|
||||
return finding.get("Raw") or finding.get("RawV2")
|
||||
if custom_detector_name(finding):
|
||||
return finding.get("Raw") or finding.get("RawV2")
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def iter_candidate_keys(input_file, plain_files):
|
||||
for item in iter_findings(input_file, ["GoogleAI", "GoogleAIStudio", "CustomRegex"]):
|
||||
finding = item.get("finding") or {}
|
||||
key = item.get("raw") or extract_key_from_finding(finding)
|
||||
if key and GEMINI_KEY_REGEX.fullmatch(key) and detector_name_from_finding(finding):
|
||||
yield item.get("source") or input_file, key, finding
|
||||
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
for path in plain_files:
|
||||
if not os.path.exists(path):
|
||||
print(f"Info: plain input {path} not found. Skipping.")
|
||||
continue
|
||||
try:
|
||||
keys = set()
|
||||
for line in iter_bounded_text_lines(path):
|
||||
keys.update(GEMINI_KEY_REGEX.findall(line))
|
||||
except (OSError, RuntimeError) as e:
|
||||
print(f"Warning: cannot read {path}: {e}")
|
||||
continue
|
||||
for idx, key in enumerate(sorted(keys), 1):
|
||||
yield f"{path}:plain:{idx}", key, {}
|
||||
|
||||
|
||||
def parse_error_response(response):
|
||||
try:
|
||||
payload = response.json()
|
||||
except json.JSONDecodeError:
|
||||
payload = {}
|
||||
|
||||
error = payload.get("error", {}) if isinstance(payload, dict) else {}
|
||||
return {
|
||||
"http_status": response.status_code,
|
||||
"code": error.get("code", response.status_code),
|
||||
"status": error.get("status", ""),
|
||||
"message": error.get("message", response.text[:500]),
|
||||
}
|
||||
|
||||
|
||||
def classify_error(error):
|
||||
http_status = int(error.get("http_status") or 0)
|
||||
status = str(error.get("status") or "").lower()
|
||||
message = str(error.get("message") or "").lower()
|
||||
|
||||
if "reported as leaked" in message or "leaked" in message:
|
||||
return "LEAKED_REVOKED"
|
||||
if "api key expired" in message or "expired" in message:
|
||||
return "EXPIRED"
|
||||
if "api key not valid" in message or "invalid api key" in message:
|
||||
return "INVALID"
|
||||
if "has not been used" in message or "it is disabled" in message or "api is disabled" in message:
|
||||
return "API_DISABLED"
|
||||
if "requests to this api" in message and "blocked" in message:
|
||||
return "RESTRICTED"
|
||||
if "api key restrictions" in message or "permission_denied" in status:
|
||||
return "RESTRICTED"
|
||||
if http_status == 429 or "resource_exhausted" in status or "quota" in message:
|
||||
return "RATE_LIMITED"
|
||||
if http_status in (400, 401):
|
||||
return "INVALID"
|
||||
if http_status == 403:
|
||||
return "RESTRICTED"
|
||||
return "UNKNOWN"
|
||||
|
||||
|
||||
def fetch_models(key, proxy, timeout, debug=False):
|
||||
try:
|
||||
response = requests.get(MODELS_URL, params={"key": key}, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as e:
|
||||
return {
|
||||
"status": "NETWORK_ERROR",
|
||||
"error": {"message": str(e)},
|
||||
"models": [],
|
||||
"model_infos": [],
|
||||
}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG /models: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}")
|
||||
|
||||
if response.status_code != 200:
|
||||
error = parse_error_response(response)
|
||||
return {
|
||||
"status": classify_error(error),
|
||||
"error": error,
|
||||
"models": [],
|
||||
"model_infos": [],
|
||||
}
|
||||
|
||||
payload = response.json()
|
||||
model_infos = payload.get("models", [])
|
||||
models = sorted({m.get("name", "").replace("models/", "") for m in model_infos if m.get("name")})
|
||||
return {
|
||||
"status": "VALID",
|
||||
"error": {},
|
||||
"models": models,
|
||||
"model_infos": model_infos,
|
||||
}
|
||||
|
||||
|
||||
def supported_methods_by_model(model_infos):
|
||||
output = {}
|
||||
for model in model_infos:
|
||||
name = model.get("name", "").replace("models/", "")
|
||||
if not name:
|
||||
continue
|
||||
output[name] = sorted(model.get("supportedGenerationMethods", []))
|
||||
return output
|
||||
|
||||
|
||||
def classify_models(models, methods_by_model):
|
||||
notable = []
|
||||
lower_models = {m.lower(): m for m in models}
|
||||
for marker in MODEL_PRIORITY:
|
||||
for lower, original in lower_models.items():
|
||||
if marker in lower and original not in notable:
|
||||
notable.append(original)
|
||||
|
||||
generation_models = sorted([
|
||||
model for model, methods in methods_by_model.items()
|
||||
if "generateContent" in methods
|
||||
])
|
||||
|
||||
if any("gemini-2.5-pro" in m.lower() for m in generation_models):
|
||||
model_class = "pro_generation"
|
||||
elif generation_models:
|
||||
model_class = "generation"
|
||||
elif models:
|
||||
model_class = "models_only"
|
||||
else:
|
||||
model_class = "no_models"
|
||||
|
||||
return notable[:20], generation_models, model_class
|
||||
|
||||
|
||||
def choose_probe_model(generation_models):
|
||||
available = set(generation_models)
|
||||
for model in PROBE_MODEL_PRIORITY:
|
||||
if model in available:
|
||||
return model
|
||||
return generation_models[0] if generation_models else None
|
||||
|
||||
|
||||
def probe_generation(key, model, proxy, timeout, debug=False):
|
||||
if not model:
|
||||
return {"status": "NO_GENERATION_MODEL", "model": None}
|
||||
|
||||
url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent"
|
||||
headers = {"x-goog-api-key": key, "Content-Type": "application/json"}
|
||||
payload = {
|
||||
"contents": [{"parts": [{"text": "ping"}]}],
|
||||
"generationConfig": {"maxOutputTokens": 1},
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as e:
|
||||
return {"status": "NETWORK_ERROR", "model": model, "error": {"message": str(e)}}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG probe {model}: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}")
|
||||
|
||||
if response.status_code == 200:
|
||||
return {"status": "GENERATION_OK", "model": model}
|
||||
|
||||
error = parse_error_response(response)
|
||||
return {"status": classify_error(error), "model": model, "error": error}
|
||||
|
||||
|
||||
def check_key(key, proxy, args):
|
||||
result = fetch_models(key, proxy, args.timeout, args.debug)
|
||||
result = redact_result_text(result, key)
|
||||
result.update({
|
||||
"checked_at": now_iso(),
|
||||
"key_masked": mask_key(key),
|
||||
"model_count": len(result.get("models", [])),
|
||||
})
|
||||
|
||||
if result["status"] != "VALID":
|
||||
result["notable_models"] = []
|
||||
result["generation_models"] = []
|
||||
result["model_class"] = "none"
|
||||
return result
|
||||
|
||||
methods_by_model = supported_methods_by_model(result.get("model_infos", []))
|
||||
notable, generation_models, model_class = classify_models(result["models"], methods_by_model)
|
||||
|
||||
result["methods_by_model"] = methods_by_model
|
||||
result["notable_models"] = notable
|
||||
result["generation_models"] = generation_models[:50]
|
||||
result["model_class"] = model_class
|
||||
result["billing_status"] = "unknown"
|
||||
|
||||
if args.probe_generation:
|
||||
probe_model = choose_probe_model(generation_models)
|
||||
result["probe"] = redact_result_text(probe_generation(key, probe_model, proxy, args.timeout, args.debug), key)
|
||||
else:
|
||||
result["probe"] = {"status": "not_probed", "model": None}
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def print_result(index, source, key, result):
|
||||
print(f"\n[{index}] Candidate {mask_key(key)} from {source}")
|
||||
print(f" STATUS: {result['status']}")
|
||||
|
||||
if result["status"] == "VALID":
|
||||
print(f" MODELS: {result.get('model_count', 0)} total; class={result.get('model_class')}")
|
||||
notable = result.get("notable_models", [])[:8]
|
||||
if notable:
|
||||
print(f" NOTABLE: {', '.join(notable)}")
|
||||
probe = result.get("probe", {})
|
||||
print(f" PROBE: {probe.get('status')} ({probe.get('model')})")
|
||||
if effective_status(result) == "VALID_RATE_LIMITED":
|
||||
print(f" OUT: {STATUS_FILES['VALID_RATE_LIMITED']}")
|
||||
else:
|
||||
error = result.get("error", {})
|
||||
message = (error.get("message") or "").replace("\n", " ")[:300]
|
||||
if message:
|
||||
print(f" MESSAGE: {message}")
|
||||
|
||||
print(f" OUT: {STATUS_FILES.get(result['status'], STATUS_FILES['UNKNOWN'])}")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Gemini / Google AI API key classifier")
|
||||
parser.add_argument("--input", default=DEFAULT_INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=None, help="Plain text file with Gemini keys. Can be repeated.")
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--probe-generation", action="store_true", help="Optionally call generateContent, preferring gemini-3.1-pro-preview when available.")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
plain_files = args.plain if args.plain is not None else DEFAULT_PLAIN_INPUT_FILES
|
||||
ensure_output_files()
|
||||
|
||||
print("--- Gemini key checker ---")
|
||||
print("Default mode: /models only. Use --probe-generation for runtime/billing probe.")
|
||||
print(f"Workspace: {SCRIPT_DIR}")
|
||||
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked_statuses = load_checked_statuses()
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
|
||||
known_keys = set(known_statuses)
|
||||
retry_statuses = set()
|
||||
if args.retry_limited:
|
||||
retry_statuses.update(("RATE_LIMITED", "VALID_RATE_LIMITED"))
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK_ERROR")
|
||||
if args.retry_valid:
|
||||
retry_statuses.update(("VALID", "VALID_RATE_LIMITED"))
|
||||
print(f"Loaded known keys: {len(known_keys)}; checked records: {len(checked_statuses)}")
|
||||
|
||||
seen_this_run = set()
|
||||
processed = 0
|
||||
skipped = 0
|
||||
|
||||
for source, key, finding in iter_candidate_keys(args.input, plain_files):
|
||||
if keycheck_input_mode() != 'postgres' and key in seen_this_run:
|
||||
cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN'
|
||||
detector = detector_name_from_finding(finding) or "GoogleAI"
|
||||
record_cached_keycheck_occurrence(SERVICE, key, cached_status, source, finding, detector)
|
||||
skipped += 1
|
||||
continue
|
||||
seen_this_run.add(key)
|
||||
|
||||
detector = detector_name_from_finding(finding) or "GoogleAI"
|
||||
if should_skip_key(
|
||||
key, checked_statuses, known_keys, args, retry_statuses,
|
||||
service=SERVICE, source=source, finding=finding, detector=detector,
|
||||
known_statuses=known_statuses,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
|
||||
processed += 1
|
||||
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(key, proxy, args)
|
||||
result["source"] = source
|
||||
event_result = {**result, "status": effective_status(result)}
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, event_result, source, finding, detector)
|
||||
|
||||
print_result(processed, source, key, result)
|
||||
append_status_file(key, result)
|
||||
append_checked_file(key, result)
|
||||
record_validation_result(SERVICE, key, {**result, "status": effective_status(result)}, source, finding, detector)
|
||||
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = effective_status(result)
|
||||
|
||||
# Small pause helps when many keys hit the same API/proxy.
|
||||
time.sleep(0.1)
|
||||
|
||||
print("\n--- Done ---")
|
||||
print(f"Processed: {processed}")
|
||||
print(f"Skipped: {skipped}")
|
||||
print(f"Results: {RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,212 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
read_plain_keys,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "github"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
PLAIN_FILE = os.path.join(OUTPUT_DIR, "github.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "githubChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "githubResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "githubAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "githubDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "githubRestricted.txt"),
|
||||
"RATE_LIMITED": os.path.join(OUTPUT_DIR, "githubRateLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "githubNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "githubUnknown.txt"),
|
||||
"REFRESH_TOKEN": os.path.join(OUTPUT_DIR, "githubRefreshToken.txt"),
|
||||
}
|
||||
|
||||
GITHUB_TOKEN_RE = re.compile(r"\b(?:gh[pousr]_[A-Za-z0-9_]{20,}|github_pat_[A-Za-z0-9_]{20,})\b")
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), PLAIN_FILE])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["Github", "GitHubOauth2"]):
|
||||
text = "\n".join(str(value or "") for value in [item.get("raw"), item.get("raw_v2")])
|
||||
for match in GITHUB_TOKEN_RE.findall(text):
|
||||
yield match, item["source"], item["finding"]
|
||||
|
||||
for item in read_plain_keys(plain_files, GITHUB_TOKEN_RE):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def is_rate_limited(response):
|
||||
remaining = response.headers.get("X-RateLimit-Remaining")
|
||||
return response.status_code in (403, 429) and remaining == "0"
|
||||
|
||||
|
||||
def check_token(token, proxy, timeout):
|
||||
if token.startswith("ghr_"):
|
||||
return {
|
||||
"status": "REFRESH_TOKEN",
|
||||
"message": "GitHub refresh tokens cannot be checked directly as bearer API tokens",
|
||||
}
|
||||
|
||||
if token.startswith("ghs_"):
|
||||
url = "https://api.github.com/installation/repositories"
|
||||
token_kind = "installation"
|
||||
else:
|
||||
url = "https://api.github.com/user"
|
||||
token_kind = "user"
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {token}",
|
||||
"Accept": "application/vnd.github+json",
|
||||
"X-GitHub-Api-Version": "2022-11-28",
|
||||
"User-Agent": "local-keycheck-github",
|
||||
}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "token_kind": token_kind}
|
||||
|
||||
message = request_error_message(response)
|
||||
scopes = response.headers.get("X-OAuth-Scopes", "")
|
||||
accepted_scopes = response.headers.get("X-Accepted-OAuth-Scopes", "")
|
||||
rate_remaining = response.headers.get("X-RateLimit-Remaining", "")
|
||||
rate_reset = response.headers.get("X-RateLimit-Reset", "")
|
||||
|
||||
if response.status_code == 200:
|
||||
payload = response.json()
|
||||
extra = {
|
||||
"token_kind": token_kind,
|
||||
"scopes": scopes,
|
||||
"accepted_scopes": accepted_scopes,
|
||||
"rate_remaining": rate_remaining,
|
||||
"rate_reset": rate_reset,
|
||||
}
|
||||
if token_kind == "installation":
|
||||
extra["repo_count"] = payload.get("total_count")
|
||||
return {"status": "VALID", "message": "installation token accepted", **extra}
|
||||
return {
|
||||
"status": "VALID",
|
||||
"message": "user token accepted",
|
||||
"login": payload.get("login"),
|
||||
"account_type": payload.get("type"),
|
||||
**extra,
|
||||
}
|
||||
|
||||
if response.status_code == 401:
|
||||
return {"status": "DEAD", "http_status": 401, "message": message, "token_kind": token_kind}
|
||||
if is_rate_limited(response):
|
||||
return {"status": "RATE_LIMITED", "http_status": response.status_code, "message": message, "token_kind": token_kind, "rate_reset": rate_reset}
|
||||
if response.status_code == 403:
|
||||
return {"status": "RESTRICTED", "http_status": 403, "message": message, "token_kind": token_kind, "scopes": scopes}
|
||||
if response.status_code in (404, 422):
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "token_kind": token_kind}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "token_kind": token_kind}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "token_kind": token_kind}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding)
|
||||
extra = result.get("login") or result.get("repo_count") or result.get("token_kind") or source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="GitHub token checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("RATE_LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_restricted:
|
||||
retry_statuses.add("RESTRICTED")
|
||||
|
||||
plain_files = args.plain or [PLAIN_FILE]
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in extract_candidates(args.input, plain_files):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] GitHub candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_token(key, proxy, args.timeout)
|
||||
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
time.sleep(0.1)
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,178 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
read_plain_keys,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
record_validation_result,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
|
||||
SERVICE = "gitlab"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
PLAIN_FILE = os.path.join(OUTPUT_DIR, "gitlab.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "gitlabChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "gitlabResults.jsonl")
|
||||
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "gitlabAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "gitlabDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "gitlabRestricted.txt"),
|
||||
"RATE_LIMITED": os.path.join(OUTPUT_DIR, "gitlabRateLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "gitlabNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "gitlabUnknown.txt"),
|
||||
}
|
||||
|
||||
GITLAB_TOKEN_RE = re.compile(r"\b(?:glpat|gloas|glcbt|glimt|glrt|glft|glsoat)-[A-Za-z0-9_\-=]{20,}\b")
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), PLAIN_FILE])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, ["Gitlab"]):
|
||||
text = "\n".join(str(value or "") for value in [item.get("raw"), item.get("raw_v2")])
|
||||
for match in GITLAB_TOKEN_RE.findall(text):
|
||||
yield match, item["source"], item["finding"]
|
||||
|
||||
for item in read_plain_keys(plain_files, GITLAB_TOKEN_RE):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def check_token(token, base_url, proxy, timeout):
|
||||
base_url = base_url.rstrip("/")
|
||||
url = f"{base_url}/api/v4/user"
|
||||
headers = {"Authorization": f"Bearer {token}", "User-Agent": "local-keycheck-gitlab"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "base_url": base_url}
|
||||
|
||||
message = request_error_message(response)
|
||||
retry_after = response.headers.get("Retry-After", "")
|
||||
rate_remaining = response.headers.get("RateLimit-Remaining") or response.headers.get("X-RateLimit-Remaining") or ""
|
||||
|
||||
if response.status_code == 200:
|
||||
payload = response.json()
|
||||
return {
|
||||
"status": "VALID",
|
||||
"message": "token accepted",
|
||||
"username": payload.get("username"),
|
||||
"name": payload.get("name"),
|
||||
"user_id": payload.get("id"),
|
||||
"base_url": base_url,
|
||||
"rate_remaining": rate_remaining,
|
||||
}
|
||||
if response.status_code == 401:
|
||||
return {"status": "DEAD", "http_status": 401, "message": message, "base_url": base_url}
|
||||
if response.status_code == 403:
|
||||
# TruffleHog treats 403 as a live token with insufficient scope or blocked account.
|
||||
return {"status": "RESTRICTED", "http_status": 403, "message": message, "base_url": base_url}
|
||||
if response.status_code == 429:
|
||||
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "base_url": base_url, "retry_after": retry_after}
|
||||
if response.status_code >= 500:
|
||||
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "base_url": base_url}
|
||||
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "base_url": base_url}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding)
|
||||
extra = result.get("username") or result.get("user_id") or source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="GitLab token checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--base-url", default="https://gitlab.com")
|
||||
parser.add_argument("--timeout", type=int, default=20)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("RATE_LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_restricted:
|
||||
retry_statuses.add("RESTRICTED")
|
||||
|
||||
plain_files = args.plain or [PLAIN_FILE]
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in extract_candidates(args.input, plain_files):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] GitLab candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_token(key, args.base_url, proxy, args.timeout)
|
||||
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
time.sleep(0.1)
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,275 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl,
|
||||
classify_common_http_status,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
read_plain_keys,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
request_error_message,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
|
||||
SERVICE = "groq"
|
||||
DETECTOR = "Groq"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "groqChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "groqResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "groqAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "groqNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "groqDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "groqRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "groqLimited.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "groqNoContext.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "groqNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "groqUnknown.txt"),
|
||||
}
|
||||
|
||||
GROQ_KEY_REGEX = re.compile(r"\bgsk_[A-Za-z0-9_-]{20,}\b")
|
||||
MODELS_URL = "https://api.groq.com/openai/v1/models"
|
||||
CHAT_URL = "https://api.groq.com/openai/v1/chat/completions"
|
||||
CHAT_MODEL_PRIORITY = (
|
||||
"llama-3.1-8b-instant",
|
||||
"llama-3.3-70b-versatile",
|
||||
"llama3-8b-8192",
|
||||
"llama3-70b-8192",
|
||||
"mixtral-8x7b-32768",
|
||||
"gemma2-9b-it",
|
||||
)
|
||||
NO_BALANCE_MARKERS = (
|
||||
"quota",
|
||||
"billing",
|
||||
"balance",
|
||||
"credit",
|
||||
"payment",
|
||||
"insufficient",
|
||||
"depleted",
|
||||
)
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, [DETECTOR]):
|
||||
key = item["raw"]
|
||||
if key and GROQ_KEY_REGEX.fullmatch(key):
|
||||
yield key, item["source"], item["finding"]
|
||||
|
||||
for item in read_plain_keys(plain_files, GROQ_KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def classify_groq_response(response):
|
||||
message = request_error_message(response).lower()
|
||||
if response.status_code == 401:
|
||||
return "DEAD"
|
||||
if response.status_code == 403:
|
||||
return "RESTRICTED"
|
||||
if response.status_code == 429:
|
||||
if any(marker in message for marker in NO_BALANCE_MARKERS):
|
||||
return "NO_BALANCE"
|
||||
return "LIMITED"
|
||||
return classify_common_http_status(response.status_code)
|
||||
|
||||
|
||||
def notable_models(payload):
|
||||
models = payload.get("data", []) if isinstance(payload, dict) else []
|
||||
ids = []
|
||||
for item in models:
|
||||
if isinstance(item, dict) and item.get("id"):
|
||||
ids.append(str(item.get("id")))
|
||||
priority = []
|
||||
for marker in ("llama", "mixtral", "gemma", "whisper"):
|
||||
for model in ids:
|
||||
if marker in model.lower() and model not in priority:
|
||||
priority.append(model)
|
||||
return priority[:20], len(ids), ids
|
||||
|
||||
|
||||
def choose_chat_model(model_ids):
|
||||
model_ids = [str(model or "") for model in model_ids if model]
|
||||
by_lower = {model.lower(): model for model in model_ids}
|
||||
for model in CHAT_MODEL_PRIORITY:
|
||||
if model.lower() in by_lower:
|
||||
return by_lower[model.lower()]
|
||||
for marker in ("llama", "mixtral", "gemma"):
|
||||
for model in model_ids:
|
||||
lowered = model.lower()
|
||||
if marker in lowered and "whisper" not in lowered and "guard" not in lowered:
|
||||
return model
|
||||
return ""
|
||||
|
||||
|
||||
def probe_chat_completion(key, model, proxy, timeout, debug=False):
|
||||
if not model:
|
||||
return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""}
|
||||
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1}
|
||||
try:
|
||||
response = requests.post(CHAT_URL, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "model": model}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG chat ping {model}: HTTP {response.status_code}: {response.text[:500].replace(key, '***REDACTED***')}")
|
||||
|
||||
if response.status_code == 200:
|
||||
return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model}
|
||||
return {
|
||||
"status": classify_groq_response(response),
|
||||
"http_status": response.status_code,
|
||||
"message": request_error_message(response).replace(key, "***REDACTED***"),
|
||||
"model": model,
|
||||
}
|
||||
|
||||
|
||||
def check_key(key, proxy, timeout, debug=False):
|
||||
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
||||
try:
|
||||
response = requests.get(MODELS_URL, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG /models: HTTP {response.status_code}: {response.text[:500].replace(key, '***REDACTED***')}")
|
||||
|
||||
if response.status_code == 200:
|
||||
try:
|
||||
payload = response.json()
|
||||
except ValueError:
|
||||
payload = {}
|
||||
models, model_count, model_ids = notable_models(payload)
|
||||
chat_model = choose_chat_model(model_ids)
|
||||
probe = probe_chat_completion(key, chat_model, proxy, timeout, debug)
|
||||
if probe.get("status") != "GENERATION_OK":
|
||||
return {
|
||||
"status": probe.get("status") or "UNKNOWN",
|
||||
"message": probe.get("message", ""),
|
||||
"model_count": model_count,
|
||||
"models": models,
|
||||
"llm_probe_status": probe.get("status"),
|
||||
"llm_probe_model": probe.get("model", chat_model),
|
||||
"llm_probe_http_status": probe.get("http_status"),
|
||||
}
|
||||
return {
|
||||
"status": "VALID",
|
||||
"message": f"chat ping ok; model={chat_model}; models={model_count}",
|
||||
"model_count": model_count,
|
||||
"models": models,
|
||||
"llm_probe_status": probe.get("status"),
|
||||
"llm_probe_model": chat_model,
|
||||
}
|
||||
|
||||
return {
|
||||
"status": classify_groq_response(response),
|
||||
"http_status": response.status_code,
|
||||
"message": request_error_message(response).replace(key, "***REDACTED***"),
|
||||
}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
extra = ",".join(result.get("models") or [])[:500] if status == "VALID" else source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Groq key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add("UNKNOWN")
|
||||
if args.retry_restricted:
|
||||
retry_statuses.add("RESTRICTED")
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.add("NO_BALANCE")
|
||||
if args.retry_valid:
|
||||
retry_statuses.add("VALID")
|
||||
|
||||
print("--- Groq key checker ---")
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in extract_candidates(args.input, args.plain):
|
||||
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] Groq candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_key(key, proxy, args.timeout, args.debug)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,142 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl, classify_common_http_status, commit_status_transaction,
|
||||
default_input_file, default_proxy_file, ensure_output_files, iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses, load_known_keys, load_proxies, mask_secret,
|
||||
read_plain_keys, record_validation_result, recover_status_transaction,
|
||||
request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event,
|
||||
)
|
||||
|
||||
SERVICE = "huggingface"
|
||||
DETECTOR_NAMES = ["HuggingFace", "Huggingface"]
|
||||
DETECTOR = "HuggingFace"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "huggingfaceChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "huggingfaceResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "huggingfaceAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "huggingfaceDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "huggingfaceRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "huggingfaceLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "huggingfaceNetwork.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "huggingfaceNoContext.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "huggingfaceUnknown.txt"),
|
||||
}
|
||||
KEY_REGEX = re.compile(r"\bhf_[A-Za-z0-9]{20,}\b")
|
||||
WHOAMI_URL = "https://huggingface.co/api/whoami-v2"
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def iter_candidate_decisions(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, DETECTOR_NAMES):
|
||||
key = item.get("credential_secret_text") or item["raw"]
|
||||
if key:
|
||||
yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key))
|
||||
for item in read_plain_keys(plain_files, KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}, True
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files):
|
||||
if valid_format:
|
||||
yield key, source, finding
|
||||
|
||||
|
||||
def check_key(key, proxy, timeout):
|
||||
try:
|
||||
response = requests.get(WHOAMI_URL, headers={"Authorization": f"Bearer {key}"}, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
if response.status_code == 200:
|
||||
data = response.json() if response.text else {}
|
||||
return {"status": "VALID", "message": "whoami accepted", "username": data.get("name") or data.get("fullname") or ""}
|
||||
status = "RESTRICTED" if response.status_code == 403 else classify_common_http_status(response.status_code)
|
||||
return {"status": status, "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***")}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="HuggingFace key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network: retry_statuses.add("NETWORK")
|
||||
if args.retry_limited: retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
|
||||
if args.retry_restricted: retry_statuses.add("RESTRICTED")
|
||||
processed = skipped = 0
|
||||
print("--- HuggingFace key checker ---")
|
||||
postgres_mode = keycheck_input_mode() == "postgres"
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain):
|
||||
if not valid_format and not postgres_mode:
|
||||
skipped += 1
|
||||
continue
|
||||
if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] HuggingFace candidate {mask_secret(key)} from {source}")
|
||||
result = (
|
||||
check_key(key, next(proxy_cycler) if proxy_cycler else None, args.timeout)
|
||||
if valid_format else
|
||||
{"status": "NO_CONTEXT", "message": "candidate does not match canonical Hugging Face token format"}
|
||||
)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,499 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
combined_provider_routing_hint,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
provider_routing_database_failed,
|
||||
read_plain_keys,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from keycheckers.provider_resolution import resolve_provider_key
|
||||
|
||||
|
||||
SERVICE = "kimi"
|
||||
DETECTOR = "KimiMoonshot"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "kimiChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "kimiResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "kimiAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "kimiNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "kimiDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "kimiRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "kimiLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "kimiNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "kimiUnknown.txt"),
|
||||
}
|
||||
|
||||
KIMI_DETECTOR_NAMES = {"kimimoonshot", "moonshotai", "moonshot", "kimi"}
|
||||
KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"}
|
||||
QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"}
|
||||
DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"}
|
||||
GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"}
|
||||
AMBIGUOUS_QWEN_DEEPSEEK_HINT = "ambiguous_qwen_deepseek"
|
||||
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
|
||||
EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment"
|
||||
CANDIDATE_PROVIDER_ROUTE_FIELD = "_keycheck_candidate_provider_route"
|
||||
KIMI_KEY_MAX_BYTES = 512
|
||||
KIMI_KEY_REGEX = re.compile(
|
||||
r"(?<![A-Za-z0-9_-])sk-[A-Za-z0-9][A-Za-z0-9_-]{20,505}(?![A-Za-z0-9_-])"
|
||||
)
|
||||
FOREIGN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
|
||||
DEFAULT_BASE_URLS = (
|
||||
"https://api.moonshot.ai/v1",
|
||||
"https://api.moonshot.cn/v1",
|
||||
)
|
||||
QWEN_CONTEXT_REGEX = re.compile(
|
||||
r"(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
DEEPSEEK_CONTEXT_REGEX = re.compile(
|
||||
r"(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)", re.IGNORECASE,
|
||||
)
|
||||
KIMI_CONTEXT_REGEX = re.compile(
|
||||
r"(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def normalize_base_url(value):
|
||||
return str(value or "").strip().rstrip("/")
|
||||
|
||||
|
||||
def split_csv(value):
|
||||
if not value:
|
||||
return []
|
||||
if isinstance(value, str):
|
||||
return [item.strip() for item in value.split(",") if item.strip()]
|
||||
return [str(item).strip() for item in value if str(item).strip()]
|
||||
|
||||
|
||||
def unique_ordered(values):
|
||||
output = []
|
||||
seen = set()
|
||||
for value in values:
|
||||
normalized = normalize_base_url(value)
|
||||
if normalized and normalized not in seen:
|
||||
seen.add(normalized)
|
||||
output.append(normalized)
|
||||
return output
|
||||
|
||||
|
||||
def endpoint_label(base_url):
|
||||
parsed = urlparse(base_url)
|
||||
return parsed.netloc or base_url
|
||||
|
||||
|
||||
def finding_detector_names(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return set()
|
||||
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
|
||||
names = {
|
||||
str(finding.get("DetectorName") or finding.get("detector") or "").strip().lower(),
|
||||
str(extra.get("name") or "").strip().lower(),
|
||||
}
|
||||
return {name for name in names if name}
|
||||
|
||||
|
||||
def finding_has_detector(finding, detector_names):
|
||||
return bool(finding_detector_names(finding) & set(detector_names))
|
||||
|
||||
|
||||
def key_from_text(*values):
|
||||
for value in values:
|
||||
for match in KIMI_KEY_REGEX.finditer(str(value or "")):
|
||||
key = match.group(0)
|
||||
if not key.startswith(FOREIGN_KEY_PREFIXES):
|
||||
return key
|
||||
return ""
|
||||
|
||||
|
||||
def key_rejection_reason(key):
|
||||
value = str(key or "")
|
||||
try:
|
||||
key_bytes = len(value.encode("utf-8", errors="strict"))
|
||||
except UnicodeEncodeError:
|
||||
return "candidate is not valid UTF-8"
|
||||
if key_bytes > KIMI_KEY_MAX_BYTES:
|
||||
return f"candidate exceeds the {KIMI_KEY_MAX_BYTES}-byte key limit"
|
||||
if value.count("sk-") != 1:
|
||||
return "candidate contains multiple key prefixes"
|
||||
if not KIMI_KEY_REGEX.fullmatch(value) or value.startswith(FOREIGN_KEY_PREFIXES):
|
||||
return "candidate does not match the bounded Kimi/Moonshot key format"
|
||||
return ""
|
||||
|
||||
|
||||
def finding_provider_routing_hint(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return ""
|
||||
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
|
||||
persisted_hint = context.get("provider_hint")
|
||||
if context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE and persisted_hint in (
|
||||
*GENERIC_SK_PROVIDERS, AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT,
|
||||
):
|
||||
return persisted_hint
|
||||
|
||||
parts = [str(context.get(key) or "") for key in ("nearby", "file")]
|
||||
metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {}
|
||||
data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {}
|
||||
for details in data.values():
|
||||
if isinstance(details, dict):
|
||||
parts.extend(str(details.get(key) or "") for key in ("file", "repository", "repo", "link", "image"))
|
||||
|
||||
text = "\n".join(parts)
|
||||
evidence = set()
|
||||
if QWEN_CONTEXT_REGEX.search(text) or finding_has_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("qwen")
|
||||
if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("deepseek")
|
||||
if KIMI_CONTEXT_REGEX.search(text) or finding_has_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("kimi")
|
||||
if persisted_hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT:
|
||||
evidence.update(("qwen", "deepseek"))
|
||||
elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT:
|
||||
evidence.update(GENERIC_SK_PROVIDERS)
|
||||
elif persisted_hint in GENERIC_SK_PROVIDERS:
|
||||
evidence.add(persisted_hint)
|
||||
if len(evidence) > 1:
|
||||
return (
|
||||
AMBIGUOUS_QWEN_DEEPSEEK_HINT
|
||||
if evidence == {"qwen", "deepseek"}
|
||||
else AMBIGUOUS_GENERIC_SK_HINT
|
||||
)
|
||||
return next(iter(evidence)) if evidence else ""
|
||||
|
||||
|
||||
def iter_candidate_keys(input_file, plain_files, trusted_retry_files=None):
|
||||
detector_names = [
|
||||
"KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi",
|
||||
"kimimoonshot", "moonshotai", "moonshot", "kimi", "CustomRegex",
|
||||
]
|
||||
routing_decisions = {}
|
||||
seen = set()
|
||||
for item in iter_findings(input_file, detector_names):
|
||||
finding = dict(item.get("finding") or {})
|
||||
finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, None)
|
||||
candidate_metadata = item.get("candidate_metadata")
|
||||
persisted_route = ""
|
||||
if keycheck_input_mode() == "postgres" and isinstance(candidate_metadata, dict):
|
||||
persisted_route = str(candidate_metadata.get("provider_hint") or "").lower()
|
||||
if persisted_route == SERVICE:
|
||||
finding[CANDIDATE_PROVIDER_ROUTE_FIELD] = SERVICE
|
||||
if persisted_route != SERVICE and not finding_has_detector(finding, KIMI_DETECTOR_NAMES):
|
||||
continue
|
||||
key = key_from_text(item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2"))
|
||||
if not key or key_rejection_reason(key):
|
||||
continue
|
||||
if persisted_route == SERVICE:
|
||||
hint = SERVICE
|
||||
else:
|
||||
local_hint = finding_provider_routing_hint(finding)
|
||||
if key not in routing_decisions:
|
||||
routing_decisions[key] = combined_provider_routing_hint(key, local_hint)
|
||||
hint = routing_decisions[key]
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
if hint == "kimi" or (
|
||||
keycheck_input_mode() == "postgres"
|
||||
and hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
):
|
||||
seen.add(key)
|
||||
yield key, item.get("source") or input_file, finding
|
||||
|
||||
owned_retry_paths = {
|
||||
os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values()
|
||||
}
|
||||
retry_files = [
|
||||
path for path in (trusted_retry_files or [])
|
||||
if os.path.normcase(os.path.abspath(path)) in owned_retry_paths
|
||||
]
|
||||
for item in read_plain_keys(retry_files, KIMI_KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key_rejection_reason(key) or key in seen:
|
||||
continue
|
||||
hint = combined_provider_routing_hint(key, "kimi")
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
if hint == "kimi":
|
||||
seen.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def redact_text(value, key):
|
||||
text = str(value or "")[:1000]
|
||||
if key:
|
||||
text = text.replace(key, "***REDACTED***")
|
||||
return KIMI_KEY_REGEX.sub("***REDACTED***", text)
|
||||
|
||||
|
||||
def parse_error(response, key):
|
||||
try:
|
||||
payload = response.json()
|
||||
except ValueError:
|
||||
payload = {}
|
||||
error = payload.get("error") if isinstance(payload, dict) else {}
|
||||
if not isinstance(error, dict):
|
||||
error = {}
|
||||
message = error.get("message") or response.text[:500]
|
||||
return {
|
||||
"http_status": response.status_code,
|
||||
"code": error.get("code") or error.get("type") or "",
|
||||
"message": redact_text(message, key),
|
||||
}
|
||||
|
||||
|
||||
def classify_error(error):
|
||||
http_status = int(error.get("http_status") or 0)
|
||||
code = str(error.get("code") or "").lower()
|
||||
message = str(error.get("message") or "").lower()
|
||||
if http_status == 401 or any(marker in code for marker in ("invalid_authentication", "invalid_api_key")):
|
||||
return "DEAD"
|
||||
if http_status == 403:
|
||||
return "RESTRICTED"
|
||||
if http_status == 429:
|
||||
if any(marker in code + " " + message for marker in ("quota", "balance", "payment")):
|
||||
return "NO_BALANCE"
|
||||
return "LIMITED"
|
||||
if 500 <= http_status <= 599:
|
||||
return "NETWORK"
|
||||
return "UNKNOWN"
|
||||
|
||||
|
||||
def check_base_url(key, base_url, proxy, timeout, debug=False):
|
||||
url = f"{normalize_base_url(base_url)}/users/me/balance"
|
||||
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {
|
||||
"status": "NETWORK", "base_url": base_url,
|
||||
"region": endpoint_label(base_url), "message": redact_text(exc, key),
|
||||
}
|
||||
if debug:
|
||||
print(
|
||||
f" DEBUG {endpoint_label(base_url)} balance: HTTP {response.status_code}: "
|
||||
f"{redact_text(response.text[:500], key)}"
|
||||
)
|
||||
if response.status_code == 200:
|
||||
try:
|
||||
payload = response.json()
|
||||
data = payload.get("data") if isinstance(payload, dict) else None
|
||||
if not isinstance(data, dict) or "available_balance" not in data:
|
||||
raise ValueError("missing data.available_balance")
|
||||
available = float(data.get("available_balance"))
|
||||
voucher = float(data.get("voucher_balance", 0) or 0)
|
||||
cash = float(data.get("cash_balance", 0) or 0)
|
||||
except (TypeError, ValueError, json.JSONDecodeError) as exc:
|
||||
return {
|
||||
"status": "UNKNOWN", "base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"message": f"invalid balance response: {exc}",
|
||||
}
|
||||
status = "VALID" if available > 0 else "NO_BALANCE"
|
||||
return {
|
||||
"status": status,
|
||||
"authenticated": True,
|
||||
"base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"balance_usd": round(available, 6),
|
||||
"voucher_balance_usd": round(voucher, 6),
|
||||
"cash_balance_usd": round(cash, 6),
|
||||
"message": f"available_balance=${available:.6f}",
|
||||
}
|
||||
error = parse_error(response, key)
|
||||
return {
|
||||
"status": classify_error(error),
|
||||
"base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"http_status": response.status_code,
|
||||
"error": error,
|
||||
"message": error.get("message") or "",
|
||||
}
|
||||
|
||||
|
||||
def choose_final_status(attempts):
|
||||
statuses = [attempt.get("status") for attempt in attempts]
|
||||
for status in ("VALID", "NO_BALANCE", "LIMITED", "RESTRICTED", "UNKNOWN", "NETWORK"):
|
||||
if status in statuses:
|
||||
return status
|
||||
return "DEAD"
|
||||
|
||||
|
||||
def check_key(key, base_urls, proxy, timeout, debug=False):
|
||||
rejection = key_rejection_reason(key)
|
||||
if rejection:
|
||||
return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True}
|
||||
attempts = []
|
||||
for base_url in base_urls:
|
||||
result = check_base_url(key, base_url, proxy, timeout, debug)
|
||||
attempts.append(result)
|
||||
if result.get("status") in ("VALID", "NO_BALANCE"):
|
||||
return {**result, "attempts": attempts}
|
||||
status = choose_final_status(attempts)
|
||||
selected = next((attempt for attempt in attempts if attempt.get("status") == status), {})
|
||||
return {**selected, "status": status, "attempts": attempts}
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status,
|
||||
result.get("message", ""), result.get("region") or source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def retry_statuses_from_args(args):
|
||||
statuses = set()
|
||||
if args.retry_network:
|
||||
statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
statuses.add("LIMITED")
|
||||
if args.retry_unknown:
|
||||
statuses.add("UNKNOWN")
|
||||
if args.retry_restricted:
|
||||
statuses.add("RESTRICTED")
|
||||
if args.retry_no_balance:
|
||||
statuses.add("NO_BALANCE")
|
||||
if args.retry_valid:
|
||||
statuses.add("VALID")
|
||||
return statuses
|
||||
|
||||
|
||||
def retry_input_files_from_args(args):
|
||||
if args.recheck_all:
|
||||
statuses = list(STATUS_FILES)
|
||||
else:
|
||||
statuses = [
|
||||
status for flag, status in (
|
||||
(args.retry_network, "NETWORK"),
|
||||
(args.retry_limited, "LIMITED"),
|
||||
(args.retry_unknown, "UNKNOWN"),
|
||||
(args.retry_restricted, "RESTRICTED"),
|
||||
(args.retry_no_balance, "NO_BALANCE"),
|
||||
(args.retry_valid, "VALID"),
|
||||
) if flag
|
||||
]
|
||||
return [STATUS_FILES[status] for status in statuses]
|
||||
|
||||
|
||||
def base_urls_from_args(args):
|
||||
custom = []
|
||||
for value in args.base_url:
|
||||
custom.extend(split_csv(value))
|
||||
custom.extend(split_csv(os.getenv("KIMI_BASE_URLS") or os.getenv("MOONSHOT_BASE_URLS")))
|
||||
defaults = [] if args.no_default_base_urls else DEFAULT_BASE_URLS
|
||||
return unique_ordered([*custom, *defaults])
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Kimi / Moonshot AI key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--base-url", action="append", default=[])
|
||||
parser.add_argument("--no-default-base-urls", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = retry_statuses_from_args(args)
|
||||
retry_files = retry_input_files_from_args(args)
|
||||
base_urls = base_urls_from_args(args)
|
||||
if not base_urls:
|
||||
raise SystemExit("No Kimi/Moonshot base URLs configured")
|
||||
|
||||
print("--- Kimi / Moonshot AI key checker ---")
|
||||
print("base_urls: " + ", ".join(endpoint_label(url) for url in base_urls))
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in iter_candidate_keys(args.input, args.plain, retry_files):
|
||||
finding = dict(finding or {})
|
||||
candidate_route = str(finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, "") or "").lower()
|
||||
if should_skip_key(
|
||||
key, checked, known, args, retry_statuses, service=SERVICE,
|
||||
source=source, finding=finding, detector=DETECTOR,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] Kimi/Moonshot candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
routing_hint = "kimi"
|
||||
if keycheck_input_mode() == "postgres":
|
||||
if candidate_route == SERVICE:
|
||||
routing_hint = SERVICE
|
||||
else:
|
||||
routing_hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding))
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
if routing_hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT):
|
||||
result = resolve_provider_key(
|
||||
key, finding, proxy, args.timeout, args.debug,
|
||||
hint=routing_hint, origin_service=SERVICE,
|
||||
)
|
||||
else:
|
||||
result = check_key(key, base_urls, proxy, args.timeout, args.debug)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,572 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import requests
|
||||
import json
|
||||
import os
|
||||
import argparse
|
||||
import re
|
||||
from itertools import cycle
|
||||
import time
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files as ensure_private_output_files,
|
||||
iter_findings,
|
||||
iter_bounded_text_lines,
|
||||
keycheck_input_mode,
|
||||
load_known_statuses,
|
||||
private_atomic_writer,
|
||||
record_cached_keycheck_occurrence,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding='utf-8', errors='replace')
|
||||
sys.stderr.reconfigure(encoding='utf-8', errors='replace')
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# --- Конфигурация ---
|
||||
SERVICE = "openai"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
ALIVE_FILE = os.path.join(OUTPUT_DIR, "openaiAlive.txt")
|
||||
DEAD_FILE = os.path.join(OUTPUT_DIR, "openaiDead.txt")
|
||||
NETWORK_FILE = os.path.join(OUTPUT_DIR, "openaiNetwork.txt")
|
||||
LIMITED_FILE = os.path.join(OUTPUT_DIR, "openaiLimited.txt")
|
||||
RESTRICTED_FILE = os.path.join(OUTPUT_DIR, "openaiRestricted.txt")
|
||||
UNKNOWN_FILE = os.path.join(OUTPUT_DIR, "openaiUnknown.txt")
|
||||
NO_TARGET_FILE = os.path.join(OUTPUT_DIR, "openaiNoTarget.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "openaiChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "openaiResults.jsonl")
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
MODEL_PRIORITY_FOR_TEST = (
|
||||
'gpt-5.6-sol',
|
||||
'gpt-5.6',
|
||||
'gpt-5.6-luna',
|
||||
'gpt-5.6-terra',
|
||||
'gpt-5',
|
||||
'o3-pro',
|
||||
'o3',
|
||||
)
|
||||
TARGET_MODELS = {'gpt-5', 'o3-pro', 'o3'}
|
||||
NON_CHAT_MODEL_MARKERS = (
|
||||
'embedding', 'image', 'audio', 'tts', 'transcribe', 'realtime', 'search', 'moderation',
|
||||
)
|
||||
PROBE_MAX_COMPLETION_TOKENS = 16
|
||||
STATUS_FILES = [ALIVE_FILE, DEAD_FILE, NETWORK_FILE, LIMITED_FILE, RESTRICTED_FILE, UNKNOWN_FILE, NO_TARGET_FILE]
|
||||
STATUS_BY_FILE = {
|
||||
ALIVE_FILE: 'ALIVE',
|
||||
DEAD_FILE: 'DEAD',
|
||||
NETWORK_FILE: 'NETWORK',
|
||||
LIMITED_FILE: 'LIMITED',
|
||||
RESTRICTED_FILE: 'RESTRICTED',
|
||||
UNKNOWN_FILE: 'UNKNOWN',
|
||||
NO_TARGET_FILE: 'NO_TARGET_MODELS',
|
||||
}
|
||||
OPENAI_KEY_REGEX = re.compile(r'sk-[A-Za-z0-9_-]{20,}')
|
||||
|
||||
# --- Вспомогательные функции ---
|
||||
def key_from_line(line):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
return None
|
||||
if '\t' in line:
|
||||
return line.split('\t', 1)[0].strip()
|
||||
if ':[' in line:
|
||||
return line.split(':[', 1)[0].strip()
|
||||
match = OPENAI_KEY_REGEX.search(line)
|
||||
if match:
|
||||
return match.group(0)
|
||||
return line.split()[0].strip()
|
||||
|
||||
def load_set_from_file(filepath):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return set()
|
||||
if not os.path.exists(filepath): return set()
|
||||
return {key for key in (key_from_line(line) for line in iter_bounded_text_lines(filepath)) if key}
|
||||
|
||||
|
||||
def iter_plain_openai_keys(paths):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
seen = set()
|
||||
items = []
|
||||
for path in paths or []:
|
||||
if not path or not os.path.exists(path):
|
||||
continue
|
||||
for line in iter_bounded_text_lines(path):
|
||||
key = key_from_line(line)
|
||||
if not key or key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
items.append({'key': key, 'source': path, 'finding': {}})
|
||||
for item in items:
|
||||
yield item
|
||||
|
||||
|
||||
def retry_plain_files(args):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return []
|
||||
files = list(args.plain or [])
|
||||
if args.recheck_all:
|
||||
files.extend(STATUS_FILES)
|
||||
else:
|
||||
if args.retry_limited:
|
||||
files.append(LIMITED_FILE)
|
||||
if args.retry_network:
|
||||
files.append(NETWORK_FILE)
|
||||
if args.retry_unknown:
|
||||
files.append(UNKNOWN_FILE)
|
||||
if args.retry_restricted:
|
||||
files.append(RESTRICTED_FILE)
|
||||
if args.retry_no_balance:
|
||||
files.append(LIMITED_FILE)
|
||||
out = []
|
||||
seen = set()
|
||||
for path in files:
|
||||
if path and path not in seen:
|
||||
seen.add(path)
|
||||
out.append(path)
|
||||
return out
|
||||
|
||||
def load_checked_statuses():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return {}
|
||||
statuses = {}
|
||||
if not os.path.exists(CHECKED_FILE):
|
||||
return statuses
|
||||
for line in iter_bounded_text_lines(CHECKED_FILE):
|
||||
parts = line.rstrip('\n').split('\t')
|
||||
if parts and parts[0]:
|
||||
statuses[parts[0]] = parts[1] if len(parts) > 1 else 'UNKNOWN'
|
||||
return statuses
|
||||
|
||||
def load_known_keys():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return set()
|
||||
known = set(load_checked_statuses().keys())
|
||||
for path in STATUS_FILES:
|
||||
known.update(load_set_from_file(path))
|
||||
return known
|
||||
|
||||
def ensure_output_files():
|
||||
ensure_private_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_BY_FILE)
|
||||
|
||||
def compact_status_file(path):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
if not os.path.exists(path):
|
||||
return
|
||||
last_by_key = {}
|
||||
order = []
|
||||
for line in iter_bounded_text_lines(path):
|
||||
key = key_from_line(line)
|
||||
if not key:
|
||||
continue
|
||||
if key not in last_by_key:
|
||||
order.append(key)
|
||||
last_by_key[key] = line
|
||||
for attempt in range(6):
|
||||
try:
|
||||
with private_atomic_writer(path) as f:
|
||||
for key in order:
|
||||
f.write(last_by_key[key])
|
||||
except PermissionError:
|
||||
if attempt == 5:
|
||||
print(f"Warning: unable to compact {path}; leaving existing file as-is")
|
||||
return
|
||||
time.sleep(0.1 * (attempt + 1))
|
||||
else:
|
||||
return
|
||||
|
||||
def compact_all_status_files():
|
||||
for path in [CHECKED_FILE, *STATUS_FILES]:
|
||||
compact_status_file(path)
|
||||
|
||||
def backfill_checked_file():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
checked = load_checked_statuses()
|
||||
changed = False
|
||||
for path, status in STATUS_BY_FILE.items():
|
||||
for key in load_set_from_file(path):
|
||||
if key not in checked:
|
||||
checked[key] = status
|
||||
changed = True
|
||||
if not changed:
|
||||
return
|
||||
with private_atomic_writer(CHECKED_FILE) as f:
|
||||
for key, status in sorted(checked.items()):
|
||||
f.write(f"{key}\t{status}\tbackfilled\n")
|
||||
|
||||
def load_proxies(proxy_file=None):
|
||||
proxy_file = proxy_file or PROXY_FILE
|
||||
if not os.path.exists(proxy_file): return None
|
||||
proxies = []
|
||||
with open(proxy_file, 'r') as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line: continue
|
||||
try:
|
||||
ip, port, login, password = line.split(':')
|
||||
proxy_url = f"http://{login}:{password}@{ip}:{port}"
|
||||
proxies.append({"http": proxy_url, "https": proxy_url})
|
||||
except ValueError:
|
||||
print(f"⚠️ Неверный формат прокси: '{line}'. Пропускаем.")
|
||||
if not proxies:
|
||||
print("⚠️ Файл proxy.txt пуст. Запросы будут идти напрямую.")
|
||||
return None
|
||||
print(f"✅ Загружено {len(proxies)} прокси.")
|
||||
return cycle(proxies)
|
||||
|
||||
def move_key_to_alive(
|
||||
key, available_target_models, service_tier, source='', finding=None,
|
||||
model_inventory=None, probe_model='',
|
||||
):
|
||||
"""
|
||||
Перемещает ключ из DEAD_FILE в ALIVE_FILE, записывая модели и service_tier.
|
||||
"""
|
||||
models_str = ",".join(sorted(list(available_target_models)))
|
||||
tier_str = str(service_tier or 'unknown').replace('\r', ' ').replace('\n', ' ')[:1000]
|
||||
print(f" -> ✅ Ключ рабочий! Модели: {models_str}, Тир: {tier_str}. Перемещаем в {ALIVE_FILE}")
|
||||
|
||||
model_inventory = sorted(set(model_inventory or available_target_models))
|
||||
result = {
|
||||
'status': 'ALIVE',
|
||||
'models': sorted(list(available_target_models)),
|
||||
'model_inventory': model_inventory,
|
||||
'model_count': len(model_inventory),
|
||||
'llm_probe_model': probe_model,
|
||||
'llm_probe_status': 'GENERATION_OK',
|
||||
'service_tier': service_tier,
|
||||
'message': f'chat ping ok; model={probe_model}; models={len(model_inventory)}',
|
||||
}
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, 'OpenAI')
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE,
|
||||
STATUS_BY_FILE,
|
||||
key,
|
||||
'ALIVE',
|
||||
status_line=f"{key}:[{models_str}]:{tier_str}\n",
|
||||
checked_line=f"{key}\tALIVE\t{time.strftime('%Y-%m-%dT%H:%M:%S')}\n",
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, 'OpenAI')
|
||||
|
||||
def redact_message(message, key):
|
||||
message = str(message).replace('\r', ' ').replace('\n', ' ')[:1000]
|
||||
if key:
|
||||
message = message.replace(key, '***REDACTED***')
|
||||
return OPENAI_KEY_REGEX.sub('***REDACTED***', message)
|
||||
|
||||
def write_key_status(key, path, status, message='', source='', finding=None, metadata=None):
|
||||
status_upper = status.upper()
|
||||
projection_status = STATUS_BY_FILE.get(path)
|
||||
if not projection_status:
|
||||
raise ValueError(f'unknown OpenAI status projection: {path}')
|
||||
message = redact_message(message, key)
|
||||
result = {**(metadata or {}), 'status': status_upper, 'message': message}
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, 'OpenAI')
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE,
|
||||
STATUS_BY_FILE,
|
||||
key,
|
||||
projection_status,
|
||||
message,
|
||||
source,
|
||||
status_line=f"{key}\t{status}\t{message}\n",
|
||||
checked_line=f"{key}\t{projection_status}\t{time.strftime('%Y-%m-%dT%H:%M:%S')}\n",
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, 'OpenAI')
|
||||
|
||||
def extract_openai_key(data):
|
||||
if data.get("DetectorName") == "OpenAI":
|
||||
return data.get("Raw") or data.get("RawV2")
|
||||
if data.get("detector") == "OpenAI":
|
||||
return data.get("raw") or data.get("raw_v2")
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict) and finding.get("DetectorName") == "OpenAI":
|
||||
return finding.get("Raw") or finding.get("RawV2")
|
||||
return None
|
||||
|
||||
# --- Функции проверки ---
|
||||
def check_authentication(key, proxy):
|
||||
print(f" [1/2] Проверка аутентификации...")
|
||||
url = "https://api.openai.com/v1/models"
|
||||
headers = {"Authorization": f"Bearer {key}"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=15)
|
||||
if response.status_code == 200:
|
||||
print(" -> Аутентификация пройдена.")
|
||||
return 'valid', response.json().get('data', [])
|
||||
elif response.status_code == 401:
|
||||
print(" -> Ошибка 401: Ключ недействителен или отозван.")
|
||||
return 'dead', response.text
|
||||
elif response.status_code == 403:
|
||||
print(f" -> Ошибка 403: ключ ограничен/заблокирован: {response.text[:300]}")
|
||||
return 'restricted', response.text
|
||||
elif response.status_code == 429:
|
||||
print(f" -> Ошибка 429: rate limit / quota: {response.text[:300]}")
|
||||
return 'limited', response.text
|
||||
else:
|
||||
print(f" -> Ошибка {response.status_code}: {response.text}")
|
||||
return 'unknown', response.text
|
||||
except requests.RequestException as e:
|
||||
print(f" -> Ошибка сети: {e}")
|
||||
return 'network', str(e)
|
||||
|
||||
def reportable_target_models(model_ids):
|
||||
return {
|
||||
model for model in model_ids
|
||||
if model in TARGET_MODELS or model.startswith('gpt-5.6-') or model == 'gpt-5.6'
|
||||
}
|
||||
|
||||
|
||||
def choose_probe_model(model_ids):
|
||||
models = [str(model or '') for model in model_ids if model]
|
||||
by_lower = {model.lower(): model for model in models}
|
||||
for model in MODEL_PRIORITY_FOR_TEST:
|
||||
if model.lower() in by_lower:
|
||||
return by_lower[model.lower()]
|
||||
for model in models:
|
||||
lowered = model.lower()
|
||||
if lowered.startswith(('gpt-', 'o')) and not any(
|
||||
marker in lowered for marker in NON_CHAT_MODEL_MARKERS
|
||||
):
|
||||
return model
|
||||
return ''
|
||||
|
||||
|
||||
def check_balance_and_tier(key, model_to_test, proxy):
|
||||
"""
|
||||
Проверяет баланс и возвращает service_tier в случае успеха.
|
||||
"""
|
||||
print(f" [2/2] Проверка баланса и тира...")
|
||||
if not model_to_test:
|
||||
print(" -> Не найдено подходящих моделей для теста баланса.")
|
||||
return 'unknown', 'no chat-capable model from /models'
|
||||
|
||||
print(f" -> Используем модель для теста: {model_to_test}")
|
||||
url = "https://api.openai.com/v1/chat/completions"
|
||||
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
payload = {
|
||||
"model": model_to_test,
|
||||
"messages": [{"role": "user", "content": "Reply with one digit."}],
|
||||
"max_completion_tokens": PROBE_MAX_COMPLETION_TOKENS,
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=20)
|
||||
if response.status_code == 200:
|
||||
# Успех, извлекаем service_tier
|
||||
response_data = response.json()
|
||||
service_tier = response_data.get('service_tier')
|
||||
return 'ok', service_tier
|
||||
elif response.status_code == 429:
|
||||
print(f" -> Ошибка 429: Нет баланса или превышен лимит.")
|
||||
return 'limited', response.text
|
||||
elif response.status_code == 401:
|
||||
print(f" -> Ошибка 401: ключ недействителен или отозван.")
|
||||
return 'dead', response.text
|
||||
elif response.status_code == 403:
|
||||
print(f" -> Ошибка 403: ключ ограничен/заблокирован: {response.text[:300]}")
|
||||
return 'restricted', response.text
|
||||
else:
|
||||
print(f" -> Ошибка {response.status_code}: {response.text}")
|
||||
return 'unknown', response.text
|
||||
except requests.RequestException as e:
|
||||
print(f" -> Ошибка сети: {e}")
|
||||
return 'network', str(e)
|
||||
|
||||
# --- Основной процесс ---
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description='OpenAI key checker')
|
||||
parser.add_argument('--input', default=INPUT_FILE)
|
||||
parser.add_argument('--plain', action='append', default=[])
|
||||
parser.add_argument('--proxy-file', default=PROXY_FILE)
|
||||
parser.add_argument('--max-keys', type=int, default=0)
|
||||
parser.add_argument('--retry-network', action='store_true')
|
||||
parser.add_argument('--retry-limited', action='store_true')
|
||||
parser.add_argument('--retry-unknown', action='store_true')
|
||||
parser.add_argument('--retry-restricted', action='store_true')
|
||||
parser.add_argument('--retry-no-balance', action='store_true')
|
||||
parser.add_argument('--recheck-all', action='store_true')
|
||||
return parser.parse_args()
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_output_files()
|
||||
compact_all_status_files()
|
||||
backfill_checked_file()
|
||||
print("--- 🚀 Запуск чекера ключей OpenAI 🚀 ---")
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked_statuses = load_checked_statuses()
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_BY_FILE)
|
||||
known_keys = set(known_statuses)
|
||||
alive_keys = load_set_from_file(ALIVE_FILE)
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add('NETWORK')
|
||||
if args.retry_limited:
|
||||
retry_statuses.update(('LIMITED', 'LIMITED_OR_NO_BALANCE', 'LIMITED_OR_QUOTA'))
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add('UNKNOWN')
|
||||
if args.retry_restricted:
|
||||
retry_statuses.add('RESTRICTED')
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.update(('NO_BALANCE', 'NO_QUOTA', 'LIMITED_OR_NO_BALANCE', 'LIMITED_OR_QUOTA'))
|
||||
print(f"📖 Загружено: {len(alive_keys)} живых ключей, {len(known_keys)} уже классифицированных ключей, {len(checked_statuses)} checked.")
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and not os.path.exists(args.input):
|
||||
print(f"❌ Файл с секретами {args.input} не найден. Завершение.")
|
||||
return
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
seen_this_run = set()
|
||||
def candidates():
|
||||
for item in iter_findings(args.input, ['OpenAI']):
|
||||
data = item.get('finding') or {}
|
||||
key = item.get('raw') or extract_openai_key(data)
|
||||
if key:
|
||||
yield {'key': key, 'source': item.get('source') or args.input, 'finding': data}
|
||||
seen_plain = set()
|
||||
for item in iter_plain_openai_keys(retry_plain_files(args)):
|
||||
key = item.get('key')
|
||||
if key and key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield item
|
||||
|
||||
for item in candidates():
|
||||
data = item.get('finding') or {}
|
||||
source_line = item.get('source') or args.input
|
||||
key = item.get('key')
|
||||
if not key:
|
||||
continue
|
||||
if keycheck_input_mode() != 'postgres' and key in seen_this_run:
|
||||
cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN'
|
||||
record_cached_keycheck_occurrence(SERVICE, key, cached_status, source_line, data, 'OpenAI')
|
||||
skipped += 1
|
||||
continue
|
||||
seen_this_run.add(key)
|
||||
if should_skip_key(
|
||||
key, checked_statuses, known_keys, args, retry_statuses,
|
||||
service=SERVICE, source=source_line, finding=data, detector='OpenAI',
|
||||
known_statuses=known_statuses,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
|
||||
print(f"\n[{processed}] 🎯 Новый кандидат: {key[:8]}...{key[-4:]} from {source_line}")
|
||||
|
||||
current_proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
|
||||
auth_status, auth_data = check_authentication(key, current_proxy)
|
||||
|
||||
if auth_status == 'valid' and auth_data:
|
||||
all_available_models_data = auth_data
|
||||
all_model_ids = sorted({
|
||||
str(model.get('id') or '') for model in all_available_models_data
|
||||
if isinstance(model, dict) and model.get('id')
|
||||
})
|
||||
found_target_models = reportable_target_models(all_model_ids)
|
||||
model_to_test = choose_probe_model(all_model_ids)
|
||||
|
||||
if not model_to_test:
|
||||
print(" -> Ключ валиден, но не имеет подходящей chat-модели. Пропускаем.")
|
||||
write_key_status(
|
||||
key, NO_TARGET_FILE, 'no_target_models', ','.join(all_model_ids)[:500],
|
||||
source_line, data, {
|
||||
'models': sorted(found_target_models),
|
||||
'model_inventory': all_model_ids,
|
||||
'model_count': len(all_model_ids),
|
||||
'llm_probe_status': 'NO_CONTEXT',
|
||||
},
|
||||
)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'NO_TARGET_MODELS'
|
||||
continue
|
||||
|
||||
balance_status, balance_data = check_balance_and_tier(key, model_to_test, current_proxy)
|
||||
probe_metadata = {
|
||||
'models': sorted(found_target_models or {model_to_test}),
|
||||
'model_inventory': all_model_ids,
|
||||
'model_count': len(all_model_ids),
|
||||
'llm_probe_model': model_to_test,
|
||||
'llm_probe_status': {
|
||||
'ok': 'GENERATION_OK',
|
||||
'limited': 'LIMITED',
|
||||
'network': 'NETWORK',
|
||||
'restricted': 'RESTRICTED',
|
||||
'dead': 'DEAD',
|
||||
}.get(balance_status, 'UNKNOWN'),
|
||||
}
|
||||
|
||||
# Проверяем, что результат не None (успешная проверка баланса)
|
||||
if balance_status == 'ok':
|
||||
move_key_to_alive(
|
||||
key, found_target_models or {model_to_test}, balance_data, source_line, data,
|
||||
model_inventory=all_model_ids, probe_model=model_to_test,
|
||||
)
|
||||
alive_keys.add(key)
|
||||
final_status = 'ALIVE'
|
||||
elif balance_status == 'limited':
|
||||
write_key_status(key, LIMITED_FILE, 'limited_or_no_balance', balance_data, source_line, data, probe_metadata)
|
||||
final_status = 'LIMITED_OR_NO_BALANCE'
|
||||
elif balance_status == 'network':
|
||||
write_key_status(key, NETWORK_FILE, 'network_error', balance_data, source_line, data, probe_metadata)
|
||||
final_status = 'NETWORK'
|
||||
elif balance_status == 'restricted':
|
||||
write_key_status(key, RESTRICTED_FILE, 'restricted', balance_data, source_line, data, probe_metadata)
|
||||
final_status = 'RESTRICTED'
|
||||
elif balance_status == 'dead':
|
||||
write_key_status(key, DEAD_FILE, 'invalid_or_revoked', balance_data, source_line, data, probe_metadata)
|
||||
final_status = 'DEAD'
|
||||
else:
|
||||
write_key_status(key, UNKNOWN_FILE, 'unknown', balance_data, source_line, data, probe_metadata)
|
||||
final_status = 'UNKNOWN'
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = final_status
|
||||
elif auth_status == 'network':
|
||||
write_key_status(key, NETWORK_FILE, 'network_error', auth_data, source_line, data)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'NETWORK'
|
||||
elif auth_status == 'limited':
|
||||
write_key_status(key, LIMITED_FILE, 'limited_or_quota', auth_data, source_line, data)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'LIMITED'
|
||||
elif auth_status == 'restricted':
|
||||
write_key_status(key, RESTRICTED_FILE, 'restricted', auth_data, source_line, data)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'RESTRICTED'
|
||||
elif auth_status == 'dead':
|
||||
write_key_status(key, DEAD_FILE, 'invalid_or_revoked', auth_data, source_line, data)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'DEAD'
|
||||
else:
|
||||
write_key_status(key, UNKNOWN_FILE, 'unknown', auth_data, source_line, data)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = 'UNKNOWN'
|
||||
|
||||
print("\n--- ✅ Проверка завершена. ---")
|
||||
print(f"Processed={processed}, skipped={skipped}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,498 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import requests
|
||||
import json
|
||||
import os
|
||||
import argparse
|
||||
from itertools import cycle
|
||||
from datetime import datetime, timezone
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding='utf-8', errors='replace')
|
||||
sys.stderr.reconfigure(encoding='utf-8', errors='replace')
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
from keycheck_common import (
|
||||
acquire_file_lock,
|
||||
append_checked,
|
||||
append_jsonl,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
env_int,
|
||||
ensure_output_files as ensure_private_output_files,
|
||||
finding_detector_secret_hash,
|
||||
iter_findings,
|
||||
iter_bounded_text_lines,
|
||||
keycheck_input_mode,
|
||||
load_known_statuses,
|
||||
load_checked_statuses,
|
||||
mask_secret,
|
||||
now_iso,
|
||||
physical_jsonl_segments,
|
||||
private_append_writer,
|
||||
private_atomic_writer,
|
||||
reconcile_keycheck_jsonl_segments,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
release_file_lock,
|
||||
repair_keycheck_jsonl_tail,
|
||||
require_provider_authority,
|
||||
rotate_jsonl_if_needed,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
sha256_text,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from runtime_security import reject_reparse_components, require_private_file
|
||||
|
||||
# --- Конфигурация ---
|
||||
SERVICE = "openrouter"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
ALIVE_FILE = os.path.join(OUTPUT_DIR, "openrouterAlive.txt")
|
||||
DEAD_FILE = os.path.join(OUTPUT_DIR, "openrouterDead.txt")
|
||||
LIMITED_FILE = os.path.join(OUTPUT_DIR, "openrouterLimited.txt")
|
||||
NO_BALANCE_FILE = os.path.join(OUTPUT_DIR, "openrouterNoBalance.txt")
|
||||
NETWORK_FILE = os.path.join(OUTPUT_DIR, "openrouterNetwork.txt")
|
||||
UNKNOWN_FILE = os.path.join(OUTPUT_DIR, "openrouterUnknown.txt")
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "openrouterChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "openrouterResults.jsonl")
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
STATUS_FILES = {
|
||||
"VALID": ALIVE_FILE,
|
||||
"NO_BALANCE": NO_BALANCE_FILE,
|
||||
"DEAD": DEAD_FILE,
|
||||
"LIMITED": LIMITED_FILE,
|
||||
"NETWORK": NETWORK_FILE,
|
||||
"UNKNOWN": UNKNOWN_FILE,
|
||||
}
|
||||
|
||||
CREDITS_URL = "https://openrouter.ai/api/v1/credits"
|
||||
|
||||
# --- Вспомогательные функции ---
|
||||
def load_set_from_file(filepath):
|
||||
"""Загружает ключи из файла в set для быстрой проверки."""
|
||||
if not os.path.exists(filepath):
|
||||
return set()
|
||||
return {line.strip().split(':')[0] for line in iter_bounded_text_lines(filepath) if line.strip()}
|
||||
|
||||
def ensure_output_files():
|
||||
ensure_private_output_files((CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()))
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def legacy_status_key(line):
|
||||
value = line.strip()
|
||||
if not value:
|
||||
return None
|
||||
if '\t' in value:
|
||||
return value.split('\t', 1)[0].strip()
|
||||
return value.split(':', 1)[0].strip()
|
||||
|
||||
|
||||
def load_openrouter_keys(path):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return set()
|
||||
if not os.path.exists(path):
|
||||
return set()
|
||||
return {key for key in (legacy_status_key(line) for line in iter_bounded_text_lines(path)) if key}
|
||||
|
||||
|
||||
def iter_plain_openrouter_keys(paths):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
seen = set()
|
||||
items = []
|
||||
for path in paths or []:
|
||||
if not path or not os.path.exists(path):
|
||||
continue
|
||||
for line in iter_bounded_text_lines(path):
|
||||
key = legacy_status_key(line)
|
||||
if not key or key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
items.append({'raw': key, 'source': path, 'finding': {}})
|
||||
for item in items:
|
||||
yield item
|
||||
|
||||
|
||||
def retry_plain_files(args):
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return []
|
||||
files = list(args.plain or [])
|
||||
if args.recheck_all:
|
||||
files.extend(STATUS_FILES.values())
|
||||
else:
|
||||
if args.retry_valid:
|
||||
files.append(ALIVE_FILE)
|
||||
if args.retry_no_balance:
|
||||
files.append(NO_BALANCE_FILE)
|
||||
if args.retry_limited:
|
||||
files.append(LIMITED_FILE)
|
||||
if args.retry_network:
|
||||
files.append(NETWORK_FILE)
|
||||
if args.retry_unknown:
|
||||
files.append(UNKNOWN_FILE)
|
||||
out = []
|
||||
seen = set()
|
||||
for path in files:
|
||||
if path and path not in seen:
|
||||
seen.add(path)
|
||||
out.append(path)
|
||||
return out
|
||||
|
||||
|
||||
def migrate_legacy_checked():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
for key in sorted(load_openrouter_keys(ALIVE_FILE)):
|
||||
if key not in checked:
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, {'status': 'VALID', 'message': 'legacy alive status migration'}, 'legacy:openrouterAlive.txt', {}, 'OpenRouter', 'legacy_status')
|
||||
append_checked(CHECKED_FILE, key, 'VALID')
|
||||
checked[key] = 'VALID'
|
||||
for key in sorted(load_openrouter_keys(DEAD_FILE)):
|
||||
if key not in checked:
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, {'status': 'DEAD', 'message': 'legacy dead status migration'}, 'legacy:openrouterDead.txt', {}, 'OpenRouter', 'legacy_status')
|
||||
append_checked(CHECKED_FILE, key, 'DEAD')
|
||||
checked[key] = 'DEAD'
|
||||
|
||||
|
||||
def _legacy_alive_checked_at(path):
|
||||
details = os.stat(path, follow_symlinks=False)
|
||||
return datetime.fromtimestamp(details.st_mtime, timezone.utc).isoformat(timespec='seconds')
|
||||
|
||||
|
||||
def _legacy_balance_event_payload(key, balance, checked_at):
|
||||
balance_text = f'{balance:.6f}'
|
||||
key_hash = sha256_text(key)
|
||||
event_id = sha256_text('|'.join([
|
||||
SERVICE,
|
||||
'legacy_status',
|
||||
'openrouterAlive.txt',
|
||||
key_hash,
|
||||
'NO_BALANCE',
|
||||
balance_text,
|
||||
]))
|
||||
return {
|
||||
'key_masked': mask_secret(key),
|
||||
'key_hash': key_hash,
|
||||
'secret_hash': key_hash,
|
||||
'detector_secret_hash': finding_detector_secret_hash({}),
|
||||
'finding_uid': '',
|
||||
'detector': 'OpenRouter',
|
||||
'source': 'legacy:openrouterAlive.txt',
|
||||
'finding': {},
|
||||
'checked_at': checked_at,
|
||||
'result_source': 'legacy_status',
|
||||
'status': 'NO_BALANCE',
|
||||
'message': f'credits={balance_text}',
|
||||
'event_id': event_id,
|
||||
}
|
||||
|
||||
|
||||
def _publish_legacy_no_balance(moved):
|
||||
lock_path = f'{NO_BALANCE_FILE}.lock'
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
require_private_file(NO_BALANCE_FILE)
|
||||
existing = load_openrouter_keys(NO_BALANCE_FILE)
|
||||
with private_append_writer(NO_BALANCE_FILE) as handle:
|
||||
for key, balance in moved:
|
||||
if key in existing:
|
||||
continue
|
||||
balance_text = f'{balance:.6f}'
|
||||
handle.write(f'{key}\tNO_BALANCE\tcredits={balance_text}\tmigrated_from_alive\n')
|
||||
existing.add(key)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def _existing_legacy_event_ids(expected):
|
||||
found = set()
|
||||
max_line_bytes = max(1024, env_int('KEYCHECK_INPUT_MAX_LINE_BYTES', 16 * 1024 * 1024))
|
||||
max_file_bytes = max(
|
||||
max_line_bytes,
|
||||
max(1, env_int('KEYCHECK_INPUT_LIST_MAX_BYTES', 32 * 1024 * 1024)),
|
||||
max(0, env_int('KEYCHECK_RESULTS_MAX_MB', 32)) * 1024 * 1024 + max_line_bytes,
|
||||
)
|
||||
paths = [path for _, path in physical_jsonl_segments(RESULTS_FILE)]
|
||||
if os.path.isfile(RESULTS_FILE):
|
||||
paths.append(os.path.abspath(RESULTS_FILE))
|
||||
for path in paths:
|
||||
reject_reparse_components(path)
|
||||
if os.path.getsize(path) > max_file_bytes:
|
||||
raise RuntimeError(f'OpenRouter result file exceeds the bounded migration scan size: {path}')
|
||||
with open(path, 'rb') as handle:
|
||||
while True:
|
||||
raw_line = handle.readline(max_line_bytes + 1)
|
||||
if not raw_line:
|
||||
break
|
||||
if len(raw_line) > max_line_bytes:
|
||||
raise RuntimeError(f'OpenRouter result line exceeds the bounded migration scan size: {path}')
|
||||
if not raw_line.endswith(b'\n'):
|
||||
raise RuntimeError(f'torn OpenRouter result line during legacy migration: {path}')
|
||||
try:
|
||||
payload = json.loads(raw_line.decode('utf-8', errors='replace'))
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
event_id = str(payload.get('event_id') or '') if isinstance(payload, dict) else ''
|
||||
if event_id not in expected:
|
||||
continue
|
||||
wanted = expected[event_id]
|
||||
for field in ('key_hash', 'status', 'source', 'result_source'):
|
||||
if str(payload.get(field) or '') != str(wanted.get(field) or ''):
|
||||
raise RuntimeError(f'conflicting OpenRouter legacy migration event: {event_id}')
|
||||
found.add(event_id)
|
||||
return found
|
||||
|
||||
|
||||
def _publish_legacy_balance_events(moved, checked_at):
|
||||
payloads = [_legacy_balance_event_payload(key, balance, checked_at) for key, balance in moved]
|
||||
expected = {payload['event_id']: payload for payload in payloads}
|
||||
lock_path = f'{RESULTS_FILE}.lock'
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
require_private_file(RESULTS_FILE)
|
||||
repair_keycheck_jsonl_tail(RESULTS_FILE)
|
||||
reconcile_keycheck_jsonl_segments(RESULTS_FILE)
|
||||
existing = _existing_legacy_event_ids(expected)
|
||||
max_bytes = max(0, env_int('KEYCHECK_RESULTS_MAX_MB', 32)) * 1024 * 1024
|
||||
for payload in payloads:
|
||||
event_id = payload['event_id']
|
||||
if event_id in existing:
|
||||
continue
|
||||
rotate_jsonl_if_needed(RESULTS_FILE, max_bytes)
|
||||
with private_append_writer(RESULTS_FILE) as handle:
|
||||
handle.write(json.dumps(payload, ensure_ascii=False, default=str) + '\n')
|
||||
existing.add(event_id)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def _publish_legacy_checked(moved, checked_at):
|
||||
lock_path = f'{CHECKED_FILE}.lock'
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
require_private_file(CHECKED_FILE)
|
||||
existing = set()
|
||||
for line in iter_bounded_text_lines(CHECKED_FILE):
|
||||
parts = line.rstrip('\r\n').split('\t')
|
||||
if len(parts) >= 2:
|
||||
existing.add((parts[0], parts[1]))
|
||||
with private_append_writer(CHECKED_FILE) as handle:
|
||||
for key, _ in moved:
|
||||
identity = (key, 'NO_BALANCE')
|
||||
if identity in existing:
|
||||
continue
|
||||
handle.write(f'{key}\tNO_BALANCE\t{checked_at}\n')
|
||||
existing.add(identity)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def _rewrite_legacy_alive(keep):
|
||||
with private_atomic_writer(ALIVE_FILE, binary=True, suffix='.legacy.tmp') as handle:
|
||||
for line in keep:
|
||||
handle.write(line.encode('utf-8'))
|
||||
|
||||
|
||||
def migrate_legacy_alive_balances():
|
||||
if keycheck_input_mode() == 'postgres':
|
||||
return
|
||||
lock_path = f'{ALIVE_FILE}.lock'
|
||||
lock = acquire_file_lock(lock_path, timeout_sec=30)
|
||||
try:
|
||||
if not os.path.exists(ALIVE_FILE):
|
||||
return
|
||||
require_private_file(ALIVE_FILE)
|
||||
checked_at = _legacy_alive_checked_at(ALIVE_FILE)
|
||||
keep = []
|
||||
moved = []
|
||||
seen = set()
|
||||
for line in iter_bounded_text_lines(ALIVE_FILE):
|
||||
text = line.strip()
|
||||
if not text:
|
||||
keep.append(line)
|
||||
continue
|
||||
key = legacy_status_key(text)
|
||||
balance = None
|
||||
if ':' in text and '\t' not in text:
|
||||
try:
|
||||
balance = float(text.rsplit(':', 1)[1])
|
||||
except ValueError:
|
||||
balance = None
|
||||
if key and balance is not None and balance <= 0:
|
||||
if key not in seen:
|
||||
moved.append((key, balance))
|
||||
seen.add(key)
|
||||
else:
|
||||
keep.append(line)
|
||||
if not moved:
|
||||
return
|
||||
_publish_legacy_no_balance(moved)
|
||||
_publish_legacy_balance_events(moved, checked_at)
|
||||
_publish_legacy_checked(moved, checked_at)
|
||||
_rewrite_legacy_alive(keep)
|
||||
finally:
|
||||
release_file_lock(lock, lock_path)
|
||||
|
||||
|
||||
def load_proxies(proxy_file=None):
|
||||
"""Загружает и подготавливает прокси."""
|
||||
proxy_file = proxy_file or PROXY_FILE
|
||||
if not os.path.exists(proxy_file):
|
||||
print("ℹ️ Файл proxy.txt не найден, запросы будут идти напрямую.")
|
||||
return None
|
||||
proxies = []
|
||||
with open(proxy_file, 'r') as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line: continue
|
||||
try:
|
||||
ip, port, login, password = line.split(':')
|
||||
proxy_url = f"http://{login}:{password}@{ip}:{port}"
|
||||
proxies.append({"http": proxy_url, "https": proxy_url})
|
||||
except ValueError:
|
||||
print(f"⚠️ Неверный формат прокси: '{line}'. Пропускаем.")
|
||||
if not proxies:
|
||||
print("⚠️ Файл proxy.txt пуст. Запросы будут идти напрямую.")
|
||||
return None
|
||||
print(f"✅ Загружено {len(proxies)} прокси.")
|
||||
return cycle(proxies)
|
||||
|
||||
def write_result(key, result, source_line, finding=None, previous_status=None):
|
||||
status = result.get('status') or 'UNKNOWN'
|
||||
credits = result.get('remaining_credits')
|
||||
extra = f"credits={credits:.6f}" if isinstance(credits, (int, float)) else source_line
|
||||
checked_at = now_iso()
|
||||
result = {**result, 'checked_at': result.get('checked_at') or checked_at}
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source_line, finding, 'OpenRouter')
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status, result.get('message', ''), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source_line, finding, 'OpenRouter')
|
||||
|
||||
# --- Функция проверки ---
|
||||
def check_openrouter_key(key, proxy):
|
||||
"""Проверяет один ключ OpenRouter и возвращает normalized result."""
|
||||
headers = {"Authorization": f"Bearer {key}"}
|
||||
try:
|
||||
resp = requests.get(CREDITS_URL, headers=headers, proxies=proxy, timeout=15)
|
||||
except requests.exceptions.RequestException as e:
|
||||
return {'status': 'NETWORK', 'message': str(e)}
|
||||
|
||||
if resp.status_code == 200:
|
||||
try:
|
||||
data = resp.json().get("data", {})
|
||||
total = float(data.get("total_credits", 0.0) or 0.0)
|
||||
used = float(data.get("total_usage", 0.0) or 0.0)
|
||||
except (TypeError, ValueError, json.JSONDecodeError) as exc:
|
||||
return {'status': 'UNKNOWN', 'http_status': resp.status_code, 'message': f'invalid credits response: {exc}'}
|
||||
remaining = total - used
|
||||
if remaining > 0:
|
||||
return {'status': 'VALID', 'remaining_credits': remaining, 'message': f'credits={remaining:.6f}'}
|
||||
return {'status': 'NO_BALANCE', 'remaining_credits': remaining, 'message': f'credits={remaining:.6f}'}
|
||||
if resp.status_code in (401, 403):
|
||||
return {'status': 'DEAD', 'http_status': resp.status_code, 'message': resp.text[:1000]}
|
||||
if resp.status_code == 429:
|
||||
return {'status': 'LIMITED', 'http_status': resp.status_code, 'message': resp.text[:1000]}
|
||||
if 500 <= resp.status_code <= 599:
|
||||
return {'status': 'NETWORK', 'http_status': resp.status_code, 'message': resp.text[:1000]}
|
||||
return {'status': 'UNKNOWN', 'http_status': resp.status_code, 'message': resp.text[:1000]}
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description='OpenRouter key checker')
|
||||
parser.add_argument('--input', default=INPUT_FILE)
|
||||
parser.add_argument('--plain', action='append', default=[])
|
||||
parser.add_argument('--proxy-file', default=PROXY_FILE)
|
||||
parser.add_argument('--max-keys', type=int, default=0)
|
||||
parser.add_argument('--retry-network', action='store_true')
|
||||
parser.add_argument('--retry-limited', action='store_true')
|
||||
parser.add_argument('--retry-unknown', action='store_true')
|
||||
parser.add_argument('--retry-no-balance', action='store_true')
|
||||
parser.add_argument('--retry-valid', action='store_true')
|
||||
parser.add_argument('--recheck-all', action='store_true')
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
# --- Основной процесс ---
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_output_files()
|
||||
migrate_legacy_alive_balances()
|
||||
migrate_legacy_checked()
|
||||
print("--- 🚀 Запуск чекера ключей OpenRouter 🚀 ---")
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
|
||||
checked_statuses = load_checked_statuses(CHECKED_FILE)
|
||||
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
|
||||
known_keys = set(known_statuses)
|
||||
for path in STATUS_FILES.values():
|
||||
known_keys.update(load_openrouter_keys(path))
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add('NETWORK')
|
||||
if args.retry_limited:
|
||||
retry_statuses.add('LIMITED')
|
||||
if args.retry_unknown:
|
||||
retry_statuses.add('UNKNOWN')
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.add('NO_BALANCE')
|
||||
if args.retry_valid:
|
||||
retry_statuses.add('VALID')
|
||||
print(f"📖 Загружено: {len(load_openrouter_keys(ALIVE_FILE))} живых ключей, {len(known_keys)} классифицированных ключей.")
|
||||
|
||||
if keycheck_input_mode() == 'jsonl' and not os.path.exists(args.input):
|
||||
print(f"❌ Файл с секретами {args.input} не найден. Завершение.")
|
||||
return
|
||||
|
||||
processed = 0
|
||||
def candidates():
|
||||
for item in iter_findings(args.input, ["OpenRouter"]):
|
||||
key = item.get("raw") or ""
|
||||
if key:
|
||||
yield item
|
||||
seen = set()
|
||||
for item in iter_plain_openrouter_keys(retry_plain_files(args)):
|
||||
key = item.get("raw") or ""
|
||||
if key and key not in seen:
|
||||
seen.add(key)
|
||||
yield item
|
||||
|
||||
for item in candidates():
|
||||
key = item.get("raw") or ""
|
||||
if not key:
|
||||
continue
|
||||
|
||||
source = item.get("source") or args.input
|
||||
finding = item.get("finding") or {}
|
||||
if should_skip_key(
|
||||
key, checked_statuses, known_keys, args, retry_statuses,
|
||||
service=SERVICE, source=source, finding=finding, detector='OpenRouter', known_statuses=known_statuses,
|
||||
):
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
|
||||
print(f"\n[{processed}] 🎯 Новый кандидат: {key[:8]}...{key[-4:]} from {source}")
|
||||
current_proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = check_openrouter_key(key, current_proxy)
|
||||
print(f" STATUS: {result.get('status')} | {str(result.get('message', ''))[:200]}")
|
||||
previous_status = known_statuses.get(key) or checked_statuses.get(key)
|
||||
write_result(key, result, source, finding, previous_status)
|
||||
known_keys.add(key)
|
||||
checked_statuses[key] = result.get('status')
|
||||
|
||||
print("\n--- ✅ Проверка завершена. ---")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,189 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import os
|
||||
|
||||
|
||||
SUPPORTED_PROVIDERS = ("deepseek", "zai", "qwen", "kimi")
|
||||
DEFAULT_PROVIDER_ORDER = SUPPORTED_PROVIDERS
|
||||
AMBIGUOUS_QWEN_DEEPSEEK_HINT = "ambiguous_qwen_deepseek"
|
||||
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
|
||||
|
||||
|
||||
def split_csv(value):
|
||||
if not value:
|
||||
return []
|
||||
if isinstance(value, str):
|
||||
values = value.split(",")
|
||||
else:
|
||||
values = value
|
||||
return [str(item).strip().lower() for item in values if str(item).strip()]
|
||||
|
||||
|
||||
def unique_supported(values):
|
||||
output = []
|
||||
seen = set()
|
||||
for value in values:
|
||||
provider = str(value or "").strip().lower()
|
||||
if provider in SUPPORTED_PROVIDERS and provider not in seen:
|
||||
seen.add(provider)
|
||||
output.append(provider)
|
||||
return output
|
||||
|
||||
|
||||
def providers_for_hint(hint):
|
||||
hint = str(hint or "").strip().lower()
|
||||
if hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT:
|
||||
return ["qwen", "deepseek"]
|
||||
if hint == AMBIGUOUS_GENERIC_SK_HINT:
|
||||
return list(SUPPORTED_PROVIDERS)
|
||||
return [hint] if hint in SUPPORTED_PROVIDERS else []
|
||||
|
||||
|
||||
def detector_provider(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return ""
|
||||
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
|
||||
names = {
|
||||
str(finding.get("DetectorName") or finding.get("DetectorType") or "").strip().lower(),
|
||||
str(extra.get("name") or "").strip().lower(),
|
||||
}
|
||||
mappings = (
|
||||
("deepseek", {"deepseek", "deepseekapikey", "deepseek_api_key"}),
|
||||
("zai", {"zaiglm"}),
|
||||
("qwen", {"qwendashscope", "qwen_dashscope", "qwen", "dashscope"}),
|
||||
("kimi", {"kimimoonshot", "moonshotai", "moonshot", "kimi"}),
|
||||
)
|
||||
for provider, detectors in mappings:
|
||||
if names & detectors:
|
||||
return provider
|
||||
return ""
|
||||
|
||||
|
||||
def ordered_providers(finding=None, hint="", origin_service="", configured_order=None):
|
||||
context = finding.get("ScannerContext") if isinstance(finding, dict) and isinstance(
|
||||
finding.get("ScannerContext"), dict
|
||||
) else {}
|
||||
hint = str(hint or context.get("provider_hint") or "").strip().lower()
|
||||
compatible = unique_supported(context.get("provider_candidates") or providers_for_hint(hint))
|
||||
if not compatible:
|
||||
compatible = providers_for_hint(hint)
|
||||
if not compatible:
|
||||
compatible = list(SUPPORTED_PROVIDERS)
|
||||
|
||||
configured = unique_supported(
|
||||
configured_order
|
||||
if configured_order is not None
|
||||
else split_csv(os.getenv("KEYCHECK_PROVIDER_RESOLUTION_ORDER"))
|
||||
)
|
||||
base_order = configured or list(DEFAULT_PROVIDER_ORDER)
|
||||
origin = str(origin_service or detector_provider(finding)).strip().lower()
|
||||
ordered = []
|
||||
if origin in compatible:
|
||||
ordered.append(origin)
|
||||
ordered.extend(provider for provider in base_order if provider in compatible)
|
||||
ordered.extend(provider for provider in compatible if provider not in ordered)
|
||||
return unique_supported(ordered)
|
||||
|
||||
|
||||
def provider_result_outcome(result):
|
||||
result = result if isinstance(result, dict) else {}
|
||||
status = str(result.get("status") or "UNKNOWN").strip().upper()
|
||||
if result.get("authenticated") is True or status in ("VALID", "ALIVE"):
|
||||
return "match"
|
||||
if result.get("candidate_rejected") or status in (
|
||||
"DEAD", "INVALID", "EXPIRED", "LEAKED_REVOKED", "INVALID_OR_REVOKED",
|
||||
):
|
||||
return "no_match"
|
||||
return "retry"
|
||||
|
||||
|
||||
def bounded_attempt(provider, result, outcome):
|
||||
result = result if isinstance(result, dict) else {}
|
||||
error = result.get("error") if isinstance(result.get("error"), dict) else {}
|
||||
return {
|
||||
"provider": provider,
|
||||
"outcome": outcome,
|
||||
"status": str(result.get("status") or "UNKNOWN").upper(),
|
||||
"authenticated": bool(result.get("authenticated")),
|
||||
"http_status": int(result.get("http_status") or error.get("http_status") or 0),
|
||||
"business_code": str(result.get("business_code") or error.get("code") or "")[:80],
|
||||
"region": str(result.get("region") or "")[:160],
|
||||
"message": str(result.get("message") or "").replace("\r", " ").replace("\n", " ")[:300],
|
||||
}
|
||||
|
||||
|
||||
def default_provider_probe(provider, key, proxy, timeout, debug=False):
|
||||
if provider == "deepseek":
|
||||
from keycheckers.deepseek import deepseekKeycheck
|
||||
|
||||
return deepseekKeycheck.check_key(key, proxy, timeout)
|
||||
if provider == "zai":
|
||||
from keycheckers.zai import zaiKeycheck
|
||||
|
||||
return zaiKeycheck.check_key(
|
||||
key, zaiKeycheck.base_urls_from_environment(), proxy, timeout, debug,
|
||||
)
|
||||
if provider == "qwen":
|
||||
from keycheckers.qwen import qwenKeycheck
|
||||
|
||||
custom = qwenKeycheck.split_csv(
|
||||
os.getenv("QWEN_BASE_URLS") or os.getenv("DASHSCOPE_BASE_URLS")
|
||||
)
|
||||
base_urls = qwenKeycheck.unique_ordered([*custom, *qwenKeycheck.DEFAULT_BASE_URLS])
|
||||
return qwenKeycheck.check_key(key, base_urls, bool(custom), proxy, timeout, debug)
|
||||
if provider == "kimi":
|
||||
from keycheckers.kimi import kimiKeycheck
|
||||
|
||||
custom = kimiKeycheck.split_csv(
|
||||
os.getenv("KIMI_BASE_URLS") or os.getenv("MOONSHOT_BASE_URLS")
|
||||
)
|
||||
base_urls = kimiKeycheck.unique_ordered([*custom, *kimiKeycheck.DEFAULT_BASE_URLS])
|
||||
return kimiKeycheck.check_key(key, base_urls, proxy, timeout, debug)
|
||||
raise ValueError(f"unsupported provider resolution adapter: {provider}")
|
||||
|
||||
|
||||
def resolve_provider_key(
|
||||
key, finding=None, proxy=None, timeout=15, debug=False, hint="",
|
||||
origin_service="", configured_order=None, probe=None,
|
||||
):
|
||||
providers = ordered_providers(finding, hint, origin_service, configured_order)
|
||||
probe = probe or default_provider_probe
|
||||
attempts = []
|
||||
retry_results = []
|
||||
for provider in providers:
|
||||
result = probe(provider, key, proxy, timeout, debug)
|
||||
result = result if isinstance(result, dict) else {"status": "UNKNOWN"}
|
||||
outcome = provider_result_outcome(result)
|
||||
attempts.append(bounded_attempt(provider, result, outcome))
|
||||
if outcome == "match":
|
||||
return {
|
||||
**result,
|
||||
"resolved_provider": provider,
|
||||
"provider_resolution": "matched",
|
||||
"provider_resolution_order": providers,
|
||||
"provider_resolution_attempts": attempts,
|
||||
"result_source": "provider_resolution",
|
||||
}
|
||||
if outcome == "retry":
|
||||
retry_results.append(result)
|
||||
|
||||
if retry_results:
|
||||
selected = retry_results[0]
|
||||
return {
|
||||
**selected,
|
||||
"provider_resolution": "retry",
|
||||
"provider_resolution_order": providers,
|
||||
"provider_resolution_attempts": attempts,
|
||||
"result_source": "provider_resolution",
|
||||
"message": str(selected.get("message") or "provider resolution remains inconclusive")[:1000],
|
||||
}
|
||||
return {
|
||||
"status": "DEAD",
|
||||
"provider_resolution": "exhausted",
|
||||
"provider_resolution_order": providers,
|
||||
"provider_resolution_attempts": attempts,
|
||||
"result_source": "provider_resolution",
|
||||
"message": "all compatible providers rejected the credential",
|
||||
}
|
||||
@@ -0,0 +1,203 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from keycheckers.provider_resolution import (
|
||||
AMBIGUOUS_GENERIC_SK_HINT,
|
||||
AMBIGUOUS_QWEN_DEEPSEEK_HINT,
|
||||
resolve_provider_key,
|
||||
)
|
||||
|
||||
|
||||
SERVICE = "provider_resolver"
|
||||
DETECTOR = "ProviderResolver"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "providerResolverChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "providerResolverResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "providerResolverAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "providerResolverNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "providerResolverDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "providerResolverRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "providerResolverLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "providerResolverNetwork.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "providerResolverNoContext.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "providerResolverUnknown.txt"),
|
||||
}
|
||||
|
||||
RESOLVABLE_KEY_REGEX = re.compile(
|
||||
r"(?<![A-Za-z0-9_.-])(?:"
|
||||
r"(?:zai|sk)-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|"
|
||||
r"[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}"
|
||||
r")(?![A-Za-z0-9_.-])"
|
||||
)
|
||||
FOREIGN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
|
||||
AMBIGUOUS_HINTS = {AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT}
|
||||
|
||||
|
||||
def key_rejection_reason(key):
|
||||
value = str(key or "")
|
||||
try:
|
||||
encoded = value.encode("utf-8", errors="strict")
|
||||
except UnicodeEncodeError:
|
||||
return "candidate is not valid UTF-8"
|
||||
if len(encoded) > 512:
|
||||
return "candidate exceeds the 512-byte key limit"
|
||||
if value.startswith(FOREIGN_KEY_PREFIXES):
|
||||
return "candidate has a foreign provider prefix"
|
||||
if not RESOLVABLE_KEY_REGEX.fullmatch(value):
|
||||
return "candidate does not match a bounded resolvable provider-key format"
|
||||
return ""
|
||||
|
||||
|
||||
def iter_candidate_keys(input_file):
|
||||
detector_names = [
|
||||
"ProviderResolver", "CustomRegex", "QwenDashScope", "Qwen_DashScope",
|
||||
"Qwen", "DashScope", "DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key",
|
||||
"KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi", "ZaiGLM",
|
||||
"qwendashscope", "qwen_dashscope", "qwen", "dashscope", "deepseek",
|
||||
"deepseekapikey", "deepseek_api_key", "kimimoonshot", "moonshotai",
|
||||
"moonshot", "kimi", "zaiglm",
|
||||
]
|
||||
for item in iter_findings(input_file, detector_names):
|
||||
finding = item.get("finding") or {}
|
||||
key = item.get("credential_secret_text") or ""
|
||||
if not key:
|
||||
for value in (item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2")):
|
||||
match = RESOLVABLE_KEY_REGEX.search(str(value or ""))
|
||||
if match:
|
||||
key = match.group(0)
|
||||
break
|
||||
if not key or key_rejection_reason(key):
|
||||
continue
|
||||
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
|
||||
metadata_hint = ""
|
||||
active_metadata = item.get("candidate_metadata")
|
||||
if isinstance(active_metadata, dict):
|
||||
metadata_hint = str(active_metadata.get("provider_hint") or "")
|
||||
metadata_candidates = active_metadata.get("provider_candidates")
|
||||
if metadata_hint or isinstance(metadata_candidates, list):
|
||||
context = dict(context)
|
||||
if metadata_hint:
|
||||
context.setdefault("provider_hint", metadata_hint)
|
||||
if isinstance(metadata_candidates, list):
|
||||
context.setdefault("provider_candidates", metadata_candidates)
|
||||
finding = dict(finding)
|
||||
finding["ScannerContext"] = context
|
||||
hint = str(context.get("provider_hint") or metadata_hint or AMBIGUOUS_GENERIC_SK_HINT)
|
||||
if hint not in AMBIGUOUS_HINTS:
|
||||
hint = AMBIGUOUS_GENERIC_SK_HINT
|
||||
yield key, item.get("source") or input_file, finding, hint
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status,
|
||||
result.get("message", ""), result.get("resolved_provider") or source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def retry_statuses_from_args(args):
|
||||
statuses = set()
|
||||
for enabled, status in (
|
||||
(args.retry_network, "NETWORK"),
|
||||
(args.retry_limited, "LIMITED"),
|
||||
(args.retry_unknown, "UNKNOWN"),
|
||||
(args.retry_restricted, "RESTRICTED"),
|
||||
(args.retry_no_balance, "NO_BALANCE"),
|
||||
(args.retry_valid, "VALID"),
|
||||
):
|
||||
if enabled:
|
||||
statuses.add(status)
|
||||
return statuses
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Ambiguous generic provider key resolver")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = retry_statuses_from_args(args)
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding, hint in iter_candidate_keys(args.input):
|
||||
if should_skip_key(
|
||||
key, checked, known, args, retry_statuses, service=SERVICE,
|
||||
source=source, finding=finding, detector=DETECTOR,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] Ambiguous provider candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
result = resolve_provider_key(
|
||||
key, finding, proxy, args.timeout, args.debug, hint=hint,
|
||||
)
|
||||
print(
|
||||
f" STATUS: {result['status']} provider={result.get('resolved_provider', '')} "
|
||||
f"| {result.get('message', '')[:200]}"
|
||||
)
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,773 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
||||
sys.stderr.reconfigure(encoding="utf-8", errors="replace")
|
||||
except (AttributeError, OSError, ValueError):
|
||||
pass
|
||||
|
||||
from keycheck_common import (
|
||||
combined_provider_routing_hint,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
provider_routing_database_failed,
|
||||
read_plain_keys,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from keycheckers.provider_resolution import resolve_provider_key
|
||||
|
||||
|
||||
SERVICE = "qwen"
|
||||
DETECTOR = "QwenDashScope"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "qwenChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "qwenResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "qwenAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "qwenDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "qwenRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "qwenLimited.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "qwenNoBalance.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "qwenNoContext.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "qwenNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "qwenUnknown.txt"),
|
||||
}
|
||||
|
||||
QWEN_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope", "dashscope", "qwen"}
|
||||
QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"}
|
||||
DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"}
|
||||
KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"}
|
||||
QWEN_KEY_MAX_BYTES = 512
|
||||
QWEN_KEY_REGEX = re.compile(
|
||||
r"(?<![A-Za-z0-9_-])sk-(?:sp-)?[A-Za-z0-9][A-Za-z0-9_-]{20,505}(?![A-Za-z0-9_-])"
|
||||
)
|
||||
QWEN_OVERSIZED_KEY_PREFIX_REGEX = re.compile(
|
||||
r"(?<![A-Za-z0-9_-])sk-(?:sp-)?[A-Za-z0-9][A-Za-z0-9_-]{506}"
|
||||
)
|
||||
OVERLAPPING_QWEN_DEEPSEEK_REGEX = re.compile(r"sk-[a-z0-9]{32}")
|
||||
FOREIGN_QWEN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
|
||||
OPENAI_LEGACY_KEY_MARKER = "T3BlbkFJ"
|
||||
QWEN_CONTEXT_REGEX = re.compile(
|
||||
r"(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
DEEPSEEK_CONTEXT_REGEX = re.compile(r"(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)", re.IGNORECASE)
|
||||
KIMI_CONTEXT_REGEX = re.compile(
|
||||
r"(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
AMBIGUOUS_PROVIDER_HINT = "ambiguous_qwen_deepseek"
|
||||
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
|
||||
GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"}
|
||||
EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment"
|
||||
CANDIDATE_PROVIDER_ROUTE_FIELD = "_keycheck_candidate_provider_route"
|
||||
DEFAULT_BASE_URLS = [
|
||||
"https://coding-intl.dashscope.aliyuncs.com/v1",
|
||||
"https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
|
||||
"https://dashscope-us.aliyuncs.com/compatible-mode/v1",
|
||||
"https://dashscope.aliyuncs.com/compatible-mode/v1",
|
||||
"https://cn-hongkong.dashscope.aliyuncs.com/compatible-mode/v1",
|
||||
]
|
||||
MODEL_MARKERS = ("qwen", "qwq", "qvq", "wan", "text-embedding", "multimodal-embedding")
|
||||
CHAT_MODEL_PRIORITY = (
|
||||
"qwen-plus",
|
||||
"qwen-turbo",
|
||||
"qwen-max",
|
||||
"qwen3-235b-a22b",
|
||||
"qwen3-32b",
|
||||
"qwen2.5-72b-instruct",
|
||||
"qwen2.5-32b-instruct",
|
||||
"qwen2.5-14b-instruct",
|
||||
"qwen2.5-7b-instruct",
|
||||
"qwq-32b",
|
||||
)
|
||||
NON_CHAT_MODEL_MARKERS = ("embedding", "rerank", "wan", "image", "audio", "tts", "asr", "vision", "vl")
|
||||
|
||||
|
||||
def normalize_base_url(value):
|
||||
value = str(value or "").strip()
|
||||
if not value:
|
||||
return ""
|
||||
return value.rstrip("/")
|
||||
|
||||
|
||||
def split_csv(value):
|
||||
if not value:
|
||||
return []
|
||||
if isinstance(value, str):
|
||||
return [item.strip() for item in value.split(",") if item.strip()]
|
||||
return [str(item).strip() for item in value if str(item).strip()]
|
||||
|
||||
|
||||
def unique_ordered(values):
|
||||
seen = set()
|
||||
output = []
|
||||
for value in values:
|
||||
normalized = normalize_base_url(value)
|
||||
if normalized and normalized not in seen:
|
||||
seen.add(normalized)
|
||||
output.append(normalized)
|
||||
return output
|
||||
|
||||
|
||||
def endpoint_label(base_url):
|
||||
parsed = urlparse(base_url)
|
||||
return parsed.netloc or base_url
|
||||
|
||||
|
||||
def is_qwen_detector(value):
|
||||
return str(value or "").lower() in QWEN_DETECTOR_NAMES
|
||||
|
||||
|
||||
def custom_detector_name(data):
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
|
||||
name = str(extra.get("name") or "")
|
||||
if str(data.get("DetectorName") or "").lower() == "customregex" and is_qwen_detector(name):
|
||||
return name
|
||||
return ""
|
||||
|
||||
|
||||
def finding_detector_names(data):
|
||||
if not isinstance(data, dict):
|
||||
return set()
|
||||
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
|
||||
names = {
|
||||
str(data.get("DetectorName") or data.get("detector") or "").strip().lower(),
|
||||
str(extra.get("name") or "").strip().lower(),
|
||||
}
|
||||
return {name for name in names if name}
|
||||
|
||||
|
||||
def finding_has_explicit_detector(data, detector_names):
|
||||
if not isinstance(data, dict):
|
||||
return False
|
||||
if finding_detector_names(data) & set(detector_names):
|
||||
return True
|
||||
nested = data.get("finding")
|
||||
return isinstance(nested, dict) and bool(finding_detector_names(nested) & set(detector_names))
|
||||
|
||||
|
||||
def detector_name_from_finding(data):
|
||||
if not isinstance(data, dict):
|
||||
return ""
|
||||
if is_qwen_detector(data.get("DetectorName")):
|
||||
return data.get("DetectorName")
|
||||
custom_name = custom_detector_name(data)
|
||||
if custom_name:
|
||||
return custom_name
|
||||
if is_qwen_detector(data.get("detector")):
|
||||
return data.get("detector")
|
||||
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict):
|
||||
if is_qwen_detector(finding.get("DetectorName")):
|
||||
return finding.get("DetectorName")
|
||||
custom_name = custom_detector_name(finding)
|
||||
if custom_name:
|
||||
return custom_name
|
||||
return ""
|
||||
|
||||
|
||||
def key_from_text(*values):
|
||||
for value in values:
|
||||
for match in QWEN_KEY_REGEX.finditer(str(value or "")):
|
||||
key = match.group(0)
|
||||
if not key.startswith(FOREIGN_QWEN_KEY_PREFIXES) and OPENAI_LEGACY_KEY_MARKER not in key:
|
||||
return key
|
||||
return ""
|
||||
|
||||
|
||||
def qwen_key_rejection_reason(key):
|
||||
value = str(key or "")
|
||||
try:
|
||||
key_bytes = len(value.encode("utf-8", errors="strict"))
|
||||
except UnicodeEncodeError:
|
||||
return "candidate is not valid UTF-8"
|
||||
if key_bytes > QWEN_KEY_MAX_BYTES:
|
||||
return f"candidate exceeds the {QWEN_KEY_MAX_BYTES}-byte key limit"
|
||||
if OPENAI_LEGACY_KEY_MARKER in value:
|
||||
return "candidate is a recognizable OpenAI legacy key"
|
||||
if value.count("sk-") != 1:
|
||||
return "candidate contains multiple concatenated key prefixes"
|
||||
if not QWEN_KEY_REGEX.fullmatch(value) or value.startswith(FOREIGN_QWEN_KEY_PREFIXES):
|
||||
return "candidate does not match the bounded Qwen key format"
|
||||
return ""
|
||||
|
||||
|
||||
def is_qwen_key(key):
|
||||
return not qwen_key_rejection_reason(key)
|
||||
|
||||
|
||||
def finding_has_oversized_qwen_key(finding, *raw_values):
|
||||
values = list(raw_values)
|
||||
if isinstance(finding, dict):
|
||||
values.extend((finding.get("Raw"), finding.get("RawV2"), finding.get("raw"), finding.get("raw_v2")))
|
||||
nested = finding.get("finding")
|
||||
if isinstance(nested, dict):
|
||||
values.extend((nested.get("Raw"), nested.get("RawV2"), nested.get("raw"), nested.get("raw_v2")))
|
||||
return any(
|
||||
QWEN_OVERSIZED_KEY_PREFIX_REGEX.search(str(value or ""))
|
||||
for value in values
|
||||
)
|
||||
|
||||
|
||||
def warn_rejected_candidate(reason, source, key=""):
|
||||
reason = str(reason or "candidate rejected")
|
||||
source = str(source or "unknown")
|
||||
if key:
|
||||
reason = reason.replace(key, "***REDACTED***")
|
||||
source = source.replace(key, "***REDACTED***")
|
||||
reason = reason.replace("\r", " ").replace("\n", " ")[:300]
|
||||
source = source.replace("\r", " ").replace("\n", " ")[:300]
|
||||
print(f"Warning: skipped Qwen candidate from {source}: {reason}", flush=True)
|
||||
|
||||
|
||||
def warn_candidate_failure(reason, source, key=""):
|
||||
reason = str(reason or "candidate failure")
|
||||
source = str(source or "unknown")
|
||||
if key:
|
||||
reason = reason.replace(key, "***REDACTED***")
|
||||
source = source.replace(key, "***REDACTED***")
|
||||
reason = QWEN_KEY_REGEX.sub("***REDACTED***", reason).replace("\r", " ").replace("\n", " ")[:300]
|
||||
source = QWEN_KEY_REGEX.sub("***REDACTED***", source).replace("\r", " ").replace("\n", " ")[:300]
|
||||
print(f"Warning: Qwen candidate failure from {source}: {reason}", flush=True)
|
||||
|
||||
|
||||
def extract_key_from_finding(data):
|
||||
if not detector_name_from_finding(data):
|
||||
return ""
|
||||
if data.get("Raw") or data.get("RawV2"):
|
||||
return key_from_text(data.get("Raw"), data.get("RawV2"))
|
||||
if data.get("raw") or data.get("raw_v2"):
|
||||
return key_from_text(data.get("raw"), data.get("raw_v2"))
|
||||
finding = data.get("finding")
|
||||
if isinstance(finding, dict):
|
||||
return key_from_text(finding.get("Raw"), finding.get("RawV2"))
|
||||
return ""
|
||||
|
||||
|
||||
def finding_provider_routing_hint(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return ""
|
||||
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
|
||||
persisted_hint = context.get("provider_hint")
|
||||
if (
|
||||
context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE
|
||||
and persisted_hint in (*GENERIC_SK_PROVIDERS, AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
):
|
||||
return persisted_hint
|
||||
parts = [str(context.get(key) or "") for key in ("nearby", "file")]
|
||||
metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {}
|
||||
data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {}
|
||||
for details in data.values():
|
||||
if not isinstance(details, dict):
|
||||
continue
|
||||
parts.extend(str(details.get(key) or "") for key in ("file", "repository", "repo", "link", "image"))
|
||||
|
||||
text = "\n".join(parts)
|
||||
evidence = set()
|
||||
if QWEN_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("qwen")
|
||||
if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("deepseek")
|
||||
if KIMI_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES):
|
||||
evidence.add("kimi")
|
||||
if persisted_hint == AMBIGUOUS_PROVIDER_HINT:
|
||||
evidence.update(("qwen", "deepseek"))
|
||||
elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT:
|
||||
evidence.update(GENERIC_SK_PROVIDERS)
|
||||
elif persisted_hint in GENERIC_SK_PROVIDERS:
|
||||
evidence.add(persisted_hint)
|
||||
if len(evidence) > 1:
|
||||
return AMBIGUOUS_PROVIDER_HINT if evidence == {"qwen", "deepseek"} else AMBIGUOUS_GENERIC_SK_HINT
|
||||
return next(iter(evidence)) if evidence else ""
|
||||
|
||||
|
||||
def finding_has_ambiguous_provider_hint(finding):
|
||||
return finding_provider_routing_hint(finding) in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def iter_candidate_keys(input_file, plain_files, trusted_retry_files=None):
|
||||
seen_plain = set()
|
||||
seen_candidates = set()
|
||||
routing_decisions = {}
|
||||
detector_names = [
|
||||
"QwenDashScope", "Qwen_DashScope", "qwendashscope", "qwen_dashscope",
|
||||
"Qwen", "DashScope", "qwen", "dashscope", "CustomRegex",
|
||||
]
|
||||
for item in iter_findings(input_file, detector_names):
|
||||
data = dict(item.get("finding") or {})
|
||||
data.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, None)
|
||||
candidate_metadata = item.get("candidate_metadata")
|
||||
persisted_route = ""
|
||||
if keycheck_input_mode() == "postgres" and isinstance(candidate_metadata, dict):
|
||||
persisted_route = str(candidate_metadata.get("provider_hint") or "").lower()
|
||||
if persisted_route == SERVICE:
|
||||
data[CANDIDATE_PROVIDER_ROUTE_FIELD] = SERVICE
|
||||
if finding_has_oversized_qwen_key(data, item.get("raw"), item.get("raw_v2")):
|
||||
warn_rejected_candidate(
|
||||
f"candidate exceeds the {QWEN_KEY_MAX_BYTES}-byte key limit",
|
||||
item.get("source") or input_file,
|
||||
)
|
||||
continue
|
||||
if persisted_route == SERVICE:
|
||||
key = key_from_text(
|
||||
item.get("raw"), item.get("raw_v2"),
|
||||
data.get("Raw"), data.get("RawV2"),
|
||||
)
|
||||
else:
|
||||
key = extract_key_from_finding(data)
|
||||
if key and is_qwen_key(key):
|
||||
if not key.startswith("sk-sp-"):
|
||||
if persisted_route == SERVICE:
|
||||
hint, lookup_failed = SERVICE, False
|
||||
else:
|
||||
local_hint = finding_provider_routing_hint(data)
|
||||
if key in routing_decisions:
|
||||
hint, lookup_failed = routing_decisions[key]
|
||||
if local_hint == AMBIGUOUS_PROVIDER_HINT or (
|
||||
local_hint and hint and local_hint != hint
|
||||
):
|
||||
hint = AMBIGUOUS_PROVIDER_HINT
|
||||
elif not hint:
|
||||
hint = local_hint
|
||||
routing_decisions[key] = (hint, lookup_failed)
|
||||
else:
|
||||
hint = combined_provider_routing_hint(key, local_hint)
|
||||
lookup_failed = provider_routing_database_failed()
|
||||
routing_decisions[key] = (hint, lookup_failed)
|
||||
if lookup_failed:
|
||||
message = "provider routing evidence lookup failed closed"
|
||||
warn_candidate_failure(message, item.get("source") or input_file, key)
|
||||
raise RuntimeError(message)
|
||||
if hint != "qwen" and not (
|
||||
keycheck_input_mode() == "postgres"
|
||||
and hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
|
||||
):
|
||||
continue
|
||||
seen_candidates.add(key)
|
||||
yield key, item.get("source") or input_file, data
|
||||
|
||||
for item in read_plain_keys(plain_files, QWEN_KEY_REGEX):
|
||||
key = item["key"]
|
||||
if not is_qwen_key(key) or not key.startswith("sk-sp-"):
|
||||
continue
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
seen_candidates.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
owned_retry_paths = {
|
||||
os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values()
|
||||
}
|
||||
retry_files = [
|
||||
path for path in (trusted_retry_files or [])
|
||||
if os.path.normcase(os.path.abspath(path)) in owned_retry_paths
|
||||
]
|
||||
for item in read_plain_keys(retry_files, QWEN_KEY_REGEX):
|
||||
key = item["key"]
|
||||
if not is_qwen_key(key) or key in seen_candidates:
|
||||
continue
|
||||
if not key.startswith("sk-sp-"):
|
||||
hint = combined_provider_routing_hint(key, "qwen")
|
||||
if provider_routing_database_failed():
|
||||
message = "provider routing evidence lookup failed closed"
|
||||
warn_candidate_failure(message, item["source"], key)
|
||||
raise RuntimeError(message)
|
||||
if hint != "qwen":
|
||||
continue
|
||||
seen_candidates.add(key)
|
||||
yield key, item["source"], {}
|
||||
|
||||
|
||||
def redact_text(text, key):
|
||||
redacted = str(text or "")[:1000]
|
||||
if key:
|
||||
redacted = redacted.replace(key, "***REDACTED***")
|
||||
return QWEN_KEY_REGEX.sub("***REDACTED***", redacted)
|
||||
|
||||
|
||||
def parse_error_response(response, key):
|
||||
try:
|
||||
payload = response.json()
|
||||
except ValueError:
|
||||
payload = {}
|
||||
error = payload.get("error") if isinstance(payload, dict) else {}
|
||||
if not isinstance(error, dict):
|
||||
error = {}
|
||||
message = error.get("message") or response.text[:500]
|
||||
return {
|
||||
"http_status": response.status_code,
|
||||
"code": error.get("code") or error.get("type") or "",
|
||||
"type": error.get("type") or "",
|
||||
"message": redact_text(message, key),
|
||||
}
|
||||
|
||||
|
||||
def classify_error(error):
|
||||
http_status = int(error.get("http_status") or 0)
|
||||
code = str(error.get("code") or "").lower()
|
||||
message = str(error.get("message") or "").lower()
|
||||
|
||||
if http_status == 401 or "invalid_api_key" in code or "incorrect api key" in message:
|
||||
return "DEAD"
|
||||
if http_status == 402 or "arrearage" in code or any(item in message for item in (
|
||||
"arrearage", "arrears", "billing", "balance", "overdue", "payment",
|
||||
"insufficient credit", "credit balance",
|
||||
)):
|
||||
return "NO_BALANCE"
|
||||
if http_status == 403:
|
||||
return "RESTRICTED"
|
||||
if http_status == 429:
|
||||
return "LIMITED"
|
||||
if 500 <= http_status <= 599:
|
||||
return "NETWORK"
|
||||
return "UNKNOWN"
|
||||
|
||||
|
||||
def choose_chat_model(models):
|
||||
models = [str(model or "").replace("models/", "") for model in models if model]
|
||||
by_lower = {model.lower(): model for model in models}
|
||||
for model in CHAT_MODEL_PRIORITY:
|
||||
if model.lower() in by_lower:
|
||||
return by_lower[model.lower()]
|
||||
for model in models:
|
||||
lowered = model.lower()
|
||||
if any(marker in lowered for marker in NON_CHAT_MODEL_MARKERS):
|
||||
continue
|
||||
if any(marker in lowered for marker in ("qwen", "qwq", "qvq")):
|
||||
return model
|
||||
return ""
|
||||
|
||||
|
||||
def probe_chat_completion(key, base_url, model, proxy, timeout, debug=False):
|
||||
if not model:
|
||||
return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""}
|
||||
url = f"{normalize_base_url(base_url)}/chat/completions"
|
||||
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1}
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)[:1000], "model": model}
|
||||
if debug:
|
||||
print(f" DEBUG {endpoint_label(base_url)} chat ping {model}: HTTP {response.status_code}: {redact_text(response.text[:500], key)}")
|
||||
if response.status_code == 200:
|
||||
return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model}
|
||||
error = parse_error_response(response, key)
|
||||
return {"status": classify_error(error), "error": error, "message": error.get("message") or "", "model": model}
|
||||
|
||||
|
||||
def parse_models(payload):
|
||||
if not isinstance(payload, dict):
|
||||
return [], []
|
||||
model_infos = payload.get("data")
|
||||
if not isinstance(model_infos, list):
|
||||
model_infos = payload.get("models") if isinstance(payload.get("models"), list) else []
|
||||
models = []
|
||||
for item in model_infos:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
model_id = item.get("id") or item.get("model") or item.get("name")
|
||||
if model_id:
|
||||
models.append(str(model_id).replace("models/", ""))
|
||||
return sorted(set(models)), model_infos
|
||||
|
||||
|
||||
def notable_models(models):
|
||||
notable = []
|
||||
for model in models:
|
||||
lowered = model.lower()
|
||||
if any(marker in lowered for marker in MODEL_MARKERS):
|
||||
notable.append(model)
|
||||
return notable[:30]
|
||||
|
||||
|
||||
def check_base_url(key, base_url, proxy, timeout, debug=False):
|
||||
url = f"{normalize_base_url(base_url)}/models"
|
||||
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {
|
||||
"base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"status": "NETWORK",
|
||||
"message": str(exc)[:1000],
|
||||
}
|
||||
|
||||
if debug:
|
||||
print(f" DEBUG {endpoint_label(base_url)} /models: HTTP {response.status_code}: {redact_text(response.text[:500], key)}")
|
||||
|
||||
if response.status_code == 200:
|
||||
try:
|
||||
payload = response.json()
|
||||
except ValueError:
|
||||
payload = {}
|
||||
models, model_infos = parse_models(payload)
|
||||
chat_model = choose_chat_model(models)
|
||||
probe = probe_chat_completion(key, base_url, chat_model, proxy, timeout, debug)
|
||||
probe_status = probe.get("status") or "UNKNOWN"
|
||||
status = "VALID" if probe_status in ("GENERATION_OK", "NO_CONTEXT") else probe_status
|
||||
probe_message = probe.get("message") or json.dumps(probe.get("error") or {}, ensure_ascii=False)[:1000]
|
||||
return {
|
||||
"base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"status": status,
|
||||
"authenticated": True,
|
||||
"model_count": len(models),
|
||||
"models": notable_models(models),
|
||||
"all_model_count": len(models),
|
||||
"model_infos_count": len(model_infos),
|
||||
"llm_probe_status": probe_status,
|
||||
"llm_probe_model": probe.get("model", chat_model),
|
||||
"message": (
|
||||
f"chat ping ok; model={chat_model}; models={len(models)}"
|
||||
if probe_status == "GENERATION_OK"
|
||||
else f"models authenticated; generation_probe={probe_status}; models={len(models)}; {probe_message}"
|
||||
)[:1000],
|
||||
"error": probe.get("error") or {},
|
||||
}
|
||||
|
||||
error = parse_error_response(response, key)
|
||||
return {
|
||||
"base_url": base_url,
|
||||
"region": endpoint_label(base_url),
|
||||
"status": classify_error(error),
|
||||
"error": error,
|
||||
"message": error.get("message") or "",
|
||||
}
|
||||
|
||||
|
||||
def choose_final_status(key, attempts, has_custom_base_urls):
|
||||
statuses = [attempt.get("status") for attempt in attempts]
|
||||
for status in ("VALID", "NO_BALANCE", "LIMITED", "RESTRICTED", "UNKNOWN", "NO_CONTEXT", "NETWORK"):
|
||||
if status in statuses:
|
||||
return status
|
||||
return "DEAD"
|
||||
|
||||
|
||||
def check_key(key, base_urls, has_custom_base_urls, proxy, timeout, debug=False):
|
||||
rejection = qwen_key_rejection_reason(key)
|
||||
if rejection:
|
||||
return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True}
|
||||
attempts = []
|
||||
for base_url in base_urls:
|
||||
result = check_base_url(key, base_url, proxy, timeout, debug)
|
||||
attempts.append(result)
|
||||
if result.get("status") == "VALID":
|
||||
return {
|
||||
"status": "VALID",
|
||||
"region": result.get("region"),
|
||||
"base_url": result.get("base_url"),
|
||||
"model_count": result.get("model_count", 0),
|
||||
"models": result.get("models", []),
|
||||
"llm_probe_status": result.get("llm_probe_status", ""),
|
||||
"llm_probe_model": result.get("llm_probe_model", ""),
|
||||
"authenticated": bool(result.get("authenticated")),
|
||||
"attempts": attempts,
|
||||
"message": f"models={result.get('model_count', 0)} region={result.get('region')}",
|
||||
}
|
||||
|
||||
status = choose_final_status(key, attempts, has_custom_base_urls)
|
||||
message = ""
|
||||
for attempt in attempts:
|
||||
if attempt.get("status") == status:
|
||||
message = attempt.get("message") or json.dumps(attempt.get("error") or {}, ensure_ascii=False)[:1000]
|
||||
break
|
||||
return {"status": status, "attempts": attempts, "message": message}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
rejection = qwen_key_rejection_reason(key)
|
||||
if rejection:
|
||||
raise ValueError(rejection)
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
if status == "VALID":
|
||||
extra = f"{result.get('region', '')};probe_model={result.get('llm_probe_model', '')};models={','.join(result.get('models', []))[:500]}"
|
||||
else:
|
||||
extra = source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE,
|
||||
STATUS_FILES,
|
||||
key,
|
||||
status,
|
||||
result.get("message", ""),
|
||||
extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def retry_statuses_from_args(args):
|
||||
retry_statuses = set()
|
||||
if args.retry_network:
|
||||
retry_statuses.add("NETWORK")
|
||||
if args.retry_limited:
|
||||
retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown:
|
||||
retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
|
||||
if args.retry_restricted:
|
||||
retry_statuses.add("RESTRICTED")
|
||||
if args.retry_no_balance:
|
||||
retry_statuses.add("NO_BALANCE")
|
||||
if args.retry_valid:
|
||||
retry_statuses.add("VALID")
|
||||
return retry_statuses
|
||||
|
||||
|
||||
def retry_input_files_from_args(args):
|
||||
if args.recheck_all:
|
||||
statuses = list(STATUS_FILES)
|
||||
else:
|
||||
statuses = []
|
||||
if args.retry_network:
|
||||
statuses.append("NETWORK")
|
||||
if args.retry_limited:
|
||||
statuses.append("LIMITED")
|
||||
if args.retry_unknown:
|
||||
statuses.extend(("UNKNOWN", "NO_CONTEXT"))
|
||||
if args.retry_restricted:
|
||||
statuses.append("RESTRICTED")
|
||||
if args.retry_no_balance:
|
||||
statuses.append("NO_BALANCE")
|
||||
if args.retry_valid:
|
||||
statuses.append("VALID")
|
||||
return list(dict.fromkeys(STATUS_FILES[status] for status in statuses))
|
||||
|
||||
|
||||
def base_urls_from_args(args):
|
||||
env_urls = split_csv(os.getenv("QWEN_BASE_URLS") or os.getenv("DASHSCOPE_BASE_URLS"))
|
||||
custom_urls = []
|
||||
for value in args.base_url:
|
||||
custom_urls.extend(split_csv(value))
|
||||
custom_urls.extend(env_urls)
|
||||
default_urls = [] if args.no_default_base_urls else DEFAULT_BASE_URLS
|
||||
return unique_ordered(custom_urls + default_urls), bool(custom_urls)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Qwen/DashScope key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--base-url", action="append", default=[], help="Extra DashScope/OpenAI-compatible base URL; can be repeated")
|
||||
parser.add_argument("--no-default-base-urls", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = retry_statuses_from_args(args)
|
||||
retry_input_files = retry_input_files_from_args(args)
|
||||
base_urls, has_custom_base_urls = base_urls_from_args(args)
|
||||
if not base_urls:
|
||||
raise SystemExit("No Qwen/DashScope base URLs configured")
|
||||
|
||||
print("--- Qwen/DashScope key checker ---")
|
||||
print("base_urls: " + ", ".join(endpoint_label(url) for url in base_urls))
|
||||
processed = 0
|
||||
skipped = 0
|
||||
for key, source, finding in iter_candidate_keys(args.input, args.plain, retry_input_files):
|
||||
finding = dict(finding or {})
|
||||
candidate_route = str(finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, "") or "").lower()
|
||||
rejection = qwen_key_rejection_reason(key)
|
||||
if rejection:
|
||||
skipped += 1
|
||||
warn_rejected_candidate(rejection, source, key)
|
||||
continue
|
||||
try:
|
||||
should_skip = should_skip_key(
|
||||
key, checked, known, args, retry_statuses, service=SERVICE,
|
||||
source=source, finding=finding, detector=DETECTOR,
|
||||
)
|
||||
except Exception as exc:
|
||||
warn_candidate_failure(f"candidate preparation failed: {exc}", source, key)
|
||||
raise
|
||||
if should_skip:
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
try:
|
||||
print(f"\n[{processed}] Qwen/DashScope candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
routing_hint = "qwen"
|
||||
if keycheck_input_mode() == "postgres":
|
||||
if candidate_route == SERVICE:
|
||||
routing_hint = SERVICE
|
||||
else:
|
||||
routing_hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding))
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
if routing_hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT):
|
||||
result = resolve_provider_key(
|
||||
key, finding, proxy, args.timeout, args.debug,
|
||||
hint=routing_hint, origin_service=SERVICE,
|
||||
)
|
||||
else:
|
||||
result = check_key(key, base_urls, has_custom_base_urls, proxy, args.timeout, args.debug)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
except Exception as exc:
|
||||
warn_candidate_failure(f"candidate processing failed: {exc}", source, key)
|
||||
raise
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,403 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
from collections import Counter
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl, classify_common_http_status, commit_status_transaction,
|
||||
default_input_file, default_proxy_file, ensure_output_files, iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses, load_known_keys, load_proxies, mask_secret,
|
||||
read_plain_keys, record_validation_result, recover_status_transaction,
|
||||
request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event,
|
||||
)
|
||||
|
||||
SERVICE = "replicate"
|
||||
DETECTOR = "Replicate"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "replicateChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "replicateResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "replicateAlive.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "replicateDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "replicateRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "replicateLimited.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "replicateNoBalance.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "replicateNetwork.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "replicateNoContext.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "replicateUnknown.txt"),
|
||||
}
|
||||
KEY_REGEX = re.compile(r"\br8_[A-Za-z0-9]{30,}\b")
|
||||
API_BASE = "https://api.replicate.com/v1"
|
||||
ACCOUNT_URL = f"{API_BASE}/account"
|
||||
RESOURCE_ENDPOINTS = {
|
||||
"predictions": f"{API_BASE}/predictions",
|
||||
"deployments": f"{API_BASE}/deployments",
|
||||
"trainings": f"{API_BASE}/trainings",
|
||||
}
|
||||
NO_BALANCE_MARKERS = (
|
||||
"balance",
|
||||
"billing",
|
||||
"credit",
|
||||
"credits",
|
||||
"payment",
|
||||
"insufficient",
|
||||
"depleted",
|
||||
"no credits",
|
||||
"out of credit",
|
||||
"run out of credit",
|
||||
)
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def auth_headers(key):
|
||||
return {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
||||
|
||||
|
||||
def redacted_error_message(response, key):
|
||||
return request_error_message(response).replace(key, "***REDACTED***")
|
||||
|
||||
|
||||
def classify_replicate_response(response, key):
|
||||
message = redacted_error_message(response, key).lower()
|
||||
if response.status_code == 402 or any(marker in message for marker in NO_BALANCE_MARKERS):
|
||||
return "NO_BALANCE"
|
||||
if response.status_code == 403:
|
||||
return "RESTRICTED"
|
||||
return classify_common_http_status(response.status_code)
|
||||
|
||||
|
||||
def api_get(key, url, proxy, timeout, debug=False):
|
||||
try:
|
||||
response = requests.get(url, headers=auth_headers(key), proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)[:1000], "payload": None}
|
||||
if debug:
|
||||
detail = "ok" if response.status_code == 200 else redacted_error_message(response, key)[:500]
|
||||
print(f" DEBUG GET {url}: HTTP {response.status_code}: {detail}")
|
||||
if response.status_code != 200:
|
||||
return {
|
||||
"status": classify_replicate_response(response, key),
|
||||
"http_status": response.status_code,
|
||||
"message": redacted_error_message(response, key),
|
||||
"payload": None,
|
||||
}
|
||||
try:
|
||||
payload = response.json() if response.text else {}
|
||||
except ValueError:
|
||||
payload = {}
|
||||
return {"status": "OK", "http_status": 200, "message": "ok", "payload": payload}
|
||||
|
||||
|
||||
def paginated_items(payload):
|
||||
if isinstance(payload, list):
|
||||
return payload
|
||||
if not isinstance(payload, dict):
|
||||
return []
|
||||
for key in ("results", "data", "items"):
|
||||
value = payload.get(key)
|
||||
if isinstance(value, list):
|
||||
return value
|
||||
return []
|
||||
|
||||
|
||||
def text_value(value):
|
||||
return str(value or "").strip()
|
||||
|
||||
|
||||
def compact_model_ref(value):
|
||||
if isinstance(value, str):
|
||||
return value.strip()
|
||||
if not isinstance(value, dict):
|
||||
return ""
|
||||
owner = text_value(value.get("owner") or value.get("model_owner"))
|
||||
name = text_value(value.get("name") or value.get("model_name"))
|
||||
if owner and name:
|
||||
return f"{owner}/{name}"
|
||||
for key in ("model", "id", "slug"):
|
||||
item = text_value(value.get(key))
|
||||
if item:
|
||||
return item
|
||||
url = text_value(value.get("url") or value.get("web_url"))
|
||||
if "replicate.com/" in url:
|
||||
return url.rstrip("/").split("replicate.com/", 1)[-1]
|
||||
return ""
|
||||
|
||||
|
||||
def model_refs_from_item(item):
|
||||
if not isinstance(item, dict):
|
||||
return []
|
||||
refs = []
|
||||
for key in ("model", "destination", "source_model", "base_model"):
|
||||
ref = compact_model_ref(item.get(key))
|
||||
if ref:
|
||||
refs.append(ref)
|
||||
for key in ("version", "latest_version", "current_release"):
|
||||
value = item.get(key)
|
||||
if isinstance(value, dict):
|
||||
ref = compact_model_ref(value.get("model") or value.get("destination"))
|
||||
if ref:
|
||||
refs.append(ref)
|
||||
return sorted(set(refs))
|
||||
|
||||
|
||||
def summarize_predictions(payload, limit=10):
|
||||
items = paginated_items(payload)
|
||||
models = sorted({text_value(item.get("model")) for item in items if isinstance(item, dict) and item.get("model")})
|
||||
statuses = Counter(text_value(item.get("status")) for item in items if isinstance(item, dict) and item.get("status"))
|
||||
samples = []
|
||||
for item in items[:limit]:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
samples.append({
|
||||
"id": text_value(item.get("id"))[:80],
|
||||
"status": text_value(item.get("status")),
|
||||
"model": text_value(item.get("model")),
|
||||
"source": text_value(item.get("source")),
|
||||
"data_removed": bool(item.get("data_removed")),
|
||||
"created_at": text_value(item.get("created_at")),
|
||||
"completed_at": text_value(item.get("completed_at")),
|
||||
})
|
||||
return {
|
||||
"prediction_count_sample": len(items),
|
||||
"prediction_has_next_page": bool(isinstance(payload, dict) and payload.get("next")),
|
||||
"prediction_status_counts": dict(statuses),
|
||||
"prediction_models": models[:50],
|
||||
"prediction_samples": samples,
|
||||
}
|
||||
|
||||
|
||||
def deployment_name(item):
|
||||
owner = text_value(item.get("owner") or item.get("deployment_owner"))
|
||||
name = text_value(item.get("name") or item.get("deployment_name"))
|
||||
if owner and name and "/" not in name:
|
||||
return f"{owner}/{name}"
|
||||
return name or owner
|
||||
|
||||
|
||||
def summarize_deployments(payload, limit=20):
|
||||
items = paginated_items(payload)
|
||||
models = sorted({ref for item in items for ref in model_refs_from_item(item)})
|
||||
deployments = []
|
||||
for item in items[:limit]:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
current_release = item.get("current_release") if isinstance(item.get("current_release"), dict) else {}
|
||||
deployments.append({
|
||||
"name": deployment_name(item),
|
||||
"model": next(iter(model_refs_from_item(item)), ""),
|
||||
"version": text_value(item.get("version") or current_release.get("version"))[:80],
|
||||
"hardware": text_value(item.get("hardware") or current_release.get("hardware")),
|
||||
"min_instances": item.get("min_instances"),
|
||||
"max_instances": item.get("max_instances"),
|
||||
})
|
||||
return {
|
||||
"deployment_count": len(items),
|
||||
"deployment_has_next_page": bool(isinstance(payload, dict) and payload.get("next")),
|
||||
"deployment_models": models[:50],
|
||||
"deployments": deployments,
|
||||
}
|
||||
|
||||
|
||||
def summarize_trainings(payload, limit=10):
|
||||
items = paginated_items(payload)
|
||||
models = sorted({ref for item in items for ref in model_refs_from_item(item)})
|
||||
statuses = Counter(text_value(item.get("status")) for item in items if isinstance(item, dict) and item.get("status"))
|
||||
samples = []
|
||||
for item in items[:limit]:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
samples.append({
|
||||
"id": text_value(item.get("id"))[:80],
|
||||
"status": text_value(item.get("status")),
|
||||
"model": next(iter(model_refs_from_item(item)), ""),
|
||||
"created_at": text_value(item.get("created_at")),
|
||||
"completed_at": text_value(item.get("completed_at")),
|
||||
})
|
||||
return {
|
||||
"training_count_sample": len(items),
|
||||
"training_has_next_page": bool(isinstance(payload, dict) and payload.get("next")),
|
||||
"training_status_counts": dict(statuses),
|
||||
"training_models": models[:50],
|
||||
"training_samples": samples,
|
||||
}
|
||||
|
||||
|
||||
def probe_account_resources(key, proxy, timeout, debug=False):
|
||||
summaries = {}
|
||||
endpoint_statuses = {}
|
||||
model_refs = set()
|
||||
ok_count = 0
|
||||
total_items = 0
|
||||
summarizers = {
|
||||
"predictions": summarize_predictions,
|
||||
"deployments": summarize_deployments,
|
||||
"trainings": summarize_trainings,
|
||||
}
|
||||
for name, url in RESOURCE_ENDPOINTS.items():
|
||||
result = api_get(key, url, proxy, timeout, debug)
|
||||
endpoint_statuses[name] = {k: v for k, v in result.items() if k in ("status", "http_status", "message")}
|
||||
if result.get("status") != "OK":
|
||||
continue
|
||||
ok_count += 1
|
||||
summary = summarizers[name](result.get("payload"))
|
||||
summaries.update(summary)
|
||||
for key_name, value in summary.items():
|
||||
if key_name.endswith("_models") and isinstance(value, list):
|
||||
model_refs.update(value)
|
||||
total_items += sum(
|
||||
int(summary.get(field, 0) or 0)
|
||||
for field in ("prediction_count_sample", "deployment_count", "training_count_sample")
|
||||
)
|
||||
if ok_count == len(RESOURCE_ENDPOINTS):
|
||||
probe_status = "RESOURCE_OK" if total_items else "NO_RESOURCES"
|
||||
elif ok_count:
|
||||
probe_status = "PARTIAL"
|
||||
else:
|
||||
probe_status = next((item.get("status") for item in endpoint_statuses.values() if item.get("status")), "UNKNOWN")
|
||||
return {
|
||||
"probe": {"status": probe_status, "endpoints": endpoint_statuses},
|
||||
"models": sorted(model_refs)[:50],
|
||||
"model_count": len(model_refs),
|
||||
**summaries,
|
||||
}
|
||||
|
||||
|
||||
def iter_candidate_decisions(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, [DETECTOR]):
|
||||
key = item.get("credential_secret_text") or item["raw"]
|
||||
if key:
|
||||
yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key))
|
||||
for item in read_plain_keys(plain_files, KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}, True
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files):
|
||||
if valid_format:
|
||||
yield key, source, finding
|
||||
|
||||
|
||||
def check_key(key, proxy, args):
|
||||
account = api_get(key, ACCOUNT_URL, proxy, args.timeout, args.debug)
|
||||
if account.get("status") != "OK":
|
||||
return {k: v for k, v in account.items() if k != "payload"}
|
||||
data = account.get("payload") if isinstance(account.get("payload"), dict) else {}
|
||||
result = {
|
||||
"status": "VALID",
|
||||
"message": "account endpoint accepted",
|
||||
"account": data.get("username") or data.get("name") or "",
|
||||
"account_type": data.get("type") or "",
|
||||
}
|
||||
if not args.no_resource_probe:
|
||||
result.update(probe_account_resources(key, proxy, args.timeout, args.debug))
|
||||
result["message"] = (
|
||||
f"account endpoint accepted; probe={result.get('probe', {}).get('status')}; "
|
||||
f"models={result.get('model_count', 0)}; "
|
||||
f"deployments={result.get('deployment_count', 0)}; "
|
||||
f"predictions={result.get('prediction_count_sample', 0)}; "
|
||||
f"trainings={result.get('training_count_sample', 0)}"
|
||||
)
|
||||
else:
|
||||
result.update({"probe": {"status": "not_probed"}, "models": [], "model_count": 0})
|
||||
return result
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
extra = ",".join(result.get("models") or [])[:1000] if status == "VALID" else source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra or source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Replicate key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--no-resource-probe", action="store_true", help="Only call /account; skip read-only predictions/deployments/trainings probes.")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network: retry_statuses.add("NETWORK")
|
||||
if args.retry_limited: retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
|
||||
if args.retry_restricted: retry_statuses.add("RESTRICTED")
|
||||
if args.retry_no_balance: retry_statuses.add("NO_BALANCE")
|
||||
if args.retry_valid: retry_statuses.add("VALID")
|
||||
processed = skipped = 0
|
||||
print("--- Replicate key checker ---")
|
||||
print("Default mode: /account plus read-only /predictions, /deployments and /trainings probes. Use --no-resource-probe for /account only.")
|
||||
print(f"proxy: {args.proxy_file}")
|
||||
postgres_mode = keycheck_input_mode() == "postgres"
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain):
|
||||
if not valid_format and not postgres_mode:
|
||||
skipped += 1
|
||||
continue
|
||||
if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] Replicate candidate {mask_secret(key)} from {source}")
|
||||
result = (
|
||||
check_key(key, next(proxy_cycler) if proxy_cycler else None, args)
|
||||
if valid_format else
|
||||
{"status": "NO_CONTEXT", "message": "candidate does not match canonical Replicate token format"}
|
||||
)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
if result.get("status") == "VALID":
|
||||
print(f" ACCOUNT: {result.get('account') or 'unknown'}")
|
||||
print(f" MODELS: {result.get('model_count', 0)} from account resources")
|
||||
notable = result.get("models") or []
|
||||
if notable:
|
||||
print(f" MODEL REFS: {', '.join(notable[:8])}")
|
||||
print(f" PROBE: {(result.get('probe') or {}).get('status')}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,231 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
append_jsonl, classify_common_http_status, commit_status_transaction,
|
||||
default_input_file, default_proxy_file, ensure_output_files, iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses, load_known_keys, load_proxies, mask_secret,
|
||||
read_plain_keys, record_validation_result, recover_status_transaction,
|
||||
request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event,
|
||||
)
|
||||
|
||||
SERVICE = "xai"
|
||||
DETECTOR_NAMES = ["XAI", "XAi", "Xai"]
|
||||
DETECTOR = "XAI"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "xaiChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "xaiResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "xaiAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "xaiNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "xaiDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "xaiRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "xaiLimited.txt"),
|
||||
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "xaiNoContext.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "xaiNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "xaiUnknown.txt"),
|
||||
}
|
||||
KEY_REGEX = re.compile(r"\bxai-[A-Za-z0-9_-]{20,}\b")
|
||||
MODELS_URL = "https://api.x.ai/v1/models"
|
||||
CHAT_URL = "https://api.x.ai/v1/chat/completions"
|
||||
CHAT_MODEL_PRIORITY = (
|
||||
"grok-4.6",
|
||||
"grok-4.5",
|
||||
"grok-4.3",
|
||||
"grok-4.20-0309-reasoning",
|
||||
"grok-4.20-0309-non-reasoning",
|
||||
)
|
||||
NO_BALANCE_MARKERS = (
|
||||
"quota",
|
||||
"billing",
|
||||
"balance",
|
||||
"credit",
|
||||
"credits",
|
||||
"payment",
|
||||
"insufficient",
|
||||
"depleted",
|
||||
"spending limit",
|
||||
"no credits",
|
||||
"used all available credits",
|
||||
"doesn't have any credits",
|
||||
)
|
||||
DEAD_MARKERS = ("incorrect api key", "invalid api key", "api key provided", "invalid-argument")
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def iter_candidate_decisions(input_file, plain_files):
|
||||
seen_plain = set()
|
||||
for item in iter_findings(input_file, DETECTOR_NAMES):
|
||||
key = item.get("credential_secret_text") or item["raw"]
|
||||
if key:
|
||||
yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key))
|
||||
for item in read_plain_keys(plain_files, KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key not in seen_plain:
|
||||
seen_plain.add(key)
|
||||
yield key, item["source"], {}, True
|
||||
|
||||
|
||||
def extract_candidates(input_file, plain_files):
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files):
|
||||
if valid_format:
|
||||
yield key, source, finding
|
||||
|
||||
|
||||
def check_key(key, proxy, timeout):
|
||||
try:
|
||||
response = requests.get(MODELS_URL, headers={"Authorization": f"Bearer {key}", "Accept": "application/json"}, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc)}
|
||||
if response.status_code == 200:
|
||||
data = response.json() if response.text else {}
|
||||
models = [item.get("id") for item in data.get("data", []) if isinstance(item, dict) and item.get("id")]
|
||||
model_inventory = sorted(set(models))
|
||||
model = choose_chat_model(models)
|
||||
probe = probe_chat_completion(key, model, proxy, timeout)
|
||||
if probe.get("status") != "GENERATION_OK":
|
||||
return {
|
||||
"status": probe.get("status") or "UNKNOWN",
|
||||
"message": probe.get("message", ""),
|
||||
"model_count": len(models),
|
||||
"models": models[:20],
|
||||
"model_inventory": model_inventory,
|
||||
"llm_probe_status": probe.get("status"),
|
||||
"llm_probe_model": probe.get("model", model),
|
||||
"llm_probe_http_status": probe.get("http_status"),
|
||||
}
|
||||
return {
|
||||
"status": "VALID", "message": f"chat ping ok; model={model}; models={len(models)}",
|
||||
"model_count": len(models), "models": models[:20], "model_inventory": model_inventory,
|
||||
"llm_probe_status": probe.get("status"), "llm_probe_model": model,
|
||||
}
|
||||
status = classify_xai_response(response)
|
||||
return {"status": status, "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***")}
|
||||
|
||||
|
||||
def classify_xai_response(response):
|
||||
message = request_error_message(response).lower()
|
||||
if any(marker in message for marker in DEAD_MARKERS):
|
||||
return "DEAD"
|
||||
if any(marker in message for marker in NO_BALANCE_MARKERS):
|
||||
return "NO_BALANCE"
|
||||
if response.status_code == 401:
|
||||
return "DEAD"
|
||||
if response.status_code == 403:
|
||||
return "RESTRICTED"
|
||||
if response.status_code == 429:
|
||||
return "LIMITED"
|
||||
return classify_common_http_status(response.status_code)
|
||||
|
||||
|
||||
def choose_chat_model(models):
|
||||
models = [str(model or "") for model in models if model]
|
||||
by_lower = {model.lower(): model for model in models}
|
||||
for model in CHAT_MODEL_PRIORITY:
|
||||
if model.lower() in by_lower:
|
||||
return by_lower[model.lower()]
|
||||
for model in models:
|
||||
if "grok" in model.lower():
|
||||
return model
|
||||
return models[0] if models else ""
|
||||
|
||||
|
||||
def probe_chat_completion(key, model, proxy, timeout):
|
||||
if not model:
|
||||
return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""}
|
||||
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1}
|
||||
try:
|
||||
response = requests.post(CHAT_URL, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {"status": "NETWORK", "message": str(exc), "model": model}
|
||||
if response.status_code == 200:
|
||||
return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model}
|
||||
return {"status": classify_xai_response(response), "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***"), "model": model}
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
extra = ",".join(result.get("models") or [])[:500] if status == "VALID" else source
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="xAI key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = set()
|
||||
if args.retry_network: retry_statuses.add("NETWORK")
|
||||
if args.retry_limited: retry_statuses.add("LIMITED")
|
||||
if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
|
||||
if args.retry_restricted: retry_statuses.add("RESTRICTED")
|
||||
if args.retry_no_balance: retry_statuses.add("NO_BALANCE")
|
||||
if args.retry_valid: retry_statuses.add("VALID")
|
||||
processed = skipped = 0
|
||||
print("--- xAI key checker ---")
|
||||
postgres_mode = keycheck_input_mode() == "postgres"
|
||||
for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain):
|
||||
if not valid_format and not postgres_mode:
|
||||
skipped += 1
|
||||
continue
|
||||
if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] xAI candidate {mask_secret(key)} from {source}")
|
||||
result = (
|
||||
check_key(key, next(proxy_cycler) if proxy_cycler else None, args.timeout)
|
||||
if valid_format else
|
||||
{"status": "NO_CONTEXT", "message": "candidate does not match canonical xAI token format"}
|
||||
)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,508 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
|
||||
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from keycheck_common import (
|
||||
combined_provider_routing_hint,
|
||||
commit_status_transaction,
|
||||
default_input_file,
|
||||
default_proxy_file,
|
||||
ensure_output_files,
|
||||
iter_findings,
|
||||
keycheck_input_mode,
|
||||
load_checked_statuses,
|
||||
load_known_keys,
|
||||
load_proxies,
|
||||
mask_secret,
|
||||
provider_routing_database_failed,
|
||||
read_plain_keys,
|
||||
record_validation_result,
|
||||
recover_status_transaction,
|
||||
require_provider_authority,
|
||||
service_output_dir,
|
||||
should_skip_key,
|
||||
write_keycheck_event,
|
||||
)
|
||||
from keycheckers.provider_resolution import (
|
||||
AMBIGUOUS_GENERIC_SK_HINT,
|
||||
AMBIGUOUS_QWEN_DEEPSEEK_HINT,
|
||||
resolve_provider_key,
|
||||
)
|
||||
|
||||
|
||||
SERVICE = "zai"
|
||||
DETECTOR = "ZaiGLM"
|
||||
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
|
||||
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
|
||||
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
|
||||
|
||||
CHECKED_FILE = os.path.join(OUTPUT_DIR, "zaiChecked.txt")
|
||||
RESULTS_FILE = os.path.join(OUTPUT_DIR, "zaiResults.jsonl")
|
||||
STATUS_FILES = {
|
||||
"VALID": os.path.join(OUTPUT_DIR, "zaiAlive.txt"),
|
||||
"NO_BALANCE": os.path.join(OUTPUT_DIR, "zaiNoBalance.txt"),
|
||||
"DEAD": os.path.join(OUTPUT_DIR, "zaiDead.txt"),
|
||||
"RESTRICTED": os.path.join(OUTPUT_DIR, "zaiRestricted.txt"),
|
||||
"LIMITED": os.path.join(OUTPUT_DIR, "zaiLimited.txt"),
|
||||
"NETWORK": os.path.join(OUTPUT_DIR, "zaiNetwork.txt"),
|
||||
"UNKNOWN": os.path.join(OUTPUT_DIR, "zaiUnknown.txt"),
|
||||
}
|
||||
|
||||
DEFAULT_BASE_URLS = (
|
||||
"https://api.z.ai/api/paas/v4",
|
||||
"https://open.bigmodel.cn/api/paas/v4",
|
||||
)
|
||||
ZAI_KEY_REGEX = re.compile(
|
||||
r"(?<![A-Za-z0-9_.-])(?:"
|
||||
r"(?:zai|sk)-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|"
|
||||
r"[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}"
|
||||
r")(?![A-Za-z0-9_.-])"
|
||||
)
|
||||
ZAI_DOTTED_KEY_REGEX = re.compile(r"^[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}$")
|
||||
ZAI_CONTEXT_REGEX = re.compile(
|
||||
r"(?:ZAI_API_KEY|GLM_API_KEY|ZHIPUAI_API_KEY|BIGMODEL_API_KEY|api\.z\.ai|"
|
||||
r"open\.bigmodel\.cn|zhipuai|chatglm)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
FOREIGN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
|
||||
AMBIGUOUS_HINTS = {AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT}
|
||||
AUTH_FAILURE_CODES = {"1000", "1001", "1003"}
|
||||
AUTHENTICATED_NO_BALANCE_CODES = {"1113"}
|
||||
AUTHENTICATED_LIMIT_CODES = {"1302", "1308", "1309", "1310", "1311"}
|
||||
AUTHENTICATED_RESTRICTED_CODES = {"1005", "1220"}
|
||||
PROBE_MODEL = "glm-5.2"
|
||||
|
||||
|
||||
def normalize_base_url(value):
|
||||
return str(value or "").strip().rstrip("/")
|
||||
|
||||
|
||||
def split_csv(value):
|
||||
if not value:
|
||||
return []
|
||||
values = value.split(",") if isinstance(value, str) else value
|
||||
return [str(item).strip() for item in values if str(item).strip()]
|
||||
|
||||
|
||||
def unique_ordered(values):
|
||||
output = []
|
||||
seen = set()
|
||||
for value in values:
|
||||
normalized = normalize_base_url(value)
|
||||
if normalized and normalized not in seen:
|
||||
seen.add(normalized)
|
||||
output.append(normalized)
|
||||
return output
|
||||
|
||||
|
||||
def base_urls_from_environment(extra=None, include_defaults=True):
|
||||
configured = []
|
||||
for value in extra or ():
|
||||
configured.extend(split_csv(value))
|
||||
configured.extend(split_csv(os.getenv("ZAI_BASE_URLS") or os.getenv("ZHIPU_BASE_URLS")))
|
||||
defaults = DEFAULT_BASE_URLS if include_defaults else ()
|
||||
return unique_ordered([*configured, *defaults])
|
||||
|
||||
|
||||
def endpoint_label(base_url):
|
||||
parsed = urlparse(base_url)
|
||||
return parsed.netloc or base_url
|
||||
|
||||
|
||||
def key_from_text(*values):
|
||||
for value in values:
|
||||
match = ZAI_KEY_REGEX.search(str(value or ""))
|
||||
if match:
|
||||
return match.group(0)
|
||||
return ""
|
||||
|
||||
|
||||
def key_rejection_reason(key):
|
||||
value = str(key or "")
|
||||
try:
|
||||
encoded = value.encode("utf-8", errors="strict")
|
||||
except UnicodeEncodeError:
|
||||
return "candidate is not valid UTF-8"
|
||||
if len(encoded) > 512:
|
||||
return "candidate exceeds the 512-byte key limit"
|
||||
if value.startswith(FOREIGN_KEY_PREFIXES):
|
||||
return "candidate has a foreign provider prefix"
|
||||
if not ZAI_KEY_REGEX.fullmatch(value):
|
||||
return "candidate does not match a bounded ZAI key format"
|
||||
return ""
|
||||
|
||||
|
||||
def finding_detector_names(finding):
|
||||
if not isinstance(finding, dict):
|
||||
return set()
|
||||
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
|
||||
return {
|
||||
name for name in (
|
||||
str(finding.get("DetectorName") or finding.get("DetectorType") or "").strip().lower(),
|
||||
str(extra.get("name") or "").strip().lower(),
|
||||
) if name
|
||||
}
|
||||
|
||||
|
||||
def finding_provider_routing_hint(finding, key=""):
|
||||
if not isinstance(finding, dict):
|
||||
return "zai" if ZAI_DOTTED_KEY_REGEX.fullmatch(str(key or "")) else ""
|
||||
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
|
||||
persisted = str(context.get("provider_hint") or "").strip().lower()
|
||||
if persisted:
|
||||
return persisted
|
||||
if "zaiglm" in finding_detector_names(finding) or ZAI_DOTTED_KEY_REGEX.fullmatch(str(key or "")):
|
||||
return "zai"
|
||||
text = "\n".join(str(context.get(name) or "") for name in ("nearby", "file"))
|
||||
return "zai" if ZAI_CONTEXT_REGEX.search(text) else ""
|
||||
|
||||
|
||||
def iter_candidate_decisions(input_file, plain_files, trusted_retry_files=None):
|
||||
detectors = [
|
||||
"ZaiGLM", "zaiglm", "CustomRegex", "QwenDashScope", "Qwen_DashScope",
|
||||
"Qwen", "DashScope", "DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key",
|
||||
"KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi",
|
||||
]
|
||||
seen = set()
|
||||
for item in iter_findings(input_file, detectors):
|
||||
finding = item.get("finding") or {}
|
||||
key = item.get("credential_secret_text") or key_from_text(
|
||||
item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2"),
|
||||
)
|
||||
if not key or key_rejection_reason(key):
|
||||
continue
|
||||
local_hint = finding_provider_routing_hint(finding, key)
|
||||
hint = combined_provider_routing_hint(key, local_hint)
|
||||
if provider_routing_database_failed():
|
||||
raise RuntimeError("provider routing evidence lookup failed closed")
|
||||
seen.add(key)
|
||||
yield key, item.get("source") or input_file, finding, hint
|
||||
|
||||
owned_retry_paths = {
|
||||
os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values()
|
||||
}
|
||||
retry_paths = [
|
||||
path for path in trusted_retry_files or ()
|
||||
if os.path.normcase(os.path.abspath(path)) in owned_retry_paths
|
||||
]
|
||||
for item in read_plain_keys([*plain_files, *retry_paths], ZAI_KEY_REGEX):
|
||||
key = item["key"]
|
||||
if key in seen or key_rejection_reason(key):
|
||||
continue
|
||||
yield key, item["source"], {}, "zai"
|
||||
|
||||
|
||||
def redact_text(value, key):
|
||||
text = str(value or "")[:1000]
|
||||
if key:
|
||||
text = text.replace(key, "***REDACTED***")
|
||||
return ZAI_KEY_REGEX.sub("***REDACTED***", text)
|
||||
|
||||
|
||||
def parse_error(response, key):
|
||||
try:
|
||||
payload = response.json()
|
||||
except ValueError:
|
||||
payload = {}
|
||||
error = payload.get("error") if isinstance(payload, dict) else {}
|
||||
if not isinstance(error, dict):
|
||||
error = {}
|
||||
return {
|
||||
"http_status": int(response.status_code),
|
||||
"code": str(error.get("code") or (payload.get("code") if isinstance(payload, dict) else "") or ""),
|
||||
"message": redact_text(
|
||||
error.get("message") or error.get("msg") or (
|
||||
payload.get("message") or payload.get("msg") if isinstance(payload, dict) else ""
|
||||
) or response.text[:500],
|
||||
key,
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def classify_error(error):
|
||||
http_status = int(error.get("http_status") or 0)
|
||||
code = str(error.get("code") or "")
|
||||
message = str(error.get("message") or "").lower()
|
||||
if http_status == 402 or any(marker in message for marker in (
|
||||
"insufficient balance", "balance is insufficient", "no balance", "account balance",
|
||||
"recharge", "payment required", "billing arrears", "credit balance",
|
||||
)):
|
||||
return "NO_BALANCE", True
|
||||
if code in AUTHENTICATED_NO_BALANCE_CODES:
|
||||
return "NO_BALANCE", True
|
||||
if any(marker in message for marker in (
|
||||
"quota", "rate limit", "rate-limit", "too many requests", "resource exhausted",
|
||||
"concurrency limit", "usage limit",
|
||||
)):
|
||||
return "LIMITED", True
|
||||
if code in AUTHENTICATED_LIMIT_CODES:
|
||||
return "LIMITED", True
|
||||
if any(marker in message for marker in (
|
||||
"permission denied", "access denied", "not authorized for", "model access", "forbidden",
|
||||
)):
|
||||
return "RESTRICTED", True
|
||||
if code in AUTHENTICATED_RESTRICTED_CODES:
|
||||
return "RESTRICTED", True
|
||||
if http_status == 401 or code in AUTH_FAILURE_CODES:
|
||||
return "DEAD", False
|
||||
if 500 <= http_status <= 599 or code in {"1200", "1230", "1234", "1305"}:
|
||||
return "NETWORK", False
|
||||
if http_status == 429:
|
||||
return "LIMITED", False
|
||||
if http_status == 403:
|
||||
return "RESTRICTED", False
|
||||
return "UNKNOWN", False
|
||||
|
||||
|
||||
def probe_chat_completion(key, base_url, model, proxy, timeout, debug=False):
|
||||
if not model:
|
||||
return {
|
||||
"status": "UNKNOWN", "model": "",
|
||||
"message": "no chat-capable model returned by /models",
|
||||
}
|
||||
url = f"{normalize_base_url(base_url)}/chat/completions"
|
||||
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": 1,
|
||||
"stream": False,
|
||||
}
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {
|
||||
"status": "NETWORK", "model": model,
|
||||
"message": redact_text(exc, key),
|
||||
}
|
||||
if debug:
|
||||
print(
|
||||
f" DEBUG {endpoint_label(base_url)} chat probe {model}: HTTP {response.status_code}: "
|
||||
f"{redact_text(response.text[:500], key)}"
|
||||
)
|
||||
if response.status_code == 200:
|
||||
try:
|
||||
response_payload = response.json()
|
||||
except ValueError as exc:
|
||||
return {
|
||||
"status": "UNKNOWN", "model": model, "http_status": response.status_code,
|
||||
"message": f"invalid chat completion response: {exc}",
|
||||
}
|
||||
if isinstance(response_payload, dict) and response_payload.get("choices"):
|
||||
return {
|
||||
"status": "GENERATION_OK", "model": model,
|
||||
"http_status": response.status_code, "message": "chat completion accepted",
|
||||
}
|
||||
error = parse_error(response, key)
|
||||
status, authenticated = classify_error(error)
|
||||
return {
|
||||
"status": status, "authenticated": authenticated, "model": model,
|
||||
"http_status": response.status_code, "business_code": error.get("code") or "",
|
||||
"error": error, "message": error.get("message") or "",
|
||||
}
|
||||
|
||||
|
||||
def check_base_url(key, base_url, proxy, timeout, debug=False):
|
||||
url = f"{normalize_base_url(base_url)}/models"
|
||||
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
||||
try:
|
||||
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
|
||||
except requests.RequestException as exc:
|
||||
return {
|
||||
"status": "NETWORK", "base_url": base_url,
|
||||
"region": endpoint_label(base_url), "message": redact_text(exc, key),
|
||||
}
|
||||
if debug:
|
||||
print(
|
||||
f" DEBUG {endpoint_label(base_url)} /models: HTTP {response.status_code}: "
|
||||
f"{redact_text(response.text[:500], key)}"
|
||||
)
|
||||
if response.status_code == 200:
|
||||
try:
|
||||
payload = response.json()
|
||||
data = payload.get("data") if isinstance(payload, dict) else None
|
||||
if not isinstance(data, list):
|
||||
raise ValueError("missing data model list")
|
||||
models = sorted({
|
||||
str(item.get("id") or item.get("name") or "")
|
||||
for item in data if isinstance(item, dict) and (item.get("id") or item.get("name"))
|
||||
})
|
||||
except (TypeError, ValueError, json.JSONDecodeError) as exc:
|
||||
return {
|
||||
"status": "UNKNOWN", "base_url": base_url,
|
||||
"region": endpoint_label(base_url), "message": f"invalid models response: {exc}",
|
||||
}
|
||||
probe_model = PROBE_MODEL
|
||||
probe = probe_chat_completion(key, base_url, probe_model, proxy, timeout, debug)
|
||||
probe_status = probe.get("status") or "UNKNOWN"
|
||||
if probe_status == "GENERATION_OK":
|
||||
status = "VALID"
|
||||
elif probe_status in {"NO_BALANCE", "LIMITED", "RESTRICTED", "NETWORK"}:
|
||||
status = probe_status
|
||||
elif probe_status == "DEAD":
|
||||
status = "RESTRICTED"
|
||||
else:
|
||||
status = "UNKNOWN"
|
||||
probe_message = probe.get("message") or json.dumps(probe.get("error") or {}, ensure_ascii=False)
|
||||
message = (
|
||||
f"chat probe ok; model={probe_model}; models={len(data)}"
|
||||
if status == "VALID"
|
||||
else f"models authenticated; generation_probe={probe_status}; model={probe_model}; "
|
||||
f"models={len(data)}; {probe_message}"
|
||||
)
|
||||
return {
|
||||
"status": status, "authenticated": True, "base_url": base_url,
|
||||
"region": endpoint_label(base_url), "model_count": len(data),
|
||||
"models": models[:30], "llm_probe_status": probe_status,
|
||||
"llm_probe_model": probe.get("model") or probe_model,
|
||||
"llm_probe_http_status": probe.get("http_status"),
|
||||
"business_code": probe.get("business_code") or "",
|
||||
"probe": probe, "error": probe.get("error") or {},
|
||||
"message": message[:1000],
|
||||
}
|
||||
error = parse_error(response, key)
|
||||
status, authenticated = classify_error(error)
|
||||
return {
|
||||
"status": status, "authenticated": authenticated, "base_url": base_url,
|
||||
"region": endpoint_label(base_url), "http_status": response.status_code,
|
||||
"business_code": error.get("code") or "", "error": error,
|
||||
"message": error.get("message") or "",
|
||||
}
|
||||
|
||||
|
||||
def check_key(key, base_urls, proxy, timeout, debug=False):
|
||||
rejection = key_rejection_reason(key)
|
||||
if rejection:
|
||||
return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True}
|
||||
attempts = []
|
||||
for base_url in base_urls:
|
||||
result = check_base_url(key, base_url, proxy, timeout, debug)
|
||||
attempts.append(result)
|
||||
if result.get("authenticated"):
|
||||
return {**result, "attempts": attempts}
|
||||
statuses = [attempt.get("status") for attempt in attempts]
|
||||
status = next(
|
||||
(candidate for candidate in ("NETWORK", "LIMITED", "RESTRICTED", "UNKNOWN", "DEAD") if candidate in statuses),
|
||||
"UNKNOWN",
|
||||
)
|
||||
selected = next((attempt for attempt in attempts if attempt.get("status") == status), {})
|
||||
return {**selected, "status": status, "attempts": attempts}
|
||||
|
||||
|
||||
def ensure_files():
|
||||
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
|
||||
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
|
||||
|
||||
|
||||
def write_result(key, result, source, finding):
|
||||
status = result.get("status") or "UNKNOWN"
|
||||
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
|
||||
commit_status_transaction(
|
||||
CHECKED_FILE, STATUS_FILES, key, status,
|
||||
result.get("message", ""), result.get("region") or source,
|
||||
)
|
||||
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
|
||||
|
||||
|
||||
def retry_statuses_from_args(args):
|
||||
statuses = set()
|
||||
for enabled, status in (
|
||||
(args.retry_network, "NETWORK"),
|
||||
(args.retry_limited, "LIMITED"),
|
||||
(args.retry_unknown, "UNKNOWN"),
|
||||
(args.retry_restricted, "RESTRICTED"),
|
||||
(args.retry_no_balance, "NO_BALANCE"),
|
||||
(args.retry_valid, "VALID"),
|
||||
):
|
||||
if enabled:
|
||||
statuses.add(status)
|
||||
return statuses
|
||||
|
||||
|
||||
def retry_input_files_from_args(args):
|
||||
if args.recheck_all:
|
||||
statuses = list(STATUS_FILES)
|
||||
else:
|
||||
statuses = list(retry_statuses_from_args(args))
|
||||
return [STATUS_FILES[status] for status in statuses]
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="ZAI / Zhipu GLM key checker")
|
||||
parser.add_argument("--input", default=INPUT_FILE)
|
||||
parser.add_argument("--plain", action="append", default=[])
|
||||
parser.add_argument("--proxy-file", default=PROXY_FILE)
|
||||
parser.add_argument("--timeout", type=int, default=15)
|
||||
parser.add_argument("--max-keys", type=int, default=0)
|
||||
parser.add_argument("--base-url", action="append", default=[])
|
||||
parser.add_argument("--no-default-base-urls", action="store_true")
|
||||
parser.add_argument("--retry-network", action="store_true")
|
||||
parser.add_argument("--retry-limited", action="store_true")
|
||||
parser.add_argument("--retry-unknown", action="store_true")
|
||||
parser.add_argument("--retry-restricted", action="store_true")
|
||||
parser.add_argument("--retry-no-balance", action="store_true")
|
||||
parser.add_argument("--retry-valid", action="store_true")
|
||||
parser.add_argument("--recheck-all", action="store_true")
|
||||
parser.add_argument("--debug", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
require_provider_authority(SERVICE)
|
||||
args = parse_args()
|
||||
ensure_files()
|
||||
proxy_cycler = load_proxies(args.proxy_file)
|
||||
checked = load_checked_statuses(CHECKED_FILE)
|
||||
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
|
||||
retry_statuses = retry_statuses_from_args(args)
|
||||
retry_files = retry_input_files_from_args(args)
|
||||
base_urls = base_urls_from_environment(args.base_url, not args.no_default_base_urls)
|
||||
if not base_urls:
|
||||
raise SystemExit("No ZAI base URLs configured")
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
postgres_mode = keycheck_input_mode() == "postgres"
|
||||
for key, source, finding, routing_hint in iter_candidate_decisions(args.input, args.plain, retry_files):
|
||||
is_ambiguous = routing_hint in AMBIGUOUS_HINTS
|
||||
if routing_hint != "zai" and not (postgres_mode and is_ambiguous):
|
||||
skipped += 1
|
||||
continue
|
||||
if should_skip_key(
|
||||
key, checked, known, args, retry_statuses, service=SERVICE,
|
||||
source=source, finding=finding, detector=DETECTOR,
|
||||
):
|
||||
skipped += 1
|
||||
continue
|
||||
if args.max_keys and processed >= args.max_keys:
|
||||
break
|
||||
processed += 1
|
||||
print(f"\n[{processed}] ZAI candidate {mask_secret(key)} from {source}")
|
||||
proxy = next(proxy_cycler) if proxy_cycler else None
|
||||
if is_ambiguous:
|
||||
result = resolve_provider_key(
|
||||
key, finding, proxy, args.timeout, args.debug,
|
||||
hint=routing_hint, origin_service=SERVICE,
|
||||
)
|
||||
else:
|
||||
result = check_key(key, base_urls, proxy, args.timeout, args.debug)
|
||||
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
|
||||
write_result(key, result, source, finding)
|
||||
known.add(key)
|
||||
checked[key] = result["status"]
|
||||
|
||||
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,768 @@
|
||||
import hashlib
|
||||
import hmac
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import socket
|
||||
import stat
|
||||
|
||||
from db_backend import parse_postgres_url
|
||||
from process_identity import verify_retained_process
|
||||
from runtime_security import (
|
||||
canonical_path,
|
||||
private_file_ready,
|
||||
read_private_json,
|
||||
reject_reparse_components,
|
||||
require_trusted_native_executable,
|
||||
sha256_file,
|
||||
)
|
||||
|
||||
|
||||
CODE_MANIFEST_SCHEMA = 5
|
||||
APPLICATION_IMPORT_SUFFIXES = ('.py', '.pyw', '.pyc', '.pyd')
|
||||
CONTROL_SCHEMA = 1
|
||||
DISCOVERY_PRODUCER_ROLE = 'discovery-producer'
|
||||
DISCOVERY_PRODUCER_SOURCES = ('gitlab', 'dockerhub', 'huggingface')
|
||||
PHASE_INACTIVE = 'INACTIVE'
|
||||
PHASE_ACTIVATING = 'ACTIVATING'
|
||||
PHASE_ACTIVE = 'ACTIVE'
|
||||
PHASE_STOPPING = 'STOPPING'
|
||||
PHASE_FAILED_HOLD = 'FAILED_HOLD'
|
||||
LIFECYCLE_PHASES = {
|
||||
PHASE_INACTIVE,
|
||||
PHASE_ACTIVATING,
|
||||
PHASE_ACTIVE,
|
||||
PHASE_STOPPING,
|
||||
PHASE_FAILED_HOLD,
|
||||
}
|
||||
|
||||
# These files collectively decide process ownership, database authority, and
|
||||
# what data may be launched or persisted by the supervisor.
|
||||
CODE_AUTHORITY_FILES = (
|
||||
'owned_process.py',
|
||||
'supervisor.py',
|
||||
'supervisor_instance.py',
|
||||
'console_runner.py',
|
||||
'scanner.py',
|
||||
'docker_shadow.py',
|
||||
'keycheck_runner.py',
|
||||
'dashboard.py',
|
||||
'postgres_runtime.py',
|
||||
'process_identity.py',
|
||||
'runtime_security.py',
|
||||
'scanner_db.py',
|
||||
'db_backend.py',
|
||||
'result_spool.py',
|
||||
'janitor.py',
|
||||
'result_bundle.py',
|
||||
'result_ingester.py',
|
||||
'jsonl_projector.py',
|
||||
'admin_api.py',
|
||||
'worker_api.py',
|
||||
'worker_assignment.py',
|
||||
'worker_package.py',
|
||||
'scan_execution.py',
|
||||
'keycheck_candidates.py',
|
||||
'paths.py',
|
||||
'target_identity.py',
|
||||
'lifecycle_authority.py',
|
||||
'audit_github_tokens.py',
|
||||
'sync_alive_github_tokens.py',
|
||||
'child_bootstrap.py',
|
||||
'runtime_bootstrap.py',
|
||||
'runtime_document.py',
|
||||
'capacity_model.py',
|
||||
'runtime_document_io.py',
|
||||
'managed_files.py',
|
||||
'host_agent_client.py',
|
||||
'host_agent_protocol.py',
|
||||
'host_agent_reconcile.py',
|
||||
'host_agent_server.py',
|
||||
'host_agent_apply.py',
|
||||
'host_agent_lifecycle.py',
|
||||
'host_agent_runtime.py',
|
||||
'host_agent_state.py',
|
||||
'worker_contracts.py',
|
||||
'worker_assignment_runner.py',
|
||||
)
|
||||
REMOTE_WORKER_CODE_AUTHORITY_FILES = (
|
||||
'db_backend.py',
|
||||
'docker_depth_experiment.py',
|
||||
'janitor.py',
|
||||
'keycheck_candidates.py',
|
||||
'lifecycle_authority.py',
|
||||
'owned_process.py',
|
||||
'paths.py',
|
||||
'process_identity.py',
|
||||
'query_policy.py',
|
||||
'remote_worker_bootstrap.py',
|
||||
'remote_worker_client.py',
|
||||
'result_bundle.py',
|
||||
'result_spool.py',
|
||||
'runtime_security.py',
|
||||
'scan_execution.py',
|
||||
'scanner.py',
|
||||
'scanner_db.py',
|
||||
'supervisor_instance.py',
|
||||
'target_identity.py',
|
||||
'worker_contracts.py',
|
||||
'worker_assignment_runner.py',
|
||||
'worker_cli.py',
|
||||
'worker_local_state.py',
|
||||
'worker_supervisor.py',
|
||||
'worker_package.py',
|
||||
)
|
||||
EXTERNAL_CODE_AUTHORITY_FILES = (
|
||||
'../runtime/check-openrouter-keys.ps1',
|
||||
'../start_core_runtime.ps1',
|
||||
'../start_runtime.ps1',
|
||||
'../stop_runtime.ps1',
|
||||
) if os.name == 'nt' else ()
|
||||
# The launchers execute before a manifest can be captured. First-launch trust
|
||||
# therefore requires offline ACL hardening; manifests only detect later drift.
|
||||
TRUFFLEHOG_MANIFEST_NAME = 'trufflehog'
|
||||
GIT_MANIFEST_NAME = 'git'
|
||||
|
||||
CHILD_INSTANCE_FILE_ENV = 'TRUF_SUPERVISOR_INSTANCE_FILE'
|
||||
CHILD_INSTANCE_ID_ENV = 'TRUF_SUPERVISOR_INSTANCE_ID'
|
||||
CHILD_TOKEN_ENV = 'TRUF_SUPERVISOR_TOKEN'
|
||||
CHILD_CONFIG_HASH_ENV = 'TRUF_SUPERVISOR_CONFIG_SHA256'
|
||||
CHILD_SCRIPT_HASH_ENV = 'TRUF_SUPERVISOR_SHA256'
|
||||
CHILD_MANIFEST_HASH_ENV = 'TRUF_SUPERVISOR_CODE_MANIFEST_SHA256'
|
||||
CHILD_DSN_HASH_ENV = 'TRUF_SUPERVISOR_DSN_SHA256'
|
||||
CHILD_KIND_ENV = 'TRUF_SUPERVISOR_CHILD_KIND'
|
||||
PRIVATE_CHILD_ENV_KEYS = (
|
||||
CHILD_INSTANCE_FILE_ENV,
|
||||
CHILD_INSTANCE_ID_ENV,
|
||||
CHILD_TOKEN_ENV,
|
||||
CHILD_CONFIG_HASH_ENV,
|
||||
CHILD_SCRIPT_HASH_ENV,
|
||||
CHILD_MANIFEST_HASH_ENV,
|
||||
CHILD_DSN_HASH_ENV,
|
||||
CHILD_KIND_ENV,
|
||||
'SCANNER_SUPERVISED',
|
||||
'TRUF_MANAGED_POSTGRES_DSN',
|
||||
'SCANNER_DB_URL',
|
||||
'DATABASE_URL',
|
||||
'SCANNER_DASHBOARD_DB_URL',
|
||||
'KEYCHECK_DB_URL',
|
||||
'TRUF_DASHBOARD_CANONICAL_LAUNCH',
|
||||
'TRUF_DASHBOARD_HOST',
|
||||
)
|
||||
_LIBPQ_PRIVATE_ENV_KEYS = frozenset({
|
||||
'PGPASSWORD',
|
||||
'PGUSER',
|
||||
'PGDATABASE',
|
||||
'PGHOST',
|
||||
'PGHOSTADDR',
|
||||
'PGPORT',
|
||||
'PGSERVICE',
|
||||
'PGSERVICEFILE',
|
||||
'PGPASSFILE',
|
||||
'PGOPTIONS',
|
||||
'PGSSLMODE',
|
||||
'PGSSLKEY',
|
||||
'PGSSLCERT',
|
||||
'PGSSLROOTCERT',
|
||||
})
|
||||
_PRIVATE_EXTERNAL_ENV_KEYS = frozenset(PRIVATE_CHILD_ENV_KEYS) | _LIBPQ_PRIVATE_ENV_KEYS
|
||||
|
||||
|
||||
class LifecycleAuthorityError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def _manifest_payload(manifest):
|
||||
return json.dumps(
|
||||
manifest,
|
||||
ensure_ascii=True,
|
||||
sort_keys=True,
|
||||
separators=(',', ':'),
|
||||
).encode('utf-8')
|
||||
|
||||
|
||||
def code_manifest_sha256(manifest):
|
||||
return hashlib.sha256(_manifest_payload(manifest)).hexdigest()
|
||||
|
||||
|
||||
def _is_reparse_point(path):
|
||||
details = os.lstat(path)
|
||||
if stat.S_ISLNK(details.st_mode):
|
||||
return True
|
||||
attributes = getattr(details, 'st_file_attributes', 0)
|
||||
reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0)
|
||||
return bool(attributes & reparse_attribute) or getattr(os.path, 'isjunction', lambda _path: False)(path)
|
||||
|
||||
|
||||
def _application_root(app_dir=None):
|
||||
candidate = os.path.abspath(os.fspath(app_dir or os.path.dirname(os.path.abspath(__file__))))
|
||||
try:
|
||||
details = os.lstat(candidate)
|
||||
if _is_reparse_point(candidate):
|
||||
raise LifecycleAuthorityError(f'application root reparse point is forbidden: {candidate}')
|
||||
except OSError as exc:
|
||||
raise LifecycleAuthorityError(f'application root is unavailable: {candidate}') from exc
|
||||
if not stat.S_ISDIR(details.st_mode):
|
||||
raise LifecycleAuthorityError(f'application root is not a directory: {candidate}')
|
||||
return canonical_path(candidate)
|
||||
|
||||
|
||||
def _external_authority_expected_path(root, name):
|
||||
if name not in EXTERNAL_CODE_AUTHORITY_FILES:
|
||||
raise LifecycleAuthorityError(f'code authority path is not an allowed external: {name}')
|
||||
project_root = os.path.normcase(os.path.abspath(os.path.dirname(root)))
|
||||
candidate = os.path.normcase(os.path.abspath(os.path.join(root, *name.split('/'))))
|
||||
try:
|
||||
contained = candidate != project_root and os.path.commonpath((project_root, candidate)) == project_root
|
||||
except ValueError:
|
||||
contained = False
|
||||
if not contained:
|
||||
raise LifecycleAuthorityError(f'external code authority path escapes the project root: {name}')
|
||||
return candidate
|
||||
|
||||
|
||||
def _require_external_authority_file(root, name):
|
||||
path = _external_authority_expected_path(root, name)
|
||||
try:
|
||||
reject_reparse_components(path)
|
||||
except OSError as exc:
|
||||
raise LifecycleAuthorityError(f'external code authority reparse point is forbidden: {name}') from exc
|
||||
try:
|
||||
details = os.stat(path, follow_symlinks=False)
|
||||
except OSError as exc:
|
||||
raise LifecycleAuthorityError(f'required external code authority file is absent: {name}') from exc
|
||||
if not stat.S_ISREG(details.st_mode):
|
||||
raise LifecycleAuthorityError(f'external code authority file is not regular: {name}')
|
||||
if canonical_path(path) != path:
|
||||
raise LifecycleAuthorityError(f'external code authority path is not exact: {name}')
|
||||
return path
|
||||
|
||||
|
||||
def _application_code_files(root):
|
||||
"""Return the exact importable application code surface."""
|
||||
names = []
|
||||
|
||||
def raise_walk_error(exc):
|
||||
raise LifecycleAuthorityError(f'unable to inspect the application root: {exc}') from exc
|
||||
|
||||
for current, directories, files in os.walk(root, followlinks=False, onerror=raise_walk_error):
|
||||
for name in directories:
|
||||
candidate = os.path.join(current, name)
|
||||
try:
|
||||
linked = _is_reparse_point(candidate)
|
||||
except OSError as exc:
|
||||
raise LifecycleAuthorityError(f'unable to inspect application directory: {candidate}') from exc
|
||||
if linked:
|
||||
relative = os.path.relpath(candidate, root).replace(os.sep, '/')
|
||||
raise LifecycleAuthorityError(f'application directory reparse point is forbidden: {relative}')
|
||||
relative_current = os.path.relpath(current, root)
|
||||
in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep))
|
||||
suffixes = ('.pyc',) if in_cache else APPLICATION_IMPORT_SUFFIXES
|
||||
for name in files:
|
||||
source_path = os.path.abspath(os.path.join(current, name))
|
||||
try:
|
||||
linked = _is_reparse_point(source_path)
|
||||
except OSError as exc:
|
||||
raise LifecycleAuthorityError(f'unable to inspect application file: {source_path}') from exc
|
||||
if linked:
|
||||
relative = os.path.relpath(source_path, root).replace(os.sep, '/')
|
||||
raise LifecycleAuthorityError(f'application file reparse point is forbidden: {relative}')
|
||||
if not name.lower().endswith(suffixes):
|
||||
continue
|
||||
path = canonical_path(source_path)
|
||||
try:
|
||||
if os.path.commonpath((root, path)) != root:
|
||||
raise LifecycleAuthorityError(f'code authority path escapes the application root: {path}')
|
||||
except ValueError as exc:
|
||||
raise LifecycleAuthorityError(f'code authority path escapes the application root: {path}') from exc
|
||||
names.append(os.path.relpath(source_path, root).replace(os.sep, '/'))
|
||||
return sorted(names)
|
||||
|
||||
|
||||
def code_authority_file_names(app_dir=None, existing_only=False):
|
||||
root = _application_root(app_dir)
|
||||
names = list(_application_code_files(root))
|
||||
for name in (*CODE_AUTHORITY_FILES, *EXTERNAL_CODE_AUTHORITY_FILES):
|
||||
path = canonical_path(os.path.join(root, name))
|
||||
if not existing_only or os.path.isfile(path):
|
||||
names.append(name)
|
||||
return tuple(dict.fromkeys(names))
|
||||
|
||||
|
||||
def resolve_manifest_executable(value, *, name=TRUFFLEHOG_MANIFEST_NAME, app_dir=None):
|
||||
if name not in {TRUFFLEHOG_MANIFEST_NAME, GIT_MANIFEST_NAME}:
|
||||
raise LifecycleAuthorityError('unsupported manifested executable')
|
||||
label = 'TruffleHog' if name == TRUFFLEHOG_MANIFEST_NAME else 'Git'
|
||||
text = str(value or '').strip()
|
||||
if not text:
|
||||
if name == GIT_MANIFEST_NAME:
|
||||
text = 'git'
|
||||
if os.name == 'nt':
|
||||
private_git = os.path.join(os.path.dirname(_application_root(app_dir)), 'runtime', 'git', 'cmd', 'git.exe')
|
||||
try:
|
||||
reject_reparse_components(private_git)
|
||||
except OSError as exc:
|
||||
raise LifecycleAuthorityError('project-private Git path contains a reparse point') from exc
|
||||
text = private_git if os.path.lexists(private_git) else text
|
||||
else:
|
||||
from paths import default_trufflehog_path
|
||||
|
||||
text = default_trufflehog_path()
|
||||
candidate = shutil.which(text) if not os.path.isabs(text) and not any(sep in text for sep in ('/', '\\')) else text
|
||||
if not candidate:
|
||||
raise LifecycleAuthorityError(f'configured {label} executable is unavailable: {text}')
|
||||
if os.name != 'nt':
|
||||
if not os.path.isabs(candidate) or candidate != os.path.normpath(candidate):
|
||||
raise LifecycleAuthorityError(f'configured {label} executable path must be exact and absolute: {candidate}')
|
||||
try:
|
||||
reject_reparse_components(candidate)
|
||||
except (OSError, ValueError) as exc:
|
||||
raise LifecycleAuthorityError(f'configured {label} executable path is unsafe: {candidate}') from exc
|
||||
path = canonical_path(candidate)
|
||||
if not os.path.isfile(path):
|
||||
raise LifecycleAuthorityError(f'configured {label} executable is not a regular file: {path}')
|
||||
return path
|
||||
|
||||
|
||||
def manifest_authority_paths(
|
||||
app_dir=None, trufflehog_path=None, policy_paths=None, existing_only=False, *,
|
||||
git_path=None, include_executables=True,
|
||||
):
|
||||
"""List authority paths; exclude executables when applying private-file policy."""
|
||||
root = _application_root(app_dir)
|
||||
paths = []
|
||||
names = list(code_authority_file_names(root, existing_only=existing_only))
|
||||
# Required externals may never disappear from read-only preflight or an
|
||||
# offline hardening plan, even when optional paths use existing_only.
|
||||
names.extend(name for name in EXTERNAL_CODE_AUTHORITY_FILES if name not in names)
|
||||
for name in names:
|
||||
if name in EXTERNAL_CODE_AUTHORITY_FILES:
|
||||
paths.append(_require_external_authority_file(root, name))
|
||||
continue
|
||||
path = canonical_path(os.path.join(root, *name.split('/')))
|
||||
if os.path.isfile(path):
|
||||
paths.append(path)
|
||||
elif not existing_only:
|
||||
raise LifecycleAuthorityError(f'code authority file is absent or outside the application root: {name}')
|
||||
if include_executables:
|
||||
if trufflehog_path or not existing_only:
|
||||
try:
|
||||
paths.append(resolve_manifest_executable(trufflehog_path))
|
||||
except LifecycleAuthorityError:
|
||||
if not existing_only or trufflehog_path:
|
||||
raise
|
||||
# Git is required even in preflight/offline hardening's existing-only mode.
|
||||
paths.append(resolve_manifest_executable(git_path, name=GIT_MANIFEST_NAME, app_dir=root))
|
||||
for value in policy_paths or ():
|
||||
if not value:
|
||||
continue
|
||||
path = canonical_path(value)
|
||||
if os.path.isfile(path):
|
||||
paths.append(path)
|
||||
elif not existing_only:
|
||||
raise LifecycleAuthorityError(f'configured policy authority file is unavailable: {path}')
|
||||
return tuple(dict.fromkeys(paths))
|
||||
|
||||
|
||||
def build_code_manifest(
|
||||
app_dir=None, trufflehog_path=None, policy_paths=None, *, git_path=None,
|
||||
include_trufflehog=True,
|
||||
):
|
||||
root = _application_root(app_dir)
|
||||
files = {}
|
||||
for name in code_authority_file_names(root):
|
||||
path = (
|
||||
_require_external_authority_file(root, name)
|
||||
if name in EXTERNAL_CODE_AUTHORITY_FILES
|
||||
else canonical_path(os.path.join(root, *name.split('/')))
|
||||
)
|
||||
try:
|
||||
contained = os.path.commonpath((root, path)) == root
|
||||
except ValueError:
|
||||
contained = False
|
||||
explicitly_external = (
|
||||
name in EXTERNAL_CODE_AUTHORITY_FILES
|
||||
and path == _external_authority_expected_path(root, name)
|
||||
)
|
||||
if (not contained and not explicitly_external) or not os.path.isfile(path):
|
||||
raise LifecycleAuthorityError(f'code authority file is absent or outside the application root: {name}')
|
||||
files[name] = {'path': path, 'sha256': sha256_file(path)}
|
||||
git_executable = resolve_manifest_executable(git_path, name=GIT_MANIFEST_NAME, app_dir=root)
|
||||
executables = {
|
||||
GIT_MANIFEST_NAME: {'path': git_executable, 'sha256': sha256_file(git_executable)},
|
||||
}
|
||||
if include_trufflehog:
|
||||
executable = resolve_manifest_executable(trufflehog_path)
|
||||
executables[TRUFFLEHOG_MANIFEST_NAME] = {
|
||||
'path': executable,
|
||||
'sha256': sha256_file(executable),
|
||||
}
|
||||
assets = {}
|
||||
for value in policy_paths or ():
|
||||
if not value:
|
||||
continue
|
||||
path = canonical_path(value)
|
||||
if not os.path.isfile(path):
|
||||
raise LifecycleAuthorityError(f'configured policy authority file is unavailable: {path}')
|
||||
assets[path] = {'path': path, 'sha256': sha256_file(path)}
|
||||
return {
|
||||
'schema': CODE_MANIFEST_SCHEMA,
|
||||
'root': root,
|
||||
'files': files,
|
||||
'executables': executables,
|
||||
'assets': assets,
|
||||
}
|
||||
|
||||
|
||||
def normalize_code_manifest(manifest, *, required_names=None, external_names=None):
|
||||
if not isinstance(manifest, dict) or manifest.get('schema') != CODE_MANIFEST_SCHEMA:
|
||||
raise LifecycleAuthorityError('unsupported code authority manifest schema')
|
||||
root_value = manifest.get('root') or ''
|
||||
root = _application_root(root_value) if root_value else ''
|
||||
values = manifest.get('files')
|
||||
expected_names = set(values) if isinstance(values, dict) else set()
|
||||
external_names = set(
|
||||
EXTERNAL_CODE_AUTHORITY_FILES if external_names is None else external_names
|
||||
)
|
||||
if not external_names <= set(EXTERNAL_CODE_AUTHORITY_FILES):
|
||||
raise LifecycleAuthorityError('code authority manifest has unsupported external files')
|
||||
required_names = set(CODE_AUTHORITY_FILES if required_names is None else required_names)
|
||||
required_names.update(external_names)
|
||||
if not root or not isinstance(values, dict) or not required_names.issubset(expected_names):
|
||||
raise LifecycleAuthorityError('code authority manifest has an incomplete file set')
|
||||
files = {}
|
||||
for name in sorted(expected_names):
|
||||
value = values.get(name)
|
||||
if not isinstance(value, dict):
|
||||
raise LifecycleAuthorityError(f'invalid code authority entry: {name}')
|
||||
if name in external_names:
|
||||
path = os.path.normcase(os.path.abspath(os.fspath(value.get('path') or '')))
|
||||
expected_path = _external_authority_expected_path(root, name)
|
||||
else:
|
||||
path = canonical_path(value.get('path') or '')
|
||||
expected_path = canonical_path(os.path.join(root, *name.split('/')))
|
||||
try:
|
||||
contained = os.path.commonpath((root, expected_path)) == root
|
||||
except ValueError:
|
||||
contained = False
|
||||
if not contained:
|
||||
raise LifecycleAuthorityError(f'code authority path is not an allowed external: {name}')
|
||||
digest = str(value.get('sha256') or '')
|
||||
if path != expected_path:
|
||||
raise LifecycleAuthorityError(f'code authority path mismatch: {name}')
|
||||
if len(digest) != 64 or any(ch not in '0123456789abcdef' for ch in digest):
|
||||
raise LifecycleAuthorityError(f'invalid code authority digest: {name}')
|
||||
files[name] = {'path': path, 'sha256': digest}
|
||||
executables = manifest.get('executables')
|
||||
executable_names = set(executables) if isinstance(executables, dict) else set()
|
||||
if executable_names not in (
|
||||
{GIT_MANIFEST_NAME},
|
||||
{TRUFFLEHOG_MANIFEST_NAME, GIT_MANIFEST_NAME},
|
||||
):
|
||||
raise LifecycleAuthorityError('code authority manifest has an incomplete executable set')
|
||||
normalized_executables = {}
|
||||
for name, label in ((TRUFFLEHOG_MANIFEST_NAME, 'TruffleHog'), (GIT_MANIFEST_NAME, 'Git')):
|
||||
if name not in executables:
|
||||
continue
|
||||
executable = executables[name]
|
||||
if not isinstance(executable, dict):
|
||||
raise LifecycleAuthorityError(f'invalid {label} authority entry')
|
||||
raw_path = executable.get('path')
|
||||
executable_digest = str(executable.get('sha256') or '')
|
||||
if not isinstance(raw_path, str) or not os.path.isabs(raw_path) or '\x00' in raw_path or len(executable_digest) != 64 or any(ch not in '0123456789abcdef' for ch in executable_digest):
|
||||
raise LifecycleAuthorityError(f'invalid {label} authority identity')
|
||||
executable_path = canonical_path(raw_path)
|
||||
if os.name != 'nt' and executable_path != raw_path:
|
||||
raise LifecycleAuthorityError(f'{label} authority path is not exact')
|
||||
normalized_executables[name] = {'path': executable_path, 'sha256': executable_digest}
|
||||
assets_value = manifest.get('assets')
|
||||
if not isinstance(assets_value, dict):
|
||||
raise LifecycleAuthorityError('code authority manifest has an invalid asset set')
|
||||
assets = {}
|
||||
for name in sorted(assets_value):
|
||||
value = assets_value[name]
|
||||
if not isinstance(value, dict):
|
||||
raise LifecycleAuthorityError(f'invalid policy authority entry: {name}')
|
||||
path = canonical_path(value.get('path') or '')
|
||||
digest = str(value.get('sha256') or '')
|
||||
if name != path or not os.path.isabs(path) or len(digest) != 64 or any(ch not in '0123456789abcdef' for ch in digest):
|
||||
raise LifecycleAuthorityError(f'invalid policy authority identity: {name}')
|
||||
assets[name] = {'path': path, 'sha256': digest}
|
||||
return {
|
||||
'schema': CODE_MANIFEST_SCHEMA,
|
||||
'root': root,
|
||||
'files': files,
|
||||
'executables': normalized_executables,
|
||||
'assets': assets,
|
||||
}
|
||||
|
||||
|
||||
def verify_code_manifest(
|
||||
manifest, expected_sha256=None, require_private_acl=False, *,
|
||||
required_names=None, external_names=None,
|
||||
):
|
||||
normalized = normalize_code_manifest(
|
||||
manifest, required_names=required_names, external_names=external_names,
|
||||
)
|
||||
digest = code_manifest_sha256(normalized)
|
||||
if expected_sha256 and not hmac.compare_digest(digest, str(expected_sha256)):
|
||||
raise LifecycleAuthorityError('code authority manifest digest mismatch')
|
||||
for name in (EXTERNAL_CODE_AUTHORITY_FILES if external_names is None else external_names):
|
||||
if normalized['files'][name]['path'] != _require_external_authority_file(normalized['root'], name):
|
||||
raise LifecycleAuthorityError(f'code authority path mismatch: {name}')
|
||||
current_code = set(_application_code_files(normalized['root']))
|
||||
manifested_code = {
|
||||
name for name, value in normalized['files'].items()
|
||||
if name.lower().endswith(APPLICATION_IMPORT_SUFFIXES)
|
||||
and os.path.commonpath((normalized['root'], value['path'])) == normalized['root']
|
||||
}
|
||||
if current_code != manifested_code:
|
||||
added = sorted(current_code - manifested_code)
|
||||
removed = sorted(manifested_code - current_code)
|
||||
detail = added[0] if added else removed[0] if removed else 'unknown'
|
||||
raise LifecycleAuthorityError(f'application code authority file set drifted: {detail}')
|
||||
entries = [(name, value, False) for name, value in normalized['files'].items()] + [
|
||||
(f'executable:{name}', value, True) for name, value in normalized['executables'].items()
|
||||
] + [(f'asset:{name}', value, False) for name, value in normalized['assets'].items()]
|
||||
for name, value, native in entries:
|
||||
if require_private_acl:
|
||||
if native and os.name != 'nt':
|
||||
try:
|
||||
require_trusted_native_executable(value['path'])
|
||||
except (OSError, ValueError) as exc:
|
||||
raise LifecycleAuthorityError(f'code authority executable is not trusted: {name}: {exc}') from exc
|
||||
elif not private_file_ready(value['path']):
|
||||
raise LifecycleAuthorityError(f'code authority ACL is not exact-private: {name}')
|
||||
try:
|
||||
current = sha256_file(value['path'])
|
||||
except OSError as exc:
|
||||
raise LifecycleAuthorityError(f'unable to verify code authority file: {name}') from exc
|
||||
if not hmac.compare_digest(current, value['sha256']):
|
||||
raise LifecycleAuthorityError(f'code authority drifted: {name}')
|
||||
return normalized
|
||||
|
||||
|
||||
def dsn_sha256(dsn):
|
||||
value = str(dsn or '')
|
||||
return hashlib.sha256(value.encode('utf-8')).hexdigest() if value else ''
|
||||
|
||||
|
||||
def supervised_child_environment(metadata, canonical_dsn, child_kind):
|
||||
return {
|
||||
'SCANNER_SUPERVISED': '1',
|
||||
CHILD_INSTANCE_FILE_ENV: str(metadata['instance_file']),
|
||||
CHILD_INSTANCE_ID_ENV: str(metadata['instance_id']),
|
||||
CHILD_TOKEN_ENV: str(metadata['token']),
|
||||
CHILD_CONFIG_HASH_ENV: str(metadata['config_sha256']),
|
||||
CHILD_SCRIPT_HASH_ENV: str(metadata['supervisor_sha256']),
|
||||
CHILD_MANIFEST_HASH_ENV: str(metadata['code_manifest_sha256']),
|
||||
CHILD_DSN_HASH_ENV: str(metadata.get('canonical_dsn_sha256') or ''),
|
||||
CHILD_KIND_ENV: str(child_kind),
|
||||
'TRUF_MANAGED_POSTGRES_DSN': str(canonical_dsn or ''),
|
||||
}
|
||||
|
||||
|
||||
def strip_supervisor_credentials(env):
|
||||
for key in list(env):
|
||||
normalized = str(key).upper()
|
||||
if normalized.startswith('TRUF_POSTGRES_') or normalized in _PRIVATE_EXTERNAL_ENV_KEYS:
|
||||
env.pop(key, None)
|
||||
return env
|
||||
|
||||
|
||||
def _send_handshake(metadata, timeout=3):
|
||||
control = metadata.get('control') or {}
|
||||
request = {
|
||||
'schema': CONTROL_SCHEMA,
|
||||
'instance_id': metadata['instance_id'],
|
||||
'token': metadata['token'],
|
||||
'action': 'handshake',
|
||||
}
|
||||
payload = json.dumps(request, ensure_ascii=True, separators=(',', ':')).encode('utf-8') + b'\n'
|
||||
with socket.create_connection((control.get('host'), int(control.get('port') or 0)), timeout=timeout) as sock:
|
||||
sock.settimeout(timeout)
|
||||
sock.sendall(payload)
|
||||
sock.shutdown(socket.SHUT_WR)
|
||||
chunks = []
|
||||
total = 0
|
||||
while True:
|
||||
chunk = sock.recv(65536)
|
||||
if not chunk:
|
||||
break
|
||||
total += len(chunk)
|
||||
if total > 1024 * 1024:
|
||||
raise LifecycleAuthorityError('supervisor handshake response is too large')
|
||||
chunks.append(chunk)
|
||||
try:
|
||||
response = json.loads(b''.join(chunks).decode('utf-8'))
|
||||
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
|
||||
raise LifecycleAuthorityError('invalid supervisor handshake response') from exc
|
||||
if (
|
||||
not isinstance(response, dict)
|
||||
or response.get('schema') != CONTROL_SCHEMA
|
||||
or response.get('instance_id') != metadata['instance_id']
|
||||
or response.get('ok') is not True
|
||||
or not isinstance(response.get('result'), dict)
|
||||
):
|
||||
raise LifecycleAuthorityError('authenticated supervisor handshake failed')
|
||||
return response['result']
|
||||
|
||||
|
||||
def _send_handshake_with_timeout_retry(metadata, timeout_retries=0):
|
||||
retries = min(1, max(0, int(timeout_retries or 0)))
|
||||
for attempt in range(retries + 1):
|
||||
try:
|
||||
return _send_handshake(metadata)
|
||||
except TimeoutError:
|
||||
if attempt >= retries:
|
||||
raise
|
||||
|
||||
|
||||
def verify_supervisor_command_line(arguments, supervisor_path, config_path):
|
||||
"""Verify the unique script binding and config value retained by the OS."""
|
||||
arguments = [str(argument) for argument in arguments]
|
||||
options = {
|
||||
'--runtime-bootstrap-entrypoint': [],
|
||||
'--config': [],
|
||||
}
|
||||
option_value_indices = set()
|
||||
for index, argument in enumerate(arguments):
|
||||
for option in options:
|
||||
if argument == option:
|
||||
value = arguments[index + 1] if index + 1 < len(arguments) else ''
|
||||
options[option].append(value)
|
||||
if index + 1 < len(arguments):
|
||||
option_value_indices.add(index + 1)
|
||||
elif argument.startswith(option + '='):
|
||||
options[option].append(argument.split('=', 1)[1])
|
||||
|
||||
expected_supervisor = canonical_path(supervisor_path)
|
||||
direct_bindings = []
|
||||
for index, argument in enumerate(arguments):
|
||||
if index in option_value_indices or not argument or argument.startswith('-'):
|
||||
continue
|
||||
try:
|
||||
if canonical_path(argument) == expected_supervisor:
|
||||
direct_bindings.append(argument)
|
||||
except (OSError, TypeError, ValueError):
|
||||
continue
|
||||
|
||||
bindings = options['--runtime-bootstrap-entrypoint'] + direct_bindings
|
||||
if len(bindings) != 1:
|
||||
raise LifecycleAuthorityError('supervisor command line must contain exactly one bound supervisor script')
|
||||
try:
|
||||
binding_matches = canonical_path(bindings[0]) == expected_supervisor
|
||||
except (OSError, TypeError, ValueError):
|
||||
binding_matches = False
|
||||
if not binding_matches:
|
||||
raise LifecycleAuthorityError('supervisor command line bound supervisor script mismatch')
|
||||
|
||||
configs = options['--config']
|
||||
if len(configs) != 1:
|
||||
raise LifecycleAuthorityError('supervisor command line must contain exactly one bound config argument')
|
||||
try:
|
||||
config_matches = canonical_path(configs[0]) == canonical_path(config_path)
|
||||
except (OSError, TypeError, ValueError):
|
||||
config_matches = False
|
||||
if not config_matches:
|
||||
raise LifecycleAuthorityError('supervisor command line bound config argument mismatch')
|
||||
|
||||
|
||||
def _verify_supervisor_process(metadata):
|
||||
process = verify_retained_process(
|
||||
metadata['pid'],
|
||||
metadata['process_creation_time'],
|
||||
metadata['executable'],
|
||||
)
|
||||
try:
|
||||
arguments = process.command_line()
|
||||
verify_supervisor_command_line(
|
||||
arguments,
|
||||
metadata['supervisor_path'],
|
||||
metadata['config_path'],
|
||||
)
|
||||
finally:
|
||||
process.close()
|
||||
|
||||
|
||||
def require_active_supervisor_child(
|
||||
config_path=None, child_kind=None, require_dsn=True, handshake_timeout_retries=0,
|
||||
):
|
||||
"""Authenticate a mutating child before it reads application inputs."""
|
||||
instance_file = os.getenv(CHILD_INSTANCE_FILE_ENV) or ''
|
||||
inherited_id = os.getenv(CHILD_INSTANCE_ID_ENV) or ''
|
||||
inherited_token = os.getenv(CHILD_TOKEN_ENV) or ''
|
||||
inherited_config_hash = os.getenv(CHILD_CONFIG_HASH_ENV) or ''
|
||||
inherited_script_hash = os.getenv(CHILD_SCRIPT_HASH_ENV) or ''
|
||||
inherited_manifest_hash = os.getenv(CHILD_MANIFEST_HASH_ENV) or ''
|
||||
inherited_dsn_hash = os.getenv(CHILD_DSN_HASH_ENV) or ''
|
||||
inherited_kind = os.getenv(CHILD_KIND_ENV) or ''
|
||||
if not all((instance_file, inherited_id, inherited_token, inherited_config_hash, inherited_script_hash, inherited_manifest_hash)):
|
||||
raise LifecycleAuthorityError('direct mutation is retired; use an authenticated active supervisor command')
|
||||
if child_kind and inherited_kind != str(child_kind):
|
||||
raise LifecycleAuthorityError('supervised child kind does not match the requested mutation entrypoint')
|
||||
|
||||
# Imported lazily to avoid a module cycle while supervisor metadata support
|
||||
# itself imports the manifest helpers above.
|
||||
from supervisor_instance import load_instance_metadata
|
||||
|
||||
try:
|
||||
metadata = load_instance_metadata(instance_file)
|
||||
except (OSError, ValueError) as exc:
|
||||
raise LifecycleAuthorityError('private supervisor instance metadata is unavailable') from exc
|
||||
if not hmac.compare_digest(metadata['instance_id'], inherited_id):
|
||||
raise LifecycleAuthorityError('supervisor child instance identity mismatch')
|
||||
if not hmac.compare_digest(metadata['token'], inherited_token):
|
||||
raise LifecycleAuthorityError('supervisor child credential mismatch')
|
||||
if metadata.get('activation_state') != PHASE_ACTIVE:
|
||||
raise LifecycleAuthorityError('supervisor is not ACTIVE; mutation is refused')
|
||||
if canonical_path(instance_file) != canonical_path(metadata.get('instance_file') or instance_file):
|
||||
raise LifecycleAuthorityError('supervisor child instance path mismatch')
|
||||
if config_path and canonical_path(config_path) != metadata['config_path']:
|
||||
raise LifecycleAuthorityError('supervisor child config path mismatch')
|
||||
|
||||
expected_pairs = (
|
||||
('config_sha256', inherited_config_hash),
|
||||
('supervisor_sha256', inherited_script_hash),
|
||||
('code_manifest_sha256', inherited_manifest_hash),
|
||||
)
|
||||
for key, inherited in expected_pairs:
|
||||
if not hmac.compare_digest(str(metadata.get(key) or ''), inherited):
|
||||
raise LifecycleAuthorityError(f'supervisor child {key} authority mismatch')
|
||||
if not hmac.compare_digest(sha256_file(metadata['config_path']), metadata['config_sha256']):
|
||||
raise LifecycleAuthorityError('supervisor config authority drifted')
|
||||
verify_code_manifest(
|
||||
metadata['code_manifest'],
|
||||
metadata['code_manifest_sha256'],
|
||||
require_private_acl=True,
|
||||
)
|
||||
_verify_supervisor_process(metadata)
|
||||
|
||||
dsn = os.getenv('TRUF_MANAGED_POSTGRES_DSN') or ''
|
||||
if require_dsn:
|
||||
try:
|
||||
parsed = parse_postgres_url(dsn)
|
||||
except ValueError as exc:
|
||||
raise LifecycleAuthorityError('a canonical managed PostgreSQL DSN is required') from exc
|
||||
if parsed['host'] != '127.0.0.1':
|
||||
raise LifecycleAuthorityError('managed PostgreSQL DSN is not loopback-bound')
|
||||
actual_dsn_hash = dsn_sha256(dsn)
|
||||
if not inherited_dsn_hash or not hmac.compare_digest(actual_dsn_hash, inherited_dsn_hash):
|
||||
raise LifecycleAuthorityError('managed PostgreSQL DSN authority mismatch')
|
||||
if not hmac.compare_digest(str(metadata.get('canonical_dsn_sha256') or ''), inherited_dsn_hash):
|
||||
raise LifecycleAuthorityError('private metadata PostgreSQL DSN authority mismatch')
|
||||
for key in ('SCANNER_DB_URL', 'DATABASE_URL'):
|
||||
if not hmac.compare_digest(str(os.getenv(key) or ''), dsn):
|
||||
raise LifecycleAuthorityError(f'{key} does not match the managed PostgreSQL DSN')
|
||||
if any(key.upper().startswith('PG') for key in os.environ):
|
||||
raise LifecycleAuthorityError('libpq PG* environment overrides are forbidden for managed children')
|
||||
|
||||
handshake = _send_handshake_with_timeout_retry(metadata, handshake_timeout_retries)
|
||||
if handshake.get('instance_id') != metadata['instance_id'] or handshake.get('activation_state') != PHASE_ACTIVE:
|
||||
raise LifecycleAuthorityError('supervisor handshake did not confirm ACTIVE authority')
|
||||
for key, value in expected_pairs:
|
||||
if not hmac.compare_digest(str(handshake.get(key) or ''), value):
|
||||
raise LifecycleAuthorityError(f'supervisor handshake {key} mismatch')
|
||||
if require_dsn and not hmac.compare_digest(str(handshake.get('canonical_dsn_sha256') or ''), inherited_dsn_hash):
|
||||
raise LifecycleAuthorityError('supervisor handshake PostgreSQL DSN authority mismatch')
|
||||
return metadata
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,324 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import fnmatch
|
||||
import hashlib
|
||||
import os
|
||||
import shutil
|
||||
|
||||
from db_backend import database_url_from_env, is_postgres_url
|
||||
from paths import apply_path_config, default_project_paths
|
||||
from migrate_runtime_safety import require_runtime_hardening_stopped
|
||||
from postgres_runtime import load_postgres_environment
|
||||
from runtime_security import ClusterAuthorityLock, reject_reparse_components
|
||||
|
||||
|
||||
DESKTOP_HF = r"C:\Users\pro100noob\Desktop\HugginFace"
|
||||
EXCLUDED_DIRS = {"__pycache__", ".git", ".opencode", "node_modules", "tmp", "runtime"}
|
||||
APP_FILES = [
|
||||
"app.py",
|
||||
"console_runner.py",
|
||||
"dashboard.py",
|
||||
"keycheck_runner.py",
|
||||
"migrate_layout.py",
|
||||
"paths.py",
|
||||
"scan_manager.py",
|
||||
"scanner.py",
|
||||
"scanner_db.py",
|
||||
"supervisor.py",
|
||||
"ui_components.py",
|
||||
"config.yaml",
|
||||
"secrets.yaml",
|
||||
"requirements.txt",
|
||||
"CHEATSHEET.md",
|
||||
"DETECTOR_NOTES.md",
|
||||
]
|
||||
APP_DIRS = [".streamlit", "keycheckers"]
|
||||
RESULT_FILES = ["found_secrets.jsonl", "scan_results.jsonl", "scan_errors.log", "scanner.db", "scanner.db-wal", "scanner.db-shm"]
|
||||
KEYCHECK_SERVICE_SCRIPTS = {
|
||||
"anthropic": [os.path.join("anthropic", "anthropicKeycheck.py")],
|
||||
"aws": [os.path.join("aws", "awsKeycheck.py")],
|
||||
"azure": [os.path.join("azure", "azureKeycheck.py")],
|
||||
"deepseek": [os.path.join("deepseek", "deepseekKeycheck.py")],
|
||||
"dockerhub": [os.path.join("dockerhub", "dockerhubKeycheck.py"), os.path.join("dockerhub", "dockerhub.txt")],
|
||||
"gcp": [os.path.join("gcp", "gcpKeycheck.py"), os.path.join("gcp", "gcp.txt")],
|
||||
"gemini": [os.path.join("gemini", "geminiKeycheck.py"), os.path.join("gemini", "gem.txt")],
|
||||
"groq": [os.path.join("groq", "groqKeycheck.py")],
|
||||
"github": [os.path.join("github", "githubKeycheck.py"), os.path.join("github", "github.txt")],
|
||||
"gitlab": [os.path.join("gitlab", "gitlabKeycheck.py"), os.path.join("gitlab", "gitlab.txt")],
|
||||
"kimi": [os.path.join("kimi", "kimiKeycheck.py")],
|
||||
"openai": ["Keycheck.py"],
|
||||
"openrouter": ["OpenrouterKeycheck.py"],
|
||||
"provider_resolver": [os.path.join("provider_resolver", "providerResolverKeycheck.py")],
|
||||
"qwen": [os.path.join("qwen", "qwenKeycheck.py")],
|
||||
"replicate": [os.path.join("replicate", "replicateKeycheck.py")],
|
||||
"xai": [os.path.join("xai", "xaiKeycheck.py")],
|
||||
"huggingface": [os.path.join("huggingface", "huggingfaceKeycheck.py")],
|
||||
"zai": [os.path.join("zai", "zaiKeycheck.py")],
|
||||
}
|
||||
KEYCHECK_TOP_LEVEL_PREFIXES = {
|
||||
"openai": "openai",
|
||||
"openrouter": "openrouter",
|
||||
}
|
||||
|
||||
|
||||
def load_config(config_path):
|
||||
if not config_path:
|
||||
return {'global': default_project_paths()}
|
||||
try:
|
||||
import yaml
|
||||
except ImportError as e:
|
||||
raise SystemExit("PyYAML is required for --config") from e
|
||||
with open(config_path, "r", encoding="utf-8") as f:
|
||||
config = apply_path_config(yaml.safe_load(f) or {}, config_path)
|
||||
return config
|
||||
|
||||
|
||||
def load_layout(config_path):
|
||||
return load_config(config_path).get('global') or {}
|
||||
|
||||
|
||||
def mkdir(path, dry_run=False):
|
||||
if dry_run:
|
||||
print(f"mkdir {path}")
|
||||
return
|
||||
os.makedirs(path, exist_ok=True)
|
||||
|
||||
|
||||
def copy_file(src, dst, overwrite=False, dry_run=False):
|
||||
if not os.path.exists(src):
|
||||
return False
|
||||
if os.path.lexists(dst) and not overwrite:
|
||||
print(f"skip existing {dst}")
|
||||
return False
|
||||
parent = os.path.dirname(dst)
|
||||
if parent:
|
||||
mkdir(parent, dry_run)
|
||||
if dry_run:
|
||||
print(f"copy {src} -> {dst}")
|
||||
return True
|
||||
try:
|
||||
reject_reparse_components(parent or os.path.dirname(os.path.abspath(dst)))
|
||||
if os.path.lexists(dst):
|
||||
reject_reparse_components(dst)
|
||||
shutil.copy2(src, dst)
|
||||
except OSError as e:
|
||||
print(f"skip locked/unavailable {src}: {e}")
|
||||
return False
|
||||
print(f"copied {src} -> {dst}")
|
||||
return True
|
||||
|
||||
|
||||
def ignore_app_dir(_dir, names):
|
||||
ignored = set()
|
||||
for name in names:
|
||||
if name in EXCLUDED_DIRS:
|
||||
ignored.add(name)
|
||||
if fnmatch.fnmatch(name, "*.pyc"):
|
||||
ignored.add(name)
|
||||
return ignored
|
||||
|
||||
|
||||
def copy_dir(src, dst, overwrite=False, dry_run=False, verified_apply=False):
|
||||
if not os.path.isdir(src):
|
||||
return False
|
||||
if os.path.lexists(dst) and not overwrite:
|
||||
print(f"skip existing {dst}")
|
||||
return False
|
||||
if dry_run:
|
||||
print(f"copytree {src} -> {dst}")
|
||||
return True
|
||||
if os.path.lexists(dst) and overwrite:
|
||||
raise RuntimeError('legacy directory replacement is retired; existing directories are never replaced')
|
||||
shutil.copytree(src, dst, ignore=ignore_app_dir, dirs_exist_ok=False)
|
||||
print(f"copied {src} -> {dst}")
|
||||
return True
|
||||
|
||||
|
||||
def create_layout(layout, dry_run=False):
|
||||
for key in ("project_dir", "runtime_dir", "results_dir", "queue_dir", "log_dir", "state_dir", "keycheck_dir", "work_dir"):
|
||||
mkdir(layout[key], dry_run)
|
||||
mkdir(os.path.join(layout["runtime_dir"], "imports"), dry_run)
|
||||
|
||||
|
||||
def copy_app_files(source_dir, layout, overwrite=False, dry_run=False, verified_apply=False):
|
||||
project_dir = layout["project_dir"]
|
||||
if os.path.abspath(source_dir) == os.path.abspath(project_dir):
|
||||
print("app source is already project_dir; app copy skipped")
|
||||
return
|
||||
for name in APP_FILES:
|
||||
copy_file(os.path.join(source_dir, name), os.path.join(project_dir, name), overwrite, dry_run)
|
||||
for name in APP_DIRS:
|
||||
copy_dir(os.path.join(source_dir, name), os.path.join(project_dir, name), overwrite, dry_run, verified_apply)
|
||||
|
||||
|
||||
def copy_scanner_runtime(old_root, layout, overwrite=False, dry_run=False, verified_apply=False):
|
||||
for name in RESULT_FILES:
|
||||
copy_file(os.path.join(old_root, name), os.path.join(layout["results_dir"], name), overwrite, dry_run)
|
||||
for pattern in ("todo_*.txt", "checked_*.txt"):
|
||||
if not os.path.isdir(old_root):
|
||||
continue
|
||||
for name in os.listdir(old_root):
|
||||
if fnmatch.fnmatch(name, pattern):
|
||||
copy_file(os.path.join(old_root, name), os.path.join(layout["queue_dir"], name), overwrite, dry_run)
|
||||
copy_dir(os.path.join(old_root, "logs"), layout["log_dir"], overwrite, dry_run, verified_apply)
|
||||
copy_dir(os.path.join(old_root, "state"), layout["state_dir"], overwrite, dry_run, verified_apply)
|
||||
copy_file(os.path.join(old_root, "runner_state.json"), os.path.join(layout["state_dir"], "runner_state.json"), overwrite, dry_run)
|
||||
|
||||
|
||||
def line_hash(line):
|
||||
return hashlib.sha256(line.strip().encode("utf-8", errors="replace")).hexdigest()
|
||||
|
||||
|
||||
def existing_line_hashes(path):
|
||||
hashes = set()
|
||||
if not os.path.exists(path):
|
||||
return hashes
|
||||
with open(path, "r", encoding="utf-8", errors="replace") as f:
|
||||
for line in f:
|
||||
if line.strip():
|
||||
hashes.add(line_hash(line))
|
||||
return hashes
|
||||
|
||||
|
||||
def import_jsonl_dedupe(inputs, output, dry_run=False):
|
||||
hashes = existing_line_hashes(output)
|
||||
added = 0
|
||||
if dry_run:
|
||||
print(f"dedupe import {len(inputs)} file(s) -> {output}")
|
||||
return 0
|
||||
mkdir(os.path.dirname(output), dry_run=False)
|
||||
with open(output, "a", encoding="utf-8") as dst:
|
||||
for path in inputs:
|
||||
if not os.path.exists(path):
|
||||
continue
|
||||
with open(path, "r", encoding="utf-8", errors="replace") as src:
|
||||
for line in src:
|
||||
if not line.strip():
|
||||
continue
|
||||
digest = line_hash(line)
|
||||
if digest in hashes:
|
||||
continue
|
||||
dst.write(line if line.endswith("\n") else line + "\n")
|
||||
hashes.add(digest)
|
||||
added += 1
|
||||
print(f"imported {added} unique finding line(s) into {output}")
|
||||
return added
|
||||
|
||||
|
||||
def copy_legacy_keychecker_outputs(layout, desktop_dir=DESKTOP_HF, overwrite=False, dry_run=False):
|
||||
if not os.path.isdir(desktop_dir):
|
||||
return
|
||||
for service in KEYCHECK_SERVICE_SCRIPTS:
|
||||
source_dir = os.path.join(desktop_dir, service)
|
||||
target_dir = os.path.join(layout["keycheck_dir"], service)
|
||||
if os.path.isdir(source_dir):
|
||||
for name in os.listdir(source_dir):
|
||||
if name.lower().endswith((".txt", ".jsonl")):
|
||||
copy_file(os.path.join(source_dir, name), os.path.join(target_dir, name), overwrite, dry_run)
|
||||
for service, prefix in KEYCHECK_TOP_LEVEL_PREFIXES.items():
|
||||
target_dir = os.path.join(layout["keycheck_dir"], service)
|
||||
for name in os.listdir(desktop_dir):
|
||||
lower = name.lower()
|
||||
if lower.startswith(prefix) and lower.endswith((".txt", ".jsonl")):
|
||||
copy_file(os.path.join(desktop_dir, name), os.path.join(target_dir, name), overwrite, dry_run)
|
||||
|
||||
|
||||
def merge_unique_lines(inputs, output, dry_run=False):
|
||||
values = []
|
||||
seen = existing_values = set()
|
||||
if os.path.exists(output):
|
||||
with open(output, "r", encoding="utf-8", errors="replace") as f:
|
||||
existing_values = {line.strip().lstrip("\ufeff") for line in f if line.strip()}
|
||||
seen = set(existing_values)
|
||||
for path in inputs:
|
||||
if not os.path.exists(path):
|
||||
continue
|
||||
with open(path, "r", encoding="utf-8", errors="replace") as f:
|
||||
for line in f:
|
||||
value = line.strip().lstrip("\ufeff")
|
||||
if value and value not in seen:
|
||||
values.append(value)
|
||||
seen.add(value)
|
||||
if dry_run:
|
||||
print(f"merge {len(values)} unique line(s) -> {output}")
|
||||
return
|
||||
mkdir(os.path.dirname(output), dry_run=False)
|
||||
with open(output, "a", encoding="utf-8") as f:
|
||||
for value in values:
|
||||
f.write(value + "\n")
|
||||
print(f"appended {len(values)} unique line(s) -> {output}")
|
||||
|
||||
|
||||
def import_huggingface_desktop(layout, desktop_dir=DESKTOP_HF, overwrite=False, dry_run=False):
|
||||
if not os.path.isdir(desktop_dir):
|
||||
print(f"Desktop HugginFace directory not found: {desktop_dir}")
|
||||
return
|
||||
jsonl_inputs = [os.path.join(desktop_dir, name) for name in os.listdir(desktop_dir) if fnmatch.fnmatch(name, "found_secrets*.jsonl")]
|
||||
import_jsonl_dedupe(jsonl_inputs, os.path.join(layout["results_dir"], "found_secrets.jsonl"), dry_run)
|
||||
merge_unique_lines([os.path.join(desktop_dir, "checked.txt")], os.path.join(layout["queue_dir"], "checked_huggingface.txt"), dry_run)
|
||||
merge_unique_lines([os.path.join(desktop_dir, "todo.txt")], os.path.join(layout["queue_dir"], "todo_huggingface.txt"), dry_run)
|
||||
copy_file(os.path.join(desktop_dir, "proxy.txt"), layout["proxy_file"], overwrite=False, dry_run=dry_run)
|
||||
copy_file(os.path.join(desktop_dir, "requirements-keycheckers.txt"), os.path.join(layout["project_dir"], "requirements-keycheckers.txt"), overwrite, dry_run)
|
||||
copy_file(os.path.join(desktop_dir, "KEYCHECKERS.md"), os.path.join(layout["project_dir"], "KEYCHECKERS.md"), overwrite, dry_run)
|
||||
for service, rel_paths in KEYCHECK_SERVICE_SCRIPTS.items():
|
||||
for rel_path in rel_paths:
|
||||
src = os.path.join(desktop_dir, rel_path)
|
||||
dst = os.path.join(layout["project_dir"], "keycheckers", service, os.path.basename(rel_path))
|
||||
copy_file(src, dst, overwrite, dry_run)
|
||||
copy_legacy_keychecker_outputs(layout, desktop_dir, overwrite, dry_run)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Copy/import legacy scanner files into the unified D:\\truf layout.")
|
||||
parser.add_argument("--config", default="config.yaml")
|
||||
parser.add_argument("--source-app", default=os.path.dirname(os.path.abspath(__file__)))
|
||||
parser.add_argument("--target-app", help="Destination app directory. Defaults to <root_dir>\\app.")
|
||||
parser.add_argument("--in-place", action="store_true", help="Use project_dir from config instead of copying to <root_dir>\\app.")
|
||||
parser.add_argument("--old-root", default=r"D:\truf")
|
||||
parser.add_argument("--desktop-hf", default=DESKTOP_HF)
|
||||
parser.add_argument("--overwrite", action="store_true")
|
||||
parser.add_argument("--dry-run", action="store_true")
|
||||
parser.add_argument("--apply", action="store_true", help="Retired; production layout mutation is disabled")
|
||||
parser.add_argument("--no-app-copy", action="store_true")
|
||||
parser.add_argument("--no-desktop-import", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
args = parse_args()
|
||||
if args.apply and args.dry_run:
|
||||
raise SystemExit('--apply and --dry-run are mutually exclusive')
|
||||
if args.apply:
|
||||
raise SystemExit(
|
||||
'migrate_layout --apply is retired because the production layout is already migrated. '
|
||||
'Use reviewed offline backup/restore tooling for any future relocation.'
|
||||
)
|
||||
config = load_config(args.config)
|
||||
layout = config.get('global') or {}
|
||||
if not args.in_place:
|
||||
layout["project_dir"] = args.target_app or os.path.join(layout["root_dir"], "app")
|
||||
dry_run = not args.apply
|
||||
load_postgres_environment(os.path.abspath(args.config), config)
|
||||
endpoint_dsn = database_url_from_env() or layout.get('database_url')
|
||||
if not is_postgres_url(endpoint_dsn):
|
||||
raise SystemExit('A caller-selected canonical PostgreSQL DSN is required for maintenance authority')
|
||||
with ClusterAuthorityLock(config, endpoint_dsn=endpoint_dsn):
|
||||
require_runtime_hardening_stopped(config)
|
||||
create_layout(layout, dry_run)
|
||||
if not args.no_app_copy:
|
||||
copy_app_files(args.source_app, layout, args.overwrite, dry_run, verified_apply=args.apply)
|
||||
copy_scanner_runtime(args.old_root, layout, args.overwrite, dry_run, verified_apply=args.apply)
|
||||
if not args.no_desktop_import:
|
||||
import_huggingface_desktop(layout, args.desktop_hf, args.overwrite, dry_run)
|
||||
if dry_run:
|
||||
print("Dry-run complete. --apply is retired; use reviewed offline backup/restore tooling for relocation.")
|
||||
return 0
|
||||
print("Migration copy/import finished. Originals were left in place.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,137 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import sqlite3
|
||||
|
||||
from scanner_db import ScannerDB
|
||||
|
||||
|
||||
TABLES = [
|
||||
'runs',
|
||||
'source_cycles',
|
||||
'target_scans',
|
||||
'target_queue',
|
||||
'scan_publication_outbox',
|
||||
'findings',
|
||||
'errors',
|
||||
'queue_snapshots',
|
||||
'config_snapshots',
|
||||
'package_repo_candidates',
|
||||
'keycheck_results',
|
||||
'keycheck_event_map',
|
||||
'finding_uid_map',
|
||||
]
|
||||
|
||||
|
||||
def sqlite_connect_ro(path):
|
||||
uri = 'file:' + os.path.abspath(path).replace('\\', '/') + '?mode=ro'
|
||||
conn = sqlite3.connect(uri, uri=True)
|
||||
conn.row_factory = sqlite3.Row
|
||||
return conn
|
||||
|
||||
|
||||
def sqlite_columns(conn, table):
|
||||
return [row['name'] for row in conn.execute(f'PRAGMA table_info({table})').fetchall()]
|
||||
|
||||
|
||||
def sqlite_count(conn, table):
|
||||
return int(conn.execute(f'SELECT COUNT(*) AS count FROM {table}').fetchone()['count'])
|
||||
|
||||
|
||||
def sqlite_row_estimate(conn, table, exact=False):
|
||||
if exact:
|
||||
return sqlite_count(conn, table)
|
||||
try:
|
||||
row = conn.execute('SELECT seq FROM sqlite_sequence WHERE name = ?', (table,)).fetchone()
|
||||
if row and row['seq'] is not None:
|
||||
return int(row['seq'])
|
||||
except sqlite3.Error:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def pg_reset_identity(db, table):
|
||||
db.conn.execute(f'''
|
||||
SELECT setval(
|
||||
pg_get_serial_sequence('{table}', 'id'),
|
||||
COALESCE((SELECT MAX(id) FROM {table}), 1),
|
||||
(SELECT MAX(id) IS NOT NULL FROM {table})
|
||||
)
|
||||
''')
|
||||
|
||||
|
||||
def postgres_safe_value(value):
|
||||
if isinstance(value, str) and '\x00' in value:
|
||||
return value.replace('\x00', '\\u0000')
|
||||
return value
|
||||
|
||||
|
||||
def copy_table(source, target, table, batch_size, dry_run=False, exact_counts=False):
|
||||
columns = sqlite_columns(source, table)
|
||||
if not columns:
|
||||
print(f'{table}: missing or empty schema in SQLite, skipped', flush=True)
|
||||
return 0
|
||||
total = sqlite_row_estimate(source, table, exact=exact_counts)
|
||||
total_label = total if total is not None else 'unknown'
|
||||
print(f'{table}: source_rows={total_label}' + ('' if exact_counts else ' estimated'), flush=True)
|
||||
if dry_run or total == 0:
|
||||
return int(total or 0)
|
||||
|
||||
column_sql = ', '.join(columns)
|
||||
placeholders = ', '.join('?' for _ in columns)
|
||||
conflict_sql = ' ON CONFLICT (id) DO NOTHING' if 'id' in columns else ''
|
||||
insert_sql = f'INSERT INTO {table} ({column_sql}) VALUES ({placeholders}){conflict_sql}'
|
||||
|
||||
copied = 0
|
||||
cursor = source.execute(f'SELECT {column_sql} FROM {table} ORDER BY id' if 'id' in columns else f'SELECT {column_sql} FROM {table}')
|
||||
while True:
|
||||
rows = cursor.fetchmany(batch_size)
|
||||
if not rows:
|
||||
break
|
||||
values = [[postgres_safe_value(row[column]) for column in columns] for row in rows]
|
||||
if target.conn.is_postgres:
|
||||
with target.conn._conn.cursor() as pg_cursor:
|
||||
with pg_cursor.copy(f'COPY {table} ({column_sql}) FROM STDIN') as copy:
|
||||
for value_row in values:
|
||||
copy.write_row(value_row)
|
||||
else:
|
||||
for value_row in values:
|
||||
target.conn.execute(insert_sql, value_row)
|
||||
copied += len(rows)
|
||||
target.conn.commit()
|
||||
print(f'{table}: copied={copied}/{total_label}', flush=True)
|
||||
if 'id' in columns:
|
||||
pg_reset_identity(target, table)
|
||||
target.conn.commit()
|
||||
return copied
|
||||
|
||||
|
||||
def truncate_target(db):
|
||||
table_sql = ', '.join(TABLES)
|
||||
db.conn.execute(f'TRUNCATE TABLE {table_sql} RESTART IDENTITY CASCADE')
|
||||
db.conn.commit()
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description='Offline migrate scanner observability SQLite DB to PostgreSQL.')
|
||||
parser.add_argument('--sqlite', required=True, help='Path to scanner_active.db or archived SQLite DB')
|
||||
parser.add_argument('--db-url', default=os.getenv('SCANNER_DB_URL') or os.getenv('DATABASE_URL'), help='PostgreSQL DSN')
|
||||
parser.add_argument('--batch-size', type=int, default=1000)
|
||||
parser.add_argument('--exact-counts', action='store_true', help='Use exact COUNT(*) per table; slow on multi-GB SQLite files')
|
||||
parser.add_argument('--truncate', action='store_true', help='Delete existing Postgres observability rows before import')
|
||||
parser.add_argument('--dry-run', action='store_true')
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
raise SystemExit(
|
||||
'migrate_observability_db.py is retired for PostgreSQL. Use migrate_runtime_safety.py '
|
||||
'--apply --sources-stopped with the bound cluster; perform legacy data import only with a reviewed, cluster-bound tool.'
|
||||
)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,21 @@
|
||||
"""Retired direct database mutation entrypoint.
|
||||
|
||||
Dashboard indexes are part of the authority-locked offline runtime-safety
|
||||
migration. Keeping a second live DDL path would bypass cluster identity and
|
||||
stopped-source checks.
|
||||
"""
|
||||
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
|
||||
def main():
|
||||
raise SystemExit(
|
||||
'optimize_dashboard_db is retired; run migrate_runtime_safety.py '
|
||||
'offline under cluster authority'
|
||||
)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
File diff suppressed because it is too large
Load Diff
+237
@@ -0,0 +1,237 @@
|
||||
import ntpath
|
||||
import os
|
||||
import re
|
||||
|
||||
from query_policy import validate_rejected_query_policy
|
||||
|
||||
|
||||
APP_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
CANONICAL_ROOT = os.path.dirname(APP_DIR)
|
||||
DEFAULT_TRUFFLEHOG = r"C:\Tools\trufflehog.exe"
|
||||
|
||||
PLACEHOLDER_RE = re.compile(r"\{([A-Za-z_][A-Za-z0-9_]*)\}")
|
||||
|
||||
|
||||
class PathResolutionError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def _norm(path):
|
||||
return os.path.normpath(str(path))
|
||||
|
||||
|
||||
def _config_dir(config_path=None):
|
||||
if config_path:
|
||||
return os.path.dirname(os.path.abspath(config_path))
|
||||
return APP_DIR
|
||||
|
||||
|
||||
def _expand(value, context):
|
||||
text = str(value)
|
||||
missing = sorted({name for name in PLACEHOLDER_RE.findall(text) if name not in context})
|
||||
if missing:
|
||||
raise PathResolutionError(f"Unknown path placeholder(s): {', '.join(missing)} in {text!r}")
|
||||
for name in PLACEHOLDER_RE.findall(text):
|
||||
text = text.replace("{" + name + "}", str(context[name]))
|
||||
return os.path.expandvars(os.path.expanduser(text))
|
||||
|
||||
|
||||
def is_command_name(value):
|
||||
text = str(value or "")
|
||||
return bool(text) and not os.path.isabs(text) and "\\" not in text and "/" not in text
|
||||
|
||||
|
||||
def is_database_url(value):
|
||||
return str(value or "").strip().lower().startswith(("postgresql://", "postgres://"))
|
||||
|
||||
|
||||
def resolve_path(value, context=None, base_dir=None, allow_command=False, required=False):
|
||||
if value is None or str(value).strip() == "":
|
||||
if required:
|
||||
raise PathResolutionError("Required path is empty")
|
||||
return value
|
||||
|
||||
context = context or {}
|
||||
text = _expand(value, context)
|
||||
if os.name != 'nt' and (ntpath.splitdrive(text)[0] or '\\' in text):
|
||||
raise PathResolutionError(f"Windows path is not supported on this platform: {text!r}")
|
||||
if allow_command and is_command_name(text):
|
||||
return text
|
||||
if os.path.isabs(text):
|
||||
return _norm(text)
|
||||
base = base_dir or context.get("project_dir") or context.get("config_dir") or os.getcwd()
|
||||
if os.name != 'nt' and (ntpath.splitdrive(str(base))[0] or '\\' in str(base)):
|
||||
raise PathResolutionError(f"Windows base path is not supported on this platform: {base!r}")
|
||||
return _norm(os.path.join(base, text))
|
||||
|
||||
|
||||
def default_trufflehog_path():
|
||||
return DEFAULT_TRUFFLEHOG if os.name == 'nt' and os.path.exists(DEFAULT_TRUFFLEHOG) else "trufflehog"
|
||||
|
||||
|
||||
def resolve_project_paths(global_config=None, config_path=None):
|
||||
global_config = global_config or {}
|
||||
context = {"config_dir": _config_dir(config_path)}
|
||||
|
||||
root_raw = (
|
||||
global_config.get("root_dir")
|
||||
or os.getenv("SCANNER_ROOT_DIR")
|
||||
or os.getenv("SCANNER_PROJECT_ROOT")
|
||||
or CANONICAL_ROOT
|
||||
)
|
||||
context["root_dir"] = resolve_path(root_raw, context, base_dir=context["config_dir"], required=True)
|
||||
|
||||
project_raw = global_config.get("project_dir") or os.getenv("SCANNER_PROJECT_DIR") or context["config_dir"]
|
||||
context["project_dir"] = resolve_path(project_raw, context, base_dir=context["config_dir"], required=True)
|
||||
|
||||
ordered_defaults = [
|
||||
("runtime_dir", os.getenv("SCANNER_RUNTIME_DIR") or os.path.join(context["root_dir"], "runtime")),
|
||||
("result_bundle_dir", os.getenv("SCANNER_RESULT_BUNDLE_DIR") or "{runtime_dir}/result_bundles"),
|
||||
("result_spool_dir", "{runtime_dir}/result_spool"),
|
||||
("results_dir", os.getenv("SCAN_RESULTS_DIR") or "{runtime_dir}/results"),
|
||||
("queue_dir", "{runtime_dir}/queues"),
|
||||
("state_dir", "{runtime_dir}/state"),
|
||||
("log_dir", "{runtime_dir}/logs"),
|
||||
("control_dir", "{runtime_dir}/control"),
|
||||
("keycheck_dir", "{runtime_dir}/keychecks"),
|
||||
("postman_cache_dir", "{runtime_dir}/postman_cache"),
|
||||
("gharchive_cache_dir", "{state_dir}/gharchive_cache"),
|
||||
("work_dir", os.getenv("TRUFFLEHOG_WORK_DIR") or os.path.join(context["root_dir"], "tmp")),
|
||||
("proxy_file", "{runtime_dir}/proxy.txt"),
|
||||
("database_path", os.getenv("SCANNER_DB_PATH") or os.getenv("SCAN_DB_PATH") or "{results_dir}/scanner.db"),
|
||||
("state_file", "{state_dir}/runner_state.json"),
|
||||
("secrets_file", "{project_dir}/secrets.yaml"),
|
||||
]
|
||||
|
||||
for key, default in ordered_defaults:
|
||||
raw = global_config.get(key) or default
|
||||
context[key] = resolve_path(raw, context, base_dir=context["project_dir"], required=True)
|
||||
|
||||
managed_database_url = os.getenv("TRUF_MANAGED_POSTGRES_DSN") or ""
|
||||
database_url = managed_database_url or global_config.get("database_url") or os.getenv("SCANNER_DB_URL") or os.getenv("DATABASE_URL") or ""
|
||||
context["database_url"] = _expand(database_url, context) if database_url else ""
|
||||
dashboard_db_url = managed_database_url or global_config.get("dashboard_db_url") or os.getenv("SCANNER_DASHBOARD_DB_URL") or context["database_url"]
|
||||
context["dashboard_db_url"] = _expand(dashboard_db_url, context) if dashboard_db_url else ""
|
||||
|
||||
trufflehog_raw = global_config.get("trufflehog_path") or os.getenv("TRUFFLEHOG_PATH") or default_trufflehog_path()
|
||||
context["trufflehog_path"] = resolve_path(
|
||||
trufflehog_raw,
|
||||
context,
|
||||
base_dir=context["project_dir"],
|
||||
allow_command=True,
|
||||
required=True,
|
||||
)
|
||||
return context
|
||||
|
||||
|
||||
def default_project_paths():
|
||||
return resolve_project_paths({}, None)
|
||||
|
||||
|
||||
def resolve_optional_path(value, path_context, base_dir=None, allow_command=False):
|
||||
if not value:
|
||||
return value
|
||||
return resolve_path(value, path_context, base_dir=base_dir or path_context.get("project_dir"), allow_command=allow_command)
|
||||
|
||||
|
||||
def resolve_postgres_data_dir(global_config=None, runtime_dir=None, base_dir=None):
|
||||
global_config = global_config or {}
|
||||
runtime_dir = runtime_dir or global_config.get('runtime_dir')
|
||||
if not runtime_dir:
|
||||
root_dir = global_config.get('root_dir') or CANONICAL_ROOT
|
||||
runtime_dir = os.path.join(root_dir, 'runtime')
|
||||
context = dict(global_config)
|
||||
context['runtime_dir'] = runtime_dir
|
||||
raw = global_config.get('postgres_data_dir') or os.path.join(runtime_dir, 'postgres', 'data')
|
||||
return resolve_path(
|
||||
raw,
|
||||
context,
|
||||
base_dir=base_dir or global_config.get('project_dir') or global_config.get('root_dir'),
|
||||
required=True,
|
||||
)
|
||||
|
||||
|
||||
def resolve_postgres_bin_dir(global_config=None, runtime_dir=None, base_dir=None):
|
||||
global_config = global_config or {}
|
||||
runtime_dir = runtime_dir or global_config.get('runtime_dir')
|
||||
if not runtime_dir:
|
||||
runtime_dir = os.path.join(global_config.get('root_dir') or CANONICAL_ROOT, 'runtime')
|
||||
context = dict(global_config, runtime_dir=runtime_dir)
|
||||
return resolve_path(
|
||||
global_config.get('postgres_bin_dir') or os.path.join(runtime_dir, 'postgres', 'pgsql', 'bin'),
|
||||
context,
|
||||
base_dir=base_dir or global_config.get('project_dir') or global_config.get('root_dir'),
|
||||
required=True,
|
||||
)
|
||||
|
||||
|
||||
def apply_path_config(config, config_path=None):
|
||||
config = config or {}
|
||||
validate_rejected_query_policy(config)
|
||||
global_config = config.setdefault("global", {})
|
||||
path_context = resolve_project_paths(global_config, config_path)
|
||||
|
||||
for key, value in path_context.items():
|
||||
global_config[key] = value
|
||||
|
||||
if global_config.get('legacy_result_spool_dir'):
|
||||
global_config['legacy_result_spool_dir'] = resolve_path(
|
||||
global_config['legacy_result_spool_dir'],
|
||||
path_context,
|
||||
base_dir=path_context['project_dir'],
|
||||
required=True,
|
||||
)
|
||||
|
||||
if global_config.get('postgres_data_dir'):
|
||||
global_config['postgres_data_dir'] = resolve_postgres_data_dir(
|
||||
global_config,
|
||||
path_context['runtime_dir'],
|
||||
base_dir=path_context['project_dir'],
|
||||
)
|
||||
|
||||
if global_config.get('postgres_bin_dir'):
|
||||
global_config['postgres_bin_dir'] = resolve_postgres_bin_dir(
|
||||
global_config,
|
||||
path_context['runtime_dir'],
|
||||
base_dir=path_context['project_dir'],
|
||||
)
|
||||
|
||||
for key in (
|
||||
'api_proxy_file', 'download_proxy_file', 'trufflehog_config',
|
||||
'dashboard_db_path', 'scan_limiter_db', 'dockerhub_tag_cache_path',
|
||||
):
|
||||
if global_config.get(key):
|
||||
global_config[key] = resolve_path(
|
||||
global_config[key],
|
||||
path_context,
|
||||
base_dir=path_context['project_dir'],
|
||||
required=True,
|
||||
)
|
||||
|
||||
supervisor = config.setdefault("supervisor", {})
|
||||
supervisor_defaults = {
|
||||
"log_dir": "{log_dir}",
|
||||
"control_dir": "{control_dir}",
|
||||
"instance_file": "{control_dir}/supervisor.instance.json",
|
||||
"lock_file": "{control_dir}/supervisor.lock",
|
||||
"supervisor_log": "{log_dir}/supervisor.log",
|
||||
"status_file": "{log_dir}/supervisor.status.txt",
|
||||
"dashboard_log": "{log_dir}/dashboard.log",
|
||||
"state_dir": "{state_dir}",
|
||||
}
|
||||
for key, default in supervisor_defaults.items():
|
||||
supervisor[key] = resolve_path(
|
||||
supervisor.get(key) or default,
|
||||
path_context,
|
||||
base_dir=path_context["project_dir"],
|
||||
required=True,
|
||||
)
|
||||
|
||||
return config
|
||||
|
||||
|
||||
def ensure_directories(paths, keys):
|
||||
for key in keys:
|
||||
path = paths.get(key)
|
||||
if path:
|
||||
os.makedirs(path, exist_ok=True)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,449 @@
|
||||
import ctypes
|
||||
import os
|
||||
import select
|
||||
import signal
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
|
||||
from runtime_security import canonical_path
|
||||
|
||||
|
||||
if os.name == 'nt':
|
||||
from ctypes import wintypes
|
||||
|
||||
class _FILETIME(ctypes.Structure):
|
||||
_fields_ = [('dwLowDateTime', wintypes.DWORD), ('dwHighDateTime', wintypes.DWORD)]
|
||||
|
||||
class _UNICODE_STRING(ctypes.Structure):
|
||||
_fields_ = [
|
||||
('Length', wintypes.USHORT),
|
||||
('MaximumLength', wintypes.USHORT),
|
||||
('Buffer', ctypes.c_void_p),
|
||||
]
|
||||
|
||||
_P_DWORD = ctypes.POINTER(wintypes.DWORD)
|
||||
_P_ULONG = ctypes.POINTER(wintypes.ULONG)
|
||||
_P_BOOL = ctypes.POINTER(wintypes.BOOL)
|
||||
_P_FILETIME = ctypes.POINTER(_FILETIME)
|
||||
_P_UNICODE_STRING = ctypes.POINTER(_UNICODE_STRING)
|
||||
_P_INT = ctypes.POINTER(ctypes.c_int)
|
||||
_P_LPWSTR = ctypes.POINTER(wintypes.LPWSTR)
|
||||
|
||||
_KERNEL32 = ctypes.WinDLL('kernel32', use_last_error=True)
|
||||
_NTDLL = ctypes.WinDLL('ntdll', use_last_error=True)
|
||||
_SHELL32 = ctypes.WinDLL('shell32', use_last_error=True)
|
||||
|
||||
_GET_EXIT_CODE_PROCESS = _KERNEL32.GetExitCodeProcess
|
||||
_GET_EXIT_CODE_PROCESS.argtypes = [wintypes.HANDLE, _P_DWORD]
|
||||
_GET_EXIT_CODE_PROCESS.restype = wintypes.BOOL
|
||||
_WAIT_FOR_SINGLE_OBJECT = _KERNEL32.WaitForSingleObject
|
||||
_WAIT_FOR_SINGLE_OBJECT.argtypes = [wintypes.HANDLE, wintypes.DWORD]
|
||||
_WAIT_FOR_SINGLE_OBJECT.restype = wintypes.DWORD
|
||||
_CLOSE_HANDLE = _KERNEL32.CloseHandle
|
||||
_CLOSE_HANDLE.argtypes = [wintypes.HANDLE]
|
||||
_CLOSE_HANDLE.restype = wintypes.BOOL
|
||||
_GET_PROCESS_TIMES = _KERNEL32.GetProcessTimes
|
||||
_GET_PROCESS_TIMES.argtypes = [
|
||||
wintypes.HANDLE, _P_FILETIME, _P_FILETIME, _P_FILETIME, _P_FILETIME,
|
||||
]
|
||||
_GET_PROCESS_TIMES.restype = wintypes.BOOL
|
||||
_QUERY_FULL_PROCESS_IMAGE_NAME = _KERNEL32.QueryFullProcessImageNameW
|
||||
_QUERY_FULL_PROCESS_IMAGE_NAME.argtypes = [
|
||||
wintypes.HANDLE, wintypes.DWORD, wintypes.LPWSTR, _P_DWORD,
|
||||
]
|
||||
_QUERY_FULL_PROCESS_IMAGE_NAME.restype = wintypes.BOOL
|
||||
_IS_PROCESS_IN_JOB = _KERNEL32.IsProcessInJob
|
||||
_IS_PROCESS_IN_JOB.argtypes = [wintypes.HANDLE, wintypes.HANDLE, _P_BOOL]
|
||||
_IS_PROCESS_IN_JOB.restype = wintypes.BOOL
|
||||
_OPEN_PROCESS = _KERNEL32.OpenProcess
|
||||
_OPEN_PROCESS.argtypes = [wintypes.DWORD, wintypes.BOOL, wintypes.DWORD]
|
||||
_OPEN_PROCESS.restype = wintypes.HANDLE
|
||||
_GET_CURRENT_PROCESS = _KERNEL32.GetCurrentProcess
|
||||
_GET_CURRENT_PROCESS.argtypes = []
|
||||
_GET_CURRENT_PROCESS.restype = wintypes.HANDLE
|
||||
_TERMINATE_PROCESS = _KERNEL32.TerminateProcess
|
||||
_TERMINATE_PROCESS.argtypes = [wintypes.HANDLE, wintypes.UINT]
|
||||
_TERMINATE_PROCESS.restype = wintypes.BOOL
|
||||
_LOCAL_FREE = _KERNEL32.LocalFree
|
||||
_LOCAL_FREE.argtypes = [wintypes.HLOCAL]
|
||||
_LOCAL_FREE.restype = wintypes.HLOCAL
|
||||
_NT_QUERY_INFORMATION_PROCESS = _NTDLL.NtQueryInformationProcess
|
||||
_NT_QUERY_INFORMATION_PROCESS.argtypes = [
|
||||
wintypes.HANDLE, wintypes.ULONG, ctypes.c_void_p, wintypes.ULONG, _P_ULONG,
|
||||
]
|
||||
_NT_QUERY_INFORMATION_PROCESS.restype = ctypes.c_long
|
||||
_COMMAND_LINE_TO_ARGV = _SHELL32.CommandLineToArgvW
|
||||
_COMMAND_LINE_TO_ARGV.argtypes = [wintypes.LPCWSTR, _P_INT]
|
||||
_COMMAND_LINE_TO_ARGV.restype = _P_LPWSTR
|
||||
else:
|
||||
_FILETIME = _UNICODE_STRING = None
|
||||
_KERNEL32 = _NTDLL = _SHELL32 = None
|
||||
|
||||
|
||||
class ProcessIdentityError(OSError):
|
||||
pass
|
||||
|
||||
|
||||
class ProcessExitedError(ProcessIdentityError):
|
||||
pass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ProcessIdentity:
|
||||
pid: int
|
||||
creation_time: str
|
||||
creation_time_unix: float
|
||||
executable: str
|
||||
in_job: object
|
||||
|
||||
def as_dict(self):
|
||||
return {
|
||||
'pid': int(self.pid),
|
||||
'creation_time': str(self.creation_time),
|
||||
'creation_time_unix': float(self.creation_time_unix),
|
||||
'executable': str(self.executable),
|
||||
'in_job': self.in_job,
|
||||
}
|
||||
|
||||
|
||||
class RetainedProcess:
|
||||
def __init__(self, identity, handle=None, pidfd=None):
|
||||
self.identity = identity
|
||||
self._handle = handle
|
||||
self._pidfd = pidfd
|
||||
self._closed = False
|
||||
|
||||
@property
|
||||
def pid(self):
|
||||
return self.identity.pid
|
||||
|
||||
def is_running(self):
|
||||
if self._closed:
|
||||
return False
|
||||
if os.name == 'nt':
|
||||
result = _WAIT_FOR_SINGLE_OBJECT(self._handle, 0)
|
||||
if result == 258:
|
||||
return True
|
||||
if result == 0:
|
||||
return False
|
||||
raise ctypes.WinError(ctypes.get_last_error())
|
||||
try:
|
||||
current = _posix_identity(self.pid)
|
||||
return current.creation_time == self.identity.creation_time
|
||||
except ProcessIdentityError:
|
||||
return False
|
||||
|
||||
def wait(self, timeout):
|
||||
timeout = max(0.0, float(timeout))
|
||||
if os.name == 'nt':
|
||||
milliseconds = min(int(timeout * 1000), 0xFFFFFFFE)
|
||||
result = _WAIT_FOR_SINGLE_OBJECT(self._handle, milliseconds)
|
||||
if result == 0:
|
||||
return True
|
||||
if result == 258:
|
||||
return False
|
||||
raise ctypes.WinError(ctypes.get_last_error())
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if not self.is_running():
|
||||
return True
|
||||
time.sleep(min(0.05, max(0.0, deadline - time.monotonic())))
|
||||
return not self.is_running()
|
||||
|
||||
def exit_code(self):
|
||||
if self._closed:
|
||||
raise ProcessIdentityError('retained process handle is closed')
|
||||
if os.name != 'nt':
|
||||
return None
|
||||
code = wintypes.DWORD()
|
||||
if not _GET_EXIT_CODE_PROCESS(self._handle, ctypes.byref(code)):
|
||||
raise ProcessIdentityError(f'unable to read process exit status: {ctypes.WinError(ctypes.get_last_error())}')
|
||||
if code.value == 259:
|
||||
return None
|
||||
return int(code.value)
|
||||
|
||||
def terminate(self):
|
||||
if self._closed:
|
||||
raise ProcessIdentityError('retained process handle is closed')
|
||||
if os.name == 'nt':
|
||||
if not _TERMINATE_PROCESS(self._handle, 1):
|
||||
raise ctypes.WinError(ctypes.get_last_error())
|
||||
return
|
||||
sender = getattr(signal, 'pidfd_send_signal', None)
|
||||
if self._pidfd is not None and sender is not None:
|
||||
sender(self._pidfd, signal.SIGTERM, None, 0)
|
||||
return
|
||||
current = _posix_identity(self.pid)
|
||||
if (
|
||||
current.creation_time != self.identity.creation_time
|
||||
or current.executable != self.identity.executable
|
||||
):
|
||||
raise ProcessIdentityError(f'process identity changed before signaling PID {self.pid}')
|
||||
os.kill(self.pid, signal.SIGTERM)
|
||||
|
||||
def command_line(self):
|
||||
if self._closed:
|
||||
raise ProcessIdentityError('retained process handle is closed')
|
||||
if os.name != 'nt':
|
||||
try:
|
||||
with open(f'/proc/{self.pid}/cmdline', 'rb') as handle:
|
||||
return [item.decode(errors='surrogateescape') for item in handle.read().split(b'\0') if item]
|
||||
except OSError as exc:
|
||||
raise ProcessIdentityError(f'unable to read process {self.pid} command line') from exc
|
||||
|
||||
needed = wintypes.ULONG()
|
||||
_NT_QUERY_INFORMATION_PROCESS(self._handle, 60, None, 0, ctypes.byref(needed))
|
||||
if not needed.value:
|
||||
raise ProcessIdentityError(f'unable to size process {self.pid} command line')
|
||||
buffer = ctypes.create_string_buffer(needed.value)
|
||||
status = _NT_QUERY_INFORMATION_PROCESS(
|
||||
self._handle, 60, buffer, needed.value, ctypes.byref(needed),
|
||||
)
|
||||
if status < 0:
|
||||
raise ProcessIdentityError(f'unable to read process {self.pid} command line (NTSTATUS 0x{status & 0xFFFFFFFF:08X})')
|
||||
value = ctypes.cast(buffer, _P_UNICODE_STRING).contents
|
||||
command = ctypes.wstring_at(value.Buffer, value.Length // ctypes.sizeof(ctypes.c_wchar))
|
||||
argc = ctypes.c_int()
|
||||
argv = _COMMAND_LINE_TO_ARGV(command, ctypes.byref(argc))
|
||||
if not argv:
|
||||
raise ProcessIdentityError(f'unable to parse process {self.pid} command line')
|
||||
try:
|
||||
return [argv[index] for index in range(argc.value)]
|
||||
finally:
|
||||
_LOCAL_FREE(argv)
|
||||
|
||||
def close(self):
|
||||
if self._closed:
|
||||
return
|
||||
self._closed = True
|
||||
if os.name == 'nt' and self._handle:
|
||||
_CLOSE_HANDLE(self._handle)
|
||||
elif self._pidfd is not None:
|
||||
try:
|
||||
os.close(self._pidfd)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, value, traceback):
|
||||
self.close()
|
||||
|
||||
def __del__(self):
|
||||
try:
|
||||
self.close()
|
||||
except BaseException:
|
||||
pass
|
||||
|
||||
|
||||
def _windows_identity(handle, pid):
|
||||
creation = _FILETIME()
|
||||
ignored_exit = _FILETIME()
|
||||
ignored_kernel = _FILETIME()
|
||||
ignored_user = _FILETIME()
|
||||
if not _GET_PROCESS_TIMES(
|
||||
handle, ctypes.byref(creation), ctypes.byref(ignored_exit),
|
||||
ctypes.byref(ignored_kernel), ctypes.byref(ignored_user),
|
||||
):
|
||||
raise ctypes.WinError(ctypes.get_last_error())
|
||||
filetime = (int(creation.dwHighDateTime) << 32) | int(creation.dwLowDateTime)
|
||||
path_buffer = ctypes.create_unicode_buffer(32768)
|
||||
path_size = wintypes.DWORD(len(path_buffer))
|
||||
if not _QUERY_FULL_PROCESS_IMAGE_NAME(handle, 0, path_buffer, ctypes.byref(path_size)):
|
||||
raise ctypes.WinError(ctypes.get_last_error())
|
||||
in_job = wintypes.BOOL()
|
||||
if not _IS_PROCESS_IN_JOB(handle, None, ctypes.byref(in_job)):
|
||||
raise ctypes.WinError(ctypes.get_last_error())
|
||||
unix_time = (filetime - 116444736000000000) / 10000000.0
|
||||
return ProcessIdentity(
|
||||
pid=int(pid),
|
||||
creation_time=f'windows-filetime:{filetime}',
|
||||
creation_time_unix=unix_time,
|
||||
executable=canonical_path(path_buffer.value),
|
||||
in_job=bool(in_job.value),
|
||||
)
|
||||
|
||||
|
||||
def _posix_identity(pid):
|
||||
stat_path = f'/proc/{int(pid)}/stat'
|
||||
try:
|
||||
with open(stat_path, 'r', encoding='ascii') as handle:
|
||||
value = handle.read()
|
||||
close_paren = value.rfind(')')
|
||||
fields = value[close_paren + 2:].split()
|
||||
start_ticks = int(fields[19])
|
||||
executable = canonical_path(os.readlink(f'/proc/{int(pid)}/exe'))
|
||||
clock_ticks = int(os.sysconf('SC_CLK_TCK'))
|
||||
boot_time = None
|
||||
with open('/proc/stat', 'r', encoding='ascii') as handle:
|
||||
for line in handle:
|
||||
if line.startswith('btime '):
|
||||
boot_time = float(line.split()[1])
|
||||
break
|
||||
if boot_time is None:
|
||||
raise ValueError('boot time unavailable')
|
||||
except (OSError, ValueError, IndexError) as exc:
|
||||
raise ProcessIdentityError(f'unable to inspect process {pid}') from exc
|
||||
return ProcessIdentity(
|
||||
pid=int(pid),
|
||||
creation_time=f'proc-start-ticks:{start_ticks}',
|
||||
creation_time_unix=boot_time + (start_ticks / float(clock_ticks)),
|
||||
executable=executable,
|
||||
in_job=False,
|
||||
)
|
||||
|
||||
|
||||
def _pidfd_live(pidfd):
|
||||
poller = select.poll()
|
||||
poller.register(pidfd, select.POLLIN)
|
||||
return not bool(poller.poll(0))
|
||||
|
||||
|
||||
def open_process(pid, *, terminate=False):
|
||||
pid = int(pid)
|
||||
if pid <= 0:
|
||||
raise ProcessIdentityError(f'invalid process ID: {pid}')
|
||||
if os.name == 'nt':
|
||||
rights = 0x00100000 | 0x00001000
|
||||
if terminate:
|
||||
rights |= 0x00000001
|
||||
handle = _OPEN_PROCESS(rights, False, pid)
|
||||
if not handle:
|
||||
native_error = ctypes.WinError(ctypes.get_last_error())
|
||||
raise ProcessIdentityError(f'unable to open process {pid}: {native_error}') from native_error
|
||||
try:
|
||||
wait_result = _WAIT_FOR_SINGLE_OBJECT(handle, 0)
|
||||
if wait_result == 0:
|
||||
exit_code = wintypes.DWORD()
|
||||
code = int(exit_code.value) if _GET_EXIT_CODE_PROCESS(
|
||||
handle, ctypes.byref(exit_code),
|
||||
) else -1
|
||||
raise ProcessExitedError(
|
||||
f'process {pid} has already exited with code {code}'
|
||||
)
|
||||
if wait_result != 258:
|
||||
raise ProcessIdentityError(
|
||||
f'unable to wait on process {pid}: {ctypes.WinError(ctypes.get_last_error())}'
|
||||
)
|
||||
exit_code = wintypes.DWORD()
|
||||
if not _GET_EXIT_CODE_PROCESS(handle, ctypes.byref(exit_code)):
|
||||
native_error = ctypes.WinError(ctypes.get_last_error())
|
||||
raise ProcessIdentityError(
|
||||
f'unable to read process {pid} exit status: {native_error}'
|
||||
) from native_error
|
||||
try:
|
||||
identity = _windows_identity(handle, pid)
|
||||
except OSError as exc:
|
||||
retry_exit_code = wintypes.DWORD()
|
||||
if (
|
||||
_GET_EXIT_CODE_PROCESS(handle, ctypes.byref(retry_exit_code))
|
||||
and retry_exit_code.value != 259
|
||||
):
|
||||
raise ProcessExitedError(
|
||||
f'process {pid} exited during identity inspection '
|
||||
f'with code {int(retry_exit_code.value)}'
|
||||
) from exc
|
||||
raise ProcessIdentityError(f'unable to inspect process {pid}') from exc
|
||||
final_wait = _WAIT_FOR_SINGLE_OBJECT(handle, 0)
|
||||
if final_wait == 0:
|
||||
raise ProcessExitedError(
|
||||
f'process {pid} exited during identity inspection'
|
||||
)
|
||||
if final_wait != 258:
|
||||
raise ProcessIdentityError(
|
||||
f'unable to confirm process {pid} liveness: '
|
||||
f'{ctypes.WinError(ctypes.get_last_error())}'
|
||||
)
|
||||
return RetainedProcess(identity, handle=handle)
|
||||
except BaseException:
|
||||
_CLOSE_HANDLE(handle)
|
||||
raise
|
||||
pidfd = None
|
||||
if hasattr(os, 'pidfd_open'):
|
||||
try:
|
||||
pidfd = os.pidfd_open(pid, 0)
|
||||
except ProcessLookupError as exc:
|
||||
raise ProcessExitedError(f'process {pid} has already exited') from exc
|
||||
except OSError as exc:
|
||||
raise ProcessIdentityError(
|
||||
f'unable to pin process {pid} with pidfd',
|
||||
) from exc
|
||||
try:
|
||||
if pidfd is not None and not _pidfd_live(pidfd):
|
||||
raise ProcessExitedError(f'process {pid} exited before identity binding')
|
||||
identity = _posix_identity(pid)
|
||||
if pidfd is not None and not _pidfd_live(pidfd):
|
||||
raise ProcessExitedError(f'process {pid} exited during identity binding')
|
||||
verified = _posix_identity(pid)
|
||||
if (
|
||||
verified.creation_time != identity.creation_time
|
||||
or verified.executable != identity.executable
|
||||
):
|
||||
raise ProcessIdentityError(
|
||||
f'process {pid} identity changed during pidfd binding',
|
||||
)
|
||||
if pidfd is not None and not _pidfd_live(pidfd):
|
||||
raise ProcessExitedError(f'process {pid} exited after identity binding')
|
||||
return RetainedProcess(identity, pidfd=pidfd)
|
||||
except BaseException:
|
||||
if pidfd is not None:
|
||||
os.close(pidfd)
|
||||
raise
|
||||
|
||||
|
||||
def current_process_identity():
|
||||
if os.name == 'nt':
|
||||
return _windows_identity(_GET_CURRENT_PROCESS(), os.getpid())
|
||||
return _posix_identity(os.getpid())
|
||||
|
||||
|
||||
def verify_retained_process(pid, creation_time, executable, *, terminate=False):
|
||||
process = open_process(pid, terminate=True) if terminate else open_process(pid)
|
||||
expected_executable = canonical_path(executable)
|
||||
if process.identity.creation_time != str(creation_time) or process.identity.executable != expected_executable:
|
||||
process.close()
|
||||
raise ProcessIdentityError(f'process identity mismatch for PID {pid}')
|
||||
return process
|
||||
|
||||
|
||||
def serialize_process_identity(identity):
|
||||
if isinstance(identity, ProcessIdentity):
|
||||
return identity.as_dict()
|
||||
raise TypeError('expected ProcessIdentity')
|
||||
|
||||
|
||||
def exact_process_identity_state(pid, creation_time, executable):
|
||||
"""Return alive, dead, reused, or unknown without PID-only inference."""
|
||||
try:
|
||||
pid = int(pid)
|
||||
except (TypeError, ValueError):
|
||||
return 'unknown'
|
||||
if pid <= 0 or not creation_time or not executable:
|
||||
return 'unknown'
|
||||
try:
|
||||
process = open_process(pid)
|
||||
except ProcessExitedError:
|
||||
return 'dead'
|
||||
except ProcessIdentityError as exc:
|
||||
cause = exc.__cause__
|
||||
winerror = getattr(cause, 'winerror', None) or getattr(exc, 'winerror', None)
|
||||
errno_value = getattr(cause, 'errno', None) or getattr(exc, 'errno', None)
|
||||
if os.name == 'nt' and winerror in (87, 1168):
|
||||
return 'dead'
|
||||
if os.name != 'nt' and errno_value in (2, 3):
|
||||
return 'dead'
|
||||
return 'unknown'
|
||||
try:
|
||||
if not process.is_running():
|
||||
return 'dead'
|
||||
if (
|
||||
str(process.identity.creation_time) != str(creation_time)
|
||||
or canonical_path(process.identity.executable) != canonical_path(executable)
|
||||
):
|
||||
return 'reused'
|
||||
return 'alive'
|
||||
except (OSError, ValueError):
|
||||
return 'unknown'
|
||||
finally:
|
||||
process.close()
|
||||
@@ -0,0 +1,126 @@
|
||||
from datetime import datetime
|
||||
import re
|
||||
|
||||
|
||||
REJECTED_QUERY_STATUS = 'rejected_zero_alive'
|
||||
REJECTED_QUERY_KEYS = {
|
||||
'source', 'query', 'status', 'evidence_cutoff', 'successful_scans',
|
||||
'findings', 'unique_credentials', 'pending_candidates',
|
||||
'ever_alive_credentials', 'reviewed_queue_rows',
|
||||
}
|
||||
REJECTED_QUERY_COUNT_KEYS = {
|
||||
'successful_scans', 'findings', 'unique_credentials',
|
||||
'pending_candidates', 'ever_alive_credentials', 'reviewed_queue_rows',
|
||||
}
|
||||
OPERATIONAL_QUERY_SENTINELS = {
|
||||
('github_archive', 'gharchive'),
|
||||
('github_archive_files', 'gharchive-files'),
|
||||
('github_gists', 'gists'),
|
||||
('github_actions', 'logs'),
|
||||
('gitlab_ci', 'logs'),
|
||||
('huggingface', 'spaces'),
|
||||
}
|
||||
SOURCE_RE = re.compile(r'[a-z][a-z0-9_]{0,63}')
|
||||
MAX_QUERY_LENGTH = 512
|
||||
MAX_EVIDENCE_COUNT = (1 << 63) - 1
|
||||
|
||||
|
||||
class QueryPolicyError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def _active_queries(source, source_config):
|
||||
raw_queries = source_config.get('queries', [])
|
||||
if isinstance(raw_queries, str):
|
||||
raw_queries = raw_queries.split(',')
|
||||
if not isinstance(raw_queries, (list, tuple)):
|
||||
raise QueryPolicyError(f'active query policy is invalid for source {source}')
|
||||
queries = set()
|
||||
for raw_query in raw_queries:
|
||||
query = str(raw_query or '').strip()
|
||||
if not query:
|
||||
raise QueryPolicyError(f'active query policy contains an empty query for source {source}')
|
||||
queries.add(query)
|
||||
return queries
|
||||
|
||||
|
||||
def validate_rejected_query_policy(config):
|
||||
config = config or {}
|
||||
raw_policy = config.get('query_policy')
|
||||
if raw_policy is None:
|
||||
return ()
|
||||
if not isinstance(raw_policy, dict) or set(raw_policy) != {'rejected'}:
|
||||
raise QueryPolicyError('query_policy must contain only the rejected registry')
|
||||
raw_entries = raw_policy.get('rejected')
|
||||
if not isinstance(raw_entries, list):
|
||||
raise QueryPolicyError('query_policy.rejected must be a list')
|
||||
|
||||
sources = config.get('sources') or {}
|
||||
if not isinstance(sources, dict):
|
||||
raise QueryPolicyError('configured sources must be a mapping')
|
||||
|
||||
normalized = []
|
||||
seen = set()
|
||||
for raw_entry in raw_entries:
|
||||
if not isinstance(raw_entry, dict) or set(raw_entry) != REJECTED_QUERY_KEYS:
|
||||
raise QueryPolicyError('rejected query evidence shape is invalid')
|
||||
source = raw_entry.get('source')
|
||||
query = raw_entry.get('query')
|
||||
if not isinstance(source, str) or not SOURCE_RE.fullmatch(source):
|
||||
raise QueryPolicyError('rejected query source is invalid')
|
||||
if source not in sources or not isinstance(sources[source], dict):
|
||||
raise QueryPolicyError(f'rejected query source is not configured: {source}')
|
||||
if (
|
||||
not isinstance(query, str)
|
||||
or query != query.strip()
|
||||
or not query
|
||||
or len(query) > MAX_QUERY_LENGTH
|
||||
):
|
||||
raise QueryPolicyError(f'rejected query text is invalid for source {source}')
|
||||
pair = (source, query)
|
||||
if pair in seen:
|
||||
raise QueryPolicyError('rejected query registry contains a duplicate pair')
|
||||
if pair in OPERATIONAL_QUERY_SENTINELS:
|
||||
raise QueryPolicyError('operational query sentinel cannot be rejected')
|
||||
if query in _active_queries(source, sources[source]):
|
||||
raise QueryPolicyError('active and rejected query policy overlap')
|
||||
if raw_entry.get('status') != REJECTED_QUERY_STATUS:
|
||||
raise QueryPolicyError('rejected query status is invalid')
|
||||
|
||||
cutoff = raw_entry.get('evidence_cutoff')
|
||||
if not isinstance(cutoff, str) or not cutoff or cutoff != cutoff.strip():
|
||||
raise QueryPolicyError('rejected query evidence cutoff is invalid')
|
||||
try:
|
||||
parsed_cutoff = datetime.fromisoformat(cutoff.replace('Z', '+00:00'))
|
||||
except ValueError as exc:
|
||||
raise QueryPolicyError('rejected query evidence cutoff is invalid') from exc
|
||||
if parsed_cutoff.tzinfo is None or parsed_cutoff.utcoffset() is None:
|
||||
raise QueryPolicyError('rejected query evidence cutoff must include a timezone')
|
||||
|
||||
counts = {}
|
||||
for name in REJECTED_QUERY_COUNT_KEYS:
|
||||
value = raw_entry.get(name)
|
||||
if (
|
||||
isinstance(value, bool)
|
||||
or not isinstance(value, int)
|
||||
or value < 0
|
||||
or value > MAX_EVIDENCE_COUNT
|
||||
):
|
||||
raise QueryPolicyError(f'rejected query {name} is invalid')
|
||||
counts[name] = value
|
||||
if counts['successful_scans'] < 1000:
|
||||
raise QueryPolicyError('rejected query has fewer than 1000 successful scans')
|
||||
if counts['pending_candidates'] != 0:
|
||||
raise QueryPolicyError('rejected query still has pending candidates')
|
||||
if counts['ever_alive_credentials'] != 0:
|
||||
raise QueryPolicyError('rejected query has historical alive credentials')
|
||||
|
||||
seen.add(pair)
|
||||
normalized.append({
|
||||
'source': source,
|
||||
'query': query,
|
||||
'status': REJECTED_QUERY_STATUS,
|
||||
'evidence_cutoff': cutoff,
|
||||
**{name: counts[name] for name in sorted(REJECTED_QUERY_COUNT_KEYS)},
|
||||
})
|
||||
return tuple(sorted(normalized, key=lambda entry: (entry['source'], entry['query'])))
|
||||
@@ -0,0 +1,162 @@
|
||||
"""Stdlib-only integrity boundary for packaged remote worker clients."""
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import runpy
|
||||
import stat
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
if not sys.dont_write_bytecode:
|
||||
raise RuntimeError('worker bootstrap could not disable bytecode writes')
|
||||
|
||||
|
||||
MAX_MANIFEST_BYTES = 1024 * 1024
|
||||
MAX_MANIFEST_FILES = 512
|
||||
WORKER_PACKAGE_SCHEMA = 3
|
||||
WORKER_PROTOCOL_VERSION = 2
|
||||
|
||||
|
||||
def _canonical(path):
|
||||
return os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(path))))
|
||||
|
||||
|
||||
def _is_reparse_point(path):
|
||||
details = os.lstat(path)
|
||||
if stat.S_ISLNK(details.st_mode):
|
||||
return True
|
||||
attributes = getattr(details, 'st_file_attributes', 0)
|
||||
reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0)
|
||||
return bool(attributes & reparse_attribute) or getattr(
|
||||
os.path, 'isjunction', lambda _path: False,
|
||||
)(path)
|
||||
|
||||
|
||||
def _relative(value, label):
|
||||
value = str(value or '')
|
||||
if (
|
||||
not value or len(value) > 512 or '\\' in value or '\x00' in value
|
||||
or value.startswith('/') or value.endswith('/')
|
||||
):
|
||||
raise RuntimeError(f'invalid worker package {label} path')
|
||||
if any(part in ('', '.', '..') for part in value.split('/')):
|
||||
raise RuntimeError(f'invalid worker package {label} path')
|
||||
return value
|
||||
|
||||
|
||||
def _sha256(path):
|
||||
digest = hashlib.sha256()
|
||||
with open(path, 'rb', buffering=0) as handle:
|
||||
for block in iter(lambda: handle.read(1024 * 1024), b''):
|
||||
digest.update(block)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def _load_manifest(path):
|
||||
details = os.stat(path, follow_symlinks=False)
|
||||
if _is_reparse_point(path) or not stat.S_ISREG(details.st_mode):
|
||||
raise RuntimeError('worker package manifest is not a regular file')
|
||||
with open(path, 'rb') as handle:
|
||||
payload = handle.read(MAX_MANIFEST_BYTES + 1)
|
||||
if len(payload) > MAX_MANIFEST_BYTES:
|
||||
raise RuntimeError('worker package manifest exceeds its byte bound')
|
||||
value = json.loads(payload.decode('utf-8', errors='strict'))
|
||||
if (
|
||||
not isinstance(value, dict)
|
||||
or type(value.get('schema')) is not int
|
||||
or value['schema'] != WORKER_PACKAGE_SCHEMA
|
||||
or type(value.get('protocol_version')) is not int
|
||||
or value['protocol_version'] != WORKER_PROTOCOL_VERSION
|
||||
):
|
||||
raise RuntimeError('worker package manifest is invalid')
|
||||
return value
|
||||
|
||||
|
||||
def _application_files(app_dir):
|
||||
files = set()
|
||||
|
||||
def raise_walk_error(exc):
|
||||
raise RuntimeError(f'unable to inspect worker application root: {exc}') from exc
|
||||
|
||||
for current, directories, names in os.walk(
|
||||
app_dir, followlinks=False, onerror=raise_walk_error,
|
||||
):
|
||||
for name in directories:
|
||||
candidate = os.path.join(current, name)
|
||||
if _is_reparse_point(candidate) or name.lower() == '__pycache__':
|
||||
raise RuntimeError('worker application directory is unsupported')
|
||||
for name in names:
|
||||
candidate = os.path.join(current, name)
|
||||
details = os.stat(candidate, follow_symlinks=False)
|
||||
if _is_reparse_point(candidate) or not stat.S_ISREG(details.st_mode):
|
||||
raise RuntimeError('worker application file is not regular')
|
||||
files.add(os.path.relpath(candidate, app_dir).replace(os.sep, '/'))
|
||||
return files
|
||||
|
||||
|
||||
def _verify_application(package_root, manifest):
|
||||
app_root = _relative(manifest.get('app_root'), 'application root')
|
||||
app_dir = _canonical(os.path.join(package_root, *app_root.split('/')))
|
||||
if app_dir != _canonical(os.path.dirname(__file__)) or _is_reparse_point(app_dir):
|
||||
raise RuntimeError('worker package application root is not canonical')
|
||||
values = manifest.get('files')
|
||||
if not isinstance(values, dict) or not 1 <= len(values) <= MAX_MANIFEST_FILES:
|
||||
raise RuntimeError('worker package file set is invalid')
|
||||
expected = set()
|
||||
for name, entry in values.items():
|
||||
name = _relative(name, 'file name')
|
||||
if not isinstance(entry, dict) or set(entry) != {'path', 'sha256'}:
|
||||
raise RuntimeError('worker package file entry is invalid')
|
||||
relative = _relative(entry.get('path'), f'file {name}')
|
||||
if relative != f'{app_root}/{name}':
|
||||
raise RuntimeError('worker package file path is not canonical')
|
||||
digest = str(entry.get('sha256') or '')
|
||||
if len(digest) != 64 or any(ch not in '0123456789abcdef' for ch in digest):
|
||||
raise RuntimeError('worker package file digest is invalid')
|
||||
path = _canonical(os.path.join(package_root, *relative.split('/')))
|
||||
try:
|
||||
contained = os.path.commonpath((app_dir, path)) == app_dir
|
||||
except ValueError:
|
||||
contained = False
|
||||
if not contained or _is_reparse_point(path) or _sha256(path) != digest:
|
||||
raise RuntimeError('worker package application integrity check failed')
|
||||
expected.add(name)
|
||||
if _application_files(app_dir) != expected:
|
||||
raise RuntimeError('worker package application file set drifted')
|
||||
return app_dir
|
||||
|
||||
|
||||
def main():
|
||||
if not (
|
||||
sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode
|
||||
):
|
||||
raise RuntimeError(
|
||||
'remote worker bootstrap requires isolated no-site bytecode-free startup (-I -S -B)'
|
||||
)
|
||||
sys.dont_write_bytecode = True
|
||||
if len(sys.argv) < 2 or sys.argv[1] != '--':
|
||||
raise RuntimeError('usage: remote_worker_bootstrap.py -- <worker args>')
|
||||
|
||||
package_root = _canonical(os.path.dirname(os.path.dirname(__file__)))
|
||||
manifest = _load_manifest(os.path.join(package_root, 'worker-package.json'))
|
||||
app_dir = _verify_application(package_root, manifest)
|
||||
dependency_dir = os.path.join(app_dir, 'dependencies')
|
||||
if (
|
||||
not os.path.isdir(dependency_dir) or _is_reparse_point(dependency_dir)
|
||||
or _canonical(dependency_dir) == app_dir
|
||||
):
|
||||
raise RuntimeError('package-local worker dependencies are unavailable')
|
||||
|
||||
entrypoint = os.path.join(app_dir, 'worker_cli.py')
|
||||
sys.path.insert(0, dependency_dir)
|
||||
sys.path.insert(0, app_dir)
|
||||
sys.argv = [entrypoint, *sys.argv[2:]]
|
||||
runpy.run_path(entrypoint, run_name='__main__')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
main()
|
||||
except Exception as exc:
|
||||
raise SystemExit('remote worker bootstrap rejected launch') from exc
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,4 @@
|
||||
requests
|
||||
boto3
|
||||
botocore
|
||||
psycopg[binary]>=3.2
|
||||
@@ -0,0 +1,10 @@
|
||||
streamlit
|
||||
starlette>=0.47.3,<1
|
||||
uvicorn>=0.53,<1
|
||||
python-multipart>=0.0.10
|
||||
pandas
|
||||
requests
|
||||
plotly
|
||||
PyYAML
|
||||
psycopg[binary]>=3.2
|
||||
zstandard==0.23.0
|
||||
@@ -0,0 +1,757 @@
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import struct
|
||||
from dataclasses import dataclass
|
||||
|
||||
from runtime_security import (
|
||||
PrivatePathState,
|
||||
durable_publish,
|
||||
ensure_private_directory,
|
||||
harden_private_file,
|
||||
inspect_private_relative_path,
|
||||
private_file_ready,
|
||||
reject_reparse_components,
|
||||
require_private_directory,
|
||||
)
|
||||
from worker_contracts import (
|
||||
AssignmentOutcome,
|
||||
MAX_DIAGNOSTIC_AGGREGATE_BYTES,
|
||||
MAX_DIAGNOSTICS_PER_ASSIGNMENT,
|
||||
decode_diagnostic_envelope,
|
||||
build_legacy_error_frame_diagnostics,
|
||||
encode_diagnostic_envelope,
|
||||
)
|
||||
|
||||
|
||||
MAGIC = b'TRUF-RB2\n'
|
||||
FORMAT_VERSION = 2
|
||||
FRAME_HEADER = struct.Struct('!cI')
|
||||
FRAME_TYPES = frozenset((b'H', b'F', b'E', b'D', b'K', b'M', b'C'))
|
||||
FRAME_ORDER = {name: index for index, name in enumerate((b'H', b'F', b'E', b'D', b'K', b'M', b'C'))}
|
||||
ID_RE = re.compile(r'^[a-f0-9]{32,64}$')
|
||||
DEFAULT_MAX_EVENT_BYTES = 64 * 1024 * 1024
|
||||
DEFAULT_MAX_FRAME_BYTES = 16 * 1024 * 1024
|
||||
FRAME_BOUNDS = {
|
||||
b'H': 1024 * 1024,
|
||||
b'F': 16 * 1024 * 1024,
|
||||
b'E': 1024 * 1024,
|
||||
b'D': 64 * 1024,
|
||||
b'K': 2 * 1024 * 1024,
|
||||
b'M': 16 * 1024 * 1024,
|
||||
b'C': 1024 * 1024,
|
||||
}
|
||||
MAX_FINDING_FRAMES = 20000
|
||||
MAX_ERROR_FRAMES = 2000
|
||||
MAX_CANDIDATE_FRAMES = 2000
|
||||
MAX_TOTAL_FRAMES = 24003 + MAX_DIAGNOSTICS_PER_ASSIGNMENT
|
||||
|
||||
|
||||
class ResultBundleError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
class ResultBundleConflictError(ResultBundleError):
|
||||
pass
|
||||
|
||||
|
||||
class ResultBundleUnavailableError(OSError):
|
||||
pass
|
||||
|
||||
|
||||
def canonical_json_bytes(value):
|
||||
try:
|
||||
return json.dumps(
|
||||
value, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
|
||||
).encode('utf-8')
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise ResultBundleError('bundle frame is not canonical JSON data') from exc
|
||||
|
||||
|
||||
def _validated_id(value, name):
|
||||
text = str(value or '').lower()
|
||||
if not ID_RE.fullmatch(text):
|
||||
raise ResultBundleError(f'invalid {name}')
|
||||
return text
|
||||
|
||||
|
||||
def _relative_ready_path(bundle_id):
|
||||
bundle_id = _validated_id(bundle_id, 'bundle_id')
|
||||
return os.path.join('ready', bundle_id[:2], f'{bundle_id}.trb')
|
||||
|
||||
|
||||
def bundle_ready_path(root, bundle_id):
|
||||
return os.path.join(os.path.abspath(root), _relative_ready_path(bundle_id))
|
||||
|
||||
|
||||
def bundle_partial_relative_path(bundle_id, reservation_token):
|
||||
bundle_id = _validated_id(bundle_id, 'bundle_id')
|
||||
producer_token = hashlib.sha256(str(reservation_token).encode('utf-8')).hexdigest()[:24]
|
||||
return os.path.join('tmp', bundle_id[:2], f'{bundle_id}.{producer_token}.partial')
|
||||
|
||||
|
||||
def bundle_partial_path(root, bundle_id, reservation_token):
|
||||
return os.path.join(
|
||||
os.path.abspath(root), bundle_partial_relative_path(bundle_id, reservation_token),
|
||||
)
|
||||
|
||||
|
||||
def ensure_bundle_reservation_paths(root, reservation):
|
||||
root = require_private_directory(os.path.abspath(root), create=False)
|
||||
reservation = (
|
||||
reservation if isinstance(reservation, BundleReservation)
|
||||
else BundleReservation.from_mapping(reservation)
|
||||
)
|
||||
for name in ('tmp', 'ready', 'quarantine'):
|
||||
ensure_private_directory(
|
||||
os.path.join(root, name, reservation.bundle_id[:2]),
|
||||
reject_reparse=True,
|
||||
)
|
||||
return reservation
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BundleReservation:
|
||||
reservation_id: int
|
||||
reservation_token: str
|
||||
bundle_id: str
|
||||
scan_event_id: str
|
||||
queue_id: int
|
||||
claim_lease_token: str
|
||||
declared_bytes: int
|
||||
ready_path: str
|
||||
source: str = ''
|
||||
platform: str = ''
|
||||
query: str = ''
|
||||
target: str = ''
|
||||
normalized_target: str = ''
|
||||
run_id: int | None = None
|
||||
cycle_id: int | None = None
|
||||
producer_instance_id: str = ''
|
||||
producer_pid: int = 0
|
||||
producer_creation_time: str = ''
|
||||
producer_executable: str = ''
|
||||
|
||||
@classmethod
|
||||
def from_mapping(cls, value):
|
||||
data = dict(value or {})
|
||||
ready_path = data.get('ready_path') or data.get('ready_relative_path') or ''
|
||||
return cls(
|
||||
reservation_id=int(data['reservation_id'] if 'reservation_id' in data else data['id']),
|
||||
reservation_token=str(data['reservation_token']),
|
||||
bundle_id=_validated_id(data['bundle_id'], 'bundle_id'),
|
||||
scan_event_id=_validated_id(data['scan_event_id'], 'scan_event_id'),
|
||||
queue_id=int(data['queue_id']),
|
||||
claim_lease_token=str(data.get('claim_lease_token') or data.get('claim_lease_token_value') or ''),
|
||||
declared_bytes=int(data.get('declared_bytes') or data.get('declared_bundle_bytes') or 0),
|
||||
ready_path=str(ready_path),
|
||||
source=str(data.get('source') or ''),
|
||||
platform=str(data.get('platform') or ''),
|
||||
query=str(data.get('query') or ''),
|
||||
target=str(data.get('target') or ''),
|
||||
normalized_target=str(data.get('normalized_target') or ''),
|
||||
run_id=data.get('run_id'),
|
||||
cycle_id=data.get('cycle_id'),
|
||||
producer_instance_id=str(data.get('producer_instance_id') or ''),
|
||||
producer_pid=int(data.get('producer_pid') or 0),
|
||||
producer_creation_time=str(data.get('producer_creation_time') or ''),
|
||||
producer_executable=str(data.get('producer_executable') or ''),
|
||||
)
|
||||
|
||||
def header(self):
|
||||
return {
|
||||
'format_version': FORMAT_VERSION,
|
||||
'reservation_id': self.reservation_id,
|
||||
'reservation_token': self.reservation_token,
|
||||
'bundle_id': self.bundle_id,
|
||||
'scan_event_id': self.scan_event_id,
|
||||
'queue_id': self.queue_id,
|
||||
'claim_lease_token': self.claim_lease_token,
|
||||
'declared_bytes': self.declared_bytes,
|
||||
'ready_relative_path': self.ready_path.replace('\\', '/'),
|
||||
'source': self.source,
|
||||
'platform': self.platform,
|
||||
'query': self.query,
|
||||
'target': self.target,
|
||||
'normalized_target': self.normalized_target,
|
||||
'run_id': self.run_id,
|
||||
'cycle_id': self.cycle_id,
|
||||
'producer_instance_id': self.producer_instance_id,
|
||||
'producer_pid': self.producer_pid,
|
||||
'producer_creation_time': self.producer_creation_time,
|
||||
'producer_executable': self.producer_executable,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BundleCommit:
|
||||
reservation_id: int
|
||||
bundle_id: str
|
||||
scan_event_id: str
|
||||
scan_event_hash: str
|
||||
relative_path: str
|
||||
actual_bytes: int
|
||||
frame_count: int
|
||||
finding_count: int
|
||||
error_count: int
|
||||
candidate_count: int
|
||||
|
||||
def as_dict(self):
|
||||
return dict(self.__dict__)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BundleMetadata(BundleCommit):
|
||||
header: dict
|
||||
result_metadata: dict
|
||||
diagnostic_count: int
|
||||
diagnostic_bytes: int
|
||||
|
||||
|
||||
class ResultBundleWriter:
|
||||
def __init__(self, root, reservation, handle, partial_path, ready_path, fault=None):
|
||||
self.root = root
|
||||
self.reservation = reservation
|
||||
self.handle = handle
|
||||
self.partial_path = partial_path
|
||||
self.ready_path = ready_path
|
||||
self.fault = fault
|
||||
self.digest = hashlib.sha256()
|
||||
self.bytes_written = 0
|
||||
self.frame_count = 0
|
||||
self.finding_count = 0
|
||||
self.error_count = 0
|
||||
self.candidate_count = 0
|
||||
self.diagnostic_count = 0
|
||||
self.diagnostic_bytes = 0
|
||||
self.diagnostic_aggregate_bytes = 0
|
||||
self._diagnostic_uids = set()
|
||||
self._last_frame_order = FRAME_ORDER[b'H']
|
||||
self.finished = False
|
||||
self.ready_published = False
|
||||
self._write_bytes(MAGIC)
|
||||
self._write_frame(b'H', reservation.header())
|
||||
|
||||
@classmethod
|
||||
def open(cls, root, reservation, fault=None, require_s_drive=False):
|
||||
root = require_private_directory(os.path.abspath(root), create=False)
|
||||
drive = os.path.splitdrive(root)[0].upper()
|
||||
if require_s_drive and drive != 'S:':
|
||||
raise ResultBundleError('production result bundle root must be on S:')
|
||||
reservation = ensure_bundle_reservation_paths(root, reservation)
|
||||
if reservation.declared_bytes <= len(MAGIC) or reservation.declared_bytes > DEFAULT_MAX_EVENT_BYTES:
|
||||
raise ResultBundleError('declared bundle byte bound is invalid')
|
||||
expected_relative = _relative_ready_path(reservation.bundle_id)
|
||||
supplied_relative = str(reservation.ready_path or expected_relative).replace('/', os.sep)
|
||||
if os.path.normcase(os.path.normpath(supplied_relative)) != os.path.normcase(os.path.normpath(expected_relative)):
|
||||
raise ResultBundleError('reservation ready path is not deterministic for its bundle ID')
|
||||
tmp_dir = os.path.join(root, 'tmp', reservation.bundle_id[:2])
|
||||
ready_dir = os.path.join(root, 'ready', reservation.bundle_id[:2])
|
||||
partial_path = bundle_partial_path(
|
||||
root, reservation.bundle_id, reservation.reservation_token,
|
||||
)
|
||||
ready_path = os.path.join(ready_dir, f'{reservation.bundle_id}.trb')
|
||||
if os.path.lexists(ready_path):
|
||||
raise ResultBundleConflictError('deterministic ready bundle path already exists')
|
||||
descriptor = os.open(
|
||||
partial_path,
|
||||
os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0),
|
||||
0o600,
|
||||
)
|
||||
try:
|
||||
os.close(descriptor)
|
||||
descriptor = None
|
||||
harden_private_file(partial_path)
|
||||
handle = open(partial_path, 'w+b', buffering=0)
|
||||
return cls(root, reservation, handle, partial_path, ready_path, fault=fault)
|
||||
except BaseException:
|
||||
if descriptor is not None:
|
||||
os.close(descriptor)
|
||||
try:
|
||||
os.remove(partial_path)
|
||||
except OSError:
|
||||
pass
|
||||
raise
|
||||
|
||||
def _inject(self, stage):
|
||||
if self.fault is not None:
|
||||
self.fault(stage, self)
|
||||
|
||||
def _write_bytes(self, payload):
|
||||
if self.bytes_written + len(payload) > self.reservation.declared_bytes:
|
||||
raise ResultBundleError('bundle exceeded its pre-reserved byte bound')
|
||||
self.handle.write(payload)
|
||||
self.digest.update(payload)
|
||||
self.bytes_written += len(payload)
|
||||
|
||||
def _write_frame(self, frame_type, value):
|
||||
if self.finished or frame_type not in FRAME_TYPES or frame_type == b'C':
|
||||
raise ResultBundleError('invalid bundle frame write')
|
||||
if self.frame_count >= MAX_TOTAL_FRAMES - 1:
|
||||
raise ResultBundleError('bundle frame count exceeds its bound')
|
||||
if FRAME_ORDER[frame_type] < self._last_frame_order:
|
||||
raise ResultBundleError('bundle frame order is not deterministic')
|
||||
payload = canonical_json_bytes(value)
|
||||
bound = min(FRAME_BOUNDS[frame_type], self.reservation.declared_bytes)
|
||||
if len(payload) > bound:
|
||||
raise ResultBundleError(f'{frame_type.decode()} frame exceeds its byte bound')
|
||||
framed = FRAME_HEADER.pack(frame_type, len(payload)) + payload
|
||||
self._inject(f'before_frame_{frame_type.decode()}')
|
||||
self._write_bytes(framed)
|
||||
self.frame_count += 1
|
||||
self._last_frame_order = FRAME_ORDER[frame_type]
|
||||
self._inject(f'after_frame_{frame_type.decode()}')
|
||||
return len(payload)
|
||||
|
||||
def write_finding(self, finding):
|
||||
if self.finding_count >= MAX_FINDING_FRAMES:
|
||||
raise ResultBundleError('bundle finding count exceeds its bound')
|
||||
self._write_frame(b'F', finding)
|
||||
self.finding_count += 1
|
||||
|
||||
def write_error(self, error):
|
||||
if self.error_count >= MAX_ERROR_FRAMES:
|
||||
raise ResultBundleError('bundle error count exceeds its bound')
|
||||
self._write_frame(b'E', {'error': str(error)})
|
||||
self.error_count += 1
|
||||
|
||||
def write_diagnostic(self, diagnostic):
|
||||
if self.diagnostic_count >= MAX_DIAGNOSTICS_PER_ASSIGNMENT:
|
||||
raise ResultBundleError('bundle diagnostic count exceeds its bound')
|
||||
try:
|
||||
if isinstance(diagnostic, dict):
|
||||
envelope = decode_diagnostic_envelope(canonical_json_bytes(diagnostic))
|
||||
else:
|
||||
envelope = decode_diagnostic_envelope(
|
||||
encode_diagnostic_envelope(diagnostic)
|
||||
)
|
||||
payload = encode_diagnostic_envelope(envelope)
|
||||
value = json.loads(payload.decode('ascii'))
|
||||
except (TypeError, ValueError, UnicodeError) as exc:
|
||||
raise ResultBundleError('bundle diagnostic frame is invalid') from exc
|
||||
if envelope.diagnostic_uid in self._diagnostic_uids:
|
||||
raise ResultBundleError('bundle diagnostic identity is duplicated')
|
||||
if (
|
||||
envelope.reservation_id != self.reservation.reservation_id
|
||||
or envelope.scan_event_id != self.reservation.scan_event_id
|
||||
or envelope.source != self.reservation.source
|
||||
):
|
||||
raise ResultBundleError('bundle diagnostic identity conflicts with its reservation')
|
||||
if (
|
||||
self.diagnostic_aggregate_bytes + len(payload) + 1
|
||||
> MAX_DIAGNOSTIC_AGGREGATE_BYTES
|
||||
):
|
||||
raise ResultBundleError('bundle diagnostic aggregate exceeds its byte bound')
|
||||
self._write_frame(b'D', value)
|
||||
self._diagnostic_uids.add(envelope.diagnostic_uid)
|
||||
self.diagnostic_count += 1
|
||||
self.diagnostic_bytes += FRAME_HEADER.size + len(payload)
|
||||
self.diagnostic_aggregate_bytes += len(payload) + 1
|
||||
|
||||
def write_candidate(self, candidate):
|
||||
if self.candidate_count >= MAX_CANDIDATE_FRAMES:
|
||||
raise ResultBundleError('bundle candidate count exceeds its bound')
|
||||
self._write_frame(b'K', candidate)
|
||||
self.candidate_count += 1
|
||||
|
||||
def finish(self, metadata):
|
||||
if self.finished:
|
||||
raise ResultBundleError('bundle writer is already finished')
|
||||
self._write_frame(b'M', metadata)
|
||||
content_hash = self.digest.hexdigest()
|
||||
footer = {
|
||||
'format_version': FORMAT_VERSION,
|
||||
'content_sha256': content_hash,
|
||||
'content_bytes': self.bytes_written,
|
||||
'byte_count': 0,
|
||||
'frame_count': self.frame_count + 1,
|
||||
'finding_count': self.finding_count,
|
||||
'error_count': self.error_count,
|
||||
'candidate_count': self.candidate_count,
|
||||
}
|
||||
if self.diagnostic_count:
|
||||
footer.update({
|
||||
'diagnostic_count': self.diagnostic_count,
|
||||
'diagnostic_bytes': self.diagnostic_bytes,
|
||||
})
|
||||
while True:
|
||||
payload = canonical_json_bytes(footer)
|
||||
framed = FRAME_HEADER.pack(b'C', len(payload)) + payload
|
||||
total = self.bytes_written + len(framed)
|
||||
if footer['byte_count'] == total:
|
||||
break
|
||||
footer['byte_count'] = total
|
||||
if total > self.reservation.declared_bytes:
|
||||
raise ResultBundleError('bundle footer exceeds its pre-reserved byte bound')
|
||||
self._inject('before_footer')
|
||||
self.handle.write(framed)
|
||||
self.bytes_written = total
|
||||
self.frame_count += 1
|
||||
self._inject('after_footer')
|
||||
self._inject('before_fsync')
|
||||
self.handle.flush()
|
||||
os.fsync(self.handle.fileno())
|
||||
self._inject('after_fsync')
|
||||
self.handle.close()
|
||||
self.handle = None
|
||||
if not private_file_ready(self.partial_path):
|
||||
raise ResultBundleError('private bundle ACL verification failed before publication')
|
||||
if os.path.getsize(self.partial_path) != self.bytes_written:
|
||||
raise ResultBundleError('bundle size changed before publication')
|
||||
self._inject('before_rename')
|
||||
durable_publish(self.partial_path, self.ready_path)
|
||||
self.ready_published = True
|
||||
self._inject('after_rename')
|
||||
if not private_file_ready(self.ready_path):
|
||||
inspection = inspect_private_relative_path(
|
||||
self.root, os.path.relpath(self.ready_path, self.root),
|
||||
)
|
||||
if inspection.state == PrivatePathState.UNKNOWN:
|
||||
raise ResultBundleUnavailableError(
|
||||
'ready bundle state is unavailable after publication'
|
||||
)
|
||||
if inspection.state == PrivatePathState.PRESENT:
|
||||
raise ResultBundleError('ready bundle ACL verification failed')
|
||||
self.finished = True
|
||||
relative = os.path.relpath(self.ready_path, self.root)
|
||||
return BundleCommit(
|
||||
reservation_id=self.reservation.reservation_id,
|
||||
bundle_id=self.reservation.bundle_id,
|
||||
scan_event_id=self.reservation.scan_event_id,
|
||||
scan_event_hash=content_hash,
|
||||
relative_path=relative.replace(os.sep, '/'),
|
||||
actual_bytes=self.bytes_written,
|
||||
frame_count=self.frame_count,
|
||||
finding_count=self.finding_count,
|
||||
error_count=self.error_count,
|
||||
candidate_count=self.candidate_count,
|
||||
)
|
||||
|
||||
def abort(self):
|
||||
if self.handle is not None:
|
||||
self.handle.close()
|
||||
self.handle = None
|
||||
if not self.ready_published:
|
||||
try:
|
||||
os.remove(self.partial_path)
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, value, traceback):
|
||||
if not self.finished:
|
||||
self.abort()
|
||||
|
||||
|
||||
class ResultBundleReader:
|
||||
def __init__(self, path, max_event_bytes=DEFAULT_MAX_EVENT_BYTES):
|
||||
self.path = os.path.abspath(path)
|
||||
self.max_event_bytes = max(1, int(max_event_bytes))
|
||||
self._validated = None
|
||||
self._validated_fingerprint = None
|
||||
|
||||
@classmethod
|
||||
def from_reservation(cls, root, reservation, max_event_bytes=DEFAULT_MAX_EVENT_BYTES):
|
||||
reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation)
|
||||
return cls(bundle_ready_path(root, reservation.bundle_id), max_event_bytes=max_event_bytes)
|
||||
|
||||
@staticmethod
|
||||
def _stat_fingerprint(value):
|
||||
return (
|
||||
int(value.st_dev), int(value.st_ino), int(value.st_mode),
|
||||
int(value.st_size), int(value.st_mtime_ns),
|
||||
)
|
||||
|
||||
def _file_fingerprint(self):
|
||||
reject_reparse_components(self.path)
|
||||
if not private_file_ready(self.path):
|
||||
raise ResultBundleUnavailableError('bundle path is not currently available as an exact private regular file')
|
||||
return self._stat_fingerprint(os.stat(self.path, follow_symlinks=False))
|
||||
|
||||
def _iter_frames(self, expected_fingerprint=None):
|
||||
fingerprint = self._file_fingerprint()
|
||||
if expected_fingerprint is not None and fingerprint != expected_fingerprint:
|
||||
raise ResultBundleError('bundle changed after validation')
|
||||
size = fingerprint[3]
|
||||
if size <= len(MAGIC) or size > self.max_event_bytes:
|
||||
raise ResultBundleError('bundle aggregate byte bound is invalid')
|
||||
with open(self.path, 'rb', buffering=0) as handle:
|
||||
opened_fingerprint = self._stat_fingerprint(os.fstat(handle.fileno()))
|
||||
if opened_fingerprint != fingerprint:
|
||||
raise ResultBundleUnavailableError('bundle identity changed while it was opened')
|
||||
magic = handle.read(len(MAGIC))
|
||||
if magic != MAGIC:
|
||||
raise ResultBundleError('bundle magic/version mismatch')
|
||||
offset = len(MAGIC)
|
||||
while offset < size:
|
||||
header = handle.read(FRAME_HEADER.size)
|
||||
if len(header) != FRAME_HEADER.size:
|
||||
raise ResultBundleError('truncated bundle frame header')
|
||||
frame_type, payload_length = FRAME_HEADER.unpack(header)
|
||||
if frame_type not in FRAME_TYPES:
|
||||
raise ResultBundleError('unknown bundle frame type')
|
||||
bound = min(FRAME_BOUNDS[frame_type], self.max_event_bytes)
|
||||
if payload_length > bound or offset + FRAME_HEADER.size + payload_length > size:
|
||||
raise ResultBundleError('bundle frame length exceeds its bound')
|
||||
payload = handle.read(payload_length)
|
||||
if len(payload) != payload_length:
|
||||
raise ResultBundleError('truncated bundle frame payload')
|
||||
try:
|
||||
value = json.loads(payload.decode('utf-8', errors='strict'))
|
||||
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
|
||||
raise ResultBundleError('bundle frame contains invalid UTF-8 JSON') from exc
|
||||
if canonical_json_bytes(value) != payload:
|
||||
raise ResultBundleError('bundle frame JSON is not canonical')
|
||||
offset += FRAME_HEADER.size + payload_length
|
||||
yield frame_type, value, header + payload, offset
|
||||
if offset != size:
|
||||
raise ResultBundleError('bundle byte count is inconsistent')
|
||||
if self._stat_fingerprint(os.fstat(handle.fileno())) != opened_fingerprint:
|
||||
raise ResultBundleError('bundle changed while it was read')
|
||||
if self._file_fingerprint() != fingerprint:
|
||||
raise ResultBundleError('bundle path changed while it was read')
|
||||
|
||||
def validate(self):
|
||||
if self._validated is not None:
|
||||
return self._validated
|
||||
digest = hashlib.sha256(MAGIC)
|
||||
header_value = None
|
||||
metadata_value = None
|
||||
footer = None
|
||||
counts = {b'F': 0, b'E': 0, b'D': 0, b'K': 0}
|
||||
diagnostic_bytes = 0
|
||||
diagnostic_aggregate_bytes = 0
|
||||
diagnostic_uids = set()
|
||||
diagnostic_scan_outcomes = []
|
||||
frame_count = 0
|
||||
last_frame_order = -1
|
||||
final_offset = len(MAGIC)
|
||||
content_bytes = None
|
||||
fingerprint = self._file_fingerprint()
|
||||
for frame_type, value, framed, offset in self._iter_frames(fingerprint):
|
||||
frame_count += 1
|
||||
if frame_count > MAX_TOTAL_FRAMES:
|
||||
raise ResultBundleError('bundle frame count exceeds its bound')
|
||||
if FRAME_ORDER[frame_type] < last_frame_order:
|
||||
raise ResultBundleError('bundle frame order is not deterministic')
|
||||
last_frame_order = FRAME_ORDER[frame_type]
|
||||
final_offset = offset
|
||||
if footer is not None:
|
||||
raise ResultBundleError('commit footer is not the final frame')
|
||||
if metadata_value is not None and frame_type != b'C':
|
||||
raise ResultBundleError('result metadata is not immediately before the commit footer')
|
||||
if frame_count == 1 and frame_type != b'H':
|
||||
raise ResultBundleError('bundle header is not the first frame')
|
||||
if frame_type in FRAME_TYPES and not isinstance(value, dict):
|
||||
raise ResultBundleError('bundle typed frame must contain a JSON object')
|
||||
if frame_type == b'E' and not isinstance(value.get('error'), str):
|
||||
raise ResultBundleError('bundle error frame is invalid')
|
||||
if frame_type == b'H':
|
||||
if header_value is not None:
|
||||
raise ResultBundleError('bundle contains duplicate headers')
|
||||
header_value = value
|
||||
elif frame_type == b'M':
|
||||
if metadata_value is not None:
|
||||
raise ResultBundleError('bundle contains duplicate metadata')
|
||||
metadata_value = value
|
||||
elif frame_type == b'C':
|
||||
footer = value
|
||||
content_bytes = offset - len(framed)
|
||||
continue
|
||||
elif frame_type == b'D':
|
||||
try:
|
||||
envelope = decode_diagnostic_envelope(canonical_json_bytes(value))
|
||||
except (TypeError, ValueError, UnicodeError) as exc:
|
||||
raise ResultBundleError('bundle diagnostic frame is invalid') from exc
|
||||
if envelope.diagnostic_uid in diagnostic_uids:
|
||||
raise ResultBundleError('bundle diagnostic identity is duplicated')
|
||||
if header_value is None or (
|
||||
envelope.reservation_id != int(header_value.get('reservation_id') or 0)
|
||||
or envelope.scan_event_id != str(header_value.get('scan_event_id') or '')
|
||||
or envelope.source != str(header_value.get('source') or '')
|
||||
):
|
||||
raise ResultBundleError('bundle diagnostic identity conflicts with its header')
|
||||
diagnostic_uids.add(envelope.diagnostic_uid)
|
||||
if envelope.assignment_outcome is not AssignmentOutcome.ACCEPTED:
|
||||
raise ResultBundleError(
|
||||
'bundle diagnostic assignment outcome is invalid'
|
||||
)
|
||||
diagnostic_scan_outcomes.append(envelope.scan_outcome.value)
|
||||
counts[b'D'] += 1
|
||||
if counts[b'D'] > MAX_DIAGNOSTICS_PER_ASSIGNMENT:
|
||||
raise ResultBundleError('bundle typed frame count exceeds its bound')
|
||||
envelope_bytes = len(encode_diagnostic_envelope(envelope))
|
||||
diagnostic_bytes += len(framed)
|
||||
diagnostic_aggregate_bytes += envelope_bytes + 1
|
||||
if (
|
||||
diagnostic_aggregate_bytes > MAX_DIAGNOSTIC_AGGREGATE_BYTES
|
||||
):
|
||||
raise ResultBundleError('bundle diagnostic aggregate exceeds its byte bound')
|
||||
elif frame_type in counts:
|
||||
counts[frame_type] += 1
|
||||
limit = {
|
||||
b'F': MAX_FINDING_FRAMES,
|
||||
b'E': MAX_ERROR_FRAMES,
|
||||
b'D': MAX_DIAGNOSTICS_PER_ASSIGNMENT,
|
||||
b'K': MAX_CANDIDATE_FRAMES,
|
||||
}[frame_type]
|
||||
if counts[frame_type] > limit:
|
||||
raise ResultBundleError('bundle typed frame count exceeds its bound')
|
||||
digest.update(framed)
|
||||
if not isinstance(header_value, dict) or not isinstance(metadata_value, dict) or not isinstance(footer, dict):
|
||||
raise ResultBundleError('bundle is missing required header, metadata, or footer')
|
||||
if frame_count < 3 or footer.get('format_version') != FORMAT_VERSION:
|
||||
raise ResultBundleError('bundle footer version is invalid')
|
||||
expected_scan_outcome = {
|
||||
'clean': 'clean',
|
||||
'found': 'found',
|
||||
'degraded': 'degraded',
|
||||
'error': 'error',
|
||||
'skipped': 'skipped',
|
||||
}.get(str(metadata_value.get('status') or 'clean'), 'error')
|
||||
if any(
|
||||
outcome != expected_scan_outcome
|
||||
for outcome in diagnostic_scan_outcomes
|
||||
):
|
||||
raise ResultBundleError(
|
||||
'bundle diagnostic scan outcome conflicts with result metadata'
|
||||
)
|
||||
base_footer_fields = {
|
||||
'format_version', 'content_sha256', 'content_bytes', 'byte_count',
|
||||
'frame_count', 'finding_count', 'error_count', 'candidate_count',
|
||||
}
|
||||
expected_footer_fields = (
|
||||
base_footer_fields | {'diagnostic_count', 'diagnostic_bytes'}
|
||||
if counts[b'D'] else base_footer_fields
|
||||
)
|
||||
if set(footer) != expected_footer_fields:
|
||||
raise ResultBundleError('bundle footer shape is invalid')
|
||||
expected = {
|
||||
'content_sha256': digest.hexdigest(),
|
||||
'content_bytes': content_bytes,
|
||||
'byte_count': final_offset,
|
||||
'frame_count': frame_count,
|
||||
'finding_count': counts[b'F'],
|
||||
'error_count': counts[b'E'],
|
||||
'candidate_count': counts[b'K'],
|
||||
}
|
||||
if counts[b'D']:
|
||||
expected.update({
|
||||
'diagnostic_count': counts[b'D'],
|
||||
'diagnostic_bytes': diagnostic_bytes,
|
||||
})
|
||||
if footer.get('content_sha256') != expected['content_sha256']:
|
||||
raise ResultBundleError('bundle content hash mismatch')
|
||||
for key in (
|
||||
'content_bytes', 'byte_count', 'frame_count', 'finding_count',
|
||||
'error_count', 'candidate_count',
|
||||
):
|
||||
value = footer.get(key)
|
||||
if isinstance(value, bool) or not isinstance(value, int) or value != expected[key]:
|
||||
raise ResultBundleError(f'bundle footer {key} mismatch')
|
||||
diagnostic_footer_fields = {'diagnostic_count', 'diagnostic_bytes'} & set(footer)
|
||||
if counts[b'D']:
|
||||
if diagnostic_footer_fields != {'diagnostic_count', 'diagnostic_bytes'}:
|
||||
raise ResultBundleError('bundle footer diagnostic accounting is missing')
|
||||
for key in ('diagnostic_count', 'diagnostic_bytes'):
|
||||
value = footer.get(key)
|
||||
if isinstance(value, bool) or not isinstance(value, int) or value != expected[key]:
|
||||
raise ResultBundleError(f'bundle footer {key} mismatch')
|
||||
elif diagnostic_footer_fields:
|
||||
raise ResultBundleError('D-less bundle has unexpected diagnostic accounting')
|
||||
bundle_id = _validated_id(header_value.get('bundle_id'), 'bundle_id')
|
||||
event_id = _validated_id(header_value.get('scan_event_id'), 'scan_event_id')
|
||||
reservation_id = header_value.get('reservation_id')
|
||||
if isinstance(reservation_id, bool) or not isinstance(reservation_id, int) or reservation_id <= 0:
|
||||
raise ResultBundleError('invalid reservation_id')
|
||||
self._validated = BundleMetadata(
|
||||
reservation_id=reservation_id,
|
||||
bundle_id=bundle_id,
|
||||
scan_event_id=event_id,
|
||||
scan_event_hash=expected['content_sha256'],
|
||||
relative_path='',
|
||||
actual_bytes=final_offset,
|
||||
frame_count=frame_count,
|
||||
finding_count=counts[b'F'],
|
||||
error_count=counts[b'E'],
|
||||
candidate_count=counts[b'K'],
|
||||
header=header_value,
|
||||
result_metadata=metadata_value,
|
||||
diagnostic_count=counts[b'D'],
|
||||
diagnostic_bytes=diagnostic_bytes,
|
||||
)
|
||||
self._validated_fingerprint = fingerprint
|
||||
return self._validated
|
||||
|
||||
def _values(self, wanted):
|
||||
validated = self.validate()
|
||||
digest = hashlib.sha256(MAGIC)
|
||||
footer = None
|
||||
for frame_type, value, framed, _ in self._iter_frames(
|
||||
self._validated_fingerprint
|
||||
):
|
||||
if frame_type == b'C':
|
||||
footer = value
|
||||
else:
|
||||
digest.update(framed)
|
||||
if frame_type == wanted:
|
||||
yield value
|
||||
if (
|
||||
digest.hexdigest() != validated.scan_event_hash
|
||||
or not isinstance(footer, dict)
|
||||
or footer.get('content_sha256') != validated.scan_event_hash
|
||||
):
|
||||
raise ResultBundleError('bundle content changed after validation')
|
||||
|
||||
def iter_findings(self):
|
||||
return self._values(b'F')
|
||||
|
||||
def iter_errors(self):
|
||||
for value in self._values(b'E'):
|
||||
yield value.get('error') if isinstance(value, dict) else value
|
||||
|
||||
def iter_candidates(self):
|
||||
return self._values(b'K')
|
||||
|
||||
def iter_diagnostics(self):
|
||||
for value in self._values(b'D'):
|
||||
envelope = decode_diagnostic_envelope(canonical_json_bytes(value))
|
||||
yield json.loads(encode_diagnostic_envelope(envelope).decode('ascii'))
|
||||
|
||||
def effective_diagnostics(self):
|
||||
validated = self.validate()
|
||||
if validated.diagnostic_count:
|
||||
return tuple(self.iter_diagnostics())
|
||||
metadata = self.metadata()
|
||||
header = self.header()
|
||||
envelopes = build_legacy_error_frame_diagnostics(
|
||||
reservation_id=validated.reservation_id,
|
||||
scan_event_id=validated.scan_event_id,
|
||||
slot_id=0,
|
||||
source=str(header.get('source') or ''),
|
||||
timestamp=(
|
||||
metadata.get('timestamp') or metadata.get('scan_started_at')
|
||||
),
|
||||
errors=tuple(self.iter_errors()),
|
||||
retryable=bool(metadata.get('retryable', False)),
|
||||
attempt=max(1, int(metadata.get('attempt') or 1)),
|
||||
)
|
||||
return tuple(
|
||||
json.loads(encode_diagnostic_envelope(envelope).decode('ascii'))
|
||||
for envelope in envelopes
|
||||
)
|
||||
|
||||
def metadata(self):
|
||||
self.validate()
|
||||
if self._file_fingerprint() != self._validated_fingerprint:
|
||||
raise ResultBundleError('bundle changed after validation')
|
||||
return dict(self.validate().result_metadata)
|
||||
|
||||
def header(self):
|
||||
self.validate()
|
||||
if self._file_fingerprint() != self._validated_fingerprint:
|
||||
raise ResultBundleError('bundle changed after validation')
|
||||
return dict(self.validate().header)
|
||||
@@ -0,0 +1,605 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
if not sys.dont_write_bytecode:
|
||||
raise RuntimeError('result ingester could not disable bytecode writes')
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
|
||||
from lifecycle_authority import require_active_supervisor_child
|
||||
from paths import apply_path_config
|
||||
from process_identity import current_process_identity, exact_process_identity_state
|
||||
from result_bundle import (
|
||||
ResultBundleError,
|
||||
ResultBundleReader,
|
||||
bundle_partial_relative_path,
|
||||
)
|
||||
from runtime_security import (
|
||||
PrivatePathState,
|
||||
durable_publish,
|
||||
durable_unlink,
|
||||
ensure_private_directory,
|
||||
private_file_ready,
|
||||
inspect_private_relative_path,
|
||||
require_private_directory,
|
||||
sha256_file,
|
||||
)
|
||||
from scanner_db import (
|
||||
DockerCoverageDispositionConflictError,
|
||||
DockerFindingAttributionLimitError,
|
||||
ScanEventConflictError,
|
||||
ScannerDB,
|
||||
)
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ResultIngester:
|
||||
def __init__(
|
||||
self, db, bundle_root, supervisor_instance_id, lease_seconds=300, fault=None,
|
||||
quarantine_max_items=10000, quarantine_max_bytes=1024 * 1024 * 1024,
|
||||
metadata_retention_days=30, metadata_retirement_batch=100,
|
||||
recover_expired_ready=False,
|
||||
):
|
||||
self.db = db
|
||||
self.bundle_root = require_private_directory(bundle_root, create=False)
|
||||
self.supervisor_instance_id = str(supervisor_instance_id)
|
||||
self.lease_seconds = max(30, int(lease_seconds))
|
||||
self.fault = fault
|
||||
self.lease = None
|
||||
self.recovery_after_id = 0
|
||||
self.quarantine_max_items = max(0, int(quarantine_max_items))
|
||||
self.quarantine_max_bytes = max(0, int(quarantine_max_bytes))
|
||||
self.metadata_retention_seconds = max(1, int(metadata_retention_days)) * 86400
|
||||
self.metadata_retirement_batch = min(500, max(1, int(metadata_retirement_batch)))
|
||||
self.next_metadata_retirement = 0.0
|
||||
self.recover_expired_ready = recover_expired_ready is True
|
||||
|
||||
def _inject(self, stage, value=None):
|
||||
if self.fault is not None:
|
||||
self.fault(stage, value)
|
||||
|
||||
def start(self):
|
||||
self.db.require_runtime_safety_schema()
|
||||
self.db.require_final_cutover()
|
||||
identity = current_process_identity()
|
||||
self.lease = self.db.acquire_pipeline_lease(
|
||||
'result_ingester', self.supervisor_instance_id, identity,
|
||||
lease_seconds=self.lease_seconds, initial_state='recovering',
|
||||
)
|
||||
if not self.lease:
|
||||
raise RuntimeError('another result ingester owns the singleton advisory lock')
|
||||
self.reconcile_terminal_artifacts()
|
||||
self.retire_terminal_metadata()
|
||||
self.recover()
|
||||
if not self.heartbeat('ready'):
|
||||
raise RuntimeError('result ingester ready lease publication failed')
|
||||
return self
|
||||
|
||||
def heartbeat(self, state='ready', error=''):
|
||||
return self.db.heartbeat_pipeline_lease(
|
||||
'result_ingester', self.lease['generation'], self.lease['lease_token'],
|
||||
lease_seconds=self.lease_seconds, state=state, error=error,
|
||||
)
|
||||
|
||||
def stop(self, error=''):
|
||||
if self.lease:
|
||||
released = self.db.release_pipeline_lease(
|
||||
'result_ingester', self.lease['generation'], self.lease['lease_token'],
|
||||
state='failed' if error else 'released', error=error,
|
||||
)
|
||||
self.lease = None
|
||||
return released
|
||||
return True
|
||||
|
||||
def _path(self, relative):
|
||||
normalized = str(relative or '').replace('/', os.sep)
|
||||
path = os.path.abspath(os.path.join(self.bundle_root, normalized))
|
||||
if os.path.commonpath((self.bundle_root, path)) != self.bundle_root or path == self.bundle_root:
|
||||
raise ResultBundleError('bundle database path escapes its configured root')
|
||||
return path
|
||||
|
||||
def _quarantine_path(self, reservation):
|
||||
bundle_id = str(reservation['bundle_id'])
|
||||
return os.path.join(
|
||||
self.bundle_root, 'quarantine', bundle_id[:2], f'{bundle_id}.trb',
|
||||
)
|
||||
|
||||
def _quarantine_relative_path(self, reservation):
|
||||
bundle_id = str(reservation['bundle_id'])
|
||||
return f'quarantine/{bundle_id[:2]}/{bundle_id}.trb'
|
||||
|
||||
def _ensure_quarantine_shard(self, reservation):
|
||||
bundle_id = str(reservation['bundle_id'])
|
||||
return ensure_private_directory(
|
||||
os.path.join(self.bundle_root, 'quarantine', bundle_id[:2]),
|
||||
reject_reparse=True,
|
||||
)
|
||||
|
||||
def _inspect(self, relative_path):
|
||||
return inspect_private_relative_path(self.bundle_root, relative_path)
|
||||
|
||||
def _defer_reservation_cleanup(self, reservation, error):
|
||||
try:
|
||||
self.db.defer_result_reservation_cleanup(reservation['id'], str(error))
|
||||
except Exception:
|
||||
logger.warning(
|
||||
'reservation cleanup backoff could not be recorded: %s', reservation['id']
|
||||
)
|
||||
|
||||
def reconcile_terminal_artifacts(self, max_pages=100):
|
||||
if not hasattr(self.db, 'bundle_terminal_temp_artifacts'):
|
||||
return
|
||||
for _ in range(max(1, int(max_pages))):
|
||||
rows = self.db.bundle_terminal_temp_artifacts(100)
|
||||
if not rows:
|
||||
return
|
||||
progressed = False
|
||||
for row in rows:
|
||||
try:
|
||||
inspection = self._inspect(row['relative_path'])
|
||||
if inspection.state == PrivatePathState.UNKNOWN:
|
||||
self.db.defer_pipeline_artifact_cleanup(
|
||||
row['id'], inspection.detail or 'artifact storage state is unknown',
|
||||
)
|
||||
continue
|
||||
if inspection.state == PrivatePathState.PRESENT:
|
||||
if not private_file_ready(inspection.path):
|
||||
self.db.defer_pipeline_artifact_cleanup(
|
||||
row['id'], 'artifact is not an exact private file',
|
||||
)
|
||||
continue
|
||||
durable_unlink(inspection.path)
|
||||
inspection = self._inspect(row['relative_path'])
|
||||
if inspection.state == PrivatePathState.ABSENT:
|
||||
self.db.mark_pipeline_artifact_deleted(row['id'])
|
||||
progressed = True
|
||||
else:
|
||||
self.db.defer_pipeline_artifact_cleanup(
|
||||
row['id'], 'artifact unlink was not confirmed',
|
||||
)
|
||||
except OSError as exc:
|
||||
try:
|
||||
self.db.defer_pipeline_artifact_cleanup(row['id'], str(exc))
|
||||
except Exception:
|
||||
logger.warning(
|
||||
'artifact cleanup backoff could not be recorded: %s', row['id']
|
||||
)
|
||||
continue
|
||||
if not progressed:
|
||||
return
|
||||
if len(rows) < 100:
|
||||
return
|
||||
return
|
||||
|
||||
def retire_terminal_metadata(self):
|
||||
if time.monotonic() < self.next_metadata_retirement:
|
||||
return
|
||||
self.next_metadata_retirement = time.monotonic() + 60
|
||||
try:
|
||||
self.db.retire_admission_intents(
|
||||
self.metadata_retention_seconds, self.metadata_retirement_batch,
|
||||
)
|
||||
self.db.retire_deleted_pipeline_artifacts(
|
||||
self.metadata_retention_seconds, self.metadata_retirement_batch,
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.warning('bounded terminal metadata retirement deferred: %s', type(exc).__name__)
|
||||
|
||||
def quarantine(self, reservation, ready_path, reason_code, detail):
|
||||
ready_relative = str(reservation['ready_relative_path']).replace('\\', '/')
|
||||
quarantine_relative = self._quarantine_relative_path(reservation)
|
||||
self._ensure_quarantine_shard(reservation)
|
||||
ready = self._inspect(ready_relative)
|
||||
quarantine = self._inspect(quarantine_relative)
|
||||
if ready.state == PrivatePathState.UNKNOWN or quarantine.state == PrivatePathState.UNKNOWN:
|
||||
raise ResultBundleError('bundle quarantine path state is unknown')
|
||||
if ready.state == PrivatePathState.PRESENT and quarantine.state == PrivatePathState.PRESENT:
|
||||
raise ResultBundleError('both ready and quarantine paths exist for one reservation')
|
||||
if ready.state == PrivatePathState.ABSENT and quarantine.state == PrivatePathState.ABSENT:
|
||||
raise ResultBundleError('bundle disappeared before quarantine')
|
||||
quarantine_path = quarantine.path
|
||||
ensure_private_directory(os.path.dirname(quarantine_path), reject_reparse=True)
|
||||
source_path = ready.path if ready.state == PrivatePathState.PRESENT else quarantine.path
|
||||
if not private_file_ready(source_path):
|
||||
raise ResultBundleError('bundle quarantine source is not an exact private file')
|
||||
byte_count = os.path.getsize(source_path)
|
||||
payload_hash = sha256_file(source_path) if byte_count else ''
|
||||
relative = quarantine_relative
|
||||
quarantine_id = self.db.quarantine_result_bundle(
|
||||
reservation['id'], reason_code, detail, relative,
|
||||
payload_sha256=payload_hash, byte_count=byte_count,
|
||||
quarantine_max_items=self.quarantine_max_items,
|
||||
quarantine_max_bytes=self.quarantine_max_bytes,
|
||||
physical_confirmed=ready.state != PrivatePathState.PRESENT,
|
||||
)
|
||||
if ready.state == PrivatePathState.PRESENT:
|
||||
durable_publish(ready.path, quarantine_path)
|
||||
confirmed = self._inspect(quarantine_relative)
|
||||
if confirmed.state != PrivatePathState.PRESENT:
|
||||
raise ResultBundleError('bundle quarantine publication was not confirmed')
|
||||
quarantine_id = self.db.quarantine_result_bundle(
|
||||
reservation['id'], reason_code, detail, relative,
|
||||
payload_sha256=payload_hash, byte_count=byte_count,
|
||||
quarantine_max_items=self.quarantine_max_items,
|
||||
quarantine_max_bytes=self.quarantine_max_bytes,
|
||||
physical_confirmed=True,
|
||||
)
|
||||
return quarantine_id
|
||||
|
||||
def recover(self, page_size=100, max_pages=100):
|
||||
pages = 0
|
||||
while pages < max(1, int(max_pages)):
|
||||
rows = self.db.active_result_reservations(self.recovery_after_id, page_size)
|
||||
if not rows:
|
||||
self.recovery_after_id = 0
|
||||
return pages
|
||||
pages += 1
|
||||
for reservation in rows:
|
||||
self.recovery_after_id = int(reservation['id'])
|
||||
if (
|
||||
reservation['state'] == 'scanning'
|
||||
and str(reservation.get('assignment_kind') or 'local') == 'remote'
|
||||
):
|
||||
continue
|
||||
ready_relative = str(reservation['ready_relative_path']).replace('\\', '/')
|
||||
ready = self._inspect(ready_relative)
|
||||
identity_state = None
|
||||
if reservation['state'] == 'scanning':
|
||||
identity_state = exact_process_identity_state(
|
||||
reservation['producer_pid'], reservation['producer_creation_time'],
|
||||
reservation['producer_executable'],
|
||||
)
|
||||
if (
|
||||
ready.state == PrivatePathState.UNKNOWN
|
||||
and identity_state in ('dead', 'reused')
|
||||
):
|
||||
try:
|
||||
ensure_private_directory(
|
||||
os.path.join(
|
||||
self.bundle_root, 'ready',
|
||||
str(reservation['bundle_id'])[:2],
|
||||
),
|
||||
reject_reparse=True,
|
||||
)
|
||||
ensure_private_directory(
|
||||
os.path.join(
|
||||
self.bundle_root, 'tmp',
|
||||
str(reservation['bundle_id'])[:2],
|
||||
),
|
||||
reject_reparse=True,
|
||||
)
|
||||
ready = self._inspect(ready_relative)
|
||||
except OSError as exc:
|
||||
self._defer_reservation_cleanup(reservation, exc)
|
||||
continue
|
||||
quarantine_relative = self._quarantine_relative_path(reservation)
|
||||
try:
|
||||
self._ensure_quarantine_shard(reservation)
|
||||
except OSError as exc:
|
||||
self._defer_reservation_cleanup(reservation, exc)
|
||||
continue
|
||||
quarantine = self._inspect(quarantine_relative)
|
||||
if quarantine.state == PrivatePathState.PRESENT:
|
||||
try:
|
||||
self.db.quarantine_result_bundle(
|
||||
reservation['id'],
|
||||
reservation.get('last_error_code') or 'recovered_quarantine',
|
||||
reservation.get('last_error_detail') or 'recovered deterministic quarantine file',
|
||||
quarantine_relative,
|
||||
payload_sha256=sha256_file(quarantine.path),
|
||||
byte_count=os.path.getsize(quarantine.path),
|
||||
quarantine_max_items=self.quarantine_max_items,
|
||||
quarantine_max_bytes=self.quarantine_max_bytes,
|
||||
physical_confirmed=True,
|
||||
)
|
||||
except OSError as exc:
|
||||
self._defer_reservation_cleanup(reservation, exc)
|
||||
continue
|
||||
if quarantine.state == PrivatePathState.UNKNOWN:
|
||||
self._defer_reservation_cleanup(
|
||||
reservation, quarantine.detail or 'quarantine state unknown',
|
||||
)
|
||||
continue
|
||||
if ready.state == PrivatePathState.UNKNOWN:
|
||||
self._defer_reservation_cleanup(
|
||||
reservation, ready.detail or 'ready state unknown',
|
||||
)
|
||||
continue
|
||||
prepared_quarantine = self.db.pending_result_bundle_quarantine(
|
||||
reservation['id']
|
||||
)
|
||||
if prepared_quarantine and ready.state == PrivatePathState.PRESENT:
|
||||
try:
|
||||
self.quarantine(
|
||||
reservation, ready.path,
|
||||
prepared_quarantine['reason_code'],
|
||||
prepared_quarantine.get('reason_detail') or 'recovered prepared quarantine',
|
||||
)
|
||||
except OSError as exc:
|
||||
self._defer_reservation_cleanup(reservation, exc)
|
||||
continue
|
||||
if reservation['state'] == 'scanning':
|
||||
if ready.state == PrivatePathState.PRESENT:
|
||||
try:
|
||||
metadata = ResultBundleReader(ready.path).validate()
|
||||
recovered = metadata.as_dict()
|
||||
recovered['relative_path'] = str(
|
||||
reservation['ready_relative_path']
|
||||
).replace('\\', '/')
|
||||
marked_ready = self.db.mark_result_bundle_ready(
|
||||
reservation['id'], recovered,
|
||||
)
|
||||
if (
|
||||
not marked_ready
|
||||
and self.recover_expired_ready
|
||||
and identity_state in ('dead', 'reused')
|
||||
):
|
||||
self.db.recover_expired_result_bundle_ready(
|
||||
reservation['id'], recovered,
|
||||
)
|
||||
except OSError as exc:
|
||||
self._defer_reservation_cleanup(reservation, exc)
|
||||
except (ValueError, ResultBundleError) as exc:
|
||||
try:
|
||||
self.quarantine(
|
||||
reservation, ready.path, 'bundle_validation_failed', str(exc),
|
||||
)
|
||||
except OSError as cleanup_exc:
|
||||
self._defer_reservation_cleanup(reservation, cleanup_exc)
|
||||
continue
|
||||
if identity_state in ('dead', 'reused'):
|
||||
try:
|
||||
ensure_private_directory(
|
||||
os.path.join(self.bundle_root, 'ready', str(reservation['bundle_id'])[:2]),
|
||||
reject_reparse=True,
|
||||
)
|
||||
ensure_private_directory(
|
||||
os.path.join(self.bundle_root, 'tmp', str(reservation['bundle_id'])[:2]),
|
||||
reject_reparse=True,
|
||||
)
|
||||
except OSError as exc:
|
||||
self._defer_reservation_cleanup(reservation, exc)
|
||||
continue
|
||||
ready = self._inspect(ready_relative)
|
||||
if ready.state != PrivatePathState.ABSENT:
|
||||
continue
|
||||
if identity_state in ('dead', 'reused'):
|
||||
partial_relative = bundle_partial_relative_path(
|
||||
reservation['bundle_id'], reservation['reservation_token'],
|
||||
).replace(os.sep, '/')
|
||||
partial = self._inspect(partial_relative)
|
||||
if partial.state == PrivatePathState.UNKNOWN:
|
||||
self._defer_reservation_cleanup(
|
||||
reservation, partial.detail or 'partial state unknown',
|
||||
)
|
||||
continue
|
||||
if partial.state == PrivatePathState.PRESENT:
|
||||
if not private_file_ready(partial.path):
|
||||
continue
|
||||
try:
|
||||
durable_unlink(partial.path)
|
||||
except OSError as exc:
|
||||
self._defer_reservation_cleanup(reservation, exc)
|
||||
continue
|
||||
partial = self._inspect(partial_relative)
|
||||
if partial.state != PrivatePathState.ABSENT:
|
||||
self._defer_reservation_cleanup(
|
||||
reservation, 'partial unlink was not confirmed',
|
||||
)
|
||||
continue
|
||||
self.db.refund_uncommitted_reservation(
|
||||
reservation['id'],
|
||||
{
|
||||
'pid': reservation['producer_pid'],
|
||||
'creation_time': reservation['producer_creation_time'],
|
||||
'executable': reservation['producer_executable'],
|
||||
},
|
||||
f'producer identity is {identity_state} and exact ready path is absent',
|
||||
partial_absence_confirmed=True,
|
||||
)
|
||||
elif reservation['state'] in ('ready', 'ingesting'):
|
||||
if ready.state == PrivatePathState.ABSENT:
|
||||
self.db.quarantine_result_bundle(
|
||||
reservation['id'], 'ready_bundle_missing',
|
||||
'database ready row has a definitively absent exact ready path',
|
||||
'', byte_count=0,
|
||||
quarantine_max_items=self.quarantine_max_items,
|
||||
quarantine_max_bytes=self.quarantine_max_bytes,
|
||||
physical_confirmed=True,
|
||||
)
|
||||
elif reservation['state'] == 'db_committed':
|
||||
bundle = self.db.result_bundle_for_reservation(reservation['id'])
|
||||
if not bundle:
|
||||
continue
|
||||
if ready.state == PrivatePathState.PRESENT:
|
||||
if not private_file_ready(ready.path):
|
||||
continue
|
||||
try:
|
||||
durable_unlink(ready.path)
|
||||
except OSError as exc:
|
||||
self._defer_reservation_cleanup(reservation, exc)
|
||||
continue
|
||||
ready = self._inspect(ready_relative)
|
||||
if ready.state == PrivatePathState.ABSENT:
|
||||
event = self.db.confirm_scan_event(
|
||||
reservation['scan_event_id'], bundle['scan_event_hash'],
|
||||
)
|
||||
if event:
|
||||
self.db.acknowledge_removed_bundle(
|
||||
reservation['id'], reservation['scan_event_id'],
|
||||
bundle['scan_event_hash'],
|
||||
)
|
||||
return pages
|
||||
|
||||
def process_one(self):
|
||||
claimed = self.db.claim_ready_result_bundle(
|
||||
self.lease['generation'], self.lease['lease_token'], self.lease_seconds,
|
||||
)
|
||||
if not claimed:
|
||||
return False
|
||||
reservation = claimed['reservation']
|
||||
bundle = claimed['bundle']
|
||||
ready_relative = str(bundle['relative_path']).replace('\\', '/')
|
||||
ready = self._inspect(ready_relative)
|
||||
try:
|
||||
if bundle['state'] == 'db_committed':
|
||||
event = self.db.confirm_scan_event(bundle['scan_event_id'], bundle['scan_event_hash'])
|
||||
if not event:
|
||||
raise ResultBundleError('db_committed bundle has no exact authoritative event')
|
||||
else:
|
||||
if ready.state == PrivatePathState.UNKNOWN:
|
||||
return False
|
||||
if ready.state == PrivatePathState.ABSENT:
|
||||
self.db.quarantine_result_bundle(
|
||||
reservation['id'], 'ready_bundle_missing',
|
||||
'claimed ready bundle is definitively absent', '',
|
||||
quarantine_max_items=self.quarantine_max_items,
|
||||
quarantine_max_bytes=self.quarantine_max_bytes,
|
||||
physical_confirmed=True,
|
||||
)
|
||||
return True
|
||||
self._inject('before_validation', reservation)
|
||||
reader = ResultBundleReader(ready.path)
|
||||
validated = reader.validate()
|
||||
self._inject('after_validation', validated)
|
||||
if (
|
||||
validated.bundle_id != str(bundle['bundle_id'])
|
||||
or validated.scan_event_id != str(bundle['scan_event_id'])
|
||||
or validated.scan_event_hash != str(bundle['scan_event_hash'])
|
||||
or validated.actual_bytes != int(bundle['actual_bytes'])
|
||||
):
|
||||
raise ResultBundleError('validated bundle totals conflict with its claimed database row')
|
||||
self._inject('before_db_commit', reservation)
|
||||
self.db.ingest_result_bundle(reader, reservation, bundle)
|
||||
self._inject('after_db_commit', reservation)
|
||||
event = self.db.confirm_scan_event(bundle['scan_event_id'], bundle['scan_event_hash'])
|
||||
self._inject('after_confirmation', event)
|
||||
if not event:
|
||||
raise ResultBundleError('database commit was not confirmed by exact event ID and hash')
|
||||
ready = self._inspect(ready_relative)
|
||||
if ready.state == PrivatePathState.UNKNOWN:
|
||||
return False
|
||||
if ready.state == PrivatePathState.PRESENT:
|
||||
self._inject('before_unlink', reservation)
|
||||
durable_unlink(ready.path)
|
||||
self._inject('after_unlink', reservation)
|
||||
ready = self._inspect(ready_relative)
|
||||
if ready.state != PrivatePathState.ABSENT:
|
||||
return False
|
||||
self._inject('before_capacity_release', reservation)
|
||||
if not self.db.acknowledge_removed_bundle(
|
||||
reservation['id'], bundle['scan_event_id'], bundle['scan_event_hash'],
|
||||
):
|
||||
raise RuntimeError('bundle capacity acknowledgement was not fenced')
|
||||
self._inject('after_capacity_release', reservation)
|
||||
return True
|
||||
except DockerCoverageDispositionConflictError as exc:
|
||||
ready = self._inspect(ready_relative)
|
||||
if ready.state == PrivatePathState.PRESENT:
|
||||
try:
|
||||
self.quarantine(
|
||||
reservation, ready.path,
|
||||
'docker_coverage_disposition_conflict', str(exc),
|
||||
)
|
||||
except OSError as cleanup_exc:
|
||||
self._defer_reservation_cleanup(reservation, cleanup_exc)
|
||||
return False
|
||||
return True
|
||||
except DockerFindingAttributionLimitError as exc:
|
||||
ready = self._inspect(ready_relative)
|
||||
if ready.state == PrivatePathState.PRESENT:
|
||||
try:
|
||||
self.quarantine(
|
||||
reservation, ready.path,
|
||||
'docker_attribution_limit_exceeded', str(exc),
|
||||
)
|
||||
except OSError as cleanup_exc:
|
||||
self._defer_reservation_cleanup(reservation, cleanup_exc)
|
||||
return False
|
||||
return True
|
||||
except ScanEventConflictError as exc:
|
||||
ready = self._inspect(ready_relative)
|
||||
if ready.state == PrivatePathState.PRESENT:
|
||||
try:
|
||||
self.quarantine(reservation, ready.path, 'scan_event_hash_conflict', str(exc))
|
||||
except OSError as cleanup_exc:
|
||||
self._defer_reservation_cleanup(reservation, cleanup_exc)
|
||||
return False
|
||||
return True
|
||||
except (ResultBundleError, ValueError) as exc:
|
||||
ready = self._inspect(ready_relative)
|
||||
if ready.state == PrivatePathState.PRESENT:
|
||||
try:
|
||||
self.quarantine(reservation, ready.path, 'bundle_validation_failed', str(exc))
|
||||
except OSError as cleanup_exc:
|
||||
self._defer_reservation_cleanup(reservation, cleanup_exc)
|
||||
return False
|
||||
return True
|
||||
raise
|
||||
except OSError as exc:
|
||||
self._defer_reservation_cleanup(reservation, exc)
|
||||
return False
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description='Singleton durable result bundle ingester')
|
||||
parser.add_argument('--config', required=True)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
metadata = require_active_supervisor_child(child_kind='result-ingester', require_dsn=True)
|
||||
args = parse_args()
|
||||
import yaml
|
||||
|
||||
with open(args.config, 'r', encoding='utf-8') as handle:
|
||||
config = apply_path_config(yaml.safe_load(handle) or {}, args.config)
|
||||
global_config = config.get('global') or {}
|
||||
db = ScannerDB(db_url=global_config['database_url'], initialize=False)
|
||||
if not db.enabled:
|
||||
raise SystemExit('result ingester PostgreSQL connection is unavailable')
|
||||
db.set_application_name('truf-result-ingester')
|
||||
worker = ResultIngester(
|
||||
db, global_config['result_bundle_dir'], metadata['instance_id'],
|
||||
lease_seconds=int(((config.get('supervisor') or {}).get('result_ingester') or {}).get('lease_seconds', 300)),
|
||||
quarantine_max_items=int(global_config.get('pipeline_quarantine_max_items', 10000)),
|
||||
quarantine_max_bytes=int(global_config.get('pipeline_quarantine_max_bytes', 1024 * 1024 * 1024)),
|
||||
metadata_retention_days=int(global_config.get('pipeline_metadata_retention_days', 30)),
|
||||
metadata_retirement_batch=int(global_config.get('pipeline_metadata_retirement_batch', 100)),
|
||||
)
|
||||
error = ''
|
||||
try:
|
||||
worker.start()
|
||||
idle = max(0.05, float(((config.get('supervisor') or {}).get('result_ingester') or {}).get('poll_sec', 0.2)))
|
||||
next_heartbeat = time.monotonic() + worker.lease_seconds / 3
|
||||
while True:
|
||||
worker.retire_terminal_metadata()
|
||||
worker.reconcile_terminal_artifacts(max_pages=1)
|
||||
worker.recover(max_pages=1)
|
||||
processed = worker.process_one()
|
||||
if time.monotonic() >= next_heartbeat:
|
||||
if not worker.heartbeat('ready'):
|
||||
raise RuntimeError('result ingester heartbeat fence was lost')
|
||||
next_heartbeat = time.monotonic() + worker.lease_seconds / 3
|
||||
if not processed:
|
||||
time.sleep(idle)
|
||||
except KeyboardInterrupt:
|
||||
pass
|
||||
except BaseException as exc:
|
||||
error = f'{type(exc).__name__}: {exc}'
|
||||
raise
|
||||
finally:
|
||||
try:
|
||||
worker.stop(error)
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
+1073
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,143 @@
|
||||
"""Stdlib-only pre-import boundary for canonical runtime entrypoints."""
|
||||
|
||||
import sys
|
||||
import os
|
||||
|
||||
if __name__ == '__main__':
|
||||
if sys.platform != 'linux' or not os.path.isfile('/.dockerenv') or os.path.abspath(__file__) != '/opt/truf/app/runtime_bootstrap.py':
|
||||
raise SystemExit('Docker development copy: runtime control is disabled outside the prepared container. See DOCKER_MIGRATION.md.')
|
||||
import runpy
|
||||
runpy.run_path('/opt/truf/app/container_runtime.py')['require_container']()
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
if not sys.dont_write_bytecode:
|
||||
raise RuntimeError('runtime bootstrap could not disable bytecode writes')
|
||||
|
||||
import runpy
|
||||
import stat
|
||||
|
||||
|
||||
RUNTIME_BOOTSTRAP_ENV = 'TRUF_RUNTIME_BOOTSTRAP'
|
||||
RUNTIME_BOOTSTRAP_VALUE = '1'
|
||||
APPLICATION_IMPORT_SUFFIXES = ('.py', '.pyw', '.pyc', '.pyd')
|
||||
SUPERVISOR_ENTRYPOINT_FLAG = '--runtime-bootstrap-entrypoint'
|
||||
TARGETS = {
|
||||
'supervisor': 'supervisor.py',
|
||||
'postgres-runtime': 'postgres_runtime.py',
|
||||
'migrate-runtime-safety': 'migrate_runtime_safety.py',
|
||||
}
|
||||
|
||||
|
||||
def _require_isolated_startup():
|
||||
if not (
|
||||
sys.flags.isolated
|
||||
and sys.flags.no_site
|
||||
and sys.flags.dont_write_bytecode
|
||||
and sys.dont_write_bytecode
|
||||
):
|
||||
raise RuntimeError('runtime bootstrap requires isolated no-site bytecode-free startup (-I -S -B)')
|
||||
|
||||
|
||||
def _canonical(path):
|
||||
return os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(path))))
|
||||
|
||||
|
||||
def _is_reparse_point(path):
|
||||
details = os.lstat(path)
|
||||
if stat.S_ISLNK(details.st_mode):
|
||||
return True
|
||||
attributes = getattr(details, 'st_file_attributes', 0)
|
||||
reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0)
|
||||
return bool(attributes & reparse_attribute) or getattr(os.path, 'isjunction', lambda _path: False)(path)
|
||||
|
||||
|
||||
def _reject_cached_bytecode(app_dir):
|
||||
def raise_walk_error(exc):
|
||||
raise RuntimeError(f'unable to inspect the application root: {exc}') from exc
|
||||
|
||||
try:
|
||||
root_details = os.lstat(app_dir)
|
||||
except OSError as exc:
|
||||
raise RuntimeError(f'application root is unavailable: {app_dir}') from exc
|
||||
if _is_reparse_point(app_dir):
|
||||
raise RuntimeError(f'application root reparse point is forbidden: {app_dir}')
|
||||
if not stat.S_ISDIR(root_details.st_mode):
|
||||
raise RuntimeError(f'application root is not a directory: {app_dir}')
|
||||
canonical_root = _canonical(app_dir)
|
||||
for current, directories, files in os.walk(app_dir, followlinks=False, onerror=raise_walk_error):
|
||||
for name in directories:
|
||||
candidate = os.path.join(current, name)
|
||||
if _is_reparse_point(candidate):
|
||||
relative = os.path.relpath(candidate, app_dir).replace(os.sep, '/')
|
||||
if name.lower() == '__pycache__':
|
||||
raise RuntimeError(f'application __pycache__ link is forbidden: {relative}')
|
||||
raise RuntimeError(f'application directory reparse point is forbidden: {relative}')
|
||||
relative_current = os.path.relpath(current, app_dir)
|
||||
in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep))
|
||||
for name in files:
|
||||
candidate = os.path.join(current, name)
|
||||
relative = os.path.relpath(candidate, app_dir).replace(os.sep, '/')
|
||||
if _is_reparse_point(candidate):
|
||||
raise RuntimeError(f'application file reparse point is forbidden: {relative}')
|
||||
if name.lower().endswith(APPLICATION_IMPORT_SUFFIXES):
|
||||
try:
|
||||
contained = os.path.commonpath((canonical_root, _canonical(candidate))) == canonical_root
|
||||
except ValueError:
|
||||
contained = False
|
||||
if not contained:
|
||||
raise RuntimeError(f'application Python authority escapes its root: {relative}')
|
||||
if in_cache and name.lower().endswith('.pyc'):
|
||||
raise RuntimeError(f'application __pycache__ bytecode is forbidden: {relative}')
|
||||
|
||||
|
||||
def _require_supervisor_entrypoint_binding(arguments, entrypoint):
|
||||
bindings = []
|
||||
for index, argument in enumerate(arguments):
|
||||
text = str(argument)
|
||||
if text == SUPERVISOR_ENTRYPOINT_FLAG:
|
||||
if index + 1 >= len(arguments):
|
||||
raise RuntimeError('supervisor runtime bootstrap entrypoint binding has no path')
|
||||
bindings.append(str(arguments[index + 1]))
|
||||
elif text.startswith(SUPERVISOR_ENTRYPOINT_FLAG + '='):
|
||||
bindings.append(text.split('=', 1)[1])
|
||||
if len(bindings) != 1:
|
||||
raise RuntimeError('supervisor runtime requires exactly one explicit bootstrap entrypoint binding')
|
||||
binding = bindings[0]
|
||||
if not os.path.isabs(binding) or _canonical(binding) != _canonical(entrypoint):
|
||||
raise RuntimeError('supervisor runtime bootstrap entrypoint binding is not canonical supervisor.py')
|
||||
|
||||
|
||||
def main():
|
||||
_require_isolated_startup()
|
||||
if len(sys.argv) < 3 or sys.argv[2] != '--':
|
||||
raise RuntimeError('usage: runtime_bootstrap.py <supervisor|postgres-runtime|migrate-runtime-safety> -- <args>')
|
||||
target_name = str(sys.argv[1]).strip().lower()
|
||||
target_file = TARGETS.get(target_name)
|
||||
if not target_file:
|
||||
raise RuntimeError(f'unsupported canonical runtime target: {target_name}')
|
||||
|
||||
app_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
_reject_cached_bytecode(app_dir)
|
||||
|
||||
entrypoint = _canonical(os.path.join(app_dir, target_file))
|
||||
arguments = list(sys.argv[3:])
|
||||
if target_name == 'supervisor':
|
||||
_require_supervisor_entrypoint_binding(arguments, entrypoint)
|
||||
|
||||
child_namespace = runpy.run_path(os.path.join(app_dir, 'child_bootstrap.py'))
|
||||
enable_dependencies = child_namespace.get('_enable_dependency_paths')
|
||||
if not callable(enable_dependencies):
|
||||
raise RuntimeError('authenticated dependency path bootstrap is unavailable')
|
||||
enable_dependencies(target_name)
|
||||
|
||||
os.environ[RUNTIME_BOOTSTRAP_ENV] = RUNTIME_BOOTSTRAP_VALUE
|
||||
sys.path.insert(0, app_dir)
|
||||
sys.argv = [entrypoint, *arguments]
|
||||
runpy.run_path(entrypoint, run_name='__main__')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
main()
|
||||
except Exception as exc:
|
||||
raise SystemExit(f'canonical runtime bootstrap rejected launch: {exc}') from exc
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,941 @@
|
||||
import hashlib
|
||||
import hmac
|
||||
import json
|
||||
import platform as host_platform
|
||||
import sys
|
||||
from contextlib import nullcontext
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Mapping
|
||||
|
||||
from result_bundle import BundleReservation, FORMAT_VERSION
|
||||
from scanner import (
|
||||
cleanup_assignment_work_dir,
|
||||
client_remote_execution_binding,
|
||||
client_scan_phase_events,
|
||||
client_scan_execution_policy,
|
||||
scan_slot_scope,
|
||||
scan_target_result,
|
||||
stage_result_bundle,
|
||||
)
|
||||
from scanner_db import normalize_target
|
||||
from target_identity import normalize_huggingface_space_id, parse_dockerhub_digest_target
|
||||
|
||||
|
||||
PROTOCOL_VERSION = 2
|
||||
REMOTE_EXECUTION_SNAPSHOT_SCHEMA = 1
|
||||
PACKAGE_DETECTOR_POLICY = '@package/detector_policy'
|
||||
MAX_REMOTE_EXECUTION_SNAPSHOT_BYTES = 64 * 1024
|
||||
_REMOTE_SCAN_POLICY_BOUNDS = {
|
||||
'trufflehog_stdout_max_mb': (1, 4096),
|
||||
'trufflehog_stderr_max_mb': (1, 4096),
|
||||
'result_bundle_max_event_bytes': (1024, 4 * 1024 * 1024 * 1024),
|
||||
'trufflehog_max_findings_per_target': (1, 1000000),
|
||||
'trufflehog_job_memory_limit_bytes': (0, 1 << 50),
|
||||
'trufflehog_windows_job_cpu_weight': (0, 10000),
|
||||
'trufflehog_windows_memory_priority': (0, 5),
|
||||
'trufflehog_diagnostic_max_lines': (1, 2000),
|
||||
'trufflehog_diagnostic_max_line_chars': (1, 8192),
|
||||
'trufflehog_diagnostic_max_line_bytes': (1, 8192),
|
||||
'trufflehog_diagnostic_max_errors': (1, 200),
|
||||
'trufflehog_diagnostic_max_warnings': (1, 200),
|
||||
'trufflehog_diagnostic_max_unclassified': (1, 20),
|
||||
}
|
||||
|
||||
|
||||
class ScanExecutionError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class QueueDispositionPolicy:
|
||||
target_retry_max_attempts: int = 3
|
||||
target_retry_base_delay_sec: int = 3600
|
||||
target_retry_max_delay_sec: int = 86400
|
||||
target_timeout_retry_delay_sec: int = 21600
|
||||
docker_layer_checkpoint_delay_sec: int = 60
|
||||
ci_soft_cooldown_days: int = 7
|
||||
soft_skip_reasons: tuple[str, ...] = ()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ScanCompatibility:
|
||||
protocol_version: int
|
||||
bundle_format_version: int
|
||||
platform_tag: str
|
||||
code_manifest_sha256: str
|
||||
effective_config_sha256: str
|
||||
detector_policy_sha256: str = ''
|
||||
|
||||
@classmethod
|
||||
def from_mapping(cls, value):
|
||||
value = dict(value or {})
|
||||
return cls(
|
||||
protocol_version=int(value.get('protocol_version') or 0),
|
||||
bundle_format_version=int(value.get('bundle_format_version') or 0),
|
||||
platform_tag=str(value.get('platform_tag') or ''),
|
||||
code_manifest_sha256=_digest(value.get('code_manifest_sha256'), 'code manifest'),
|
||||
effective_config_sha256=_digest(
|
||||
value.get('effective_config_sha256'), 'effective config',
|
||||
),
|
||||
detector_policy_sha256=_digest(
|
||||
value.get('detector_policy_sha256'), 'detector policy', optional=True,
|
||||
),
|
||||
)
|
||||
|
||||
def as_dict(self):
|
||||
return dict(self.__dict__)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class WorkerBuildCompatibility:
|
||||
protocol_version: int
|
||||
bundle_format_version: int
|
||||
platform_tag: str
|
||||
code_manifest_sha256: str
|
||||
detector_policy_sha256: str
|
||||
|
||||
@classmethod
|
||||
def from_mapping(cls, value):
|
||||
value = dict(value or {})
|
||||
if set(value) != {
|
||||
'protocol_version', 'bundle_format_version', 'platform_tag',
|
||||
'code_manifest_sha256', 'detector_policy_sha256',
|
||||
}:
|
||||
raise ValueError('worker build compatibility shape is invalid')
|
||||
return cls(
|
||||
protocol_version=int(value.get('protocol_version') or 0),
|
||||
bundle_format_version=int(value.get('bundle_format_version') or 0),
|
||||
platform_tag=str(value.get('platform_tag') or ''),
|
||||
code_manifest_sha256=_digest(value.get('code_manifest_sha256'), 'code manifest'),
|
||||
detector_policy_sha256=_digest(
|
||||
value.get('detector_policy_sha256'), 'detector policy',
|
||||
),
|
||||
)
|
||||
|
||||
def as_dict(self):
|
||||
return dict(self.__dict__)
|
||||
|
||||
|
||||
def _digest(value, label, optional=False):
|
||||
value = str(value or '')
|
||||
if optional and not value:
|
||||
return ''
|
||||
if len(value) != 64 or any(char not in '0123456789abcdef' for char in value):
|
||||
raise ValueError(f'invalid {label} digest')
|
||||
return value
|
||||
|
||||
|
||||
def local_platform_tag():
|
||||
machine = host_platform.machine().strip().lower().replace('amd64', 'x86_64')
|
||||
system = 'windows' if sys.platform == 'win32' else 'linux' if sys.platform.startswith('linux') else ''
|
||||
if not system or machine not in {'x86_64', 'aarch64', 'arm64'}:
|
||||
raise ScanExecutionError('unsupported worker platform')
|
||||
return f'{system}-{machine.replace("arm64", "aarch64")}'
|
||||
|
||||
|
||||
def validate_scan_compatibility(required, local):
|
||||
required = required if isinstance(required, ScanCompatibility) else ScanCompatibility.from_mapping(required)
|
||||
local = local if isinstance(local, ScanCompatibility) else ScanCompatibility.from_mapping(local)
|
||||
if required.protocol_version != PROTOCOL_VERSION or local.protocol_version != PROTOCOL_VERSION:
|
||||
raise ScanExecutionError('worker protocol is incompatible')
|
||||
if required.bundle_format_version != FORMAT_VERSION or local.bundle_format_version != FORMAT_VERSION:
|
||||
raise ScanExecutionError('result bundle format is incompatible')
|
||||
for name in (
|
||||
'platform_tag', 'code_manifest_sha256', 'effective_config_sha256',
|
||||
'detector_policy_sha256',
|
||||
):
|
||||
if not hmac.compare_digest(str(getattr(required, name)), str(getattr(local, name))):
|
||||
raise ScanExecutionError(f'worker {name.replace("_", " ")} is incompatible')
|
||||
return required
|
||||
|
||||
|
||||
def validate_worker_build_compatibility(
|
||||
required, local, *, expected_protocol_version=PROTOCOL_VERSION,
|
||||
):
|
||||
required = (
|
||||
required if isinstance(required, WorkerBuildCompatibility)
|
||||
else WorkerBuildCompatibility.from_mapping(required)
|
||||
)
|
||||
local = (
|
||||
local if isinstance(local, WorkerBuildCompatibility)
|
||||
else WorkerBuildCompatibility.from_mapping(local)
|
||||
)
|
||||
if (
|
||||
required.protocol_version != expected_protocol_version
|
||||
or local.protocol_version != expected_protocol_version
|
||||
):
|
||||
raise ScanExecutionError('worker protocol is incompatible')
|
||||
if required.bundle_format_version != FORMAT_VERSION or local.bundle_format_version != FORMAT_VERSION:
|
||||
raise ScanExecutionError('result bundle format is incompatible')
|
||||
for name in ('platform_tag', 'code_manifest_sha256', 'detector_policy_sha256'):
|
||||
if not hmac.compare_digest(str(getattr(required, name)), str(getattr(local, name))):
|
||||
raise ScanExecutionError(f'worker {name.replace("_", " ")} is incompatible')
|
||||
return required
|
||||
|
||||
|
||||
_COMMON_SCAN_KWARGS = {
|
||||
'timeout_sec', 'detectors', 'exclude_detectors', 'no_verification',
|
||||
'trufflehog_config', 'token',
|
||||
}
|
||||
_SOURCE_SCAN_KWARGS = {
|
||||
'git': {'git_plan'},
|
||||
'github': {'git_plan', 'max_depth', 'max_commit_age_days', 'commit_lookup_pages',
|
||||
'skip_if_commit_lookup_fails'},
|
||||
'github_archive': {'max_depth', 'max_commit_age_days', 'commit_lookup_pages',
|
||||
'skip_if_commit_lookup_fails'},
|
||||
'gitlab': {'git_plan', 'external_trufflehog_lifecycle', 'max_depth',
|
||||
'max_commit_age_days', 'commit_lookup_pages', 'skip_if_commit_lookup_fails'},
|
||||
'docker': {'docker_layer_work', 'trufflehog_concurrency', 'docker_recovery_limits',
|
||||
'docker_recovery_min_free_bytes'},
|
||||
'huggingface': set(),
|
||||
'npm': {'max_artifact_size_mb'},
|
||||
'pypi': {'max_artifact_size_mb'},
|
||||
'package_git': {'max_depth', 'max_commit_age_days', 'commit_lookup_pages',
|
||||
'skip_if_commit_lookup_fails'},
|
||||
'postman': {'max_artifact_size_mb'},
|
||||
'github_gists': {'max_artifact_size_mb'},
|
||||
'github_archive_files': {'max_artifact_size_mb'},
|
||||
'github_actions': {
|
||||
'ci_runs_per_repo', 'ci_lookback_days', 'ci_max_log_archive_mb',
|
||||
'ci_max_log_file_mb', 'ci_failed_first', 'ci_scan_artifacts',
|
||||
'ci_max_artifacts_per_run', 'ci_max_artifact_archive_mb',
|
||||
'ci_max_artifact_file_mb', 'ci_max_artifact_files',
|
||||
'ci_target_max_download_mb', 'fetch_timeout',
|
||||
},
|
||||
'gitlab_ci': {
|
||||
'ci_pipelines_per_project', 'ci_jobs_per_pipeline', 'ci_lookback_days',
|
||||
'ci_max_trace_mb', 'ci_scan_artifacts', 'ci_max_artifacts_per_pipeline',
|
||||
'ci_max_artifact_archive_mb', 'ci_max_artifact_file_mb',
|
||||
'ci_max_artifact_files', 'ci_target_max_download_mb', 'fetch_timeout',
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def validate_scan_kwargs(platform, scan_kwargs):
|
||||
platform = str(platform or '').strip().lower()
|
||||
if platform not in _SOURCE_SCAN_KWARGS:
|
||||
raise ScanExecutionError('unsupported scan platform')
|
||||
values = dict(scan_kwargs or {})
|
||||
unknown = set(values) - _COMMON_SCAN_KWARGS - _SOURCE_SCAN_KWARGS[platform]
|
||||
if unknown:
|
||||
raise ScanExecutionError('scan settings contain unsupported fields')
|
||||
timeout = values.get('timeout_sec')
|
||||
if isinstance(timeout, bool):
|
||||
raise ScanExecutionError('scan timeout is invalid')
|
||||
try:
|
||||
timeout = float(timeout)
|
||||
except (TypeError, ValueError, OverflowError):
|
||||
raise ScanExecutionError('scan timeout is invalid') from None
|
||||
if not 1 <= timeout <= 86400:
|
||||
raise ScanExecutionError('scan timeout is outside the worker bound')
|
||||
values['timeout_sec'] = timeout
|
||||
return values
|
||||
|
||||
|
||||
def normalize_remote_scan_policy(value):
|
||||
values = dict(value or {})
|
||||
expected = {
|
||||
'drop_detectors', 'strict_git_provider_token_filter',
|
||||
*_REMOTE_SCAN_POLICY_BOUNDS,
|
||||
}
|
||||
if set(values) != expected:
|
||||
raise ScanExecutionError('remote scan policy shape is invalid')
|
||||
raw_drop = values['drop_detectors']
|
||||
if isinstance(raw_drop, str):
|
||||
raw_drop = raw_drop.split(',')
|
||||
if not isinstance(raw_drop, (list, tuple)) or len(raw_drop) > 256:
|
||||
raise ScanExecutionError('remote detector drop policy is invalid')
|
||||
drop_detectors = []
|
||||
for item in raw_drop:
|
||||
if not isinstance(item, str):
|
||||
raise ScanExecutionError('remote detector drop policy is invalid')
|
||||
item = item.strip().lower()
|
||||
if not item:
|
||||
continue
|
||||
if len(item) > 128 or any(ord(char) < 32 or ord(char) == 127 for char in item):
|
||||
raise ScanExecutionError('remote detector drop policy is invalid')
|
||||
drop_detectors.append(item)
|
||||
strict = values['strict_git_provider_token_filter']
|
||||
if not isinstance(strict, bool):
|
||||
raise ScanExecutionError('remote Git provider token policy is invalid')
|
||||
normalized = {
|
||||
'drop_detectors': sorted(set(drop_detectors)),
|
||||
'strict_git_provider_token_filter': strict,
|
||||
}
|
||||
for name, (minimum, maximum) in _REMOTE_SCAN_POLICY_BOUNDS.items():
|
||||
raw = values[name]
|
||||
if not isinstance(raw, int) or isinstance(raw, bool):
|
||||
raise ScanExecutionError('remote scan policy limit is invalid')
|
||||
number = raw
|
||||
if number < minimum or number > maximum:
|
||||
raise ScanExecutionError('remote scan policy limit is outside its bounds')
|
||||
normalized[name] = number
|
||||
return normalized
|
||||
|
||||
|
||||
def remote_execution_identity(
|
||||
platform, scan_kwargs, event_scan_options, queue_policy, limits, scan_policy,
|
||||
):
|
||||
normalized_scan = validate_scan_kwargs(platform, scan_kwargs)
|
||||
event_options = dict(event_scan_options or {})
|
||||
if 'token' in event_options or 'git_plan' in event_options:
|
||||
raise ScanExecutionError('event scan settings contain private or planned fields')
|
||||
expected_event = {
|
||||
name: value for name, value in normalized_scan.items()
|
||||
if name not in {'token', 'git_plan'}
|
||||
}
|
||||
if event_options != expected_event:
|
||||
raise ScanExecutionError('event scan settings do not match execution settings')
|
||||
try:
|
||||
policy = (
|
||||
queue_policy if isinstance(queue_policy, QueueDispositionPolicy)
|
||||
else QueueDispositionPolicy(**dict(queue_policy or {}))
|
||||
)
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise ScanExecutionError('queue disposition policy is invalid') from exc
|
||||
policy_value = {
|
||||
'target_retry_max_attempts': int(policy.target_retry_max_attempts),
|
||||
'target_retry_base_delay_sec': int(policy.target_retry_base_delay_sec),
|
||||
'target_retry_max_delay_sec': int(policy.target_retry_max_delay_sec),
|
||||
'target_timeout_retry_delay_sec': int(policy.target_timeout_retry_delay_sec),
|
||||
'docker_layer_checkpoint_delay_sec': int(policy.docker_layer_checkpoint_delay_sec),
|
||||
'ci_soft_cooldown_days': int(policy.ci_soft_cooldown_days),
|
||||
'soft_skip_reasons': list(policy.soft_skip_reasons),
|
||||
}
|
||||
limit_values = dict(limits or {})
|
||||
if set(limit_values) != {'candidate_max_items', 'candidate_max_bytes'}:
|
||||
raise ScanExecutionError('worker assignment limits are invalid')
|
||||
normalized_limits = {
|
||||
'candidate_max_items': int(limit_values['candidate_max_items']),
|
||||
'candidate_max_bytes': int(limit_values['candidate_max_bytes']),
|
||||
}
|
||||
if (
|
||||
not 1 <= normalized_limits['candidate_max_items'] <= 100000
|
||||
or not 1024 <= normalized_limits['candidate_max_bytes'] <= 64 * 1024 * 1024
|
||||
):
|
||||
raise ScanExecutionError('worker assignment limits are outside their bounds')
|
||||
execution = {
|
||||
'source': str(platform or '').strip().lower(),
|
||||
'scan_kwargs': event_options,
|
||||
'scan_policy': normalize_remote_scan_policy(scan_policy),
|
||||
'queue_policy': policy_value,
|
||||
'limits': normalized_limits,
|
||||
}
|
||||
return canonical_json_sha256(execution), execution
|
||||
|
||||
|
||||
def validate_remote_assignment_compatibility(
|
||||
required, local_build, platform, scan_kwargs, event_scan_options, queue_policy, limits,
|
||||
scan_policy, *, expected_protocol_version=PROTOCOL_VERSION,
|
||||
):
|
||||
required = required if isinstance(required, ScanCompatibility) else ScanCompatibility.from_mapping(required)
|
||||
validate_worker_build_compatibility({
|
||||
'protocol_version': required.protocol_version,
|
||||
'bundle_format_version': required.bundle_format_version,
|
||||
'platform_tag': required.platform_tag,
|
||||
'code_manifest_sha256': required.code_manifest_sha256,
|
||||
'detector_policy_sha256': required.detector_policy_sha256,
|
||||
}, local_build, expected_protocol_version=expected_protocol_version)
|
||||
effective, _ = remote_execution_identity(
|
||||
platform, scan_kwargs, event_scan_options, queue_policy, limits, scan_policy,
|
||||
)
|
||||
if not hmac.compare_digest(effective, required.effective_config_sha256):
|
||||
raise ScanExecutionError('worker effective config is incompatible')
|
||||
return required
|
||||
|
||||
|
||||
def _remote_snapshot_envelope(value):
|
||||
if not isinstance(value, dict):
|
||||
raise ScanExecutionError('remote execution snapshot must be an object')
|
||||
value = dict(value)
|
||||
if set(value) != {
|
||||
'schema', 'compatibility', 'execution', 'planning', 'credential_ref',
|
||||
} or value.get('schema') != REMOTE_EXECUTION_SNAPSHOT_SCHEMA:
|
||||
raise ScanExecutionError('remote execution snapshot shape is invalid')
|
||||
compatibility = ScanCompatibility.from_mapping(value.get('compatibility'))
|
||||
execution = dict(value.get('execution') or {})
|
||||
if set(execution) != {
|
||||
'source', 'scan_kwargs', 'scan_policy', 'queue_policy', 'limits',
|
||||
}:
|
||||
raise ScanExecutionError('remote execution snapshot settings are invalid')
|
||||
source = str(execution.get('source') or '').strip().lower()
|
||||
effective, normalized_execution = remote_execution_identity(
|
||||
source, execution.get('scan_kwargs'), execution.get('scan_kwargs'),
|
||||
execution.get('queue_policy'), execution.get('limits'),
|
||||
execution.get('scan_policy'),
|
||||
)
|
||||
if normalized_execution['scan_kwargs'].get('trufflehog_config') != PACKAGE_DETECTOR_POLICY:
|
||||
raise ScanExecutionError('remote execution snapshot policy path is invalid')
|
||||
if not hmac.compare_digest(effective, compatibility.effective_config_sha256):
|
||||
raise ScanExecutionError('remote execution snapshot effective config is invalid')
|
||||
planning = dict(value.get('planning') or {})
|
||||
credential_ref = dict(value.get('credential_ref') or {})
|
||||
if set(credential_ref) != {'source', 'auth_entry'}:
|
||||
raise ScanExecutionError('remote execution snapshot credential reference is invalid')
|
||||
queue_source = str(credential_ref.get('source') or '').strip().lower()
|
||||
auth_entry = str(credential_ref.get('auth_entry') or '')
|
||||
if len(auth_entry) > 128 or '\x00' in auth_entry:
|
||||
raise ScanExecutionError('remote execution snapshot credential reference is invalid')
|
||||
return compatibility, normalized_execution, planning, queue_source, auth_entry
|
||||
|
||||
|
||||
def _normalize_exact_git_v1_planning(planning):
|
||||
planning = dict(planning or {})
|
||||
if set(planning) != {
|
||||
'kind', 'git_baseline_depth', 'git_ref_resolution_attempts',
|
||||
'git_ref_resolution_timeout_sec', 'git_ref_resolution_max_bytes',
|
||||
} or planning.get('kind') != 'exact_git_v1':
|
||||
raise ScanExecutionError('remote execution snapshot planning is invalid')
|
||||
|
||||
try:
|
||||
baseline_depth = int(planning['git_baseline_depth'])
|
||||
attempts = int(planning['git_ref_resolution_attempts'])
|
||||
timeout = float(planning['git_ref_resolution_timeout_sec'])
|
||||
max_bytes = int(planning['git_ref_resolution_max_bytes'])
|
||||
except (TypeError, ValueError, OverflowError) as exc:
|
||||
raise ScanExecutionError('remote execution snapshot planning is invalid') from exc
|
||||
if (
|
||||
isinstance(planning['git_baseline_depth'], bool)
|
||||
or isinstance(planning['git_ref_resolution_attempts'], bool)
|
||||
or isinstance(planning['git_ref_resolution_timeout_sec'], bool)
|
||||
or isinstance(planning['git_ref_resolution_max_bytes'], bool)
|
||||
or not 1 <= baseline_depth <= 1000000
|
||||
or not 1 <= attempts <= 20
|
||||
or not 0.1 <= timeout <= 300
|
||||
or not 1024 <= max_bytes <= 64 * 1024 * 1024
|
||||
):
|
||||
raise ScanExecutionError('remote execution snapshot planning is outside its bounds')
|
||||
return {
|
||||
'kind': 'exact_git_v1',
|
||||
'git_baseline_depth': baseline_depth,
|
||||
'git_ref_resolution_attempts': attempts,
|
||||
'git_ref_resolution_timeout_sec': timeout,
|
||||
'git_ref_resolution_max_bytes': max_bytes,
|
||||
}
|
||||
|
||||
|
||||
def _normalize_kind_only_planning(planning, kind):
|
||||
planning = dict(planning or {})
|
||||
if planning != {'kind': kind}:
|
||||
raise ScanExecutionError('remote execution snapshot planning is invalid')
|
||||
return {'kind': kind}
|
||||
|
||||
|
||||
def _normalized_remote_snapshot(
|
||||
value, *, queue_sources, worker_platform, planning_kind,
|
||||
planning_normalizer, public_credential,
|
||||
):
|
||||
compatibility, execution, planning, queue_source, auth_entry = (
|
||||
_remote_snapshot_envelope(value)
|
||||
)
|
||||
platform = execution['source']
|
||||
if (
|
||||
queue_source not in queue_sources
|
||||
or (worker_platform is None and platform != queue_source)
|
||||
or (worker_platform is not None and platform != worker_platform)
|
||||
or (public_credential and auth_entry)
|
||||
):
|
||||
raise ScanExecutionError('remote execution snapshot source capability is invalid')
|
||||
normalized_planning = planning_normalizer(planning)
|
||||
if normalized_planning.get('kind') != planning_kind:
|
||||
raise ScanExecutionError('remote execution snapshot planning kind is invalid')
|
||||
normalized = {
|
||||
'schema': REMOTE_EXECUTION_SNAPSHOT_SCHEMA,
|
||||
'compatibility': compatibility.as_dict(),
|
||||
'execution': execution,
|
||||
'planning': normalized_planning,
|
||||
'credential_ref': {'source': queue_source, 'auth_entry': auth_entry},
|
||||
}
|
||||
encoded = json.dumps(
|
||||
normalized, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
|
||||
allow_nan=False,
|
||||
).encode('utf-8')
|
||||
if len(encoded) > MAX_REMOTE_EXECUTION_SNAPSHOT_BYTES:
|
||||
raise ScanExecutionError('remote execution snapshot exceeds its byte bound')
|
||||
return normalized
|
||||
|
||||
|
||||
def normalize_exact_git_execution_snapshot(value):
|
||||
return _normalized_remote_snapshot(
|
||||
value,
|
||||
queue_sources=frozenset(('github', 'gitlab')),
|
||||
worker_platform=None,
|
||||
planning_kind='exact_git_v1',
|
||||
planning_normalizer=_normalize_exact_git_v1_planning,
|
||||
public_credential=False,
|
||||
)
|
||||
|
||||
|
||||
def normalize_docker_direct_execution_snapshot(value):
|
||||
return _normalized_remote_snapshot(
|
||||
value,
|
||||
queue_sources=frozenset(('dockerhub',)),
|
||||
worker_platform='docker',
|
||||
planning_kind='docker_direct_v1',
|
||||
planning_normalizer=lambda planning: _normalize_kind_only_planning(
|
||||
planning, 'docker_direct_v1',
|
||||
),
|
||||
public_credential=True,
|
||||
)
|
||||
|
||||
|
||||
def normalize_huggingface_space_execution_snapshot(value):
|
||||
return _normalized_remote_snapshot(
|
||||
value,
|
||||
queue_sources=frozenset(('huggingface',)),
|
||||
worker_platform='huggingface',
|
||||
planning_kind='huggingface_space_v1',
|
||||
planning_normalizer=lambda planning: _normalize_kind_only_planning(
|
||||
planning, 'huggingface_space_v1',
|
||||
),
|
||||
public_credential=True,
|
||||
)
|
||||
|
||||
|
||||
def normalize_remote_execution_snapshot(value):
|
||||
if not isinstance(value, dict) or not isinstance(value.get('planning'), dict):
|
||||
raise ScanExecutionError('remote execution snapshot planning is invalid')
|
||||
kind = value['planning'].get('kind')
|
||||
normalizer = {
|
||||
'exact_git_v1': normalize_exact_git_execution_snapshot,
|
||||
'docker_direct_v1': normalize_docker_direct_execution_snapshot,
|
||||
'huggingface_space_v1': normalize_huggingface_space_execution_snapshot,
|
||||
}.get(kind)
|
||||
if normalizer is None:
|
||||
raise ScanExecutionError('remote execution snapshot planning kind is unsupported')
|
||||
return normalizer(value)
|
||||
|
||||
|
||||
def normalize_docker_direct_execution_target(value):
|
||||
try:
|
||||
return parse_dockerhub_digest_target(value)
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise ScanExecutionError('Docker direct target is invalid') from exc
|
||||
|
||||
|
||||
def normalize_huggingface_space_execution_target(value):
|
||||
try:
|
||||
return normalize_huggingface_space_id(value)
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise ScanExecutionError('HuggingFace Space target is invalid') from exc
|
||||
|
||||
|
||||
def remote_execution_snapshot_sha256(value):
|
||||
normalized = normalize_remote_execution_snapshot(value)
|
||||
encoded = json.dumps(
|
||||
normalized, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
|
||||
allow_nan=False,
|
||||
).encode('utf-8')
|
||||
return hashlib.sha256(encoded).hexdigest()
|
||||
|
||||
|
||||
def _normalize_remote_assignment_deadlines(value, reservation, scan_kwargs):
|
||||
if not isinstance(value, dict) or set(value) != {
|
||||
'target_scan_timeout_seconds', 'result_upload_body_timeout_seconds',
|
||||
'assignment_ttl_seconds', 'assignment_issued_at',
|
||||
'assignment_deadline_at',
|
||||
}:
|
||||
raise ScanExecutionError('worker assignment deadlines shape is invalid')
|
||||
deadlines = dict(value)
|
||||
for name in (
|
||||
'target_scan_timeout_seconds', 'result_upload_body_timeout_seconds',
|
||||
'assignment_ttl_seconds',
|
||||
):
|
||||
if type(deadlines[name]) is not int or deadlines[name] <= 0:
|
||||
raise ScanExecutionError('worker assignment deadline value is invalid')
|
||||
issued_at = deadlines['assignment_issued_at']
|
||||
deadline_at = deadlines['assignment_deadline_at']
|
||||
if (
|
||||
type(issued_at) is not str
|
||||
or type(deadline_at) is not str
|
||||
or issued_at != reservation.get('remote_issued_at')
|
||||
or deadline_at != reservation.get('remote_expires_at')
|
||||
):
|
||||
raise ScanExecutionError('worker assignment deadline changed after reservation')
|
||||
try:
|
||||
issued = datetime.fromisoformat(issued_at)
|
||||
deadline = datetime.fromisoformat(deadline_at)
|
||||
except ValueError as exc:
|
||||
raise ScanExecutionError('worker assignment deadline timestamp is invalid') from exc
|
||||
if (
|
||||
issued.tzinfo is None
|
||||
or deadline.tzinfo is None
|
||||
or issued.utcoffset() != timedelta(0)
|
||||
or deadline.utcoffset() != timedelta(0)
|
||||
or issued.isoformat(timespec='seconds') != issued_at
|
||||
or deadline.isoformat(timespec='seconds') != deadline_at
|
||||
or deadline - issued != timedelta(seconds=deadlines['assignment_ttl_seconds'])
|
||||
):
|
||||
raise ScanExecutionError('worker assignment deadline timestamp is invalid')
|
||||
if deadlines['target_scan_timeout_seconds'] != scan_kwargs.get('timeout_sec'):
|
||||
raise ScanExecutionError('worker assignment target scan timeout changed')
|
||||
return deadlines
|
||||
|
||||
|
||||
def _validate_remote_assignment(
|
||||
assignment, local_build, expected_protocol_version,
|
||||
package_capabilities=None,
|
||||
):
|
||||
if not isinstance(assignment, dict) or set(assignment) != {
|
||||
'reservation', 'deadlines', 'compatibility', 'scan_kwargs', 'event_scan_options',
|
||||
'queue_policy', 'limits', 'scan_policy', 'execution_snapshot',
|
||||
'execution_snapshot_sha256', 'execution_plan',
|
||||
}:
|
||||
raise ScanExecutionError('worker assignment shape is invalid')
|
||||
reservation_value = dict(assignment.get('reservation') or {})
|
||||
try:
|
||||
reservation = BundleReservation.from_mapping(reservation_value)
|
||||
except (KeyError, TypeError, ValueError) as exc:
|
||||
raise ScanExecutionError('worker assignment reservation is invalid') from exc
|
||||
source = str(reservation.source or '').strip().lower()
|
||||
platform = str(reservation.platform or '').strip().lower()
|
||||
if (
|
||||
reservation_value.get('assignment_kind') != 'remote'
|
||||
or not source or not platform
|
||||
):
|
||||
raise ScanExecutionError('worker assignment reservation is not remote')
|
||||
|
||||
snapshot = normalize_remote_execution_snapshot(
|
||||
assignment.get('execution_snapshot'),
|
||||
)
|
||||
snapshot_sha256 = _digest(
|
||||
assignment.get('execution_snapshot_sha256'), 'execution snapshot',
|
||||
)
|
||||
if not hmac.compare_digest(
|
||||
snapshot_sha256, remote_execution_snapshot_sha256(snapshot),
|
||||
):
|
||||
raise ScanExecutionError('worker assignment execution snapshot hash changed')
|
||||
if snapshot['compatibility']['protocol_version'] != expected_protocol_version:
|
||||
raise ScanExecutionError('worker assignment snapshot protocol is incompatible')
|
||||
required = ScanCompatibility.from_mapping(assignment.get('compatibility'))
|
||||
if required.as_dict() != snapshot['compatibility']:
|
||||
raise ScanExecutionError('worker assignment compatibility changed after admission')
|
||||
validate_remote_assignment_compatibility(
|
||||
required, local_build, platform, assignment.get('scan_kwargs'),
|
||||
assignment.get('event_scan_options'), assignment.get('queue_policy'),
|
||||
assignment.get('limits'), assignment.get('scan_policy'),
|
||||
expected_protocol_version=expected_protocol_version,
|
||||
)
|
||||
effective, execution = remote_execution_identity(
|
||||
platform, assignment.get('scan_kwargs'),
|
||||
assignment.get('event_scan_options'), assignment.get('queue_policy'),
|
||||
assignment.get('limits'), assignment.get('scan_policy'),
|
||||
)
|
||||
if execution != snapshot['execution']:
|
||||
raise ScanExecutionError('worker assignment settings changed after admission')
|
||||
if (
|
||||
snapshot['credential_ref']['source'] != source
|
||||
or snapshot['execution']['source'] != platform
|
||||
or not hmac.compare_digest(effective, required.effective_config_sha256)
|
||||
or not hmac.compare_digest(
|
||||
str(reservation_value.get('remote_effective_config_sha256') or ''),
|
||||
required.effective_config_sha256,
|
||||
)
|
||||
):
|
||||
raise ScanExecutionError('worker assignment identity changed after admission')
|
||||
|
||||
planning_kind = snapshot['planning']['kind']
|
||||
if expected_protocol_version == 1 and planning_kind != 'exact_git_v1':
|
||||
raise ScanExecutionError('legacy worker assignment planning kind is invalid')
|
||||
capability = (source, platform, planning_kind)
|
||||
if package_capabilities is not None:
|
||||
capabilities = {
|
||||
tuple(value) for value in package_capabilities
|
||||
if isinstance(value, (list, tuple)) and len(value) == 3
|
||||
}
|
||||
if capability not in capabilities:
|
||||
raise ScanExecutionError(
|
||||
'worker assignment capability is not supported by this package'
|
||||
)
|
||||
|
||||
plan = assignment.get('execution_plan')
|
||||
if not isinstance(plan, dict) or set(plan) != {
|
||||
'kind', 'execution_target', 'bound_plan',
|
||||
} or plan.get('kind') != planning_kind:
|
||||
raise ScanExecutionError('worker assignment execution plan is invalid')
|
||||
scan_kwargs = dict(assignment.get('scan_kwargs') or {})
|
||||
event_scan_options = dict(assignment.get('event_scan_options') or {})
|
||||
deadlines = _normalize_remote_assignment_deadlines(
|
||||
assignment.get('deadlines'), reservation_value, scan_kwargs,
|
||||
)
|
||||
if planning_kind == 'exact_git_v1':
|
||||
if (
|
||||
source not in {'github', 'gitlab'} or platform != source
|
||||
or str(plan.get('execution_target') or '') != reservation.target
|
||||
or not isinstance(plan.get('bound_plan'), dict)
|
||||
or scan_kwargs.get('git_plan') != plan['bound_plan']
|
||||
or scan_kwargs.get('docker_layer_work') is not None
|
||||
):
|
||||
raise ScanExecutionError('worker assignment exact Git plan is invalid')
|
||||
execution_target = reservation.target
|
||||
elif planning_kind == 'docker_direct_v1':
|
||||
if source != 'dockerhub' or platform != 'docker':
|
||||
raise ScanExecutionError('worker assignment Docker capability is invalid')
|
||||
parsed = normalize_docker_direct_execution_target(reservation.target)
|
||||
execution_target = parsed['image']
|
||||
if parsed['normalized_target'] != reservation.normalized_target:
|
||||
raise ScanExecutionError('worker assignment Docker identity is invalid')
|
||||
elif planning_kind == 'huggingface_space_v1':
|
||||
if source != 'huggingface' or platform != 'huggingface':
|
||||
raise ScanExecutionError('worker assignment HuggingFace capability is invalid')
|
||||
execution_target = normalize_huggingface_space_execution_target(
|
||||
reservation.target,
|
||||
)
|
||||
if normalize_target(execution_target, platform) != reservation.normalized_target:
|
||||
raise ScanExecutionError('worker assignment HuggingFace identity is invalid')
|
||||
else:
|
||||
raise ScanExecutionError('worker assignment planning kind is unsupported')
|
||||
if planning_kind != 'exact_git_v1' and (
|
||||
plan.get('bound_plan') is not None
|
||||
or str(plan.get('execution_target') or '') != execution_target
|
||||
or scan_kwargs != event_scan_options
|
||||
or any(name in scan_kwargs for name in (
|
||||
'token', 'git_plan', 'docker_layer_work',
|
||||
))
|
||||
):
|
||||
raise ScanExecutionError('worker assignment direct plan is not credential-free')
|
||||
return {
|
||||
'reservation': reservation,
|
||||
'snapshot': snapshot,
|
||||
'snapshot_sha256': snapshot_sha256,
|
||||
'compatibility': required,
|
||||
'planning_kind': planning_kind,
|
||||
'execution_target': execution_target,
|
||||
'deadlines': deadlines,
|
||||
'execution_plan': {
|
||||
'kind': planning_kind,
|
||||
'execution_target': execution_target,
|
||||
'bound_plan': plan.get('bound_plan'),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def validate_protocol1_remote_assignment(assignment, local_build):
|
||||
return _validate_remote_assignment(assignment, local_build, 1)
|
||||
|
||||
|
||||
def validate_protocol2_remote_assignment(
|
||||
assignment, local_build, package_capabilities=None,
|
||||
):
|
||||
return _validate_remote_assignment(
|
||||
assignment, local_build, PROTOCOL_VERSION, package_capabilities,
|
||||
)
|
||||
|
||||
|
||||
def _first_error_line(result):
|
||||
for error in result.get('errors') or ():
|
||||
for line in str(error).splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
payload = json.loads(line)
|
||||
except (TypeError, ValueError):
|
||||
return line[:300]
|
||||
return str(payload.get('error') or payload.get('msg') or line)[:300]
|
||||
return ''
|
||||
|
||||
|
||||
def _docker_result_resets_attempts(result):
|
||||
if result.get('docker_layer_plan') is None or not result.get('retryable', False):
|
||||
return False
|
||||
execution = result.get('docker_layer_execution')
|
||||
records = execution.get('blobs') if isinstance(execution, dict) else None
|
||||
descriptors = result['docker_layer_plan'].get('descriptors')
|
||||
if not isinstance(records, list) or not isinstance(descriptors, list):
|
||||
return False
|
||||
if not any(
|
||||
item.get('coverage_state') in ('selected', 'shared_pending')
|
||||
for item in descriptors if isinstance(item, dict)
|
||||
):
|
||||
return False
|
||||
return not any(
|
||||
item.get('status') in ('retryable_failed', 'terminal_failed')
|
||||
for item in records if isinstance(item, dict)
|
||||
)
|
||||
|
||||
|
||||
def queue_disposition_for_result(result, platform, attempts, policy, *, now=None):
|
||||
policy = policy if isinstance(policy, QueueDispositionPolicy) else QueueDispositionPolicy(**policy)
|
||||
attempts = max(0, int(attempts or 0))
|
||||
max_attempts = max(1, int(policy.target_retry_max_attempts or 1))
|
||||
now = now or datetime.now(timezone.utc)
|
||||
skipped = str(result.get('skipped') or '')
|
||||
if skipped in set(policy.soft_skip_reasons):
|
||||
return {
|
||||
'queue_status': 'deferred', 'queue_error': skipped,
|
||||
'available_after': (now + timedelta(days=max(1, policy.ci_soft_cooldown_days))).isoformat(timespec='seconds'),
|
||||
'reset_attempts': True,
|
||||
}
|
||||
if result.get('docker_layer_plan') is not None:
|
||||
if not result.get('errors'):
|
||||
status, available_after, reset = 'done', None, False
|
||||
elif not bool(result.get('retryable', False)):
|
||||
status, available_after, reset = 'failed', None, False
|
||||
else:
|
||||
reset = _docker_result_resets_attempts(result)
|
||||
if not reset and attempts >= max_attempts:
|
||||
status, available_after = 'failed', None
|
||||
else:
|
||||
status = 'deferred'
|
||||
available_after = (now + timedelta(seconds=max(
|
||||
1, int(policy.docker_layer_checkpoint_delay_sec or 1),
|
||||
))).isoformat(timespec='seconds')
|
||||
return {
|
||||
'queue_status': status,
|
||||
'queue_error': _first_error_line(result) if result.get('errors') else None,
|
||||
'available_after': available_after, 'reset_attempts': reset,
|
||||
}
|
||||
if not result.get('errors'):
|
||||
return {
|
||||
'queue_status': 'done', 'queue_error': None,
|
||||
'available_after': None, 'reset_attempts': False,
|
||||
}
|
||||
timed_out = bool((result.get('scan_meta') or {}).get('command_timed_out')) \
|
||||
or result.get('error_class') == 'timeout'
|
||||
if timed_out:
|
||||
status = 'failed' if attempts >= max_attempts else 'deferred'
|
||||
delay = max(60, int(policy.target_timeout_retry_delay_sec or 60))
|
||||
elif result.get('source_failure'):
|
||||
status = 'failed' if not result.get('retryable', True) and attempts >= max_attempts else 'deferred'
|
||||
delay = max(1, int(policy.target_retry_max_delay_sec or 1))
|
||||
elif not result.get('retryable', True) or attempts >= max_attempts:
|
||||
status, delay = 'failed', 0
|
||||
else:
|
||||
status = 'deferred'
|
||||
base = max(1, int(policy.target_retry_base_delay_sec or 1))
|
||||
maximum = max(base, int(policy.target_retry_max_delay_sec or base))
|
||||
delay = min(maximum, base * (2 ** max(0, attempts - 1)))
|
||||
return {
|
||||
'queue_status': status,
|
||||
'queue_error': _first_error_line(result),
|
||||
'available_after': (
|
||||
(now + timedelta(seconds=delay)).isoformat(timespec='seconds')
|
||||
if status == 'deferred' else None
|
||||
),
|
||||
'reset_attempts': bool(
|
||||
(result.get('source_failure') and result.get('retryable', True))
|
||||
or _docker_result_resets_attempts(result)
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def stage_scan_result_in_scope(
|
||||
result, reservation, bundle_root, event_scan_options, queue_policy, *, attempts,
|
||||
candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024,
|
||||
require_s_drive=False, fault=None, diagnostic_slot_id=0,
|
||||
):
|
||||
reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation)
|
||||
disposition = queue_disposition_for_result(
|
||||
result, reservation.platform, attempts, queue_policy,
|
||||
)
|
||||
return stage_result_bundle(
|
||||
result, reservation, bundle_root, event_scan_options, disposition,
|
||||
candidate_max_items=candidate_max_items,
|
||||
candidate_max_bytes=candidate_max_bytes,
|
||||
require_s_drive=require_s_drive, fault=fault,
|
||||
diagnostic_slot_id=diagnostic_slot_id,
|
||||
diagnostic_attempt=max(1, int(attempts or 1)),
|
||||
)
|
||||
|
||||
|
||||
def execute_planned_result_in_scope(
|
||||
reservation, bundle_root, scan_kwargs, event_scan_options, queue_policy, *,
|
||||
attempts, candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024,
|
||||
execution_target=None, scan_meta_defaults=None, require_s_drive=False,
|
||||
phase_callback=None, bundle_fault=None, diagnostic_slot_id=0,
|
||||
):
|
||||
reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation)
|
||||
scan_kwargs = validate_scan_kwargs(reservation.platform, scan_kwargs)
|
||||
target = reservation.target if execution_target is None else execution_target
|
||||
expected = reservation.normalized_target or normalize_target(reservation.target, reservation.platform)
|
||||
if normalize_target(target, reservation.platform) != expected:
|
||||
raise ScanExecutionError('execution target does not match the reservation')
|
||||
with client_scan_phase_events(phase_callback):
|
||||
result = scan_target_result(
|
||||
target, reservation.platform, reservation.scan_event_id, scan_kwargs,
|
||||
)
|
||||
result['target'] = reservation.target
|
||||
result['scan_type'] = reservation.platform
|
||||
if scan_meta_defaults:
|
||||
metadata = result.setdefault('scan_meta', {})
|
||||
if not isinstance(metadata, dict):
|
||||
raise ScanExecutionError('scanner metadata is invalid')
|
||||
for name, value in dict(scan_meta_defaults).items():
|
||||
metadata.setdefault(name, value)
|
||||
if phase_callback is not None:
|
||||
phase_callback('cleaning')
|
||||
cleanup = cleanup_assignment_work_dir()
|
||||
phase_callback('cleaning', cleanup)
|
||||
phase_callback('bundling')
|
||||
return stage_scan_result_in_scope(
|
||||
result, reservation, bundle_root, event_scan_options, queue_policy,
|
||||
attempts=attempts, candidate_max_items=candidate_max_items,
|
||||
candidate_max_bytes=candidate_max_bytes, require_s_drive=require_s_drive,
|
||||
fault=bundle_fault, diagnostic_slot_id=diagnostic_slot_id,
|
||||
)
|
||||
|
||||
|
||||
def execute_planned_claim(
|
||||
reservation, bundle_root, scan_kwargs, event_scan_options, queue_policy, scan_policy, *,
|
||||
attempts, candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024,
|
||||
lease=None, execution_target=None, scan_meta_defaults=None,
|
||||
require_s_drive=False, phase_callback=None, bundle_fault=None,
|
||||
diagnostic_slot_id=0,
|
||||
):
|
||||
timeout = validate_scan_kwargs(
|
||||
reservation.platform if isinstance(reservation, BundleReservation) else reservation.get('platform'),
|
||||
scan_kwargs,
|
||||
)['timeout_sec']
|
||||
platform = reservation.platform if isinstance(reservation, BundleReservation) else reservation.get('platform')
|
||||
policy = normalize_remote_scan_policy(scan_policy)
|
||||
if phase_callback is not None:
|
||||
phase_callback('waiting_permit', {'boundary': 'scan_slot_scope'})
|
||||
with client_scan_execution_policy(policy):
|
||||
with scan_slot_scope(['scan-target', platform], timeout, lease=lease):
|
||||
return execute_planned_result_in_scope(
|
||||
reservation, bundle_root, scan_kwargs, event_scan_options, queue_policy,
|
||||
attempts=attempts, candidate_max_items=candidate_max_items,
|
||||
candidate_max_bytes=candidate_max_bytes, execution_target=execution_target,
|
||||
scan_meta_defaults=scan_meta_defaults, require_s_drive=require_s_drive,
|
||||
phase_callback=phase_callback, bundle_fault=bundle_fault,
|
||||
diagnostic_slot_id=diagnostic_slot_id,
|
||||
)
|
||||
|
||||
|
||||
def execute_protocol2_remote_claim(
|
||||
validated_assignment, bundle_root, scan_kwargs, event_scan_options,
|
||||
queue_policy, scan_policy, *, attempts, candidate_max_items=2000,
|
||||
candidate_max_bytes=2 * 1024 * 1024, lease=None,
|
||||
scan_meta_defaults=None, require_s_drive=False, phase_callback=None,
|
||||
bundle_fault=None, diagnostic_slot_id=0,
|
||||
):
|
||||
kind = str(validated_assignment.get('planning_kind') or '')
|
||||
authority = (
|
||||
client_remote_execution_binding(kind)
|
||||
if kind in {'docker_direct_v1', 'huggingface_space_v1'}
|
||||
else nullcontext()
|
||||
)
|
||||
with authority:
|
||||
return execute_planned_claim(
|
||||
validated_assignment['reservation'], bundle_root, scan_kwargs,
|
||||
event_scan_options, queue_policy, scan_policy,
|
||||
attempts=attempts,
|
||||
candidate_max_items=candidate_max_items,
|
||||
candidate_max_bytes=candidate_max_bytes,
|
||||
lease=lease,
|
||||
execution_target=validated_assignment['execution_target'],
|
||||
scan_meta_defaults={
|
||||
**dict(scan_meta_defaults or {}), 'planning_kind': kind,
|
||||
},
|
||||
require_s_drive=require_s_drive,
|
||||
phase_callback=phase_callback, bundle_fault=bundle_fault,
|
||||
diagnostic_slot_id=diagnostic_slot_id,
|
||||
)
|
||||
|
||||
|
||||
def canonical_json_sha256(value):
|
||||
return hashlib.sha256(json.dumps(
|
||||
value, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
|
||||
).encode('utf-8')).hexdigest()
|
||||
@@ -0,0 +1,71 @@
|
||||
import json
|
||||
import threading
|
||||
|
||||
|
||||
RETIRED_MESSAGE = 'Legacy ScanManager mutation controls are retired; use the authenticated supervisor.'
|
||||
|
||||
|
||||
class ScanManager:
|
||||
"""Read-only compatibility shell for the retired legacy scanner UI."""
|
||||
|
||||
def __init__(self):
|
||||
self.current_scan = None
|
||||
self.scan_thread = None
|
||||
self.results = []
|
||||
self.progress = {
|
||||
'total': 0,
|
||||
'completed': 0,
|
||||
'failed': 0,
|
||||
'secrets_found': 0,
|
||||
'current_target': None,
|
||||
'status': 'retired',
|
||||
}
|
||||
self.scan_lock = threading.Lock()
|
||||
self.scanned_targets = set()
|
||||
self.scan_history = []
|
||||
|
||||
def start_scan(self, scan_type, targets, scan_options):
|
||||
return False, RETIRED_MESSAGE
|
||||
|
||||
def pause_scan(self):
|
||||
return False
|
||||
|
||||
def resume_scan(self):
|
||||
return False
|
||||
|
||||
def stop_scan(self):
|
||||
return False
|
||||
|
||||
def is_scanning(self):
|
||||
return False
|
||||
|
||||
def get_progress(self):
|
||||
with self.scan_lock:
|
||||
return self.progress.copy()
|
||||
|
||||
def get_results(self):
|
||||
with self.scan_lock:
|
||||
return self.results.copy()
|
||||
|
||||
def clear_results(self):
|
||||
return False
|
||||
|
||||
def get_scan_info(self):
|
||||
with self.scan_lock:
|
||||
return self.current_scan.copy() if self.current_scan else None
|
||||
|
||||
def export_results(self, format='json'):
|
||||
with self.scan_lock:
|
||||
if str(format).lower() == 'json':
|
||||
return json.dumps(self.results, indent=2, default=str)
|
||||
return None
|
||||
|
||||
def clear_scan_history(self):
|
||||
return False
|
||||
|
||||
def get_scan_history(self):
|
||||
with self.scan_lock:
|
||||
return self.scan_history.copy()
|
||||
|
||||
def get_scanned_targets_count(self):
|
||||
return len(self.scanned_targets)
|
||||
+17045
File diff suppressed because it is too large
Load Diff
+30695
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,423 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import json
|
||||
import gzip
|
||||
import os
|
||||
import tarfile
|
||||
import tempfile
|
||||
import uuid
|
||||
import zipfile
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from types import SimpleNamespace
|
||||
|
||||
import scanner as scanner_module
|
||||
from console_runner import queue_error_disposition, target_retry_delay_sec
|
||||
from db_backend import redact_database_url
|
||||
from scanner import (
|
||||
apply_trufflehog_diagnostics,
|
||||
build_authenticated_git_url,
|
||||
cached_gharchive_hour,
|
||||
convert_package_git_unavailable_to_skip,
|
||||
docker_tag_platform_support,
|
||||
fetch_dockerhub_images,
|
||||
normalize_git_repo_candidate,
|
||||
redact_command_args,
|
||||
run_command,
|
||||
safe_extract_package_zip,
|
||||
safe_extract_tar,
|
||||
)
|
||||
from scanner_db import ScannerDB, normalize_target, target_status
|
||||
from runtime_security import ensure_private_directory, harden_private_file
|
||||
|
||||
|
||||
def diagnostic(message, error, **extra):
|
||||
return json.dumps({'level': 'error', 'msg': message, 'error': error, **extra})
|
||||
|
||||
|
||||
def assert_diagnostic_policy():
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(
|
||||
result,
|
||||
diagnostic('non-critical error processing chunk', 'error reading chunk: brotli: PADDING_2'),
|
||||
0,
|
||||
'pypi',
|
||||
)
|
||||
assert not result.get('errors')
|
||||
assert result.get('warnings') and result.get('degraded')
|
||||
assert target_status(result) == 'degraded'
|
||||
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(result, diagnostic('Skipping result: invalid', 'empty raw'), 0, 'npm')
|
||||
assert not result.get('errors') and result.get('warnings')
|
||||
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(
|
||||
result,
|
||||
diagnostic('error reading chunk', 'read tcp: connection reset by peer'),
|
||||
0,
|
||||
'pypi',
|
||||
)
|
||||
assert result.get('errors') and result.get('retryable') is True
|
||||
assert result.get('error_class') == 'network'
|
||||
|
||||
for detail in ('unexpected EOF', 'permission denied', 'no space left on device'):
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(result, diagnostic('error reading chunk', detail), 0, 'pypi')
|
||||
assert result.get('errors'), detail
|
||||
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(result, diagnostic('error reading chunk', 'brotli: PADDING_2'), 0, 'git')
|
||||
assert result.get('errors')
|
||||
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(result, diagnostic('space', 'no repo found for repo'), 0, 'huggingface')
|
||||
assert result.get('skipped') and not result.get('errors')
|
||||
assert target_status(result) == 'skipped'
|
||||
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(
|
||||
result,
|
||||
diagnostic('error processing image', 'no child with platform linux/amd64 in index image:tag'),
|
||||
1,
|
||||
'docker',
|
||||
)
|
||||
assert result.get('skipped') and not result.get('errors')
|
||||
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(
|
||||
result,
|
||||
diagnostic('error processing layer', 'gzip: invalid header') + '\n' + json.dumps({'level': 'info-0', 'msg': 'finished scanning'}),
|
||||
0,
|
||||
'docker',
|
||||
)
|
||||
assert result.get('degraded') and not result.get('errors')
|
||||
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(
|
||||
result,
|
||||
diagnostic('a detector ignored the context timeout', 'context deadline exceeded') + '\n' + json.dumps({'level': 'info-0', 'msg': 'finished scanning'}),
|
||||
0,
|
||||
'docker',
|
||||
)
|
||||
assert result.get('degraded') and not result.get('errors')
|
||||
assert result.get('warning_classes') == ['detector_timeout']
|
||||
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(
|
||||
result,
|
||||
diagnostic('error processing layer', 'gzip: invalid header'),
|
||||
1,
|
||||
'docker',
|
||||
)
|
||||
assert result.get('errors') and not result.get('degraded')
|
||||
|
||||
result = {'findings': [], 'errors': []}
|
||||
apply_trufflehog_diagnostics(result, '', 2, 'filesystem')
|
||||
assert result.get('errors') and result.get('retryable') is True
|
||||
|
||||
result['findings'] = [{'DetectorName': 'Example'}]
|
||||
assert target_status(result) == 'error'
|
||||
|
||||
result = {'errors': [diagnostic('error running scan', 'remote: Repository not found.')], 'findings': []}
|
||||
convert_package_git_unavailable_to_skip(result)
|
||||
assert result.get('skipped') and not result.get('errors')
|
||||
|
||||
|
||||
def assert_docker_platform_policy():
|
||||
assert docker_tag_platform_support({'images': [{'os': 'linux', 'architecture': 'amd64'}]}) is True
|
||||
assert docker_tag_platform_support({'images': [{'os': 'linux', 'architecture': 'arm64'}]}) is False
|
||||
assert docker_tag_platform_support({'images': []}) is None
|
||||
assert docker_tag_platform_support({'images': [{'os': 'linux'}]}) is None
|
||||
assert docker_tag_platform_support({'images': [{'os': 'unknown', 'architecture': 'unknown'}]}) is None
|
||||
assert docker_tag_platform_support({'images': [{'os': 'linux', 'architecture': 'arm64'}, {'os': 'unknown', 'architecture': 'unknown'}]}) is None
|
||||
assert docker_tag_platform_support({}) is None
|
||||
assert normalize_target('Owner/Repo:Prod', 'docker') == 'owner/repo:prod'
|
||||
assert normalize_git_repo_candidate('https://[invalid url, do not cite]/repo') is None
|
||||
|
||||
|
||||
def assert_docker_partial_pagination_policy():
|
||||
class Response:
|
||||
def __init__(self, page):
|
||||
self.page = page
|
||||
|
||||
def raise_for_status(self):
|
||||
if self.page > 1:
|
||||
raise RuntimeError('page outside result set')
|
||||
|
||||
def json(self):
|
||||
return {'count': 1, 'results': [{'repo_name': 'owner/repo'}]}
|
||||
|
||||
original = scanner_module.api_request
|
||||
try:
|
||||
def fake_request(method, url, **kwargs):
|
||||
page = int(url.split('page=', 1)[1].split('&', 1)[0])
|
||||
return Response(page)
|
||||
|
||||
scanner_module.api_request = fake_request
|
||||
assert fetch_dockerhub_images('test', 3, per_page=10, resolve_tags=False) == ['owner/repo']
|
||||
finally:
|
||||
scanner_module.api_request = original
|
||||
|
||||
|
||||
def assert_retry_policy():
|
||||
assert target_retry_delay_sec(1, 60, 3600) == 60
|
||||
assert target_retry_delay_sec(2, 60, 3600) == 120
|
||||
assert target_retry_delay_sec(10, 60, 3600) == 3600
|
||||
|
||||
args = SimpleNamespace(target_retry_max_attempts=3, target_timeout_retry_delay_sec=21600)
|
||||
status, available_after, attempts, max_attempts = queue_error_disposition(
|
||||
None, 'github', 'github', 'https://github.com/o/r',
|
||||
{'errors': ['timeout'], 'error_class': 'timeout', 'scan_meta': {'command_timed_out': True}},
|
||||
args, {'attempts': 3},
|
||||
)
|
||||
assert status == 'failed' and available_after is None and attempts == 3 and max_attempts == 3
|
||||
|
||||
|
||||
def assert_timeout_output_policy():
|
||||
marker = '{"DetectorName":"PartialFinding"}'
|
||||
old_work_dir = scanner_module.scan_config.work_dir
|
||||
old_min_free = scanner_module.scan_config.min_free_gb
|
||||
old_authority_check = scanner_module.require_trufflehog_launch_authority
|
||||
with tempfile.TemporaryDirectory() as temp_dir:
|
||||
work_dir = os.path.join(temp_dir, 'work')
|
||||
ensure_private_directory(work_dir, reject_reparse=True)
|
||||
scanner_module.initialize_scanner_runtime(preflight_complete=True, register_cleanup=False)
|
||||
scanner_module.scan_config.work_dir = work_dir
|
||||
scanner_module.scan_config.min_free_gb = 0
|
||||
scanner_module.require_trufflehog_launch_authority = lambda _command: None
|
||||
try:
|
||||
stdout, stderr, returncode = run_command(
|
||||
[sys.executable, '-c', f'import time; print({marker!r}, flush=True); time.sleep(5)'],
|
||||
1,
|
||||
)
|
||||
finally:
|
||||
scanner_module.scan_config.work_dir = old_work_dir
|
||||
scanner_module.scan_config.min_free_gb = old_min_free
|
||||
scanner_module.require_trufflehog_launch_authority = old_authority_check
|
||||
assert returncode == -1 and marker in stdout and 'timed out' in stderr.lower()
|
||||
|
||||
|
||||
def assert_archive_and_secret_safety():
|
||||
url, secrets = build_authenticated_git_url('https://attacker.example/repo.git', 'github', 'sentinel-token')
|
||||
assert url == 'https://attacker.example/repo.git' and not secrets and 'sentinel-token' not in url
|
||||
redacted = redact_command_args(['git', 'https://x-access-token:sentinel-token@github.com/org/repo.git', '--token', 'sentinel-token'])
|
||||
assert all('sentinel-token' not in value for value in redacted)
|
||||
|
||||
with tempfile.TemporaryDirectory() as temp_dir:
|
||||
tar_path = os.path.join(temp_dir, 'unsafe.tar')
|
||||
with tarfile.open(tar_path, 'w') as archive:
|
||||
link = tarfile.TarInfo('link')
|
||||
link.type = tarfile.SYMTYPE
|
||||
link.linkname = '..'
|
||||
archive.addfile(link)
|
||||
try:
|
||||
safe_extract_tar(tar_path, os.path.join(temp_dir, 'tar-out'))
|
||||
raise AssertionError('unsafe tar link was accepted')
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
zip_path = os.path.join(temp_dir, 'large.zip')
|
||||
with zipfile.ZipFile(zip_path, 'w') as archive:
|
||||
archive.writestr('large.txt', b'x' * 2048)
|
||||
try:
|
||||
safe_extract_package_zip(
|
||||
zip_path, os.path.join(temp_dir, 'zip-out'),
|
||||
max_total_size_mb=0, max_file_size_mb=0,
|
||||
)
|
||||
except ValueError:
|
||||
raise AssertionError('disabled package zip bounds rejected a valid archive')
|
||||
|
||||
crowded_tar = os.path.join(temp_dir, 'crowded.tar')
|
||||
with tarfile.open(crowded_tar, 'w') as archive:
|
||||
archive.addfile(tarfile.TarInfo('one'))
|
||||
archive.addfile(tarfile.TarInfo('two'))
|
||||
try:
|
||||
safe_extract_tar(crowded_tar, os.path.join(temp_dir, 'crowded-out'), max_files=1)
|
||||
raise AssertionError('tar member bound was not enforced')
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
assert 'query-secret' not in redact_database_url('postgresql://u@localhost/db?password=query-secret')
|
||||
malformed = ScannerDB(db_path=os.path.join(tempfile.gettempdir(), 'must-not-open.db'), db_url='postgreql://bad')
|
||||
assert malformed.postgres_required and not malformed.enabled and malformed.path is None
|
||||
|
||||
with tempfile.TemporaryDirectory() as temp_dir:
|
||||
runtime_dir = os.path.join(temp_dir, 'runtime')
|
||||
cache_dir = os.path.join(runtime_dir, 'gharchive')
|
||||
ensure_private_directory(runtime_dir)
|
||||
ensure_private_directory(cache_dir)
|
||||
hour = datetime(2026, 7, 11, 12, tzinfo=timezone.utc)
|
||||
cache_path = os.path.join(cache_dir, '2026-07-11-12.json.gz')
|
||||
with gzip.open(cache_path, 'wb') as archive:
|
||||
archive.write(b'{"type":"PushEvent"}\n')
|
||||
harden_private_file(cache_path)
|
||||
old_runtime = scanner_module.scan_config.runtime_dir
|
||||
old_free = scanner_module.scan_config.gharchive_cache_min_free_bytes
|
||||
try:
|
||||
scanner_module.scan_config.runtime_dir = runtime_dir
|
||||
scanner_module.scan_config.gharchive_cache_min_free_bytes = 0
|
||||
assert scanner_module.canonical_path(
|
||||
cached_gharchive_hour(hour, cache_dir, request_timeout=1, retries=1)
|
||||
) == scanner_module.canonical_path(cache_path)
|
||||
finally:
|
||||
scanner_module.scan_config.runtime_dir = old_runtime
|
||||
scanner_module.scan_config.gharchive_cache_min_free_bytes = old_free
|
||||
|
||||
|
||||
def assert_queue_policy():
|
||||
old_urls = {key: os.environ.pop(key, None) for key in ('SCANNER_DB_URL', 'DATABASE_URL')}
|
||||
try:
|
||||
with tempfile.TemporaryDirectory() as temp_dir:
|
||||
db = ScannerDB(db_path=os.path.join(temp_dir, 'scanner.db'))
|
||||
try:
|
||||
assert db.enabled
|
||||
target = 'owner/repo:tag'
|
||||
db.enqueue_targets('dockerhub', 'docker', 'q', [target])
|
||||
claimed = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 3600, return_rows=True)
|
||||
assert [row['target'] for row in claimed] == [target]
|
||||
first_claim = claimed[0]
|
||||
row = db.target_queue_item('dockerhub', 'docker', target)
|
||||
assert row['status'] == 'in_progress' and int(row['attempts']) == 1
|
||||
assert db.reclaim_target_leases('dockerhub', 'docker', 'owner-b') == 0
|
||||
|
||||
future = (datetime.now(timezone.utc) + timedelta(hours=1)).isoformat(timespec='seconds')
|
||||
db.complete_target_queue_item(
|
||||
'dockerhub', 'docker', target, None, 'deferred', 'network', future,
|
||||
queue_id=first_claim['id'], lease_token=first_claim['lease_token'],
|
||||
)
|
||||
db.sync_target_queue_from_files('dockerhub', 'docker', [target], [], 'q')
|
||||
row = db.target_queue_item('dockerhub', 'docker', target)
|
||||
assert row['status'] == 'deferred' and row['available_after'] == future
|
||||
updated_at = db.conn.execute(
|
||||
'SELECT updated_at FROM target_queue WHERE source = ? AND normalized_target = ?',
|
||||
('dockerhub', normalize_target(target, 'docker')),
|
||||
).fetchone()['updated_at']
|
||||
db.sync_target_queue_from_files('dockerhub', 'docker', [target], [], 'q')
|
||||
assert db.conn.execute(
|
||||
'SELECT updated_at FROM target_queue WHERE source = ? AND normalized_target = ?',
|
||||
('dockerhub', normalize_target(target, 'docker')),
|
||||
).fetchone()['updated_at'] == updated_at
|
||||
assert db.claim_targets('dockerhub', 'docker', 1, 'owner-b', 3600) == []
|
||||
|
||||
db.conn.execute(
|
||||
"UPDATE target_queue SET available_after = ? WHERE source = ?",
|
||||
('2000-01-01T00:00:00+00:00', 'dockerhub'),
|
||||
)
|
||||
db.conn.commit()
|
||||
reclaimed = db.claim_targets('dockerhub', 'docker', 1, 'owner-b', 3600, return_rows=True)
|
||||
assert [item['target'] for item in reclaimed] == [target]
|
||||
db.complete_target_queue_item(
|
||||
'dockerhub', 'docker', target, None, 'failed', 'permanent',
|
||||
queue_id=reclaimed[0]['id'], lease_token=reclaimed[0]['lease_token'],
|
||||
)
|
||||
db.sync_target_queue_from_files('dockerhub', 'docker', [], [target], 'q')
|
||||
row = db.target_queue_item('dockerhub', 'docker', target)
|
||||
assert row['status'] == 'failed' and row['available_after'] is None
|
||||
|
||||
done_target = 'owner/other:tag'
|
||||
db.enqueue_targets('dockerhub', 'docker', 'q', [done_target])
|
||||
done_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 3600, return_rows=True)[0]
|
||||
db.complete_target_queue_item(
|
||||
'dockerhub', 'docker', done_target, None, 'done',
|
||||
queue_id=done_claim['id'], lease_token=done_claim['lease_token'],
|
||||
)
|
||||
db.enqueue_targets('dockerhub', 'docker', 'q', [done_target], requeue_done=True)
|
||||
row = db.target_queue_item('dockerhub', 'docker', done_target)
|
||||
assert row['status'] == 'pending' and int(row['attempts']) == 0
|
||||
done_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 3600, return_rows=True)[0]
|
||||
db.complete_target_queue_item(
|
||||
'dockerhub', 'docker', done_target, None, 'done',
|
||||
queue_id=done_claim['id'], lease_token=done_claim['lease_token'],
|
||||
)
|
||||
|
||||
legacy_target = 'owner/legacy:tag'
|
||||
db.enqueue_targets('dockerhub', 'docker', 'q', [legacy_target])
|
||||
db.conn.execute(
|
||||
"UPDATE target_queue SET status = 'pending', available_after = ? WHERE normalized_target = ?",
|
||||
(future, normalize_target(legacy_target, 'docker')),
|
||||
)
|
||||
db.conn.commit()
|
||||
db.sync_target_queue_from_files('dockerhub', 'docker', [legacy_target], [], 'q')
|
||||
row = db.target_queue_item('dockerhub', 'docker', legacy_target)
|
||||
assert row['status'] == 'deferred' and row['available_after'] == future
|
||||
|
||||
stale_target = 'owner/stale:tag'
|
||||
db.enqueue_targets('dockerhub', 'docker', 'q', [stale_target])
|
||||
stale_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 60, max_attempts=3, return_rows=True)[0]
|
||||
assert stale_claim['target'] == stale_target
|
||||
db.conn.execute(
|
||||
"UPDATE target_queue SET lease_expires_at = ? WHERE normalized_target = ?",
|
||||
('2000-01-01T00:00:00+00:00', normalize_target(stale_target, 'docker')),
|
||||
)
|
||||
db.conn.commit()
|
||||
newer_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-b', 60, max_attempts=3, return_rows=True)[0]
|
||||
assert newer_claim['target'] == stale_target
|
||||
assert newer_claim['lease_token'] != stale_claim['lease_token']
|
||||
assert not db.complete_target_queue_item(
|
||||
'dockerhub', 'docker', stale_target, None, 'done',
|
||||
queue_id=stale_claim['id'], lease_token=stale_claim['lease_token'],
|
||||
)
|
||||
row = db.target_queue_item('dockerhub', 'docker', stale_target)
|
||||
assert row['status'] == 'in_progress' and row['lease_owner'] == 'owner-b'
|
||||
|
||||
run_id = db.start_run('smoke', ['smoke'])
|
||||
cycle_id = db.start_source_cycle(run_id, 'dockerhub', 'docker', 'search', 'q', 1, 1, None, {}, {})
|
||||
result = {
|
||||
'target': stale_target, 'scan_type': 'docker', 'findings': [], 'errors': [],
|
||||
'scan_event_id': str(uuid.uuid4()),
|
||||
'timestamp': datetime.now(timezone.utc).isoformat(timespec='seconds'),
|
||||
}
|
||||
before = db.conn.execute('SELECT COUNT(*) AS count FROM target_scans').fetchone()['count']
|
||||
stale = db.record_and_complete_target_result(
|
||||
run_id, cycle_id, 'dockerhub', 'q', stale_target, result, {},
|
||||
row['id'], 'owner-a', 'done',
|
||||
lease_token=stale_claim['lease_token'],
|
||||
)
|
||||
assert stale and stale['stale']
|
||||
assert db.conn.execute('SELECT COUNT(*) AS count FROM target_scans').fetchone()['count'] == before + 1
|
||||
row = db.target_queue_item('dockerhub', 'docker', stale_target)
|
||||
assert row['status'] == 'in_progress' and row['lease_token'] == newer_claim['lease_token']
|
||||
result = dict(result, scan_event_id=str(uuid.uuid4()))
|
||||
owned = db.record_and_complete_target_result(
|
||||
run_id, cycle_id, 'dockerhub', 'q', stale_target, result, {},
|
||||
row['id'], 'owner-b', 'done',
|
||||
lease_token=newer_claim['lease_token'],
|
||||
)
|
||||
assert owned and not owned['stale']
|
||||
assert db.conn.execute('SELECT COUNT(*) AS count FROM target_scans').fetchone()['count'] == before + 2
|
||||
|
||||
exhausted_target = 'owner/exhausted:tag'
|
||||
db.enqueue_targets('dockerhub', 'docker', 'q', [exhausted_target])
|
||||
assert db.claim_targets('dockerhub', 'docker', 1, 'owner-x', 60, max_attempts=1) == [exhausted_target]
|
||||
db.conn.execute(
|
||||
"UPDATE target_queue SET lease_expires_at = ? WHERE normalized_target = ?",
|
||||
('2000-01-01T00:00:00+00:00', normalize_target(exhausted_target, 'docker')),
|
||||
)
|
||||
db.conn.commit()
|
||||
assert db.claim_targets('dockerhub', 'docker', 1, 'owner-y', 60, max_attempts=1) == []
|
||||
assert db.target_queue_item('dockerhub', 'docker', exhausted_target)['status'] == 'failed'
|
||||
finally:
|
||||
db.close()
|
||||
finally:
|
||||
for key, value in old_urls.items():
|
||||
if value is not None:
|
||||
os.environ[key] = value
|
||||
|
||||
|
||||
def main():
|
||||
for key in ('SCANNER_DB_URL', 'DATABASE_URL', 'TRUF_MANAGED_POSTGRES_DSN', 'KEYCHECK_DB_URL'):
|
||||
os.environ.pop(key, None)
|
||||
assert_diagnostic_policy()
|
||||
assert_docker_platform_policy()
|
||||
assert_docker_partial_pagination_policy()
|
||||
assert_retry_policy()
|
||||
assert_timeout_output_policy()
|
||||
assert_archive_and_secret_safety()
|
||||
assert_queue_policy()
|
||||
print('scanner error policy smoke: OK')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
+6562
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,378 @@
|
||||
import hmac
|
||||
import ipaddress
|
||||
import os
|
||||
import re
|
||||
import secrets
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from process_identity import current_process_identity, verify_retained_process
|
||||
from lifecycle_authority import (
|
||||
LIFECYCLE_PHASES,
|
||||
PHASE_ACTIVATING,
|
||||
LifecycleAuthorityError,
|
||||
build_code_manifest,
|
||||
code_manifest_sha256,
|
||||
normalize_code_manifest,
|
||||
verify_code_manifest,
|
||||
verify_supervisor_command_line,
|
||||
)
|
||||
from runtime_security import (
|
||||
atomic_write_private_json,
|
||||
canonical_path,
|
||||
durable_unlink,
|
||||
PrivateFileLock,
|
||||
read_private_json,
|
||||
sha256_file,
|
||||
write_private_json_exclusive,
|
||||
)
|
||||
|
||||
|
||||
INSTANCE_SCHEMA = 2
|
||||
CONTROL_SCHEMA = 1
|
||||
SHUTDOWN_RECEIPT_SCHEMA = 1
|
||||
|
||||
|
||||
class InstanceMetadataError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
class InstanceLockError(OSError):
|
||||
pass
|
||||
|
||||
|
||||
def instance_lock_path(instance_path):
|
||||
return os.path.splitext(os.path.abspath(instance_path))[0] + '.lock'
|
||||
|
||||
|
||||
def shutdown_receipt_path(instance_path):
|
||||
return os.path.splitext(os.path.abspath(instance_path))[0] + '.exit.json'
|
||||
|
||||
|
||||
class SupervisorInstanceLock:
|
||||
"""Lifetime singleton lock keyed by the canonical private instance path."""
|
||||
|
||||
def __init__(self, instance_path, lock_path=None):
|
||||
self.instance_path = canonical_path(instance_path)
|
||||
self.path = os.path.normcase(os.path.abspath(lock_path or instance_lock_path(instance_path)))
|
||||
self._lock = PrivateFileLock(self.path)
|
||||
self._acquired = False
|
||||
|
||||
@property
|
||||
def acquired(self):
|
||||
return self._acquired
|
||||
|
||||
def acquire(self):
|
||||
if self._acquired:
|
||||
return self
|
||||
try:
|
||||
self._lock.acquire()
|
||||
except BlockingIOError as exc:
|
||||
raise InstanceLockError('another supervisor owns this private runtime control lock') from exc
|
||||
except OSError as exc:
|
||||
raise InstanceLockError(str(exc)) from exc
|
||||
self._acquired = True
|
||||
return self
|
||||
|
||||
def release(self):
|
||||
if not self._acquired:
|
||||
return
|
||||
self._acquired = False
|
||||
self._lock.release()
|
||||
|
||||
def __enter__(self):
|
||||
return self.acquire()
|
||||
|
||||
def __exit__(self, exc_type, value, traceback):
|
||||
self.release()
|
||||
|
||||
|
||||
def utc_now_iso():
|
||||
return datetime.now(timezone.utc).isoformat(timespec='seconds')
|
||||
|
||||
|
||||
def is_loopback_host(host):
|
||||
try:
|
||||
return ipaddress.ip_address(str(host)).is_loopback
|
||||
except ValueError:
|
||||
return str(host).strip().lower() == 'localhost'
|
||||
|
||||
|
||||
def build_instance_metadata(
|
||||
launch_nonce,
|
||||
supervisor_path,
|
||||
config_path,
|
||||
control_host,
|
||||
control_port,
|
||||
manages_postgres,
|
||||
identity=None,
|
||||
instance_id=None,
|
||||
token=None,
|
||||
activation_state=PHASE_ACTIVATING,
|
||||
expected_config_sha256=None,
|
||||
expected_supervisor_sha256=None,
|
||||
code_manifest=None,
|
||||
expected_code_manifest_sha256=None,
|
||||
canonical_dsn_sha256='',
|
||||
lifecycle_mode='background',
|
||||
instance_file=None,
|
||||
):
|
||||
if not launch_nonce:
|
||||
raise InstanceMetadataError('launch nonce is required')
|
||||
if not is_loopback_host(control_host):
|
||||
raise InstanceMetadataError('control endpoint must be loopback-only')
|
||||
identity = identity or current_process_identity()
|
||||
supervisor_path = canonical_path(supervisor_path)
|
||||
config_path = canonical_path(config_path)
|
||||
actual_supervisor_sha256 = sha256_file(supervisor_path)
|
||||
actual_config_sha256 = sha256_file(config_path)
|
||||
if expected_supervisor_sha256 and not hmac.compare_digest(actual_supervisor_sha256, str(expected_supervisor_sha256)):
|
||||
raise InstanceMetadataError('supervisor script changed after parent authority capture')
|
||||
if expected_config_sha256 and not hmac.compare_digest(actual_config_sha256, str(expected_config_sha256)):
|
||||
raise InstanceMetadataError('supervisor config changed after parent authority capture')
|
||||
code_manifest = normalize_code_manifest(code_manifest or build_code_manifest())
|
||||
manifest_sha256 = code_manifest_sha256(code_manifest)
|
||||
if expected_code_manifest_sha256 and not hmac.compare_digest(manifest_sha256, str(expected_code_manifest_sha256)):
|
||||
raise InstanceMetadataError('code manifest changed after parent authority capture')
|
||||
lifecycle_mode = str(lifecycle_mode)
|
||||
if lifecycle_mode not in ('background', 'foreground'):
|
||||
raise InstanceMetadataError('invalid supervisor lifecycle mode')
|
||||
return {
|
||||
'schema': INSTANCE_SCHEMA,
|
||||
'instance_id': instance_id or secrets.token_urlsafe(24),
|
||||
'token': token or secrets.token_urlsafe(48),
|
||||
'launch_nonce': str(launch_nonce),
|
||||
'pid': int(identity.pid),
|
||||
'process_creation_time': str(identity.creation_time),
|
||||
'executable': canonical_path(identity.executable),
|
||||
'supervisor_path': supervisor_path,
|
||||
'supervisor_sha256': actual_supervisor_sha256,
|
||||
'config_path': config_path,
|
||||
'config_sha256': actual_config_sha256,
|
||||
'code_manifest': code_manifest,
|
||||
'code_manifest_sha256': manifest_sha256,
|
||||
'canonical_dsn_sha256': str(canonical_dsn_sha256 or ''),
|
||||
'instance_file': canonical_path(instance_file) if instance_file else '',
|
||||
'lifecycle_mode': lifecycle_mode,
|
||||
'control': {'host': str(control_host), 'port': int(control_port)},
|
||||
'startup_time': utc_now_iso(),
|
||||
'manages_postgres': bool(manages_postgres),
|
||||
'activation_state': str(activation_state),
|
||||
'private_file_ready': True,
|
||||
}
|
||||
|
||||
|
||||
def validate_instance_metadata(value):
|
||||
if not isinstance(value, dict) or value.get('schema') != INSTANCE_SCHEMA:
|
||||
raise InstanceMetadataError('unsupported supervisor instance schema')
|
||||
required_strings = (
|
||||
'instance_id', 'token', 'launch_nonce', 'process_creation_time', 'executable',
|
||||
'supervisor_path', 'supervisor_sha256', 'config_path', 'config_sha256', 'startup_time',
|
||||
'code_manifest_sha256', 'lifecycle_mode',
|
||||
)
|
||||
for key in required_strings:
|
||||
if not isinstance(value.get(key), str) or not value[key]:
|
||||
raise InstanceMetadataError(f'invalid supervisor instance field: {key}')
|
||||
for key in ('supervisor_sha256', 'config_sha256', 'code_manifest_sha256'):
|
||||
if not re.fullmatch(r'[0-9a-f]{64}', value[key]):
|
||||
raise InstanceMetadataError(f'invalid supervisor instance hash: {key}')
|
||||
canonical_dsn_sha256 = str(value.get('canonical_dsn_sha256') or '')
|
||||
if canonical_dsn_sha256 and not re.fullmatch(r'[0-9a-f]{64}', canonical_dsn_sha256):
|
||||
raise InstanceMetadataError('invalid supervisor instance DSN authority')
|
||||
try:
|
||||
manifest = normalize_code_manifest(value.get('code_manifest'))
|
||||
except (OSError, ValueError) as exc:
|
||||
raise InstanceMetadataError(str(exc)) from exc
|
||||
if not hmac.compare_digest(code_manifest_sha256(manifest), value['code_manifest_sha256']):
|
||||
raise InstanceMetadataError('supervisor instance code manifest digest mismatch')
|
||||
if len(value['instance_id']) > 256 or len(value['token']) < 32 or len(value['token']) > 512:
|
||||
raise InstanceMetadataError('invalid supervisor instance credentials')
|
||||
try:
|
||||
pid = int(value.get('pid'))
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise InstanceMetadataError('invalid supervisor instance PID') from exc
|
||||
if pid <= 0:
|
||||
raise InstanceMetadataError('invalid supervisor instance PID')
|
||||
control = value.get('control')
|
||||
if not isinstance(control, dict) or not is_loopback_host(control.get('host')):
|
||||
raise InstanceMetadataError('invalid supervisor control endpoint')
|
||||
try:
|
||||
port = int(control.get('port'))
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise InstanceMetadataError('invalid supervisor control port') from exc
|
||||
if not 0 < port <= 65535:
|
||||
raise InstanceMetadataError('invalid supervisor control port')
|
||||
if value.get('private_file_ready') is not True:
|
||||
raise InstanceMetadataError('supervisor instance is not marked private-file-ready')
|
||||
activation_state = str(value.get('activation_state') or PHASE_ACTIVATING).upper()
|
||||
if activation_state not in LIFECYCLE_PHASES:
|
||||
raise InstanceMetadataError('invalid supervisor activation state')
|
||||
lifecycle_mode = str(value.get('lifecycle_mode') or '')
|
||||
if lifecycle_mode not in ('background', 'foreground'):
|
||||
raise InstanceMetadataError('invalid supervisor lifecycle mode')
|
||||
normalized = dict(value)
|
||||
normalized['pid'] = pid
|
||||
normalized['control'] = {'host': str(control['host']), 'port': port}
|
||||
normalized['executable'] = canonical_path(value['executable'])
|
||||
normalized['supervisor_path'] = canonical_path(value['supervisor_path'])
|
||||
normalized['config_path'] = canonical_path(value['config_path'])
|
||||
normalized['code_manifest'] = manifest
|
||||
normalized['code_manifest_sha256'] = value['code_manifest_sha256']
|
||||
normalized['canonical_dsn_sha256'] = canonical_dsn_sha256
|
||||
normalized['instance_file'] = canonical_path(value['instance_file']) if value.get('instance_file') else ''
|
||||
normalized['lifecycle_mode'] = lifecycle_mode
|
||||
normalized['manages_postgres'] = bool(value.get('manages_postgres'))
|
||||
normalized['activation_state'] = activation_state
|
||||
return normalized
|
||||
|
||||
|
||||
def write_instance_metadata(path, metadata):
|
||||
write_private_json_exclusive(path, validate_instance_metadata(metadata))
|
||||
|
||||
|
||||
def load_instance_metadata(path):
|
||||
return validate_instance_metadata(read_private_json(path))
|
||||
|
||||
|
||||
def update_instance_activation(path, instance_id, activation_state):
|
||||
activation_state = str(activation_state).upper()
|
||||
if activation_state not in LIFECYCLE_PHASES:
|
||||
raise InstanceMetadataError('invalid supervisor activation state')
|
||||
current = load_instance_metadata(path)
|
||||
if not hmac.compare_digest(current['instance_id'], str(instance_id)):
|
||||
raise InstanceMetadataError('supervisor activation instance mismatch')
|
||||
current['activation_state'] = activation_state
|
||||
atomic_write_private_json(path, validate_instance_metadata(current))
|
||||
return current
|
||||
|
||||
|
||||
def remove_instance_if_matches(path, instance_id, instance_lock=None, lock_path=None):
|
||||
owned_lock = None
|
||||
if instance_lock is None:
|
||||
try:
|
||||
owned_lock = SupervisorInstanceLock(path, lock_path=lock_path).acquire()
|
||||
instance_lock = owned_lock
|
||||
except OSError:
|
||||
return False
|
||||
try:
|
||||
current = load_instance_metadata(path)
|
||||
except (OSError, ValueError):
|
||||
if owned_lock:
|
||||
owned_lock.release()
|
||||
return False
|
||||
if not hmac.compare_digest(current['instance_id'], str(instance_id)):
|
||||
if owned_lock:
|
||||
owned_lock.release()
|
||||
return False
|
||||
try:
|
||||
before = os.stat(path, follow_symlinks=False)
|
||||
confirmed = load_instance_metadata(path)
|
||||
after = os.stat(path, follow_symlinks=False)
|
||||
identity_before = (before.st_dev, before.st_ino, before.st_size, before.st_mtime_ns)
|
||||
identity_after = (after.st_dev, after.st_ino, after.st_size, after.st_mtime_ns)
|
||||
if identity_before != identity_after or not hmac.compare_digest(confirmed['instance_id'], str(instance_id)):
|
||||
return False
|
||||
durable_unlink(path)
|
||||
return True
|
||||
except (OSError, ValueError):
|
||||
return False
|
||||
finally:
|
||||
if owned_lock:
|
||||
owned_lock.release()
|
||||
|
||||
|
||||
def verify_instance_process(
|
||||
metadata,
|
||||
supervisor_path=None,
|
||||
config_path=None,
|
||||
allow_config_drift=False,
|
||||
allow_code_drift=False,
|
||||
):
|
||||
metadata = validate_instance_metadata(metadata)
|
||||
if supervisor_path and metadata['supervisor_path'] != canonical_path(supervisor_path):
|
||||
raise InstanceMetadataError('supervisor path does not match instance metadata')
|
||||
if config_path and metadata['config_path'] != canonical_path(config_path):
|
||||
raise InstanceMetadataError('config path does not match instance metadata')
|
||||
if not allow_code_drift:
|
||||
try:
|
||||
verify_code_manifest(metadata['code_manifest'], metadata['code_manifest_sha256'])
|
||||
if sha256_file(metadata['supervisor_path']) != metadata['supervisor_sha256']:
|
||||
raise InstanceMetadataError('supervisor script hash does not match instance metadata')
|
||||
except OSError as exc:
|
||||
raise InstanceMetadataError(f'unable to recompute supervisor code authority: {exc}') from exc
|
||||
except ValueError as exc:
|
||||
raise InstanceMetadataError(str(exc)) from exc
|
||||
try:
|
||||
config_matches = sha256_file(metadata['config_path']) == metadata['config_sha256']
|
||||
except OSError as exc:
|
||||
if not allow_config_drift:
|
||||
raise InstanceMetadataError(f'unable to recompute supervisor config authority hash: {exc}') from exc
|
||||
config_matches = False
|
||||
if not config_matches and not allow_config_drift:
|
||||
raise InstanceMetadataError('supervisor config hash drifted; only authenticated shutdown is allowed')
|
||||
process = verify_retained_process(
|
||||
metadata['pid'],
|
||||
metadata['process_creation_time'],
|
||||
metadata['executable'],
|
||||
)
|
||||
try:
|
||||
arguments = process.command_line()
|
||||
if metadata['lifecycle_mode'] == 'background' and '--background-child' not in arguments:
|
||||
raise InstanceMetadataError('retained Python process is not a background supervisor child')
|
||||
if metadata['lifecycle_mode'] == 'foreground' and '--background-child' in arguments:
|
||||
raise InstanceMetadataError('retained Python process lifecycle mode mismatch')
|
||||
try:
|
||||
verify_supervisor_command_line(
|
||||
arguments,
|
||||
metadata['supervisor_path'],
|
||||
metadata['config_path'],
|
||||
)
|
||||
except LifecycleAuthorityError as exc:
|
||||
raise InstanceMetadataError(str(exc)) from exc
|
||||
except BaseException:
|
||||
process.close()
|
||||
raise
|
||||
return process
|
||||
|
||||
|
||||
def write_shutdown_receipt(instance_path, instance_id, exit_code):
|
||||
value = {
|
||||
'schema': SHUTDOWN_RECEIPT_SCHEMA,
|
||||
'instance_id': str(instance_id),
|
||||
'exit_code': int(exit_code),
|
||||
'completed_at': utc_now_iso(),
|
||||
}
|
||||
atomic_write_private_json(shutdown_receipt_path(instance_path), value)
|
||||
|
||||
|
||||
def load_shutdown_receipt(instance_path, instance_id):
|
||||
value = read_private_json(shutdown_receipt_path(instance_path))
|
||||
if value.get('schema') != SHUTDOWN_RECEIPT_SCHEMA:
|
||||
raise InstanceMetadataError('unsupported supervisor shutdown receipt schema')
|
||||
if not hmac.compare_digest(str(value.get('instance_id') or ''), str(instance_id)):
|
||||
raise InstanceMetadataError('supervisor shutdown receipt instance mismatch')
|
||||
try:
|
||||
code = int(value.get('exit_code'))
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise InstanceMetadataError('invalid supervisor shutdown receipt exit code') from exc
|
||||
return code
|
||||
|
||||
|
||||
def remove_shutdown_receipt(instance_path, instance_id=None):
|
||||
path = shutdown_receipt_path(instance_path)
|
||||
try:
|
||||
if instance_id is not None:
|
||||
load_shutdown_receipt(instance_path, instance_id)
|
||||
durable_unlink(path)
|
||||
return True
|
||||
except (OSError, ValueError):
|
||||
return False
|
||||
|
||||
|
||||
def authenticate_request(request, instance_id, token):
|
||||
if not isinstance(request, dict) or request.get('schema') != CONTROL_SCHEMA:
|
||||
return False
|
||||
request_instance = request.get('instance_id')
|
||||
request_token = request.get('token')
|
||||
if not isinstance(request_instance, str) or not isinstance(request_token, str):
|
||||
return False
|
||||
return hmac.compare_digest(request_instance, str(instance_id)) and hmac.compare_digest(request_token, str(token))
|
||||
@@ -0,0 +1,356 @@
|
||||
import sys
|
||||
|
||||
sys.dont_write_bytecode = True
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
import threading
|
||||
|
||||
import yaml
|
||||
|
||||
from migrate_runtime_safety import require_runtime_hardening_stopped
|
||||
from db_backend import database_url_from_env, is_postgres_url
|
||||
from paths import apply_path_config
|
||||
from postgres_runtime import load_postgres_environment
|
||||
from runtime_security import (
|
||||
ClusterAuthorityLock,
|
||||
canonical_path,
|
||||
durable_replace,
|
||||
harden_private_file,
|
||||
PrivateFileLock,
|
||||
private_file_ready,
|
||||
require_private_directory,
|
||||
require_private_file,
|
||||
)
|
||||
|
||||
|
||||
GITHUB_TOKEN_PREFIXES = ('ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_')
|
||||
ACCEPTED_ALIVE_STATUSES = {'ALIVE', 'VALID', 'VALID_2FA'}
|
||||
DOCKERHUB_TOKEN_RE = re.compile(r'^dckr_pat_[A-Za-z0-9_-]{27}$')
|
||||
PROVIDER_DEFAULTS = {
|
||||
'github': {
|
||||
'alive_file': os.path.join('runtime', 'keychecks', 'github', 'githubAlive.txt'),
|
||||
'pool': 'github_main',
|
||||
'name_prefix': 'gh',
|
||||
},
|
||||
'dockerhub': {
|
||||
'alive_file': os.path.join('runtime', 'keychecks', 'dockerhub', 'dockerhubAlive.txt'),
|
||||
'pool': 'dockerhub_main',
|
||||
'name_prefix': 'dockerhub',
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def read_alive_tokens(path):
|
||||
tokens = []
|
||||
seen = set()
|
||||
with open(path, 'r', encoding='utf-8', errors='replace') as f:
|
||||
for line in f:
|
||||
# Status files are TSV-like: token, status, message, extra.
|
||||
fields = line.rstrip('\r\n').split('\t')
|
||||
token = fields[0].strip() if fields else ''
|
||||
if not token or not token.startswith(GITHUB_TOKEN_PREFIXES):
|
||||
continue
|
||||
if any(character.isspace() for character in token):
|
||||
raise ValueError('alive token input contains whitespace in a token field')
|
||||
status = fields[1].strip().upper() if len(fields) > 1 else ''
|
||||
if status not in ACCEPTED_ALIVE_STATUSES:
|
||||
raise ValueError(f'alive token input contains an unaccepted or missing status: {status or "(missing)"}')
|
||||
if token in seen:
|
||||
continue
|
||||
seen.add(token)
|
||||
tokens.append(token)
|
||||
return tokens
|
||||
|
||||
|
||||
def read_alive_credentials(path, provider):
|
||||
provider = str(provider or 'github').strip().lower()
|
||||
if provider == 'github':
|
||||
return [{'token': token} for token in read_alive_tokens(path)], 0
|
||||
if provider != 'dockerhub':
|
||||
raise ValueError(f'unsupported alive credential provider: {provider}')
|
||||
|
||||
credentials = []
|
||||
seen = {}
|
||||
skipped_missing_username = 0
|
||||
with open(path, 'r', encoding='utf-8', errors='replace') as handle:
|
||||
for line in handle:
|
||||
fields = line.rstrip('\r\n').split('\t')
|
||||
identity = fields[0].strip() if fields else ''
|
||||
if not identity:
|
||||
continue
|
||||
if ':' in identity:
|
||||
username, token = identity.rsplit(':', 1)
|
||||
username = username.strip()
|
||||
token = token.strip()
|
||||
else:
|
||||
username = ''
|
||||
token = identity
|
||||
if not DOCKERHUB_TOKEN_RE.fullmatch(token):
|
||||
continue
|
||||
status = fields[1].strip().upper() if len(fields) > 1 else ''
|
||||
if status not in {'VALID', 'VALID_2FA'}:
|
||||
raise ValueError(f'alive DockerHub input contains an unaccepted or missing status: {status or "(missing)"}')
|
||||
if not username:
|
||||
skipped_missing_username += 1
|
||||
continue
|
||||
if len(username) > 256 or ':' in username or any(character.isspace() for character in username):
|
||||
raise ValueError('alive DockerHub input contains an invalid username field')
|
||||
previous = seen.get(token)
|
||||
if previous is not None:
|
||||
if previous.casefold() != username.casefold():
|
||||
raise ValueError('alive DockerHub input contains conflicting usernames for one token')
|
||||
continue
|
||||
seen[token] = username
|
||||
credentials.append({'username': username, 'token': token})
|
||||
return credentials, skipped_missing_username
|
||||
|
||||
|
||||
def next_name(existing_names, prefix):
|
||||
pattern = re.compile(rf'^{re.escape(prefix)}_(\d+)$')
|
||||
max_index = 0
|
||||
for name in existing_names:
|
||||
match = pattern.match(str(name or ''))
|
||||
if match:
|
||||
max_index = max(max_index, int(match.group(1)))
|
||||
return f'{prefix}_{max_index + 1}'
|
||||
|
||||
|
||||
def _atomic_write_private_yaml(path, value):
|
||||
parent = require_private_directory(os.path.dirname(os.path.abspath(path)), create=False)
|
||||
payload = yaml.safe_dump(value, allow_unicode=True, sort_keys=False, width=120).encode('utf-8')
|
||||
temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.tmp'
|
||||
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0)
|
||||
descriptor = os.open(temporary, flags, 0o600)
|
||||
try:
|
||||
os.close(descriptor)
|
||||
descriptor = None
|
||||
harden_private_file(temporary)
|
||||
with open(temporary, 'wb') as handle:
|
||||
handle.write(payload)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
if not private_file_ready(temporary):
|
||||
raise OSError(f'private temporary secrets ACL changed: {temporary}')
|
||||
durable_replace(temporary, path)
|
||||
if not private_file_ready(path):
|
||||
raise OSError(f'private secrets ACL changed during publication: {path}')
|
||||
finally:
|
||||
if descriptor is not None:
|
||||
os.close(descriptor)
|
||||
try:
|
||||
if os.path.exists(temporary):
|
||||
os.remove(temporary)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def sync_tokens(
|
||||
secrets_path,
|
||||
alive_path,
|
||||
pool_name,
|
||||
name_prefix,
|
||||
apply=False,
|
||||
*,
|
||||
provider='github',
|
||||
replace_conflicting_usernames=False,
|
||||
canonical_secrets_path=None,
|
||||
authority_lock=None,
|
||||
stopped_verified=False,
|
||||
):
|
||||
if authority_lock is None or not getattr(authority_lock, 'acquired', False):
|
||||
raise RuntimeError('alive-token sync requires an acquired cluster authority lock')
|
||||
if stopped_verified is not True:
|
||||
raise RuntimeError('alive-token sync requires verified stopped runtime proof')
|
||||
if not canonical_secrets_path or canonical_path(secrets_path) != canonical_path(canonical_secrets_path):
|
||||
raise RuntimeError('alive-token sync secrets path must exactly match canonical global.secrets_file')
|
||||
if not os.path.exists(alive_path):
|
||||
raise SystemExit(f'alive token file not found: {alive_path}')
|
||||
if not os.path.exists(secrets_path):
|
||||
raise SystemExit(f'secrets file not found: {secrets_path}')
|
||||
|
||||
require_private_file(secrets_path)
|
||||
require_private_file(alive_path)
|
||||
lock = PrivateFileLock(f'{secrets_path}.sync.lock').acquire()
|
||||
try:
|
||||
require_private_file(secrets_path)
|
||||
require_private_file(alive_path)
|
||||
with open(secrets_path, 'r', encoding='utf-8') as f:
|
||||
secrets = yaml.safe_load(f) or {}
|
||||
|
||||
auth_pools = secrets.setdefault('auth_pools', {})
|
||||
pool = auth_pools.setdefault(pool_name, [])
|
||||
if not isinstance(pool, list):
|
||||
raise SystemExit(f'auth_pools.{pool_name} must be a list')
|
||||
existing_before = len(pool)
|
||||
|
||||
provider = str(provider or 'github').strip().lower()
|
||||
if provider not in PROVIDER_DEFAULTS:
|
||||
raise ValueError(f'unsupported alive credential provider: {provider}')
|
||||
existing_tokens = set()
|
||||
existing_entries = {}
|
||||
existing_names = set()
|
||||
normalized_existing = 0
|
||||
normalized_usernames = 0
|
||||
for entry in pool:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
if entry.get('name'):
|
||||
existing_names.add(str(entry.get('name')))
|
||||
token = str(entry.get('token') or '')
|
||||
stripped = token.strip()
|
||||
if stripped != token:
|
||||
normalized_existing += 1
|
||||
if apply:
|
||||
entry['token'] = stripped
|
||||
if stripped:
|
||||
existing_tokens.add(stripped)
|
||||
existing_entries.setdefault(stripped, []).append(entry)
|
||||
if provider == 'dockerhub':
|
||||
username = str(entry.get('username') or '')
|
||||
stripped_username = username.strip()
|
||||
if stripped_username != username:
|
||||
normalized_usernames += 1
|
||||
if apply:
|
||||
entry['username'] = stripped_username
|
||||
|
||||
alive_credentials, skipped_missing_username = read_alive_credentials(alive_path, provider)
|
||||
added = []
|
||||
username_filled = 0
|
||||
username_conflicts = 0
|
||||
username_replaced = 0
|
||||
for credential in alive_credentials:
|
||||
token = credential['token']
|
||||
if token in existing_tokens:
|
||||
if provider == 'dockerhub':
|
||||
entries = existing_entries.get(token, [])
|
||||
usernames = {
|
||||
str(entry.get('username') or '').strip().casefold()
|
||||
for entry in entries if str(entry.get('username') or '').strip()
|
||||
}
|
||||
expected = credential['username'].casefold()
|
||||
if len(usernames) > 1 or (usernames and expected not in usernames):
|
||||
if replace_conflicting_usernames:
|
||||
username_replaced += 1
|
||||
if apply:
|
||||
for entry in entries:
|
||||
entry['username'] = credential['username']
|
||||
else:
|
||||
username_conflicts += 1
|
||||
continue
|
||||
if not usernames:
|
||||
username_filled += 1
|
||||
if apply:
|
||||
for entry in entries:
|
||||
entry['username'] = credential['username']
|
||||
continue
|
||||
name = next_name(existing_names, name_prefix)
|
||||
existing_names.add(name)
|
||||
existing_tokens.add(token)
|
||||
new_entry = {'name': name}
|
||||
if provider == 'dockerhub':
|
||||
new_entry['username'] = credential['username']
|
||||
new_entry['token'] = token
|
||||
added.append(new_entry)
|
||||
|
||||
if apply and (
|
||||
added or normalized_existing or normalized_usernames
|
||||
or username_filled or username_replaced
|
||||
):
|
||||
pool.extend(added)
|
||||
_atomic_write_private_yaml(secrets_path, secrets)
|
||||
|
||||
return {
|
||||
'alive_unique': len(alive_credentials),
|
||||
'existing_before': existing_before,
|
||||
'added': len(added),
|
||||
'normalized_existing': normalized_existing,
|
||||
'normalized_usernames': normalized_usernames,
|
||||
'username_filled': username_filled,
|
||||
'username_conflicts': username_conflicts,
|
||||
'username_replaced': username_replaced,
|
||||
'skipped_missing_username': skipped_missing_username,
|
||||
'pool_after': len(pool) + (0 if apply else len(added)),
|
||||
}
|
||||
finally:
|
||||
lock.release()
|
||||
|
||||
|
||||
def load_config(path):
|
||||
with open(path, 'r', encoding='utf-8') as handle:
|
||||
return apply_path_config(yaml.safe_load(handle) or {}, path)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description='Sync alive provider credentials into a private auth pool without printing them.')
|
||||
parser.add_argument('--provider', choices=sorted(PROVIDER_DEFAULTS), default='github')
|
||||
parser.add_argument('--secrets', help='Must exactly match global.secrets_file from --config')
|
||||
parser.add_argument('--alive-file')
|
||||
parser.add_argument('--pool')
|
||||
parser.add_argument('--name-prefix')
|
||||
parser.add_argument('--config', default=os.path.join('app', 'config.yaml'))
|
||||
parser.add_argument('--dry-run', action='store_true')
|
||||
parser.add_argument('--apply', action='store_true', help='Apply under verified offline maintenance authority')
|
||||
parser.add_argument(
|
||||
'--replace-conflicting-usernames', action='store_true',
|
||||
help='Replace an existing DockerHub username only when the same token has an authoritative alive pair',
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
args = parse_args()
|
||||
if args.apply and args.dry_run:
|
||||
raise SystemExit('--apply and --dry-run are mutually exclusive')
|
||||
config_path = canonical_path(args.config)
|
||||
require_private_file(config_path)
|
||||
config = load_config(config_path)
|
||||
configured_value = (config.get('global') or {}).get('secrets_file')
|
||||
if not configured_value:
|
||||
raise SystemExit('global.secrets_file is required')
|
||||
configured_secrets = canonical_path(configured_value)
|
||||
requested_secrets = canonical_path(args.secrets) if args.secrets else configured_secrets
|
||||
if requested_secrets != configured_secrets:
|
||||
raise SystemExit('--secrets must exactly match canonical global.secrets_file')
|
||||
load_postgres_environment(config_path, config)
|
||||
endpoint_dsn = database_url_from_env() or (config.get('global') or {}).get('database_url')
|
||||
if not is_postgres_url(endpoint_dsn):
|
||||
raise SystemExit('A caller-selected canonical PostgreSQL DSN is required for maintenance authority')
|
||||
provider = str(getattr(args, 'provider', 'github') or 'github').strip().lower()
|
||||
defaults = PROVIDER_DEFAULTS.get(provider)
|
||||
if defaults is None:
|
||||
raise SystemExit(f'unsupported provider: {provider}')
|
||||
alive_file = args.alive_file or defaults['alive_file']
|
||||
pool_name = args.pool or defaults['pool']
|
||||
name_prefix = args.name_prefix or defaults['name_prefix']
|
||||
with ClusterAuthorityLock(config, endpoint_dsn=endpoint_dsn) as authority_lock:
|
||||
require_runtime_hardening_stopped(config)
|
||||
result = sync_tokens(
|
||||
configured_secrets,
|
||||
alive_file,
|
||||
pool_name,
|
||||
name_prefix,
|
||||
apply=args.apply,
|
||||
provider=provider,
|
||||
replace_conflicting_usernames=bool(getattr(args, 'replace_conflicting_usernames', False)),
|
||||
canonical_secrets_path=configured_secrets,
|
||||
authority_lock=authority_lock,
|
||||
stopped_verified=True,
|
||||
)
|
||||
mode = 'updated' if args.apply else 'dry_run'
|
||||
print(
|
||||
f"{mode}: provider={provider} pool={pool_name} alive_unique={result['alive_unique']} "
|
||||
f"existing_before={result['existing_before']} added={result['added']} "
|
||||
f"username_filled={result.get('username_filled', 0)} "
|
||||
f"username_conflicts={result.get('username_conflicts', 0)} "
|
||||
f"username_replaced={result.get('username_replaced', 0)} "
|
||||
f"skipped_missing_username={result.get('skipped_missing_username', 0)} "
|
||||
f"normalized_existing={result['normalized_existing']} "
|
||||
f"normalized_usernames={result.get('normalized_usernames', 0)} pool_after={result['pool_after']}"
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user