Initial server source import

This commit is contained in:
sashatrask
2026-09-30 20:30:56 +03:00
commit 170dd941b9
498 changed files with 261563 additions and 0 deletions
+237
View File
@@ -0,0 +1,237 @@
# Deny by default. File-only exceptions keep credentials and caches out of the context.
**
!app/app.py
!app/audit_github_tokens.py
!app/capacity_model.py
!app/child_bootstrap.py
!app/console_runner.py
!app/container_import.py
!app/container_import_config.py
!app/container_projection_recovery.py
!app/container_runtime.py
!app/dashboard.py
!app/db_backend.py
!app/admin_api.py
!app/docker_depth_experiment.py
!app/docker_depth_operator.py
!app/docker_depth_report.py
!app/docker_shadow.py
!app/host_agent_client.py
!app/host_agent_apply.py
!app/host_agent_lifecycle.py
!app/host_agent_protocol.py
!app/host_agent_reconcile.py
!app/host_agent_runtime.py
!app/host_agent_server.py
!app/host_agent_state.py
!app/janitor.py
!app/jsonl_projector.py
!app/keycheck_accounting_smoke.py
!app/keycheck_candidates.py
!app/keycheck_runner.py
!app/keycheckers/__init__.py
!app/keycheckers/keycheck_common.py
!app/keycheckers/provider_resolution.py
!app/keycheckers/anthropic/anthropicKeycheck.py
!app/keycheckers/aws/awsKeycheck.py
!app/keycheckers/azure/azureKeycheck.py
!app/keycheckers/deepseek/deepseekKeycheck.py
!app/keycheckers/dockerhub/dockerhubKeycheck.py
!app/keycheckers/gcp/gcpKeycheck.py
!app/keycheckers/gemini/geminiKeycheck.py
!app/keycheckers/github/githubKeycheck.py
!app/keycheckers/gitlab/gitlabKeycheck.py
!app/keycheckers/groq/groqKeycheck.py
!app/keycheckers/huggingface/huggingfaceKeycheck.py
!app/keycheckers/kimi/kimiKeycheck.py
!app/keycheckers/openai/Keycheck.py
!app/keycheckers/openrouter/OpenrouterKeycheck.py
!app/keycheckers/provider_resolver/providerResolverKeycheck.py
!app/keycheckers/qwen/qwenKeycheck.py
!app/keycheckers/replicate/replicateKeycheck.py
!app/keycheckers/xai/xaiKeycheck.py
!app/keycheckers/zai/zaiKeycheck.py
!app/lifecycle_authority.py
!app/managed_files.py
!app/migrate_layout.py
!app/migrate_observability_db.py
!app/migrate_runtime_safety.py
!app/optimize_dashboard_db.py
!app/owned_process.py
!app/paths.py
!app/postgres_runtime.py
!app/process_identity.py
!app/query_policy.py
!app/result_bundle.py
!app/result_ingester.py
!app/result_spool.py
!app/runtime_bootstrap.py
!app/runtime_document.py
!app/runtime_document_io.py
!app/runtime_security.py
!app/scan_manager.py
!app/scanner.py
!app/scan_execution.py
!app/scanner_db.py
!app/scanner_error_policy_smoke.py
!app/supervisor.py
!app/supervisor_instance.py
!app/sync_alive_github_tokens.py
!app/target_identity.py
!app/ui_components.py
!app/worker_api.py
!app/worker_assignment.py
!app/worker_assignment_runner.py
!app/worker_cli.py
!app/worker_contracts.py
!app/worker_local_state.py
!app/worker_package.py
!app/worker_package_builder.py
!app/worker_supervisor.py
!app/remote_worker_bootstrap.py
!app/remote_worker_client.py
!app/requirements.txt
!app/requirements-keycheckers.txt
!app/config.linux.yaml
!app/trufflehog-custom-detectors.yaml
!app/.streamlit/config.toml
!start_runtime.ps1
!start_core_runtime.ps1
!stop_runtime.ps1
!Dockerfile
!.dockerignore
!compose.yaml
!compose.edge.yaml
!compose.shared-host.yaml
!docs/remote-worker-quickstart-ru.md
!docs/remote-worker-cheatsheet-windows-ru.md
!docs/remote-worker-cheatsheet-linux-ru.md
!docs/remote-worker-cheatsheet-docker-ru.md
!deploy/edge/Dockerfile
!deploy/edge/Caddyfile
!deploy/edge/Caddyfile.shared-host
!deploy/edge/host-caddy-shared.caddy
!deploy/edge/entrypoint.sh
!deploy/edge/README.md
!deploy/edge/admin-denylist.caddy
!deploy/edge/automatic-tls.caddy
!deploy/fail2ban/truf_caddy_admin_denylist.py
!deploy/fail2ban/Dockerfile.edge-e2e
!deploy/fail2ban/edge_e2e_docker_shim.py
!deploy/fail2ban/fail2ban.d-edge-e2e.local
!deploy/fail2ban/filter.d-truf-admin-auth.conf
!deploy/fail2ban/jail.d-truf-admin-auth.local
!deploy/fail2ban/action.d-truf-caddy-admin-denylist.conf
!deploy/fail2ban/fail2ban.d-truf-persistence.local
!deploy/host-agent/truf_host_agent.py
!deploy/host-agent/truf_host_agent_install.py
!deploy/host-agent/truf-host-agent.conf
!deploy/host-agent/truf-host-agent.service
!deploy/host-agent/truf-host-agent.socket
!deploy/systemd/truf-caddy-admin-denylist-expire.service
!deploy/systemd/truf-caddy-admin-denylist-expire.timer
!docker/requirements.in
!docker/requirements.lock
!docker/Dockerfile.edge-e2e
!docker/requirements-test.in
!docker/requirements-test.lock
!docker/requirements-worker.in
!docker/requirements-worker.lock
!docker/worker-package-pins.json
!docker/verify.py
!docker/build-dependencies/requirements.in
!docker/build-dependencies/requirements.lock
!tests/container_unit.py
!tests/container_e2e.py
!tests/container_import_stop_e2e.py
!tests/container_projection_recovery_e2e.py
!tests/owned_process_helper.py
!tests/parity_helpers.py
!tests/packaged_worker_e2e_server.py
!tests/edge_e2e_backend.py
!tests/edge_e2e_client.py
!tests/test_synthetic_llm_pipeline.py
!tests/test_docker_foundation.py
!tests/test_db_backend_safety.py
!tests/test_owned_process.py
!tests/test_owned_process_linux.py
!tests/test_owned_process_boundary.py
!tests/test_runtime_bootstrap_authority.py
!tests/test_runtime_document.py
!tests/test_runtime_document_io.py
!tests/test_managed_files.py
!tests/test_operations_schema.py
!tests/test_operations_control.py
!tests/test_host_agent_protocol.py
!tests/test_host_agent_linux.py
!tests/test_host_agent_apply.py
!tests/test_host_agent_deploy.py
!tests/test_host_agent_lifecycle.py
!tests/test_host_agent_reconcile.py
!tests/test_host_agent_runtime.py
!tests/test_host_agent_state.py
!tests/test_supervisor_foreground_shutdown.py
!tests/test_supervisor_startup_rollback.py
!tests/test_supervisor_managed_postgres_gate.py
!tests/test_observer_only_coordinated_shutdown.py
!tests/test_supervisor_safety.py
!tests/test_discovery_producer_supervisor.py
!tests/test_distributed_core_profile.py
!tests/test_discovery_only_cycle.py
!tests/test_discovery_request_budgets.py
!tests/test_operations_service.py
!tests/test_postgres_runtime.py
!tests/test_container_security.py
!tests/test_runtime_security.py
!tests/test_postgres_empty_initialization.py
!tests/test_container_migration_paths.py
!tests/test_container_provider_portability.py
!tests/test_container_e2e_helpers.py
!tests/test_container_runtime.py
!tests/test_container_import.py
!tests/test_container_import_config.py
!tests/test_container_projection_recovery.py
!tests/test_result_bundle_v2.py
!tests/test_pipeline_cutover_invariants.py
!tests/test_custom_provider_detector_compatibility.py
!tests/test_scan_execution.py
!tests/test_worker_api.py
!tests/test_worker_api_runtime.py
!tests/test_worker_assignment.py
!tests/test_worker_cli.py
!tests/test_worker_contracts.py
!tests/test_worker_local_state.py
!tests/test_worker_package.py
!tests/test_worker_supervisor.py
!tests/test_multisource_execution_snapshot.py
!tests/test_remote_direct_credentials.py
!tests/test_docker_staging_bounds.py
!tests/test_remote_worker_db.py
!tests/test_admin_api.py
!tests/test_edge_deployment.py
!tests/fixtures/worker_tls_cert.pem
!tests/fixtures/worker_tls_key.pem
# Defense in depth if the allowlist is expanded later.
**/.env*
**/secrets*
**/credentials*
**/*service-account*
**/__pycache__/
**/.venv/
**/venv/
**/node_modules/
**/*.py[cod]
**/*.db*
**/*.sqlite*
**/*.jsonl*
**/*.log*
.git/
.opencode/
runtime/
tmp/
state/
logs/
data/
docker/imports/
truf-cluster-authority-*/
+14
View File
@@ -0,0 +1,14 @@
TRUF_POSTGRES_DB=truf
TRUF_POSTGRES_USER=truf
TRUF_POSTGRES_PASSWORD=change-me-long-random-password
TRUF_POSTGRES_PORT=5432
# Optional tuning overrides.
TRUF_POSTGRES_MAX_CONNECTIONS=100
TRUF_POSTGRES_SHARED_BUFFERS=512MB
TRUF_POSTGRES_EFFECTIVE_CACHE_SIZE=2GB
TRUF_POSTGRES_CHECKPOINT_TIMEOUT=15min
TRUF_POSTGRES_LOG_MIN_DURATION_STATEMENT=2000
# App DSN template. Keep the real password out of config snapshots/logs.
SCANNER_DB_URL=postgresql://truf:change-me-long-random-password@127.0.0.1:5432/truf
+12
View File
@@ -0,0 +1,12 @@
* text=auto
*.py text eol=lf
*.sh text eol=lf
*.yaml text eol=lf
*.yml text eol=lf
*.toml text eol=lf
*.md text eol=lf
*.txt text eol=lf
Dockerfile text eol=lf
.dockerignore text eol=lf
.gitignore text eol=lf
.gitattributes text eol=lf
+44
View File
@@ -0,0 +1,44 @@
# Local credentials and provider input pools.
.env*
!.env*.example
secrets.yaml*
secrets.*.yaml
!secrets.example.yaml
credentials*.json
*-service-account*.json
/app/keycheckers/**/*.txt
# Runtime data, findings, queues, logs, and cluster authority.
/runtime/
/tmp/
/state/
/logs/
/docker/test-results/
/docker/imports/
/data/
/truf-cluster-authority-*/
/checked_*.txt
/todo_*.txt
/runner_state.json
*.db
*.db-*
*.sqlite
*.sqlite-*
*.sqlite3
*.sqlite3-*
*.jsonl
*.jsonl.*
*.log
*.log.*
# Reproducible dependencies and generated files.
__pycache__/
*.py[cod]
.pytest_cache/
.venv/
venv/
node_modules/
build/
dist/
.coverage
htmlcov/
+15
View File
@@ -0,0 +1,15 @@
# Engineering Decisions
## Provider Source Execution
- Keep source adapters minimal. The server validates assignment shape, canonical target identity, and the immutable identity required by the protocol.
- Git planning may bind an exact commit. Docker planning may resolve a mutable tag to an immutable digest. These are identity operations, not provider-access proofs.
- The worker is the final authority for real provider access. It reports success, a permanent target failure, or a retryable provider failure; the server settles or retries from that result.
- Discovery credentials are not assignment fields unless a separately approved capability explicitly defines credential delivery.
- Do not add per-target server preflight requests, durable public-access proofs, proof TTL/freshness columns, access-evidence migrations, broad child-environment credential scrubbing, credential sandboxes, or post-hoc redaction pipelines by default.
- Before adding any such defensive or security-specific mechanism, stop and obtain explicit user approval. Record the approved behavior in an OpenSpec requirement and task before implementation.
- Do not treat existing defensive code as precedent for duplicating the same machinery for another source.
- The worker machine's ambient environment belongs to its operator. Assignment code must not silently rewrite HOME, XDG, Git, Docker, or provider environments merely to enforce a nominally tokenless assignment.
- Prefer direct, bounded provider-error classification over preventive infrastructure: authentication/access/not-found failures are permanent when target-scoped; rate limits, network failures, and provider 5xx responses are retryable.
These rules apply to future source adapters and to changes in `worker_assignment.py`, `scan_execution.py`, `remote_worker_client.py`, `scanner.py`, `scanner_db.py`, and discovery producers.
+418
View File
@@ -0,0 +1,418 @@
# Docker Development Copy
Status: runnable Linux container deployment with a passed fresh offline E2E.
Production source/provider parity and original-data migration are not proven.
Original Windows installation: `D:\truf`. Development copy: `D:\truf-docker`.
Only the development copy is changed; the original `D:\truf` and storage on `S:`
remain untouched. No original STOP or START was performed for this migration.
Docker/Compose installation in WSL was user-approved and has occurred. Earlier
statements about unavailable Docker, no installations, and no images describe
historical stages, not the current deployment.
The current acceptance evidence is from the isolated Docker runs on 2026-09-15.
The historical original-runtime lineage proof was not rerun.
## Copy And Local History
These are historical source-copy/baseline records, not a fresh Git inventory.
- The user changed the initial full-snapshot request to a source-only copy.
- The retained source selection was 295 files, approximately 6.82 MiB before Git and migration edits.
- Application Python, tests, configuration, documentation, OpenSpec artifacts, and project scripts were retained.
- Databases, WAL/SHM files, results, queues, logs, runtime state, provider input pools, real secrets, native Windows bundles, and dependency caches were omitted or removed from the copy.
- The partial external-data copy `D:\truf-docker-data` was removed. Original external storage on `S:` was not modified.
- SHA-256 comparison verified the selected source files before migration edits. The clone's `.gitignore` was strengthened before staging.
- Initial local commit: `1b3c7fc`, `chore: snapshot source for Docker migration`, 292 tracked files on `main`.
- Three copied OpenCode package metadata files remain ignored by their original nested ignore policy.
- No remote, push, Git configuration change, or second commit was made. Migration edits remain separate from the baseline.
## Safety Boundary
Copied root PowerShell tools and direct canonical Python runtime/control CLI
launches retain their staging refusals. The supported container path is the
image's `tini -> python3 -u -I -S -B app/container_runtime.py` entrypoint, which
uses authenticated bootstrap dispatch rather than removing host safeguards.
Do not use copied Windows maintenance/import scripts to operate this deployment.
The original `app/config.yaml` is preserved in the initial Git commit and in the
unchanged original installation. In this working tree it was renamed to
`app/config.linux.yaml`. The initial slice changed filesystem/deployment paths
only; subsequent container contracts fix private storage and control paths.
There is no automatically selected `app/config.yaml`; the container entrypoint
defaults explicitly to `/opt/truf/app/config.linux.yaml`. This production profile
is distinct from the verifier's deliberately narrowed `/data/config/e2e.yaml`.
The image requires read-only application storage, UID/GID 10001 for runtime
commands, a private native Linux named volume at `/data`, and private tmpfs at
`/run/truf`. Provisioning alone runs as root, with only CHOWN, DAC_OVERRIDE, and
FOWNER added to the dropped capability set. Private application/data directories
are mode 0700 and files 0600. System executables remain root-owned and immutable;
they must not be chowned to the application user to satisfy private-file checks.
PostgreSQL 16 runs under the supervisor in the same container, with generated
private credentials, loopback connectivity, and exact cluster identity binding.
External PostgreSQL authority is not implemented; changing a DSN or starting a
separate PostgreSQL service does not implement that backend. Never reuse physical
Windows PGDATA on Linux. Any original-data migration requires separately
approved logical export/import and validation. No host data bind mount, Docker
socket, privileged mode, or Docker-in-Docker is required for DockerHub scanning.
## Implemented Foundation
- Path defaults derive from this checkout instead of `D:\truf` or the current working directory.
- Generated filesystem templates use portable separators. Explicit YAML still overrides environment defaults.
- POSIX path resolution rejects Windows drive, UNC, device, and backslash syntax rather than silently joining it to a Linux directory.
- Generic path defaults retain the runtime-relative result-bundle directory and PostgreSQL's `runtime/postgres/data` suffix; the container profile explicitly selects the separate `/data` paths listed below.
- The Windows TruffleHog fallback is checked only on Windows. Managed PostgreSQL DSN precedence is unchanged.
- `.gitignore` excludes credentials and consumables. `.dockerignore` denies everything except reviewed, explicitly named build/source files, including nested provider modules.
- `.gitattributes` specifies LF for Linux-facing source/configuration files without changing global Git settings.
- Offline tests cover path behavior, the Linux profile, context allowlist, and refusal of copied control entrypoints.
## Implemented Lifecycle Changes
The lifecycle changes in `app/owned_process.py` and `app/supervisor.py` implement
Linux ownership and coordinated foreground shutdown. Native Linux ownership
tests and the fresh container E2E now pass within their selected scope; this is
not acceptance of every production source/provider or unsafe failure scenario.
- Linux startup now reports complete kernel-derived identities, verifies the host and payload sessions, and requires a verified child subreaper. Unreadable or already-exited executables fail startup rather than receiving a PID-only identity.
- Nested observers have independent sessions. Cleanup signals the still-pinned payload group before reaping its leader, then drains adopted children. Completed adopted observers are also reaped while the root remains alive.
- Payload signal status is reproduced only after cleanup, including SIGKILL as a real negative subprocess return code. Administrative stop remains a distinct nonzero result.
- Linux failed-start cleanup retains its observer through repeated interruptions until exit is confirmed; it does not hard-kill that observer on a timer.
- A configured owner sends an explicit stop byte over its retained pipe, so another inherited writer cannot suppress cancellation. Forked proxy finalization only detaches its local descriptor, without signalling the owner's job or taking an inherited mutex.
- Startup status descriptors have a single atomic owner. An unclaimed reader is cancelled before waiting for host cleanup, avoiding double close, reused-FD reads, and a blocked status writer during interrupted thread startup.
- Windows keeps its Job Object containment and resource checks. Its isolated host now uses the base interpreter rather than a virtual-environment redirecting launcher, so the retained process and acknowledged identity have the same PID.
- The supervisor's POSIX SIGTERM callback only latches a shutdown request. Locked checkpoints close admission before activation or further starts, and preserve coordinated child-before-PostgreSQL teardown.
- Metadata publication and shutdown interruptions retain closed gates and unsafe authority in `FAILED_HOLD`. Partial activation uses full coordinated cleanup; failures remain nonzero even if a later cleanup retry succeeds.
- Foreground and background shutdown publish the same instance-bound receipt after fallible control/log cleanup. The waiting authenticated stopper or locked stale reconciliation removes metadata; foreground no longer deletes it before the stopper can verify completion.
- PostgreSQL stop and close share one remaining timeout. This is not a bound on the complete shutdown sequence: unsafe authority is retained indefinitely rather than released when a timer expires.
Additional container work is implemented, rather than still pending:
- Source-start rollback retains uncertain owners and closes admission; locked stale-metadata reconciliation uses exact identity rather than PID alone.
- Durable state stays on `/data`; recoverable control metadata is under `/run/truf/control` and does not survive recreation.
- Linux manifests distinguish private application files from root-owned native binaries; the obsolete Windows OpenRouter PowerShell dependency is not the Linux provider entrypoint.
- `Dockerfile` packages Python 3.12.14, PostgreSQL 16.15, Git, tini, and hash-locked Python dependencies in the production server. TruffleHog 3.97.4 is pinned only in worker and test targets; the remote-only production server does not contain it.
- Fresh provisioning, empty-cluster initialization, 27 schema migrations, final cutover, and initialization/identity markers are implemented. Partial initialization fails closed and is not automatically repaired or adopted.
- Compose applies a read-only root filesystem, dropped capabilities, no-new-privileges, 2 CPUs, 6 GiB memory, 512 PIDs, 256 MiB shared memory, and bounded log rotation. Tini forwards SIGTERM to the foreground runtime/supervisor, not indiscriminately to its process group.
- Dependency-aware health checks require authenticated ACTIVE control, READY PostgreSQL, required workers and durable pipeline leases, schema/cutover validity, the expected PG16 data directory, and writable nonfull persistent storage.
Containment covers managed nested `OwnedProcess` trees, not arbitrary session
escapes or an observer independently killed by SIGKILL/OOM. The application
retains unsafe authority indefinitely in `FAILED_HOLD`, but Compose's stop grace
is only 10 minutes. Docker can then force termination; that is unsafe shutdown,
not successful coordinated cleanup. Two clean E2E stops do not prove safe
termination of `FAILED_HOLD` at that deadline.
## Current Linux Paths
These are the implemented image, volume, and tmpfs contracts.
| Purpose | Path |
| --- | --- |
| Image application root | `/opt/truf` |
| Application code | `/opt/truf/app` |
| New Linux runtime state | `/data/runtime-linux` |
| Separately initialized Linux PostgreSQL cluster | `/data/postgres-linux` |
| Durable result bundles | `/data/scanner-result-bundles` |
| Scanner scratch | `/data/scanner-work` |
| Private imported provider credentials | `/data/config/secrets.yaml` |
| Generated PostgreSQL password | `/data/postgres-password` |
| Ephemeral control and authority state | `/run/truf/control`, `/run/truf/authority` |
| Worker/test TruffleHog executable | `/usr/local/bin/trufflehog` (absent from the production server) |
The original physical cluster was PostgreSQL 16, but it was not retained in this
source-only copy. Do not mount Windows PostgreSQL data into a Linux server.
Initialize isolated development data; any later production-data migration needs
a separately approved logical export/import and validation procedure.
## Development Linux/WSL Procedure
This volume-only procedure is retained for isolated development and migration verification;
it is not the production edge installation procedure. Production operators must use
`deploy/edge/README.md`, including the fixed host-agent installer, active documents under
`/etc/truf/runtime`, immutable package manifests under `/etc/truf/worker-packages`, the
host-agent socket, and the combined base plus edge Compose invocation.
The following are operator commands, not commands executed by this documentation
update. Use a Linux shell or WSL with Python 3, Git, Docker's Linux daemon, and
the Compose plugin (`docker compose`). Verified host versions were Docker 29.8.0
and Compose 5.5.1. Keep Docker-managed named volumes on native Linux storage, not NTFS
or an original-runtime directory. Build steps need package/download network
access; the offline test runs below do not. Do not run the full legacy suite.
### Build
From the development checkout, define an explicitly scoped Compose helper. The
example uses the passwordless sudo Docker access used by the recorded verifier;
omit `sudo -n` if your account already has direct daemon access. Do not change
daemon permissions or install/reset WSL as part of these instructions.
```sh
cd /mnt/d/truf-docker
dc() { sudo -n docker compose --project-name truf-docker --project-directory "$PWD" --env-file /dev/null --file compose.yaml "$@"; }
dc --profile test build runtime test
```
On native Linux, substitute the development checkout path for `/mnt/d/truf-docker`.
This creates `truf-local:runtime` and `truf-local:test`. All lifecycle commands
below must retain this project name and checkout so they use the same volume.
The explicit env file avoids implicitly loading a checkout `.env`; do not supply
unreviewed Docker/Compose environment overrides or proxy credentials.
### Provision And Initialize
Use these one-off commands only while the runtime is stopped. `--no-deps` avoids
implicitly starting other services. Provision is network-disabled, generates a
new private database password, and seeds an empty provider-secret mapping. A
valid already-provisioned layout is checked without regenerating credentials;
nonempty or partially provisioned layouts are refused.
```sh
dc run --rm --no-deps --pull never -T provision
```
For an authorized production-profile run, import an existing private YAML mapping
from stdin. Replace the placeholder filename with an approved Linux-side secret
file, not a file in the original installation. Do not put secret values in command
arguments, the image, the checkout, or this document. This is not a Compose secret
mount: the locked, atomic import writes `/data/config/secrets.yaml` with private
ownership/mode and refuses an active runtime or PostgreSQL PID file. Imports can
also be repeated after confirmed shutdown for credential rotation.
```sh
dc run --rm --no-deps --pull never -T runtime import-secrets < /absolute/private/provider-secrets.yaml
dc run --rm --no-deps --pull never -T runtime initialize
```
Initialization creates an independent PG16 cluster, starts maintenance mode,
applies base/schema migrations and final cutover, confirms PostgreSQL stopped,
then publishes the initialization marker. A matching initialized volume is not
reinitialized. On partial initialization, stop and inspect offline; do not delete
markers, change identity, or rerun repair scripts to force admission.
### Start, Status, And Health
Starting `compose.yaml` uses the production configuration and can launch enabled
sources and provider workers with real network access. It is NOT the offline E2E
procedure and must only be used with separately authorized targets/credentials.
`run` initializes if needed before execing the noninteractive autostart supervisor;
the explicit initialization step above makes that first-install phase visible.
```sh
dc up --detach --no-deps --no-build --pull never runtime
dc ps runtime
dc exec -T runtime /usr/local/bin/python3 -I -S -B /opt/truf/app/container_runtime.py status
dc exec -T runtime /usr/local/bin/python3 -I -S -B /opt/truf/app/container_runtime.py health
```
`status` and `health` both call the same readiness function and return JSON on
success, nonzero on failure. They are not general stopped-runtime inventory
commands. Run them with `exec` in the existing runtime, not `compose run`, because
a new container has a different control tmpfs. During initial startup, readiness
can fail until activation and worker leases complete; Compose checks every 30
seconds with a 15-second timeout, 240-second start period, and three retries.
Readiness is not proof that every provider works or that production load is safe.
### Stop And Recreate
```sh
dc stop --timeout 600 runtime
cid=$(dc ps --all --quiet runtime)
sudo -n docker container inspect --format 'status={{.State.Status}} exit={{.State.ExitCode}} oom={{.State.OOMKilled}} restarts={{.RestartCount}}' "$cid"
```
Require an exited container, exit code 0, and OOM false; investigate unexpected
restarts. A successful `compose stop` invocation alone is not graceful-exit proof.
Do not shorten the timeout, force-kill, remove the data volume, or treat a
10-minute forced termination as safe. Runtime restart policy is `on-failure:3`;
it is not a substitute for investigating uncertain ownership or partial state.
To replace a confirmed-stopped container while retaining its existing data:
```sh
dc up --detach --no-deps --no-build --pull never --force-recreate runtime
dc exec -T runtime /usr/local/bin/python3 -I -S -B /opt/truf/app/container_runtime.py health
```
Allow readiness to complete again. Never use `down --volumes` on data you intend
to retain. The old `docker-compose.postgres.yml` is a noncanonical manual-recovery
fixture, not a deployment or an external-authority implementation.
### Offline Verification
Use the reviewed container selection, not unrestricted pytest discovery:
```sh
dc --profile test run --rm --no-deps --pull never -T test
python3 -I -S -B docker/test_verify.py
python3 -I -S -B docker/verify.py
```
The selected container suite runs without `/data` or provider credentials, with
network disabled and temporary fixtures in `/tmp`; native local child processes
and loopback control sockets are intentional. Only this unit-test service allows
execution from its 512 MiB `/tmp` for temporary venv and askpass fixtures. Both
the production and E2E services use the same 128 MiB `noexec,nosuid,nodev` `/tmp`.
`docker/test_verify.py` is a separate six-test stdlib regression suite, passing
on both Windows and WSL Linux;
its total is not silently added to the selected container-suite count.
`docker/verify.py` requires already-built local runtime/test images and an already
Git-ignored `docker/test-results/latest.json`. It performs no builds or pulls and
does not edit ignore files. Run the verifier itself as the normal Linux user; it
tries direct Docker access, then `sudo -n docker`. Git is used for the read-only
evidence ignore guard, not for mutations.
The verifier invokes only `compose.e2e.yaml`, generates a unique `truf-e2e-*`
project, pins the local images by ID, and uses fresh private `data` and `tools`
named volumes. All services have network disabled, proxy settings cleared, no
host data binds or published ports, and no automatic restarts. It preserves the
production runtime image/entrypoint but prepares a narrowed offline configuration:
a local Git fixture and real TruffleHog with verification disabled, then the real
scan/bundle/ingestion/projection pipeline and OpenAI worker. ONLY that worker's
HTTP transport is substituted with a synthetic 401 response; no live provider
request is made. Production source selection/provider parity is not tested by
this profile. Do not manually merge it into production Compose or rerun prepare
after recreation.
The verifier checks healthy activation, pipeline lineage, one synthetic HTTP
request, graceful stop, recreation on the same volume, unchanged persisted
counts/hashes, no duplicates and no second HTTP request, health, and a second
graceful stop. Its aggregate check budget is 3600 seconds, each health wait at
most 240 seconds, each stop 600 seconds, and failure handling has a separate
720-second budget. Nonzero, forced, or OOM exits cannot pass.
On success it removes only its ownership-verified containers and volumes; use
`python3 -I -S -B docker/verify.py --keep` instead to retain stopped test artifacts.
Failures retain artifacts after a guarded stop attempt, possibly with running
containers if ownership or stopping cannot be proven. Record the printed project
name for investigation; do not use broad prune/down/kill commands. Evidence is
written to `docker/test-results/latest.json` as counts, hashes, image IDs, statuses,
and durations without raw command logs or secret values.
## Current Verified Evidence
The fresh recorded E2E in `docker/test-results/latest.json` has `status.result`
and `status.checks` both `passed`, `status.cleanup` equal to `removed`, and zero
owned containers, networks, or volumes remaining. The final run on 2026-09-15
took 54.374 seconds and repeated the earlier successful fresh-volume run.
- Fresh PG16 initialization and all 27 migrations completed with identity/cutover checks; both authenticated health checks passed.
- Real local Git and native TruffleHog produced one finding, one scan, one result bundle/reservation, and one queue attempt through ingestion and projection, with zero pipeline quarantine/errors.
- The real OpenAI worker used only the offline synthetic 401 transport: one first HTTP request, one linked keycheck result/current state, two projection jobs, and three projection appends.
- Recreation used a new container on the same volume without rerunning prepare. Persisted artifact IDs, migration/cutover hashes, SQL summary, output bytes and projection ledger hashes matched; repeated keycheck made zero HTTP requests and introduced no duplicate rows/appends.
- Both graceful stops recorded container exit 0, OOM false, and zero restarts. Both separate read-only stopped-volume checks confirmed private PG storage, initialization, and absence of `postmaster.pid`.
- Shutdown receipt publication is inferred ONLY from the foreground zero-exit contract. The receipt itself was not read after stop because `/run/truf` tmpfs had disappeared; evidence labels this `exit_contract_only_tmpfs_removed`.
- Final selected container regression suite: 578 passed, seven Windows-only tests skipped on Linux, in 22.25 seconds. This includes all six fresh-volume regressions. Separately, `docker/test_verify.py` passed all six tests on both Windows and WSL Linux. These are scope-specific counts, not counts stored in the E2E JSON.
- All 157 Python files under `app`, `tests`, and `docker` parsed successfully. `git diff --check` passed; existing PowerShell LF/CRLF notices were not whitespace failures. No changes were staged or committed and no Git remote was added.
Recorded image IDs (not a promise that mutable local tags still point to them):
- Runtime: `sha256:92502a2581ebcabe79ddc28744dabcfb3000b99b7c0c5262c4c09aff9dc0267b`.
- Test: `sha256:4ce11325728ba3e58e6643c1c8e800f317179d5c7c50e7e80568b58f62dbdfd0`.
## Remaining Limitations
`DOCKER_READINESS_AUDIT.md` records the original Windows audit. Its historical
line references and pending foundation tasks are not a current implementation
checklist. The remaining acceptance boundaries are:
1. `FAILED_HOLD` can outlive Docker's 10-minute grace and be terminated unsafely. No general crash/OOM/forced-stop recovery proof is claimed.
2. Production source/provider parity, live providers, real credentials, broad discovery, sustained load, throughput, and resource sizing have not been validated by the offline E2E.
3. There is no arm64 build/runtime proof, even though TruffleHog has an arm64 checksum entry.
4. External PostgreSQL authority is not implemented. Only a fresh supervisor-owned native Linux PG16 cluster is supported here.
5. There is no original Windows database migration, original-data equivalence, or production cutover proof. The historical read-only lineage record below is not such a migration proof and was not executed by this update.
## Historical Verification (Superseded Status)
The sections below preserve earlier scoped verification records. Their test
counts, file inventories, no-install/no-image statements, staging status, and
then-open container checks apply only to those stages and are superseded by the
current evidence above. At the earlier lifecycle stage Docker CLI was unavailable
and no image build/container/database migration had yet been performed. That is
no longer the current state. No historical original-runtime operation below is
claimed as an execution by this documentation update or by the fresh offline E2E.
### WSL Ownership Verification
Verified on 2026-09-14 in `Ubuntu-24.04`, WSL version 2, Linux kernel
`6.18.33.2-microsoft-standard-WSL2`, CPython 3.12.3, as unprivileged UID 1000:
- The initial successful startup took approximately 44 seconds, exceeding the earlier 10-second probe limit. WSL reported automatic NAT-to-VirtioProxy networking fallback. No network setting was changed, and the tests needed no network access.
- All nine previously skipped `LinuxOwnedProcessIntegrationTests` first passed on the real kernel. They were then included in the expanded run: 38 tests passed, zero failures/errors/skips, in 3.682 seconds excluding WSL startup.
- The expanded selection also covers mocked failure paths, ordinary output/timeout behavior, static process-safety checks, isolated-host startup, an offline temporary virtual environment, and synthetic credential filtering.
- The first expanded run exposed a test-fixture issue: Ubuntu's standard-library `sitecustomize.py` shadowed the virtual environment's fixture in the positive control. The test now puts its own module first through a temporary `PYTHONPATH`, supplied to both control and isolated-host scenarios. Both startup-hook markers must appear in the control and remain absent for the isolated host. No application code was changed for this correction.
- The existing Windows selection was rerun after that test-only fix: 159 passed, nine Linux-only skips. Those nine skips are covered by the successful WSL run, not left untested.
- The Linux runner used only the standard-library `unittest`, `python3 -I -S -B`, an empty inherited environment via `env -i`, and explicit safe locale/path/temp settings. No packages were installed and no pytest plugins were loaded. It asserted the effective `tempfile` directory before collection; a 90-second test watchdog was separate from the longer WSL startup allowance.
- Code was read from `/mnt/d/truf-docker`; `HOME`, `TMPDIR`, `TEMP`, and `TMP` were confined to `/mnt/c/Users/PRO100~1/AppData/Local/Temp/opencode`. Fixtures used mounted Windows storage, not a new native Linux data volume. This does not validate Linux storage ownership, permissions, or container mounts.
- No canonical runtime CLI, original database, provider, Docker service, or copied recovery Compose fixture was launched. Staging refusals remain unchanged.
Exact expanded `unittest` selection, with the clone's `tests` directory explicitly
added to the isolated runner's module search path:
- `test_owned_process_linux`
- `test_owned_process.OwnedProcessTests`
- `test_owned_process.StaticProcessSafetyTests`
- `test_owned_process_boundary.OwnedProcessHostBoundaryTests`
- `test_owned_process_boundary.CredentialBoundaryTests.test_host_environment_strips_mixed_case_database_credentials_only`
### Earlier Windows Lifecycle Verification
Verified on 2026-09-14 with Windows CPython 3.12.3 after the lifecycle changes:
- 159 selected tests passed; nine native Linux tests were skipped because they require a real Linux kernel and `/proc`. The passing selection includes native Windows Job/virtual-environment checks, mocked Linux kernel operations, mocked supervisor lifecycle scenarios, and the previous foundation checks.
- Regressions cover interrupted status-reader startup, cancellation before host reaping, duplicate control writers, foreign-proxy detachment, sticky failure results, receipt verification, shutdown ordering, and graceful TERM between activation/start checkpoints.
- The nine Linux-only fixtures cover live kernel identities and sessions, transitive cleanup on normal exit and actual pipe EOF, explicit stop with another writer retained, forked-proxy finalization, live adopted-child reaping, SIGKILL/SIGTERM status, and rejected startup after nested children are ready. Test signals use pidfds bound to the acknowledged payload identity.
- All 145 application/test Python files parsed successfully. The source-only artifact check found 301 files excluding `.git`, with no credential pools, databases, results, runtime/cache directories, or links.
- `git diff --check` passed. Seven existing PowerShell LF/CRLF warnings are not whitespace failures. The index and remote list remain empty; the only commit is still `1b3c7fc`.
- Independent scoped static reviews were followed by regression fixes and reruns. The last ownership review found no remaining concrete issue in the reviewed corrections; this is not a Linux conformance result.
- At that stage, a bounded WSL probe timed out without output; the subsequent WSL investigation and successful tests are recorded above. No WSL reset/install, image build, application launch, database access, or live provider test was performed by that lifecycle continuation.
The combined selection was limited to these modules/node IDs:
- `tests/test_owned_process_linux.py`
- `tests/test_owned_process.py`
- `tests/test_owned_process_boundary.py::OwnedProcessHostBoundaryTests`
- `tests/test_owned_process_boundary.py::CredentialBoundaryTests::test_host_environment_strips_mixed_case_database_credentials_only`
- `tests/test_supervisor_foreground_shutdown.py`
- `tests/test_supervisor_managed_postgres_gate.py`
- `tests/test_observer_only_coordinated_shutdown.py`
- `tests/test_docker_foundation.py`
- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_default_data_directory_is_unchanged`
- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_external_data_directory_expands_without_moving_runtime_assets`
- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_external_data_identity_mismatch_remains_fail_closed`
- `tests/test_postgres_runtime.py::PostgresRuntimePathTests::test_external_data_identity_verifies_only_when_exactly_bound`
- `tests/test_supervisor_safety.py::ManagedConfigurationAuthorityTests::test_managed_dsn_overrides_config_database_urls`
The test child used `python -X utf8 -B`, pytest `-q --tb=short -rs`,
`-p no:cacheprovider -o addopts= --confcutdir=tests`, disabled plugin autoload,
and empty `PYTHONPATH`, `PYTEST_ADDOPTS`, and `PYTEST_PLUGINS`. Runtime/DSN
overrides were removed only from that child's environment. All of `TMPDIR`,
`TEMP`, and `TMP` pointed at the approved temporary work area, and the runner
asserted the actual `tempfile` directory before collecting tests.
### Previously Recorded Verification
The following earlier results are retained as historical records. They are not
new original-runtime operations performed by this lifecycle continuation.
Recorded on 2026-09-14 with Windows CPython 3.12.3:
- 19 targeted offline tests passed: all 14 foundation tests, four existing PostgreSQL path/identity tests, and the existing managed-DSN precedence test.
- The existing external-data path fixture now uses portable separators too; intentional Windows-path rejection remains separately covered.
- Python CLI tests verify the complete literal refusal AST before invoking an isolated interpreter. PowerShell safety checks parse scripts without executing them, so a broken refusal cannot make the test control the host.
- All 143 Python application/test files parsed successfully; no syntax errors.
- Parsed YAML comparison against the initial Git commit found exactly 31 changed deployment-path fields; all other settings were identical.
- The final source tree contained 299 files, approximately 6.85 MiB excluding `.git`, with no real credentials/pools, databases, results, runtime/cache directories, or links found by the artifact check.
- The original supervisor and PostgreSQL were still running; no copy writers remained. There were no staged changes or Git remotes, and the only commit remained the initial baseline.
- A read-only original-runtime lineage proof joined one completed queue reservation through its bundle, scan, finding, keycheck candidate/result, projection jobs, appends, and stream generations. Exact event/hash relationships held, and both scan projections plus the keycheck projection matched their recorded byte offsets, lengths, record counts, and SHA-256 digests; no target or credential value was emitted.
Tests ran with bytecode writes, pytest plugin autoload, and pytest cache disabled;
application/database environment overrides were removed from the test process.
Temporary test files were confined to the approved temporary work area. The
PowerShell review and Windows POSIX-path emulation do not prove Linux container
behavior, and the context allowlist check is not an actual Docker build.
Do not run the entire existing test suite against this machine: it includes
native-process, socket, database, and live integration scenarios.
+364
View File
@@ -0,0 +1,364 @@
# Docker Readiness Audit
Date: 2026-09-14. Scope: the application and runtime launch chain in `D:\truf`.
This is an inspection report, not an implementation. Application code, configuration, secrets, databases and runtime data were not changed. The target assumed here is a Linux container. Windows containers, target CPU architecture, deployment host and resource budget have not been specified.
## Verdict
**The project is not ready to containerize unchanged. Adding a Dockerfile around the PowerShell launchers is insufficient.** There are both packaging gaps and concrete defects in the POSIX process/security paths. PostgreSQL is also part of a local process-ownership protocol, not just a replaceable connection URL.
The first deployment should retain **one supervisor and its authenticated children in one container, with one runtime replica**. PostgreSQL can initially remain locally managed in that container, or become a separate service after an explicit external-database authority mode is implemented. Neither option is currently a configuration-only change.
No deployment Dockerfile or `.dockerignore` was found in the inspected application/root. `docker-compose.postgres.yml` is explicitly a **noncanonical manual recovery fixture**, not the production runtime definition. DockerHub scanning in the application is unrelated to deployment packaging.
Priority definitions:
- **P0 / B01-B14:** resolve before a working, safely restartable Linux deployment. Some items need packaging/provisioning rather than application changes.
- **P1 / R01-R07:** resolve before unattended operation with persistent data.
- **Conditional / C01-C07:** required only for the stated feature or deployment choice. These are not all prerequisites for a headless, single-runtime deployment.
## Runtime Map
| Component | Actual entry points and role |
| --- | --- |
| Canonical startup | `app/runtime_bootstrap.py`, `app/child_bootstrap.py`; isolated interpreter startup and authenticated imports |
| Lifecycle | `app/supervisor.py`, `app/supervisor_instance.py`, `app/lifecycle_authority.py`; admission, ownership, control, manifests and shutdown |
| Subprocess containment | `app/owned_process.py`, `app/process_identity.py` |
| Scanning | `app/console_runner.py`, `app/scanner.py`; native TruffleHog and Git |
| Key checking | `app/keycheck_runner.py`, `app/keycheckers/`; Python provider processes, HTTP and AWS SDK |
| Database | `app/postgres_runtime.py`, `app/db_backend.py`, `app/scanner_db.py`; managed PostgreSQL and normalized runtime schema |
| Result pipeline | `app/result_bundle.py`, `app/result_ingester.py`, `app/jsonl_projector.py`, `app/janitor.py` |
| Optional UI | `app/dashboard.py`; supervised read-only Streamlit dashboard |
| Retired entry points | `app/app.py:1-13` and `app/scan_manager.py:8`; do not use as the container application |
## P0: Deployment Blockers
### B01. Replace Windows Path Assumptions, Not Just Environment Variables
**Evidence:** `app/config.yaml:8-12,30-37,70-89,109-120,140-150`; `app/paths.py:7-8,17-18,46-59,70-110`.
The configuration contains `D:\truf`, `S:\postgres-data`, `S:\scanner-result-bundles`, `S:\scanner-work`, `C:\Tools\trufflehog.exe` and backslash-based derived paths. POSIX treats a Windows drive path as relative and a backslash as an ordinary filename character. `os.path.normpath()` does not translate them.
YAML `root_dir` wins over `SCANNER_ROOT_DIR`/`SCANNER_PROJECT_ROOT`; YAML `trufflehog_path` wins over `TRUFFLEHOG_PATH`. Several defaults remain Windows-specific even if those YAML values are removed. The managed PostgreSQL DSN has its own precedence and must remain consistent with the selected authority mode.
**Isolated reproduction:** executing the actual path functions with POSIX path semantics, a config location under `/opt/truf/app`, and Linux environment overrides produced `/opt/truf/app/D:\truf` for the root and `/opt/truf/app/C:\Tools\trufflehog.exe` for TruffleHog. With empty YAML, the root became portable but the default log path still became `/srv/truf/runtime\logs`. No application was imported or started.
**Correction:** supply a complete Linux config/profile and make path defaults platform-aware with joins or portable separators. Cover global paths, supervisor instance/status/lock paths, policy assets, caches and maintenance paths. Define and document precedence rather than assuming environment variables override YAML. Preserve the working Windows profile; do not replace backslashes indiscriminately in arbitrary settings or stored data.
### B02. Use the Canonical Foreground Entrypoint
**Evidence:** `start_runtime.ps1:18`; `start_core_runtime.ps1:23`; `app/runtime_bootstrap.py:24-31,115-129`; `app/supervisor.py:4468-4481,4878-4882,5000-5001,5277-5278`.
The launchers use `--background`, return after starting a child, and therefore have the wrong lifetime for a container entrypoint. Interactive mode is not automatically disabled without a TTY; EOF can end its loop. Noninteractive mode without autostart is not sufficient either.
**Correction:** use exec-style startup of the canonical supervisor, without daemonization, with explicit `--non-interactive --autostart`. Keep `-I -S -B` and the bootstrap entrypoint binding. Use `--no-dashboard` for the initial headless deployment. Do not launch workers directly or substitute `streamlit run app.py`.
Illustrative command contract for the embedded-PostgreSQL option, **only after the other fixes and offline provisioning**; not executed during this audit:
```text
python3 -u -I -S -B /opt/truf/app/runtime_bootstrap.py supervisor -- --runtime-bootstrap-entrypoint /opt/truf/app/supervisor.py --config /opt/truf/app/config.yaml --non-interactive --autostart --no-dashboard --no-clear --with-postgres
```
Select sources explicitly. `start_core_runtime.ps1` and `app/config.linux.yaml` use the exact distributed producer set `gitlab,dockerhub,huggingface`; operational workers such as `keychecks` are configured independently. `--once` is not a guarantee that the whole supervised pipeline is a terminating batch job (`app/supervisor.py:974`).
### B03. Fix the POSIX OwnedProcess Identity Handshake
**Evidence:** `app/owned_process.py:1006-1014`; `app/scanner.py:11479-11495,1825-1842`.
The POSIX handshake returns payload/host identities without `creation_time`; the payload executable is taken from command text. The scanner passes that identity into its ownership marker, which requires `pid`, `creation_time` and `executable`. Access to the missing field can raise `KeyError` on the real native-scanner launch path.
**Correction:** return complete, canonical, verified process identities before startup acknowledgement. Preserve the stdlib-only containment bootstrap and exact identity checks; a PID alone is insufficient. Add a real POSIX handshake-to-marker test, not a mock that supplies the missing field.
### B04. Make Nested POSIX Process Containment Actually Contain the Tree
**Evidence:** `app/owned_process.py:442-445,917-921,989-993`; `app/supervisor.py:1180-1189`; `app/keycheck_runner.py:2226-2233`; `app/scanner.py:11479`.
An outer payload owns process group A. Its inner containment host inherits A, but the inner provider/native payload creates session/group B. Killing A can kill the inner host while leaving B running. The dead host can no longer reliably process its parent pipe and stop B. This affects source restarts and dependency-loss handling inside a still-running container, not only final container termination.
**Correction:** implement nested containment with verified tree termination, such as isolated observer hosts with cascading stop/acknowledgement, or an appropriately designed cgroup mechanism. Do not replace identity-based ownership with name-based process killing. An init/reaper alone does not fix this defect.
### B05. Integrate Container Signals and a Real Shutdown Budget
**Evidence:** `app/supervisor.py:2338-2361,3447,3551,5293-5305`; `app/postgres_runtime.py:824`; `app/config.yaml:168-170,185`.
The supervisor handles `KeyboardInterrupt` but does not register a SIGTERM handler. Docker's normal stop signal therefore is not wired into coordinated shutdown; PID 1 also has special Linux signal semantics. The shutdown path includes admission closure, pipeline draining, sequential child stops and PostgreSQL shutdown. A short container grace period can interrupt that protocol. Locally managed PostgreSQL is daemonized through `pg_ctl`, so reaping also needs attention.
**Correction:** connect SIGTERM to the existing STOPPING/shutdown event flow, provide an init/reaper, and forward signals to the supervisor rather than indiscriminately to its whole process tree. Size `stop_grace_period` from the total measured shutdown deadline, not only the PostgreSQL timeout. Current configuration includes a 120-second PostgreSQL shutdown timeout and a 180-second background-shutdown timeout; neither proves that a particular total container grace is sufficient.
An explicit SIGINT stop signal could be an interim tested workaround, not a substitute for the complete TERM/PID1/nested-process fix. An unconfirmed stop intentionally enters `FAILED_HOLD`; no finite grace period can guarantee a clean outcome there. Preserve authority, expose failure and require an escalation procedure instead of releasing locks optimistically.
### B06. Make Restart Safe Across Container PID Reuse
**Evidence:** `app/supervisor.py:5067`; related identity and instance handling in `app/supervisor_instance.py` and `app/process_identity.py`.
Persisted, otherwise valid instance metadata combined with a reused PID can block a new foreground supervisor. PID reuse is particularly predictable across fresh container PID namespaces.
**Correction:** reconcile stale instance state using exact identity under the correct authority lock, and/or place runtime-instance/control metadata in deliberately ephemeral storage. Keep persistent business data separate from per-instance PID, control, shutdown-receipt and session state. Never delete a lock or metadata merely because it is old. Verify that the previous runtime cannot still own the database before recovery.
### B07. Provision Non-Root Ownership, Private Paths and a Usable Lock Root
**Evidence:** `app/runtime_security.py:391-400,651-685,842-889,989-998`; `app/postgres_runtime.py:258-271`.
POSIX private-file policy requires ownership by the effective UID and no group/other permissions. Lifecycle preflight is read-only and requires configured directories to exist already, including application/root paths, runtime data, bundle subdirectories and PostgreSQL paths. A fresh named volume or a default root-owned Docker secret does not automatically meet this contract. The PostgreSQL process inherits the runtime UID; Linux PostgreSQL cannot run as root.
The authority lock root is hardcoded to `/var/lock/truf`. `/var/lock` is a symlink on many Linux images and conflicts with the no-symlink policy; creating it as an unprivileged user is another problem.
**Correction:** choose a stable non-root UID/GID, provision all required paths and file ownership before normal startup, and use a real prepared private authority directory, for example `/run/truf/authority`. Make its location explicit instead of depending on a distribution's `/var/lock` layout. Data/config modes normally need owner-only access. Verify the actual behavior of named volumes, secret mounts and Docker Desktop mounts; do not solve this with `chmod 777` or privileged mode.
Provisioning may need a separate controlled initialization step. The steady-state application should not require root, and startup should retain its fail-closed validation rather than silently repairing arbitrary mounted data.
### B08. Separate Executable Trust Policy From Data-File Hardening
**Evidence:** `app/runtime_security.py:651-685`; `app/lifecycle_authority.py:432-443,636-639`; `app/migrate_runtime_safety.py:2502-2503,2508-2521`.
The current hardener sets every POSIX file to `0600`, removing native executable bits. Conversely, ordinary system-installed `0755`, root-owned Git/TruffleHog executables do not satisfy the current exact-private manifest policy. The offline hardener also hardens each file's parent and runtime trees; pointing it at a system binary can attempt to harden a shared system directory.
**Correction:** define and verify executable permissions separately, preserving `x` and a trusted owner. Choose deliberately between private executable copies and an explicit immutable-system-binary trust policy. Account for the complete Git installation and PostgreSQL libraries/helpers, not just one binary. Do not run the existing recursive hardener over `/usr/bin` or blindly apply `0600` to a native runtime. Test hardening idempotency without breaking execution.
### B09. Package the Code Authority and Isolated Import Layout Correctly
**Evidence:** `app/lifecycle_authority.py:21,38-70,378-443`; `app/runtime_bootstrap.py:47-83`; `app/child_bootstrap.py:177-195`.
The manifest unconditionally includes `../runtime/check-openrouter-keys.ps1`, `../start_runtime.ps1` and `../stop_runtime.ps1`. Excluding all PowerShell or all `runtime/` content before changing this contract can break authentication even on Linux. Application-tree symlinks and cached application bytecode are rejected. Creating an ordinary virtualenv under the application tree can introduce both.
**Correction:** make the external authority-file set OS-aware, or retain these inert first-party files at the required relative locations until that change is made. Ship all required application modules and detector/policy assets, without application `.pyc`/`__pycache__` artifacts or symlinked application paths. Keep the dependency environment outside the inspected `app/` tree. Install dependencies for the exact interpreter used by isolated bootstrap; arbitrary `PYTHONPATH` and user-site packages are not a substitute.
Use immutable releases with a full controlled restart. Live edits to mounted code/config are incompatible with manifest drift detection (`app/supervisor.py:3045`).
### B10. Decide and Implement the PostgreSQL Authority Topology
**Evidence:** `app/postgres_runtime.py:112-130,249-255,642-671,1335`; `app/db_backend.py:42-101`; `docker-compose.postgres.yml:1-20`.
Current managed mode expects loopback, local PostgreSQL executables, a local data directory, bound cluster identity and an inspectable local postmaster process. Changing the DSN host to a Compose service name does not implement external PostgreSQL support. The URL parser also rejects query parameters, so appending `?sslmode=...` is not currently a supported TLS configuration route.
| Option | Required work |
| --- | --- |
| Locally managed PostgreSQL in the runtime container | Preserve a shared PID/network namespace and lifecycle owner. Supply Linux PostgreSQL executables at the currently fixed `runtime/postgres/pgsql/bin` layout, or make the binary paths configurable. Keep PGDATA separate from binaries and use the compatible non-root UID. Include init/reaping and coordinated database shutdown. |
| Separate PostgreSQL container/service | Add an explicit external authority/backend mode across `postgres_runtime.py`, DSN validation, supervisor lifecycle and readiness. It must not require local postmaster PIDs/data paths/binaries or attempt local start/stop. Preserve authenticated endpoint/cluster identity checks, fencing and schema readiness. Define explicit TLS settings if required. |
Simply disabling authority, process or endpoint checks is not an acceptable implementation. A pre-existing PostgreSQL instance can be observed without being owned; do not assume the supervisor will stop it. `maintenance-start` returns after startup and is not a PostgreSQL container service entrypoint (`app/postgres_runtime.py:1463`).
Keep `docker-compose.postgres.yml` separate: it is profile-gated, uses `restart: no`, Windows bind paths and a deliberately noncanonical endpoint. Its `postgres:16` image is not evidence of the version required by the authoritative cluster. Do not silently promote this recovery database to production authority.
### B11. Add an Explicit Offline Provisioning and Data-Migration Procedure
**Evidence:** `app/postgres_runtime.py:321-356,400`; `app/scanner_db.py:238,5145-5177,20520-20530`; `app/migrate_runtime_safety.py:2971-3008`.
`bootstrap_cluster_identity()` does not run `initdb`; it expects an existing cluster and executable set. It rejects supervisor metadata, `postmaster.pid` and a listening endpoint. The identity binds paths, binaries and cluster identity, so an old Windows identity file must not be reused as a Linux authority binding.
Workers also require the runtime safety schema and a valid `postgres-normalized-v2-authority` final-cutover marker with evidence. A fresh PostgreSQL service reporting `pg_isready` is not an application-ready database.
**Correction:** distinguish two offline phases. First, initialize/restore and bind the embedded cluster identity while the target PostgreSQL server is stopped. Then run database schema/cutover work with PostgreSQL available in controlled maintenance mode but all normal sources/pipeline workers stopped. Use `--initialize-base` for a genuinely fresh installation, not as a substitute for understanding an existing dataset. Complete the applicable normalization, projection reconciliation and final-cutover checks before admitting workers.
Determine the real source/target PostgreSQL versions before transfer. Prefer a planned logical dump/restore for the Windows-to-Linux move unless another backup method is explicitly validated as compatible; do not assume copying Windows PGDATA works. Back up and transfer matching result bundles/projection data as well. Review legacy absolute Windows locators and use the applicable migration/reconciliation paths, not blanket database string replacement.
`app/migrate_layout.py:18,280` and the legacy spool default in `app/migrate_runtime_safety.py:74` also contain host-specific paths; do not use their defaults as Linux provisioning instructions. Optional `pg_trgm` creation is attempted defensively, not a proven unconditional startup prerequisite. No live migration should be run until restore/rollback and exclusive ownership are established.
### B12. Build a Complete, Reproducible Linux Dependency Set
**Evidence:** `app/requirements.txt:1-9`; `app/requirements-keycheckers.txt:1-4`; `app/child_bootstrap.py:24-34,177-195`; `app/keycheck_runner.py:2168-2175`; `app/lifecycle_authority.py:242-291,378-390`; `app/scanner.py:11168,11403-11405,11932,12993,13012,13834`.
- Installing only `requirements.txt` misses `boto3`/`botocore`; isolated bootstrap requires them for every keycheck-provider. Installing only `requirements-keycheckers.txt` misses `PyYAML`. Install the union or define complete, tested profiles. `zstandard` is currently required for every scanner bootstrap, not just an enabled Docker source.
- Pin a tested, patched CPython minor and dependency resolution, including a compatible boto3/botocore pair. The host has Python 3.12.3, but that is neither a recommended security patch level nor a Linux compatibility result. Archive extraction uses version-sensitive tarfile APIs; a strict minimum of 3.12 was not established because some security APIs were backported.
- Supply Linux TruffleHog and full Git for the target architecture, with release/checksum verification. The resolver currently prefers a present private `runtime/git/cmd/git.exe` without an OS check. Exclude Windows vendor binaries and make the Linux resolution explicit. Git and TruffleHog are required by the current global manifest even for a restricted source set.
- Ensure TruffleHog's subprocess `PATH` resolves the same intended Git installation as the manifest, including its HTTPS transport helper. Copying a lone `git` executable is insufficient.
- Validate native wheels/ABI and stdlib `ssl`, `sqlite3`, `zlib`, `bz2`, `lzma`, plus CA certificates. Check shared-library requirements of the selected TruffleHog and, if embedded, PostgreSQL build. Windows wheels and extensions cannot be reused. A glibc-based image is a simpler first target than assuming Alpine/musl compatibility.
- `psycopg[binary]` with a supported wheel does not automatically require `libpq-dev`/`pg_config`; source-build requirements depend on wheel availability. The zstandard CLI does not replace the Python package. Go/CGO are build dependencies only if the selected TruffleHog is compiled from source.
- Validate the actual TruffleHog CLI and output contract: the wrapper uses Git/Docker/filesystem/HuggingFace paths, archive flags and branch/SHA/local-development options. A successful version probe alone does not validate these. The parser also uses the `finished scanning` diagnostic to distinguish complete work from an incomplete command.
### B13. Prevent Secrets and Host State From Entering the Image
**Evidence:** root `.gitignore`; root/application file layout; `app/runtime_security.py:872-889,989-998`; manifest exceptions in `app/lifecycle_authority.py:66-70`.
The working directory contains credential files, credential backups, databases, scan output, keycheck output, state, logs, temporary data and bundled Windows tools. Their contents were not read for this audit. `.gitignore` does not protect a Docker build context.
**Correction:** create `.dockerignore` plus an allowlisted `COPY` strategy. Exclude real `.env*`, secret/backup/lock variants, databases including WAL/SHM, findings/results, queues, logs, state, scratch data, caches, local tool state and Windows vendor distributions. Account explicitly for the currently manifested first-party runtime scripts instead of blindly excluding them. Do not bake credentials into layers, build arguments or a committed Compose file.
Provide non-secret example configuration and inject secrets at runtime. Test owner and mode compatibility under B07: a root-owned `0444` secret mount does not satisfy the current effective-UID private policy. Keep code/config read-only after provisioning where feasible. The optional credential-writeback workflow is covered separately in C04.
### B14. Persist the Whole Data Pipeline and Preserve Filesystem Semantics
**Evidence:** `app/config.yaml:11-19,34-37,79-81,109-120,140,148`; `app/result_bundle.py:67-85,331-354,388-403`; `app/jsonl_projector.py:109-153`; `app/keycheck_runner.py:2201-2203`.
PostgreSQL is not the only durable store. Result reservations refer to payload bundles on disk. Bundles are flushed/fsynced and atomically published from `tmp` to `ready` under one root; the commit reference is relative to that root. Losing the bundle volume while keeping PostgreSQL can lose pending ingestion inputs. Output publication also has file-level state and locks.
| Data class | Deployment treatment |
| --- | --- |
| PostgreSQL data | Durable volume; version-compatible backup/restore; one authority |
| Result bundles | Durable volume with `tmp`, `ready` and `quarantine` together; preserve atomic rename/fsync behavior |
| Results and keycheck outputs | Preserve JSONL, rotation/publication state and needed replay inputs; maintain owner-only access |
| Queue/state/resolver and SQLite caches | Classify individually; persist required resume state, distinguish rebuildable caches from authoritative PostgreSQL data |
| Work clones/download/extraction scratch | Separate bounded writable storage; do not assume it fits memory-backed tmpfs |
| Control/PID/session metadata | Deliberately per-instance storage or exact-identity reconciliation; do not restore stale runtime identity as business data |
| Logs | Bounded retention or a secure collector; do not grow the container writable layer indefinitely |
| Config and secrets | Separately provisioned/injected; not bundled into data/image backups indiscriminately |
Do not mount bundle `tmp` on a different filesystem from `ready`. Validate ownership, no-symlink policy, locks and durable atomic publication on the actual storage driver. Do not assume Windows binds, SMB or NFS have the required POSIX behavior. Named volumes backed by a suitable native Linux filesystem are the safer first choice, but still require testing.
Mounting only the old `runtime/` directory misses the configured `S:` locations. Root-level `scanner.db` and old output files are not automatically the authoritative deployment dataset. Establish the transfer inventory before copying.
Current capacity settings include a 3 GiB bundle budget, a 192 MiB per-event cap, a 2 GiB projection backlog budget and a 20 GiB free-space floor. Provision space for concurrent work, PostgreSQL/WAL, bundles and outputs, or deliberately retune those policies. A small default container disk can refuse scans even while the process is healthy. Test a coordinated database-plus-bundle restore, not just `pg_dump` in isolation.
## P1: Unattended Operation
### R01. Retain Ownership When Failed Startup Cleanup Cannot Confirm Exit
`app/supervisor.py:1200-1208` swallows errors from terminate/wait and clears the retained process reference. Preserve the owner/identity and enter the existing failed-hold path if rollback cannot prove that a child stopped. Otherwise a failed start can leave an untracked process. Verify this after the POSIX containment correction.
### R02. Preserve Signal/OOM Exit Status
`app/owned_process.py:930` attempts to reproduce a signalled payload exit through signal handling, but installing a handler for SIGKILL is invalid. A payload terminated with `-9` can be reported as host exit 127. Correct the signal-exit reproduction and test OOM/SIGKILL separately from ordinary program failures; do not label this as a container memory-policy fix by itself.
### R03. Unify Foreground Shutdown Completion
`app/supervisor.py:5327-5339` writes shutdown receipts only for background children, while POSIX inspection of a non-child process cannot retrieve its exit code. This can make the existing stop workflow report failure after a foreground container runtime has actually exited. Define one completion protocol for both launch modes. Until then, authenticated `--cmd shutdown` plus independently waiting for the supervisor/container to exit is different from trusting the shutdown acknowledgement alone.
### R04. Add Dependency-Aware Health and Recovery Semantics
`app/supervisor.py:5225,1285,3361-3367,3551` distinguishes activation, held workers, initial ingester readiness and failed-hold state. A live PID, ACTIVE handshake, Streamlit health response or PostgreSQL TCP response is not enough to certify the pipeline.
Expose a read-only machine health result covering supervisor phase, authenticated database/cluster identity, schema/cutover readiness, ingester/projector heartbeat or singleton lease, configured required workers, storage/backlog health and any unrecoverable hold. Allow an honest startup period without admitting work prematurely. The initial source gate opens once; explicitly decide whether later dependency loss should close it or allow bounded asynchronous intake, and test that policy through outage, backlog exhaustion and recovery.
Distinguish degraded readiness from a dead process. A Docker healthcheck alone does not restart an unhealthy container; a restart policy normally responds to process exit. Do not configure blind health-triggered replacement that discards a `FAILED_HOLD` ownership dispute.
### R05. Replace Host Resource Assumptions With Container Budgets
`app/scanner.py:291,1221,1279-1290,11460-11472`; `app/owned_process.py:965`; `app/config.yaml:90-95,145`.
Windows Job memory/CPU/priority settings are not enforced by the POSIX branch. The configured TruffleHog Job memory limit is 4 GiB; putting that value in YAML does not create a Linux limit. CPU counts may describe the host rather than the effective quota. The optional bonus scan slot uses Windows resource counters and fails closed on Linux.
Set explicit workload concurrency, cgroup CPU/memory/PID budgets and storage limits. Account for PostgreSQL, Python, native payloads, one containment-host process per owned job and threads. A whole-container memory cap is not equivalent to the old per-tree Windows Job cap. Explicitly disable the bonus slot initially or implement quota-aware Linux telemetry without weakening admission safety. Derive limits from representative tests rather than multiplying configured maxima into an asserted minimum RAM requirement.
### R06. Make Logs Observable Without Depending on Ignored Python Variables
`app/supervisor.py:240` and `app/keycheck_runner.py:2168-2175` launch isolated interpreters. `-I` ignores `PYTHONUNBUFFERED` and `PYTHONIOENCODING`; adding those variables to Compose is not a reliable buffering/encoding fix. Use explicit interpreter flags such as `-u` or configure streams, and verify child output under the container locale. Retain necessary file logs with rotation/collection and keep credential-bearing output private and redacted from generic health messages.
### R07. Enforce Provider Process Deadlines
`app/keycheck_runner.py:2263` waits for the provider process without a process-level timeout. A provider can outlive a scheduler deadline even when individual HTTP operations have timeouts. Add a bounded process deadline/watchdog using corrected owned-tree termination; preserve partial durable results and lease recovery. A liveness check must not silently treat a stuck provider as productive work.
## Conditional Requirements
### C01. Authenticated Git Needs a POSIX Askpass Helper
**Applies when:** Git requests credentials, including relevant Git/HuggingFace/package paths.
`app/scanner.py:11426-11433` unconditionally creates a Windows `git-askpass.cmd` using batch syntax when `TRUF_GIT_TOKEN` is set. Callers include `app/scanner.py:12384-12388,12600-12605`. Anonymous clones can hide the defect.
Provide a POSIX helper with a correct interpreter/shebang, LF and private executable permissions; retain the Windows branch. Read the token from the controlled environment, not a credential-bearing URL or argv. If the work volume is `noexec`, a prepackaged trusted helper outside that scratch volume is preferable to weakening the whole volume. Coordinate this with executable hardening and immutable code policy. Test using fake credentials and a local/mocked Git interaction.
### C02. Dashboard Publication Requires an Explicit Security Design
**Applies when:** the UI must be accessed from outside the runtime container.
`app/.streamlit/config.toml:1-4` sets `127.0.0.1:5000`; `app/supervisor.py:3608-3624` and `app/dashboard.py:2528-2538` independently reject non-loopback hosts. Dashboard launch also requires authenticated supervisor-child context. Changing only Streamlit configuration or publishing a Docker port will not make the in-container loopback listener reachable.
Either add an explicit secured container-bind mode in both guards, or use a proxy/tunnel in the **same network namespace** that can reach the existing loopback listener. A normal separate bridge-network proxy cannot reach it. Add access control and TLS at the appropriate boundary; read-only database access does not make scan/credential observability safe for public exposure. Preserve the supervised launch contract.
Keep the control interface `127.0.0.1:8765` private (`app/supervisor.py:3726-3733`; `app/config.yaml:182-183`). Use authenticated bootstrap commands such as `--cmd status`, `--cmd shutdown` and `--attach` through `docker exec` in the same container and UID. Do not publish port 8765 or broadly remove loopback restrictions. This entire UI exposure change can be deferred by using `--no-dashboard`.
### C03. Restricted Egress, Proxies, Custom CA and IPv6 Need Explicit Support
**Applies when:** deployment cannot use the existing direct outbound network behavior.
- `app/scanner.py:374-375,560,11397` uses different routing for discovery and downloads/native Git/TruffleHog. Some paths deliberately remove proxy environment variables or use direct clients. Configured download-proxy flags do not themselves implement that routing. `HTTP_PROXY` alone is not enough for a proxy-only deployment.
- `app/scanner.py:480-484` accepts proxy formats that differ from `app/keycheckers/keycheck_common.py:2200`; OpenAI/Gemini/OpenRouter also have duplicated parsers. Unify or explicitly constrain all formats and fallback behavior, including escaped credentials and ambient environment proxies. Add PySocks/`requests[socks]` only if SOCKS is required; it is not currently declared.
- `app/scanner.py:13932` ignores ambient CA settings on the downloader path. Plumb the trusted CA explicitly for corporate interception/custom trust instead of disabling verification. `app/scanner.py:105` forces IPv4 by default; consider `SCANNER_FORCE_IPV4=0` only if the target network needs IPv6 and the path is tested.
- Allow the selected sources' API, registry/CDN, download and redirect destinations, plus configured provider/resolver endpoints. DNS/private-address protections can reject destinations (`app/scanner.py:9340-9389,9416`). Preserve SSRF safeguards while making any intended private-network exception explicit. Provider resolution may contact DeepSeek/Z.ai/Qwen/Kimi depending on configured order (`app/keycheckers/provider_resolution.py:64-156`).
Use mocks/local fixtures for proxy, TLS, redirect and DNS tests. Do not use recovered credentials to test network readiness.
### C04. Offline Credential Writeback Needs a Different Mount Contract
**Applies when:** `sync_alive_github_tokens.py` will update the canonical secrets file.
`app/sync_alive_github_tokens.py:120-138,162-175,257-262` requires verified stopped authority, an adjacent lock, a same-directory private temporary file and atomic replacement. A read-only secret can be suitable for normal runtime but not for this maintenance operation. A single-file bind mount also cannot be assumed to support replacement of its mountpoint.
Choose a separate external/offline rotation workflow or a private writable containing directory for this maintenance mode. Preserve atomic publication and exact canonical path checks. Do not make all application code/secrets permanently writable merely to support an optional operation.
### C05. Multiple Replicas or Split Workers Require New Coordination
**Applies when:** scaling the runtime or moving authenticated workers into separate containers.
`app/runtime_security.py:502` uses filesystem-scoped locks. `app/lifecycle_authority.py:641` relies on local process verification, metadata and control reachability. `app/jsonl_projector.py:140-153` has both a singleton database lease and a file lock; ingester/projector are not arbitrary scalable workers.
Local lock paths in separate container filesystems do not provide a cross-container exclusion guarantee. Worker authentication also does not become remote authentication simply because a directory is mounted. A scale-out design needs explicit shared/distributed fencing, control/identity transport, data ownership and volume semantics. Until then, use one owner and one runtime replica, prevent the old host runtime from remaining active, and do not suggest `docker compose --scale` as an operational option.
### C06. Decide Whether Windows Archive-Name Rules Remain Policy
**Applies when:** Linux deployments should accept archive entries valid on POSIX but invalid on Windows.
`app/scanner.py:13760-13775` still rejects Windows reserved names, colons and trailing dot/space on Linux. This is a policy limitation, not an unconditional container boot defect. Either document it unchanged or separate OS-specific name restrictions. Keep traversal, entry-type, size and expansion-budget protections intact.
### C07. Extend the Manifest if First-Party Linux Native Modules Are Added
**Applies when:** native `.so` application modules are introduced inside the authenticated application tree.
`app/lifecycle_authority.py:21` includes `.pyd` but not Linux `.so` in application import suffixes. Add the appropriate native extension suffix policy and tests when such first-party modules exist. This is not a reason to add every installed dependency to the current application-code manifest or to block the present pure-Python application solely on this basis.
## What Is Not Required
- No Docker daemon, Docker socket mount, Docker CLI, DinD, privileged container or image-architecture emulation is needed for the inspected DockerHub scanning path. It reads image content rather than executing the image (`app/scanner.py:12980,13645`).
- `DOCKER_CONFIG` is an authentication input, not a need for Docker Desktop. Recovery intentionally rejects implicit keychain/helper assumptions; preserve the managed credential pool (`app/scanner.py:4790,13433-13456`).
- npm/PyPI content is scan data. Node.js and a browser are not runtime dependencies of these scanner/keychecker paths. AWS CLI, `gcloud` and `az` are not required merely because those providers are checked.
- Go is not needed in the final image when supplying a compatible prebuilt TruffleHog. GCP keychecker RSA handling does not establish a dependency on Google SDK/cryptography/openssl CLI (`app/keycheckers/gcp/gcpKeycheck.py:288`).
- Existing POSIX `/proc` identity support, `flock`, PostgreSQL command branches, activation/STOPPING states, leases and owned-versus-observed database semantics should be preserved and completed, not rewritten wholesale.
- A general path-case rename or repository-wide CRLF rewrite was not justified. Fix genuinely platform-specific helpers and configured paths instead.
## Documentation and Operational Corrections
Create a deployment Compose definition separate from the recovery fixture, an allowlisted image build, a non-secret Linux configuration example, an ownership/volume provisioning procedure, and a backup/restore/upgrade runbook. These are missing deployment deliverables, not files generated by this audit.
Document the exact source profile and foreground lifecycle, explicit health semantics, stop/restart deadlines, singleton restriction, volume classes, secret maintenance, pinned versions and supported architecture. Remove Windows freeze-counter/diagnostic scripts from the Linux launch chain (`start_freeze_counters.ps1:27`); keeping a script as inert manifest data is different from executing it.
Correct any assumption that provider checks are free/read-only readiness probes. `app/KEYCHECKERS.md:44,87,109` must be reconciled with actual provider defaults. Qwen and several other providers can perform generation by default (`app/keycheckers/qwen/qwenKeycheck.py:544`); AWS/Replicate/Azure paths can probe IAM, resources or RBAC, with additional optional model requests. TruffleHog `no-verification` does not disable the separate keychecker subsystem (`app/config.yaml:136`). Healthchecks and image smoke tests must not invoke those real credential checks.
## Recommended Implementation Order
1. Choose Linux distribution/CPU architecture, PostgreSQL topology, UI requirement, source set, egress policy, non-root UID and storage/resource budgets. Confirm which existing data is authoritative and define rollback.
2. Fix portable paths, POSIX identity/containment, signal/restart behavior and executable/ACL policy. Add focused offline Linux regression tests while preserving the current Windows contracts.
3. Assemble the pinned dependency/native-tool image and code-authority layout. Add `.dockerignore`, a foreground entrypoint contract, private provisioning and a separate deployment Compose definition. Keep one runtime replica.
4. Build and exercise a disposable Linux environment with fake credentials and local fixtures. Validate process lifetime, shutdown, restart, permissions, imports, native CLI contracts, health and resource limits before touching real data.
5. Implement the chosen PostgreSQL mode. Rehearse fresh initialization and a restored dataset, schema/cutover migration, bundle/projection reconciliation and coordinated recovery. Embedded identity binding needs a stopped target PostgreSQL; SQL migration needs PostgreSQL available with normal workers stopped.
6. Complete the conditional features actually needed: authenticated Git, secured UI, restricted-network support or credential maintenance. Defer unrelated scale-out work.
7. Perform a controlled real-data cutover only after backup/restore rehearsal, exclusive ownership and rollback checks. Start with conservative concurrency and verify health/backlog behavior before increasing load.
## Acceptance Tests
| Area | Required evidence before claiming support |
| --- | --- |
| Build and ABI | Build on each supported target architecture; resolved dependency check; stdlib/native imports through the intended isolated interpreter; TruffleHog/Git help/version and library compatibility |
| Authority image layout | No rejected application bytecode/symlinks; required manifest files/assets present; stable code/config hashes; dependencies outside the application tree |
| Paths and permissions | Linux path resolution for every configured directory/file; non-root fresh-volume provisioning; correct private owners/modes; executable bits survive hardening; usable real lock root |
| Process identity | Real POSIX host/payload handshake can create the scanner owner marker; exact identity survives normal lifecycle checks |
| Process containment | Nested provider/native grandchildren terminate on source restart, parent death and failed startup; no orphan payloads or unreaped zombies |
| Container lifecycle | Foreground no-TTY autostart; SIGTERM during active work; confirmed drain/stop; bounded ordinary shutdown; explicit failed-hold escalation; restart after reused PID/stale metadata |
| Database | Fresh schema and valid cutover; wrong cluster rejected; offline identity rebinding; authenticated outage/recovery; external mode, if chosen, has no local PG start/stop dependency |
| Durability | Container recreation preserves reservations/bundles/publications; interrupted atomic publication recovers safely; coordinated PG-plus-bundle backup can actually be restored |
| Limits | Full disk, low free-space floor, bounded backlog, constrained CPU/RAM/PIDs, OOM exit status, bonus-slot denial and provider process deadline |
| Network | Mocked direct/proxy/SOCKS-as-needed, parser formats, CA, IPv4/IPv6-as-needed, redirects and DNS/SSRF behavior |
| Git | POSIX authenticated askpass with fake credentials, private executable permissions and the selected `noexec` work-volume arrangement |
| Archives | gzip/zstd/PAX, malformed archives/missing decoders, entry-name policy and traversal/size protections on the chosen patched CPython |
| Optional UI | Reachable only by the intended secured path; supervised authentication intact; health distinguished from full pipeline readiness; control port not published |
| Safe probes | No provider generation, credential validation or production endpoint activity from build/health tests |
Existing tests to extend/select carefully:
- `tests/test_owned_process.py:79-82,98`: the identity assertion covers PID, and tree cleanup coverage is Windows-specific.
- `tests/test_temp_owner_child_safety.py:49-59`: a mock supplies complete identity and can hide the real POSIX handshake defect.
- `tests/test_supervisor_safety.py:978`: a mocked zero exit code can hide the foreground completion gap.
- `tests/test_pipeline_postgres_integration.py:69`: adapt `.exe` assumptions to a disposable Linux PostgreSQL setup, not the authoritative host database.
- `tests/test_runtime_security.py:227`: add actual container UID/mount/symlink/executable-policy cases.
- `tests/test_api_proxy_routing.py:62,92,215,248`: extend mocked routing, proxy-parser and trust behavior.
- `tests/test_docker_codec_recovery.py:72,130,149,175` and `tests/test_resource_lifecycle_fixes.py:191`: extend codec and cgroup/admission coverage.
- Some tests exercise installed/native/live paths, including `tests/test_docker_codec_recovery.py:666` and `tests/test_huggingface_long_paths.py:324`. Separate offline tests from explicitly opted-in integration tests; do not run the entire suite against existing data or credentials by default.
## Verification Performed and Limits
- Inspected the application, configuration, entrypoints, dependency manifests, security/process/DB/result-pipeline code, recovery Compose and relevant tests. Findings cite inspected file/line locations; line numbers may move with later edits.
- Reproduced the Windows-path/config-precedence failure using the actual pure path functions under POSIX path semantics, without application startup or filesystem mutation.
- Parsed all 61 application Python files with Python 3.12.3 using AST-only analysis: zero syntax errors. This is not an import, dependency, Linux execution or behavior test.
- Docker CLI was unavailable in this session: `docker version --format '{{json .Server}}'` failed because the command was not found. This does not prove that the host has no Docker installation or can never run containers.
- No Docker build, Compose deployment, Linux process integration test or full pytest suite was run. No supervisor, scanner, provider checker or PostgreSQL server was started. Secret contents, real credential checks and database migrations were not used for verification.
- Only this report was added. The findings identify the correction surface visible from repository inspection; target-platform tests may reveal additional issues. No claim of Docker readiness is made until the acceptance checks pass.
+368
View File
@@ -0,0 +1,368 @@
FROM python:3.12-slim-bookworm@sha256:782412e85d0f0984994c290652577d4018aff08145c85b262bb63dc0c7522254 AS python-base
ENV PATH=/usr/local/bin:/usr/bin:/bin:/usr/lib/postgresql/16/bin \
HOME=/data/home \
LANG=C.UTF-8 \
LC_ALL=C.UTF-8 \
PYTHONDONTWRITEBYTECODE=1
RUN /usr/local/bin/python3 -I -S -B -c "import sys; assert sys.version_info[:3] == (3, 12, 14), sys.version"
FROM python-base AS lock-generator
COPY docker/build-dependencies/requirements.lock /tmp/compiler.lock
RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \
--require-hashes --only-binary=:all: --no-compile --no-cache-dir \
-r /tmp/compiler.lock \
&& rm /tmp/compiler.lock
WORKDIR /src
COPY app/requirements.txt app/requirements-keycheckers.txt ./app/
COPY docker/requirements.in docker/requirements.lock docker/requirements-test.in docker/requirements-test.lock docker/requirements-worker.in docker/requirements-worker.lock ./docker/
COPY docker/build-dependencies/requirements.in docker/build-dependencies/requirements.lock ./docker/build-dependencies/
ENV CUSTOM_COMPILE_COMMAND="See docker/build-dependencies/README.md for the pinned Python 3.12.14 pip-tools generation command."
CMD ["python3", "-m", "piptools", "compile", "--generate-hashes", "--allow-unsafe", "--resolver=backtracking", "--strip-extras", "--no-emit-index-url", "--no-emit-trusted-host", "--index-url=https://pypi.org/simple", "--pip-args=--only-binary=:all:", "--output-file=docker/requirements.lock", "docker/requirements.in"]
FROM python-base AS trufflehog-download
ARG TARGETARCH
RUN python3 -I -S -B - "$TARGETARCH" <<'PY'
import hashlib
import os
import shutil
import sys
import tarfile
import urllib.request
checksums = {
"amd64": "dc24007c2f233bd61c05beabeb44aa27ea9b43288166279209abe0458c5ce76b",
"arm64": "7e65e771d2a247964056aa5edba0f8ae3945895e5dce867fe0ffbc7b0128239a",
}
architecture = sys.argv[1]
if architecture not in checksums:
raise SystemExit("TruffleHog is pinned only for linux/amd64 and linux/arm64")
name = f"trufflehog_3.97.4_linux_{architecture}.tar.gz"
url = "https://github.com/trufflesecurity/trufflehog/releases/download/v3.97.4/" + name
digest = hashlib.sha256()
size = 0
with urllib.request.urlopen(url, timeout=60) as response, open("/tmp/trufflehog.tar.gz", "wb") as output:
while chunk := response.read(1024 * 1024):
digest.update(chunk)
size += len(chunk)
output.write(chunk)
if digest.hexdigest() != checksums[architecture]:
raise SystemExit("TruffleHog archive SHA-256 mismatch")
if architecture == "amd64" and size != 34970205:
raise SystemExit("TruffleHog amd64 archive length mismatch")
with tarfile.open("/tmp/trufflehog.tar.gz", "r:gz") as archive:
member = archive.getmember("trufflehog")
if not member.isfile():
raise SystemExit("TruffleHog archive executable must be a regular file")
with archive.extractfile(member) as source, open("/trufflehog", "wb") as output:
shutil.copyfileobj(source, output)
os.chmod("/trufflehog", 0o755)
os.unlink("/tmp/trufflehog.tar.gz")
print(f"Verified {name}: {size} bytes, sha256:{digest.hexdigest()}")
PY
FROM python-base AS worker-dependencies
COPY docker/requirements-worker.lock /tmp/requirements-worker.lock
RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \
--require-hashes --only-binary=:all: --no-compile --no-cache-dir \
--target /worker-dependencies -r /tmp/requirements-worker.lock \
&& PYTHONPATH=/worker-dependencies python3 -I -S -B - <<'PY'
import sys
sys.path.insert(0, "/worker-dependencies")
import requests
import yaml
import zstandard
PY
RUN rm /tmp/requirements-worker.lock
FROM python-base AS worker-native-dependencies
RUN <<'SH'
set -eu
rm -f /etc/apt/sources.list /etc/apt/sources.list.d/debian.sources
printf '%s\n' \
'Types: deb' \
'URIs: https://snapshot.debian.org/archive/debian/20260914T000000Z/' \
'Suites: bookworm bookworm-updates' \
'Components: main' \
'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \
'Check-Valid-Until: no' \
'' \
'Types: deb' \
'URIs: https://snapshot.debian.org/archive/debian-security/20260914T000000Z/' \
'Suites: bookworm-security' \
'Components: main' \
'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \
'Check-Valid-Until: no' \
> /etc/apt/sources.list.d/debian.sources
export DEBIAN_FRONTEND=noninteractive
apt-get -o Acquire::Retries=3 -o Acquire::https::Timeout=30 -o APT::Update::Error-Mode=any update
apt-get install -y --no-install-recommends \
ca-certificates=20250419~deb12u1 \
git=1:2.39.5-0+deb12u3 \
tini=0.19.0-1+b3
install -d -o 10001 -g 10001 -m 0700 /data /data/home
/usr/sbin/groupadd --gid 10001 truf
/usr/sbin/useradd --uid 10001 --gid 10001 --no-create-home --home-dir /data/home --shell /usr/sbin/nologin truf
install -d -o 0 -g 0 -m 0755 /worker-git/bin /worker-git/libexec /worker-git/share
cp -aL /usr/bin/git /worker-git/bin/git
cp -aL /usr/lib/git-core /worker-git/libexec/git-core
cp -aL /usr/share/git-core /worker-git/share/git-core
find /worker-git -type d -exec chmod 0755 {} +
find /worker-git -type f -exec chmod go-w {} +
test -x /worker-git/bin/git
test -x /worker-git/libexec/git-core/git-remote-https
rm -rf /var/lib/apt/lists/* /var/cache/apt/archives/* /var/log/apt/*
SH
COPY --from=trufflehog-download --chown=0:0 --chmod=0755 /trufflehog /usr/local/bin/trufflehog
FROM worker-native-dependencies AS worker-package-build
ARG TARGETARCH
COPY --from=worker-dependencies --chown=0:0 /worker-dependencies /build/app/dependencies
COPY app/ /build/app/
COPY docs/remote-worker-quickstart-ru.md /build/README_RU.md
COPY docs/remote-worker-cheatsheet-windows-ru.md /build/
COPY docs/remote-worker-cheatsheet-linux-ru.md /build/
COPY docs/remote-worker-cheatsheet-docker-ru.md /build/
COPY docker/worker-package-pins.json /build/worker-package-pins.json
RUN case "$TARGETARCH" in \
amd64) platform_tag=linux-x86_64 ;; \
arm64) platform_tag=linux-aarch64 ;; \
*) echo "unsupported worker architecture" >&2; exit 1 ;; \
esac \
&& python3 -u -I -S -B /build/app/worker_package_builder.py assemble \
--output /opt/truf-worker \
--source-app /build/app \
--dependencies /build/app/dependencies \
--detector-policy /build/app/trufflehog-custom-detectors.yaml \
--trufflehog /usr/local/bin/trufflehog \
--git-root /worker-git \
--git-executable bin/git \
--platform-tag "$platform_tag" \
--operator-readme /build/README_RU.md \
--operator-cheatsheet /build/remote-worker-cheatsheet-windows-ru.md \
--operator-cheatsheet /build/remote-worker-cheatsheet-linux-ru.md \
--operator-cheatsheet /build/remote-worker-cheatsheet-docker-ru.md \
--build-inputs /build/worker-package-pins.json
FROM worker-native-dependencies AS worker
ENV GIT_EXEC_PATH=/opt/truf-worker/runtime/git/libexec/git-core \
GIT_TEMPLATE_DIR=/opt/truf-worker/runtime/git/share/git-core/templates
COPY --from=worker-package-build --chown=0:0 /opt/truf-worker /opt/truf-worker
RUN chown -R 10001:10001 /opt/truf-worker/app \
&& find /opt/truf-worker/app -type d -exec chmod 0700 {} + \
&& find /opt/truf-worker/app -type f -exec chmod 0600 {} + \
&& chown 10001:10001 /opt/truf-worker/worker-package.json \
&& chmod 0600 /opt/truf-worker/worker-package.json \
&& find /opt/truf-worker/bin /opt/truf-worker/runtime -type d -exec chmod 0755 {} + \
&& find /opt/truf-worker/bin /opt/truf-worker/runtime -type f -exec chmod go-w {} + \
&& test ! -e /opt/truf-worker/app/keycheck_runner.py \
&& test ! -d /opt/truf-worker/app/keycheckers \
&& test ! -e /usr/lib/postgresql \
&& test ! -e /usr/bin/psql
USER 10001:10001
RUN /opt/truf-worker/bin/trufflehog --version >/dev/null \
&& /opt/truf-worker/runtime/git/bin/git --version >/dev/null \
&& /usr/local/bin/python3 -I -S -B - <<'PY'
import sys
sys.path[:0] = ['/opt/truf-worker/app', '/opt/truf-worker/app/dependencies']
from importlib.util import find_spec
from worker_cli import parse_args
from worker_package import verify_worker_package
package = verify_worker_package('/opt/truf-worker/worker-package.json')
assert package['manifest']['schema'] == 3
assert package['manifest']['protocol_version'] == 2
assert {
(item['source'], item['platform'], item['planning_kind'])
for item in package['manifest']['capabilities']
} == {
('gitlab', 'gitlab', 'exact_git_v1'),
('dockerhub', 'docker', 'docker_direct_v1'),
('huggingface', 'huggingface', 'huggingface_space_v1'),
}
assert set(package['runtime_trees']) == {'git'}
assert parse_args(['run', '--server', 'https://worker.example', '--token', 'x' * 32]).command == 'run'
assert all(find_spec(name) is None for name in ('httpx', 'psycopg', 'starlette', 'streamlit'))
PY
WORKDIR /data
ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/python3", "-u", "-I", "-S", "-B", "/opt/truf-worker/app/remote_worker_bootstrap.py", "--"]
CMD ["run"]
FROM python-base AS native-dependencies
ADD --checksum=sha256:0144068502a1eddd2a0280ede10ef607d1ec592ce819940991203941564e8e76 https://www.postgresql.org/media/keys/ACCC4CF8.asc /usr/share/keyrings/postgresql.asc
RUN <<'SH'
set -eu
chmod 0644 /usr/share/keyrings/postgresql.asc
rm -f /etc/apt/sources.list /etc/apt/sources.list.d/debian.sources
printf '%s\n' \
'Types: deb' \
'URIs: https://snapshot.debian.org/archive/debian/20260914T000000Z/' \
'Suites: bookworm bookworm-updates' \
'Components: main' \
'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \
'Check-Valid-Until: no' \
'' \
'Types: deb' \
'URIs: https://snapshot.debian.org/archive/debian-security/20260914T000000Z/' \
'Suites: bookworm-security' \
'Components: main' \
'Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg' \
'Check-Valid-Until: no' \
> /etc/apt/sources.list.d/debian.sources
printf '%s\n' \
'deb [signed-by=/usr/share/keyrings/postgresql.asc] https://apt-archive.postgresql.org/pub/repos/apt bookworm-pgdg-archive main' \
> /etc/apt/sources.list.d/postgresql.list
# Only these exact PGDG packages may supplement the immutable Debian snapshot.
printf '%s\n' \
'Package: postgresql-16 postgresql-client-16 libpq5' \
'Pin: version 16.15-1.pgdg12+2' \
'Pin-Priority: 1001' \
'' \
'Package: postgresql-common postgresql-client-common' \
'Pin: version 293.pgdg12+1' \
'Pin-Priority: 1001' \
'' \
'Package: *' \
'Pin: origin apt-archive.postgresql.org' \
'Pin-Priority: -1' \
> /etc/apt/preferences.d/postgresql
printf '#!/bin/sh\nexit 101\n' > /usr/sbin/policy-rc.d
chmod 0755 /usr/sbin/policy-rc.d
export DEBIAN_FRONTEND=noninteractive
apt-get -o Acquire::Retries=3 -o Acquire::https::Timeout=30 -o APT::Update::Error-Mode=any update
apt-get install -y --no-install-recommends \
postgresql-common=293.pgdg12+1 \
postgresql-client-common=293.pgdg12+1
# Set this after common is installed but before installing any server package.
printf '\ncreate_main_cluster = false\n' >> /etc/postgresql-common/createcluster.conf
apt-get install -y --no-install-recommends \
ca-certificates=20250419~deb12u1 \
git=1:2.39.5-0+deb12u3 \
tini=0.19.0-1+b3 \
postgresql-16=16.15-1.pgdg12+2 \
postgresql-client-16=16.15-1.pgdg12+2 \
libpq5=16.15-1.pgdg12+2
test ! -d /var/lib/postgresql/16/main
rm -f /etc/ssl/private/ssl-cert-snakeoil.key /etc/ssl/certs/ssl-cert-snakeoil.pem
rm -rf /var/lib/apt/lists/* /var/cache/apt/archives/* /var/log/apt/*
/usr/sbin/groupadd --gid 10001 truf
/usr/sbin/useradd --uid 10001 --gid 10001 --no-create-home --home-dir /data/home --shell /usr/sbin/nologin truf
install -d -o 10001 -g 10001 -m 0700 /data /data/home
SH
RUN python3 -I -S -B - <<'PY'
import os
from pathlib import Path
# Keep the package's complete Git helper tree; regular hard links preserve argv[0].
for directory in (Path("/usr/local/bin"), Path("/usr/lib/git-core")):
for path in directory.iterdir():
if path.is_symlink() and os.access(path, os.X_OK):
target = path.resolve(strict=True)
if not target.is_file() or target.stat().st_uid != 0:
raise SystemExit(f"Untrusted executable target: {path}")
path.unlink()
os.link(target, path)
# Prefer native PG16 clients over the distribution's symlinked version wrappers.
for target in Path("/usr/lib/postgresql/16/bin").iterdir():
path = Path("/usr/bin") / target.name
if path.is_symlink():
path.unlink()
os.link(target, path)
for name in ("/usr/local/bin/python3", "/usr/bin/git", "/usr/lib/git-core/git-remote-https", "/usr/bin/tini"):
path = Path(name)
details = path.lstat()
if path.is_symlink() or not path.is_file() or details.st_uid != 0 or details.st_mode & 0o022:
raise SystemExit(f"Untrusted native executable: {path}")
PY
FROM native-dependencies AS dependencies
COPY docker/requirements.lock /tmp/requirements.lock
RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \
--require-hashes --only-binary=:all: --no-compile --no-cache-dir \
-r /tmp/requirements.lock \
&& python3 -m pip --isolated check \
&& rm /tmp/requirements.lock
USER 10001:10001
# Both public targets inherit these exact runtime contents; the default stays runtime.
FROM dependencies AS runtime-base
USER 0:0
RUN install -d -o 10001 -g 10001 -m 0700 /opt/truf /opt/truf/app /opt/truf/tests
USER 10001:10001
COPY --chown=10001:10001 app/ /opt/truf/app/
RUN python3 -I -S -B - <<'PY'
import os
import stat
for directory, directories, files in os.walk("/opt/truf/app", followlinks=False):
for path in [directory, *(os.path.join(directory, name) for name in directories + files)]:
details = os.lstat(path)
if not (stat.S_ISDIR(details.st_mode) or stat.S_ISREG(details.st_mode)):
raise SystemExit(f"Application links/special files are forbidden: {path}")
if os.path.basename(path) == "__pycache__" or path.endswith((".pyc", ".pyo")):
raise SystemExit(f"Application bytecode is forbidden: {path}")
if (details.st_uid, details.st_gid) != (10001, 10001):
raise SystemExit(f"Application ownership mismatch: {path}")
os.chmod(path, 0o700 if stat.S_ISDIR(details.st_mode) else 0o600)
PY
WORKDIR /opt/truf/app
ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/python3", "-u", "-I", "-S", "-B", "/opt/truf/app/container_runtime.py"]
CMD ["run"]
FROM runtime-base AS test
USER 0:0
COPY --from=trufflehog-download --chown=0:0 --chmod=0755 /trufflehog /usr/local/bin/trufflehog
COPY docker/requirements-test.lock /tmp/requirements-test.lock
RUN python3 -m pip --isolated install --index-url=https://pypi.org/simple \
--require-hashes --only-binary=:all: --no-compile --no-cache-dir \
-r /tmp/requirements-test.lock \
&& python3 -m pip --isolated check \
&& rm /tmp/requirements-test.lock
USER 10001:10001
COPY --chown=10001:10001 tests/ /opt/truf/tests/
COPY --chown=10001:10001 --chmod=0600 .dockerignore /opt/truf/.dockerignore
COPY --chown=10001:10001 --chmod=0600 Dockerfile /opt/truf/Dockerfile
COPY --chown=10001:10001 --chmod=0600 start_runtime.ps1 start_core_runtime.ps1 stop_runtime.ps1 /opt/truf/
COPY --chown=10001:10001 --chmod=0600 compose.yaml compose.edge.yaml /opt/truf/
COPY --chown=10001:10001 deploy/ /opt/truf/deploy/
COPY --chown=10001:10001 docker/ /opt/truf/docker/
RUN python3 -I -S -B - <<'PY'
import os
from pathlib import Path
import stat
edge_e2e = {
path.name for path in Path('/opt/truf/tests').glob('edge_e2e_*.py')
}
if edge_e2e != {'edge_e2e_backend.py', 'edge_e2e_client.py'}:
raise SystemExit(f'Unexpected edge E2E test-stage inputs: {sorted(edge_e2e)}')
for directory, directories, files in os.walk("/opt/truf/tests", followlinks=False):
for path in [directory, *(os.path.join(directory, name) for name in directories + files)]:
details = os.lstat(path)
if not (stat.S_ISDIR(details.st_mode) or stat.S_ISREG(details.st_mode)):
raise SystemExit(f"Test links/special files are forbidden: {path}")
if os.path.basename(path) == "__pycache__" or path.endswith((".pyc", ".pyo")):
raise SystemExit(f"Test bytecode is forbidden: {path}")
if (details.st_uid, details.st_gid) != (10001, 10001):
raise SystemExit(f"Test ownership mismatch: {path}")
os.chmod(path, 0o700 if stat.S_ISDIR(details.st_mode) else 0o600)
PY
WORKDIR /opt/truf
ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/python3", "-u", "-I", "-S", "-B", "/opt/truf/tests/container_unit.py"]
CMD []
FROM runtime-base AS runtime
+102
View File
@@ -0,0 +1,102 @@
# Pipeline Quarantine Audit
Initial snapshot and remediation: `2026-08-15`
## Current Impact
- PostgreSQL quarantine rows: `177`
- Accounted capacity: `4354 items / 1,817,503,389 bytes`
- Configured admission limit: `10000 items / 1,073,741,824 bytes`
- New scan admission is closed because the byte limit is exceeded.
- `49,215` admission intents have already ended with `quarantine_admission_closed`.
`pipeline: ready` means that PostgreSQL and workers are healthy. It does not mean that new scan admission is open.
## Result Bundles
Two rows account for `4004 items / 1,212,153,856 bytes`.
| Quarantine ID | Source | Original error | Physical size | Read-only validation now |
|---|---|---|---:|---|
| `29` | Hugging Face `spaces` | Transient Windows `Permission denied` | `2,870 B` | Valid; 3 frames, no findings/errors/candidates |
| `91` | Docker Hub query `tokenizer` | Transient Windows `Permission denied` | `5,354 B` | Valid; 8 frames, 1 finding, 4 errors, no candidates |
The bundle contents are valid. Their large capacity cost comes from worst-case reservations transferred into quarantine, not their physical file sizes.
Current code now reports a temporarily unavailable private bundle as an availability error. Result ingester defers it instead of classifying it as invalid content.
Recommendation: recover both bundles through an audited offline path rather than discard them. ID `91` contains one finding and must not be deleted without an explicit decision.
## Keycheck Quarantine
There are `175` pending keycheck quarantine rows.
| Class | Rows | Distinct credentials | Assessment |
|---|---:|---:|---|
| Repeated DeepSeek unconsumed rechecks | `104` | `1` | Duplicate hourly retries; current state is now `NO_CONTEXT` |
| DeepSeek unconsumed findings | `15` | `6` | Legitimate historical candidates filtered by routing |
| Azure Foundry unconsumed | `15` | `12` | Historical provider-consumption issue |
| Hugging Face unconsumed | `7` | `5` | Historical provider-consumption issue |
| Replicate unconsumed | `6` | `3` | Historical provider-consumption issue |
| xAI unconsumed | `3` | `3` | Historical provider-consumption issue |
| DeepSeek route mismatch | `24` | `14` | Misrouted non-DeepSeek detectors; source findings remain in PostgreSQL |
| Azure route mismatch | `1` | `1` | Historical Azure Foundry route mismatch; current state is `UNKNOWN` |
The active growth came from one DeepSeek credential whose current state was `NETWORK`. Hourly network retry created a fresh candidate, provider routing silently rejected it, and the candidate entered quarantine after three unconsumed attempts.
## Preventive Changes
- Non-DeepSeek routing decisions now complete as `NO_CONTEXT` instead of leaving a leased candidate unconsumed.
- DeepSeek has a canonical `deepseekNoContext.txt` status projection.
- Recheck generation now refuses to enqueue a credential while it has an unresolved `provider_candidate_unconsumed` or `candidate_provider_route_mismatch` quarantine row.
- Temporarily unavailable result bundle files are deferred instead of quarantined as validation failures.
Verification:
- Targeted tests: `60 passed`.
- Manual `recheck deepseek network`: code `0`, no candidates processed.
- DeepSeek unconsumed quarantine remained exactly `119`; no new row was created.
## Approved Remediation
1. Stop sources and acquire the offline migration guard.
2. Recover bundle IDs `29` and `91` through deterministic re-ingestion.
3. Discard the `104` duplicate DeepSeek retry rows after exact manifest review.
4. Discard the `24` confirmed DeepSeek route-mismatch rows after exact manifest review.
5. Requeue one current candidate per distinct credential from the remaining legitimate unconsumed groups; keep source attribution.
6. Review the single Azure route mismatch separately.
7. Resolve superseded duplicate candidate rows with an audited discard manifest.
8. Verify physical artifacts, capacity accounting, projections and reopened scan admission before restarting sources.
All destructive decisions must use an exact private `truf-pipeline-quarantine-review-v1` manifest containing quarantine ID, reason code, payload hash and action. The offline review command requires both `--apply` and `--sources-stopped`.
## Applied Result
The user approved recovery plus selective cleanup.
- Manifest: `runtime/control/quarantine-remediation-20260815.json`
- Manifest SHA-256: `c3143e6b0e15e07b6afacffd50b449444b9932e75001f633ac82b5b79e21fe3f`
- Reviewed: `177`
- Approved retry: `32`
- Audited discard: `145`
- Duplicate/conflicting reviews: `0`
- Final quarantine capacity: `0 items / 0 bytes`
Bundle outcomes:
| Quarantine ID | Final reservation | Final bundle | Target disposition | Preserved contents |
|---|---|---|---|---|
| `29` | `acknowledged` | `acknowledged` | `done` | Clean empty result |
| `91` | `acknowledged` | `acknowledged` | `deferred` | `1 finding`, `4 errors` |
Keycheck outcomes from the retried unique credentials:
| Service | Final current statuses |
|---|---|
| Azure | `12 FOUNDRY_UNRESOLVED`, `1 UNKNOWN` |
| DeepSeek | `6 NO_CONTEXT` |
| Hugging Face | `5 NO_CONTEXT` |
| Replicate | `3 NO_CONTEXT` |
| xAI | `3 NO_CONTEXT` |
The malformed legacy provider candidates are now terminal current-state records instead of repeatedly deferred/quarantined candidates. Scan admission reopened and scanner workers returned to `3/3` active operation.
+273
View File
@@ -0,0 +1,273 @@
# Truf Runtime Cheatsheet
## Быстрый старт
Команды выполняются из `D:\truf` в PowerShell.
```powershell
# Запустить весь canonical runtime
.\start_runtime.ps1
# Запустить explicit core set с keychecks, без dashboard
.\start_core_runtime.ps1
# Подключиться к интерактивной консоли supervisor
.\attach_runtime.ps1
# Координированно остановить весь runtime и PostgreSQL
.\stop_runtime.ps1
```
Если PowerShell блокирует запуск скриптов:
```powershell
powershell.exe -NoProfile -ExecutionPolicy Bypass -File .\start_runtime.ps1
```
Для полного рестарта используй именно:
```powershell
.\stop_runtime.ps1
.\start_runtime.ps1
```
Для полного рестарта в core-only режиме:
```powershell
.\stop_runtime.ps1
.\start_core_runtime.ps1
```
Не используй `restart all` как замену полному рестарту: pipeline workers защищены от ручного рестарта, пока scanner sources работают.
## Что запускается
| Компонент | Назначение |
|---|---|
| PostgreSQL | Единственный authoritative storage |
| `result-ingester` | Переносит scan bundles в PostgreSQL |
| `jsonl-projector` | Создаёт compatibility JSONL projections |
| `janitor` | Обслуживает runtime queues и временные данные |
| `github` | GitHub scanner loop |
| `gitlab` | GitLab scanner loop |
| `huggingface` | Hugging Face scanner loop |
| `dockerhub` | Docker Hub scanner loop |
| `package_git` | Package/repository scanner loop |
| `keychecks` | Почасовой provider checker scheduler |
Одновременно выполняется максимум `3` scan workers. Dashboard при обычном запуске выключен.
## PowerShell-скрипты
| Скрипт | Что делает |
|---|---|
| `start_runtime.ps1` | Запускает freeze diagnostics, проверяет identity PostgreSQL и поднимает background supervisor |
| `start_core_runtime.ps1` | Поднимает PostgreSQL, pipeline, janitor и три discovery-only producer; dashboard выключен |
| `stop_runtime.ps1` | Выполняет authenticated coordinated shutdown supervisor, children и PostgreSQL |
| `attach_runtime.ps1` | Открывает интерактивную supervisor-консоль; `quit` только отключает консоль |
| `start_freeze_counters.ps1` | Запускает Windows performance counters в `H:\truf-diagnostics` |
| `monitor_runtime_lag.ps1` | Пишет CPU/RAM/disk/runtime lag в CSV |
| `cleanup_stale_agentui_vite.ps1` | Отдельная уборка старых AgentUI/Vite процессов; без `-Apply` только dry run |
| `runtime\check-openrouter-keys.ps1` | Retired; намеренно завершается ошибкой |
В проекте нет собственных `.bat`/`.cmd`. Найденные BAT внутри `runtime\postgres\pgsql\pgAdmin 4` принадлежат pgAdmin и для Truf не используются.
## Full и Core-only режимы
`start_runtime.ps1` использует allowlist `supervisor.enabled_sources` из `config.linux.yaml`. Сейчас этот allowlist уже равен distributed core set, поэтому оба start-скрипта запускают одинаковые discovery producer.
`start_core_runtime.ps1` фиксирует core set прямо в wrapper и не зависит от будущего расширения default allowlist:
```text
gitlab,dockerhub,huggingface
```
PostgreSQL, `result-ingester`, `jsonl-projector`, `janitor` и независимо включённый `keychecks` также запускаются. Dashboard не запускается.
Чтобы сменить режим, сначала останови текущий supervisor через `.\stop_runtime.ps1`, затем запусти нужный start-скрипт. `stop_runtime.ps1` одинаков для обоих режимов.
## Core Sources
| Source | Что производит на сервере |
|---|---|
| `gitlab` | Ищет недавно активные GitLab projects и ставит их в очередь remote workers |
| `dockerhub` | Ищет Docker Hub images и ставит в очередь только immutable `repo@sha256:...` targets |
| `huggingface` | Ищет новейшие Hugging Face Spaces и ставит их в очередь remote workers |
`result-ingester`, `jsonl-projector`, `janitor` и `keychecks` отображаются как отдельные system workers, но не являются discovery sources. GitHub и `package_git` остаются доступными legacy/manual source, однако в distributed core profile не входят.
## Janitor
Janitor обслуживает только scanner work area (`S:\scanner-work`), а не PostgreSQL и не provider status files.
- Каждые `60` секунд ищет временные каталоги разрешённых типов.
- Рассматривает только каталоги старше `7200` секунд.
- Требует приватный `.scanner-owner.json` с точным process identity.
- Удаляет каталог только если owner и parent гарантированно мертвы.
- Не следует по symlink, junction или другим reparse points.
- Один проход ограничен `50` каталогами, `10000` entries, `1 GiB`, `30` секундами и depth `64`.
- Не имеет PostgreSQL credentials и не удаляет findings, keycheck history, current state или найденные секреты.
Примеры диагностики:
```powershell
# Один диагностический замер
.\monitor_runtime_lag.ps1 -Once
# Свой файл и интервал
.\monitor_runtime_lag.ps1 -OutputPath H:\truf-diagnostics\lag.csv -IntervalSeconds 10
# Безопасный просмотр кандидатов на очистку Vite
.\cleanup_stale_agentui_vite.ps1
# Реальная очистка найденного точного набора
.\cleanup_stale_agentui_vite.ps1 -Apply
```
## Supervisor-команды
Сначала запусти `.\attach_runtime.ps1`, затем используй команды ниже.
| Команда | Назначение |
|---|---|
| `help` | Полная встроенная справка |
| `status` | Свежий status table |
| `watch` | Live status; `q` возвращает в prompt |
| `auth <source|all>` | Состояние auth pools |
| `logs <source> [N]` | Последние `N` строк bounded-лога |
| `command <source|all>` | Фактическая child-команда, log и state paths |
| `start <source|all>` | Запустить остановленный source |
| `stop <source|all>` | Остановить и оставить остановленным |
| `restart <source|all>` | Перезапустить отдельный source |
| `pause <source|all>` | Остановить и отметить paused |
| `resume <source|all>` | Снять pause и запустить |
| `once <source|all>` | Один проход source с `--once` |
| `mode <source|all> loop|once|repeat` | Изменить режим source |
| `set <source|all> interval <sec>` | Интервал repeat mode |
| `set <source|all> restart on|off` | Автоматический restart после сбоя |
| `set <source|all> restart_delay <sec>` | Начальная задержка restart |
| `dashboard status|start|stop|restart` | Управление dashboard |
| `shutdown` | Полный coordinated shutdown |
| `quit` | В attach-режиме только отсоединиться |
`reload` намеренно отключён. После изменения `config.yaml` или runtime-кода нужен полный `stop_runtime.ps1` + `start_runtime.ps1`.
Source alias: `docker` означает `dockerhub`.
## Статусы
| Статус | Значение |
|---|---|
| `running` | Child сейчас работает |
| `waiting` | Ожидает следующего запуска/retry |
| `blocked` | Ждёт стабильной готовности PostgreSQL |
| `paused` | Остановлен командой `pause` |
| `done` | Успешный one-shot завершён |
| `failed` | Child завершился с ошибкой, restart выключен |
`desired=running` показывает желаемое состояние. `rs` означает текущую серию ошибок / общее число automatic restarts. Старый `exit=1` рядом с уже `running` source относится к предыдущей попытке запуска.
## Keycheck Recheck
Формат:
```text
recheck <service|all> [type ...] [options]
```
Если type не указан, выполняется полный `--recheck-all` выбранного service.
### Типы
| Type | Что ставится в очередь |
|---|---|
| `network` | Текущие transient network statuses |
| `ratelimited` | Limited/rate-limited и связанные no-balance statuses |
| `unknown` | Unknown и no-context |
| `restricted` | Restricted |
| `nobalance` | No-balance/no-quota |
| `valid` или `alive` | Текущие alive credentials |
| `all` | Все известные credentials |
| `legacy-vertex` | Только GCP: импортировать и проверить отсутствующие legacy Vertex TXT credentials |
### Опции
| Опция | Значение |
|---|---|
| `--force` | Остановить уже работающий keycheck batch и начать этот |
| `--max-keys N` | Ограничить число credentials |
| `--proxy-file PATH` | Временно переопределить proxy file |
| `--no-resource-probe` | Отключить resource probe; сейчас используется Replicate |
| `--no-summary` | Не пересобирать summary/status projections после batch |
`--input PATH` является legacy/offline compatibility option и в canonical PostgreSQL runtime не используется.
### Примеры
```text
# Повторить только network failures у всех providers
recheck all network
# Перепроверить все текущие alive GCP credentials
recheck gcp valid
# Полностью перепроверить Qwen
recheck qwen all
# Проверить максимум 5 alive Replicate без resource probe
recheck replicate valid --max-keys 5 --no-resource-probe
# Импортировать/дедуплицировать старые GCP Vertex TXT записи и проверить только их
recheck gcp legacy-vertex
# Прервать текущий keycheck batch и запустить новый
recheck gcp valid --force
```
Scheduled keychecks запускаются раз в `3600` секунд. По умолчанию проверяются новые candidates и повторяются только `NETWORK`; alive/limited/unknown/restricted/no-balance автоматически каждый час не перепроверяются.
## Текущие GCP Vertex Probes
| Provider | Модели | Locations | Проверка |
|---|---|---|---|
| Google | `gemini-3.6-flash`, `gemini-3.1-pro-preview` | `global`, `us`, `eu` | `countTokens`, без генерации |
| Anthropic | `claude-opus-5`, `claude-opus-4-7`, `claude-opus-4-6`, `claude-fable-5` | `global`, `us`, `eu`, `us-east5`, `europe-west1` | `rawPredict`, до 1 output token |
Anthropic probe является реальным минимальным inference-вызовом и может иметь небольшой расход.
## Куда идут данные
1. Scanner sources создают result bundles.
2. `result-ingester` пишет findings и keycheck candidates в PostgreSQL.
3. Provider checker арендует candidate и выполняет API probe через `runtime\proxy.txt`.
4. Новый результат добавляется в append-only `keycheck_results`.
5. `keycheck_current_state` переключается на последний результат.
6. `jsonl-projector` создаёт compatibility JSONL.
7. Summary projection атомарно обновляет status TXT.
PostgreSQL является source of truth. TXT/JSONL в `runtime\keychecks` являются compatibility projections, а не входом для обычного recheck.
## Полезные пути
| Путь | Назначение |
|---|---|
| `app\config.yaml` | Основная конфигурация runtime, sources и probes |
| `runtime\proxy.txt` | Proxy для provider checks |
| `runtime\logs\supervisor.status.txt` | Последний status snapshot |
| `runtime\logs\supervisor.log` | Supervisor log |
| `runtime\logs\keychecks.log` | Общий keycheck log |
| `runtime\keychecks\summary.tsv` | Текущий provider summary |
| `runtime\keychecks\alive_summary.tsv` | Краткий alive summary |
| `runtime\keychecks\<service>` | Compatibility status/results files provider-а |
| `runtime\control\supervisor.instance.json` | Private control metadata; вручную не редактировать |
| `S:\postgres-data` | Canonical PostgreSQL cluster |
| `S:\scanner-work` | Scanner scratch/work area |
## Безопасность
- Не запускай provider scripts напрямую.
- Не передавай raw credentials через CLI.
- Не редактируй `supervisor.instance.json`.
- Для управления используй только authenticated supervisor.
- Для полного рестарта используй canonical start/stop scripts.
- Не удаляй PostgreSQL cluster или runtime queues вручную.
+285
View File
@@ -0,0 +1,285 @@
# Windows Snapshot Import
## Current Status
As of 2026-09-16, source capture succeeded, but the target import failed during
maintenance cleanup and was manually interrupted. The destination remains
**failed, unmarked and stopped**, not verified-stopped or ready for normal run.
The user then explicitly authorized removal of copied SQLite databases,
backups, `found_secrets` outputs and archival files, preserving Windows originals.
**The staged snapshot is now intentionally incomplete: `files.tar` was deleted.**
Its retained manifest describes the original capture, not the reduced target.
Do not rewrite the manifest, retry import, restart the retained container, or
recapture/repopulate the removed copies automatically. The procedures below
describe the original full-snapshot workflow, not a resume procedure for this
pruned destination. Any future recovery needs a separately reviewed plan.
| Execution evidence | Result |
| --- | --- |
| Source snapshot publication | Manifest published 2026-09-15T19:36:28.010915+00:00; source supervisor/PG stopped flags true |
| Approved manifest SHA-256 | `08344147133c37d4b6f404cf4fac3e59d58f94917f1fa58a77cbb68c36db7e8a` |
| Original capture inventory | 50,501 files, 34,851,776,467 bytes; 49,897 active and 604 archival files; 61 tables and 38 sequences |
| Retained PostgreSQL dump | `database.dump`, 2,619,119,892 bytes; not deleted or modified by cleanup |
| Failed import container | `63286fd554f832fd3a1f073e7c923977e209149e940b684cdbf35e4479ebd5ba`; exited 129, PID 0, restarts 0, restart policy `no` |
| Pinned runtime/cleanup image | `sha256:ecf1ee044fd6e936359a5955e0a42b452b8098a3f9d822272ab373b697761de2` |
| Cleanup verification | 635 original Windows files checked for presence/size and unchanged metadata; original PG control hashes unchanged; Windows `postgres.exe` count 0 |
| Retained destination verification | Metadata of 49,866 remaining inventory files and 1,883 PG files unchanged; PG control/config hashes unchanged; initialized marker absent; application remains stopped |
| Final verified-stopped acceptance | NOT ACHIEVED; cleanup does not repair the failed import |
### Authorized Copy Cleanup
Only these copied locations were removed on 2026-09-16:
| Copied location/family | Files | Bytes removed |
| --- | ---: | ---: |
| `/data/windows-archive` including old SQLite backups and archived configurations | 604 | 10,827,425,254 |
| `/data/runtime-linux/results/scanner*.db` and associated WAL/SHM/journal files | 11 | 10,281,779,360 |
| `/data/runtime-linux/results/found_secrets.*`, including generations, manifest and publication ledger | 20 | 5,584,368,562 |
| Total from native `truf-docker_data` volume | 635 | 26,693,573,176 |
| Completed staging directory's `files.tar` on Windows D: | 1 | 34,917,959,680 |
Original `D:\truf`, `S:\postgres-data` and source bundles were not deleted or
modified. Target PostgreSQL, its dump, translated configuration, credentials,
proxies, queues, other result streams, bundles, caches and their required
publication ledgers were retained. The one-off cleanup used a network-disabled
utility container with only the verified native target volume writable; it did
not run PostgreSQL, the importer, scanners, providers or application services.
The volume gained approximately 26.69 GB of filesystem free space. Approximately
34.92 GB was freed on Windows D:. This did not compact the WSL VHDX on S: or
return all newly free ext4 blocks to the Windows host; S: reported
30,467,690,496 bytes free after cleanup. No WSL/storage reconfiguration was done.
Removing `found_secrets` files does not reset PostgreSQL projector cursors or
pending append/rotation proofs. A future authorized startup must first address
that projection state explicitly; removing files or their SQLite ledger alone
is not a safe live-cursor reset. Do not erase PostgreSQL findings, counters or
pipeline evidence to make the removed files appear consistent.
Reported regression evidence, not rerun by this documentation change: selected
Docker suite **663 passed, 7 Windows-only skipped**; synthetic snapshot tests
**58 passed**; pure config tests **12 passed**; host importer tests **66 passed,
1 POSIX-only skipped**. None is proof of this snapshot's capture or import E2E.
## Scope And Paths
The authorized operation copies the original logical database, proxies, secrets
and reviewed durable files. It does not move/delete the originals, migrate the
source schema, or execute providers, scanners, keycheckers or archived scripts.
Only exclusive PostgreSQL maintenance is allowed during capture/import. Leave
both the original supervisor and original PostgreSQL stopped after capture,
and the destination stopped after import.
Current private staging directory:
- Windows: `D:\truf-docker\docker\imports\windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3`
- WSL: `/mnt/d/truf-docker/docker/imports/windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3`
- Container: the same directory bound read-only at `/import`.
A completed snapshot contains exactly `manifest.json`, `files.tar` and
`database.dump`. Do not add reports or other files inside it. Windows staging
remains private to the capturing account and SYSTEM; preserve its ACLs rather
than making it world-readable for Docker. `docker/imports/` is excluded from
Git and the image build context. Never put dump/tar contents, credentials,
proxy values, application data or raw logs in Git, images or terminal output.
The directory listed above currently retains only `manifest.json` and
`database.dump` after the authorized cleanup. It is not a completed import input.
The destination is the base Compose native Linux volume `truf-docker_data`,
mounted at `/data`, not a Windows bind mount. Paths in braces below are reviewed
families; optional archival inputs are copied only when present.
| Original source | Destination within `/data` |
| --- | --- |
| `S:\postgres-data` via a full PG16 logical dump | `/data/postgres-linux`, independently initialized native Linux PG16 |
| `D:\truf\runtime\{results,queues,state,keychecks,postman_cache,result_spool}` | `/data/runtime-linux/{results,queues,state,keychecks,postman_cache,result_spool}` |
| `D:\truf\runtime\proxy.txt` | `/data/runtime-linux/proxy.txt` |
| `D:\truf\app\{secrets.yaml,trufflehog-custom-detectors.yaml}` | `/data/config/{secrets.yaml,trufflehog-custom-detectors.yaml}` |
| `D:\truf\app\config.yaml` | `/data/windows-archive/app/config.yaml`; translated profile at `/data/config/windows-import.yaml` |
| `S:\scanner-result-bundles\{tmp,ready,quarantine}` | `/data/scanner-result-bundles/{tmp,ready,quarantine}` |
| Reviewed archival inputs under `D:\truf` | `/data/windows-archive/` with their original relative paths |
Archival scope includes `D:\truf\state`, `runtime\backups`, `runtime\imports`,
non-authority JSON reports from `runtime\control`, `app\.streamlit\config.toml`,
`.env.postgres`, `docker-compose.postgres.yml`, `runner_state.json`,
`runtime\keychecks.7z`, `runtime\orkey.txt`, `runtime\check-openrouter-keys.ps1`,
`runtime\*.md`, root `checked_*.txt`/`todo_*.txt`, root/app `scanner.db*`,
app `config.yaml.*`/`secrets.yaml.*`, root result projection families and
`*.publication-ledger.sqlite3*`. The legacy copy tree is also archival:
`D:\truf\runtime\keychecks \u2014 \u043a\u043e\u043f\u0438\u044f`
(Unicode escapes describe the actual folder name, not a literal shell path).
Within `runtime\state`, `scan_limiter*.db*`, `*.tmp*` and
`janitor.cursor.json` are archival only, never active Linux authority/state.
Excluded: physical PGDATA/WAL, Windows PostgreSQL binaries/logs, live control
authority, locks/PIDs, `S:\scanner-work`, `runtime\downloads`, runtime git/traces/
freeze-diagnostics, `gharchive_cache`, `.git`, `.opencode`, tests and code caches.
Ordinary logs are excluded outside retained result/keycheck projection families;
`scan_errors.log*` is deliberately durable data, not a diagnostic to display.
The manifest records the actual selected inventory and exclusion counts.
The archived `.env.postgres` is never sourced or used for the target connection.
Provision generates `/data/postgres-password`; Linux uses this new
password and a different PG16 system identifier, not the source password or
physical cluster. Archived Windows configs are not executable runtime profiles.
## Capture And Capacity Gates
Main runs `D:\truf-docker\docker\windows_snapshot.py` using native PowerShell,
not context-mode or another Windows Job wrapper: original PostgreSQL correctly
refuses Job membership. The exporter temporarily starts only source maintenance
PostgreSQL and must confirm its stop before publication. Do not rerun capture
into the current attempt directory, reuse partial files, or kill a process that
is retaining authority while stop remains unconfirmed.
After capture exits 0 and publishes its final manifest, main records its digest
from the trusted Windows path, then checks that the WSL-visible manifest has
the same digest. Do not replace the approved pin with a newly computed digest
merely to bypass a mismatch. Native PowerShell digest command:
```powershell
(Get-FileHash -LiteralPath 'D:\truf-docker\docker\imports\windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3\manifest.json' -Algorithm SHA256).Hash.ToLowerInvariant()
```
Check Windows `D:` staging capacity and, independently, physical free space on
`S:`, which backs the Docker/WSL VHDX, and free space on native Linux `/data`.
Record the actual VHDX/daemon storage location; a large Linux `df` result does
not prove the Windows host can grow the VHDX. Budget its anticipated growth
while retaining at least **20 GiB physical free on S:**. The importer cannot
measure or enforce this host-side reserve.
The Linux preflight requires `file_bytes + database_bytes + 20 GiB` free, using
source physical database size from manifest metadata. Older v1 metadata without
that size uses `max(24 GiB, 4 * dump_bytes)` as the database estimate. At least
**20 GiB must still be free on Linux after import**. Check both host and guest
capacity during and after restoration; compressed dump size alone is not a
capacity estimate. Do not delete original data to make space.
## Offline Procedure
Run these steps separately in WSL Bash only after main approves the capture and
capacity evidence. Use the existing Linux Docker daemon and already-built,
importer-integrated `truf-local:runtime` image; no builds or pulls here. Keep the
fixed project/directory below. Do not use the generic initialize/start procedure
in `DOCKER_MIGRATION.md` for this full-schema snapshot.
```bash
TRUF_WINDOWS_SNAPSHOT='/mnt/d/truf-docker/docker/imports/windows-20260915-59a1c0aa23ec411b86f25c5eb9d2a4d3'
TRUF_WINDOWS_SNAPSHOT_SHA256='PENDING'
dci() {
if [[ ! "${TRUF_WINDOWS_SNAPSHOT_SHA256:-}" =~ ^[0-9a-f]{64}$ ]]; then
printf '%s\n' 'STOP: set the approved lowercase manifest SHA-256.' >&2
return 1
fi
sudo -n env \
TRUF_WINDOWS_SNAPSHOT="${TRUF_WINDOWS_SNAPSHOT:?Set the completed snapshot directory}" \
TRUF_WINDOWS_SNAPSHOT_SHA256="$TRUF_WINDOWS_SNAPSHOT_SHA256" \
docker compose \
--project-name truf-docker \
--project-directory /mnt/d/truf-docker \
--env-file /dev/null \
--file /mnt/d/truf-docker/compose.yaml \
--file /mnt/d/truf-docker/compose.snapshot-import.yaml \
"$@"
}
sha256sum "$TRUF_WINDOWS_SNAPSHOT/manifest.json"
dci config --quiet
sudo -n docker image inspect --format '{{.Id}}' truf-local:runtime
sudo -n docker volume inspect --format '{{.Name}} {{.Driver}} {{.Mountpoint}}' truf-docker_data
sudo -n docker container inspect --format '{{.State.Status}}' truf-docker-snapshot-import
```
Replace `PENDING` with the previously approved pin, not a credential. The
`sudo -n env NAME=value ... docker compose` form explicitly passes the two
non-secret interpolation inputs even when sudo strips shell exports. Do not
use `sudo -E` or pass source connection/provider variables. `--env-file /dev/null`
prevents implicit checkout `.env` loading, but does not sanitize shell exports;
use a clean operator shell and no unreviewed Docker/Compose overrides.
Main must separately confirm the image identity, **absence** of
`truf-docker_data`, and absence of the retained import container name before
provisioning. Only specific no-such-volume/no-such-container responses establish
absence; daemon/permission errors do not. If either already exists, stop for
review instead of adopting, overwriting or deleting it.
```bash
dci run --rm --no-deps --pull never -T provision
```
Proceed only after successful provision. This unchanged base service creates
the private layout, generated password and empty placeholders, not an application
schema. **Do not call `initialize`, `import-secrets`, or normal `run` first.**
The importer uses `initialize-empty` internally and restores the entire custom
dump into a virgin schema before raw table/sequence comparison and permitted
target-only recovery/migrations.
```bash
dci run --detach --no-deps --pull never -T \
--name truf-docker-snapshot-import runtime
```
This is the single retained maintenance container: no `--rm`, no automatic
restart, no dependency startup, no healthcheck and `network_mode: none`.
The override preserves the base image/entrypoint, non-root UID/GID, read-only
rootfs, capabilities/security policy, native `/data` volume, tmpfs and resource
limits. It adds only read-only `/import`; `create_host_path: false` rejects a
missing source directory rather than silently creating one.
Import starts with `/opt/truf/app/config.linux.yaml` in both the environment
and explicit `--config`. **Do not merge `compose.windows-import.yaml` here.**
The translated private profile does not exist at initial preflight; the importer
creates and selects it internally only after validating/extracting the snapshot.
## Stopped Verification
```bash
sudo -n docker container wait truf-docker-snapshot-import
sudo -n docker container inspect --format \
'status={{.State.Status}} exit={{.State.ExitCode}} oom={{.State.OOMKilled}} restarts={{.RestartCount}} network={{.HostConfig.NetworkMode}} restart={{.HostConfig.RestartPolicy.Name}} auto_remove={{.HostConfig.AutoRemove}}' \
truf-docker-snapshot-import
```
Require `status=exited exit=0 oom=false restarts=0 network=none restart=no
auto_remove=false`. `wait` prints the container exit code; the command's own
exit status alone is not import success. Waiting may take hours or hold while
maintenance stop is uncertain. Do not impose a timeout that kills the container.
Do not display raw `docker logs`, Compose logs, database logs, full environment
dumps or application data. Only structured importer numeric phase/count/byte
events and allowlisted aggregate/hash evidence are suitable for progress.
Phase 13 is emitted before final publication and is not success proof.
Main must privately inspect the following evidence from the stopped retained
container, for example with `docker cp` into a separate owner-only evidence
directory outside `/import`. Do not start another runtime to inspect it; `health`
and `status` are live readiness actions, not stopped-import verification.
- `/data/config/windows-import-manifest.json`: its byte SHA-256 equals the approved staging manifest pin; source stopped flags are true. The raw evidence, report and initialized marker all carry that same `manifest_sha256`. Report archive/dump hashes match this manifest and the verified staging files.
- `/data/config/windows-import-raw.json`: its SHA-256 matches report `raw_evidence_sha256`; `table_counts` equals manifest `database.table_counts`, `sequences_provided` is true and `sequences_verified` equals manifest `database.sequence_count`. The importer checks actual sequence values before transformations; this evidence records their verified count, not their values. Record only aggregate tables/rows/sequences.
- `/data/config/windows-import.yaml`: hash bytes without displaying values; SHA-256 matches report `config_sha256`.
- `/data/config/windows-import-report.json`: `status` is `verified-stopped`, `maintenance_stopped` is true, all `pipeline_after` counts are zero, final Linux reserve is at least 20 GiB, and cutover/migration/preserved-evidence checks succeeded. Record any reported fenced recovery or Postman rebasing; these may legitimately change final target counts or move incoming tmp/ready bundles after raw comparison.
- `/data/initialized.json`: exists as the last publication, has format `truf-container-data-v1`, `pg_major` 16 and the approved `manifest_sha256`; `import_report_sha256` matches the actual report bytes. Its system identifier equals report `linux_system_identifier` and differs from the source identifier.
- Confirm destination `/data/postgres-linux/postmaster.pid` is absent, original supervisor/PostgreSQL remain stopped, and host/guest space reserves still hold. Record all results in the PENDING table before declaring verified-stopped.
## Failure And Release
Any nonzero exit, OOM, missing/mismatched evidence or uncertain stop is not
verified-stopped. Retain the container, volume and private snapshot. A partial
import is failed/unmarked, not automatically resumable; early rejection can
leave no report. A report saying verified-stopped without a matching final
initialized marker and clean container exit is still insufficient.
Do not automatically retry/restart, overwrite, delete, remove locks/markers,
initialize, prune, run `down --volumes`, or force-kill an authority-holding
importer. Review the retained state first; any new attempt needs an explicit
decision and separately approved fresh destination, not cleanup by this runbook.
There is **no automatic normal run**. A future live run requires separate user
authorization after verified-stopped acceptance. Only then use base
`compose.yaml` plus `compose.windows-import.yaml`, without the snapshot override,
so runtime, health and status all select `/data/config/windows-import.yaml`.
Do not start either the original or destination supervisor as part of import.
+398
View File
@@ -0,0 +1,398 @@
# Worker Operator Experience Handoff
> Historical handoff. The current continuation entry point is
> `docs/session-handoff/README.md`. This file retains implementation and artifact
> provenance, but its stop point and immediate-next-actions section are obsolete.
Last updated: 2026-09-25
This is the historical implementation record for the OpenSpec change
`add-worker-operator-experience`. For current continuation instructions, read
`docs/session-handoff/README.md`. Do not repeat completed production validation
or rebuild accepted artifacts unless a current verification fails.
## User intent and constraints
- Continue autonomously from this handoff and finish the change end to end.
- The user explicitly requested a file handoff because invoking conversation
compression appears to stop or destabilize all OpenCode sessions. Avoid
proactively invoking the compression tool in the continuation session.
- Workspace: `D:\truf-workers`.
- Use the configured SSH server named `sec` only. Never call or connect through
the configured server named `prod`.
- Never print or record tokens, credentials, the private admin prefix, raw
targets/findings, runtime YAML, or worker command lines containing auth data.
- Do not add a new masking, redaction, credential-sandbox, or other security
scope without explicit approval and an OpenSpec requirement.
- Do not remove or revert unrelated workspace files. The repository baseline is
entirely untracked (`git status --short` shows the whole tree as `??`), so Git
cannot provide a meaningful task-specific diff.
- Do not archive the OpenSpec change unless the user explicitly asks. Completing
tasks and reporting "ready to archive" is expected.
## Current OpenSpec state
- Change: `add-worker-operator-experience`
- Schema: `spec-driven`
- Artifact status: proposal, design, specs, and tasks are complete.
- Apply progress before final closure: 23/27 tasks complete.
- File: `openspec/changes/add-worker-operator-experience/tasks.md`
- Tasks 1.1 through 5.5 are checked.
- Remaining unchecked tasks:
- 6.1: validate the canonical from-zero operator guide.
- 6.2: complete test matrix, reproducible packages, manifest registration,
and documented identities.
- 6.3: bounded Windows and WSL/Docker production validation and restoration.
- 6.4: durable dated report with timings, watchdog evidence, snapshots,
transcript, known limits, rollout, and rollback.
Do not check 6.1-6.4 until the remaining focused matrix, report, and strict
OpenSpec validation have passed.
## Implemented scope
The change now includes:
- Versioned worker phase/events and canonical transitions.
- Monotonic sequence handling and JSON/NDJSON contracts.
- Unified bounded diagnostics with deterministic identities.
- PostgreSQL progress/diagnostic persistence and admin queries.
- Server-owned global/per-source assignment deadline policy.
- Private local state, logs, history, diagnostic artifacts, and retention.
- Status, attach, logs/history, JSON/NDJSON, bounded follow/tail CLI behavior.
- Cross-platform worker supervisor, control protocol, drain/stop, shutdown
receipts, stale instance handling, and recovery slots.
- Per-assignment contained runner, controller protocol, watchdog, timeout bundle
publication, restart adoption, and abandoned-root cleanup.
- Authenticated progress endpoint and diagnostic ingestion.
- Admin assignment/progress/diagnostic experience.
- Windows portable and Linux image packaging for the supervisor runtime.
- Canonical operator runbook in `docs/remote-worker-operations.md`.
Important implementation files include:
- `app/worker_contracts.py`
- `app/worker_local_state.py`
- `app/worker_supervisor.py`
- `app/worker_cli.py`
- `app/worker_assignment_runner.py`
- `app/remote_worker_client.py`
- `app/worker_api.py`
- `app/scanner_db.py`
- `app/admin_api.py`
- `app/worker_package.py`
- `app/worker_package_builder.py`
- `docker/verify_packaged_workers.py`
- Worker-related tests under `tests/`
## Final correctness fixes
### Recovered ready-bundle transition
`WorkerSlot` could recover a published ready bundle while its persisted event
phase was still `assigned`. Upload code emitted `uploading` only from
`bundling/backoff`, then attempted the invalid transition
`assigned -> awaiting_receipt`.
Fix in `app/remote_worker_client.py`:
- Emit `UPLOADING` when the current event phase is `ASSIGNED`, as well as the
existing bundling/backoff cases.
- Regression in `tests/test_worker_api.py` validates the event sequence
`assigned -> uploading -> awaiting_receipt`.
The focused worker API/local-state/supervisor suite passed 100 tests after this
fix.
### Packaged E2E abandoned work invariant
Completed runner roots are intentionally retained under `work/abandoned` for at
least 60 seconds; retention maintenance normally runs every 300 seconds. The E2E
harness incorrectly required the total work file count to be zero, causing a
false `linux_direct_claims_timeout` after Linux had correctly claimed both direct
assignments.
Fix in `docker/verify_packaged_workers.py`:
- Linux and Windows work-tree identities now include `active_entries`.
- Files/directories beneath top-level `abandoned` are retained but not active.
- Direct-assignment and final-cleanup predicates require zero active entries,
while preserving strict state and bundle identity checks.
- Outage marker waits also check worker liveness, so an exited worker fails
immediately rather than timing out after four minutes.
### Windows `prepare-worker.ps1` ACL defect
Testing a freshly extracted ZIP exposed a real release bug. The old generated
script ran `icacls ... /grant:r ... /T`; on descendants this produced
inheritance-only ACEs, returned success, and made packaged `python.exe`
inaccessible.
Final fix in `app/worker_package_builder.py`:
1. Set private inheritable full-control ACEs for the current user, SYSTEM, and
Administrators on the package root only.
2. Run `icacls (Join-Path $root '*') /inheritance:d /T /C` so each descendant
converts inherited ACLs to explicit protected ACLs with the correct file or
directory flags.
Regression in `tests/test_worker_package.py` checks the generated script and, on
Windows, executes it and verifies `private_directory_ready(root)` plus
`private_file_ready(child)`. `tests/test_worker_package.py` passes 20 tests.
Do not use the earlier `/reset /T` idea: inherited ACLs are not accepted because
runtime trust requires protected explicit ACLs.
### Watchdog test timing stabilization
The broad focused suite exposed two false failures because three tests created a
100 ms absolute watchdog deadline before runner protocol-root/state setup. Under
the complete Windows suite that setup could consume the deadline, exercising the
startup-deadline branch instead of the intended blocked-operation watchdog.
Test-only changes in `tests/test_worker_assignment_runner.py`:
- Affected tests:
- `test_watchdog_kills_while_state_persistence_is_blocked`
- `test_blocked_startup_gate_write_enters_preparing_timeout_result_path`
- `test_watchdog_kills_while_event_drain_is_blocked`
- Scan deadline: 1 second -> 2 seconds.
- Watchdog deadline: 0.1 second -> 1 second.
- Injected block: 0.4 second -> 1.4 seconds.
- Kill bound: 0.3 second -> 1.3 seconds.
This preserves the independent watchdog assertion and does not weaken product
code. The exact three-test rerun passed: `3 passed in 5.17s`.
## Accepted reproducible artifacts
### Windows final pair: I and J
Paths:
- `build/operator-experience-validation/windows-i.zip`
- `build/operator-experience-validation/windows-i.zip.json`
- `build/operator-experience-validation/windows-j.zip`
- `build/operator-experience-validation/windows-j.zip.json`
Both independently built archives are identical:
- Bytes: `134850988`
- Archive SHA-256:
`6ea9290736a059f1e17d8e89d9cf83506fa4abe2ba2f3731a7422a7b0f386e97`
- Package manifest identity:
`78a962b2bd3fa411413c79e9a8ffb021608a08ff020b1ad851f4505ea634b2b6`
- Build-input identity:
`6991ebbce6ae758c2bdd19a6ae934335aa585a50f86b18ccde8d88bca40ce436`
- Raw `worker-package.json` SHA-256:
`e0b17d70fcb868fe39fac45ab6e05a17c6d40852e6034010fb63b6cab31f8a3c`
Acceptance used a fresh extraction, not the builder output:
- `build/pwe-final-i-extracted`
- The package's own corrected `prepare-worker.ps1` was run once.
- Direct package verification then passed.
The older Windows G/H archives are obsolete for acceptance because they contain
the broken preparation script. Their package manifest identity happens to be the
same because the support script is outside that manifest, but their archive
identity is not accepted. Do not publish or register G/H as final Windows ZIPs.
### Linux final pair: G and H
Tags:
- `truf-worker-test:operator-experience-final-3g`
- `truf-worker-test:operator-experience-final-3h`
Both were built with provenance disabled and are reproducible:
- Worker package identity:
`45588f2cf406b41b239cfa3b8a9dc83fe84b587229bc997b2729016e1f0dde42`
- Image manifest / accepted image ID:
`sha256:3a088f5743121d823aae132234a29730a84339cecbfda5fc601e8e942f9948c3`
- Config:
`sha256:687a1c4c51c1b962c7fa7ea0cc4b04d159e7ba4f94ef347940c9fb225f7cb87d`
- Raw `worker-package.json` SHA-256:
`ee926cce3c19e9e6094753f51fa902415bd7364c24fa649cd0c1b659c0aa4d60`
Extracted final manifest:
- `build/operator-experience-validation/linux-worker-package-g.json`
Test image:
- Tag: `truf-worker-test:operator-experience-final-3`
- ID:
`sha256:1a22c396dbf329e20caf77f88b7c7a310bda86befbcf3b917f10ded3720ee712`
## Final packaged E2E
Passed run:
- Run ID: `35f3f52e232067c1`
- Safe summary: `build/pwe-35f3f52e232067c1/summary.json`
- Windows input: freshly extracted and prepared Windows I.
- Linux input: Linux G.
- Status: passed.
- Cleanup: complete.
- Foreign Docker state: unchanged.
- Windows and Linux normalized evidence matched.
- Restart, outage, durable bundle, direct assignment, direct bundle, receipt,
shutdown, local cleanup, and cross-platform evidence gates all passed.
Do not copy raw target values from the summary into reports or chat. Only the
safe aggregate facts above are needed.
All Docker resources from final and diagnosed failed runs were cleaned by exact
owned IDs/names. Some local `build/pwe-*` failure evidence directories remain and
are safe to leave. `build/pwe-final-g` may still have unusable ACLs after running
the old broken preparation script; do not use it. `build/pwe-final-i-extracted`
is the accepted extracted Windows directory.
## Production validation evidence
Bounded production validation was completed before final package acceptance and
production was restored afterward.
Safe aggregate results:
- Assignments issued: 34.
- Accepted: 33.
- One intentional expected expiry.
- Accepted assignments ingested, settled, and projected: 33.
- Unresolved, precommit, and quarantine counts: zero.
- Natural timeout evidence reservation: 1453.
- Full-stage progress/watchdog evidence reservation: 1455.
Evidence files:
- `build/operator-experience-validation/final-evidence.json`
- `build/operator-experience-validation/progress-v3-evidence.json`
- `build/operator-experience-validation/timeout-evidence.json`
- `build/operator-experience-validation/server-baseline.json`
These files are the source for duration percentiles, phase/watchdog evidence,
diagnostic/admin snapshots, and reconciled counts in the final report. Derive
only aggregate/sanitized facts. Do not reproduce raw targets, findings, secrets,
or private route names.
Final production state after restoration:
- Operations controls: normal/open, revision 126.
- Standard WSL production worker user: enabled, assignment cap 1.
- Standard production device: enabled and not revoked.
- Temporary validation identities: disabled/revoked.
- Runtime canonical health: healthy.
- Edge remained up.
Do not repeat production assignments merely to write the report. Existing
evidence is sufficient.
## Registered trusted manifests
Registration was completed only after the final packaged E2E passed, using SSH
server `sec` only.
Remote paths:
- `/etc/truf/worker-packages/linux-worker-package-v2.json`
- `/etc/truf/worker-packages/windows-worker-package-v3.json`
Final remote SHA-256 values match the accepted manifests:
- Linux: `ee926cce3c19e9e6094753f51fa902415bd7364c24fa649cd0c1b659c0aa4d60`
- Windows: `e0b17d70fcb868fe39fac45ab6e05a17c6d40852e6034010fb63b6cab31f8a3c`
Both are `root:root` mode `0644`. Existing
`.pre-operator-experience` backups were preserved unchanged. Upload temp files
were removed. After registration, canonical runtime health succeeded and Docker
reported `truf-docker-runtime-1` healthy. No restart or config mutation was
needed.
## Test state
Completed checks:
- Worker API/local-state/supervisor focused suite: 100 passed.
- Worker package tests after ACL fix: 20 passed.
- Exact three watchdog timing tests after stabilization: 3 passed.
- Full packaged Windows/Linux E2E: passed, run `35f3f52e232067c1`.
- Production health after final manifest registration: passed.
The broad focused matrix was run before the watchdog test timing patch:
```powershell
python -B -m pytest tests/test_worker_api.py tests/test_worker_api_runtime.py tests/test_worker_assignment.py tests/test_worker_assignment_runner.py tests/test_worker_cli.py tests/test_worker_contracts.py tests/test_worker_local_state.py tests/test_worker_observability_db.py tests/test_worker_package.py tests/test_worker_runner_handoff_linux.py tests/test_worker_supervisor.py tests/test_remote_worker_db.py tests/test_scan_execution.py tests/test_admin_api.py -q
```
Result before the timing-only patch:
- 353 passed.
- 3 skipped.
- 2 false timing failures described above.
The two failures and the nearby equivalent test pass after the patch, but the
complete 14-file command has not yet been rerun. This is the exact current stop
point.
An unrestricted repository-wide pytest run is not a useful release gate in this
checkout because unrelated private/generated assets and platform assumptions are
absent. Its known baseline was `3031 passed, 134 skipped, 68 failed`. Do not try
to fix unrelated failures as part of this change. The focused change matrix,
packaged E2E, production proof, and strict OpenSpec validation are the gates.
## Immediate next actions
1. Rerun the exact 14-file focused matrix shown above. Expected result after the
timing patch is 355 passed and 3 skipped. If it fails, diagnose only genuine
worker-operator regressions; do not broaden scope.
2. Create the durable report:
`docs/worker-operator-experience-validation-2026-09-24.md`.
3. In the report, include only sanitized aggregate evidence:
- Scope and acceptance criteria.
- Final Windows I/J and Linux G/H identities from this handoff.
- Packaged E2E run `35f3f52e232067c1` and cleanup/foreign-state result.
- Production issued/accepted/reconciled counts.
- Duration percentiles derived from `final-evidence.json`.
- Watchdog/full-stage evidence from `progress-v3-evidence.json`.
- Natural timeout evidence from `timeout-evidence.json`.
- Diagnostic/admin snapshot facts without private content.
- Sanitized operator command transcript.
- Known limits, especially no public registry/auto-updater and intentional
abandoned-root retention.
- Rollout and rollback/restoration facts, controls revision 126, and final
healthy state.
4. Re-read `docs/remote-worker-operations.md` against task 6.1. It already covers
package acquisition/build, Windows preparation, install/first run, lifecycle,
status/attach/logs/history, phases/deadlines, diagnostics, drain/stop,
recovery, update, and removal. Make only a minimal correction if the final
artifact/report facts expose an actual gap.
5. Run strict validation:
```powershell
openspec validate add-worker-operator-experience --strict
```
6. If the focused matrix, report, runbook review, and strict validation pass,
change only task checkboxes 6.1-6.4 in
`openspec/changes/add-worker-operator-experience/tasks.md` from `[ ]` to `[x]`.
7. Re-run `openspec instructions apply --change "add-worker-operator-experience" --json`
and confirm progress 27/27 with state `all_done`.
8. Give the user a concise completion result and say the change is ready to
archive. Do not archive it without an explicit request.
## Report safety checklist
Before saving or quoting the final report, verify it contains none of:
- Tokens or credentials.
- Raw worker targets or findings.
- Runtime YAML or secret environment values.
- The private admin route prefix.
- Worker argv/auth command lines.
- Unbounded log or diagnostic bodies.
Allowed report content includes hashes, aggregate counts, reservation numeric
IDs used as evidence references, phase names, durations/percentiles, safe test
counts, generic command names, and public artifact paths within this workspace.
+40
View File
@@ -0,0 +1,40 @@
# Remote Worker Development Workspace
This is an independent source-only snapshot of the current `D:\truf-docker` working tree, not a copy of its running system.
## Snapshot
- Source HEAD for provenance: `1b3c7fc4948c5cf2fc389065db3693a27b65300c`.
- Current modified and selected untracked source files are included; this snapshot is not equivalent to that commit alone.
- Copied 325 files, 8,150,338 bytes (about 7.8 MiB), with SHA-256 equality checked for every copied file.
- Source-side deletions are preserved, including the absence of `app/config.yaml`.
- No database, PGDATA, runtime directory, finding/keycheck output, logs, imports, caches, dependencies, or executable binaries were copied.
- No actual `.env`, secrets file, provider credential pool, or source Git history was copied. The tracked `.env.postgres.example` is only a template.
- Git was initialized independently. No commit, remote, runtime container, or Docker data volume was created for this workspace.
- Source `.opencode` skills, the old session handoff, and the loose operator note were not copied. Existing OpenSpec change artifacts remain unchanged and unarchived.
## Active Plan
`openspec/changes/add-minimal-remote-scan-workers/` contains the completed proposal, design, requirements, implementation checklist, and isolated worker implementation. It preserves existing scan and server-side keycheck logic.
## Safe Local Checks
Run from this directory:
```powershell
python -I -S -B docker/test_verify.py -v
python -I -S -B tests/container_unit.py --check-selection
openspec validate add-minimal-remote-scan-workers --strict --no-interactive
```
The first two commands use the standard library only, do not import the application, and do not start Docker, PostgreSQL, a scanner, or provider checks. The selection check validates test declarations, not their execution.
The original planning-stage verification passed. Current implementation evidence is recorded by the change checklist and isolated test outputs, including cross-platform packaged-client scans and the empty-database end-to-end gates.
## Before Runtime Testing
The inherited deployment files are SOURCE REFERENCES, not an isolated test setup. In particular, `compose.yaml` still names the production-style `truf-docker` project and shared `truf-local:*` image tags; `docker-compose.postgres.yml` and import overrides must not be used here. Do not run plain `docker compose up`, import a snapshot, invoke old native launchers, or run unrestricted pytest.
The first implementation tasks must establish unique test project/image/volume names, neutral configuration, scrubbed inherited credentials/DSNs/proxies, disabled live discovery, and synthetic source/provider transports. Review existing `compose.e2e.yaml`, `docker/verify.py`, and `tests/container_unit.py` for reuse before adding any new test infrastructure. Never mount `D:\truf`, `D:\truf-docker`, their runtime directories, or existing Docker data volumes. A future test database must initialize empty and contain only synthetic fixtures.
Local test implementation and worker packaging must use this workspace, not the active source or runtime. Production deployment/import and archiving unrelated changes require separate authorization.
+10
View File
@@ -0,0 +1,10 @@
[server]
headless = true
address = "127.0.0.1"
port = 5000
[theme]
base = "light"
[browser]
gatherUsageStats = false
+289
View File
@@ -0,0 +1,289 @@
# Truf Runtime Operations
All scanner, keycheck, dashboard, and PostgreSQL lifecycle mutation is owned by `supervisor.py`. Direct `console_runner.py`, mutating `keycheck_runner.py`, and legacy `app.py` controls are retired.
PostgreSQL is the sole authority for scan results, queue completion, keycheck results, and keycheck current state. Scanner sources publish private version-2 bundles to `S:\scanner-result-bundles`; the singleton result ingester commits them transactionally. JSONL and status files are asynchronous, rebuildable compatibility projections and may lag without rolling back a committed scan.
The scanner has `max_active_scans=3` guaranteed fair permits plus at most one memory-gated non-Docker bonus permit. A permit covers staging, TruffleHog, normalization, bundle fsync, and the atomic ready rename only. PostgreSQL ingestion, JSONL projection, and keychecks do not hold scan permits.
Run commands from `D:\truf\app` unless a full path is shown.
## Start And Stop
Canonical production start:
```powershell
..\start_runtime.ps1
```
Canonical full coordinated shutdown:
```powershell
..\stop_runtime.ps1
```
Equivalent authenticated launch after cluster identity has been verified:
```powershell
python -I -S -B runtime_bootstrap.py supervisor -- --runtime-bootstrap-entrypoint D:\truf\app\supervisor.py --config config.yaml --background --no-dashboard --with-postgres
```
Direct read/control operations remain supported by `supervisor.py`:
```powershell
python supervisor.py --config config.yaml --background-status
..\attach_runtime.ps1
python supervisor.py --config config.yaml --cmd "status"
python supervisor.py --config config.yaml --stop-background --with-postgres
```
In an attached prompt, `q` only detaches; `shutdown` requests full coordinated shutdown. Prefer `..\stop_runtime.ps1` for canonical full shutdown.
The dashboard is currently disabled. To use the read-only dashboard, enable it in `config.yaml` and perform a coordinated runtime restart. The legacy scanner UI is intentionally retired.
## Source Commands
Use the foreground supervisor prompt or authenticated `--cmd` requests:
```powershell
python supervisor.py --config config.yaml --cmd "status"
python supervisor.py --config config.yaml --cmd "start github"
python supervisor.py --config config.yaml --cmd "once gitlab"
python supervisor.py --config config.yaml --cmd "restart dockerhub"
python supervisor.py --config config.yaml --cmd "pause npm"
python supervisor.py --config config.yaml --cmd "resume npm"
python supervisor.py --config config.yaml --cmd "stop package_git"
python supervisor.py --config config.yaml --cmd "stop all"
python supervisor.py --config config.yaml --cmd "start all"
python supervisor.py --config config.yaml --cmd "logs pypi 80"
python supervisor.py --config config.yaml --cmd "command github"
```
`stop all` stops managed children while the supervisor and PostgreSQL remain running. `start all` includes `pypi`; do not use it when `pypi` must remain stopped.
Configure discovery mode, queries, custom target files, timeouts, workers, and source-specific arguments in `config.yaml` before starting or restarting a source. Do not pass tokens or mutable scan options through a direct runner command.
## Keychecks
The supervisor manages keychecks as the `keychecks` pseudo-source:
```powershell
python supervisor.py --config config.yaml --cmd "start keychecks"
python supervisor.py --config config.yaml --cmd "recheck all network"
python supervisor.py --config config.yaml --cmd "recheck gemini all --max-keys 100"
python supervisor.py --config config.yaml --cmd "recheck replicate valid --max-keys 25"
python supervisor.py --config config.yaml --cmd "recheck all --summary-only"
python supervisor.py --config config.yaml --cmd "logs keychecks 80"
```
Provider probe arguments belong under `keychecks.service_args` in `config.yaml`. Normal providers claim fenced PostgreSQL `keycheck_candidates` and commit `keycheck_results` plus `keycheck_current_state` directly. Files under `D:\truf\runtime\keychecks` are compatibility projections, not current-state authority. At most four provider children run concurrently, each with a bounded candidate slice so later services cannot starve.
## Configuration And Secrets
Primary files:
```text
D:\truf\app\config.yaml
D:\truf\app\secrets.yaml
D:\truf\.env.postgres
D:\truf\runtime\proxy.txt
```
`config.yaml` contains paths, source settings, and auth-pool names. Actual source tokens belong in `secrets.yaml`; PostgreSQL credentials belong in `.env.postgres`. Managed children receive one canonical loopback PostgreSQL DSN after all configurable environment overrides.
Important runtime paths:
```text
D:\truf\runtime\results
S:\scanner-result-bundles
D:\truf\runtime\result_spool (legacy import compatibility only)
D:\truf\runtime\queues
D:\truf\runtime\state
D:\truf\runtime\logs
D:\truf\runtime\control
D:\truf\runtime\keychecks
D:\truf\runtime\postman_cache
D:\truf\runtime\postgres\data
D:\truf\tmp
```
Runtime startup performs read-only ACL/owner/reparse preflight and never repairs paths.
### Remote Assignment Capacity
`global.result_bundle_max_event_bytes` and `supervisor.worker_api.max_bundle_bytes` are hard per-bundle limits and remain 64 MiB. They are not admission reservations. Each unresolved remote assignment instead charges the persisted `global.remote_assignment_reserve_bytes` baseline of 2 MiB on both the bundle and projection byte axes; local scans retain their existing worst-case reservation behavior.
Remote admission enforces `global.remote_assignment_max_active: 50` atomically in addition to each user's typed `active_assignment_cap`. The intended 50-assignment user must therefore have its cap set to 50 through the authenticated worker administration path. Lower either cap to reduce concurrency; do not raise the global cap above the validated maximum.
A valid remote bundle larger than 2 MiB atomically expands its persisted bundle charge to actual bytes before the server returns an acceptance receipt. Temporary aggregate bundle exhaustion returns retryable capacity backpressure, and the worker must retry the identical durable upload. Projection serialization similarly expands a leased job to exact aggregate bytes before any append. If projection capacity is unavailable, the untouched job returns to `pending`; capacity backpressure alone never quarantines it.
The production keycheck limits of 131,072 items and 128 MiB cover fifty baseline candidate reservations. Bundle and projection aggregate capacities and projection headroom remain independent safety bounds. Runtime-document validation rejects a hard bundle limit above 64 MiB, a remote baseline below 2 MiB or above the hard limit, a global cap above 50, and any aggregate axis that cannot hold all configured baselines.
## Authority Model
One cross-session lock is derived only from the canonical bundled PostgreSQL data directory. It is held by every lifecycle-owning foreground/background supervisor and by PostgreSQL bootstrap/verification, migration, reconciliation, and offline hardening. Changing control directory, instance file, or port cannot split authority; different data directories have independent locks.
The configurable control-directory lock remains a secondary per-instance safety layer. Duplicate launch failure never sends coordinated shutdown to a different owner.
Before spawn, the launcher captures exact config, supervisor, and code-manifest authority. The manifest also covers the result bundle, ingester, projector, keycheck candidate, and isolated janitor modules.
The child completes canonical DSN validation and all controller/backend/source/keycheck/dashboard construction before publishing `ACTIVATING`. PostgreSQL, dashboard, and source ticks are forbidden until authenticated parent activation and exact post-activation recheck publish `ACTIVE`.
If activation becomes uncertain, rollback uses authenticated shutdown and the full configured deadline. It never force-terminates an exact published candidate that may be active. An unconfirmed exact candidate is left running and reported rather than risking an orphaned PostgreSQL tree.
Authenticated shutdown atomically publishes `STOPPING` and closes all start gates under the control lock before acknowledgement. Mutating commands and snapshot-triggered polls cannot start children after this transition.
Code/config drift closes start gates and triggers safe shutdown. Authenticated shutdown remains available through private instance credentials, endpoint binding, and retained process identity even when on-disk code changed. Tokens, raw DSNs, and secret-derived hashes are not written to logs.
## Child Authentication
Managed scanner, result-ingester, JSONL-projector, keycheck, janitor, and dashboard children must prove all of the following before mutation. The janitor receives no database capability and persists only an exact-private bounded local cursor:
1. Private per-instance metadata matches inherited instance credentials.
2. The retained supervisor process and config command line match metadata.
3. The supervisor handshake reports `ACTIVE`.
4. Config, supervisor, and complete code-manifest hashes match.
5. `SCANNER_DB_URL`, `DATABASE_URL`, and the inherited managed DSN name the same canonical loopback PostgreSQL authority.
An environment marker by itself grants no authority. Empty or foreign DSNs fail before application files, `ScannerDB`, dependency checks, provider checks, or scanning.
## PostgreSQL Maintenance
One-time offline cluster identity binding, with all runtime processes stopped:
```powershell
python postgres_runtime.py bootstrap --config config.yaml
```
Read-only identity verification:
```powershell
python postgres_runtime.py verify --config config.yaml
```
Identity-verified maintenance start/stop, without scanner or keycheck children:
```powershell
python postgres_runtime.py maintenance-start --config config.yaml
python postgres_runtime.py maintenance-stop --config config.yaml
```
### Docker Depth Experiment Operator
Keep `sources.dockerhub.docker_depth_experiment.enabled: false` while reviewing and applying the cohort and hold. From `D:\truf\app`, use the existing private `runtime\state` directory and always stop maintenance PostgreSQL in a `finally` step if an operator command fails:
```powershell
..\stop_runtime.ps1
python postgres_runtime.py maintenance-start --config config.yaml
python docker_depth_operator.py --config config.yaml --status
python docker_depth_operator.py --config config.yaml --generate-cohort-manifest D:\truf\runtime\state\docker-depth-cohort-review.json
$cohortSha256 = Read-Host 'Reviewed cohort SHA-256'
python docker_depth_operator.py --config config.yaml --apply-cohort-manifest D:\truf\runtime\state\docker-depth-cohort-review.json --approve-sha256 $cohortSha256 --apply --sources-stopped
python docker_depth_operator.py --config config.yaml --generate-hold-manifest D:\truf\runtime\state\docker-depth-hold-review.json
$holdSha256 = Read-Host 'Reviewed hold SHA-256'
python docker_depth_operator.py --config config.yaml --apply-hold-manifest D:\truf\runtime\state\docker-depth-hold-review.json --approve-sha256 $holdSha256 --apply --sources-stopped
python postgres_runtime.py maintenance-stop --config config.yaml
```
Review the private files out of band and approve exactly the SHA-256 printed by their generation commands. The operator has no DSN option, never prints queries or targets, and never starts or stops PostgreSQL. After the hold apply and maintenance stop, changing only `enabled` from `false` to `true` is the separate activation decision; its semantic `config_sha256` must remain unchanged. Run `--status` between another maintenance start/stop pair to verify that hash before a separately approved `..\start_runtime.ps1`.
An attempt-limit hold caused by the retired zero-graph resolver defect has a separate one-time reviewed recovery. Keep enabled config, stop sources, start maintenance PostgreSQL, review the private manifest, and approve only its exact printed SHA-256:
```powershell
python docker_depth_operator.py --config config.yaml --generate-resolver-refund-manifest D:\truf\runtime\state\docker-depth-resolver-refund-review.json
$refundSha256 = Read-Host 'Reviewed resolver refund SHA-256'
python docker_depth_operator.py --config config.yaml --apply-resolver-refund-manifest D:\truf\runtime\state\docker-depth-resolver-refund-review.json --approve-sha256 $refundSha256 --apply --sources-stopped
```
A genuine remote `resolver_attempt_limit` hold uses a separate reviewed disposition. The manifest deterministically selects the next fresh repository or records terminal remote-unavailable scarcity when none remains:
```powershell
python docker_depth_operator.py --config config.yaml --generate-resolver-disposition-manifest D:\truf\runtime\state\docker-depth-resolver-disposition-review.json
$dispositionSha256 = Read-Host 'Reviewed resolver disposition SHA-256'
python docker_depth_operator.py --config config.yaml --apply-resolver-disposition-manifest D:\truf\runtime\state\docker-depth-resolver-disposition-review.json --approve-sha256 $dispositionSha256 --apply --sources-stopped
```
Release is reviewed only after the experiment reaches `completed` under enabled config:
```powershell
..\stop_runtime.ps1
python postgres_runtime.py maintenance-start --config config.yaml
python docker_depth_operator.py --config config.yaml --generate-reactivation-manifest D:\truf\runtime\state\docker-depth-reactivation-review.json
$reactivationSha256 = Read-Host 'Reviewed reactivation SHA-256'
python docker_depth_operator.py --config config.yaml --apply-reactivation-manifest D:\truf\runtime\state\docker-depth-reactivation-review.json --approve-sha256 $reactivationSha256 --apply --sources-stopped
python postgres_runtime.py maintenance-stop --config config.yaml
```
Runtime-safety schema migration, with supervisor, dashboard, scanner, and keycheck sessions stopped:
```powershell
python migrate_runtime_safety.py --config config.yaml --apply --sources-stopped
```
Final cutover refuses a nonempty legacy result spool, any `scan_publication_outbox` row, any legacy `raw_result_json` row, or a prepared legacy JSONL-ledger append. Runtime workers refuse to start until the migration records the singleton PostgreSQL cutover marker.
Import bounded batches from the retired durable spool until `remaining=0`:
```powershell
python migrate_runtime_safety.py --config config.yaml --import-legacy-spool --max-rows 1000 --apply --sources-stopped
```
Convert bounded legacy raw rows to normalized-v2 data, then rerun the normal migration to restore the cutover marker:
```powershell
python migrate_runtime_safety.py --config config.yaml --backfill-normalized-results --max-rows 1000 --max-bytes 201326592 --max-seconds 30 --apply --sources-stopped
```
If the cutover gate reports legacy outbox rows, project only that bounded backlog offline before retrying migration:
```powershell
python migrate_runtime_safety.py --config config.yaml --drain-legacy-outbox --max-rows 1000 --apply --sources-stopped
```
If a legacy event is too expensive for the per-finding ledger drain, first apply the additive schema (the command remains nonzero while the gate is closed), then transfer exact outbox references into the singleton projector queue without rewriting historical payloads:
```powershell
python migrate_runtime_safety.py --config config.yaml --import-legacy-outbox-to-projection --max-rows 1000 --apply --sources-stopped
```
Build a full bounded, resumable PostgreSQL-derived projection in a dedicated empty directory. Repeat until `completed=true`; live projection files are never overwritten:
```powershell
python migrate_runtime_safety.py --config config.yaml --rebuild-jsonl-output "D:\truf\runtime\rebuild" --max-rows 1000 --max-bytes 201326592 --max-seconds 30 --apply --sources-stopped
```
Quarantine review accepts only a private `truf-pipeline-quarantine-review-v1` manifest with exact `id`, `reason_code`, `payload_sha256`, and `action` (`discard`, deterministic `retry`, or Docker layer `rescan`) entries. `rescan` retires the stale bundle and returns its immutable target to the normal fresh-claim path:
```powershell
python migrate_runtime_safety.py --config config.yaml --review-pipeline-quarantine "D:\review\quarantine.json" --max-rows 1000 --apply --sources-stopped
```
Bounded todo reconciliation:
```powershell
python migrate_runtime_safety.py --config config.yaml --apply --sources-stopped --todo "D:\truf\runtime\queues\todo_github.txt" --source github --platform github --max-rows 1000 --max-bytes 4194304 --max-seconds 5
```
Offline layout hardening creates required directories parent-first, recursively hardens runtime-owned trees, and hardens existing config, secrets, proxy, detector config, and PostgreSQL environment files:
```powershell
python migrate_runtime_safety.py --config config.yaml --harden-runtime --apply --sources-stopped
```
Maintenance and runtime exclude each other through the same cluster authority lock. PostgreSQL sessions use `search_path=public`; `pg_catalog` keeps implicit precedence, authority built-ins are explicitly qualified, and PUBLIC `CREATE` on `public` fails closed.
`D:\truf\docker-compose.postgres.yml` is a disabled, noncanonical manual-recovery fixture. It is behind the `noncanonical-manual-recovery` profile, has no restart policy, requires an explicit unused `TRUF_DOCKER_POSTGRES_PORT`, and binds data under `D:\truf`; never use it to start or replace the supervisor-owned cluster. Any legacy Docker named volume is intentionally left untouched.
## Dashboard Secrecy
Default dashboard queries and frames do not contain raw credentials. Raw scanner and validation values are available only behind explicit default-false per-session reveal controls with a local warning. Treat the database itself as sensitive because persisted findings still contain raw values.
The dashboard binds to loopback only. Do not expose it through `0.0.0.0`, a reverse proxy, screenshots, or shared logs.
## Cleanup
The authenticated `janitor` child is the only stale-tree recovery worker. It does not import `scanner`, and it enforces exact PID/creation-time/executable identities plus enumeration, candidate, entry, byte, time, and depth budgets. Its exact-private local cursor rotates layouts after every inspected entry so a huge first layout cannot starve later trees. Unknown identity retains the tree. Sources may only attempt bounded cleanup of a directory they just used; there is no startup sweep, low-space sweep, periodic supervisor scanner import, or `atexit` cleanup.
Definitively aborted unreferenced admission intents and deleted artifacts are retired in bounded keyset batches after 30 days. Retirement updates an aggregate SHA-256 chain and count before deleting exact rows; pending, committed/referenced, active, and open-quarantine authority is never eligible.
Do not manually delete results, result spool events, queue state, PostgreSQL data, or control metadata. Use authenticated coordinated shutdown before offline maintenance.
+351
View File
@@ -0,0 +1,351 @@
# Detector Notes
Working notes about TruffleHog detector behavior and local post-processing ideas.
## GitHub / GitLab Noise
Current TruffleHog source contains both modern and legacy detectors.
GitHub v2 detects modern prefixed PATs:
```text
(ghp|gho|ghu|ghs|ghr|github_pat)_[a-zA-Z0-9_]{36,255}
```
GitHub v1 detects legacy 40-character hex tokens near words such as `github`, `gh`, `pat`, or `token`:
```text
(?:github|gh|pat|token).{0,40}([a-f0-9]{40})
```
GitHubOauth2 detects a 20-character client id and a 40-character client secret near `github`:
```text
client_id: [a-zA-Z0-9]{20}
client_secret: [a-f0-9]{40}
Raw = client_id
RawV2 = client_id + client_secret
```
GitLab v2 detects modern PATs:
```text
glpat-[a-zA-Z0-9\-=_]{20,22}
```
GitLab v1 detects any 20-22 character token-like value near `gitlab` and skips `glpat-` so v2 can handle it:
```text
gitlab ... ([a-zA-Z0-9\-=_]{20,22})
```
Observed local results show high false-positive volume for unverified GitHub v1, GitHubOauth2, and GitLab v1 detections. The current scanner post-filter drops unverified GitHub/GitLab findings that do not match known modern token prefixes. This reduces noise but can hide real legacy/OAuth credentials if verification cannot run.
TODO: Prefer a confidence model over hard dropping:
```text
verified -> high confidence
modern prefix shape -> high/medium confidence
legacy GitHub v1 / GitHubOauth2 / GitLab v1 -> low confidence unless verified
```
Dashboard should hide low-confidence findings by default, but allow explicit review.
## GCP
### GCP service account JSON
Detector: `GCP`
TruffleHog detects JSON blobs containing `auth_provider_x509_cert_url` and parses service-account style credentials.
Useful fields already present in the JSON:
```text
type
project_id
private_key_id
private_key
client_email
client_id
auth_uri
token_uri
auth_provider_x509_cert_url
client_x509_cert_url
```
TruffleHog output behavior:
```text
Raw = client_email, or full key JSON if client_email is missing
RawV2 = full cleaned credential JSON
Redacted = client_email
ExtraData.project = project_id
AnalysisInfo.principal = client_email
AnalysisInfo.type = type
```
Practical enrichment fields:
```text
gcp_project_id
gcp_client_email
gcp_client_id
gcp_private_key_id
gcp_credential_type
```
This detector has enough context to verify/function without extra source-code lookup if `RawV2` is preserved.
### GCP Application Default Credentials
Detector: `GCPApplicationDefaultCredentials`
TruffleHog detects ADC JSON containing `client_secret` and `.apps.googleusercontent.com` client IDs.
Useful fields:
```text
client_id
client_secret
refresh_token
type
```
TruffleHog output behavior:
```text
Raw = client_id without .apps.googleusercontent.com suffix
RawV2 = client_id_without_suffix + refresh_token
Redacted = shortened refresh_token
ExtraData may contain verification details when verified
```
Risk: `RawV2` is concatenated and does not retain `client_secret` cleanly. The raw finding JSON may not be enough to reconstruct the original ADC JSON unless the source line/file is available.
Practical enrichment fields:
```text
gcp_client_id
gcp_refresh_token_redacted
gcp_credential_type
```
TODO: For ADC findings, use source context around the finding to parse the whole JSON and preserve `client_secret`/`refresh_token` as structured fields.
### Google AQ authentication keys
`AQ.` credentials are currently classified through the Gemini Developer API at
`generativelanguage.googleapis.com`. That result does not establish Vertex AI access.
TODO: Add a separate Vertex AI Express probe for `AQ.` credentials against the supported
`aiplatform.googleapis.com` key-authenticated methods. Keep Gemini Developer API and Vertex
results independent, and do not infer access to the full project/location-scoped Vertex API
from the key prefix or from a successful Gemini Developer API check.
## Azure
### Azure Container Registry
Detector: `AzureContainerRegistry`
Detector finds registry hosts and ACR password-like values.
Patterns:
```text
registry: <name>.azurecr.io
password: [a-zA-Z0-9+/]{42}+ACR[a-zA-Z0-9]{6}
```
TruffleHog output behavior:
```text
Raw = password
RawV2 = {"username":"<registry>","password":"<password>"}
Redacted = registry name
```
Verification uses:
```text
https://<registry>.azurecr.io/v2/
BasicAuth(username=<registry>, password=<password>)
```
Practical enrichment fields:
```text
azure_acr_registry
azure_acr_login_server = <registry>.azurecr.io
```
This detector has enough context in `RawV2` to be useful.
### Azure OpenAI
Detector: `AzureOpenAI`
Detector finds API keys and Azure OpenAI endpoints.
Patterns:
```text
endpoint: <service>.openai.azure.com
key: 32 lowercase hex chars near api_key/openai_key keywords
```
TruffleHog output behavior:
```text
Raw = api key
RawV2 = key:endpoint when endpoint is paired during verification or when only one endpoint exists
Redacted = shortened key
```
Verification calls:
```text
https://<endpoint>/openai/deployments?api-version=2023-03-15-preview
Header: Api-Key: <key>
```
Practical enrichment fields:
```text
azure_openai_endpoint
azure_openai_resource_name
```
TODO: If RawV2 is empty, scan nearby source context for `.openai.azure.com` to pair keys with endpoints.
### Azure DevOps PAT
Detector: `AzureDevopsPersonalAccessToken`
Detector finds a 52-character token and an organization-like string near `azure`.
TruffleHog output behavior:
```text
Raw = PAT
RawV2 = PAT + organization
```
Verification calls:
```text
https://dev.azure.com/<organization>/_apis/projects
BasicAuth(username="", password=<PAT>)
```
Risk: `RawV2` is concatenated without delimiter, so organization extraction from RawV2 is ambiguous unless the token length is known.
Practical enrichment fields:
```text
azure_devops_org
```
TODO: Parse organization from raw finding JSON/source context rather than relying on concatenated RawV2 alone.
## DockerHub
Detector: `Dockerhub`
DockerHub v2 detects modern PATs:
```text
dckr_pat_[a-zA-Z0-9_-]{27}
```
DockerHub v1 detects UUID-like legacy tokens near `docker`.
Both versions try to pair the token with nearby usernames or emails:
```text
username: user/usr/-u/id nearby value or email address
Raw = token
RawV2 = username:token when username/email is found
```
Verification calls:
```text
POST https://hub.docker.com/v2/users/login
{"username":"<username>","password":"<token>"}
```
If verified, ExtraData can include:
```text
hub_username
hub_email
hub_scope
2fa_required
```
Practical enrichment fields:
```text
dockerhub_username
dockerhub_email
dockerhub_scope
dockerhub_2fa_required
```
Risk: A token without nearby username cannot be verified by this detector, but may still be useful if a username can be found elsewhere in the same package/repo.
TODO: For unverified DockerHub PATs with empty RawV2, scan nearby context and package/repo metadata for plausible DockerHub usernames.
## Proposed Enrichment Layer
Add a post-processing enrichment layer after TruffleHog result parsing and before DB insert.
Input:
```text
finding JSON
target metadata
source file path/line if available
optional nearby source context
```
Output fields stored in DB/dashboard:
```text
provider
credential_kind
credential_confidence
required_context_missing
principal
project_id
tenant_id
organization
registry
endpoint
username
email
scope
resource
```
Suggested confidence levels:
```text
verified
structured_complete
prefix_shape_complete
token_only_missing_context
legacy_unverified
noisy_unverified
```
Priority implementation:
1. Parse GCP service-account JSON from `RawV2`.
2. Parse Azure ACR `RawV2` JSON.
3. Parse Azure OpenAI `RawV2` as `key:endpoint` when available.
4. Parse DockerHub `RawV2` as `username:token` and ExtraData when verified.
5. Add low-confidence classification for GitHub/GitLab legacy detectors instead of hard-dropping them.
6. Optionally read nearby file context for detectors where RawV2 lacks required context.
+13
View File
@@ -0,0 +1,13 @@
# Offline JSONL Reconciliation
PostgreSQL is authoritative. JSONL and provider status files are asynchronous, bounded, rebuildable compatibility projections and may lag. Normal keychecks never consume `found_secrets.jsonl`; reconciliation is explicit offline compatibility work only.
Provider `*Checked.txt` files are rebuilt from PostgreSQL by the keycheck runner and are not append streams owned by the JSONL projector.
1. Stop the supervisor and acquire the same cluster authority used by `migrate_runtime_safety.py`.
2. Make an immutable backup of the current file, every numbered segment, the manifest, the publication ledger, and any `*.torn-tail.bin` file.
3. Validate every retained segment as newline-terminated UTF-8 JSON. Quarantine, rather than concatenate, any final partial record.
4. Do not use keycheck `input_state.json` as a retention checkpoint. PostgreSQL candidates and current state are authoritative; immutable compatibility generations may be retired by the configured generation limit.
5. For pre-v2 multi-GiB history, do not raise online bounds or backfill it during startup. Preserve it and use a separately reviewed, bounded offline rebuild/import operation.
6. The singleton projector recovers prepared appends by exact generation, offset, length, and SHA-256; partial tails are quarantined and truncated to the prepared offset before retry.
7. Rotation renames the active generation atomically and never copies full history. A deterministic poison job is quarantined individually and later jobs continue.
+192
View File
@@ -0,0 +1,192 @@
# Keychecker Layout
Normal input and authority:
```text
PostgreSQL keycheck_candidates fenced provider work queue
PostgreSQL keycheck_results authoritative history
PostgreSQL keycheck_current_state authoritative current classification
D:\truf\runtime\proxy.txt optional provider proxy input
```
`found_secrets.jsonl`, `*Results.jsonl`, and `*Checked.txt`/status files are projector-owned compatibility outputs. They may lag and normal providers do not read or write them.
Shared helper:
```text
D:\truf\app\keycheckers\keycheck_common.py
```
`keycheck_runner.py --input-mode postgres` is the default managed mode. `--input` is accepted only with explicit `--input-mode jsonl` for reviewed offline/import compatibility. Provider completion inserts the result, updates current state, completes the exact candidate lease, releases candidate capacity, and creates its projection job in one PostgreSQL transaction.
## DeepSeek
```powershell
python supervisor.py --config config.yaml --cmd "recheck deepseek all --max-keys 100"
```
Output folder:
```text
deepseek\deepseekAlive.txt
deepseek\deepseekNoBalance.txt
deepseek\deepseekDead.txt
deepseek\deepseekLimited.txt
deepseek\deepseekNetwork.txt
deepseek\deepseekUnknown.txt
deepseek\deepseekChecked.txt
deepseek\deepseekResults.jsonl
```
## Qwen / DashScope
```powershell
python supervisor.py --config config.yaml --cmd "recheck qwen all --max-keys 100"
```
The checker validates `QwenDashScope` findings with `GET /models` against the public DashScope OpenAI-compatible region endpoints. Coding Plan keys (`sk-sp-...`) use `https://coding-intl.dashscope.aliyuncs.com/v1` by default. Workspace-specific endpoints can be added with `--base-url` via `keychecks.service_args.qwen` or `QWEN_BASE_URLS`.
Output folder:
```text
qwen\qwenAlive.txt
qwen\qwenNoBalance.txt
qwen\qwenNoContext.txt
qwen\qwenDead.txt
qwen\qwenLimited.txt
qwen\qwenRestricted.txt
qwen\qwenNetwork.txt
qwen\qwenUnknown.txt
qwen\qwenChecked.txt
qwen\qwenResults.jsonl
```
## Kimi / Moonshot AI
```powershell
python supervisor.py --config config.yaml --cmd "recheck kimi all --max-keys 100"
```
The explicit `MOONSHOT_API_KEY` / `KIMI_API_KEY` detector is validated without generation by calling `GET /v1/users/me/balance` on the independent global and China Moonshot endpoints.
Output folder:
```text
kimi\kimiAlive.txt
kimi\kimiNoBalance.txt
kimi\kimiDead.txt
kimi\kimiLimited.txt
kimi\kimiRestricted.txt
kimi\kimiNetwork.txt
kimi\kimiUnknown.txt
kimi\kimiChecked.txt
kimi\kimiResults.jsonl
```
## Groq
```powershell
python supervisor.py --config config.yaml --cmd "recheck groq all --max-keys 100"
```
The checker validates TruffleHog `Groq` findings with `GET https://api.groq.com/openai/v1/models` and does not run generation probes.
Output folder:
```text
groq\groqAlive.txt
groq\groqDead.txt
groq\groqLimited.txt
groq\groqRestricted.txt
groq\groqNetwork.txt
groq\groqUnknown.txt
groq\groqChecked.txt
groq\groqResults.jsonl
```
## Replicate / xAI / HuggingFace
```powershell
python supervisor.py --config config.yaml --cmd "recheck replicate all --max-keys 100"
python supervisor.py --config config.yaml --cmd "recheck xai all --max-keys 100"
python supervisor.py --config config.yaml --cmd "recheck huggingface all --max-keys 100"
```
These checkers validate built-in TruffleHog findings through non-generating endpoints: Replicate account lookup, xAI model list, and HuggingFace whoami.
## Anthropic
```powershell
python supervisor.py --config config.yaml --cmd "recheck anthropic all --max-keys 100"
```
Output folder:
```text
anthropic\anthropicAlive.txt
anthropic\anthropicNoQuota.txt
anthropic\anthropicDead.txt
anthropic\anthropicLimited.txt
anthropic\anthropicRestricted.txt
anthropic\anthropicNetwork.txt
anthropic\anthropicUnknown.txt
anthropic\anthropicChecked.txt
anthropic\anthropicResults.jsonl
```
## AWS
Default mode only checks STS identity:
```powershell
python supervisor.py --config config.yaml --cmd "recheck aws all --max-keys 100"
```
Optional Bedrock probing is configured under `keychecks.service_args.aws`, then run:
```powershell
python supervisor.py --config config.yaml --cmd "recheck aws all --max-keys 100"
```
Output folder:
```text
aws\awsAlive.txt
aws\awsBedrock.txt
aws\awsAdmin.txt
aws\awsCanary.txt
aws\awsQuarantined.txt
aws\awsAccessDenied.txt
aws\awsDead.txt
aws\awsNetwork.txt
aws\awsUnknown.txt
aws\awsChecked.txt
aws\awsResults.jsonl
```
Canary AWS credentials are detected before active AWS probes when TruffleHog provides `ExtraData.is_canary` / canary message. If metadata is absent, STS ARN containing `canarytokens` is also classified as `awsCanary.txt` and IAM/Bedrock probes are skipped.
## Azure
```powershell
python supervisor.py --config config.yaml --cmd "recheck azure all --max-keys 100"
```
This checks Azure service-principal findings from `DetectorName=Azure` using `tenantId`, `clientId`, `clientSecret` from `RawV2`.
`DetectorName=AzureOpenAI` is placed into `azureOpenAIUnresolved.txt` unless an endpoint/resource name is available.
Output folder:
```text
azure\azureAlive.txt
azure\azureDead.txt
azure\azureRestricted.txt
azure\azureNetwork.txt
azure\azureUnknown.txt
azure\azureOpenAIUnresolved.txt
azure\azureChecked.txt
azure\azureResults.jsonl
```
## Retry Flags
Common flags:
```text
--retry-network
--retry-limited
--retry-unknown
--recheck-all
--max-keys N
```
Network/proxy failures are committed with the `network` status group and can be selected for a bounded PostgreSQL recheck. `*Network.txt` is only its asynchronous compatibility projection.
+4555
View File
File diff suppressed because it is too large Load Diff
+13
View File
@@ -0,0 +1,13 @@
"""Retired legacy mutation UI.
Scanner lifecycle control is intentionally available only through supervisor.py.
The read-only observability UI remains dashboard.py.
"""
import streamlit as st
st.set_page_config(page_title='Scanner UI Retired', page_icon='LOCK', layout='centered')
st.title('Legacy scanner controls are retired')
st.error('This UI cannot start, pause, resume, cancel, or configure scans.')
st.info('Use the authenticated supervisor commands for lifecycle control and dashboard.py for read-only observability.')
+15
View File
@@ -0,0 +1,15 @@
"""Retired direct GitHub credential audit entrypoint."""
import sys
sys.dont_write_bytecode = True
def main():
raise SystemExit(
'This direct credential audit is retired. Use authenticated supervisor-managed GitHub keychecks.'
)
if __name__ == '__main__':
main()
+55
View File
@@ -0,0 +1,55 @@
MAX_RESULT_BUNDLE_BYTES = 64 * 1024 * 1024
REMOTE_ASSIGNMENT_BASELINE_BYTES = 2 * 1024 * 1024
REMOTE_ASSIGNMENT_MAX_ACTIVE = 50
def validate_remote_assignment_capacity(config):
values = {}
fields = (
'result_bundle_max_event_bytes',
'remote_assignment_reserve_bytes',
'remote_assignment_max_active',
'result_bundle_max_total_bytes',
'projection_backlog_max_bytes',
'projection_backlog_headroom_bytes',
'keycheck_queue_max_items',
'keycheck_queue_max_bytes',
'keycheck_candidates_per_event',
'keycheck_candidate_bytes_per_event',
)
for name in fields:
value = config.get(name)
if type(value) is not int or value < 0:
raise ValueError(name)
values[name] = value
hard_limit = values['result_bundle_max_event_bytes']
reserve = values['remote_assignment_reserve_bytes']
active = values['remote_assignment_max_active']
if not REMOTE_ASSIGNMENT_BASELINE_BYTES <= reserve <= hard_limit:
raise ValueError('remote_assignment_reserve_bytes')
if not REMOTE_ASSIGNMENT_BASELINE_BYTES <= hard_limit <= MAX_RESULT_BUNDLE_BYTES:
raise ValueError('result_bundle_max_event_bytes')
if not 1 <= active <= REMOTE_ASSIGNMENT_MAX_ACTIVE:
raise ValueError('remote_assignment_max_active')
required_bytes = active * reserve
if values['result_bundle_max_total_bytes'] < required_bytes:
raise ValueError('result_bundle_max_total_bytes')
projection_admission_bytes = (
values['projection_backlog_max_bytes']
- values['projection_backlog_headroom_bytes']
)
if projection_admission_bytes < required_bytes:
raise ValueError('projection_backlog_max_bytes')
if (
values['keycheck_queue_max_items']
< active * values['keycheck_candidates_per_event']
):
raise ValueError('keycheck_queue_max_items')
if (
values['keycheck_queue_max_bytes']
< active * values['keycheck_candidate_bytes_per_event']
):
raise ValueError('keycheck_queue_max_bytes')
return values
+486
View File
@@ -0,0 +1,486 @@
"""Stdlib-only authentication boundary for supervised application children."""
import sys
sys.dont_write_bytecode = True
if not sys.dont_write_bytecode:
raise RuntimeError('supervised child bootstrap could not disable bytecode writes')
import hashlib
import hmac
import json
import os
import runpy
import socket
import stat
MAX_METADATA_BYTES = 256 * 1024
MAX_HANDSHAKE_BYTES = 1024 * 1024
MAX_PYVENV_BYTES = 64 * 1024
MANIFEST_SCHEMA = 5
APPLICATION_IMPORT_SUFFIXES = ('.py', '.pyw', '.pyc', '.pyd')
CONTROL_SCHEMA = 1
REQUIRED_DEPENDENCIES = {
'supervisor': ('psycopg', 'yaml'),
'postgres-runtime': ('psycopg', 'yaml'),
'migrate-runtime-safety': ('psycopg', 'yaml'),
'scanner': ('psycopg', 'requests', 'urllib3', 'yaml', 'zstandard'),
'discovery-producer': ('psycopg', 'requests', 'urllib3', 'yaml', 'zstandard'),
'docker-shadow': ('psycopg', 'requests', 'urllib3', 'yaml', 'zstandard'),
'keycheck': ('psycopg', 'requests', 'yaml'),
'dashboard': ('pandas', 'plotly', 'psycopg', 'streamlit', 'yaml'),
'keycheck-provider': ('boto3', 'botocore', 'psycopg', 'requests', 'yaml'),
'janitor': ('yaml',),
'result-ingester': ('psycopg', 'yaml'),
'jsonl-projector': ('psycopg', 'yaml'),
'worker-api': ('psycopg', 'requests', 'starlette', 'urllib3', 'uvicorn', 'yaml', 'zstandard'),
}
class ChildRuntimeError(RuntimeError):
pass
ENV = {
'instance_file': 'TRUF_SUPERVISOR_INSTANCE_FILE',
'instance_id': 'TRUF_SUPERVISOR_INSTANCE_ID',
'token': 'TRUF_SUPERVISOR_TOKEN',
'config_sha256': 'TRUF_SUPERVISOR_CONFIG_SHA256',
'supervisor_sha256': 'TRUF_SUPERVISOR_SHA256',
'code_manifest_sha256': 'TRUF_SUPERVISOR_CODE_MANIFEST_SHA256',
'dsn_sha256': 'TRUF_SUPERVISOR_DSN_SHA256',
'kind': 'TRUF_SUPERVISOR_CHILD_KIND',
}
def _canonical(path):
return os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(path))))
def _is_reparse_point(path):
details = os.lstat(path)
if stat.S_ISLNK(details.st_mode):
return True
attributes = getattr(details, 'st_file_attributes', 0)
reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0)
return bool(attributes & reparse_attribute) or getattr(os.path, 'isjunction', lambda _path: False)(path)
def _contained(path, roots):
for root in roots:
try:
if path != root and os.path.commonpath((root, path)) == root:
return True
except ValueError:
continue
return False
def _validated_site_directory(path, roots):
if not path or not os.path.isdir(path):
return ''
candidate = _canonical(path)
trusted_roots = tuple(_canonical(root) for root in roots if root)
if os.path.basename(candidate).lower() not in ('site-packages', 'dist-packages'):
raise RuntimeError(f'interpreter dependency path is not a site-packages directory: {candidate}')
if not _contained(candidate, trusted_roots):
raise RuntimeError(f'interpreter dependency path escapes its trusted root: {candidate}')
return candidate
def _append_site_directories(paths, roots):
existing = {_canonical(path) for path in sys.path if path}
for path in paths:
candidate = _validated_site_directory(path, roots)
if candidate and candidate not in existing:
# Direct insertion intentionally does not evaluate .pth hook lines.
sys.path.append(candidate)
existing.add(candidate)
def _venv_configuration():
# Preserve a venv's bin/python symlink location while locating pyvenv.cfg.
executable_dir = os.path.dirname(os.path.normcase(os.path.abspath(sys.executable)))
roots = [executable_dir]
if os.path.basename(executable_dir).lower() in ('bin', 'scripts'):
roots.insert(0, os.path.dirname(executable_dir))
for root in roots:
config_path = os.path.join(root, 'pyvenv.cfg')
if not os.path.isfile(config_path):
continue
details = os.stat(config_path, follow_symlinks=False)
if not stat.S_ISREG(details.st_mode) or details.st_size > MAX_PYVENV_BYTES:
raise RuntimeError('interpreter pyvenv.cfg is not a bounded regular file')
with open(config_path, 'rb') as handle:
payload = handle.read(MAX_PYVENV_BYTES + 1)
if len(payload) > MAX_PYVENV_BYTES:
raise RuntimeError('interpreter pyvenv.cfg exceeds its byte bound')
include_system = False
for raw_line in payload.decode('utf-8', errors='strict').splitlines():
key, separator, value = raw_line.partition('=')
if separator and key.strip().lower() == 'include-system-site-packages':
include_system = value.strip().lower() in ('1', 'true', 'yes')
return _canonical(root), include_system
return '', True
def _venv_site_directories(root):
if os.name == 'nt':
return [os.path.join(root, 'Lib', 'site-packages')]
version = f'python{sys.version_info.major}.{sys.version_info.minor}'
return [
os.path.join(root, library, version, name)
for library in ('lib', 'lib64')
for name in ('site-packages', 'dist-packages')
]
def _system_site_directories():
import sysconfig
roots = tuple(dict.fromkeys((_canonical(sys.base_prefix), _canonical(sys.base_exec_prefix))))
paths = sysconfig.get_paths(vars={
'base': sys.base_prefix,
'platbase': sys.base_exec_prefix,
})
return [paths.get('purelib'), paths.get('platlib')], roots
def _user_site_directories():
if os.name == 'nt':
try:
import ctypes
appdata = ctypes.create_unicode_buffer(32768)
if ctypes.windll.shell32.SHGetFolderPathW(None, 0x001A, None, 0, appdata) != 0:
return [], ()
root = _canonical(os.path.join(appdata.value, 'Python'))
version = f'Python{sys.version_info.major}{sys.version_info.minor}'
return [os.path.join(root, version, 'site-packages')], (root,)
except (AttributeError, OSError, ValueError):
return [], ()
try:
import pwd
home = _canonical(pwd.getpwuid(os.getuid()).pw_dir)
except (ImportError, KeyError, OSError):
return [], ()
version = f'python{sys.version_info.major}.{sys.version_info.minor}'
if sys.platform == 'darwin':
root = _canonical(os.path.join(home, 'Library', 'Python', f'{sys.version_info.major}.{sys.version_info.minor}'))
return [os.path.join(root, 'lib', 'python', 'site-packages')], (root,)
root = _canonical(os.path.join(home, '.local'))
return [
os.path.join(root, 'lib', version, 'site-packages'),
os.path.join(root, 'lib', version, 'dist-packages'),
], (root,)
def _missing_dependencies(kind):
import importlib.util
return [name for name in REQUIRED_DEPENDENCIES[kind] if importlib.util.find_spec(name) is None]
def _enable_dependency_paths(kind):
venv_root, include_system = _venv_configuration()
if venv_root:
_append_site_directories(_venv_site_directories(venv_root), (venv_root,))
if include_system:
system_paths, system_roots = _system_site_directories()
_append_site_directories(system_paths, system_roots)
if _missing_dependencies(kind):
user_paths, user_roots = _user_site_directories()
_append_site_directories(user_paths, user_roots)
missing = _missing_dependencies(kind)
if missing:
raise RuntimeError('required authenticated child dependencies are unavailable: ' + ', '.join(missing))
def _sha256_file(path):
digest = hashlib.sha256()
with open(path, 'rb') as handle:
while True:
block = handle.read(1024 * 1024)
if not block:
return digest.hexdigest()
digest.update(block)
def _read_object(path):
details = os.stat(path, follow_symlinks=False)
if not stat.S_ISREG(details.st_mode) or details.st_size <= 0 or details.st_size > MAX_METADATA_BYTES:
raise RuntimeError('supervisor metadata is not a bounded regular file')
with open(path, 'rb') as handle:
payload = handle.read(MAX_METADATA_BYTES + 1)
if len(payload) > MAX_METADATA_BYTES:
raise RuntimeError('supervisor metadata exceeds its byte bound')
value = json.loads(payload.decode('utf-8'))
if not isinstance(value, dict):
raise RuntimeError('supervisor metadata root is invalid')
return value
def _manifest_digest(manifest):
payload = json.dumps(manifest, ensure_ascii=True, sort_keys=True, separators=(',', ':')).encode('utf-8')
return hashlib.sha256(payload).hexdigest()
def _application_code_files(root):
names = set()
def raise_walk_error(exc):
raise RuntimeError(f'unable to inspect the application root: {exc}') from exc
for current, directories, files in os.walk(root, followlinks=False, onerror=raise_walk_error):
for name in directories:
candidate = os.path.join(current, name)
if _is_reparse_point(candidate):
relative = os.path.relpath(candidate, root).replace(os.sep, '/')
raise RuntimeError(f'application directory reparse point is forbidden: {relative}')
relative_current = os.path.relpath(current, root)
in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep))
suffixes = ('.pyc',) if in_cache else APPLICATION_IMPORT_SUFFIXES
for name in files:
source_path = os.path.abspath(os.path.join(current, name))
if _is_reparse_point(source_path):
relative = os.path.relpath(source_path, root).replace(os.sep, '/')
raise RuntimeError(f'application file reparse point is forbidden: {relative}')
if not name.lower().endswith(suffixes):
continue
path = _canonical(source_path)
try:
contained = os.path.commonpath((root, path)) == root
except ValueError:
contained = False
if not contained:
raise RuntimeError('application Python authority escapes its root')
names.add(os.path.relpath(source_path, root).replace(os.sep, '/'))
return names
def _reject_cached_bytecode(root):
def raise_walk_error(exc):
raise RuntimeError(f'unable to inspect the application root: {exc}') from exc
try:
root_details = os.lstat(root)
except OSError as exc:
raise RuntimeError(f'application root is unavailable: {root}') from exc
if _is_reparse_point(root):
raise RuntimeError(f'application root reparse point is forbidden: {root}')
if not stat.S_ISDIR(root_details.st_mode):
raise RuntimeError(f'application root is not a directory: {root}')
for current, directories, files in os.walk(root, followlinks=False, onerror=raise_walk_error):
for name in directories:
candidate = os.path.join(current, name)
if _is_reparse_point(candidate):
relative = os.path.relpath(candidate, root).replace(os.sep, '/')
if name.lower() == '__pycache__':
raise RuntimeError(f'application __pycache__ link is forbidden: {relative}')
raise RuntimeError(f'application directory reparse point is forbidden: {relative}')
relative_current = os.path.relpath(current, root)
in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep))
for name in files:
candidate = os.path.join(current, name)
relative = os.path.relpath(candidate, root).replace(os.sep, '/')
if _is_reparse_point(candidate):
raise RuntimeError(f'application file reparse point is forbidden: {relative}')
if in_cache and name.lower().endswith('.pyc'):
raise RuntimeError(f'application __pycache__ bytecode is forbidden: {relative}')
def _verify_manifest(metadata, inherited):
manifest = metadata.get('code_manifest')
if not isinstance(manifest, dict) or manifest.get('schema') != MANIFEST_SCHEMA:
raise RuntimeError('unsupported child code manifest')
expected_digest = str(metadata.get('code_manifest_sha256') or '')
if not hmac.compare_digest(_manifest_digest(manifest), expected_digest):
raise RuntimeError('child code manifest digest mismatch')
if not hmac.compare_digest(expected_digest, inherited['code_manifest_sha256']):
raise RuntimeError('inherited child code manifest mismatch')
root_value = manifest.get('root') or ''
files = manifest.get('files')
executables = manifest.get('executables')
assets = manifest.get('assets')
if not root_value or not isinstance(files, dict) or not isinstance(executables, dict) or not isinstance(assets, dict):
raise RuntimeError('child code manifest is incomplete')
raw_root = os.path.abspath(os.fspath(root_value))
_reject_cached_bytecode(raw_root)
root = _canonical(raw_root)
manifested_code = set()
for name, value in files.items():
if not isinstance(value, dict):
raise RuntimeError('child code manifest file entry is invalid')
expected_path = _canonical(os.path.join(root, *str(name).split('/')))
path = _canonical(value.get('path') or '')
if path != expected_path or not hmac.compare_digest(_sha256_file(path), str(value.get('sha256') or '')):
raise RuntimeError(f'child code authority drifted: {name}')
try:
contained = os.path.commonpath((root, path)) == root
except ValueError:
contained = False
if contained and str(name).lower().endswith(APPLICATION_IMPORT_SUFFIXES):
manifested_code.add(str(name).replace('\\', '/'))
current_code = _application_code_files(root)
if current_code != manifested_code:
added = sorted(current_code - manifested_code)
removed = sorted(manifested_code - current_code)
detail = added[0] if added else removed[0] if removed else 'unknown'
raise RuntimeError(f'application code authority file set drifted: {detail}')
for group_name, values in (('executable', executables), ('asset', assets)):
for name, value in values.items():
if not isinstance(value, dict):
raise RuntimeError(f'child {group_name} authority entry is invalid')
path = _canonical(value.get('path') or '')
if not os.path.isabs(path) or not hmac.compare_digest(_sha256_file(path), str(value.get('sha256') or '')):
raise RuntimeError(f'child {group_name} authority drifted: {name}')
return root
def _handshake(metadata):
control = metadata.get('control') or {}
request = {
'schema': CONTROL_SCHEMA,
'instance_id': metadata['instance_id'],
'token': metadata['token'],
'action': 'handshake',
}
encoded = json.dumps(request, ensure_ascii=True, separators=(',', ':')).encode('utf-8') + b'\n'
chunks = []
total = 0
with socket.create_connection((control.get('host'), int(control.get('port') or 0)), timeout=3) as client:
client.settimeout(3)
client.sendall(encoded)
client.shutdown(socket.SHUT_WR)
while True:
chunk = client.recv(65536)
if not chunk:
break
total += len(chunk)
if total > MAX_HANDSHAKE_BYTES:
raise RuntimeError('supervisor handshake exceeds its byte bound')
chunks.append(chunk)
response = json.loads(b''.join(chunks).decode('utf-8'))
if (
not isinstance(response, dict)
or response.get('schema') != CONTROL_SCHEMA
or response.get('instance_id') != metadata['instance_id']
or response.get('ok') is not True
or not isinstance(response.get('result'), dict)
):
raise RuntimeError('authenticated supervisor handshake failed')
return response['result']
def _authenticate(kind):
inherited = {name: str(os.getenv(variable) or '') for name, variable in ENV.items()}
if not all(inherited.values()):
raise RuntimeError('direct mutation is retired; use an authenticated active supervisor command')
if inherited['kind'] != kind:
raise RuntimeError('supervised child kind does not match the bootstrap entrypoint')
instance_file = _canonical(inherited['instance_file'])
metadata = _read_object(instance_file)
if metadata.get('schema') != 2 or _canonical(metadata.get('instance_file') or '') != instance_file:
raise RuntimeError('supervisor child instance metadata is invalid')
for key in ('instance_id', 'token', 'config_sha256', 'supervisor_sha256', 'code_manifest_sha256'):
if not hmac.compare_digest(str(metadata.get(key) or ''), inherited[key]):
raise RuntimeError(f'supervisor child {key} authority mismatch')
if str(metadata.get('activation_state') or '').upper() != 'ACTIVE':
raise RuntimeError('supervisor is not ACTIVE; child launch is refused')
if not hmac.compare_digest(_sha256_file(metadata['config_path']), inherited['config_sha256']):
raise RuntimeError('supervisor config authority drifted')
if not hmac.compare_digest(_sha256_file(metadata['supervisor_path']), inherited['supervisor_sha256']):
raise RuntimeError('supervisor script authority drifted')
root = _verify_manifest(metadata, inherited)
dsn = str(os.getenv('TRUF_MANAGED_POSTGRES_DSN') or '')
if kind == 'janitor':
if dsn or os.getenv('SCANNER_DB_URL') or os.getenv('DATABASE_URL'):
raise RuntimeError('janitor child must not receive database mutation capability')
else:
dsn_digest = hashlib.sha256(dsn.encode('utf-8')).hexdigest() if dsn else ''
if not dsn or not hmac.compare_digest(dsn_digest, inherited['dsn_sha256']):
raise RuntimeError('managed PostgreSQL DSN authority mismatch')
for variable in ('SCANNER_DB_URL', 'DATABASE_URL'):
if not hmac.compare_digest(str(os.getenv(variable) or ''), dsn):
raise RuntimeError(f'{variable} does not match managed PostgreSQL authority')
handshake = _handshake(metadata)
if handshake.get('activation_state') != 'ACTIVE' or handshake.get('instance_id') != metadata['instance_id']:
raise RuntimeError('supervisor handshake did not confirm ACTIVE authority')
for key in ('config_sha256', 'supervisor_sha256', 'code_manifest_sha256'):
if not hmac.compare_digest(str(handshake.get(key) or ''), inherited[key]):
raise RuntimeError(f'supervisor handshake {key} mismatch')
if not hmac.compare_digest(str(handshake.get('canonical_dsn_sha256') or ''), inherited['dsn_sha256']):
raise RuntimeError('supervisor handshake PostgreSQL authority mismatch')
return root, metadata
def main():
if not sys.flags.isolated or not sys.flags.no_site or not sys.flags.dont_write_bytecode:
raise RuntimeError('supervised child bootstrap requires isolated no-site bytecode-free startup (-I -S -B)')
if len(sys.argv) < 2:
raise SystemExit('supervised child bootstrap kind is required')
kind = str(sys.argv[1]).strip().lower()
root, metadata = _authenticate(kind)
_enable_dependency_paths(kind)
arguments = list(sys.argv[2:])
if kind != 'keycheck-provider' and arguments[:1] == ['--']:
arguments.pop(0)
if kind in ('scanner', 'discovery-producer'):
entrypoint = os.path.join(root, 'console_runner.py')
elif kind == 'docker-shadow':
entrypoint = os.path.join(root, 'docker_shadow.py')
elif kind == 'keycheck':
entrypoint = os.path.join(root, 'keycheck_runner.py')
elif kind == 'dashboard':
entrypoint = os.path.join(root, 'dashboard.py')
elif kind == 'janitor':
entrypoint = os.path.join(root, 'janitor.py')
elif kind == 'result-ingester':
entrypoint = os.path.join(root, 'result_ingester.py')
elif kind == 'jsonl-projector':
entrypoint = os.path.join(root, 'jsonl_projector.py')
elif kind == 'worker-api':
entrypoint = os.path.join(root, 'worker_api.py')
elif kind == 'keycheck-provider':
if not arguments:
raise RuntimeError('keycheck provider bootstrap entrypoint is required')
relative = arguments.pop(0).replace('\\', '/')
if not arguments or arguments.pop(0) != '--':
raise RuntimeError('keycheck provider bootstrap separator is required')
if '--' in arguments:
raise RuntimeError('duplicate keycheck provider bootstrap separator')
entrypoint = _canonical(os.path.join(root, *relative.split('/')))
provider_root = _canonical(os.path.join(root, 'keycheckers'))
try:
allowed = os.path.commonpath((provider_root, entrypoint)) == provider_root
except ValueError:
allowed = False
if not allowed or not relative.lower().endswith('.py'):
raise RuntimeError('keycheck provider bootstrap entrypoint is outside authority')
else:
raise RuntimeError('unsupported supervised child bootstrap kind')
entrypoint = _canonical(entrypoint)
files = (metadata.get('code_manifest') or {}).get('files') or {}
if not any(_canonical(value.get('path') or '') == entrypoint for value in files.values() if isinstance(value, dict)):
raise RuntimeError('child entrypoint is absent from immutable authority')
sys.path.insert(0, root)
try:
if kind == 'dashboard':
sys.argv = ['streamlit', 'run', entrypoint, *arguments]
runpy.run_module('streamlit', run_name='__main__', alter_sys=True)
else:
sys.argv = [entrypoint, *arguments]
runpy.run_path(entrypoint, run_name='__main__')
except Exception as exc:
raise ChildRuntimeError(str(exc)) from exc
if __name__ == '__main__':
try:
main()
except ChildRuntimeError as exc:
raise SystemExit(f'supervised child runtime failed: {exc}') from exc
except Exception as exc:
raise SystemExit(f'supervised child bootstrap rejected launch: {exc}') from exc
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+127
View File
@@ -0,0 +1,127 @@
"""Pure translation of the reviewed Windows config; YAML and file copies are caller-owned."""
from copy import deepcopy
import ntpath
# Only these reviewed absolute Windows paths have known container replacements.
_FIXED_GLOBAL_PATHS = {
'root_dir': (r'D:\truf', '/opt/truf'),
'project_dir': (r'D:\truf\app', '/opt/truf/app'),
'runtime_dir': (r'D:\truf\runtime', '/data/runtime-linux'),
'postgres_data_dir': (r'S:\postgres-data', '/data/postgres-linux'),
'postgres_bin_dir': (r'D:\truf\runtime\postgres\pgsql\bin', '/usr/lib/postgresql/16/bin'),
'result_bundle_dir': (r'S:\scanner-result-bundles', '/data/scanner-result-bundles'),
'work_dir': (r'S:\scanner-work', '/data/scanner-work'),
'control_dir': (r'D:\truf\runtime\control', '/run/truf/control'),
'trufflehog_path': (r'C:\Tools\trufflehog.exe', '/usr/local/bin/trufflehog'),
'proxy_file': (r'D:\truf\runtime\proxy.txt', '/data/runtime-linux/proxy.txt'),
'secrets_file': (r'D:\truf\app\secrets.yaml', '/data/config/secrets.yaml'),
'trufflehog_config': (
r'D:\truf\app\trufflehog-custom-detectors.yaml',
'/data/config/trufflehog-custom-detectors.yaml',
),
}
_PATH_FIELDS = {
'global': (
'result_spool_dir', 'legacy_result_spool_dir', 'results_dir', 'queue_dir',
'state_dir', 'log_dir', 'keycheck_dir', 'postman_cache_dir', 'gharchive_cache_dir',
'database_path', 'state_file', 'api_proxy_file', 'download_proxy_file',
'dashboard_db_path', 'scan_limiter_db', 'dockerhub_tag_cache_path',
),
'supervisor': ('log_dir', 'supervisor_log', 'status_file', 'dashboard_log', 'state_dir'),
'keychecks': ('input', 'proxy_file', 'keycheck_dir', 'summary_tsv', 'summary_json', 'alive_summary_tsv'),
}
_SOURCE_PATH_FIELDS = ('target_file', 'trufflehog_config', 'postman_cache_dir', 'gharchive_cache_dir')
_WINDOWS_KNOBS = (
'trufflehog_job_memory_limit_bytes',
'trufflehog_windows_job_cpu_weight',
'trufflehog_windows_memory_priority',
)
def translate_windows_config(original, baseline):
"""Return an independent config and sorted, changed dotted key paths (never values).
``baseline`` is the parsed config.linux.yaml, not a general merge source.
Fixed paths must match the container contract. Other known path fields keep
relative paths/templates with Linux separators; unreviewed Windows absolute
paths raise ValueError naming only the key. No environment, filesystem,
runtime, database, or YAML operations are performed.
"""
if not isinstance(original, dict) or not isinstance(baseline, dict):
raise TypeError('original and baseline must be dictionaries')
config = deepcopy(original)
adjusted = set()
def assign(mapping, key, value, prefix):
if key not in mapping or mapping[key] != value:
mapping[key] = deepcopy(value)
adjusted.add(prefix + '.' + key)
def path_value(value, key_path, approved=None):
if value is None:
return value
if not isinstance(value, str):
raise ValueError('Expected path string at ' + key_path)
if ntpath.splitdrive(value)[0] or value.startswith('\\'):
if approved is None or ntpath.normcase(ntpath.normpath(value)) != ntpath.normcase(ntpath.normpath(approved[0])):
raise ValueError('Unsupported Windows path at ' + key_path)
return approved[1]
return value.replace('\\', '/')
linux_paths = {key: pair[1] for key, pair in _FIXED_GLOBAL_PATHS.items()}
control = _FIXED_GLOBAL_PATHS['control_dir']
supervisor_paths = {'control_dir': control}
for key, filename in (('instance_file', 'supervisor.instance.json'), ('lock_file', 'supervisor.lock')):
supervisor_paths[key] = (control[0] + '\\' + filename, control[1] + '/' + filename)
for section, fixed_paths in (('global', _FIXED_GLOBAL_PATHS), ('supervisor', supervisor_paths)):
mapping = config.setdefault(section, {})
for key, pair in fixed_paths.items():
key_path = section + '.' + key
value = path_value(mapping.get(key), key_path, pair)
if key == 'trufflehog_path' and value not in (None, '', 'trufflehog', 'trufflehog.exe', pair[1]):
raise ValueError('Unsupported executable path at ' + key_path)
# The copied original policy intentionally replaces the image policy.
if key != 'trufflehog_config':
try:
baseline_path = baseline[section][key].format_map(linux_paths)
except (KeyError, AttributeError, ValueError):
raise ValueError('Invalid Linux baseline path at ' + key_path) from None
if baseline_path != pair[1]:
raise ValueError('Invalid Linux baseline path at ' + key_path)
assign(mapping, key, pair[1], section)
groups = [(section, config.get(section, {}), fields) for section, fields in _PATH_FIELDS.items()]
groups.extend(('sources.' + name, source, _SOURCE_PATH_FIELDS)
for name, source in config.get('sources', {}).items())
for prefix, mapping, fields in groups:
for key in fields:
if key not in mapping:
continue
approved = None
if key in ('proxy_file', 'api_proxy_file', 'download_proxy_file'):
approved = _FIXED_GLOBAL_PATHS['proxy_file']
elif key == 'trufflehog_config':
approved = _FIXED_GLOBAL_PATHS[key]
value = path_value(mapping[key], prefix + '.' + key, approved)
if key == 'trufflehog_config' and value in (
'{project_dir}/trufflehog-custom-detectors.yaml', 'trufflehog-custom-detectors.yaml',
):
value = linux_paths[key]
assign(mapping, key, value, prefix)
for key in ('max_active_scans', 'opportunistic_scan_slots') + _WINDOWS_KNOBS:
assign(config['global'], key, baseline['global'][key], 'global')
for key in ('interactive', 'autostart', 'control_host', 'control_port'):
assign(config['supervisor'], key, baseline['supervisor'][key], 'supervisor')
assign(config['supervisor'].setdefault('dashboard', {}), 'enabled',
baseline['supervisor']['dashboard']['enabled'], 'supervisor.dashboard')
for name, source in config.get('sources', {}).items():
source_baseline = baseline.get('sources', {}).get(name, {})
for key in _WINDOWS_KNOBS:
if key in source or key in source_baseline:
assign(source, key, source_baseline.get(key, baseline['global'][key]), 'sources.' + name)
return config, sorted(adjusted)
+416
View File
@@ -0,0 +1,416 @@
"""Explicit maintenance-only loss acknowledgement for the reviewed copied output.
No CLI, startup hook, connection creation, PostgreSQL lifecycle, or output rebuild.
The caller must retain initialize.lock and ClusterAuthorityLock through this call
AND subsequent positively verified maintenance stop, including every exception or
uncertain commit. It must keep the target isolated with no workers/network clients.
Use an idle, writable, autocommit=True psycopg connection with dict_row and quiet server log
settings. The supplied runtime is the prepared container_runtime module, not its
main()/initialize() entrypoint. Never call this during normal initialized startup.
The immutable PREPARED journal describes intent, not a fabricated completed rename.
An exact before state can be applied; an exact after state is a read-only retry.
Partial journals and later output require review, never automatic cleanup/rewind.
"""
import hashlib
import json
import os
from pathlib import Path
import re
import time
from datetime import datetime, timezone
DATA = Path('/data')
RUN = Path('/run/truf')
APPROVED_MANIFEST_SHA256 = '08344147133c37d4b6f404cf4fac3e59d58f94917f1fa58a77cbb68c36db7e8a'
JOURNAL_NAME = 'found-secrets-loss-g13-g14.prepared.json'
FORMAT = 'truf-found-secrets-loss-g13-g14-v1'
LOSS = 'Previously published copied output intentionally lost; all PostgreSQL history retained. No rotation or rename occurred.'
OLD_GENERATION, NEW_GENERATION, OLD_OFFSET = 13, 14, 97783145
APPEND_COUNT, ROTATION_COUNT = 38024, 13
MAX_JOURNAL = 1024 * 1024
STREAM_COLUMNS = {'stream_name', 'base_relative_path', 'current_generation', 'rotation_bytes',
'max_generations', 'created_at', 'updated_at'}
CURSOR_COLUMNS = {'stream_name', 'generation', 'committed_offset', 'last_append_id', 'last_job_id',
'last_event_id', 'last_event_hash', 'updated_at'}
class ProjectionRecoveryError(RuntimeError):
"""Safe diagnostic only; caller still owns maintenance/stop authority."""
def _encoded(value):
return (json.dumps(value, ensure_ascii=True, sort_keys=True,
separators=(',', ':'), allow_nan=False) + '\n').encode('ascii')
def recover_found_secrets_projection(runtime, connection, *, system_identifier, manifest_sha256,
initialize_lock, authority_lock):
"""Apply only the pinned g13 loss transition, or recognize its exact retry.
Caller-owned locks must be acquired runtime_security lock objects for the fixed
target. This function never closes the connection or releases those locks.
Its own file lock and transaction-scoped advisory/table locks exclude writers.
Returned 'committed'/'already-committed' is not target readiness or stop proof.
journal_sha256 hashes the complete immutable file, not just its inner record.
"""
stage = 'preflight'
deadline = time.monotonic() + 10800
try:
from container_import import _fsync_dir, _identifier, _input, _integer, _json, _manifest, _regular, _write, MAX_MANIFEST
from runtime_security import ClusterAuthorityLock, PrivateFileLock
runtime.require_container()
if (runtime.DATA != DATA or runtime.RUN != RUN
or runtime.INITIALIZED != DATA / 'initialized.json'
or runtime.INITIALIZE_LOCK != DATA / 'initialize.lock'
or manifest_sha256 != APPROVED_MANIFEST_SHA256
or not isinstance(system_identifier, str)
or re.fullmatch(r'[1-9][0-9]{0,19}', system_identifier) is None
or connection.closed or connection.broken or connection.autocommit is not True
or int(connection.info.transaction_status) != 0):
raise ValueError()
config_dir, results = DATA / 'config', DATA / 'runtime-linux/results'
identity_path = DATA / 'runtime-linux/postgres/cluster_identity.json'
manifest_path = config_dir / 'windows-import-manifest.json'
journal_path = config_dir / JOURNAL_NAME
projector_lock = results / '.jsonl-projector.lock'
endpoint = _encoded({'database': 'truf', 'host': '127.0.0.1', 'port': 5432,
'schema': 'public'}).decode('ascii').strip()
data_hash = hashlib.sha256(str(DATA / 'postgres-linux').encode('utf-8')).hexdigest()
endpoint_hash = hashlib.sha256(endpoint.encode('ascii')).hexdigest()
def files_stopped():
if (time.monotonic() >= deadline or getattr(runtime, '_shutdown_requested', True) is not False
or not isinstance(initialize_lock, PrivateFileLock) or initialize_lock.acquired is not True
or initialize_lock.path != os.path.normcase(str(runtime.INITIALIZE_LOCK))
or not isinstance(authority_lock, ClusterAuthorityLock) or authority_lock.acquired is not True
or authority_lock.data_directory != str(DATA / 'postgres-linux')
or authority_lock.endpoint_identity != endpoint
or authority_lock.path != str(DATA / 'runtime-linux/postgres' / f'.cluster-authority-{data_hash}.lock')
or authority_lock.endpoint_path != str(RUN / 'authority' / f'endpoint-{endpoint_hash}.lock')):
raise ValueError()
for directory in (DATA, config_dir, results, identity_path.parent,
DATA / 'runtime-linux/logs', RUN, RUN / 'control'):
runtime.private_path(directory, directory=True)
_regular(runtime.INITIALIZE_LOCK, runtime)
for path in (runtime.INITIALIZED, RUN / 'control/supervisor.instance.json',
RUN / 'control/supervisor.pid', DATA / 'runtime-linux/logs/supervisor.instance.json',
DATA / 'runtime-linux/logs/supervisor.pid'):
try:
path.lstat()
except FileNotFoundError:
continue
raise ValueError()
with os.scandir(results) as entries:
for index, entry in enumerate(entries):
if index >= 100000 or entry.name.casefold() == 'found_secrets' or entry.name.casefold().startswith('found_secrets.'):
raise ValueError()
def read_private(path, limit):
with _input(path, runtime) as (handle, before):
if not 0 < before[4] <= limit:
raise ValueError()
raw = handle.read(limit + 1)
if len(raw) != before[4]:
raise ValueError()
return raw, _json(raw)
files_stopped()
identity_raw, identity = read_private(identity_path, MAX_JOURNAL)
manifest_raw, _ = read_private(manifest_path, MAX_MANIFEST)
manifest, _ = _manifest(manifest_raw, manifest_sha256)
expected_identity = {'pg_major': 16, 'system_identifier': system_identifier,
'data_directory': str(DATA / 'postgres-linux'), 'database': 'truf',
'user': 'truf', 'port': 5432}
if (not isinstance(identity, dict) or any(identity.get(k) != v for k, v in expected_identity.items())
or manifest['database']['system_identifier'] == system_identifier
or 'sequence_states' not in manifest['database']):
raise ValueError()
binding = {'system_identifier': system_identifier, 'manifest_sha256': manifest_sha256,
'identity_sha256': hashlib.sha256(identity_raw).hexdigest()}
def online():
if time.monotonic() >= deadline or runtime._shutdown_requested:
raise ValueError()
connection.execute('SELECT pg_catalog.pg_stat_clear_snapshot()')
row = connection.execute("""SELECT pg_catalog.current_database() AS database,
current_user AS user_name, pg_catalog.current_setting('data_directory') AS data_directory,
pg_catalog.current_setting('port')::int AS port,
pg_catalog.current_setting('server_version_num')::int AS version_num,
pg_catalog.pg_is_in_recovery() AS in_recovery,
(SELECT system_identifier::text FROM pg_catalog.pg_control_system()) AS system_identifier,
(SELECT rolsuper FROM pg_catalog.pg_roles WHERE rolname = current_user) AS superuser,
(SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE backend_type = 'client backend'
AND pid <> pg_catalog.pg_backend_pid()) AS other_clients,
pg_catalog.current_schema() AS schema_name,
pg_catalog.current_setting('search_path') AS search_path,
pg_catalog.current_setting('transaction_read_only') AS read_only,
pg_catalog.current_setting('fsync') AS fsync,
pg_catalog.current_setting('full_page_writes') AS full_page_writes,
EXISTS (SELECT 1 FROM pg_catalog.pg_namespace n CROSS JOIN LATERAL pg_catalog.aclexplode(
COALESCE(n.nspacl, pg_catalog.acldefault('n', n.nspowner))) acl
WHERE n.nspname = 'public' AND acl.grantee = 0 AND acl.privilege_type = 'CREATE') AS public_create""").fetchone()
expected = {'database': 'truf', 'user_name': 'truf', 'data_directory': str(DATA / 'postgres-linux'),
'port': 5432, 'system_identifier': system_identifier, 'in_recovery': False,
'superuser': True, 'other_clients': 0, 'schema_name': 'public',
'search_path': 'public', 'read_only': 'off', 'public_create': False,
'fsync': 'on', 'full_page_writes': 'on'}
if (not isinstance(row, dict) or any(row.get(k) != v for k, v in expected.items())
or _integer(row.get('version_num')) // 10000 != 16):
raise ValueError()
def metadata():
rows = connection.execute("""SELECT s.stream_name, c.stream_name AS cursor_stream_name,
pg_catalog.row_to_json(s) AS stream, pg_catalog.row_to_json(c) AS cursor
FROM public.projection_streams s FULL JOIN public.projection_cursors c
ON c.stream_name = s.stream_name""").fetchall()
seen, found = set(), None
scans = {'scan_results': 'scan_results.jsonl', 'found_secrets': 'found_secrets.jsonl',
'scan_errors': 'scan_errors.log'}
for row in rows:
name, stream, cursor = row['stream_name'], row['stream'], row['cursor']
if (not isinstance(name, str) or name in seen or row['cursor_stream_name'] != name
or not isinstance(stream, dict) or set(stream) != STREAM_COLUMNS
or not isinstance(cursor, dict) or set(cursor) != CURSOR_COLUMNS
or stream['stream_name'] != name or cursor['stream_name'] != name
or _integer(stream['current_generation']) != _integer(cursor['generation'])):
raise ValueError()
seen.add(name)
_integer(cursor['committed_offset'])
_integer(stream['rotation_bytes'], 1)
_integer(stream['max_generations'])
expected_path = scans.get(name)
if expected_path is None:
match = re.fullmatch(r'keycheck:([a-z0-9][a-z0-9_.-]{0,63}):(results|status)', name)
if not match:
raise ValueError()
suffix = 'Results.jsonl' if match[2] == 'results' else 'Checked.txt'
expected_path = f'{match[1]}/{match[1]}{suffix}'
if stream['base_relative_path'] != expected_path:
raise ValueError()
if name == 'found_secrets':
found = {'stream': stream, 'cursor': cursor}
if len(seen) != 34 or not scans.keys() <= seen:
raise ValueError()
return found
tables = sorted(manifest['database']['table_counts'])
sequences = manifest['database']['sequence_states']
if set(sequences) != {'public'}:
raise ValueError()
def preserved():
proof = {}
for table in tables:
if time.monotonic() >= deadline or runtime._shutdown_requested:
raise ValueError()
where = " WHERE t.stream_name <> 'found_secrets'" if table in ('projection_streams', 'projection_cursors') else ''
digest, count = hashlib.sha256(), 0
with connection.cursor(name='found_loss_digest') as cursor:
cursor.itersize = 1000
cursor.execute("SELECT pg_catalog.encode(pg_catalog.sha256(pg_catalog.convert_to("
"pg_catalog.row_to_json(t)::text, 'UTF8')), 'hex') COLLATE \"C\" AS digest FROM public."
+ _identifier(table) + ' AS t' + where + ' ORDER BY digest')
for row in cursor:
value = row['digest']
if not isinstance(value, str) or re.fullmatch(r'[0-9a-f]{64}', value) is None:
raise ValueError()
digest.update(value.encode('ascii'))
count += 1
if count % 1000 == 0 and (time.monotonic() >= deadline or runtime._shutdown_requested):
raise ValueError()
if count != manifest['database']['table_counts'][table] - int(bool(where)):
raise ValueError()
proof[table] = {'rows': count, 'sha256': digest.hexdigest()}
rows = connection.execute("""SELECT c.relname AS name FROM pg_catalog.pg_class c
JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
WHERE n.nspname = 'public' AND c.relkind = 'S' ORDER BY c.relname""").fetchall()
if {row['name'] for row in rows} != set(sequences['public']):
raise ValueError()
for row in rows:
state = connection.execute('SELECT last_value, is_called FROM public.' + _identifier(row['name'])).fetchone()
if _encoded(state) != _encoded(sequences['public'][row['name']]):
raise ValueError()
return {'tables': proof, 'sequences': {'count': len(rows),
'sha256': hashlib.sha256(_encoded(sequences)).hexdigest()}}
with PrivateFileLock(str(projector_lock)):
_regular(projector_lock, runtime)
with connection.transaction():
stage = 'database-fences'
connection.execute('SET TRANSACTION ISOLATION LEVEL READ COMMITTED')
online()
connection.execute("SET LOCAL lock_timeout = '5s'")
connection.execute("SET LOCAL statement_timeout = '10800s'")
connection.execute("SET LOCAL temp_file_limit = '4GB'")
connection.execute("SET LOCAL work_mem = '128MB'")
connection.execute('SET LOCAL max_parallel_workers_per_gather = 0')
connection.execute("SET LOCAL row_security = off")
connection.execute('SET LOCAL synchronous_commit = on')
for setting, value in (('log_min_error_statement', 'panic'), ('log_min_messages', 'panic'),
('log_statement', 'none'), ('log_min_duration_statement', '-1')):
connection.execute('SET LOCAL ' + setting + " = '" + value + "'")
for arguments in ((1414681926, 1785753445), (1414681926, 1768842867), (781273968142991337,)):
lock = connection.execute('SELECT pg_catalog.pg_try_advisory_xact_lock('
+ ','.join(['%s'] * len(arguments)) + ') AS locked', arguments).fetchone()
if not lock or lock['locked'] is not True:
raise ValueError()
catalog_sql = """SELECT c.relname AS name, c.relkind AS kind FROM pg_catalog.pg_class c
JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
WHERE n.nspname = 'public' AND c.relkind IN ('r','p','f') ORDER BY c.relname"""
catalog = connection.execute(catalog_sql).fetchall()
if len(catalog) != len(tables) or {row['name'] for row in catalog} != set(tables) or any(row['kind'] != 'r' for row in catalog):
raise ValueError()
connection.execute('LOCK TABLE ' + ','.join('public.' + _identifier(table) for table in tables)
+ ' IN EXCLUSIVE MODE NOWAIT')
online()
stage = 'history-gates'
gates = connection.execute("""WITH a AS (
SELECT count(*) AS appends, max(a.generation) AS append_highwater,
count(*) FILTER (WHERE a.state IS DISTINCT FROM 'appended' OR j.id IS NULL
OR j.status IS DISTINCT FROM 'completed' OR j.capacity_released IS DISTINCT FROM 1
OR j.job_kind IS DISTINCT FROM 'scan_event' OR (j.required_stream_mask & 2) IS DISTINCT FROM 2
OR a.event_id IS DISTINCT FROM j.event_id OR a.event_hash IS DISTINCT FROM j.event_hash
OR a.generation IS NULL OR a.generation NOT BETWEEN 0 AND 13
OR a.byte_offset IS NULL OR a.byte_offset < 0 OR a.byte_length IS NULL OR a.byte_length < 0
OR a.record_count IS NULL OR a.record_count < 0
OR (a.generation = 13 AND a.byte_offset::numeric + a.byte_length::numeric > 97783145)) AS bad_appends
FROM public.projection_appends a LEFT JOIN public.projection_jobs j ON j.id = a.job_id
WHERE a.stream_name = 'found_secrets'), r AS (
SELECT count(*) AS rotations, max(to_generation) AS rotation_highwater,
count(*) FILTER (WHERE state IS DISTINCT FROM 'completed' OR from_generation IS NULL OR from_generation < 0
OR to_generation IS DISTINCT FROM from_generation + 1 OR to_generation > 13
OR source_bytes IS NULL OR source_bytes < 0
OR segment_relative_path IS DISTINCT FROM
'found_secrets.g' || lpad(from_generation::text, 6, '0') || '.jsonl') AS bad_rotations
FROM public.projection_rotations WHERE stream_name = 'found_secrets')
SELECT a.*, r.*, (SELECT count(*) FROM public.projection_append_audit) AS audits,
(SELECT count(*) FROM public.projection_jobs WHERE (required_stream_mask & 2) <> 0
AND status <> 'completed') AS pending,
(SELECT count(*) FROM public.result_reservations
WHERE state IN ('scanning','ready','ingesting','db_committed')) AS reservations,
(SELECT count(*) FROM public.target_queue q LEFT JOIN public.result_reservations r
ON r.id = q.current_result_reservation_id WHERE q.status = 'in_progress'
OR r.state IN ('scanning','ready','ingesting','db_committed')) AS queue_leases,
(SELECT count(*) FROM public.docker_content_blobs
WHERE state IN ('leased','submitted') OR lease_reservation_id IS NOT NULL) AS blob_leases,
(SELECT count(*) FROM pg_catalog.pg_trigger WHERE NOT tgisinternal
AND tgrelid IN ('public.projection_streams'::regclass,'public.projection_cursors'::regclass)) AS triggers,
(SELECT count(*) FROM pg_catalog.pg_rewrite
WHERE ev_class IN ('public.projection_streams'::regclass,'public.projection_cursors'::regclass)) AS rules
FROM a CROSS JOIN r""").fetchone()
expected_gates = {'appends': APPEND_COUNT, 'append_highwater': 13, 'bad_appends': 0,
'rotations': ROTATION_COUNT, 'rotation_highwater': 13, 'bad_rotations': 0,
'audits': 0, 'pending': 0, 'reservations': 0, 'queue_leases': 0,
'blob_leases': 0, 'triggers': 0, 'rules': 0}
if _encoded(gates) != _encoded(expected_gates):
raise ValueError()
current = metadata()
proof = preserved()
stage = 'journal'
try:
journal_path.lstat()
except FileNotFoundError:
journal = None
else:
raw, journal = read_private(journal_path, MAX_JOURNAL)
if (not isinstance(journal, dict) or set(journal) != {'record', 'sha256'}
or raw != _encoded(journal)
or journal['sha256'] != hashlib.sha256(_encoded(journal['record'])).hexdigest()):
raise ValueError()
if journal is None:
stamp = datetime.now(timezone.utc).isoformat(timespec='seconds')
before = current
record = {'format': FORMAT, 'state': 'PREPARED', 'loss': LOSS, 'binding': binding,
'prepared_at': stamp, 'before': before, 'preserved': proof,
'digest_algorithm': 'sha256-concatenated-sorted-pg-row-sha256-hex-v1'}
else:
record = journal['record']
if (not isinstance(record, dict) or set(record) != {'format', 'state', 'loss', 'binding',
'prepared_at', 'before', 'after', 'preserved', 'digest_algorithm'}
or record['format'] != FORMAT or record['state'] != 'PREPARED' or record['loss'] != LOSS
or _encoded(record['binding']) != _encoded(binding)
or _encoded(record['preserved']) != _encoded(proof)
or record['digest_algorithm'] != 'sha256-concatenated-sorted-pg-row-sha256-hex-v1'):
raise ValueError()
stamp, before = record['prepared_at'], record['before']
if (not isinstance(stamp, str) or datetime.fromisoformat(stamp).isoformat(timespec='seconds') != stamp
or not stamp.endswith('+00:00') or set(before) != {'stream', 'cursor'}
or set(before['stream']) != STREAM_COLUMNS or set(before['cursor']) != CURSOR_COLUMNS
or before['stream']['stream_name'] != 'found_secrets'
or before['stream']['base_relative_path'] != 'found_secrets.jsonl'
or before['cursor']['stream_name'] != 'found_secrets'
or _integer(before['stream']['current_generation']) != OLD_GENERATION
or _integer(before['cursor']['generation']) != OLD_GENERATION
or _integer(before['cursor']['committed_offset']) != OLD_OFFSET):
raise ValueError()
_integer(before['cursor']['last_append_id'], 1)
_integer(before['cursor']['last_job_id'], 1)
last = connection.execute("""SELECT id, job_id, stream_name, generation, byte_offset,
byte_length, event_id, event_hash, state FROM public.projection_appends WHERE id = %s""",
(before['cursor']['last_append_id'],)).fetchone()
if (not last or last['id'] != before['cursor']['last_append_id']
or last['job_id'] != before['cursor']['last_job_id'] or last['stream_name'] != 'found_secrets'
or last['generation'] != OLD_GENERATION or last['state'] != 'appended'
or _integer(last['byte_offset']) + _integer(last['byte_length']) != OLD_OFFSET
or last['event_id'] != before['cursor']['last_event_id']
or last['event_hash'] != before['cursor']['last_event_hash']
or not isinstance(last['event_id'], str) or not last['event_id']
or re.fullmatch(r'[0-9a-f]{64}', last['event_hash'] or '') is None):
raise ValueError()
after = {key: dict(value) for key, value in before.items()}
after['stream'].update(current_generation=NEW_GENERATION, updated_at=stamp)
after['cursor'].update(generation=NEW_GENERATION, committed_offset=0, last_append_id=None, updated_at=stamp)
if journal is not None and _encoded(record['after']) != _encoded(after):
raise ValueError()
already = _encoded(current) == _encoded(after)
if (not already and _encoded(current) != _encoded(before)) or (already and journal is None):
raise ValueError()
if journal is None:
record['after'] = after
journal = {'record': record, 'sha256': hashlib.sha256(_encoded(record)).hexdigest()}
encoded = _encoded(journal)
if len(encoded) > MAX_JOURNAL:
raise ValueError()
_write(runtime, journal_path, encoded)
if read_private(journal_path, MAX_JOURNAL)[0] != encoded:
raise ValueError()
# A prior failure may have left complete bytes without a confirmed fsync.
with _input(journal_path, runtime) as (handle, _):
if handle.read(MAX_JOURNAL + 1) != _encoded(journal):
raise ValueError()
os.fsync(handle.fileno())
_fsync_dir(config_dir)
if not already:
stage = 'compare-and-swap'
result = connection.execute("""UPDATE public.projection_streams AS s
SET current_generation = 14, updated_at = %s
WHERE s.stream_name = 'found_secrets' AND pg_catalog.to_jsonb(s) = %s::jsonb""",
(stamp, _encoded(before['stream']).decode('ascii')))
if result.rowcount != 1:
raise ValueError()
result = connection.execute("""UPDATE public.projection_cursors AS c
SET generation = 14, committed_offset = 0, last_append_id = NULL, updated_at = %s
WHERE c.stream_name = 'found_secrets' AND pg_catalog.to_jsonb(c) = %s::jsonb""",
(stamp, _encoded(before['cursor']).decode('ascii')))
if result.rowcount != 1:
raise ValueError()
if _encoded(metadata()) != _encoded(after) or _encoded(preserved()) != _encoded(proof):
raise ValueError()
stage = 'precommit'
files_stopped()
if (read_private(identity_path, MAX_JOURNAL)[0] != identity_raw
or read_private(manifest_path, MAX_MANIFEST)[0] != manifest_raw
or read_private(journal_path, MAX_JOURNAL)[0] != _encoded(journal)
or _encoded(connection.execute(catalog_sql).fetchall()) != _encoded(catalog)):
raise ValueError()
online()
stage = 'commit'
return {'status': 'already-committed' if already else 'committed', 'journal_path': str(journal_path),
'journal_sha256': hashlib.sha256(_encoded(journal)).hexdigest(), **binding, 'generation': NEW_GENERATION}
except BaseException:
raise ProjectionRecoveryError('Projection recovery refused at ' + stage
+ '; retain caller maintenance authority and verify stop.') from None
+594
View File
@@ -0,0 +1,594 @@
"""Isolated entrypoint for the private, read-only Docker deployment."""
import argparse
import http.client
import json
import os
from pathlib import Path
import re
import runpy
import secrets
import signal
import stat
import subprocess
import sys
from urllib.parse import quote
sys.dont_write_bytecode = True
if not sys.dont_write_bytecode:
raise RuntimeError('container runtime could not disable bytecode writes')
APP = Path('/opt/truf/app')
DATA = Path('/data')
RUN = Path('/run/truf')
UID = GID = 10001
DEFAULT_CONFIG = APP / 'config.linux.yaml'
PROVISIONED = DATA / '.provisioned.json'
INITIALIZED = DATA / 'initialized.json'
INITIALIZE_LOCK = DATA / 'initialize.lock'
PASSWORD = DATA / 'postgres-password'
PROVIDER_SECRETS = DATA / 'config/secrets.yaml'
FORMAT = 'truf-container-data-v1'
WORKER_HEALTH_TOKEN = '0' * 64
DIRECTORIES = (
'home', 'config', 'managed-files', 'runtime-linux', 'runtime-linux/results',
'runtime-linux/queues', 'runtime-linux/state',
'runtime-linux/state/gharchive_cache', 'runtime-linux/logs',
'runtime-linux/keychecks', 'runtime-linux/postman_cache',
'runtime-linux/result_spool', 'runtime-linux/postgres',
'runtime-linux/postgres/logs', 'postgres-linux', 'scanner-work',
'scanner-result-bundles', 'scanner-result-bundles/tmp',
'scanner-result-bundles/ready', 'scanner-result-bundles/quarantine',
)
_shutdown_requested = False
def private_path(path, *, directory=False):
path = Path(path)
if not path.is_absolute() or '..' in path.parts:
raise RuntimeError('private path must be absolute and normalized')
for component in (*reversed(path.parents), path):
if stat.S_ISLNK(component.lstat().st_mode):
raise RuntimeError('private paths must not contain symlinks')
details = path.lstat()
expected_type = stat.S_ISDIR if directory else stat.S_ISREG
if (not expected_type(details.st_mode) or details.st_uid != UID
or details.st_gid != GID
or stat.S_IMODE(details.st_mode) != (0o700 if directory else 0o600)):
raise RuntimeError('private path ownership, type, or mode is invalid: ' + str(path))
return path
def require_container(*, provisioning=False):
if (sys.platform != 'linux' or Path(__file__) != APP / 'container_runtime.py'
or not Path('/.dockerenv').is_file()):
raise RuntimeError('runtime commands are restricted to the prepared Docker image')
if not (sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode):
raise RuntimeError('container entrypoint requires python -I -S -B')
if os.getuid() != os.geteuid() or os.geteuid() != (0 if provisioning else UID):
raise RuntimeError('unexpected container runtime UID')
if not provisioning and os.getgid() != GID:
raise RuntimeError('unexpected container runtime GID')
if not os.statvfs(APP).f_flag & os.ST_RDONLY:
raise RuntimeError('the application image must be mounted read-only')
private_path(APP.parent, directory=True)
private_path(APP, directory=True)
private_path(APP / 'container_runtime.py')
with open('/proc/self/mountinfo', 'rb') as handle:
payload = handle.read(1024 * 1024 + 1)
if len(payload) > 1024 * 1024:
raise RuntimeError('mount inventory exceeds its bound')
mounts = {}
for line in payload.splitlines():
fields = line.split()
if len(fields) > 6 and b'-' in fields:
mounts[fields[4]] = fields[fields.index(b'-') + 1]
if mounts.get(b'/data') not in (b'ext4', b'xfs', b'btrfs', b'zfs'):
raise RuntimeError('/data must be an independent native Linux data volume')
if mounts.get(b'/run/truf') != b'tmpfs':
raise RuntimeError('/run/truf must be an independent private tmpfs')
private_path(DATA, directory=True)
private_path(RUN, directory=True)
os.umask(0o077)
def _read_json(path):
path = private_path(path)
if path.stat().st_size > 4096:
raise RuntimeError('container marker exceeds its bound')
value = json.loads(path.read_text(encoding='utf-8'))
if not isinstance(value, dict) or value.get('format') != FORMAT:
raise RuntimeError('unrecognized container data marker')
return value
def _write_new(path, payload, *, provisioning=False):
try:
private_path(path.parent, directory=True)
descriptor = os.open(
path,
os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW,
0o600,
)
with os.fdopen(descriptor, 'wb') as handle:
if provisioning:
os.fchown(handle.fileno(), UID, GID)
handle.write(payload)
handle.flush()
os.fsync(handle.fileno())
descriptor = os.open(path.parent, os.O_RDONLY | os.O_DIRECTORY)
try:
os.fsync(descriptor)
finally:
os.close(descriptor)
private_path(path)
finally:
payload = None
def provision():
import fcntl
lock = DATA / '.provision.lock'
try:
descriptor = os.open(lock, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600)
os.fchown(descriptor, UID, GID)
except FileExistsError:
private_path(lock)
descriptor = os.open(lock, os.O_WRONLY | os.O_NOFOLLOW)
with os.fdopen(descriptor, 'wb') as handle:
fcntl.flock(handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
if PROVISIONED.exists():
_read_json(PROVISIONED)
managed_files = DATA / 'managed-files'
try:
os.mkdir(managed_files, 0o700)
os.chown(managed_files, UID, GID)
except FileExistsError:
pass
for name in DIRECTORIES:
private_path(DATA / name, directory=True)
private_path(PASSWORD)
private_path(PROVIDER_SECRETS)
print('Container data layout is already provisioned; nothing was changed.', flush=True)
return
if set(os.listdir(DATA)) - {'home', '.provision.lock'}:
raise RuntimeError('refusing to provision nonempty or partially initialized data')
home = DATA / 'home'
if home.exists() and any(home.iterdir()):
raise RuntimeError('refusing to provision a nonempty home directory')
for name in DIRECTORIES:
path = DATA / name
if not path.exists():
os.mkdir(path, 0o700)
os.chown(path, UID, GID)
private_path(path, directory=True)
_write_new(PASSWORD, (secrets.token_urlsafe(48) + '\n').encode('ascii'), provisioning=True)
_write_new(PROVIDER_SECRETS, b'{}\n', provisioning=True)
_write_new(DATA / 'runtime-linux/proxy.txt', b'', provisioning=True)
_write_new(PROVISIONED, (json.dumps({'format': FORMAT, 'uid': UID, 'gid': GID}) + '\n').encode('ascii'), provisioning=True)
print('Fresh private container data layout provisioned; PostgreSQL is not initialized yet.', flush=True)
def prepare_environment(
config_path, *, validate_documents=True, return_config_sha256=False,
):
_read_json(PROVISIONED)
for name in ('authority', 'control', 'tmp'):
path = RUN / name
try:
path.mkdir(mode=0o700)
except FileExistsError:
pass
private_path(path, directory=True)
config_path = Path(config_path)
if config_path != DEFAULT_CONFIG and config_path.parent != DATA / 'config':
raise RuntimeError('configuration must be image-owned or in /data/config')
private_path(config_path)
password = private_path(PASSWORD).read_text(encoding='ascii').rstrip('\n')
if re.fullmatch(r'[A-Za-z0-9_-]{32,128}', password) is None:
raise RuntimeError('the generated PostgreSQL password is invalid')
for name in tuple(os.environ):
upper = name.upper()
if (upper.startswith(('PG', 'TRUF_', 'SCANNER_', 'SCAN_', 'TRUFFLEHOG_', 'KEYCHECK_'))
or upper in ('DATABASE_URL', 'PYTHONPATH', 'PYTHONHOME')):
del os.environ[name]
url = 'postgresql://truf:' + quote(password, safe='') + '@127.0.0.1:5432/truf'
os.environ.update({
'PATH': '/usr/local/bin:/usr/bin:/bin:/usr/lib/postgresql/16/bin',
'HOME': '/data/home', 'TMPDIR': str(RUN / 'tmp'),
'TMP': str(RUN / 'tmp'), 'TEMP': str(RUN / 'tmp'),
'TRUF_CONTAINER_CONFIG': str(config_path),
'TRUF_POSTGRES_DB': 'truf', 'TRUF_POSTGRES_USER': 'truf',
'TRUF_POSTGRES_PORT': '5432', 'TRUF_POSTGRES_PASSWORD': password,
'SCANNER_DB_URL': url, 'DATABASE_URL': url, 'TRUF_MANAGED_POSTGRES_DSN': url,
'TRUF_DB_CONNECT_TIMEOUT_SEC': '3', 'TRUF_DB_STATEMENT_TIMEOUT_MS': '5000',
'TRUF_DB_LOCK_TIMEOUT_MS': '2000', 'TRUF_DB_IDLE_TRANSACTION_TIMEOUT_MS': '10000',
})
bootstrap = runpy.run_path(str(APP / 'child_bootstrap.py'))
bootstrap['_enable_dependency_paths']('supervisor')
sys.path.insert(0, str(APP))
from paths import apply_path_config
from runtime_document_io import (
load_managed_runtime_config,
validate_managed_runtime_files,
)
from runtime_security import preflight_lifecycle_paths
def check_resolved(candidate):
expected = {
'root_dir': '/opt/truf', 'project_dir': str(APP),
'runtime_dir': '/data/runtime-linux', 'postgres_data_dir': '/data/postgres-linux',
'postgres_bin_dir': '/usr/lib/postgresql/16/bin',
'result_bundle_dir': '/data/scanner-result-bundles', 'work_dir': '/data/scanner-work',
'control_dir': str(RUN / 'control'), 'secrets_file': str(PROVIDER_SECRETS),
}
if any(candidate['global'].get(name) != value for name, value in expected.items()):
raise RuntimeError('configuration escapes the fixed container storage contract')
if candidate['supervisor'].get('control_dir') != str(RUN / 'control'):
raise RuntimeError('supervisor control must remain on private ephemeral storage')
preflight_lifecycle_paths(
str(config_path), candidate, authority_profile='server',
)
documents = (
validate_managed_runtime_files(str(config_path))
if validate_documents
else load_managed_runtime_config(str(config_path))
)
config = documents.config
config_sha256 = documents.config_sha256
documents = None
config = apply_path_config(config, str(config_path))
check_resolved(config)
if return_config_sha256:
return config, config_sha256
return config
def _bootstrap_command(target, *arguments):
return [sys.executable, '-u', '-I', '-S', '-B', str(APP / 'runtime_bootstrap.py'), target, '--', *arguments]
def initialize(config_path, config, *, expected_config_sha256=None):
from postgres_runtime import postgres_runtime_paths
from runtime_document_io import load_managed_runtime_config
from runtime_security import PrivateFileLock, read_private_json, write_private_json_exclusive
def require_stable_config():
if expected_config_sha256 is None:
return
current = load_managed_runtime_config(str(config_path))
current_sha256 = current.config_sha256
current = None
if current_sha256 != expected_config_sha256:
raise RuntimeError('configuration changed after managed runtime validation')
require_stable_config()
paths = postgres_runtime_paths(config)
def migrate(*, initialize_base):
if _shutdown_requested:
return False
try:
# Even an uncertain maintenance start must enter the verified stop path.
require_stable_config()
subprocess.run(_bootstrap_command(
'postgres-runtime', 'maintenance-start', '--config', str(config_path),
), check=True)
if _shutdown_requested:
return False
require_stable_config()
arguments = [
'migrate-runtime-safety', '--config', str(config_path),
]
if initialize_base:
arguments.append('--initialize-base')
arguments.extend(('--apply', '--sources-stopped'))
subprocess.run(_bootstrap_command(*arguments), check=True)
return True
finally:
subprocess.run(_bootstrap_command(
'postgres-runtime', 'maintenance-stop', '--config', str(config_path),
), check=True)
with PrivateFileLock(str(INITIALIZE_LOCK)):
if INITIALIZED.exists():
marker = _read_json(INITIALIZED)
identity = read_private_json(paths['identity_path'])
if (not marker.get('system_identifier')
or marker.get('system_identifier') != identity.get('system_identifier')
or marker.get('pg_major') != 16 or identity.get('pg_major') != 16):
raise RuntimeError('initialization marker does not match the bound cluster')
migrated = migrate(initialize_base=False)
if migrated:
print('Existing PostgreSQL migrated and confirmed stopped.', flush=True)
return
if Path(paths['identity_path']).exists() or any(Path(paths['data_dir']).iterdir()):
raise RuntimeError('partial initialization requires offline inspection; no automatic repair is allowed')
if _shutdown_requested:
return
require_stable_config()
subprocess.run(_bootstrap_command('postgres-runtime', 'initialize-empty', '--config', str(config_path)), check=True)
if _shutdown_requested:
return
if not migrate(initialize_base=True):
return
identity = read_private_json(paths['identity_path'])
require_stable_config()
write_private_json_exclusive(str(INITIALIZED), {
'format': FORMAT, 'system_identifier': identity['system_identifier'],
'pg_major': identity['pg_major'],
})
print('Independent PostgreSQL initialized, migrated, cut over, and confirmed stopped.', flush=True)
def _probe_worker_api(config):
worker = config['supervisor']['worker_api']
connection = None
try:
connection = http.client.HTTPConnection(
worker['address'], int(worker['port']), timeout=2,
)
connection.request(
'POST', '/api/v1/worker/claim', body=b'',
headers={
'Authorization': 'Bearer ' + WORKER_HEALTH_TOKEN,
'Content-Length': '0',
},
)
response = connection.getresponse()
payload = response.read(4097)
if (
len(payload) > 4096
or response.status != 401
or response.getheader('WWW-Authenticate') != 'Bearer'
or json.loads(payload) != {
'error': {
'code': 'unauthorized',
'message': 'worker credentials are invalid',
},
}
):
raise RuntimeError('worker API is not ready')
except Exception:
raise RuntimeError('worker API is not ready') from None
finally:
if connection is not None:
connection.close()
def health(config, *, require_worker_api=False, require_discovery_producers=False):
# Never create a competing SQL session while first-install migration is exclusive.
_read_json(INITIALIZED)
from scanner_db import ScannerDB
from lifecycle_authority import DISCOVERY_PRODUCER_SOURCES
from supervisor import get_control_snapshot
from supervisor_instance import load_instance_metadata
metadata = load_instance_metadata(config['supervisor']['instance_file'])
snapshot = get_control_snapshot(metadata)
if (snapshot.get('activation_state') != 'ACTIVE'
or snapshot.get('postgres', {}).get('state') != 'READY'
or snapshot.get('postgres', {}).get('ready') is not True):
raise RuntimeError('supervisor and PostgreSQL are not ready')
signatures = {row[0]: row for row in snapshot.get('signature', ())}
required = ['result-ingester', 'jsonl-projector']
if config['supervisor'].get('janitor', {}).get('enabled', True):
required.append('janitor')
worker_api_enabled = config['supervisor'].get('worker_api', {}).get(
'enabled', False,
)
if require_worker_api and not worker_api_enabled:
raise RuntimeError('worker API is required but disabled')
if worker_api_enabled:
required.append('worker-api')
for name in required:
row = signatures.get(name)
if not row or tuple(row[1:3]) != ('running', 'running') or not row[3] or row[7]:
raise RuntimeError('a required pipeline worker is not running')
if any(row[1] == 'failed' or row[8] for row in signatures.values()):
raise RuntimeError('a managed source is failed or retains uncertain ownership')
if require_discovery_producers:
source_config = config.get('sources') or {}
supervisor_sources = config['supervisor'].get('sources') or {}
enabled_producers = [
name for name in DISCOVERY_PRODUCER_SOURCES
if (
(supervisor_sources.get(name) or {}).get('enabled')
if 'enabled' in (supervisor_sources.get(name) or {})
else (source_config.get(name) or {}).get('enabled', False)
)
]
for name in enabled_producers:
row = signatures.get(name)
ready = bool(
row
and row[2] == 'running'
and not row[7]
and not row[8]
and (
(row[1] == 'running' and row[3])
or (row[1] == 'waiting' and row[4] == 0)
)
)
if not ready:
raise RuntimeError('a required discovery producer is not ready')
if require_worker_api:
_probe_worker_api(config)
db = ScannerDB(db_url=os.environ['SCANNER_DB_URL'], initialize=False)
try:
if not db.enabled or not db.conn.is_postgres:
raise RuntimeError('PostgreSQL application connection is unavailable')
db.set_application_name('truf-container-health')
db.conn.execute('SET default_transaction_read_only = on')
db.conn.commit()
db.require_runtime_safety_schema()
db.require_final_cutover()
for name in ('result_ingester', 'jsonl_projector'):
if not db.pipeline_worker_health(name, metadata['instance_id'])['healthy']:
raise RuntimeError('a required durable worker lease is not ready')
row = db.conn.execute("SELECT current_setting('data_directory') AS data_directory, current_setting('server_version_num') AS version").fetchone()
db.conn.commit()
if row['data_directory'] != '/data/postgres-linux' or int(row['version']) // 10000 != 16:
raise RuntimeError('PostgreSQL identity does not match the container')
finally:
db.close()
for name in ('work_dir', 'result_bundle_dir', 'results_dir'):
path = config['global'][name]
if not os.access(path, os.W_OK | os.X_OK) or os.statvfs(path).f_bavail == 0:
raise RuntimeError('required persistent storage is not writable or is full')
return {'healthy': True, 'activation_state': 'ACTIVE', 'postgres': 'READY', 'workers': sorted(signatures)}
def import_secrets(config_path, config, *, expected_config_sha256=None):
from postgres_runtime import postgres_runtime_paths
from runtime_document_io import (
load_managed_runtime_config,
validate_managed_runtime_files,
)
from runtime_security import ClusterAuthorityLock, PrivateFileLock, durable_replace, fsync_directory
import yaml
with PrivateFileLock(str(INITIALIZE_LOCK)), ClusterAuthorityLock(
config, create_parent=False, endpoint_dsn=os.environ['SCANNER_DB_URL'],
):
if (Path(config['supervisor']['instance_file']).exists()
or (Path(postgres_runtime_paths(config)['data_dir']) / 'postmaster.pid').exists()):
raise RuntimeError('stop the runtime before replacing provider credentials')
payload = sys.stdin.buffer.read(1024 * 1024 + 1)
try:
if not payload or len(payload) > 1024 * 1024:
raise RuntimeError('credential input must be a nonempty YAML document below 1 MiB')
validated = validate_managed_runtime_files(
str(config_path), secrets_bytes=payload,
)
except BaseException:
payload = None
raise
payload = None
validated_config_sha256 = validated.config_sha256
if (
expected_config_sha256 is not None
and validated_config_sha256 != expected_config_sha256
):
validated = None
raise RuntimeError('configuration changed while provider credentials were being validated')
try:
serialized = yaml.safe_dump(
validated.secrets, allow_unicode=True,
).encode('utf-8')
except BaseException:
validated = None
raise
validated = None
temporary = DATA / ('config/secrets-import-' + secrets.token_hex(16))
try:
try:
_write_new(temporary, serialized)
finally:
serialized = None
private_path(PROVIDER_SECRETS)
current = load_managed_runtime_config(str(config_path))
current_sha256 = current.config_sha256
current = None
if current_sha256 != validated_config_sha256:
raise RuntimeError(
'configuration changed while provider credentials were being validated'
)
durable_replace(str(temporary), str(PROVIDER_SECRETS))
fsync_directory(str(PROVIDER_SECRETS.parent))
finally:
temporary.unlink(missing_ok=True)
print('Private provider credentials replaced; no credentials were printed.', flush=True)
def main(argv=None):
global _shutdown_requested
parser = argparse.ArgumentParser(description=__doc__, allow_abbrev=False)
parser.add_argument('action', choices=('provision', 'initialize', 'run', 'health', 'status', 'import-secrets', 'import-snapshot'))
parser.add_argument('--config', default=os.environ.get('TRUF_CONTAINER_CONFIG', str(DEFAULT_CONFIG)))
parser.add_argument('--manifest-sha256')
parser.add_argument('--require-worker-api', action='store_true')
parser.add_argument('--require-discovery-producers', action='store_true')
args = parser.parse_args(argv)
require_container(provisioning=args.action == 'provision')
if (
(args.require_worker_api or args.require_discovery_producers)
and args.action not in ('health', 'status')
):
raise RuntimeError('strict health requirements are restricted to health and status')
if args.action == 'import-snapshot':
if re.fullmatch(r'[0-9a-f]{64}', args.manifest_sha256 or '') is None:
raise RuntimeError('snapshot import requires --manifest-sha256 with the approved manifest digest')
if Path(args.config) != DEFAULT_CONFIG:
raise RuntimeError('snapshot import must begin with the image-owned default configuration')
elif args.manifest_sha256 is not None:
raise RuntimeError('--manifest-sha256 is restricted to snapshot import')
if args.action == 'provision':
provision()
return 0
bind_config = args.action in ('initialize', 'run', 'import-secrets')
prepared = prepare_environment(
args.config,
validate_documents=args.action != 'import-secrets',
return_config_sha256=bind_config,
)
if bind_config:
config, config_sha256 = prepared
else:
config = prepared
if args.action in ('health', 'status'):
print(json.dumps(health(
config, require_worker_api=args.require_worker_api,
require_discovery_producers=args.require_discovery_producers,
), sort_keys=True), flush=True)
return 0
if args.action == 'import-secrets':
import_secrets(
args.config, config,
expected_config_sha256=config_sha256,
)
return 0
def request_shutdown(_signum, _frame):
global _shutdown_requested
_shutdown_requested = True
previous = signal.signal(signal.SIGTERM, request_shutdown)
try:
if args.action == 'import-snapshot':
from container_import import import_snapshot
return import_snapshot(sys.modules[__name__], args.manifest_sha256)
initialize(
Path(args.config), config,
expected_config_sha256=config_sha256,
)
if _shutdown_requested or args.action == 'initialize':
return 0
from runtime_document_io import load_managed_runtime_config
current = load_managed_runtime_config(str(args.config))
current_sha256 = current.config_sha256
current = None
if current_sha256 != config_sha256:
raise RuntimeError('configuration changed after managed runtime validation')
command = _bootstrap_command(
'supervisor', '--runtime-bootstrap-entrypoint', str(APP / 'supervisor.py'),
'--config', args.config, '--with-postgres', '--non-interactive',
'--autostart', '--no-dashboard',
)
os.execv(sys.executable, command)
finally:
signal.signal(signal.SIGTERM, previous)
if __name__ == '__main__':
try:
raise SystemExit(main())
except Exception as exc:
print('Container runtime rejected: ' + str(exc), file=sys.stderr, flush=True)
raise SystemExit(1) from None
+2581
View File
File diff suppressed because it is too large Load Diff
+836
View File
@@ -0,0 +1,836 @@
import os
import re
import sqlite3
import time
from urllib.parse import parse_qsl, quote, unquote, urlencode, urlsplit, urlunsplit
POSTGRES_SCHEMES = ('postgresql://', 'postgres://')
DEFAULT_POSTGRES_CONNECT_TIMEOUT_SEC = 10
DEFAULT_POSTGRES_STATEMENT_TIMEOUT_MS = 30000
DEFAULT_POSTGRES_LOCK_TIMEOUT_MS = 10000
DEFAULT_POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS = 30000
DEFAULT_POSTGRES_TCP_USER_TIMEOUT_MS = 30000
POSTGRES_APPLICATION_SCHEMA = 'public'
POSTGRES_CHILD_START_RETRY_ATTEMPTS = 3
POSTGRES_CHILD_START_RETRY_MARKERS = (
'server closed the connection unexpectedly',
'connection reset by peer',
'connection was forcibly closed by the remote host',
)
HOST_AGENT_POSTGRES_SOCKET_DIRECTORY = '/run/truf-postgres'
HOST_AGENT_POSTGRES_DATABASE = 'truf'
HOST_AGENT_POSTGRES_USER = 'truf'
HOST_AGENT_POSTGRES_PORT = 5432
class DatabaseUrlError(ValueError):
pass
def is_postgres_url(value):
return str(value or '').strip().lower().startswith(POSTGRES_SCHEMES)
def database_url_from_env():
# Supervised processes receive TRUF_MANAGED_POSTGRES_DSN as the canonical
# connection authority. Use that same precedence everywhere that derives
# an endpoint identity or opens a managed connection.
return (
os.getenv('TRUF_MANAGED_POSTGRES_DSN')
or os.getenv('SCANNER_DB_URL')
or os.getenv('DATABASE_URL')
)
def parse_postgres_url(value):
"""Parse one URL-form libpq DSN without allowing alternate authorities."""
text = str(value or '').strip()
if not text or any(character in text for character in ('\x00', '\r', '\n')):
raise DatabaseUrlError('invalid PostgreSQL database URL')
try:
parsed = urlsplit(text)
port = parsed.port or 5432
host = parsed.hostname or ''
username = unquote(parsed.username or '')
password = unquote(parsed.password or '')
database = unquote((parsed.path or '')[1:]) if (parsed.path or '').startswith('/') else ''
except (TypeError, ValueError) as exc:
raise DatabaseUrlError('invalid PostgreSQL database URL') from exc
if parsed.scheme.lower() not in ('postgresql', 'postgres'):
raise DatabaseUrlError('database URL must use the PostgreSQL scheme')
if parsed.query:
raise DatabaseUrlError('PostgreSQL database URL query parameters are forbidden')
if parsed.fragment:
raise DatabaseUrlError('PostgreSQL database URL fragments are forbidden')
if not parsed.netloc or not host or not username or not database:
raise DatabaseUrlError('PostgreSQL database URL must include one host, user, and database')
authority = parsed.netloc.rsplit('@', 1)[-1]
decoded_authority = unquote(authority)
if (
',' in decoded_authority
or any(character.isspace() for character in decoded_authority)
or '%' in authority
or any(character in host for character in (',', '/', '\\', '\x00'))
):
raise DatabaseUrlError('PostgreSQL database URL must contain exactly one literal host authority')
if not 0 < int(port) <= 65535:
raise DatabaseUrlError('PostgreSQL database URL port is invalid')
if any(character in database for character in ('/', '\\', '?', '#', '\x00')):
raise DatabaseUrlError('PostgreSQL database URL database name contains encoded authority syntax')
if parsed.path.count('/') != 1:
raise DatabaseUrlError('PostgreSQL database URL must contain exactly one database path segment')
return {
'parsed': parsed,
'host': host.lower(),
'port': int(port),
'database': database,
'user': username,
'password': password,
}
def canonical_postgres_url(value, database, user, port, host='127.0.0.1'):
"""Return one libpq URL whose endpoint cannot be redirected by DSN options."""
values = parse_postgres_url(value)
expected_host = str(host).lower()
if values['host'] != expected_host or values['port'] != int(port):
raise DatabaseUrlError('PostgreSQL database URL endpoint does not match managed cluster authority')
if values['database'] != str(database) or values['user'] != str(user):
raise DatabaseUrlError('PostgreSQL database URL identity does not match managed cluster authority')
credentials = quote(str(user), safe='')
if values['parsed'].password is not None:
credentials += ':' + quote(values['password'], safe='')
netloc = f'{credentials}@{expected_host}:{int(port)}'
return urlunsplit(('postgresql', netloc, '/' + quote(str(database), safe=''), '', ''))
def redact_database_url(value):
text = str(value or '')
if not is_postgres_url(text):
if text.strip().lower().startswith(('postgres', 'postgre')):
return 'postgresql://***'
if '://' in text:
return text.split('://', 1)[0] + '://***'
return text
try:
parsed = urlsplit(text)
username = parsed.username or ''
host = parsed.hostname or ''
port = f':{parsed.port}' if parsed.port else ''
netloc = parsed.netloc
if parsed.password:
netloc = f'{username}:***@{host}{port}' if username else f'***@{host}{port}'
query = []
for key, value in parse_qsl(parsed.query, keep_blank_values=True):
key_lower = key.lower()
sensitive = any(part in key_lower for part in ('password', 'passwd', 'pwd', 'token', 'secret', 'credential'))
query.append((key, '***' if sensitive else value))
return urlunsplit((parsed.scheme, netloc, parsed.path, urlencode(query), parsed.fragment))
except Exception:
return 'postgresql://***'
def _split_sql_script(script):
statements = []
current = []
quote = None
escape = False
for char in str(script or ''):
current.append(char)
if escape:
escape = False
continue
if char == '\\':
escape = True
continue
if quote:
if char == quote:
quote = None
continue
if char in ("'", '"'):
quote = char
continue
if char == ';':
statement = ''.join(current).strip()
if statement:
statements.append(statement[:-1].strip())
current = []
tail = ''.join(current).strip()
if tail:
statements.append(tail)
return [statement for statement in statements if statement]
def _convert_qmark_to_psycopg(sql):
out = []
quote = None
escape = False
for char in str(sql or ''):
if escape:
out.append(char)
escape = False
continue
if char == '\\':
out.append(char)
escape = True
continue
if quote:
out.append('%%' if char == '%' else char)
if char == quote:
quote = None
continue
if char in ("'", '"'):
out.append(char)
quote = char
continue
if char == '?':
out.append('%s')
elif char == '%':
out.append('%%')
else:
out.append(char)
return ''.join(out)
def _postgres_schema_sql(script):
converted = str(script or '').replace(
'INTEGER PRIMARY KEY AUTOINCREMENT',
'BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY',
)
# PostgreSQL cannot create either side of the target_queue/target_scans
# cycle with both inline FKs. The offline migration adds this edge after
# both tables exist; SQLite can retain it in the base schema.
converted = converted.replace(
',\n FOREIGN KEY(queue_id) REFERENCES target_queue(id)',
'',
)
for future_foreign_key in (
',\n FOREIGN KEY(result_reservation_id) REFERENCES result_reservations(id)',
',\n FOREIGN KEY(reservation_id) REFERENCES result_reservations(id)',
',\n FOREIGN KEY(current_result_reservation_id) REFERENCES result_reservations(id)',
',\n FOREIGN KEY(candidate_id) REFERENCES keycheck_candidates(id)',
',\n FOREIGN KEY(credential_id) REFERENCES keycheck_credentials(id)',
',\n FOREIGN KEY(last_append_id) REFERENCES projection_appends(id)',
',\n FOREIGN KEY(experiment_id) REFERENCES docker_depth_experiments(id)',
):
converted = converted.replace(future_foreign_key, '')
for column in (
'run_id', 'cycle_id', 'target_scan_id', 'finding_id', 'keycheck_result_id',
'last_run_id', 'last_cycle_id', 'queue_id', 'byte_offset', 'line_number',
'current_result_reservation_id', 'result_reservation_id', 'reservation_id',
'projection_job_id', 'keycheck_candidate_id', 'candidate_id', 'credential_id',
'keycheck_result_id', 'target_scan_id', 'job_id', 'last_append_id',
'last_job_id', 'last_result_id', 'object_id', 'lease_reservation_id',
'covered_reservation_id', 'declared_bytes', 'verified_bytes',
'experiment_id', 'pass_id', 'page_id', 'source_cycle_id', 'retry_work_id',
'repository_queue_id', 'first_cycle_id', 'last_cycle_id', 'first_page_id',
'last_page_id', 'target_queue_id', 'manifest_id', 'manifest_layer_id',
'eligibility_page_id', 'experiment_repository_id', 'experiment_target_id',
'scan_binding_id', 'manifest_size_bytes', 'layer_size_bytes',
'fence_generation', 'resolver_generation', 'dispatch_order', 'total_count',
'user_id', 'remote_user_id', 'remote_device_id',
'expected_revision', 'resulting_revision', 'revision',
'before_bytes', 'after_bytes', 'previous_event_id',
):
converted = converted.replace(f'{column} INTEGER', f'{column} BIGINT')
converted = converted.replace(
'CREATE VIEW IF NOT EXISTS keycheck_latest_state AS',
'CREATE OR REPLACE VIEW keycheck_latest_state AS',
)
return converted
def _sqlite_check_constraints(sql):
text = str(sql or '')
constraints = {}
index = 0
position = 0
quote = None
while position < len(text):
char = text[position]
if quote:
if char == quote:
if position + 1 < len(text) and text[position + 1] == quote:
position += 2
continue
quote = None
position += 1
continue
if char in ("'", '"', '`'):
quote = char
position += 1
continue
if (
text[position:position + 5].lower() != 'check'
or (position and (text[position - 1].isalnum() or text[position - 1] == '_'))
or (
position + 5 < len(text)
and (text[position + 5].isalnum() or text[position + 5] == '_')
)
):
position += 1
continue
opening = position + 5
while opening < len(text) and text[opening].isspace():
opening += 1
if opening >= len(text) or text[opening] != '(':
position += 5
continue
depth = 1
closing = opening + 1
expression_quote = None
while closing < len(text) and depth:
current = text[closing]
if expression_quote:
if current == expression_quote:
if closing + 1 < len(text) and text[closing + 1] == expression_quote:
closing += 2
continue
expression_quote = None
elif current in ("'", '"', '`'):
expression_quote = current
elif current == '(':
depth += 1
elif current == ')':
depth -= 1
closing += 1
if depth:
break
prefix = text[:position]
named = re.search(
r'\bCONSTRAINT\s+(?:"([A-Za-z_][A-Za-z0-9_$]*)"|'
r'([A-Za-z_][A-Za-z0-9_$]*))\s*$',
prefix,
re.IGNORECASE,
)
name = (named.group(1) or named.group(2)) if named else f'__unnamed_check_{index}'
expression = text[opening + 1:closing - 1]
constraint = {
'expression': expression,
'definition': f'CHECK ({expression})',
'valid': True,
}
if name in constraints:
constraints[name]['valid'] = False
name = f'__duplicate_check_{index}_{name}'
constraint['valid'] = False
constraints[name] = constraint
index += 1
position = closing
return constraints
class DatabaseConnection:
def __init__(self, dialect, conn, application_schema=None):
self.dialect = dialect
self._conn = conn
self.application_schema = application_schema if dialect == 'postgres' else None
@property
def is_postgres(self):
return self.dialect == 'postgres'
@property
def is_sqlite(self):
return self.dialect == 'sqlite'
def execute(self, sql, params=None):
params = tuple(params or ())
if self.is_postgres:
params = tuple(value.replace('\x00', '') if isinstance(value, str) else value for value in params)
cur = self._conn.cursor()
cur.execute(_convert_qmark_to_psycopg(sql), params)
return cur
return self._conn.execute(sql, params)
def executescript(self, script):
if self.is_sqlite:
return self._conn.executescript(script)
for statement in _split_sql_script(_postgres_schema_sql(script)):
self.execute(statement)
return None
def commit(self):
return self._conn.commit()
def rollback(self):
return self._conn.rollback()
def close(self):
return self._conn.close()
def table_columns(self, table):
if self.is_postgres:
rows = self.execute(
'''SELECT a.attname AS name
FROM pg_catalog.pg_attribute a
JOIN pg_catalog.pg_class c ON c.oid = a.attrelid
JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
WHERE n.nspname = ?
AND c.relname = ?
AND a.attnum > 0
AND NOT a.attisdropped''',
(self.application_schema, table),
).fetchall()
return {row['name'] for row in rows}
return {row['name'] for row in self.execute(f'PRAGMA table_info({table})').fetchall()}
def table_column_details(self, table):
if self.is_postgres:
rows = self.execute(
'''SELECT a.attname AS name,
pg_catalog.format_type(a.atttypid, a.atttypmod) AS type,
a.attnotnull AS not_null,
a.attidentity AS identity_generation,
a.attgenerated AS generated_kind,
pg_catalog.pg_get_expr(d.adbin, d.adrelid) AS default_sql,
(d.oid IS NOT NULL) AS has_default,
pg_catalog.pg_get_serial_sequence(
pg_catalog.quote_ident(n.nspname) || '.' || pg_catalog.quote_ident(c.relname),
a.attname
) AS sequence_name,
COALESCE(i.indisprimary, false) AS primary_key
FROM pg_catalog.pg_attribute a
JOIN pg_catalog.pg_class c ON c.oid = a.attrelid
JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
LEFT JOIN pg_catalog.pg_attrdef d ON d.adrelid = a.attrelid AND d.adnum = a.attnum
LEFT JOIN pg_catalog.pg_index i ON i.indrelid = a.attrelid
AND i.indisprimary AND a.attnum = ANY(i.indkey)
WHERE n.nspname = ?
AND c.relname = ?
AND a.attnum > 0
AND NOT a.attisdropped
ORDER BY a.attnum''',
(self.application_schema, table),
).fetchall()
return {
row['name']: {
'type': str(row['type'] or '').lower(),
'not_null': bool(row['not_null']),
'default': str(row['default_sql'] or ''),
'has_default': bool(row['has_default']),
'primary_key': bool(row['primary_key']),
'identity': str(row['identity_generation'] or ''),
'generated': str(row['generated_kind'] or ''),
'sequence': str(row['sequence_name'] or ''),
}
for row in rows
}
rows = self.execute(f'PRAGMA table_info({table})').fetchall()
return {
row['name']: {
'type': str(row['type'] or '').lower(),
'not_null': bool(row['notnull']) or bool(row['pk']),
'default': str(row['dflt_value'] or ''),
'has_default': row['dflt_value'] is not None,
'primary_key': bool(row['pk']),
'identity': '',
'generated': '',
'sequence': '',
}
for row in rows
}
def table_indexes(self, table):
if self.is_postgres:
rows = self.execute(
'''SELECT idx.relname AS name,
i.indisunique AS is_unique,
i.indisprimary AS is_primary,
i.indisvalid AS is_valid,
i.indisready AS is_ready,
i.indislive AS is_live,
pg_catalog.pg_get_expr(i.indpred, i.indrelid) AS predicate,
ARRAY(
SELECT pg_catalog.pg_get_indexdef(i.indexrelid, position, true)
FROM pg_catalog.generate_series(1, i.indnkeyatts) AS position
ORDER BY position
) AS columns,
pg_catalog.pg_get_indexdef(i.indexrelid) AS sql
FROM pg_catalog.pg_index i
JOIN pg_catalog.pg_class tbl ON tbl.oid = i.indrelid
JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace
JOIN pg_catalog.pg_class idx ON idx.oid = i.indexrelid
WHERE n.nspname = ? AND tbl.relname = ?''',
(self.application_schema, table),
).fetchall()
return {
row['name']: {
'unique': bool(row['is_unique']),
'primary': bool(row['is_primary']),
'valid': bool(row['is_valid']),
'ready': bool(row['is_ready']),
'live': bool(row['is_live']),
'predicate': str(row['predicate'] or ''),
'columns': [str(value).strip('"') for value in (row['columns'] or [])],
'sql': str(row['sql'] or ''),
}
for row in rows
}
output = {}
for row in self.execute(f'PRAGMA index_list({table})').fetchall():
name = row['name']
columns = [item['name'] for item in self.execute(f'PRAGMA index_info({name})').fetchall()]
sql_row = self.execute(
"SELECT sql FROM sqlite_master WHERE type = 'index' AND name = ?",
(name,),
).fetchone()
sql = str(sql_row['sql'] if sql_row else '')
predicate = sql.split(' WHERE ', 1)[1] if ' WHERE ' in sql.upper() else ''
if ' WHERE ' in sql.upper():
position = sql.upper().index(' WHERE ')
predicate = sql[position + 7:]
output[name] = {
'unique': bool(row['unique']),
'primary': str(row['origin'] or '') == 'pk',
'valid': True,
'ready': True,
'live': True,
'predicate': predicate,
'columns': columns,
'sql': sql,
}
return output
def table_foreign_keys(self, table):
if self.is_postgres:
rows = self.execute(
'''SELECT con.conname AS name,
ARRAY(
SELECT src.attname
FROM pg_catalog.unnest(con.conkey) WITH ORDINALITY AS keys(attnum, position)
JOIN pg_catalog.pg_attribute src
ON src.attrelid = con.conrelid AND src.attnum = keys.attnum
ORDER BY keys.position
) AS columns,
ref_n.nspname AS referenced_schema,
ref.relname AS referenced_table,
ARRAY(
SELECT dst.attname
FROM pg_catalog.unnest(con.confkey) WITH ORDINALITY AS keys(attnum, position)
JOIN pg_catalog.pg_attribute dst
ON dst.attrelid = con.confrelid AND dst.attnum = keys.attnum
ORDER BY keys.position
) AS referenced_columns,
con.confupdtype::text AS update_action,
con.confdeltype::text AS delete_action,
con.convalidated AS is_valid
FROM pg_catalog.pg_constraint con
JOIN pg_catalog.pg_class tbl ON tbl.oid = con.conrelid
JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace
JOIN pg_catalog.pg_class ref ON ref.oid = con.confrelid
JOIN pg_catalog.pg_namespace ref_n ON ref_n.oid = ref.relnamespace
WHERE con.contype = 'f' AND n.nspname = ? AND tbl.relname = ?''',
(self.application_schema, table),
).fetchall()
action_names = {
'a': 'NO ACTION', 'r': 'RESTRICT', 'c': 'CASCADE',
'n': 'SET NULL', 'd': 'SET DEFAULT',
}
return {
row['name']: {
'columns': [str(value) for value in (row['columns'] or [])],
'referenced_schema': str(row['referenced_schema'] or ''),
'referenced_table': str(row['referenced_table'] or ''),
'referenced_columns': [str(value) for value in (row['referenced_columns'] or [])],
'update_action': action_names.get(str(row['update_action'] or ''), str(row['update_action'] or '')),
'delete_action': action_names.get(str(row['delete_action'] or ''), str(row['delete_action'] or '')),
'valid': bool(row['is_valid']),
}
for row in rows
}
output = {}
for row in self.execute(f'PRAGMA foreign_key_list({table})').fetchall():
name = f'fk_{row["id"]}'
current = output.setdefault(name, {
'columns': [],
'referenced_schema': 'main',
'referenced_table': str(row['table'] or ''),
'referenced_columns': [],
'update_action': str(row['on_update'] or '').upper(),
'delete_action': str(row['on_delete'] or '').upper(),
'valid': True,
})
current['columns'].append(str(row['from'] or ''))
current['referenced_columns'].append(str(row['to'] or ''))
return output
def table_check_constraints(self, table):
if self.is_postgres:
rows = self.execute(
'''SELECT con.conname AS name,
pg_catalog.pg_get_expr(con.conbin, con.conrelid, true) AS expression,
pg_catalog.pg_get_constraintdef(con.oid, true) AS definition,
con.convalidated AS is_valid
FROM pg_catalog.pg_constraint con
JOIN pg_catalog.pg_class tbl ON tbl.oid = con.conrelid
JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace
WHERE con.contype = 'c' AND n.nspname = ? AND tbl.relname = ?''',
(self.application_schema, table),
).fetchall()
return {
str(row['name']): {
'expression': str(row['expression'] or ''),
'definition': str(row['definition'] or ''),
'valid': bool(row['is_valid']),
}
for row in rows
}
row = self.execute(
"SELECT sql FROM sqlite_master WHERE type = 'table' AND name = ?",
(table,),
).fetchone()
return _sqlite_check_constraints(row['sql'] if row else '')
def table_triggers(self, table):
if self.is_postgres:
rows = self.execute(
'''SELECT trg.tgname AS name,
trg.tgenabled <> 'D' AS enabled,
pg_catalog.pg_get_triggerdef(trg.oid, true) AS sql,
pg_catalog.pg_get_functiondef(trg.tgfoid) AS function_sql
FROM pg_catalog.pg_trigger trg
JOIN pg_catalog.pg_class tbl ON tbl.oid = trg.tgrelid
JOIN pg_catalog.pg_namespace n ON n.oid = tbl.relnamespace
WHERE n.nspname = ? AND tbl.relname = ?
AND NOT trg.tgisinternal''',
(self.application_schema, table),
).fetchall()
return {
str(row['name']): {
'enabled': bool(row['enabled']),
'sql': str(row['sql'] or ''),
'function_sql': str(row['function_sql'] or ''),
}
for row in rows
}
rows = self.execute(
"SELECT name, sql FROM sqlite_master WHERE type = 'trigger' AND tbl_name = ?",
(table,),
).fetchall()
return {
str(row['name']): {
'enabled': True,
'sql': str(row['sql'] or ''),
'function_sql': '',
}
for row in rows
}
def table_exists(self, table):
if self.is_postgres:
row = self.execute(
'''SELECT c.oid AS name FROM pg_catalog.pg_class c
JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
WHERE n.nspname = ? AND c.relname = ? AND c.relkind IN ('r', 'p')''',
(self.application_schema, table),
).fetchone()
return bool(row and row['name'])
row = self.execute("SELECT name FROM sqlite_master WHERE type = 'table' AND name = ?", (table,)).fetchone()
return bool(row)
def insert_returning_id(self, sql, params=None):
if self.is_postgres:
cur = self.execute(f'{sql.rstrip()} RETURNING id', params)
row = cur.fetchone()
return row['id'] if row else None
cur = self.execute(sql, params)
return cur.lastrowid if int(getattr(cur, 'rowcount', 0) or 0) != 0 else None
def json_extract(self, column, path):
if self.is_sqlite:
return f"json_extract({column}, '{path}')"
parts = str(path or '').lstrip('$.').split('.')
pg_path = ','.join(part for part in parts if part)
return f"(NULLIF({column}, '')::jsonb #>> '{{{pg_path}}}')"
def connect_sqlite(path, timeout_sec=30, read_only=False, immutable=False, check_same_thread=True):
if read_only:
params = 'mode=ro&immutable=1' if immutable else 'mode=ro'
uri = 'file:' + str(path).replace('\\', '/') + '?' + params
conn = sqlite3.connect(uri, uri=True, timeout=max(1, int(timeout_sec or 30)), check_same_thread=check_same_thread)
else:
conn = sqlite3.connect(path, timeout=max(1, int(timeout_sec or 30)), check_same_thread=check_same_thread)
conn.row_factory = sqlite3.Row
return DatabaseConnection('sqlite', conn)
def _bounded_int(value, default, minimum=1):
try:
return max(minimum, int(value))
except (TypeError, ValueError):
return max(minimum, int(default))
def connect_postgres(
url,
connect_timeout_sec=None,
statement_timeout_ms=None,
lock_timeout_ms=None,
idle_in_transaction_timeout_ms=None,
tcp_user_timeout_ms=None,
):
parse_postgres_url(url)
try:
import psycopg
from psycopg.rows import dict_row
except ImportError as exc:
raise RuntimeError('PostgreSQL backend requires psycopg[binary]. Install app requirements first.') from exc
connect_timeout_sec = _bounded_int(
connect_timeout_sec if connect_timeout_sec is not None else os.getenv('TRUF_DB_CONNECT_TIMEOUT_SEC'),
DEFAULT_POSTGRES_CONNECT_TIMEOUT_SEC,
)
statement_timeout_ms = _bounded_int(
statement_timeout_ms if statement_timeout_ms is not None else os.getenv('TRUF_DB_STATEMENT_TIMEOUT_MS'),
DEFAULT_POSTGRES_STATEMENT_TIMEOUT_MS,
)
lock_timeout_ms = _bounded_int(
lock_timeout_ms if lock_timeout_ms is not None else os.getenv('TRUF_DB_LOCK_TIMEOUT_MS'),
DEFAULT_POSTGRES_LOCK_TIMEOUT_MS,
)
idle_in_transaction_timeout_ms = _bounded_int(
idle_in_transaction_timeout_ms if idle_in_transaction_timeout_ms is not None else os.getenv('TRUF_DB_IDLE_TRANSACTION_TIMEOUT_MS'),
DEFAULT_POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS,
)
tcp_user_timeout_ms = _bounded_int(
tcp_user_timeout_ms if tcp_user_timeout_ms is not None else os.getenv('TRUF_DB_TCP_USER_TIMEOUT_MS'),
DEFAULT_POSTGRES_TCP_USER_TIMEOUT_MS,
minimum=1000,
)
options = ' '.join((
f'-c search_path={POSTGRES_APPLICATION_SCHEMA}',
f'-c statement_timeout={statement_timeout_ms}',
f'-c lock_timeout={lock_timeout_ms}',
f'-c idle_in_transaction_session_timeout={idle_in_transaction_timeout_ms}',
))
for attempt in range(POSTGRES_CHILD_START_RETRY_ATTEMPTS):
try:
conn = psycopg.connect(
url,
row_factory=dict_row,
connect_timeout=connect_timeout_sec,
options=options,
tcp_user_timeout=tcp_user_timeout_ms,
keepalives=1,
keepalives_idle=5,
keepalives_interval=5,
keepalives_count=2,
)
break
except Exception as exc:
transient_child_start = any(
marker in str(exc).lower() for marker in POSTGRES_CHILD_START_RETRY_MARKERS
)
if not transient_child_start or attempt + 1 >= POSTGRES_CHILD_START_RETRY_ATTEMPTS:
raise
time.sleep(0.05 * (attempt + 1))
try:
cursor = conn.cursor()
cursor.execute(
"""SELECT pg_catalog.current_schema() AS schema_name,
pg_catalog.current_setting('search_path') AS search_path,
EXISTS (
SELECT 1
FROM pg_catalog.pg_namespace n
CROSS JOIN LATERAL pg_catalog.aclexplode(
COALESCE(n.nspacl, pg_catalog.acldefault('n', n.nspowner))
) acl
WHERE n.nspname = 'public'
AND acl.grantee = 0
AND acl.privilege_type = 'CREATE'
) AS public_create"""
)
row = cursor.fetchone()
cursor.close()
schema_name = row.get('schema_name') if isinstance(row, dict) else row[0] if row else None
search_path = row.get('search_path') if isinstance(row, dict) else row[1] if row else None
public_create = row.get('public_create', False) if isinstance(row, dict) else row[2] if row and len(row) > 2 else False
normalized_path = re.sub(r'[\s\"]', '', str(search_path or '').lower())
if schema_name != POSTGRES_APPLICATION_SCHEMA or normalized_path != 'public' or bool(public_create):
raise RuntimeError('PostgreSQL application schema/search_path validation failed')
conn.rollback()
except Exception:
try:
conn.close()
except Exception:
pass
raise
return DatabaseConnection('postgres', conn, application_schema=POSTGRES_APPLICATION_SCHEMA)
def connect_host_agent_postgres():
"""Open the one fixed peer-authenticated host-agent authority."""
if os.name != 'posix' or not hasattr(os, 'geteuid') or os.geteuid() != 0:
raise RuntimeError('PostgreSQL host-agent authority requires root on POSIX')
try:
import psycopg
from psycopg.rows import dict_row
except ImportError as exc:
raise RuntimeError(
'PostgreSQL backend requires psycopg[binary]. Install app requirements first.'
) from exc
options = ' '.join((
f'-c search_path={POSTGRES_APPLICATION_SCHEMA}',
f'-c statement_timeout={DEFAULT_POSTGRES_STATEMENT_TIMEOUT_MS}',
f'-c lock_timeout={DEFAULT_POSTGRES_LOCK_TIMEOUT_MS}',
f'-c idle_in_transaction_session_timeout={DEFAULT_POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS}',
))
conn = psycopg.connect(
dbname=HOST_AGENT_POSTGRES_DATABASE,
user=HOST_AGENT_POSTGRES_USER,
host=HOST_AGENT_POSTGRES_SOCKET_DIRECTORY,
port=HOST_AGENT_POSTGRES_PORT,
row_factory=dict_row,
connect_timeout=DEFAULT_POSTGRES_CONNECT_TIMEOUT_SEC,
options=options,
sslmode='disable',
)
try:
cursor = conn.cursor()
cursor.execute(
"""SELECT pg_catalog.current_schema() AS schema_name,
pg_catalog.current_setting('search_path') AS search_path,
CURRENT_USER AS current_user,
current_database() AS database_name,
pg_catalog.inet_server_addr() IS NULL AS unix_socket,
pg_catalog.current_setting('port')::integer AS port,
EXISTS (
SELECT 1
FROM pg_catalog.pg_namespace n
CROSS JOIN LATERAL pg_catalog.aclexplode(
COALESCE(n.nspacl, pg_catalog.acldefault('n', n.nspowner))
) acl
WHERE n.nspname = 'public'
AND acl.grantee = 0
AND acl.privilege_type = 'CREATE'
) AS public_create"""
)
row = cursor.fetchone()
cursor.close()
normalized_path = re.sub(
r'[\s\"]', '', str((row or {}).get('search_path') or '').lower()
)
if (
not isinstance(row, dict)
or row.get('schema_name') != POSTGRES_APPLICATION_SCHEMA
or normalized_path != POSTGRES_APPLICATION_SCHEMA
or row.get('current_user') != HOST_AGENT_POSTGRES_USER
or row.get('database_name') != HOST_AGENT_POSTGRES_DATABASE
or row.get('unix_socket') is not True
or row.get('port') != HOST_AGENT_POSTGRES_PORT
or bool(row.get('public_create'))
):
raise RuntimeError('PostgreSQL host-agent authority validation failed')
conn.rollback()
except Exception:
try:
conn.close()
except Exception:
pass
raise
return DatabaseConnection(
'postgres', conn, application_schema=POSTGRES_APPLICATION_SCHEMA,
)
File diff suppressed because it is too large Load Diff
+546
View File
@@ -0,0 +1,546 @@
import sys
sys.dont_write_bytecode = True
import argparse
import contextlib
import hmac
import json
import os
import re
from db_backend import canonical_postgres_url, is_postgres_url
from docker_depth_experiment import (
DOCKER_DEPTH_QUERY_COUNT,
_docker_depth_authority,
_experiment_identity_matches,
_stored_cohort_plan,
apply_docker_depth_cohort_manifest,
apply_docker_depth_hold_manifest,
apply_docker_depth_reactivation_manifest,
apply_docker_depth_resolver_disposition_manifest,
apply_docker_depth_resolver_refund_manifest,
canonical_docker_depth_plan_hash,
generate_docker_depth_cohort_manifest,
generate_docker_depth_hold_manifest,
generate_docker_depth_reactivation_manifest,
generate_docker_depth_resolver_disposition_manifest,
generate_docker_depth_resolver_refund_manifest,
summarize_docker_depth_fresh_coverage,
validate_docker_depth_cohort_manifest,
validate_docker_depth_config,
validate_docker_depth_hold_manifest,
validate_docker_depth_reactivation_manifest,
validate_docker_depth_resolver_disposition_manifest,
validate_docker_depth_resolver_refund_manifest,
validate_dockerhub_discovery_policies,
)
from migrate_runtime_safety import (
postgres_migration_guard,
require_local_sources_stopped,
)
from paths import apply_path_config
from postgres_runtime import (
canonical_database_url,
load_postgres_environment,
verify_cluster_identity,
)
from runtime_security import (
ClusterAuthorityLock,
MAX_EXTENDED_PRIVATE_JSON_BYTES,
read_private_json,
reject_reparse_components,
require_private_directory,
require_private_file,
preflight_lifecycle_paths,
write_private_json_exclusive,
)
from scanner_db import ScannerDB
APPLICATION_NAME = 'truf-docker-depth-operator'
MANIFEST_MAX_BYTES = MAX_EXTENDED_PRIVATE_JSON_BYTES
_SHA256_RE = re.compile(r'^[a-f0-9]{64}$')
_DISABLED_ACTIONS = frozenset({
'generate-cohort', 'apply-cohort', 'generate-hold', 'apply-hold',
})
_ENABLED_ACTIONS = frozenset({
'generate-reactivation', 'apply-reactivation',
'generate-resolver-disposition', 'apply-resolver-disposition',
'generate-resolver-refund', 'apply-resolver-refund',
})
def load_config(path):
"""Load path-expanded YAML only; validation and runtime access are separate."""
try:
import yaml
except ImportError as exc:
raise RuntimeError('PyYAML is required') from exc
with open(path, 'r', encoding='utf-8') as handle:
return apply_path_config(yaml.safe_load(handle) or {}, path)
def provenance_policy_sha256(validated):
source = validated.normalized_config['sources']['dockerhub']
policies = validate_dockerhub_discovery_policies(
source, validated.experiment.queries,
)
hashes = {policy['policy_sha256'] for policy in policies}
if len(policies) != DOCKER_DEPTH_QUERY_COUNT or len(hashes) != 1:
raise RuntimeError('Docker depth provenance policy authority is ambiguous')
return next(iter(hashes))
def _action_name(args):
for attribute, name in (
('status', 'status'),
('generate_cohort_manifest', 'generate-cohort'),
('apply_cohort_manifest', 'apply-cohort'),
('generate_hold_manifest', 'generate-hold'),
('apply_hold_manifest', 'apply-hold'),
('generate_reactivation_manifest', 'generate-reactivation'),
('apply_reactivation_manifest', 'apply-reactivation'),
('generate_resolver_disposition_manifest', 'generate-resolver-disposition'),
('apply_resolver_disposition_manifest', 'apply-resolver-disposition'),
('generate_resolver_refund_manifest', 'generate-resolver-refund'),
('apply_resolver_refund_manifest', 'apply-resolver-refund'),
):
if getattr(args, attribute, None):
return name
raise RuntimeError('Docker depth operator action is unavailable')
def _require_action_arguments(parser, args, action):
applying = action.startswith('apply-')
supplied_apply_option = bool(
args.confirm_apply or args.sources_stopped or args.approve_sha256
)
if applying:
if not args.confirm_apply or not args.sources_stopped or not args.approve_sha256:
parser.error(
'apply actions require --approve-sha256, --apply, and --sources-stopped'
)
if not _SHA256_RE.fullmatch(args.approve_sha256):
parser.error('--approve-sha256 must be one lowercase SHA-256 value')
elif supplied_apply_option:
parser.error('approval options are valid only for apply actions')
def _require_action_config_state(experiment, action):
if action in _DISABLED_ACTIONS and experiment.enabled:
raise RuntimeError('Docker depth reviewed preparation requires disabled config')
if action in _ENABLED_ACTIONS and not experiment.enabled:
raise RuntimeError('Docker depth reviewed release requires enabled config')
def _manifest_path(path, *, existing):
absolute = reject_reparse_components(os.path.abspath(os.fspath(path)))
require_private_directory(os.path.dirname(absolute), create=False)
if existing:
require_private_file(absolute)
elif os.path.lexists(absolute):
require_private_file(absolute)
return absolute
def _publish_manifest(path, manifest):
absolute = _manifest_path(path, existing=False)
if os.path.lexists(absolute):
if read_private_json(absolute, max_bytes=MANIFEST_MAX_BYTES) != manifest:
raise RuntimeError('A different reviewed manifest already exists')
return absolute, False
try:
write_private_json_exclusive(
absolute, manifest, max_bytes=MANIFEST_MAX_BYTES,
)
except FileExistsError:
if read_private_json(absolute, max_bytes=MANIFEST_MAX_BYTES) != manifest:
raise RuntimeError('Reviewed manifest publication raced a different file')
return absolute, False
require_private_file(absolute)
return absolute, True
def _read_approved_manifest(path, validator, experiment, policy_sha256, approved):
absolute = _manifest_path(path, existing=True)
manifest = read_private_json(absolute, max_bytes=MANIFEST_MAX_BYTES)
normalized, manifest_sha256 = validator(
manifest, experiment, policy_sha256,
)
if not hmac.compare_digest(manifest_sha256, approved):
raise ValueError('Reviewed manifest approval hash conflicts')
return absolute, normalized, manifest_sha256
def _prepare_action(args, action, experiment, policy_sha256):
if action.startswith('generate-'):
attribute = action.replace('-', '_') + '_manifest'
path = _manifest_path(getattr(args, attribute), existing=False)
return {'path': path}
if not action.startswith('apply-'):
return {}
kind = action.removeprefix('apply-')
validator = {
'cohort': validate_docker_depth_cohort_manifest,
'hold': validate_docker_depth_hold_manifest,
'reactivation': validate_docker_depth_reactivation_manifest,
'resolver-disposition': validate_docker_depth_resolver_disposition_manifest,
'resolver-refund': validate_docker_depth_resolver_refund_manifest,
}[kind]
path = getattr(args, f'apply_{kind.replace("-", "_")}_manifest')
absolute, manifest, manifest_sha256 = _read_approved_manifest(
path, validator, experiment, policy_sha256, args.approve_sha256,
)
return {
'path': absolute,
'manifest': manifest,
'manifest_sha256': manifest_sha256,
}
def _verify_online_cluster_identity(db, dsn, identity):
canonical = canonical_postgres_url(
dsn, identity['database'], identity['user'], identity['port'],
)
if canonical != dsn:
raise RuntimeError('Managed PostgreSQL DSN is not canonical')
row = db.conn.execute(
'''SELECT pg_catalog.current_database() AS database,
CURRENT_USER AS user_name,
pg_catalog.current_setting('data_directory') AS data_directory,
pg_catalog.current_setting('port')::integer AS port,
(SELECT system_identifier::text
FROM pg_catalog.pg_control_system()) AS system_identifier'''
).fetchone()
checks = {
'database': (str(row['database']), str(identity['database'])),
'user': (str(row['user_name']), str(identity['user'])),
'data_directory': (
os.path.normcase(os.path.realpath(os.path.abspath(row['data_directory']))),
os.path.normcase(os.path.realpath(os.path.abspath(identity['data_directory']))),
),
'port': (int(row['port']), int(identity['port'])),
'system_identifier': (
str(row['system_identifier']), str(identity['system_identifier']),
),
}
if any(actual != expected for actual, expected in checks.values()):
raise RuntimeError('Online PostgreSQL identity does not match private authority')
db.conn.commit()
@contextlib.contextmanager
def operator_database(config_path, config, *, read_only):
preflight_lifecycle_paths(config_path, config)
load_postgres_environment(config_path, config)
dsn = canonical_database_url()
if not dsn or not is_postgres_url(dsn):
raise RuntimeError('Canonical managed PostgreSQL DSN is unavailable')
with ClusterAuthorityLock(config, endpoint_dsn=dsn):
require_local_sources_stopped(config)
identity = verify_cluster_identity(config)
dsn = canonical_postgres_url(
dsn, identity['database'], identity['user'], identity['port'],
)
db = ScannerDB(db_url=dsn, initialize=False)
try:
if not db.enabled:
raise RuntimeError('Managed PostgreSQL connection is unavailable')
_verify_online_cluster_identity(db, dsn, identity)
db.set_application_name(APPLICATION_NAME)
with postgres_migration_guard(db):
db.require_runtime_safety_schema()
db.require_final_cutover()
if read_only:
db.conn.execute('SET default_transaction_read_only = on')
db.conn.commit()
yield db
finally:
db.close()
def _status(db, experiment, policy_sha256):
authority = _docker_depth_authority(experiment, policy_sha256)
coverage = summarize_docker_depth_fresh_coverage(
db, experiment, policy_sha256,
)
row = db.conn.execute(
'SELECT * FROM docker_depth_experiments WHERE experiment_key = ?',
(authority['experiment_key'],),
).fetchone()
state = 'absent'
plan_sha256 = ''
hold_manifest_sha256 = ''
fence_active = 0
counts = {
'owned_policy_events': 0,
'planned_queries': 0,
'planned_repositories': 0,
'targets': 0,
'unreleased_holds': 0,
}
if row:
if not _experiment_identity_matches(row, authority):
raise RuntimeError('Docker depth persisted authority drifted')
state = str(row['state'])
plan_sha256 = str(row['plan_sha256'] or '')
hold_manifest_sha256 = str(row['hold_manifest_sha256'] or '')
fence_active = int(any(
row[name] is not None
for name in ('fence_owner', 'fence_token', 'fence_expires_at')
))
if plan_sha256:
stored = _stored_cohort_plan(db.conn, row, authority)
if canonical_docker_depth_plan_hash(stored) != plan_sha256:
raise RuntimeError('Docker depth persisted cohort hash drifted')
count_row = db.conn.execute(
'''SELECT
(SELECT COUNT(*) FROM docker_depth_experiment_queries
WHERE experiment_id = ?) AS planned_queries,
(SELECT COUNT(*) FROM docker_depth_experiment_repositories
WHERE experiment_id = ?) AS planned_repositories,
(SELECT COUNT(*) FROM docker_depth_experiment_targets
WHERE experiment_id = ?) AS targets,
(SELECT COUNT(*) FROM target_queue_policy_events
WHERE experiment_id = ?) AS owned_policy_events,
(SELECT COUNT(*)
FROM target_queue_policy_events cold_event
LEFT JOIN target_queue_policy_events reverse_event
ON reverse_event.reverses_event_id = cold_event.id
WHERE cold_event.experiment_id = ?
AND cold_event.action = 'cold'
AND reverse_event.id IS NULL) AS unreleased_holds''',
(row['id'], row['id'], row['id'], row['id'], row['id']),
).fetchone()
counts = {name: int(count_row[name]) for name in counts}
db.conn.commit()
return {
'action': 'status',
'config_enabled': bool(experiment.enabled),
'config_sha256': authority['config_sha256'],
'counts': counts,
'experiment_present': bool(row),
'fence_active': fence_active,
'fresh_coverage': coverage,
'hold_manifest_sha256': hold_manifest_sha256,
'ordered_queries_sha256': authority['ordered_queries_sha256'],
'plan_sha256': plan_sha256,
'provenance_policy_sha256': authority['provenance_policy_sha256'],
'selector_sha256': authority['selector_sha256'],
'state': state,
}
def _execute_action(db, action, prepared, experiment, policy_sha256):
if action == 'status':
return _status(db, experiment, policy_sha256)
if action == 'generate-cohort':
manifest, manifest_sha256 = generate_docker_depth_cohort_manifest(
db, experiment, policy_sha256,
)
path, created = _publish_manifest(prepared['path'], manifest)
return {
'action': action,
'files_created': int(created),
'manifest_sha256': manifest_sha256,
'path': os.path.basename(path),
'plan_sha256': manifest['plan_sha256'],
'queries': len(manifest['plan']['queries']),
'repositories': sum(
len(item['repositories']) for item in manifest['plan']['queries']
),
}
if action == 'apply-cohort':
result = apply_docker_depth_cohort_manifest(
db, experiment, policy_sha256, prepared['manifest'],
prepared['manifest_sha256'],
)
return {
'action': action,
'manifest_sha256': prepared['manifest_sha256'],
'path': os.path.basename(prepared['path']),
'plan_sha256': result['plan_sha256'],
'plans_persisted': int(bool(result['planned'])),
'queries': len(result['plan']['queries']),
'repositories': sum(
len(item['repositories']) for item in result['plan']['queries']
),
}
if action == 'generate-hold':
manifest, manifest_sha256 = generate_docker_depth_hold_manifest(
db, experiment, policy_sha256,
)
path, created = _publish_manifest(prepared['path'], manifest)
return {
'action': action,
'conflicts': manifest['conflict_count'],
'entries': manifest['entry_count'],
'files_created': int(created),
'manifest_sha256': manifest_sha256,
'path': os.path.basename(path),
'plan_sha256': manifest['plan_sha256'],
}
if action == 'apply-hold':
result = apply_docker_depth_hold_manifest(
db, experiment, policy_sha256, prepared['manifest'],
prepared['manifest_sha256'],
)
return {
'action': action,
'conflicts': int(result['conflicts']),
'duplicates': int(result['duplicates']),
'manifest_sha256': prepared['manifest_sha256'],
'path': os.path.basename(prepared['path']),
'transitioned': int(result['transitioned']),
}
if action == 'generate-reactivation':
manifest, manifest_sha256 = generate_docker_depth_reactivation_manifest(
db, experiment, policy_sha256,
)
path, created = _publish_manifest(prepared['path'], manifest)
return {
'action': action,
'entries': manifest['entry_count'],
'files_created': int(created),
'hold_manifest_sha256': manifest['hold_manifest_sha256'],
'manifest_sha256': manifest_sha256,
'path': os.path.basename(path),
}
if action == 'apply-reactivation':
result = apply_docker_depth_reactivation_manifest(
db, experiment, policy_sha256, prepared['manifest'],
prepared['manifest_sha256'],
)
return {
'action': action,
'duplicates': int(result['duplicates']),
'manifest_sha256': prepared['manifest_sha256'],
'path': os.path.basename(prepared['path']),
'transitioned': int(result['transitioned']),
}
if action == 'generate-resolver-disposition':
manifest, manifest_sha256 = (
generate_docker_depth_resolver_disposition_manifest(
db, experiment, policy_sha256,
)
)
path, created = _publish_manifest(prepared['path'], manifest)
return {
'action': action,
'files_created': int(created),
'manifest_sha256': manifest_sha256,
'outcome': manifest['entries'][0]['outcome'],
'path': os.path.basename(path),
}
if action == 'apply-resolver-disposition':
result = apply_docker_depth_resolver_disposition_manifest(
db, experiment, policy_sha256, prepared['manifest'],
prepared['manifest_sha256'],
)
return {
'action': action,
'applied': int(result['applied']),
'duplicates': int(result['duplicates']),
'manifest_sha256': prepared['manifest_sha256'],
'outcome': result['outcome'],
'path': os.path.basename(prepared['path']),
'state': result['state'],
}
if action == 'generate-resolver-refund':
manifest, manifest_sha256 = generate_docker_depth_resolver_refund_manifest(
db, experiment, policy_sha256, prepared['evidence_log_path'],
)
path, created = _publish_manifest(prepared['path'], manifest)
return {
'action': action,
'entries': manifest['entry_count'],
'files_created': int(created),
'held_entries': manifest['held_entry_count'],
'manifest_sha256': manifest_sha256,
'path': os.path.basename(path),
'refund_attempts': manifest['refund_attempts'],
}
if action == 'apply-resolver-refund':
result = apply_docker_depth_resolver_refund_manifest(
db, experiment, policy_sha256, prepared['manifest'],
prepared['manifest_sha256'], prepared['evidence_log_path'],
)
return {
'action': action,
'duplicates': int(result['duplicates']),
'manifest_sha256': prepared['manifest_sha256'],
'path': os.path.basename(prepared['path']),
'refunded': int(result['refunded']),
'state': result['state'],
}
raise RuntimeError('Docker depth operator action is unsupported')
def parse_args(argv=None):
parser = argparse.ArgumentParser(
description='Offline reviewed operator for the bounded Docker depth experiment.',
)
parser.add_argument(
'--config', default=os.path.join(os.path.dirname(__file__), 'config.yaml'),
)
actions = parser.add_mutually_exclusive_group(required=True)
actions.add_argument('--status', action='store_true')
actions.add_argument('--generate-cohort-manifest')
actions.add_argument('--apply-cohort-manifest')
actions.add_argument('--generate-hold-manifest')
actions.add_argument('--apply-hold-manifest')
actions.add_argument('--generate-reactivation-manifest')
actions.add_argument('--apply-reactivation-manifest')
actions.add_argument('--generate-resolver-disposition-manifest')
actions.add_argument('--apply-resolver-disposition-manifest')
actions.add_argument('--generate-resolver-refund-manifest')
actions.add_argument('--apply-resolver-refund-manifest')
parser.add_argument('--approve-sha256')
parser.add_argument('--apply', dest='confirm_apply', action='store_true')
parser.add_argument('--sources-stopped', action='store_true')
args = parser.parse_args(argv)
action = _action_name(args)
_require_action_arguments(parser, args, action)
return args, action
def main(argv=None):
try:
args, action = parse_args(argv)
config_path = os.path.abspath(args.config)
config = load_config(config_path)
validated = validate_docker_depth_config(
config, managed_postgres=True, final_cutover=True,
)
if validated.experiment is None:
raise RuntimeError('Docker depth experiment configuration is unavailable')
experiment = validated.experiment
_require_action_config_state(experiment, action)
policy_sha256 = provenance_policy_sha256(validated)
prepared = _prepare_action(
args, action, experiment, policy_sha256,
)
if action in ('generate-resolver-refund', 'apply-resolver-refund'):
log_path = reject_reparse_components(os.path.abspath(os.path.join(
validated.normalized_config['global']['log_dir'], 'dockerhub.log',
)))
require_private_file(log_path)
prepared['evidence_log_path'] = log_path
with operator_database(
config_path, validated.normalized_config,
read_only=not action.startswith('apply-'),
) as db:
report = _execute_action(
db, action, prepared, experiment, policy_sha256,
)
print(json.dumps(report, ensure_ascii=True, sort_keys=True))
return 0
except Exception as exc:
raise SystemExit(
f'Docker depth operator failed closed: {type(exc).__name__}'
) from None
if __name__ == '__main__':
main()
File diff suppressed because it is too large Load Diff
+849
View File
@@ -0,0 +1,849 @@
import sys
sys.dont_write_bytecode = True
if not sys.dont_write_bytecode:
raise RuntimeError('Docker shadow runner could not disable bytecode writes')
import argparse
import copy
import hashlib
import math
import os
import re
import secrets
import socket
import time
import scanner
import console_runner
from keycheck_candidates import extract_candidates
from lifecycle_authority import require_active_supervisor_child
from scanner_db import (
DOCKER_ADAPTIVE_GATE_MAX_CONTROLS,
DOCKER_ADAPTIVE_GATE_MIN_CONTROLS,
DOCKER_ADAPTIVE_LAYER_CLASS_ORDER,
DOCKER_ADAPTIVE_SELECTOR_VERSION,
DOCKER_ADAPTIVE_SHADOW_SELECTION_METRIC_KEYS,
ScannerDB,
canonical_docker_layer_plan_bytes,
docker_content_media_class,
docker_layer_execution_policy_sha256,
docker_layer_selection_policy_sha256,
finding_identity,
select_docker_adaptive_payload,
validate_docker_adaptive_checkpoint,
validate_docker_adaptive_shadow_selection_metrics,
validate_docker_layer_execution,
validate_docker_layer_limits,
validate_docker_layer_plan,
validate_docker_layer_resolution,
)
_SHA256_RE = re.compile(r'[a-f0-9]{64}')
_ROUTED_SERVICE_RE = re.compile(r'[a-z0-9][a-z0-9_.-]{0,63}')
SHADOW_FAILURE_METRIC_KEYS = (
'diagnostic_full_incomplete',
'diagnostic_adaptive_incomplete',
'diagnostic_blob_chunk_processing',
'diagnostic_blob_detector_timeout',
'diagnostic_blob_network',
'diagnostic_blob_timeout',
'diagnostic_blob_mixed',
'diagnostic_blob_other',
)
_SHADOW_BLOB_FAILURE_METRICS = {
'chunk_processing': 'diagnostic_blob_chunk_processing',
'detector_timeout': 'diagnostic_blob_detector_timeout',
'network': 'diagnostic_blob_network',
'transfer_timeout': 'diagnostic_blob_timeout',
'timeout': 'diagnostic_blob_timeout',
'mixed': 'diagnostic_blob_mixed',
}
class DockerShadowPrivacyError(RuntimeError):
pass
def empty_selection_metrics():
return {name: 0 for name in DOCKER_ADAPTIVE_SHADOW_SELECTION_METRIC_KEYS}
def empty_failure_metrics():
return {name: 0 for name in SHADOW_FAILURE_METRIC_KEYS}
def _blob_failure_metric(error_code):
return _SHADOW_BLOB_FAILURE_METRICS.get(
str(error_code or ''), 'diagnostic_blob_other',
)
def _canonical_private_plan(plan):
plan = validate_docker_layer_plan(plan)
payload = canonical_docker_layer_plan_bytes(plan)
return plan, hashlib.sha256(payload).hexdigest()
def _validate_duplicate_descriptors(descriptors):
identities = {}
for descriptor in descriptors:
identity = (
descriptor['kind'], descriptor['size'],
docker_content_media_class(descriptor['kind'], descriptor['media_type']),
)
previous = identities.setdefault(descriptor['digest'], identity)
if previous != identity:
raise ValueError('Docker shadow duplicate digest metadata conflicts')
def build_private_adaptive_plan(
resolved, payload_classes, limits, checkpoint, scan_policy_sha256,
):
resolved = validate_docker_layer_resolution(resolved)
limits = validate_docker_layer_limits(limits)
checkpoint = validate_docker_adaptive_checkpoint(checkpoint)
scan_policy_sha256 = str(scan_policy_sha256 or '').lower()
if not _SHA256_RE.fullmatch(scan_policy_sha256):
raise ValueError('Docker shadow scan policy must be a lowercase SHA-256')
descriptors = [resolved['config'], *resolved['layers']]
_validate_duplicate_descriptors(descriptors)
decisions = select_docker_adaptive_payload(
descriptors, payload_classes, limits, covered_digests=(),
)
metrics = empty_selection_metrics()
planned = []
omitted = 0
for descriptor, selected, reason, payload_class in decisions:
if reason == 'config_selected':
metrics['selected_config'] += 1
elif reason.startswith('selected_'):
metrics[reason] += 1
elif reason == 'already_covered':
metrics['reuse_already_covered'] += 1
elif reason == 'duplicate_digest':
metrics['reuse_duplicate_digest'] += 1
elif not selected:
metric_name = f'omitted_{reason}'
if metric_name not in metrics:
raise ValueError('Docker shadow selection reason is not aggregate-safe')
metrics[metric_name] += 1
omitted += 1
planned.append({
**descriptor,
'payload_class': payload_class,
'selected': bool(selected),
'selection_reason': reason,
'coverage_state': 'selected' if selected else 'skipped',
'lease_token': None,
'attempt': 0,
'max_attempts': limits['blob_max_attempts'],
})
plan, plan_sha256 = _canonical_private_plan({
'version': 2,
'image': resolved['image'],
'repository': resolved['repository'],
'manifest_digest': resolved['manifest_digest'],
'platform_os': resolved['platform_os'],
'platform_arch': resolved['platform_arch'],
'manifest_media_type': resolved['manifest_media_type'],
'limits': limits,
'selector_version': DOCKER_ADAPTIVE_SELECTOR_VERSION,
'selection_policy_sha256': docker_layer_selection_policy_sha256(limits),
'scan_policy_sha256': scan_policy_sha256,
'execution_policy_sha256': docker_layer_execution_policy_sha256(
scan_policy_sha256, limits,
),
'checkpoint': checkpoint,
'descriptors': planned,
})
return {
'plan': plan,
'plan_sha256': plan_sha256,
'selection_metrics': validate_docker_adaptive_shadow_selection_metrics(metrics),
'omitted_descriptor_count': omitted,
}
def _checkpoint_order(descriptor):
if descriptor['kind'] == 'config':
return (0, 0, -descriptor['position'], descriptor['size'], descriptor['digest'])
try:
class_rank = DOCKER_ADAPTIVE_LAYER_CLASS_ORDER.index(descriptor['payload_class'])
except ValueError as exc:
raise ValueError('Docker shadow descriptor class is not schedulable') from exc
return (1, class_rank, -descriptor['position'], descriptor['size'], descriptor['digest'])
def lease_private_adaptive_checkpoint(plan, token_factory=None):
plan = validate_docker_layer_plan(plan)
if plan['version'] != 2:
raise ValueError('Docker shadow checkpoints require a version-two plan')
if any(
descriptor['coverage_state'] in ('leased', 'shared_pending')
for descriptor in plan['descriptors']
):
raise ValueError('Docker shadow plan already contains active leases')
pending = {}
for descriptor in plan['descriptors']:
if descriptor['coverage_state'] == 'selected':
pending.setdefault(descriptor['digest'], []).append(descriptor)
if not pending:
return None
groups = []
for digest, descriptors in pending.items():
attempts = {item['attempt'] for item in descriptors}
maximums = {item['max_attempts'] for item in descriptors}
if len(attempts) != 1 or len(maximums) != 1:
raise ValueError('Docker shadow duplicate attempts conflict')
if next(iter(attempts)) >= next(iter(maximums)):
raise ValueError('Docker shadow plan exceeded its private attempt budget')
representative = min(descriptors, key=_checkpoint_order)
groups.append((representative, digest))
groups.sort(key=lambda item: _checkpoint_order(item[0]))
leased_digests = set()
leased_bytes = 0
max_blobs = plan['checkpoint']['max_blobs']
max_bytes = plan['checkpoint']['max_bytes']
for descriptor, digest in groups:
if len(leased_digests) >= max_blobs:
continue
if leased_digests and leased_bytes + descriptor['size'] > max_bytes:
continue
leased_digests.add(digest)
leased_bytes += descriptor['size']
if not leased_digests:
raise ValueError('Docker shadow checkpoint made no bounded progress')
token_factory = token_factory or (lambda: secrets.token_urlsafe(32))
tokens = {digest: str(token_factory()) for digest in leased_digests}
leased_plan = copy.deepcopy(plan)
for descriptor in leased_plan['descriptors']:
if descriptor['digest'] not in leased_digests:
continue
descriptor['coverage_state'] = 'leased'
descriptor['lease_token'] = tokens[descriptor['digest']]
descriptor['attempt'] += 1
leased_plan, plan_sha256 = _canonical_private_plan(leased_plan)
return {'plan': leased_plan, 'plan_sha256': plan_sha256}
def apply_private_adaptive_execution(plan, execution):
plan, plan_sha256 = _canonical_private_plan(plan)
execution = validate_docker_layer_execution(execution, plan, plan_sha256)
records = {record['digest']: record for record in execution['blobs']}
next_plan = copy.deepcopy(plan)
for descriptor in next_plan['descriptors']:
if descriptor['coverage_state'] != 'leased':
continue
status = records[descriptor['digest']]['status']
if status == 'covered':
next_state = 'covered'
elif status == 'retryable_failed' and descriptor['attempt'] < descriptor['max_attempts']:
next_state = 'selected'
else:
next_state = 'terminal_failed'
descriptor['coverage_state'] = next_state
descriptor['lease_token'] = None
return _canonical_private_plan(next_plan)[0]
def private_result_identities(result, normalized_target):
if not isinstance(result, dict):
raise ValueError('Docker shadow private result must be an object')
routed = set()
detectors = set()
try:
scanner.strip_nearby_context_for_persistence(result)
findings = result.get('findings') or []
if not isinstance(findings, list):
raise ValueError('Docker shadow private findings must be a list')
for finding in findings:
if not isinstance(finding, dict):
raise ValueError('Docker shadow private finding must be an object')
detector_hash = str(
finding_identity('dockerhub', normalized_target, finding)[2] or ''
).lower()
if not _SHA256_RE.fullmatch(detector_hash):
raise DockerShadowPrivacyError('Docker shadow detector identity is invalid')
detectors.add(detector_hash)
for candidate in extract_candidates(finding):
service = str(candidate.service or '')
provider_key_hash = str(candidate.provider_key_hash or '').lower()
if (
not _ROUTED_SERVICE_RE.fullmatch(service)
or not _SHA256_RE.fullmatch(provider_key_hash)
):
raise DockerShadowPrivacyError('Docker shadow routed identity is invalid')
routed.add((service, provider_key_hash))
finally:
findings = result.get('findings') if isinstance(result, dict) else None
if isinstance(findings, list):
for finding in findings:
if isinstance(finding, dict):
finding.clear()
findings.clear()
result.clear()
return frozenset(routed), frozenset(detectors)
def run_timed_private_side(
side, operation, durable_checkpoint, *, timeout_sec,
monotonic_ns=time.monotonic_ns,
):
if side not in ('full', 'adaptive'):
raise ValueError('Docker shadow side is invalid')
with scanner.scan_slot_scope(['docker-shadow', side], timeout_sec=timeout_sec):
started_ns = monotonic_ns()
metrics = operation()
if not isinstance(metrics, dict) or any(
not isinstance(name, str)
or isinstance(value, bool)
or not isinstance(value, int)
or value < 0
for name, value in metrics.items()
):
raise ValueError('Docker shadow private sink accepts only non-negative aggregates')
durable_checkpoint()
elapsed_ns = max(1, monotonic_ns() - started_ns)
return metrics, max(1, math.ceil(elapsed_ns / 1_000_000))
def _clear_private_result(result):
if not isinstance(result, dict):
return
findings = result.get('findings')
if isinstance(findings, list):
for finding in findings:
if isinstance(finding, dict):
finding.clear()
findings.clear()
result.clear()
def _full_scan_incomplete_reason(result):
if result.get('source_failure'):
return 'full_scan_source_failure'
if result.get('skipped'):
return 'full_scan_skipped'
return 'full_scan_error'
def _adaptive_scan_incomplete_reason(exc):
if isinstance(exc, scanner.DockerLayerInfrastructureError):
return 'adaptive_scan_infrastructure'
if isinstance(exc, scanner.DockerContentTransferError):
return 'adaptive_scan_transfer'
if isinstance(exc, TimeoutError):
return 'adaptive_scan_timeout'
if isinstance(exc, scanner.DockerRemoteAccessError):
return 'adaptive_scan_remote_access'
if isinstance(exc, ValueError):
return 'adaptive_scan_contract'
return 'adaptive_scan_error'
def shadow_failure_reason_code(exc):
if isinstance(exc, DockerShadowPrivacyError):
return 'privacy_violation'
if isinstance(exc, (KeyboardInterrupt, SystemExit)):
return 'operator_interrupt'
if exc.__class__.__name__ == 'ScanSlotFatalError':
return 'scan_slot_fatal'
if isinstance(exc, TimeoutError):
return 'timeout'
if isinstance(exc, ValueError):
return 'invalid_contract'
if isinstance(exc, RuntimeError):
message = str(exc).lower()
if 'checkpoint' in message:
return 'checkpoint_failure'
if 'fence' in message or 'lease' in message:
return 'report_fence_failure'
if 'cohort' in message or 'control' in message:
return 'control_failure'
return 'runtime_failure'
return 'unexpected_failure'
def execute_private_adaptive_scan(
normalized_target, *, limits, checkpoint, scan_policy_sha256, timeout_sec,
platform_os='linux', platform_arch='amd64', min_free_bytes=0,
scan_kwargs=None,
):
timeout_sec = max(1, int(timeout_sec))
deadline = time.monotonic() + timeout_sec
scan_kwargs = dict(scan_kwargs or {})
resolved, bearer_auth = scanner.resolve_docker_content_manifest(
normalized_target,
platform_os=str(platform_os or 'linux'),
platform_arch=str(platform_arch or 'amd64'),
deadline=deadline,
)
payload_classes, bearer_auth = scanner.fetch_docker_config_payload_classes(
resolved, bearer_auth, deadline=deadline,
min_free_bytes=max(0, int(min_free_bytes or 0)),
)
built = build_private_adaptive_plan(
resolved, payload_classes, limits, checkpoint, scan_policy_sha256,
)
plan = built['plan']
selection_metrics = dict(built['selection_metrics'])
failure_metrics = empty_failure_metrics()
routed = set()
detectors = set()
selected_unique = {
item['digest']: item['max_attempts']
for item in plan['descriptors'] if item['selected']
}
max_checkpoints = max(1, sum(selected_unique.values()))
checkpoint_count = 0
while True:
leased = lease_private_adaptive_checkpoint(plan)
if leased is None:
break
checkpoint_count += 1
if checkpoint_count > max_checkpoints or time.monotonic() >= deadline:
raise RuntimeError('Docker shadow adaptive checkpoint budget was exhausted')
work = {
'plan': leased['plan'],
'plan_sha256': leased['plan_sha256'],
'bearer_auth': bearer_auth,
'min_free_bytes': max(0, int(min_free_bytes or 0)),
'deadline': deadline,
}
result = scanner.scan_docker_layer_plan(
normalized_target, work,
timeout_sec=max(1, math.ceil(deadline - time.monotonic())),
detectors=scan_kwargs.get('detectors'),
exclude_detectors=scan_kwargs.get('exclude_detectors'),
no_verification=bool(scan_kwargs.get('no_verification', False)),
trufflehog_config=scan_kwargs.get('trufflehog_config'),
log_target=False,
)
try:
execution = result.get('docker_layer_execution')
next_plan = apply_private_adaptive_execution(
leased['plan'], result.get('docker_layer_execution'),
)
records = {
record['digest']: record for record in execution['blobs']
}
newly_terminal = {
item['digest'] for item in next_plan['descriptors']
if item['coverage_state'] == 'terminal_failed'
and item['digest'] in records
}
for digest in newly_terminal:
failure_metrics[
_blob_failure_metric(records[digest]['error_code'])
] += 1
plan = next_plan
checkpoint_routed, checkpoint_detectors = private_result_identities(
result, normalized_target,
)
except Exception:
_clear_private_result(result)
raise
routed.update(checkpoint_routed)
detectors.update(checkpoint_detectors)
del checkpoint_routed, checkpoint_detectors
if any(
item['coverage_state'] in ('selected', 'leased', 'shared_pending')
for item in plan['descriptors']
):
raise RuntimeError('Docker shadow adaptive plan did not reach a terminal state')
selection_metrics['adaptive_checkpoints'] += checkpoint_count
terminal_digests = {
item['digest'] for item in plan['descriptors']
if item['coverage_state'] == 'terminal_failed'
}
if sum(failure_metrics.values()) != len(terminal_digests):
raise RuntimeError('Docker shadow terminal failure accounting is inconsistent')
return {
'routed': frozenset(routed),
'detectors': frozenset(detectors),
'selection_metrics': validate_docker_adaptive_shadow_selection_metrics(
selection_metrics,
),
'omitted_descriptor_count': int(built['omitted_descriptor_count']),
'failure_count': len(terminal_digests),
'failure_metrics': failure_metrics,
}
def private_full_side_metrics(db, control, scan_kwargs):
reference_routed, reference_detectors = db.docker_adaptive_shadow_control_identities(
control['target_scan_id'],
)
reference_routed = set(reference_routed)
reference_detectors = set(reference_detectors)
rerun_routed = set()
rerun_detectors = set()
result = None
try:
max_attempts = min(10, max(1, int(scan_kwargs.get('max_attempts', 3) or 3)))
incomplete_reason = 'full_scan_error'
attempts = 0
for attempts in range(1, max_attempts + 1):
result = scanner.scan_docker_image(
control['normalized_target'],
timeout_sec=int(scan_kwargs['timeout_sec']),
detectors=scan_kwargs.get('detectors'),
exclude_detectors=scan_kwargs.get('exclude_detectors'),
no_verification=bool(scan_kwargs.get('no_verification', False)),
trufflehog_config=scan_kwargs.get('trufflehog_config'),
config_dir=scanner.docker_token_manager.get_next_config(),
trufflehog_concurrency=int(scan_kwargs.get('trufflehog_concurrency', 0) or 0),
docker_recovery_limits=scan_kwargs.get('docker_recovery_limits'),
docker_recovery_min_free_bytes=int(scan_kwargs.get('docker_recovery_min_free_bytes', 20 << 30)),
log_target=False,
)
if not isinstance(result, dict):
raise ValueError('Docker shadow full result must be an object')
if not (
result.get('errors') or result.get('skipped')
or result.get('source_failure')
or result.get('warnings') or result.get('degraded')
or ((result.get('scan_meta') or {}).get('docker_layer_scope') or {}).get('coverage_complete') is False
or ((result.get('scan_meta') or {}).get('docker_full_recovery') or {}).get('coverage_complete') is False
):
rerun_routed, rerun_detectors = map(
set, private_result_identities(result, control['normalized_target']),
)
result = None
return {
'full_routed_count': len(reference_routed),
'full_detector_count': len(reference_detectors),
'failure_count': 0,
'safety_regression_count': int(
rerun_routed != reference_routed
or rerun_detectors != reference_detectors
),
**empty_failure_metrics(),
}
incomplete_reason = _full_scan_incomplete_reason(result)
retryable = bool(result.get('retryable', True))
_clear_private_result(result)
result = None
if not retryable:
break
print(
'Docker adaptive shadow full side incomplete: '
f'reason_code={incomplete_reason} attempts={attempts}'
)
metrics = empty_failure_metrics()
metrics['diagnostic_full_incomplete'] = 1
metrics.update({
'full_routed_count': len(reference_routed),
'full_detector_count': len(reference_detectors),
'failure_count': 1,
'safety_regression_count': 0,
})
return metrics
finally:
_clear_private_result(result)
reference_routed.clear()
reference_detectors.clear()
rerun_routed.clear()
rerun_detectors.clear()
def private_adaptive_side_metrics(
db, control, *, limits, checkpoint, scan_policy_sha256, scan_kwargs,
platform_os='linux', platform_arch='amd64', min_free_bytes=0,
):
reference_routed, reference_detectors = db.docker_adaptive_shadow_control_identities(
control['target_scan_id'],
)
reference_routed = set(reference_routed)
reference_detectors = set(reference_detectors)
adaptive_routed = set()
adaptive_detectors = set()
outcome = None
try:
max_attempts = min(10, max(1, int(scan_kwargs.get('max_attempts', 3) or 3)))
incomplete_reason = 'adaptive_scan_error'
attempts = 0
for attempts in range(1, max_attempts + 1):
try:
outcome = execute_private_adaptive_scan(
control['normalized_target'], limits=limits, checkpoint=checkpoint,
scan_policy_sha256=scan_policy_sha256,
timeout_sec=int(scan_kwargs['timeout_sec']),
platform_os=platform_os, platform_arch=platform_arch,
min_free_bytes=min_free_bytes, scan_kwargs=scan_kwargs,
)
adaptive_routed = set(outcome.pop('routed'))
adaptive_detectors = set(outcome.pop('detectors'))
metrics = {
'adaptive_routed_count': len(adaptive_routed),
'routed_intersection_count': len(reference_routed & adaptive_routed),
'adaptive_detector_count': len(adaptive_detectors),
'detector_intersection_count': len(reference_detectors & adaptive_detectors),
'omitted_descriptor_count': int(outcome['omitted_descriptor_count']),
'failure_count': int(outcome['failure_count']),
}
metrics.update(outcome['selection_metrics'])
metrics.update(outcome['failure_metrics'])
return metrics
except DockerShadowPrivacyError:
raise
except scanner.ScanSlotFatalError:
raise
except MemoryError:
raise
except Exception as exc:
incomplete_reason = _adaptive_scan_incomplete_reason(exc)
retryable = bool(getattr(exc, 'retryable', not isinstance(exc, ValueError)))
if not retryable:
break
finally:
if isinstance(outcome, dict):
outcome.clear()
outcome = None
adaptive_routed.clear()
adaptive_detectors.clear()
print(
'Docker adaptive shadow adaptive side incomplete: '
f'reason_code={incomplete_reason} attempts={attempts}'
)
metrics = empty_selection_metrics()
metrics.update(empty_failure_metrics())
metrics['diagnostic_adaptive_incomplete'] = 1
metrics.update({
'adaptive_routed_count': 0,
'routed_intersection_count': 0,
'adaptive_detector_count': 0,
'detector_intersection_count': 0,
'omitted_descriptor_count': 0,
'failure_count': 1,
})
return metrics
finally:
if isinstance(outcome, dict):
outcome.clear()
reference_routed.clear()
reference_detectors.clear()
adaptive_routed.clear()
adaptive_detectors.clear()
def _shadow_config(config):
supervisor = config.get('supervisor') if isinstance(config, dict) else None
supervisor = supervisor if isinstance(supervisor, dict) else {}
value = supervisor.get('docker_shadow')
if not isinstance(value, dict):
raise ValueError('Docker shadow supervisor configuration is required')
allowed = {'enabled', 'cohort_size', 'lease_seconds'}
if set(value) - allowed:
raise ValueError('Docker shadow supervisor configuration has unsupported fields')
if not console_runner.bool_config(value.get('enabled'), False):
raise ValueError('Docker shadow operator command is disabled')
cohort_size = int(value.get('cohort_size', DOCKER_ADAPTIVE_GATE_MIN_CONTROLS))
lease_seconds = int(value.get('lease_seconds', 3600))
if not DOCKER_ADAPTIVE_GATE_MIN_CONTROLS <= cohort_size <= DOCKER_ADAPTIVE_GATE_MAX_CONTROLS:
raise ValueError('Docker shadow cohort size must be between 50 and 100')
if not 60 <= lease_seconds <= 86400:
raise ValueError('Docker shadow lease duration is outside the supported range')
return {'cohort_size': cohort_size, 'lease_seconds': lease_seconds}
def _scan_kwargs(source_args):
return {
'timeout_sec': max(1, int(source_args.timeout)),
'detectors': source_args.detectors,
'exclude_detectors': source_args.exclude_detectors,
'no_verification': bool(source_args.no_verification),
'trufflehog_config': source_args.trufflehog_config,
'trufflehog_concurrency': max(0, int(source_args.trufflehog_concurrency or 0)),
'max_attempts': max(
1, int(getattr(source_args, 'target_retry_max_attempts', 3) or 3),
),
}
def run_shadow(config_path):
require_active_supervisor_child(
config_path, child_kind='docker-shadow', require_dsn=True,
)
config = console_runner.load_config(config_path)
settings = _shadow_config(config)
source_config = (config.get('sources') or {}).get('dockerhub')
if not isinstance(source_config, dict):
raise ValueError('Docker shadow requires the DockerHub source configuration')
global_config = config.get('global') or {}
if not isinstance(global_config, dict):
raise ValueError('Docker shadow global configuration must be a mapping')
console_runner.apply_global_config(global_config)
scanner.initialize_scanner_runtime(preflight_complete=True)
secrets_config = console_runner.load_secrets(config, config_path)
state = {
'version': 1,
'sources': {'dockerhub': console_runner.default_source_state()},
}
console_runner.configure_source_auth(
'dockerhub', source_config, state=state, secrets=secrets_config,
)
source_args = console_runner.build_args_from_source_config(
'dockerhub', source_config, global_config, '', auth_entry=None,
)
scanner.scan_config.drop_detectors = scanner.csv_items(source_args.drop_detectors)
scanner.scan_config.trufflehog_job_memory_limit_bytes = int(
source_args.trufflehog_job_memory_limit_bytes
)
scan_kwargs = _scan_kwargs(source_args)
limits = console_runner.docker_layer_limits(source_args)
scan_kwargs['docker_recovery_limits'] = limits
scan_kwargs['docker_recovery_min_free_bytes'] = int(getattr(source_args, 'docker_layer_min_free_bytes', 20 << 30))
checkpoint = console_runner.docker_adaptive_checkpoint(source_args)
scan_policy_sha256 = console_runner.docker_layer_scan_policy_sha256(
source_args, scan_kwargs,
)
execution_policy_sha256 = docker_layer_execution_policy_sha256(
scan_policy_sha256, limits,
)
selection_policy_sha256 = docker_layer_selection_policy_sha256(limits)
selection_salt = hashlib.sha256(
('docker-shadow-controls-v1:' + scan_policy_sha256 + ':'
+ execution_policy_sha256 + ':' + selection_policy_sha256).encode('ascii')
).hexdigest()
db_url = str(os.getenv('TRUF_MANAGED_POSTGRES_DSN') or '')
if not db_url:
raise RuntimeError('Docker shadow canonical PostgreSQL authority is unavailable')
db = ScannerDB(db_url=db_url, initialize=False)
controls = []
report = None
owner = 'docker-shadow:{}:{}'.format(
os.getpid(), hashlib.sha256(socket.gethostname().encode('utf-8')).hexdigest()[:16],
)
try:
if not db.enabled:
raise RuntimeError('Docker shadow PostgreSQL connection is unavailable')
db.set_application_name('truf-docker-adaptive-shadow')
controls = db.docker_adaptive_shadow_controls(
scan_policy_sha256, settings['cohort_size'], selection_salt,
)
report = db.start_docker_adaptive_shadow_report(
scan_policy_sha256, execution_policy_sha256, selection_policy_sha256,
settings['cohort_size'], owner, lease_seconds=settings['lease_seconds'],
)
aggregate = {
'completed_pairs': 0,
'full_routed_count': 0,
'adaptive_routed_count': 0,
'routed_intersection_count': 0,
'full_detector_count': 0,
'adaptive_detector_count': 0,
'detector_intersection_count': 0,
'full_slot_ms': 0,
'adaptive_slot_ms': 0,
'omitted_descriptor_count': 0,
'failure_count': 0,
'privacy_violation_count': 0,
'safety_regression_count': 0,
}
selection_metrics = empty_selection_metrics()
failure_metrics = empty_failure_metrics()
def durable_checkpoint():
db.checkpoint_docker_adaptive_shadow_report(
report['report_token'], owner, report['lease_token'],
lease_seconds=settings['lease_seconds'],
)
for index, control in enumerate(controls):
sides = ('full', 'adaptive') if index % 2 == 0 else ('adaptive', 'full')
for side in sides:
if side == 'full':
operation = lambda control=control: private_full_side_metrics(
db, control, scan_kwargs,
)
else:
operation = lambda control=control: private_adaptive_side_metrics(
db, control, limits=limits, checkpoint=checkpoint,
scan_policy_sha256=scan_policy_sha256,
scan_kwargs=scan_kwargs,
platform_os=source_args.docker_platform_os,
platform_arch=source_args.docker_platform_arch,
min_free_bytes=source_args.docker_layer_min_free_bytes,
)
side_metrics, elapsed_ms = run_timed_private_side(
side, operation, durable_checkpoint,
timeout_sec=scan_kwargs['timeout_sec'],
)
aggregate[f'{side}_slot_ms'] += elapsed_ms
for name, value in side_metrics.items():
if name in selection_metrics:
selection_metrics[name] += value
elif name in failure_metrics:
failure_metrics[name] += value
else:
aggregate[name] += value
side_metrics.clear()
aggregate['completed_pairs'] += 1
control.clear()
completed = db.finish_docker_adaptive_shadow_report(
report['report_token'], owner, report['lease_token'],
selection_metrics=selection_metrics, **aggregate,
)
print(
'Docker adaptive shadow report: '
f'id={completed["report_id"]} passed={str(completed["passed"]).lower()} '
f'controls={completed["completed_pairs"]} '
f'routed_recall_ppm={completed["routed_recall_ppm"]} '
f'slot_ratio_ppm={completed["slot_ratio_ppm"]} '
'failure_categories=' + ','.join(
f'{name.removeprefix("diagnostic_")}:{failure_metrics[name]}'
for name in SHADOW_FAILURE_METRIC_KEYS
if failure_metrics[name]
)
)
return 0 if completed['passed'] else 2
except BaseException as exc:
if report is not None:
try:
db.fail_docker_adaptive_shadow_report(
report['report_token'], owner, report['lease_token'],
privacy_violation_count=int(isinstance(exc, DockerShadowPrivacyError)),
safety_regression_count=0,
)
except Exception:
pass
print(
'Docker adaptive shadow report failed safely: '
f'reason_code={shadow_failure_reason_code(exc)}'
)
return 1
finally:
for control in controls:
if isinstance(control, dict):
control.clear()
controls.clear()
db.close()
scanner.docker_token_manager.cleanup()
def parse_args(argv=None):
parser = argparse.ArgumentParser(description='Run one private Docker adaptive shadow report')
parser.add_argument('--config', required=True)
return parser.parse_args(argv)
def main(argv=None):
args = parse_args(argv)
return run_shadow(args.config)
if __name__ == '__main__':
raise SystemExit(main())
File diff suppressed because it is too large Load Diff
+131
View File
@@ -0,0 +1,131 @@
import os
import socket
import stat
import struct
import time
from host_agent_protocol import (
CLIENT_CONNECT_TIMEOUT_SECONDS,
EXCHANGE_TIMEOUT_SECONDS,
HOST_AGENT_SOCKET_PATH,
MAX_RESPONSE_PAYLOAD_BYTES,
HostAgentAction,
HostAgentProtocolError,
HostAgentRequest,
HostAgentStatus,
decode_response_frame,
encode_request_frame,
receive_frame,
require_eof,
send_frame,
)
class HostAgentClientError(RuntimeError):
def __init__(self, category):
self.category = category
super().__init__('host operations agent request failed')
class HostAgentUnavailableError(HostAgentClientError):
pass
class HostAgentRejectedError(HostAgentClientError):
pass
def _fixed_socket_is_safe():
try:
details = os.lstat(HOST_AGENT_SOCKET_PATH)
except OSError:
return False
return stat.S_ISSOCK(details.st_mode) and details.st_uid == 0
def _peer_credentials(sock):
if not hasattr(socket, 'SO_PEERCRED'):
raise HostAgentUnavailableError('peer_credentials_unavailable')
try:
raw = sock.getsockopt(
socket.SOL_SOCKET, socket.SO_PEERCRED, struct.calcsize('3i'),
)
pid, uid, gid = struct.unpack('3i', raw)
except (OSError, struct.error) as exc:
raise HostAgentUnavailableError('peer_credentials_unavailable') from exc
# A peer outside the client's PID namespace is reported as PID 0 even
# though its UID/GID remain authoritative through SO_PEERCRED.
if pid < 0:
raise HostAgentUnavailableError('peer_identity_invalid')
return pid, uid, gid
class HostAgentClient:
def __init__(self):
pass
@staticmethod
def is_available():
return _fixed_socket_is_safe()
def dispatch(
self, *, operation_id, action, active_config_sha256,
active_secrets_sha256, candidate_config_sha256,
candidate_secrets_sha256,
):
try:
request = HostAgentRequest(
operation_id=operation_id,
action=HostAgentAction(action),
active_config_sha256=active_config_sha256,
active_secrets_sha256=active_secrets_sha256,
candidate_config_sha256=candidate_config_sha256,
candidate_secrets_sha256=candidate_secrets_sha256,
)
frame = encode_request_frame(request)
except (HostAgentProtocolError, TypeError, ValueError) as exc:
raise HostAgentRejectedError('request_invalid') from exc
if not _fixed_socket_is_safe():
raise HostAgentUnavailableError('socket_unavailable')
deadline = time.monotonic() + EXCHANGE_TIMEOUT_SECONDS
connection = None
response = None
try:
family = getattr(socket, 'AF_UNIX', None)
if family is None:
raise HostAgentUnavailableError('unix_socket_unavailable')
connection = socket.socket(family, socket.SOCK_STREAM)
connection.settimeout(min(
CLIENT_CONNECT_TIMEOUT_SECONDS,
max(0.001, deadline - time.monotonic()),
))
connection.connect(HOST_AGENT_SOCKET_PATH)
_pid, peer_uid, _gid = _peer_credentials(connection)
if peer_uid != 0:
raise HostAgentUnavailableError('server_identity_invalid')
send_frame(connection, frame, deadline=deadline)
connection.shutdown(socket.SHUT_WR)
response = decode_response_frame(receive_frame(
connection, maximum=MAX_RESPONSE_PAYLOAD_BYTES, deadline=deadline,
))
require_eof(connection, deadline=deadline)
except HostAgentClientError:
raise
except HostAgentProtocolError as exc:
raise HostAgentUnavailableError('response_invalid') from exc
except (OSError, TimeoutError) as exc:
raise HostAgentUnavailableError('transport_unavailable') from exc
finally:
if connection is not None:
try:
connection.close()
except OSError:
pass
connection = frame = request = family = None
if response.operation_id != operation_id:
raise HostAgentUnavailableError('response_identity_invalid')
if response.status is HostAgentStatus.ACCEPTED:
return response
if response.status is HostAgentStatus.REJECTED:
raise HostAgentRejectedError('request_rejected')
raise HostAgentUnavailableError('request_unavailable')
File diff suppressed because it is too large Load Diff
+315
View File
@@ -0,0 +1,315 @@
import hmac
import json
import re
import struct
import time
import uuid
from dataclasses import dataclass
from enum import Enum
HOST_AGENT_SOCKET_PATH = '/run/truf/host-agent.sock'
HOST_AGENT_RUNTIME_UID = 10001
MAX_REQUEST_PAYLOAD_BYTES = 1024
MAX_RESPONSE_PAYLOAD_BYTES = 256
CLIENT_CONNECT_TIMEOUT_SECONDS = 1.0
SERVER_READ_TIMEOUT_SECONDS = 2.0
EXCHANGE_TIMEOUT_SECONDS = 5.0
_FRAME_HEADER_BYTES = 4
_SHA256_RE = re.compile(r'^[0-9a-f]{64}$')
_REQUEST_FIELDS = frozenset((
'operation_id', 'action',
'active_config_sha256', 'active_secrets_sha256',
'candidate_config_sha256', 'candidate_secrets_sha256',
))
_RESPONSE_FIELDS = frozenset(('operation_id', 'status'))
class HostAgentProtocolError(ValueError):
def __init__(self, category):
self.category = category
super().__init__('host agent protocol message is invalid')
class HostAgentAction(str, Enum):
APPLY_CONFIG = 'apply-config'
APPLY_SECRETS = 'apply-secrets'
APPLY_BOTH = 'apply-both'
RESTART = 'restart'
class HostAgentStatus(str, Enum):
ACCEPTED = 'accepted'
UNAVAILABLE = 'unavailable'
REJECTED = 'rejected'
INVALID = 'invalid'
@dataclass(frozen=True, slots=True)
class HostAgentRequest:
operation_id: str
action: HostAgentAction
active_config_sha256: str
active_secrets_sha256: str
candidate_config_sha256: str | None
candidate_secrets_sha256: str | None
@dataclass(frozen=True, slots=True)
class HostAgentResponse:
operation_id: str | None
status: HostAgentStatus
def _canonical_uuid(value):
if not isinstance(value, str) or not value:
raise HostAgentProtocolError('operation_id')
try:
parsed = uuid.UUID(value)
except (ValueError, AttributeError) as exc:
raise HostAgentProtocolError('operation_id') from exc
if parsed.int == 0 or str(parsed) != value:
raise HostAgentProtocolError('operation_id')
return value
def _sha256(value, field, *, optional=False):
if optional and value is None:
return None
if not isinstance(value, str) or _SHA256_RE.fullmatch(value) is None:
raise HostAgentProtocolError(field)
return value
def _canonical_json(value):
try:
return json.dumps(
value, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
allow_nan=False,
).encode('ascii')
except (TypeError, ValueError, UnicodeError) as exc:
raise HostAgentProtocolError('json') from exc
def _strict_json(payload, *, maximum):
if type(payload) is not bytes or not 1 <= len(payload) <= maximum:
raise HostAgentProtocolError('bounds')
def reject_duplicate(pairs):
result = {}
for key, value in pairs:
if key in result:
raise HostAgentProtocolError('duplicate_field')
result[key] = value
return result
try:
text = payload.decode('utf-8', errors='strict')
value = json.loads(
text, object_pairs_hook=reject_duplicate,
parse_constant=lambda _value: (_ for _ in ()).throw(
HostAgentProtocolError('constant')
),
)
except HostAgentProtocolError:
raise
except (UnicodeDecodeError, json.JSONDecodeError, TypeError, ValueError) as exc:
raise HostAgentProtocolError('json') from exc
finally:
text = None
if not isinstance(value, dict):
raise HostAgentProtocolError('shape')
if not hmac.compare_digest(_canonical_json(value), payload):
raise HostAgentProtocolError('canonical')
return value
def _normalize_request(value):
if not isinstance(value, dict) or set(value) != _REQUEST_FIELDS:
raise HostAgentProtocolError('shape')
try:
action = HostAgentAction(value.get('action'))
except (TypeError, ValueError) as exc:
raise HostAgentProtocolError('action') from exc
active_config = _sha256(value.get('active_config_sha256'), 'active_config_sha256')
active_secrets = _sha256(value.get('active_secrets_sha256'), 'active_secrets_sha256')
candidate_config = _sha256(
value.get('candidate_config_sha256'), 'candidate_config_sha256', optional=True,
)
candidate_secrets = _sha256(
value.get('candidate_secrets_sha256'), 'candidate_secrets_sha256', optional=True,
)
required = {
HostAgentAction.APPLY_CONFIG: (True, False),
HostAgentAction.APPLY_SECRETS: (False, True),
HostAgentAction.APPLY_BOTH: (True, True),
HostAgentAction.RESTART: (False, False),
}[action]
if (candidate_config is not None, candidate_secrets is not None) != required:
raise HostAgentProtocolError('candidate_identity')
return HostAgentRequest(
operation_id=_canonical_uuid(value.get('operation_id')),
action=action,
active_config_sha256=active_config,
active_secrets_sha256=active_secrets,
candidate_config_sha256=candidate_config,
candidate_secrets_sha256=candidate_secrets,
)
def _request_value(request):
if not isinstance(request, HostAgentRequest):
raise HostAgentProtocolError('request_type')
return {
'operation_id': request.operation_id,
'action': request.action.value if isinstance(request.action, HostAgentAction) else request.action,
'active_config_sha256': request.active_config_sha256,
'active_secrets_sha256': request.active_secrets_sha256,
'candidate_config_sha256': request.candidate_config_sha256,
'candidate_secrets_sha256': request.candidate_secrets_sha256,
}
def encode_request_payload(request):
normalized = _normalize_request(_request_value(request))
payload = _canonical_json(_request_value(normalized))
if len(payload) > MAX_REQUEST_PAYLOAD_BYTES:
raise HostAgentProtocolError('bounds')
return payload
def decode_request_payload(payload):
return _normalize_request(_strict_json(payload, maximum=MAX_REQUEST_PAYLOAD_BYTES))
def _normalize_response(value):
if not isinstance(value, dict) or set(value) != _RESPONSE_FIELDS:
raise HostAgentProtocolError('shape')
try:
status = HostAgentStatus(value.get('status'))
except (TypeError, ValueError) as exc:
raise HostAgentProtocolError('status') from exc
operation_id = value.get('operation_id')
if status is HostAgentStatus.INVALID:
if operation_id is not None:
raise HostAgentProtocolError('operation_id')
else:
operation_id = _canonical_uuid(operation_id)
return HostAgentResponse(operation_id=operation_id, status=status)
def _response_value(response):
if not isinstance(response, HostAgentResponse):
raise HostAgentProtocolError('response_type')
return {
'operation_id': response.operation_id,
'status': response.status.value if isinstance(response.status, HostAgentStatus) else response.status,
}
def encode_response_payload(response):
normalized = _normalize_response(_response_value(response))
payload = _canonical_json(_response_value(normalized))
if len(payload) > MAX_RESPONSE_PAYLOAD_BYTES:
raise HostAgentProtocolError('bounds')
return payload
def decode_response_payload(payload):
return _normalize_response(_strict_json(payload, maximum=MAX_RESPONSE_PAYLOAD_BYTES))
def _encode_frame(payload, maximum):
if type(payload) is not bytes or not 1 <= len(payload) <= maximum:
raise HostAgentProtocolError('bounds')
return struct.pack('!I', len(payload)) + payload
def _decode_frame(frame, maximum):
if type(frame) is not bytes or len(frame) < _FRAME_HEADER_BYTES:
raise HostAgentProtocolError('frame')
length = struct.unpack('!I', frame[:_FRAME_HEADER_BYTES])[0]
if not 1 <= length <= maximum or len(frame) != _FRAME_HEADER_BYTES + length:
raise HostAgentProtocolError('frame')
return frame[_FRAME_HEADER_BYTES:]
def encode_request_frame(request):
return _encode_frame(encode_request_payload(request), MAX_REQUEST_PAYLOAD_BYTES)
def decode_request_frame(frame):
return decode_request_payload(_decode_frame(frame, MAX_REQUEST_PAYLOAD_BYTES))
def encode_response_frame(response):
return _encode_frame(encode_response_payload(response), MAX_RESPONSE_PAYLOAD_BYTES)
def decode_response_frame(frame):
return decode_response_payload(_decode_frame(frame, MAX_RESPONSE_PAYLOAD_BYTES))
def _remaining(deadline):
remaining = deadline - time.monotonic()
if remaining <= 0:
raise HostAgentProtocolError('timeout')
return remaining
def receive_frame(sock, *, maximum, deadline):
header = _receive_exact(sock, _FRAME_HEADER_BYTES, deadline)
length = struct.unpack('!I', header)[0]
if not 1 <= length <= maximum:
raise HostAgentProtocolError('bounds')
return header + _receive_exact(sock, length, deadline)
def _receive_exact(sock, length, deadline):
chunks = bytearray()
try:
while len(chunks) < length:
sock.settimeout(_remaining(deadline))
chunk = sock.recv(length - len(chunks))
if not chunk:
raise HostAgentProtocolError('truncated')
chunks.extend(chunk)
return bytes(chunks)
except HostAgentProtocolError:
raise
except (OSError, TimeoutError) as exc:
raise HostAgentProtocolError('transport') from exc
finally:
chunks.clear()
chunk = None
def require_eof(sock, *, deadline):
try:
sock.settimeout(_remaining(deadline))
if sock.recv(1):
raise HostAgentProtocolError('trailing_data')
except HostAgentProtocolError:
raise
except (OSError, TimeoutError) as exc:
raise HostAgentProtocolError('transport') from exc
def send_frame(sock, frame, *, deadline):
if type(frame) is not bytes:
raise HostAgentProtocolError('frame')
view = memoryview(frame)
try:
while view:
sock.settimeout(_remaining(deadline))
sent = sock.send(view)
if sent <= 0:
raise HostAgentProtocolError('transport')
view = view[sent:]
except HostAgentProtocolError:
raise
except (OSError, TimeoutError) as exc:
raise HostAgentProtocolError('transport') from exc
finally:
view.release()
+94
View File
@@ -0,0 +1,94 @@
"""Bounded runtime reconciliation of fixed host-agent result evidence."""
import json
import os
from pathlib import Path
import stat
from host_agent_state import MAX_STATE_BYTES
from runtime_security import reject_reparse_components
HOST_RESULT_DIRECTORY = Path('/data/host-agent-results')
HOST_ROOT_UID = 0
HOST_RUNTIME_GID = 10001
class HostResultError(RuntimeError):
pass
def fixed_result_directory_is_safe():
try:
reject_reparse_components(HOST_RESULT_DIRECTORY)
details = os.stat(HOST_RESULT_DIRECTORY, follow_symlinks=False)
return (
stat.S_ISDIR(details.st_mode)
and details.st_uid == HOST_ROOT_UID
and details.st_gid == HOST_RUNTIME_GID
and stat.S_IMODE(details.st_mode) == 0o750
)
except Exception:
return False
def _read_result(operation_id):
path = HOST_RESULT_DIRECTORY / f'{operation_id}.json'
descriptor = None
try:
flags = os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0) | getattr(os, 'O_NOFOLLOW', 0)
descriptor = os.open(path, flags)
before = os.fstat(descriptor)
if (
not stat.S_ISREG(before.st_mode)
or before.st_nlink != 1
or before.st_uid != HOST_ROOT_UID
or before.st_gid != HOST_RUNTIME_GID
or stat.S_IMODE(before.st_mode) != 0o640
):
raise HostResultError('host result metadata is invalid')
with os.fdopen(descriptor, 'rb') as handle:
descriptor = None
payload = handle.read(MAX_STATE_BYTES + 1)
after = os.fstat(handle.fileno())
current = os.stat(path, follow_symlinks=False)
identity = lambda item: (
item.st_dev, item.st_ino, item.st_size,
getattr(item, 'st_mtime_ns', None), getattr(item, 'st_ctime_ns', None),
)
if (
len(payload) > MAX_STATE_BYTES
or identity(before) != identity(after)
or identity(after) != identity(current)
):
raise HostResultError('host result changed during read')
value = json.loads(payload.decode('ascii'))
canonical = json.dumps(
value, sort_keys=True, separators=(',', ':'), ensure_ascii=True,
allow_nan=False,
).encode('ascii')
if canonical != payload:
raise HostResultError('host result is not canonical')
return payload
except FileNotFoundError:
return None
except HostResultError:
raise
except Exception:
raise HostResultError('host result is invalid') from None
finally:
if descriptor is not None:
os.close(descriptor)
def reconcile_pending_host_results(database, *, limit=32):
if not fixed_result_directory_is_safe():
return 0
reconciled = 0
for operation in database.pending_runtime_agent_operations(limit=limit):
envelope = _read_result(operation['operation_id'])
if envelope is None:
continue
if database.reconcile_runtime_operation_result(envelope) is not None:
reconciled += 1
return reconciled
+214
View File
@@ -0,0 +1,214 @@
"""Fixed production authority and asynchronous host-operation dispatch."""
import hmac
from pathlib import Path, PurePosixPath
import threading
from host_agent_apply import HostApplyError, HostApplySession
from host_agent_lifecycle import execute_fixed_operation
from host_agent_protocol import HostAgentStatus, encode_request_payload
from host_agent_state import HostOperationState
from runtime_document import (
MAX_CONFIG_DOCUMENT_BYTES,
_resolve_package_manifest_path,
load_yaml_document,
)
from runtime_security import read_stable_root_file
from scanner_db import ScannerDB
from worker_package import (
MAX_WORKER_PACKAGE_MANIFEST_BYTES,
load_worker_package_manifest_bytes,
)
HOST_RUNTIME_ACTIVE_ROOT = Path('/etc/truf/runtime')
HOST_WORKER_PACKAGE_ROOT = Path('/etc/truf/worker-packages')
class HostRuntimeError(RuntimeError):
def __init__(self, category):
self.category = str(category)
super().__init__('host operation runtime failed')
def _stable_root_file(path, maximum):
try:
return read_stable_root_file(path, maximum, HOST_WORKER_PACKAGE_ROOT)
except Exception:
raise HostRuntimeError('package_evidence') from None
def _host_manifest_path(resolved):
value = PurePosixPath(resolved)
try:
relative = value.relative_to(PurePosixPath('/data/worker-packages'))
except ValueError:
raise HostRuntimeError('package_evidence') from None
if not relative.parts or any(part in ('', '.', '..') for part in relative.parts):
raise HostRuntimeError('package_evidence')
return HOST_WORKER_PACKAGE_ROOT.joinpath(*relative.parts)
def load_fixed_package_capabilities(config_payload):
config = load_yaml_document(
config_payload, max_bytes=MAX_CONFIG_DOCUMENT_BYTES,
)
try:
profiles = config['supervisor']['worker_api']['compatibility_profiles']
except (KeyError, TypeError):
raise HostRuntimeError('package_evidence') from None
if type(profiles) is not dict:
raise HostRuntimeError('package_evidence')
evidence = {}
try:
for profile_name, profile in profiles.items():
if type(profile_name) is not str or type(profile) is not dict:
raise HostRuntimeError('package_evidence')
reference = profile.get('package_manifest')
resolved = _resolve_package_manifest_path(config, reference)
if resolved is None:
raise HostRuntimeError('package_evidence')
payload = _stable_root_file(
_host_manifest_path(resolved), MAX_WORKER_PACKAGE_MANIFEST_BYTES,
)
manifest = load_worker_package_manifest_bytes(payload)
evidence[profile_name] = {
'package_manifest': reference,
'capabilities': manifest['capabilities'],
}
return evidence
except HostRuntimeError:
raise
except Exception:
raise HostRuntimeError('package_evidence') from None
finally:
config = profiles = profile_name = profile = reference = None
resolved = payload = manifest = None
class FixedHostOperationDispatcher:
def __init__(self):
self._guard = threading.Lock()
self._active_request = None
self._worker = None
self._closing = False
def _execute(self, session, database, state):
try:
try:
execute_fixed_operation(session, state=state)
except BaseException:
# The durable executor owns safety/result handling. Do not let
# thread tracebacks disclose host details at this outer boundary.
pass
finally:
try:
session.close()
except BaseException:
pass
try:
database.close()
except BaseException:
pass
finally:
with self._guard:
self._active_request = None
self._worker = None
@staticmethod
def _record_validation_failure(state, request):
try:
phase = state.initialize('original')
if phase.get('phase') == 'prepared':
if phase.get('publication_state') != 'original':
return False
phase = state.advance(
'prepared', 'failed', 'original',
forward_category='validation_failed',
safe_detail='validation_failed',
)
if (
phase.get('phase') != 'failed'
or phase.get('publication_state') != 'original'
or phase.get('forward_category') != 'validation_failed'
or phase.get('safe_detail') != 'validation_failed'
):
return False
state.publish_result(
'failed',
safe_category='validation_failed',
safe_detail='validation_failed',
resulting_identity={
'active_config_sha256': request.active_config_sha256,
'active_secrets_sha256': request.active_secrets_sha256,
},
)
return True
except Exception:
return False
def handle(self, request):
encoded = encode_request_payload(request)
with self._guard:
if self._closing:
return HostAgentStatus.UNAVAILABLE
if self._active_request is not None:
return (
HostAgentStatus.ACCEPTED
if hmac.compare_digest(encoded, self._active_request)
else HostAgentStatus.REJECTED
)
state = HostOperationState(request)
if state.terminal_result() is not None:
return HostAgentStatus.ACCEPTED
database = None
session = None
try:
database = ScannerDB.host_agent_authority()
session = HostApplySession(
request, database,
package_capability_provider=load_fixed_package_capabilities,
)
session.__enter__()
state.initialize(session.publication_state)
worker = threading.Thread(
target=self._execute,
args=(session, database, state),
name='truf-host-operation',
daemon=False,
)
self._active_request = encoded
self._worker = worker
worker.start()
except Exception as error:
accepted = (
isinstance(error, HostApplyError)
and error.category == 'validation'
and session is not None
and session.claim is not None
and self._record_validation_failure(state, request)
)
self._active_request = None
self._worker = None
if session is not None:
try:
session.close()
except BaseException:
pass
if database is not None:
try:
database.close()
except BaseException:
pass
return (
HostAgentStatus.ACCEPTED
if accepted else HostAgentStatus.UNAVAILABLE
)
return HostAgentStatus.ACCEPTED
def close(self):
with self._guard:
self._closing = True
worker = self._worker
if worker is not None:
worker.join()
+167
View File
@@ -0,0 +1,167 @@
import os
import socket
import stat
import struct
import time
from host_agent_protocol import (
EXCHANGE_TIMEOUT_SECONDS,
HOST_AGENT_RUNTIME_UID,
HOST_AGENT_SOCKET_PATH,
MAX_REQUEST_PAYLOAD_BYTES,
SERVER_READ_TIMEOUT_SECONDS,
HostAgentProtocolError,
HostAgentResponse,
HostAgentStatus,
decode_request_frame,
encode_response_frame,
receive_frame,
require_eof,
send_frame,
)
SYSTEMD_LISTEN_FD = 3
ACCEPT_POLL_SECONDS = 1.0
class HostAgentServerError(RuntimeError):
def __init__(self, category):
self.category = category
super().__init__('host operations agent server failed')
def _peer_credentials(connection):
if not hasattr(socket, 'SO_PEERCRED'):
raise HostAgentServerError('peer_credentials_unavailable')
try:
raw = connection.getsockopt(
socket.SOL_SOCKET, socket.SO_PEERCRED, struct.calcsize('3i'),
)
pid, uid, gid = struct.unpack('3i', raw)
except (OSError, struct.error) as exc:
raise HostAgentServerError('peer_credentials_unavailable') from exc
return pid, uid, gid
def unavailable_handler(_request):
return HostAgentStatus.UNAVAILABLE
def serve_connection(connection, *, handler=unavailable_handler, accepted_at=None):
if not isinstance(connection, socket.socket):
raise HostAgentServerError('connection_invalid')
started = time.monotonic() if accepted_at is None else accepted_at
try:
peer_pid, peer_uid, _peer_gid = _peer_credentials(connection)
except HostAgentServerError:
return False
if peer_pid <= 0 or peer_uid != HOST_AGENT_RUNTIME_UID:
return False
request = None
response = None
try:
read_deadline = min(
started + SERVER_READ_TIMEOUT_SECONDS,
started + EXCHANGE_TIMEOUT_SECONDS,
)
frame = receive_frame(
connection, maximum=MAX_REQUEST_PAYLOAD_BYTES,
deadline=read_deadline,
)
require_eof(connection, deadline=read_deadline)
request = decode_request_frame(frame)
except HostAgentProtocolError:
response = HostAgentResponse(None, HostAgentStatus.INVALID)
else:
try:
outcome = handler(request)
if isinstance(outcome, HostAgentResponse):
response = outcome
else:
response = HostAgentResponse(
request.operation_id, HostAgentStatus(outcome),
)
if response.operation_id != request.operation_id:
raise HostAgentServerError('handler_identity_invalid')
except BaseException as exc:
if not isinstance(exc, Exception):
raise
response = HostAgentResponse(
request.operation_id, HostAgentStatus.UNAVAILABLE,
)
try:
send_frame(
connection, encode_response_frame(response),
deadline=started + EXCHANGE_TIMEOUT_SECONDS,
)
except HostAgentProtocolError:
return False
finally:
frame = request = response = outcome = None
return True
def _validate_listener(listener):
if not isinstance(listener, socket.socket):
raise HostAgentServerError('listener_invalid')
unix_family = getattr(socket, 'AF_UNIX', None)
if unix_family is None:
raise HostAgentServerError('listener_invalid')
try:
socket_type = listener.getsockopt(socket.SOL_SOCKET, socket.SO_TYPE)
accepting = listener.getsockopt(socket.SOL_SOCKET, socket.SO_ACCEPTCONN)
except OSError as exc:
raise HostAgentServerError('listener_invalid') from exc
if (
listener.family != unix_family
or socket_type != socket.SOCK_STREAM
or accepting != 1
):
raise HostAgentServerError('listener_invalid')
try:
if listener.getsockname() != HOST_AGENT_SOCKET_PATH:
raise HostAgentServerError('listener_path_invalid')
details = os.lstat(HOST_AGENT_SOCKET_PATH)
except HostAgentServerError:
raise
except OSError as exc:
raise HostAgentServerError('listener_unavailable') from exc
if not stat.S_ISSOCK(details.st_mode) or details.st_uid != 0:
raise HostAgentServerError('listener_owner_invalid')
def inherited_systemd_listener():
geteuid = getattr(os, 'geteuid', None)
if geteuid is None or geteuid() != 0:
raise HostAgentServerError('root_required')
if os.environ.get('LISTEN_PID') != str(os.getpid()):
raise HostAgentServerError('socket_activation_invalid')
if os.environ.get('LISTEN_FDS') != '1':
raise HostAgentServerError('socket_activation_invalid')
try:
listener = socket.socket(fileno=SYSTEMD_LISTEN_FD)
_validate_listener(listener)
except BaseException:
try:
listener.close()
except (OSError, UnboundLocalError):
pass
raise
return listener
def serve_forever(listener, *, handler=unavailable_handler, stop_event=None):
_validate_listener(listener)
if stop_event is not None:
listener.settimeout(ACCEPT_POLL_SECONDS)
while stop_event is None or not stop_event.is_set():
try:
connection, _address = listener.accept()
except InterruptedError:
continue
except TimeoutError:
continue
with connection:
serve_connection(connection, handler=handler)
+412
View File
@@ -0,0 +1,412 @@
"""Durable fixed-path evidence for privileged runtime operations."""
import hashlib
import json
import os
from pathlib import Path
import stat
from host_agent_protocol import decode_request_payload, encode_request_payload
from runtime_security import fsync_directory, reject_reparse_components
HOST_ROOT_UID = 0
HOST_ROOT_GID = 0
HOST_RUNTIME_GID = 10001
HOST_STATE_ROOT = Path('/var/lib/truf/host-agent')
HOST_STATE_ROOT_MODE = 0o700
HOST_OPERATION_DIRECTORY = HOST_STATE_ROOT / 'operations'
HOST_OPERATION_DIRECTORY_MODE = 0o700
HOST_RESULT_DIRECTORY = HOST_STATE_ROOT / 'results'
HOST_RESULT_DIRECTORY_MODE = 0o750
HOST_FAILED_HOLD_PATH = HOST_STATE_ROOT / 'failed-hold.json'
MAX_STATE_BYTES = 16 * 1024
_PHASES = {
'prepared', 'forward_started', 'rollback_started',
'succeeded', 'failed', 'rolled_back', 'failed_hold',
}
_PUBLICATION_STATES = {'original', 'partial', 'candidate'}
_TERMINAL_RESULTS = {'succeeded', 'failed', 'rolled_back', 'failed_hold'}
_TRANSITIONS = {
'prepared': {'forward_started', 'rollback_started', 'failed'},
'forward_started': {'rollback_started', 'succeeded'},
'rollback_started': {'rolled_back', 'failed_hold'},
}
class HostStateError(RuntimeError):
def __init__(self, category, *, cancellation=None):
self.category = str(category)
self.cancellation = cancellation
super().__init__('host runtime state failed')
def _canonical(value):
try:
payload = json.dumps(
value, sort_keys=True, separators=(',', ':'), ensure_ascii=True,
allow_nan=False,
).encode('ascii')
except (TypeError, ValueError):
raise HostStateError('evidence') from None
if not payload or len(payload) > MAX_STATE_BYTES:
raise HostStateError('evidence')
return payload
def _require_directory(path, *, gid, mode):
try:
reject_reparse_components(path)
details = os.stat(path, follow_symlinks=False)
if not stat.S_ISDIR(details.st_mode):
raise OSError('not a directory')
if os.name != 'nt' and (
details.st_uid != HOST_ROOT_UID
or details.st_gid != gid
or stat.S_IMODE(details.st_mode) != mode
):
raise OSError('directory metadata')
except Exception:
raise HostStateError('filesystem') from None
def _read_file(path, *, gid, mode):
descriptor = None
try:
reject_reparse_components(Path(path).parent)
flags = os.O_RDONLY | getattr(os, 'O_CLOEXEC', 0)
if hasattr(os, 'O_BINARY'):
flags |= os.O_BINARY
if hasattr(os, 'O_NOFOLLOW'):
flags |= os.O_NOFOLLOW
descriptor = os.open(path, flags)
before = os.fstat(descriptor)
if (
not stat.S_ISREG(before.st_mode) or before.st_nlink != 1
or (
os.name != 'nt'
and (
before.st_uid != HOST_ROOT_UID or before.st_gid != gid
or stat.S_IMODE(before.st_mode) != mode
)
)
):
raise OSError('file metadata')
with os.fdopen(descriptor, 'rb') as handle:
descriptor = None
payload = handle.read(MAX_STATE_BYTES + 1)
after = os.fstat(handle.fileno())
current = os.stat(path, follow_symlinks=False)
identity = lambda item: (
item.st_dev, item.st_ino, item.st_size,
getattr(item, 'st_mtime_ns', None),
None if os.name == 'nt' else getattr(item, 'st_ctime_ns', None),
)
if (
identity(before) != identity(after)
or identity(after) != identity(current)
or len(payload) > MAX_STATE_BYTES
):
raise OSError('file changed')
return payload
except FileNotFoundError:
raise
except Exception:
raise HostStateError('filesystem') from None
finally:
if descriptor is not None:
os.close(descriptor)
def _decode_canonical(payload):
try:
value = json.loads(payload.decode('ascii'))
except (UnicodeDecodeError, json.JSONDecodeError):
raise HostStateError('evidence') from None
if not isinstance(value, dict) or _canonical(value) != payload:
raise HostStateError('evidence')
return value
def _write_stage(path, payload, *, gid, mode):
stage = Path(path).parent / f'.{Path(path).name}.stage'
descriptor = None
created = False
published = False
try:
try:
details = os.stat(stage, follow_symlinks=False)
if (
not stat.S_ISREG(details.st_mode) or details.st_nlink != 1
or (os.name != 'nt' and details.st_uid != HOST_ROOT_UID)
):
raise OSError('unsafe stage')
os.unlink(stage)
fsync_directory(stage.parent)
except FileNotFoundError:
pass
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_CLOEXEC', 0)
if hasattr(os, 'O_BINARY'):
flags |= os.O_BINARY
if hasattr(os, 'O_NOFOLLOW'):
flags |= os.O_NOFOLLOW
descriptor = os.open(stage, flags, mode)
created = True
if os.name != 'nt':
os.fchmod(descriptor, mode)
details = os.fstat(descriptor)
if details.st_uid != HOST_ROOT_UID or details.st_gid != gid:
os.fchown(descriptor, HOST_ROOT_UID, gid)
view = memoryview(payload)
written = 0
while written < len(view):
count = os.write(descriptor, view[written:])
if count <= 0:
raise OSError('short write')
written += count
os.fsync(descriptor)
os.close(descriptor)
descriptor = None
os.replace(stage, path)
created = False
published = True
fsync_directory(Path(path).parent)
stored = _read_file(path, gid=gid, mode=mode)
if not hashlib.sha256(stored).digest() == hashlib.sha256(payload).digest():
raise HostStateError('evidence')
except HostStateError:
if published:
raise HostStateError('uncertain') from None
raise
except BaseException as error:
if published:
cancellation = error if not isinstance(error, Exception) else None
raise HostStateError(
'uncertain', cancellation=cancellation,
) from None
if not isinstance(error, Exception):
raise
raise HostStateError('filesystem') from None
finally:
if descriptor is not None:
os.close(descriptor)
if created:
try:
os.unlink(stage)
fsync_directory(stage.parent)
except OSError:
pass
def _publish_exact(path, payload, *, gid, mode):
try:
existing = _read_file(path, gid=gid, mode=mode)
except FileNotFoundError:
_write_stage(path, payload, gid=gid, mode=mode)
return payload
if existing != payload:
raise HostStateError('conflict')
return existing
def failed_hold_operation():
try:
payload = _read_file(
HOST_FAILED_HOLD_PATH, gid=HOST_ROOT_GID, mode=0o600,
)
except FileNotFoundError:
return None
value = _decode_canonical(payload)
if (
set(value) != {
'schema', 'operation_id', 'action', 'forward_category',
'publication_state', 'containment_confirmed',
}
or value.get('schema') != 1
or not isinstance(value.get('operation_id'), str)
or value.get('publication_state') not in _PUBLICATION_STATES
or type(value.get('containment_confirmed')) is not bool
or not isinstance(value.get('forward_category'), str)
):
raise HostStateError('evidence')
return value['operation_id']
class HostOperationState:
def __init__(self, request):
self.request = decode_request_payload(encode_request_payload(request))
self.operation_path = (
HOST_OPERATION_DIRECTORY / f'{self.request.operation_id}.json'
)
self.result_path = HOST_RESULT_DIRECTORY / f'{self.request.operation_id}.json'
def _phase_record(
self, phase, publication_state, *, forward_category=None,
safe_detail=None, containment_confirmed=None,
):
if (
phase not in _PHASES
or publication_state not in _PUBLICATION_STATES
or forward_category is not None
and not isinstance(forward_category, str)
or safe_detail is not None
and not isinstance(safe_detail, str)
or containment_confirmed is not None
and type(containment_confirmed) is not bool
):
raise HostStateError('evidence')
return {
'schema': 1,
'operation_id': self.request.operation_id,
'action': self.request.action.value,
'active_config_sha256': self.request.active_config_sha256,
'active_secrets_sha256': self.request.active_secrets_sha256,
'candidate_config_sha256': self.request.candidate_config_sha256,
'candidate_secrets_sha256': self.request.candidate_secrets_sha256,
'phase': phase,
'publication_state': publication_state,
'forward_category': forward_category,
'safe_detail': safe_detail,
'containment_confirmed': containment_confirmed,
}
def _read_phase(self):
payload = _read_file(self.operation_path, gid=HOST_ROOT_GID, mode=0o600)
value = _decode_canonical(payload)
if set(value) != set(self._phase_record('prepared', 'original')):
raise HostStateError('evidence')
expected = self._phase_record(
value.get('phase'), value.get('publication_state'),
forward_category=value.get('forward_category'),
safe_detail=value.get('safe_detail'),
containment_confirmed=value.get('containment_confirmed'),
)
if value != expected:
raise HostStateError('evidence')
return value
def initialize(self, publication_state='original'):
_require_directory(
HOST_STATE_ROOT, gid=HOST_ROOT_GID, mode=HOST_STATE_ROOT_MODE,
)
_require_directory(
HOST_OPERATION_DIRECTORY,
gid=HOST_ROOT_GID,
mode=HOST_OPERATION_DIRECTORY_MODE,
)
hold = failed_hold_operation()
if hold is not None and hold != self.request.operation_id:
raise HostStateError('failed_hold')
try:
return self._read_phase()
except FileNotFoundError:
value = self._phase_record('prepared', publication_state)
_write_stage(
self.operation_path, _canonical(value),
gid=HOST_ROOT_GID, mode=0o600,
)
return value
def advance(
self, expected_phase, next_phase, publication_state, *,
forward_category=None, safe_detail=None, containment_confirmed=None,
):
current = self._read_phase()
value = self._phase_record(
next_phase, publication_state,
forward_category=forward_category,
safe_detail=safe_detail,
containment_confirmed=containment_confirmed,
)
if current == value:
return value
if (
current['phase'] != expected_phase
or next_phase not in _TRANSITIONS.get(expected_phase, set())
):
raise HostStateError('state')
_write_stage(
self.operation_path, _canonical(value),
gid=HOST_ROOT_GID, mode=0o600,
)
return value
def terminal_result(self):
try:
payload = _read_file(
self.result_path, gid=HOST_RUNTIME_GID, mode=0o640,
)
except FileNotFoundError:
return None
value = _decode_canonical(payload)
if (
set(value) != {
'schema', 'operation_id', 'action', 'result', 'safe_category',
'safe_detail', 'resulting_identity',
}
or value.get('schema') != 1
or value.get('operation_id') != self.request.operation_id
or value.get('action') != self.request.action.value
or value.get('result') not in _TERMINAL_RESULTS
):
raise HostStateError('evidence')
return value
def publish_result(
self, result, *, safe_category, safe_detail, resulting_identity,
):
if result not in _TERMINAL_RESULTS:
raise HostStateError('evidence')
if result == 'succeeded':
if safe_category is not None or safe_detail is not None:
raise HostStateError('evidence')
elif not isinstance(safe_category, str) or not isinstance(safe_detail, str):
raise HostStateError('evidence')
if resulting_identity is not None and (
not isinstance(resulting_identity, dict)
or set(resulting_identity) != {
'active_config_sha256', 'active_secrets_sha256',
}
or any(
not isinstance(value, str) or len(value) != 64
for value in resulting_identity.values()
)
):
raise HostStateError('evidence')
value = {
'schema': 1,
'operation_id': self.request.operation_id,
'action': self.request.action.value,
'result': result,
'safe_category': safe_category,
'safe_detail': safe_detail,
'resulting_identity': resulting_identity,
}
_require_directory(
HOST_RESULT_DIRECTORY,
gid=HOST_RUNTIME_GID,
mode=HOST_RESULT_DIRECTORY_MODE,
)
_publish_exact(
self.result_path, _canonical(value), gid=HOST_RUNTIME_GID, mode=0o640,
)
return value
def publish_failed_hold(
self, *, forward_category, publication_state, containment_confirmed,
):
marker = {
'schema': 1,
'operation_id': self.request.operation_id,
'action': self.request.action.value,
'forward_category': str(forward_category),
'publication_state': publication_state,
'containment_confirmed': bool(containment_confirmed),
}
_publish_exact(
HOST_FAILED_HOLD_PATH, _canonical(marker),
gid=HOST_ROOT_GID, mode=0o600,
)
return marker
+455
View File
@@ -0,0 +1,455 @@
import sys
sys.dont_write_bytecode = True
if not sys.dont_write_bytecode:
raise RuntimeError('janitor could not disable bytecode writes')
import argparse
import hashlib
import json
import os
import stat
import time
from dataclasses import dataclass
from datetime import datetime, timezone
from lifecycle_authority import require_active_supervisor_child
from paths import apply_path_config
from process_identity import exact_process_identity_state
from runtime_security import (
atomic_write_private_json,
canonical_path,
fsync_directory,
is_reparse_point,
private_directory_ready,
private_file_ready,
read_private_json,
reject_reparse_components,
require_private_directory,
)
MARKER_NAME = '.scanner-owner.json'
MARKER_SCHEMA = 2
APPROVED_LAYOUTS = (
('work', '', ('trufflehog-', 'trufflehog-run-', 'trufflehog-probe-', 'hg-run-', 'docker-config-', 'docker-layer-', 'tmp-', 'worker-assignment-')),
('work', 'docker-config', ('docker-config-',)),
('work', 'hg', ('hg-run-',)),
('work', 'tmp', ('trufflehog-', 'trufflehog-run-', 'hg-run-', 'docker-config-', 'docker-layer-', 'tmp-')),
('work', os.path.join('tmp', 'docker-config'), ('docker-config-',)),
('work', 'abandoned', ('worker-assignment-',)),
)
@dataclass
class JanitorBudget:
max_candidates: int = 50
max_entries: int = 10000
max_bytes: int = 1024 * 1024 * 1024
max_seconds: float = 30.0
max_depth: int = 64
max_enumerated: int = 1000
candidates: int = 0
entries: int = 0
bytes: int = 0
started_at: float = 0.0
exhausted: bool = False
enumerated: int = 0
def __post_init__(self):
self.started_at = self.started_at or time.monotonic()
def consume(self, size=0, candidate=False, depth=0, allow_oversized=False):
if candidate:
self.candidates += 1
else:
self.entries += 1
self.bytes += max(0, int(size or 0))
candidate_limit = self.candidates > self.max_candidates
entry_limit = self.entries > self.max_entries
byte_limit = self.bytes > self.max_bytes
depth_limit = depth > self.max_depth
time_limit = time.monotonic() - self.started_at >= self.max_seconds
self.exhausted = bool(
candidate_limit or entry_limit or byte_limit or depth_limit or time_limit
)
# Unlinking one regular file is bounded metadata work regardless of its
# payload size. Directory traversal remains bounded by the other limits.
oversized_progress = bool(
allow_oversized and byte_limit
and not (candidate_limit or entry_limit or depth_limit or time_limit)
)
return not self.exhausted or oversized_progress
def consume_enumerated(self):
self.enumerated += 1
self.exhausted = bool(
self.enumerated > self.max_enumerated
or time.monotonic() - self.started_at >= self.max_seconds
)
return not self.exhausted
def _marker_relative_path(root, path):
relative = os.path.relpath(path, root)
if relative == '.' or relative.startswith('..' + os.sep) or os.path.isabs(relative):
raise ValueError('candidate escapes the approved janitor root')
return relative.replace(os.sep, '/')
def validate_marker(root, path, marker, allowed_executables, minimum_age_sec, now=None):
if marker.get('schema') != MARKER_SCHEMA or marker.get('root_kind') != 'work':
return False, 'unsupported_marker'
try:
if marker.get('relative_path') != _marker_relative_path(root, path):
return False, 'path_mismatch'
created = datetime.fromisoformat(str(marker.get('created_at') or '').replace('Z', '+00:00'))
if created.tzinfo is None:
created = created.replace(tzinfo=timezone.utc)
now_value = now or datetime.now(timezone.utc)
if (now_value - created).total_seconds() < max(0, float(minimum_age_sec)):
return False, 'too_young'
except (TypeError, ValueError):
return False, 'invalid_time_or_path'
allowed = {canonical_path(value) for value in allowed_executables if value}
identities = {}
states = {}
prefixes = ('owner', 'parent')
if any(f'child_{field}' in marker for field in ('pid', 'creation_time', 'executable')):
prefixes += ('child',)
for prefix in prefixes:
if prefix == 'child' and any(not marker.get(f'child_{field}') for field in ('pid', 'creation_time', 'executable')):
return False, 'child_identity_invalid'
identity = {
'pid': marker.get(f'{prefix}_pid'),
'creation_time': marker.get(f'{prefix}_creation_time'),
'executable': marker.get(f'{prefix}_executable'),
}
try:
executable = canonical_path(identity['executable'])
except (OSError, TypeError, ValueError):
return False, f'{prefix}_identity_invalid'
if executable not in allowed:
return False, f'{prefix}_executable_unapproved'
identity['executable'] = executable
identities[prefix] = identity
states[prefix] = exact_process_identity_state(
identity['pid'], identity['creation_time'], identity['executable'],
)
if 'child' in states and states['child'] != 'dead':
return False, 'child_live_or_unknown'
if states['owner'] != 'dead':
return False, 'owner_live_or_unknown'
same_identity = all(
identities['owner'].get(field) == identities['parent'].get(field)
for field in ('pid', 'creation_time', 'executable')
)
if same_identity and states['parent'] != 'dead':
return False, 'parent_owned_live_or_unknown'
return True, 'eligible'
def bounded_remove_tree(path, budget, marker_name=MARKER_NAME):
"""Delete without recursion or reparse traversal; leave the root marker last."""
path = os.path.abspath(path)
reject_reparse_components(path)
if is_reparse_point(path) or not private_directory_ready(path):
raise OSError(f'janitor candidate is not an exact private directory: {path}')
marker_path = os.path.join(path, marker_name)
stack = []
root_iterator = os.scandir(path)
stack.append((path, root_iterator, 0))
try:
while stack:
if time.monotonic() - budget.started_at >= budget.max_seconds:
budget.exhausted = True
return False
directory, iterator, depth = stack[-1]
try:
entry = next(iterator)
except StopIteration:
iterator.close()
stack.pop()
if directory == path:
if os.path.lexists(marker_path):
details = os.stat(marker_path, follow_symlinks=False)
# The verified owner marker is removed only after every payload
# entry is gone, so finishing the empty root must make progress
# even when one oversized payload exhausted this pass's budget.
budget.consume(details.st_size, depth=depth + 1)
if not private_file_ready(marker_path):
raise OSError('janitor owner marker lost its private identity')
os.remove(marker_path)
os.rmdir(directory)
fsync_directory(os.path.dirname(directory))
return True
os.rmdir(directory)
continue
if entry.path == marker_path:
continue
details = entry.stat(follow_symlinks=False)
is_regular = stat.S_ISREG(details.st_mode)
if not budget.consume(
details.st_size, depth=depth + 1, allow_oversized=is_regular,
):
return False
if entry.is_symlink() or is_reparse_point(entry.path):
raise OSError(f'janitor candidate contains a link or reparse point: {entry.path}')
if stat.S_ISDIR(details.st_mode):
child_iterator = os.scandir(entry.path)
stack.append((entry.path, child_iterator, depth + 1))
elif is_regular:
os.chmod(entry.path, stat.S_IWRITE | stat.S_IREAD)
os.remove(entry.path)
else:
raise OSError(f'janitor candidate contains an unsupported entry: {entry.path}')
finally:
for _, iterator, _ in stack:
iterator.close()
return False
def _layout_cursor_name(root, relative_parent, prefixes):
identity = '|'.join((
canonical_path(root), str(relative_parent).replace(os.sep, '/'), ','.join(prefixes),
))
return 'layout:' + hashlib.sha256(identity.encode('utf-8')).hexdigest()
class JanitorCursorStore:
SCHEMA = 1
def __init__(self, path, root):
self.path = os.path.abspath(path)
self.root_hash = hashlib.sha256(canonical_path(root).encode('utf-8')).hexdigest()
self.dirty = False
self.state = {
'schema': self.SCHEMA,
'root_sha256': self.root_hash,
'next_layout': 0,
'layouts': {},
}
self.iterators = {}
self.seeking = {}
if os.path.lexists(self.path):
loaded = read_private_json(self.path, max_bytes=256 * 1024)
if (
not isinstance(loaded, dict)
or loaded.get('schema') != self.SCHEMA
or loaded.get('root_sha256') != self.root_hash
or not isinstance(loaded.get('layouts'), dict)
):
raise RuntimeError('janitor cursor authority is invalid')
self.state = loaded
else:
require_private_directory(os.path.dirname(self.path), create=True)
atomic_write_private_json(self.path, self.state)
def _save(self):
atomic_write_private_json(self.path, self.state)
self.dirty = False
def _mark_dirty(self):
self.dirty = True
def flush(self):
if self.dirty:
self._save()
def next_layout(self, count):
index = int(self.state.get('next_layout') or 0) % max(1, int(count))
self.state['next_layout'] = (index + 1) % max(1, int(count))
self._mark_dirty()
return index
def next_entry(self, layout_name, parent):
iterator = self.iterators.get(layout_name)
if iterator is None:
iterator = os.scandir(parent)
self.iterators[layout_name] = iterator
last_name = str((self.state['layouts'].get(layout_name) or {}).get('last_name') or '')
self.seeking[layout_name] = bool(last_name)
try:
entry = next(iterator)
except StopIteration:
iterator.close()
self.iterators.pop(layout_name, None)
self.seeking.pop(layout_name, None)
current = self.state['layouts'].setdefault(layout_name, {})
current['last_name'] = ''
current['wrap_count'] = int(current.get('wrap_count') or 0) + 1
self._mark_dirty()
return None, False
current = self.state['layouts'].setdefault(layout_name, {'last_name': '', 'wrap_count': 0})
target = str(current.get('last_name') or '')
if self.seeking.get(layout_name):
if entry.name == target:
self.seeking[layout_name] = False
return entry, True
current['last_name'] = entry.name
self._mark_dirty()
return entry, False
def close(self):
for iterator in self.iterators.values():
iterator.close()
self.iterators.clear()
class _MemoryCursorStore(JanitorCursorStore):
def __init__(self):
self.path = ''
self.root_hash = ''
self.dirty = False
self.state = {'schema': 1, 'root_sha256': '', 'next_layout': 0, 'layouts': {}}
self.iterators = {}
self.seeking = {}
def _save(self):
self.dirty = False
return None
def iter_candidates(root, budget, cursor_store):
layouts = list(APPROVED_LAYOUTS)
completed_layouts = set()
while not budget.exhausted and len(completed_layouts) < len(layouts):
if (
budget.enumerated >= budget.max_enumerated
or time.monotonic() - budget.started_at >= budget.max_seconds
):
budget.exhausted = True
return
index = cursor_store.next_layout(len(layouts))
root_kind, relative_parent, prefixes = layouts[index]
layout_name = _layout_cursor_name(root, relative_parent, prefixes)
if layout_name in completed_layouts:
continue
parent = os.path.join(root, relative_parent) if relative_parent else root
try:
if not os.path.isdir(parent) or is_reparse_point(parent):
completed_layouts.add(layout_name)
continue
entry, seeking = cursor_store.next_entry(layout_name, parent)
except OSError:
completed_layouts.add(layout_name)
continue
if entry is None:
completed_layouts.add(layout_name)
continue
if not budget.consume_enumerated():
return
if seeking:
continue
if not entry.name.startswith(prefixes) or not entry.is_dir(follow_symlinks=False):
continue
if not budget.consume(candidate=True):
return
yield root_kind, layout_name, entry.name, entry.path
def run_janitor_pass(
root, allowed_executables, minimum_age_sec=7200, budget=None,
cursor_store=None, excluded_relative_paths=(),
):
budget = budget or JanitorBudget()
root = require_private_directory(root, create=False)
cursor_store = cursor_store or _MemoryCursorStore()
excluded = set()
for value in excluded_relative_paths:
relative = str(value or '').replace('\\', '/')
if (
not relative or relative.startswith('/') or relative.endswith('/')
or any(part in ('', '.', '..') for part in relative.split('/'))
):
raise ValueError('janitor exclusion path is invalid')
excluded.add(relative)
if len(excluded) > 4096:
raise ValueError('janitor exclusion set exceeds its bound')
report = {'considered': 0, 'removed': 0, 'retained': 0, 'errors': 0, 'exhausted': False}
try:
for _, _, _, path in iter_candidates(root, budget, cursor_store):
report['considered'] += 1
if _marker_relative_path(root, path) in excluded:
report['retained'] += 1
continue
marker_path = os.path.join(path, MARKER_NAME)
try:
if not private_file_ready(marker_path):
report['retained'] += 1
continue
marker = read_private_json(marker_path)
eligible, _ = validate_marker(
root, path, marker, allowed_executables, minimum_age_sec,
)
if not eligible:
report['retained'] += 1
continue
if bounded_remove_tree(path, budget):
report['removed'] += 1
else:
report['retained'] += 1
except (OSError, ValueError):
report['errors'] += 1
if budget.exhausted:
break
finally:
cursor_store.flush()
report['exhausted'] = budget.exhausted
report['entries'] = budget.entries
report['bytes'] = budget.bytes
report['enumerated'] = budget.enumerated
return report
def parse_args():
parser = argparse.ArgumentParser(description='Bounded scanner work-directory janitor')
parser.add_argument('--config', required=True)
return parser.parse_args()
def main():
metadata = require_active_supervisor_child(child_kind='janitor', require_dsn=False)
args = parse_args()
import yaml
with open(args.config, 'r', encoding='utf-8') as handle:
config = apply_path_config(yaml.safe_load(handle) or {}, args.config)
global_config = config.get('global') or {}
janitor_config = ((config.get('supervisor') or {}).get('janitor') or {})
manifest = metadata.get('code_manifest') or {}
allowed = [sys.executable]
allowed.extend(
item.get('path') for item in (manifest.get('executables') or {}).values()
if isinstance(item, dict) and item.get('path')
)
interval = max(5.0, float(janitor_config.get('interval_sec', 60) or 60))
cursor_store = JanitorCursorStore(
os.path.join(global_config['state_dir'], 'janitor.cursor.json'),
global_config['work_dir'],
)
try:
while True:
budget = JanitorBudget(
max_candidates=max(1, int(janitor_config.get('max_candidates', 50) or 50)),
max_entries=max(1, int(janitor_config.get('max_entries', 10000) or 10000)),
max_bytes=max(1, int(janitor_config.get('max_bytes', 1024 * 1024 * 1024) or 1)),
max_seconds=max(0.1, float(janitor_config.get('max_seconds', 30) or 30)),
max_depth=max(1, int(janitor_config.get('max_depth', 64) or 64)),
max_enumerated=max(1, int(janitor_config.get('max_enumerated', 1000) or 1000)),
)
report = run_janitor_pass(
global_config['work_dir'], allowed,
minimum_age_sec=max(0, int(janitor_config.get('minimum_age_sec', 7200) or 0)),
budget=budget, cursor_store=cursor_store,
)
print(json.dumps(report, ensure_ascii=True, sort_keys=True), flush=True)
time.sleep(interval)
finally:
cursor_store.close()
if __name__ == '__main__':
main()
+736
View File
@@ -0,0 +1,736 @@
import sys
sys.dont_write_bytecode = True
if not sys.dont_write_bytecode:
raise RuntimeError('JSONL projector could not disable bytecode writes')
import argparse
import hashlib
import json
import os
import re
import time
from dataclasses import dataclass
from lifecycle_authority import require_active_supervisor_child
from paths import apply_path_config
from process_identity import current_process_identity
from runtime_security import (
PrivateFileLock,
PrivatePathState,
durable_publish,
durable_unlink,
harden_private_file,
inspect_private_relative_path,
private_file_ready,
reject_reparse_components,
require_private_directory,
)
from scanner_db import ScannerDB
STREAM_MASKS = {'scan_results': 1, 'found_secrets': 2, 'scan_errors': 4}
MAX_SERIALIZED_EVENT_BYTES = 192 * 1024 * 1024
MAX_TAIL_QUARANTINE_BYTES = MAX_SERIALIZED_EVENT_BYTES
@dataclass(frozen=True)
class SerializedStream:
stream_name: str
path: str
byte_length: int
payload_sha256: str
record_count: int
artifact_id: int = 0
class _HashedWriter:
def __init__(self, handle, max_bytes):
self.handle = handle
self.max_bytes = int(max_bytes)
self.digest = hashlib.sha256()
self.bytes_written = 0
def write(self, payload):
payload = payload.encode('utf-8') if isinstance(payload, str) else bytes(payload)
if self.bytes_written + len(payload) > self.max_bytes:
raise ValueError('projection serialization exceeds its event byte bound')
self.handle.write(payload)
self.digest.update(payload)
self.bytes_written += len(payload)
def _write_json_line(writer, value):
encoder = json.JSONEncoder(
ensure_ascii=False, sort_keys=True, separators=(',', ':'), default=str,
)
for chunk in encoder.iterencode(value):
writer.write(chunk)
writer.write(b'\n')
def _write_json_value(writer, value):
encoder = json.JSONEncoder(
ensure_ascii=False, sort_keys=True, separators=(',', ':'), default=str,
)
for chunk in encoder.iterencode(value):
writer.write(chunk)
def _write_json_array(writer, values):
writer.write(b'[')
first = True
for value in values:
if not first:
writer.write(b',')
_write_json_value(writer, value)
first = False
writer.write(b']')
def _write_scan_result(writer, header, findings, errors):
values = dict(header)
keys = sorted(set(values) | {'findings', 'errors'})
writer.write(b'{')
for index, key in enumerate(keys):
if index:
writer.write(b',')
_write_json_value(writer, key)
writer.write(b':')
if key == 'findings':
_write_json_array(writer, findings)
elif key == 'errors':
_write_json_array(writer, errors)
else:
_write_json_value(writer, values[key])
writer.write(b'}\n')
class JsonlProjector:
def __init__(
self, db, results_dir, supervisor_instance_id, lease_seconds=300,
fault=None, keycheck_dir=None, quarantine_max_items=10000,
quarantine_max_bytes=1024 * 1024 * 1024, artifact_tracking=True,
projection_max_bytes=2 * 1024 * 1024 * 1024,
):
self.db = db
self.results_dir = require_private_directory(results_dir, create=False)
self.keycheck_dir = require_private_directory(
keycheck_dir or os.path.join(os.path.dirname(self.results_dir), 'keychecks'),
create=False,
)
self.supervisor_instance_id = str(supervisor_instance_id)
self.lease_seconds = max(30, int(lease_seconds))
self.fault = fault
self.lease = None
self.file_lock = None
self.temp_dir = require_private_directory(
os.path.join(self.results_dir, '.projection-tmp'), create=True,
)
self.quarantine_dir = require_private_directory(
os.path.join(self.results_dir, '.projection-quarantine'), create=True,
)
self.quarantine_max_items = max(0, int(quarantine_max_items))
self.quarantine_max_bytes = max(0, int(quarantine_max_bytes))
self.projection_max_bytes = max(0, int(projection_max_bytes))
self.artifact_tracking = bool(artifact_tracking)
def _inject(self, stage, value=None):
if self.fault is not None:
self.fault(stage, value)
def start(self):
self.db.require_runtime_safety_schema()
self.db.require_final_cutover()
self.file_lock = PrivateFileLock(
os.path.join(self.results_dir, '.jsonl-projector.lock')
).acquire()
self.lease = self.db.acquire_pipeline_lease(
'jsonl_projector', self.supervisor_instance_id, current_process_identity(),
lease_seconds=self.lease_seconds, initial_state='starting',
)
if not self.lease:
self.file_lock.release()
self.file_lock = None
raise RuntimeError('another JSONL projector owns the singleton advisory lock')
if self.artifact_tracking:
self.reconcile_terminal_temps()
self.recover_rotations()
if not self.heartbeat():
raise RuntimeError('JSONL projector ready lease publication failed')
return self
def heartbeat(self, error=''):
return self.db.heartbeat_pipeline_lease(
'jsonl_projector', self.lease['generation'], self.lease['lease_token'],
lease_seconds=self.lease_seconds, state='ready', error=error,
)
def _rollback_database(self):
connection = getattr(self.db, 'conn', None)
if connection is not None:
try:
connection.rollback()
except BaseException:
pass
def stop(self, error=''):
try:
if self.lease:
self._rollback_database()
try:
self.db.release_pipeline_lease(
'jsonl_projector', self.lease['generation'], self.lease['lease_token'],
state='failed' if error else 'released', error=error,
)
except BaseException:
self._rollback_database()
raise
self.lease = None
finally:
if self.file_lock:
self.file_lock.release()
self.file_lock = None
def _results_path(self, relative, stream_name=''):
root = self.keycheck_dir if str(stream_name).startswith('keycheck:') else self.results_dir
path = os.path.abspath(os.path.join(root, str(relative).replace('/', os.sep)))
if os.path.commonpath((root, path)) != root or path == root:
raise ValueError('projection path escapes results_dir')
reject_reparse_components(os.path.dirname(path))
return path
@staticmethod
def _segment_relative(base_relative, generation):
base, extension = os.path.splitext(base_relative)
return f'{base}.g{int(generation):06d}{extension}'
def _create_private_empty(self, path):
if os.path.exists(path):
if not private_file_ready(path):
raise OSError(f'projection active file is not private: {path}')
return
descriptor = os.open(
path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0), 0o600,
)
os.close(descriptor)
harden_private_file(path)
def _prepared_path(self, job, stream_name):
safe_stream_name = re.sub(r'[^A-Za-z0-9_.-]+', '_', stream_name)
event_hash = str(job['event_hash'] or '').lower()
if not re.fullmatch(r'[a-f0-9]{64}', event_hash):
raise ValueError('projection job event hash is not a canonical SHA-256 identity')
path = os.path.join(
self.temp_dir,
f'job-{int(job["id"])}-{safe_stream_name}-{event_hash}.prepared',
)
relative = os.path.relpath(path, self.results_dir).replace(os.sep, '/')
artifact_id = 0
if self.artifact_tracking:
artifact_id = self.db.register_pipeline_artifact(
'jsonl_projector', 'prepared_stream', job['id'], stream_name,
relative, state='expected', byte_count=int(job['capacity_bytes']),
)
if os.path.lexists(path):
if not private_file_ready(path):
raise OSError(f'projection prepared file is not private: {path}')
else:
descriptor = os.open(
path,
os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0),
0o600,
)
os.close(descriptor)
harden_private_file(path)
return path, artifact_id
def _delete_registered_artifact(self, path, artifact_id):
relative = os.path.relpath(path, self.results_dir).replace(os.sep, '/')
inspection = inspect_private_relative_path(self.results_dir, relative)
if inspection.state == PrivatePathState.UNKNOWN:
raise OSError('projection artifact state is unknown during cleanup')
if inspection.state == PrivatePathState.PRESENT:
durable_unlink(inspection.path)
inspection = inspect_private_relative_path(self.results_dir, relative)
if inspection.state != PrivatePathState.ABSENT:
raise OSError('projection artifact unlink was not confirmed')
if artifact_id:
self.db.mark_pipeline_artifact_deleted(artifact_id)
def reconcile_terminal_temps(self, max_pages=100):
if not self.artifact_tracking or not hasattr(self.db, 'projection_terminal_temp_artifacts'):
return
for _ in range(max(1, int(max_pages))):
rows = self.db.projection_terminal_temp_artifacts(100)
if not rows:
return
for row in rows:
path = self._results_path(row['relative_path'])
self._delete_registered_artifact(path, row['id'])
if len(rows) < 100:
return
return
def recover_rotations(self):
for rotation in self.db.pending_projection_rotations(100):
active = self._results_path(rotation['base_relative_path'], rotation['stream_name'])
segment = self._results_path(rotation['segment_relative_path'], rotation['stream_name'])
require_private_directory(os.path.dirname(active), create=True)
active_exists = os.path.isfile(active)
segment_exists = os.path.isfile(segment)
if active_exists and segment_exists:
if (
os.path.getsize(active) != 0
or os.path.getsize(segment) != int(rotation['source_bytes'])
):
raise RuntimeError('projection rotation has conflicting active and immutable names')
elif active_exists:
if os.path.getsize(active) != int(rotation['source_bytes']):
raise RuntimeError('projection rotation active size changed')
durable_publish(active, segment)
elif not segment_exists:
raise RuntimeError('projection rotation lost both exact source names')
self._create_private_empty(active)
if not self.db.complete_projection_rotation(rotation['id']):
raise RuntimeError('projection rotation completion fence failed')
def _serialize(self, job):
if job['job_kind'] == 'scan_event':
scan = self.db.projection_scan_header(
job['target_scan_id'], max_bytes=self.projection_max_bytes,
)
if not isinstance(scan, dict) or not isinstance(scan.get('result'), dict):
raise ValueError('authoritative scan result cannot be reconstructed')
result = scan['result']
normalized = scan['storage'] == 'normalized_v2'
stream_specs = [
(name, mask) for name, mask in STREAM_MASKS.items()
if int(job['required_stream_mask']) & mask
]
elif job['job_kind'] == 'keycheck_event':
result = self.db.keycheck_result_for_projection(job['keycheck_result_id'])
if not isinstance(result, dict):
raise ValueError('authoritative keycheck result cannot be reconstructed')
service = str(result['service'])
stream_specs = [
(name, mask) for name, mask in (
(f'keycheck:{service}:results', 8),
(f'keycheck:{service}:status', 16),
) if int(job['required_stream_mask']) & mask
]
else:
raise ValueError(f'unsupported projection job kind: {job["job_kind"]}')
streams = []
prepared_paths = []
try:
for stream_name, mask in stream_specs:
path, artifact_id = self._prepared_path(job, stream_name)
prepared_paths.append((path, artifact_id))
count = 0
with open(path, 'wb', buffering=0) as handle:
writer = _HashedWriter(handle, self.projection_max_bytes)
if stream_name == 'scan_results':
findings = (
self.db.iter_projection_findings(job['target_scan_id'])
if normalized else iter(result.get('findings') or ())
)
errors = (
self.db.iter_projection_errors(job['target_scan_id'])
if normalized else iter(result.get('errors') or ())
)
_write_scan_result(writer, result, findings, errors)
count = 1
elif stream_name == 'found_secrets':
findings = (
self.db.iter_projection_findings(job['target_scan_id'])
if normalized else iter(result.get('findings') or ())
)
for finding in findings:
_write_json_line(writer, finding)
count += 1
elif stream_name == 'scan_errors':
timestamp = result.get('timestamp') or ''
scan_type = result.get('scan_type') or ''
target = result.get('target') or ''
event_id = result.get('scan_event_id') or job['event_id']
errors = (
self.db.iter_projection_errors(job['target_scan_id'])
if normalized else iter(result.get('errors') or ())
)
for index, error in enumerate(errors, 1):
row_id = hashlib.sha256(
f'{event_id}|{index}'.encode('utf-8')
).hexdigest()
writer.write(
f'{row_id}\t{timestamp}\t{scan_type}\t{target}\t{error}\n'
)
count += 1
elif stream_name.endswith(':results'):
payload = {
'event_id': result['event_id'],
'service': result['service'],
'status': result['status'],
'status_group': result['status_group'],
'checked_at': result['checked_at'],
'key_hash': result['key_hash'],
'secret_hash': result['secret_hash'],
'key_masked': result['key_masked'],
'finding_uid': result['finding_uid'],
'detector': result['detector_name'],
'source': result['source'],
'message': result['message'],
'metadata': json.loads(result['metadata_json'] or '{}'),
'result_source': result['result_source'],
}
_write_json_line(writer, payload)
count = 1
else:
secret = result.get('credential_secret_text') or result.get('credential_secret_json') or ''
message = str(result.get('message') or '').replace('\r', ' ').replace('\n', ' ')[:1000]
writer.write(
f'{secret}\t{result["status"]}\t{result["checked_at"]}\t{message}\n'
)
count = 1
handle.flush()
os.fsync(handle.fileno())
streams.append(SerializedStream(
stream_name, path, writer.bytes_written, writer.digest.hexdigest(), count,
artifact_id,
))
if self.artifact_tracking:
self.db.register_pipeline_artifact(
'jsonl_projector', 'prepared_stream', job['id'], stream_name,
os.path.relpath(path, self.results_dir).replace(os.sep, '/'),
state='present', payload_sha256=writer.digest.hexdigest(),
byte_count=writer.bytes_written,
)
return streams
except BaseException:
self._rollback_database()
for path, artifact_id in prepared_paths:
try:
self._delete_registered_artifact(path, artifact_id)
except BaseException:
pass
raise
def _hash_region(self, path, offset, length):
digest = hashlib.sha256()
remaining = int(length)
with open(path, 'rb', buffering=0) as handle:
handle.seek(int(offset))
while remaining:
block = handle.read(min(1024 * 1024, remaining))
if not block:
raise OSError('projection append proof is truncated')
digest.update(block)
remaining -= len(block)
return digest.hexdigest()
def _quarantine_tail(self, active, offset, job, stream_name, append):
size = os.path.getsize(active)
length = max(0, size - int(offset))
safe_stream_name = re.sub(r'[^A-Za-z0-9_.-]+', '_', stream_name)
tail_hash = self._hash_region(active, offset, length) if length else hashlib.sha256(b'').hexdigest()
path = os.path.join(
self.quarantine_dir,
f'append-{append["id"]}-o{int(offset)}-l{length}-{tail_hash}.tail',
)
evidence_error = None
evidence_registered = False
try:
if length:
if length > MAX_TAIL_QUARANTINE_BYTES:
raise ValueError('projection partial tail exceeds quarantine byte bound')
relative = os.path.relpath(path, self.results_dir).replace(os.sep, '/')
registration = self.db.register_projection_tail_quarantine(
job['id'], append['id'], stream_name, relative, tail_hash, length,
self.quarantine_max_items, self.quarantine_max_bytes,
)
evidence_registered = True
if not os.path.lexists(path):
temporary = path + '.partial'
temp_relative = os.path.relpath(temporary, self.results_dir).replace(os.sep, '/')
temp_artifact = self.db.register_pipeline_artifact(
'jsonl_projector', 'projection_tail_temp', job['id'],
f'{append["id"]}:{int(offset)}:{length}:{tail_hash}',
temp_relative, state='expected', byte_count=length,
)
if os.path.lexists(temporary):
if not private_file_ready(temporary):
raise OSError('existing projection tail temporary is not private')
self._delete_registered_artifact(temporary, temp_artifact)
temp_artifact = self.db.register_pipeline_artifact(
'jsonl_projector', 'projection_tail_temp', job['id'],
f'{append["id"]}:{int(offset)}:{length}:{tail_hash}',
temp_relative, state='expected', byte_count=length,
)
descriptor = os.open(
temporary,
os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0),
0o600,
)
try:
os.close(descriptor)
descriptor = None
harden_private_file(temporary)
with open(active, 'rb', buffering=0) as source, open(temporary, 'wb', buffering=0) as target:
source.seek(int(offset))
remaining = length
while remaining:
block = source.read(min(1024 * 1024, remaining))
if not block:
raise OSError('projection partial tail changed while quarantining')
target.write(block)
remaining -= len(block)
target.flush()
os.fsync(target.fileno())
if (
os.path.getsize(temporary) != length
or self._hash_region(temporary, 0, length) != tail_hash
):
raise OSError('projection tail quarantine proof failed before publication')
self.db.register_pipeline_artifact(
'jsonl_projector', 'projection_tail_temp', job['id'],
f'{append["id"]}:{int(offset)}:{length}:{tail_hash}',
temp_relative, state='present', payload_sha256=tail_hash,
byte_count=length,
)
durable_publish(temporary, path)
temporary_state = inspect_private_relative_path(
self.results_dir, temp_relative,
)
if temporary_state.state != PrivatePathState.ABSENT:
raise OSError('projection tail temporary retirement was not confirmed')
self.db.mark_pipeline_artifact_deleted(temp_artifact)
finally:
if descriptor is not None:
os.close(descriptor)
elif (
not private_file_ready(path)
or os.path.getsize(path) != length
or self._hash_region(path, 0, length) != tail_hash
):
raise OSError('immutable projection tail evidence is invalid')
if not self.db.confirm_projection_tail_artifact(
registration['artifact_id'], tail_hash, length,
):
raise RuntimeError('projection tail artifact confirmation lost its fence')
except BaseException as exc:
evidence_error = exc
finally:
with open(active, 'r+b', buffering=0) as handle:
handle.truncate(int(offset))
handle.flush()
os.fsync(handle.fileno())
if os.path.getsize(active) != int(offset):
raise OSError('projection partial tail corrective truncation was not confirmed')
if evidence_error is not None:
raise evidence_error
def _rotate_if_needed(self, state, payload_bytes):
active = self._results_path(state['base_relative_path'], state['stream_name'])
require_private_directory(os.path.dirname(active), create=True)
self._create_private_empty(active)
current_size = os.path.getsize(active)
if current_size != int(state['committed_offset']):
state = self.db.initialize_projection_stream_offset(
state['stream_name'], current_size,
)
if current_size != int(state['committed_offset']):
raise RuntimeError('projection active size does not match its committed cursor')
if not current_size or current_size + payload_bytes <= int(state['rotation_bytes']):
return state
segment_relative = self._segment_relative(
state['base_relative_path'], state['current_generation'],
)
rotation = self.db.prepare_projection_rotation(
state['stream_name'], current_size, segment_relative,
)
segment = self._results_path(segment_relative, state['stream_name'])
self._inject('before_rotation_rename', rotation)
durable_publish(active, segment)
self._inject('after_rotation_rename', rotation)
self._create_private_empty(active)
if not self.db.complete_projection_rotation(rotation['id']):
raise RuntimeError('projection rotation completion fence failed')
updated = self.db.projection_stream_state(state['stream_name'])
oldest = int(updated['current_generation']) - int(updated['max_generations'])
if oldest >= 0:
old_relative = self._segment_relative(updated['base_relative_path'], oldest)
old_path = self._results_path(old_relative, updated['stream_name'])
if os.path.isfile(old_path):
durable_unlink(old_path)
return updated
def _append_stream(self, job, serialized):
state = self.db.projection_stream_state(serialized.stream_name)
if not state:
raise ValueError(f'projection stream is absent: {serialized.stream_name}')
existing = self.db.projection_append_for_job(job['id'], serialized.stream_name)
if existing is None:
state = self._rotate_if_needed(state, serialized.byte_length)
append = self.db.prepare_projection_append(
job['id'], job['lease_token'], serialized.stream_name,
state['generation'], serialized.byte_length,
serialized.payload_sha256, serialized.record_count,
)
else:
append = existing
if (
int(append['byte_length']) != serialized.byte_length
or str(append['payload_sha256']) != serialized.payload_sha256
or int(append['record_count']) != serialized.record_count
):
raise ValueError('prepared projection append conflicts with deterministic serialization')
active = self._results_path(state['base_relative_path'], serialized.stream_name)
require_private_directory(os.path.dirname(active), create=True)
self._create_private_empty(active)
offset = int(append['byte_offset'])
end = offset + int(append['byte_length'])
size = os.path.getsize(active)
if append['state'] == 'appended':
if size < end or self._hash_region(active, offset, append['byte_length']) != append['payload_sha256']:
raise ValueError('acknowledged projection append proof is invalid')
return
if size >= end:
if size == end and self._hash_region(active, offset, append['byte_length']) == append['payload_sha256']:
if not self.db.complete_projection_append(append['id'], job['id'], job['lease_token']):
raise RuntimeError('projection append replay acknowledgement failed')
return
self._quarantine_tail(active, offset, job, serialized.stream_name, append)
self._inject('after_tail_recovery', append)
elif size > offset:
self._quarantine_tail(active, offset, job, serialized.stream_name, append)
self._inject('after_tail_recovery', append)
elif size < offset:
raise ValueError('projection active file is shorter than its prepared offset')
self._inject('before_append', append)
with open(active, 'ab', buffering=0) as target, open(serialized.path, 'rb', buffering=0) as source:
while True:
block = source.read(1024 * 1024)
if not block:
break
target.write(block)
target.flush()
os.fsync(target.fileno())
self._inject('after_append_fsync', append)
if os.path.getsize(active) != end or self._hash_region(active, offset, append['byte_length']) != append['payload_sha256']:
raise RuntimeError('projection append proof failed after fsync')
if not self.db.complete_projection_append(append['id'], job['id'], job['lease_token']):
raise RuntimeError('projection append completion fence failed')
def process_one(self):
self.reconcile_terminal_temps(max_pages=1)
job = self.db.claim_projection_job(
self.lease['generation'], self.lease['lease_token'], self.lease_seconds,
)
if not job:
return False
streams = []
primary_failure = False
try:
streams = self._serialize(job)
actual_bytes = sum(item.byte_length for item in streams)
expanded = self.db.expand_projection_job_capacity(
job['id'], job['lease_token'], actual_bytes,
self.projection_max_bytes,
)
if expanded is False:
return True
if expanded is None:
raise RuntimeError('projection capacity expansion lost its lease fence')
job = expanded
for serialized in streams:
self._append_stream(job, serialized)
if not self.db.complete_projection_job(job['id'], job['lease_token']):
raise RuntimeError('projection job completion fence failed')
return True
except (ValueError, TypeError, UnicodeError, json.JSONDecodeError) as exc:
self._rollback_database()
try:
self.db.quarantine_projection_job(
job['id'], job['lease_token'], 'deterministic_projection_error', str(exc),
quarantine_max_items=self.quarantine_max_items,
quarantine_max_bytes=self.quarantine_max_bytes,
)
except BaseException:
primary_failure = True
self._rollback_database()
raise
return True
except BaseException:
primary_failure = True
self._rollback_database()
raise
finally:
for serialized in streams:
try:
self._delete_registered_artifact(
serialized.path, serialized.artifact_id,
)
except OSError:
pass
except BaseException:
self._rollback_database()
if not primary_failure:
raise
def parse_args():
parser = argparse.ArgumentParser(description='Singleton PostgreSQL-backed JSONL projector')
parser.add_argument('--config', required=True)
return parser.parse_args()
def main():
metadata = require_active_supervisor_child(child_kind='jsonl-projector', require_dsn=True)
args = parse_args()
import yaml
with open(args.config, 'r', encoding='utf-8') as handle:
config = apply_path_config(yaml.safe_load(handle) or {}, args.config)
global_config = config.get('global') or {}
settings = ((config.get('supervisor') or {}).get('jsonl_projector') or {})
db = ScannerDB(db_url=global_config['database_url'], initialize=False)
if not db.enabled:
raise SystemExit('JSONL projector PostgreSQL connection is unavailable')
db.set_application_name('truf-jsonl-projector')
worker = JsonlProjector(
db, global_config['results_dir'], metadata['instance_id'],
lease_seconds=int(settings.get('lease_seconds', 300)),
keycheck_dir=global_config['keycheck_dir'],
quarantine_max_items=int(global_config.get('pipeline_quarantine_max_items', 10000)),
quarantine_max_bytes=int(global_config.get('pipeline_quarantine_max_bytes', 1024 * 1024 * 1024)),
projection_max_bytes=int(global_config.get(
'projection_backlog_max_bytes', 2 * 1024 * 1024 * 1024,
)),
)
error = ''
try:
worker.start()
idle = max(0.05, float(settings.get('poll_sec', 0.2)))
next_heartbeat = time.monotonic() + worker.lease_seconds / 3
while True:
processed = worker.process_one()
if time.monotonic() >= next_heartbeat:
if not worker.heartbeat():
raise RuntimeError('JSONL projector heartbeat fence was lost')
next_heartbeat = time.monotonic() + worker.lease_seconds / 3
if not processed:
time.sleep(idle)
except KeyboardInterrupt:
pass
except BaseException as exc:
error = f'{type(exc).__name__}: {exc}'
raise
finally:
try:
worker.stop(error)
finally:
db.close()
if __name__ == '__main__':
main()
+71
View File
@@ -0,0 +1,71 @@
import sys
sys.dont_write_bytecode = True
import os
import sqlite3
import tempfile
from keycheckers.keycheck_common import (
append_checked,
append_status,
record_cached_keycheck_occurrence,
)
from keycheck_runner import ingest_keycheck_results_to_db
from scanner_db import ScannerDB
from runtime_security import ensure_private_directory
def main():
for key in ('SCANNER_DB_URL', 'DATABASE_URL', 'TRUF_MANAGED_POSTGRES_DSN', 'KEYCHECK_DB_URL'):
os.environ.pop(key, None)
with tempfile.TemporaryDirectory(prefix="keycheck-accounting-") as tmp:
db_path = os.path.join(tmp, "scanner.db")
output_dir = os.path.join(tmp, "keychecks", "openai")
ensure_private_directory(output_dir, reject_reparse=True)
db = ScannerDB(db_path=db_path)
db.close()
alive_file = os.path.join(output_dir, "openaiAlive.txt")
checked_file = os.path.join(output_dir, "openaiChecked.txt")
key = "sk-test-keycheck-accounting-1234567890"
append_status(alive_file, key, "ALIVE", "fixture", "smoke")
append_checked(checked_file, key, "ALIVE")
os.environ["KEYCHECK_DB_PATH"] = db_path
os.environ["KEYCHECK_OUTPUT_DIR"] = output_dir
os.environ["KEYCHECK_SERVICE"] = "openai"
finding = {"DetectorName": "OpenAI", "Raw": key}
ok = record_cached_keycheck_occurrence("openai", key, "ALIVE", "fixture.jsonl:1", finding, "OpenAI")
if not ok:
raise SystemExit("cached occurrence write returned false")
inserted = ingest_keycheck_results_to_db(
{"database_path": db_path, "keycheck_dir": os.path.join(tmp, "keychecks")},
["openai"],
max_rows=10,
)
if inserted != 1:
raise SystemExit(f"unexpected ingest count: {inserted}")
conn = sqlite3.connect(db_path)
try:
row = conn.execute(
"SELECT service, status, status_group, metadata_json FROM keycheck_results"
).fetchone()
finally:
conn.close()
if not row:
raise SystemExit("missing keycheck_results row")
service, status, status_group, metadata_json = row
if (service, status, status_group) != ("openai", "ALIVE", "alive"):
raise SystemExit(f"unexpected row status: {(service, status, status_group)}")
if "cached_status" not in metadata_json:
raise SystemExit("missing cached_status metadata")
print("keycheck accounting smoke ok")
if __name__ == "__main__":
main()
+524
View File
@@ -0,0 +1,524 @@
import hashlib
import json
import re
from dataclasses import dataclass, field
MAX_CANDIDATE_SECRET_BYTES = 1024 * 1024
MAX_CANDIDATE_METADATA_BYTES = 64 * 1024
@dataclass(frozen=True)
class CandidateSpec:
service: str
candidate_kind: str
credential_hash: str
provider_key_hash: str
secret_hash: str
secret_text: str | None = None
secret_json: str | None = None
key_masked: str = ''
endpoint: str = ''
principal: str = ''
metadata: dict = field(default_factory=dict)
def as_frame(self, attribution=None):
return {
'service': self.service,
'candidate_kind': self.candidate_kind,
'credential_hash': self.credential_hash,
'provider_key_hash': self.provider_key_hash,
'secret_hash': self.secret_hash,
'secret_text': self.secret_text,
'secret_json': self.secret_json,
'key_masked': self.key_masked,
'endpoint': self.endpoint,
'principal': self.principal,
'metadata': self.metadata,
'attribution': dict(attribution or {}),
}
DETECTOR_SERVICES = {
'openai': 'openai',
'anthropic': 'anthropic',
'qwendashscope': 'qwen',
'qwen_dashscope': 'qwen',
'qwen': 'qwen',
'dashscope': 'qwen',
'deepseek': 'deepseek',
'deepseekapikey': 'deepseek',
'deepseek_api_key': 'deepseek',
'zaiglm': 'zai',
'kimimoonshot': 'kimi',
'moonshotai': 'kimi',
'moonshot': 'kimi',
'kimi': 'kimi',
'openrouter': 'openrouter',
'groq': 'groq',
'replicate': 'replicate',
'xai': 'xai',
'huggingface': 'huggingface',
'github': 'github',
'githuboauth2': 'github',
'gitlab': 'gitlab',
'aws': 'aws',
'gcp': 'gcp',
'gcpapplicationdefaultcredentials': 'gcp',
'googleai': 'gemini',
'googleaistudio': 'gemini',
'azure': 'azure',
'azureopenai': 'azure',
'azurecontainerregistry': 'azure',
'azurefoundryendpointbeforekey': 'azure',
'azurefoundrykeybeforeendpoint': 'azure',
'dockerhub': 'dockerhub',
}
GEMINI_KEY_RE = re.compile(
r'AIza[0-9A-Za-z_-]{20,}|(?<![0-9A-Za-z_-])AQ\.[0-9A-Za-z_-]{50}(?![0-9A-Za-z_-])'
)
AZURE_OPENAI_KEY_RE = re.compile(r'\b[a-fA-F0-9]{32}\b')
AZURE_OPENAI_ENDPOINT_RE = re.compile(r'([a-z0-9-]+\.openai\.azure\.com)', re.IGNORECASE)
AZURE_FOUNDRY_ENDPOINT_RE = re.compile(
r'([a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models|services|inference)\.ai\.azure\.com)',
re.IGNORECASE,
)
AZURE_FOUNDRY_KEY_ASSIGNMENT_RE = re.compile(
r'(?is)(?:authorization|api[_-]?key|key|token|secret|credential|bearer)'
r'[^\n:=]{0,80}[:=]\s*["\']?(?:bearer\s+)?([A-Za-z0-9_./+=\-]{20,512})'
)
NON_FOUNDRY_KEY_PREFIXES = (
'sk-', 'sk_', 'sk-or-', 'xai-', 'ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_',
'github_pat_', 'glpat-', 'glrt-', 'hf_', 'AIza', 'AQ.', 'zai-', 'gsk_',
'r8_', 'nvapi-',
)
GITHUB_TOKEN_RE = re.compile(r'\b(?:gh[pousr]_[A-Za-z0-9_]{20,}|github_pat_[A-Za-z0-9_]{20,})\b')
GITLAB_TOKEN_RE = re.compile(r'\b(?:glpat|gloas|glcbt|glimt|glrt|glft|glsoat)-[A-Za-z0-9_\-=]{20,}\b')
DOCKER_PAT_RE = re.compile(r'\bdckr_pat_[A-Za-z0-9_-]{27}\b')
QWEN_KEY_RE = re.compile(r'\b(?:sk-sp-[A-Za-z0-9_-]{16,}|sk-[A-Za-z0-9_-]{20,})\b')
KIMI_KEY_RE = re.compile(r'\bsk-[A-Za-z0-9][A-Za-z0-9_-]{20,505}\b')
ZAI_KEY_RE = re.compile(
r'(?<![A-Za-z0-9_.-])(?:'
r'(?:zai|sk)-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|'
r'[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}'
r')(?![A-Za-z0-9_.-])'
)
GENERIC_SK_PROVIDERS = {'qwen', 'deepseek', 'kimi', 'zai'}
AMBIGUOUS_PROVIDER_HINTS = {'ambiguous_qwen_deepseek', 'ambiguous_generic_sk'}
def _json(value):
return json.dumps(value, ensure_ascii=True, sort_keys=True, separators=(',', ':'))
def _provider_json(value):
return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(',', ':'))
def _bounded(value, max_bytes):
text = str(value or '')
encoded = text.encode('utf-8', errors='strict')
if len(encoded) > max_bytes:
raise ValueError('keycheck candidate field exceeds its byte bound')
return text
def _mask(value):
text = str(value or '')
if len(text) <= 8:
return '*' * len(text)
return text[:4] + ('*' * min(24, len(text) - 8)) + text[-4:]
def _service_for_finding(finding):
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
hint = str(context.get('provider_hint') or '').lower()
if hint in AMBIGUOUS_PROVIDER_HINTS:
return 'provider_resolver'
if hint in GENERIC_SK_PROVIDERS:
return hint
detector = re.sub(r'[^a-z0-9_]', '', str(
finding.get('DetectorName') or finding.get('DetectorType') or ''
).lower())
service = DETECTOR_SERVICES.get(detector, '')
if service:
return service
extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {}
name = re.sub(r'[^a-z0-9_]', '', str(extra.get('name') or '').lower())
return DETECTOR_SERVICES.get(name, '')
def _make_candidate(
service, candidate_kind, probe_material, raw_material, *, secret_text=None,
secret_json=None, endpoint='', principal='', metadata=None,
):
probe_material = _bounded(probe_material, MAX_CANDIDATE_SECRET_BYTES)
raw_material = _bounded(raw_material, MAX_CANDIDATE_SECRET_BYTES)
provider_key_hash = hashlib.sha256(probe_material.encode('utf-8')).hexdigest()
secret_hash = hashlib.sha256(raw_material.encode('utf-8')).hexdigest()
credential_hash = hashlib.sha256('|'.join((
'truf-credential-v2', service, probe_material,
)).encode('utf-8')).hexdigest()
metadata = dict(metadata or {})
if len(_json(metadata).encode('utf-8')) > MAX_CANDIDATE_METADATA_BYTES:
raise ValueError('keycheck candidate metadata exceeds its byte bound')
return CandidateSpec(
service=service,
candidate_kind=candidate_kind,
credential_hash=credential_hash,
provider_key_hash=provider_key_hash,
secret_hash=secret_hash,
secret_text=secret_text,
secret_json=secret_json,
key_masked=_mask(probe_material),
endpoint=endpoint,
principal=principal,
metadata=metadata,
)
def stored_provider_key_hash(service, candidate_kind, secret_text, secret_json, endpoint='', principal=''):
service = str(service or '').lower()
candidate_kind = str(candidate_kind or '')
secret_text = str(secret_text or '')
secret_json = str(secret_json or '')
endpoint = str(endpoint or '').lower()
principal = str(principal or '')
if service == 'gcp':
parsed = _json_object(secret_json)
probe = _provider_json(parsed) if parsed is not None else secret_json
elif service == 'azure' and candidate_kind == 'azure_service_principal':
parsed = _json_object(secret_json) or {}
probe = ':'.join(str(parsed.get(name) or '') for name in ('tenantId', 'clientId', 'clientSecret'))
if probe == '::':
probe = ':'.join(str(parsed.get(name) or '') for name in ('tenant_id', 'client_id', 'client_secret'))
elif service == 'azure' and candidate_kind == 'azure_container_registry':
parsed = _json_object(secret_json) or {}
probe = f"{parsed.get('username') or principal}:{parsed.get('password') or ''}"
elif service == 'azure' and endpoint:
probe = f'{endpoint}:{secret_text}'
elif service == 'dockerhub' and principal:
probe = f'{principal}:{secret_text}'
else:
probe = secret_text or secret_json
probe = _bounded(probe, MAX_CANDIDATE_SECRET_BYTES)
if not probe:
raise ValueError('stored keycheck credential has no provider probe material')
return hashlib.sha256(probe.encode('utf-8')).hexdigest()
def _json_object(value):
try:
parsed = json.loads(str(value or ''))
except (TypeError, ValueError):
return None
return parsed if isinstance(parsed, dict) else None
def _raw_material(finding):
value = finding.get('RawV2') or finding.get('Raw')
if value:
return str(value)
structured = finding.get('StructuredData')
return _json(structured) if isinstance(structured, dict) and structured else ''
def _azure_foundry_keyish(value):
text = re.sub(r'(?i)^bearer\s+', '', str(value or '').strip().strip('"\'`,;')).strip()
lowered = text.lower()
if not (20 <= len(text) <= 512) or any(character.isspace() for character in text):
return False
if any(marker in lowered for marker in ('http://', 'https://', '{{', '${', '<', 'azure.com')):
return False
if text.startswith(NON_FOUNDRY_KEY_PREFIXES):
return False
return bool(re.search(r'[A-Za-z]', text) and re.search(r'[0-9]', text))
def extract_azure_foundry_parts(raw, raw_v2=''):
materials = [str(value or '') for value in (raw_v2, raw) if value]
endpoint = next((
match.group(1).lower()
for value in materials
for match in [AZURE_FOUNDRY_ENDPOINT_RE.search(value)]
if match
), '')
if not endpoint:
return None
for value in materials:
for key in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(value):
key = str(key).strip().strip('"\'`,;')
if _azure_foundry_keyish(key):
return {'key': key, 'endpoint': endpoint}
match = AZURE_FOUNDRY_ENDPOINT_RE.search(value)
if not match:
continue
before = value[:match.start()].strip(' \t\r\n:=,;\'"/')
after = value[match.end():].strip(' \t\r\n:=,;\'"/')
for key in (after, before):
if _azure_foundry_keyish(key):
return {'key': key, 'endpoint': endpoint}
return None
def extract_candidates(finding, attribution=None):
if not isinstance(finding, dict):
return
service = _service_for_finding(finding)
if not service:
return
detector = str(finding.get('DetectorName') or finding.get('DetectorType') or '')
detector_key = re.sub(r'[^a-z0-9_]', '', detector.lower())
postman = finding.get('PostmanContext') if isinstance(finding.get('PostmanContext'), dict) else {}
raw = str(finding.get('Raw') or '')
raw_v2 = str(finding.get('RawV2') or '')
raw_material = _raw_material(finding)
base_metadata = {
'detector_name': detector,
'finding_uid': str(finding.get('finding_uid') or ''),
}
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
provider_candidates = [
str(provider).lower() for provider in context.get('provider_candidates') or ()
if str(provider).lower() in GENERIC_SK_PROVIDERS
]
provider_hint = str(context.get('provider_hint') or '')
if provider_hint:
base_metadata['provider_hint'] = provider_hint
if provider_candidates:
base_metadata['provider_candidates'] = list(dict.fromkeys(provider_candidates))
endpoint = str(postman.get('endpoint') or '')
principal = str(postman.get('principal') or postman.get('username') or '')
if service == 'aws':
probe = raw_v2 or raw
if ':' in probe:
yield _make_candidate(
service, 'aws_access_key_pair', probe, raw_material,
secret_text=probe, metadata={**base_metadata, 'raw_v2': probe},
)
return
if service == 'gcp':
parsed = _json_object(raw_v2)
if parsed is None:
context = finding.get('ScannerContext') if isinstance(finding.get('ScannerContext'), dict) else {}
parsed = _json_object(context.get('nearby'))
if parsed is not None:
probe = _provider_json(parsed)
yield _make_candidate(
service, 'gcp_json', probe, raw_material or probe,
secret_json=probe, metadata={**base_metadata, 'raw_v2': probe},
)
return
if service == 'azure':
parsed = _json_object(raw_v2)
if detector_key == 'azure':
if parsed:
tenant = parsed.get('tenantId') or parsed.get('tenant_id')
client = parsed.get('clientId') or parsed.get('client_id')
secret = parsed.get('clientSecret') or parsed.get('client_secret')
if tenant and client and secret:
probe = f'{tenant}:{client}:{secret}'
canonical_json = _json(parsed)
yield _make_candidate(
service, 'azure_service_principal', probe, raw_material,
secret_json=canonical_json,
metadata={**base_metadata, 'raw_v2': canonical_json},
)
return
if detector_key == 'azurecontainerregistry':
if parsed:
username = parsed.get('username')
password = parsed.get('password')
if username and password:
probe = f'{username}:{password}'
canonical_json = _json(parsed)
yield _make_candidate(
service, 'azure_container_registry', probe, raw_material,
secret_json=canonical_json, principal=str(username),
metadata={**base_metadata, 'raw_v2': canonical_json},
)
return
if detector_key == 'azureopenai':
match = re.match(r'^([a-fA-F0-9]{32}):(.+\.openai\.azure\.com)$', raw_v2)
key = match.group(1) if match else raw
found_endpoint = match.group(2).lower() if match else endpoint.lower()
if key:
probe = f'{found_endpoint}:{key}' if found_endpoint else key
provider_raw_v2 = f'{key}:{found_endpoint}' if found_endpoint else raw_v2
yield _make_candidate(
service, 'azure_openai', probe, raw_material,
secret_text=key, endpoint=found_endpoint,
metadata={**base_metadata, 'raw_v2': provider_raw_v2},
)
return
foundry = extract_azure_foundry_parts(raw, raw_v2)
if foundry:
foundry_endpoint = foundry['endpoint']
foundry_key = foundry['key']
probe = f'{foundry_endpoint}:{foundry_key}'
yield _make_candidate(
service, 'azure_foundry', probe, raw_material,
secret_text=foundry_key, endpoint=foundry_endpoint,
metadata={**base_metadata, 'raw_v2': probe},
)
return
if service == 'dockerhub':
token_match = DOCKER_PAT_RE.search(raw_v2) or DOCKER_PAT_RE.search(raw)
if not token_match:
return
token = token_match.group(0)
username = ''
if ':' in raw_v2 and raw_v2.rsplit(':', 1)[-1] == token:
username = raw_v2.rsplit(':', 1)[0]
extra = finding.get('ExtraData') if isinstance(finding.get('ExtraData'), dict) else {}
analysis = finding.get('AnalysisInfo') if isinstance(finding.get('AnalysisInfo'), dict) else {}
username = username or str(extra.get('hub_username') or analysis.get('username') or '')
probe = f'{username}:{token}' if username else token
yield _make_candidate(
service, 'dockerhub_pat', probe, raw_material,
secret_text=token, principal=username,
metadata={**base_metadata, 'raw_v2': probe},
)
return
patterns = {
'github': GITHUB_TOKEN_RE,
'gitlab': GITLAB_TOKEN_RE,
'gemini': GEMINI_KEY_RE,
'qwen': QWEN_KEY_RE,
'kimi': KIMI_KEY_RE,
'zai': ZAI_KEY_RE,
'provider_resolver': KIMI_KEY_RE,
}
pattern = patterns.get(service)
values = pattern.findall(raw + '\n' + raw_v2) if pattern else [raw or raw_v2]
seen = set()
for value in values:
value = str(value or '').strip()
if not value or value in seen:
continue
seen.add(value)
yield _make_candidate(
service, 'provider_key', value, raw_material or value,
secret_text=value, endpoint=endpoint, principal=principal,
metadata={**base_metadata, 'raw_v2': raw_v2},
)
def extract_structured_candidates(artifact_context, attribution=None):
if not isinstance(artifact_context, dict):
return
for finding in artifact_context.get('findings') or ():
yield from extract_candidates(finding, attribution)
contexts = artifact_context.get('contexts') or ()
origin = str(artifact_context.get('origin') or 'structured-artifact')
endpoint_values = []
for context in contexts:
if not isinstance(context, dict):
continue
text = ' '.join(str(context.get(key) or '') for key in ('value', 'endpoint', 'host', 'key'))
endpoint_values.extend(AZURE_OPENAI_ENDPOINT_RE.findall(text))
endpoint_values.extend(AZURE_FOUNDRY_ENDPOINT_RE.findall(text))
seen = set()
for index, context in enumerate(contexts):
if not isinstance(context, dict):
continue
value = str(context.get('value') or '').strip()
if not value or '{{' in value or '${' in value:
continue
path = str(context.get('path') or '')
text = ' '.join((
value, str(context.get('endpoint') or ''), str(context.get('host') or ''),
str(context.get('key') or ''),
))
for key in GEMINI_KEY_RE.findall(value):
identity = ('gemini', key)
if identity in seen:
continue
seen.add(identity)
yield _make_candidate(
'gemini', 'structured_postman', key, key, secret_text=key,
metadata={
'detector_name': 'GoogleAIStudio', 'origin': origin,
'structured_origin': f'{origin}:{path or index}',
},
)
azure_key = AZURE_OPENAI_KEY_RE.search(value)
if azure_key:
endpoints = AZURE_OPENAI_ENDPOINT_RE.findall(text) or [
endpoint for endpoint in endpoint_values
if AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint)
]
for endpoint in endpoints[:5]:
key = azure_key.group(0)
secret = f'{key}:{str(endpoint).lower()}'
identity = ('azure-openai', secret)
if identity in seen:
continue
seen.add(identity)
yield _make_candidate(
'azure', 'structured_postman', f'{str(endpoint).lower()}:{key}',
secret, secret_text=key, endpoint=str(endpoint).lower(), metadata={
'detector_name': 'AzureOpenAI', 'origin': origin,
'raw_v2': secret,
'structured_origin': f'{origin}:{path or index}',
},
)
key_label = str(context.get('key') or '').lower()
normalized_key_label = re.sub(r'[^a-z0-9]+', '_', key_label).strip('_')
structured_provider = None
structured_detector = ''
structured_pattern = None
if normalized_key_label in ('dashscope_api_key', 'qwen_api_key'):
structured_provider = 'qwen'
structured_detector = 'QwenDashScope'
structured_pattern = QWEN_KEY_RE
elif normalized_key_label in ('moonshot_api_key', 'kimi_api_key'):
structured_provider = 'kimi'
structured_detector = 'KimiMoonshot'
structured_pattern = KIMI_KEY_RE
elif normalized_key_label in (
'zai_api_key', 'z_ai_api_key', 'glm_api_key',
'zhipuai_api_key', 'bigmodel_api_key',
):
structured_provider = 'zai'
structured_detector = 'ZaiGLM'
structured_pattern = ZAI_KEY_RE
if structured_provider and structured_pattern.fullmatch(value):
identity = (structured_provider, value)
if identity not in seen:
seen.add(identity)
yield _make_candidate(
structured_provider, 'structured_postman', value, value,
secret_text=value, metadata={
'detector_name': structured_detector, 'origin': origin,
'structured_origin': f'{origin}:{path or index}',
},
)
foundry_endpoints = AZURE_FOUNDRY_ENDPOINT_RE.findall(text)
if (
foundry_endpoints and 20 <= len(value) <= 512
and not any(character.isspace() for character in value)
and any(token in key_label for token in ('key', 'token', 'secret', 'authorization'))
):
for endpoint in foundry_endpoints[:5]:
secret = f'{str(endpoint).lower()}:{value}'
identity = ('azure-foundry', secret)
if identity in seen:
continue
seen.add(identity)
yield _make_candidate(
'azure', 'structured_postman', secret, secret,
secret_text=value, endpoint=str(endpoint).lower(), metadata={
'detector_name': 'AzureFoundryEndpointBeforeKey', 'origin': origin,
'raw_v2': secret,
'structured_origin': f'{origin}:{path or index}',
},
)
def candidate_uid(scan_event_id, finding_uid_or_origin, service, credential_hash):
return hashlib.sha256('|'.join((
'truf-keycheck-candidate-v1', str(scan_event_id), str(finding_uid_or_origin),
str(service), str(credential_hash),
)).encode('utf-8')).hexdigest()
File diff suppressed because it is too large Load Diff
+1
View File
@@ -0,0 +1 @@
"""Keychecker package for the unified scanner layout."""
@@ -0,0 +1,274 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
import re
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
read_plain_keys,
recover_status_transaction,
request_error_message,
record_validation_result,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
SERVICE = "anthropic"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "anthropicChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "anthropicResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "anthropicAlive.txt"),
"NO_QUOTA": os.path.join(OUTPUT_DIR, "anthropicNoQuota.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "anthropicDead.txt"),
"LIMITED": os.path.join(OUTPUT_DIR, "anthropicLimited.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "anthropicRestricted.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "anthropicNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "anthropicUnknown.txt"),
}
ANTHROPIC_REGEX = re.compile(r"sk-ant-(?:api03|admin01)-[A-Za-z0-9\-_]{93}AA|sk-ant-[A-Za-z0-9\-_]{86}")
ANTHROPIC_ADMIN_PREFIX = "sk-ant-admin01-"
ANTHROPIC_ADMIN_API_KEYS_URL = "https://api.anthropic.com/v1/organizations/api_keys"
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def extract_candidates(input_file, plain_files):
seen_plain = set()
for item in iter_findings(input_file, ["Anthropic"]):
key = item["raw"]
if key and ANTHROPIC_REGEX.fullmatch(key):
yield key, item["source"], item["finding"]
for item in read_plain_keys(plain_files, ANTHROPIC_REGEX):
key = item["key"]
if key not in seen_plain:
seen_plain.add(key)
yield key, item["source"], {}
def tier_from_rpm(rpm):
mapping = {5: "Free Tier", 50: "Tier 1", 1000: "Tier 2", 2000: "Tier 3", 4000: "Tier 4"}
return mapping.get(rpm, "Scale/Unknown")
def anthropic_rate_headers(response):
output = {}
for name, value in response.headers.items():
lowered = name.lower()
if lowered.startswith("anthropic-ratelimit-"):
output[lowered.replace("anthropic-ratelimit-", "rate_").replace("-", "_")] = value
return output
def list_models(key, proxy, timeout):
headers = {
"anthropic-version": "2023-06-01",
"x-api-key": key,
}
try:
response = requests.get("https://api.anthropic.com/v1/models", headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"models_error": str(exc)[:500]}
if response.status_code != 200:
return {"models_status": response.status_code, "models_error": request_error_message(response)}
try:
data = response.json()
except ValueError:
return {"models_status": response.status_code, "models_error": "invalid JSON response"}
models = []
for item in data.get("data") or []:
if isinstance(item, dict) and item.get("id"):
models.append(item["id"])
return {
"models_status": response.status_code,
"models_count": len(models),
"models": models[:50],
}
def check_admin_key(key, proxy, timeout):
headers = {
"anthropic-version": "2023-06-01",
"x-api-key": key,
}
try:
response = requests.get(
ANTHROPIC_ADMIN_API_KEYS_URL,
headers=headers,
params={"limit": 1},
proxies=proxy,
timeout=timeout,
)
except requests.RequestException as exc:
return {"status": "NETWORK", "admin": True, "message": str(exc)[:500]}
if response.status_code == 200:
return {
"status": "VALID",
"admin": True,
"http_status": 200,
"message": "Anthropic Admin API access confirmed",
}
status = {
401: "DEAD",
403: "RESTRICTED",
429: "LIMITED",
}.get(response.status_code, "UNKNOWN")
message = request_error_message(response).replace(key, "***REDACTED***")
return {
"status": status,
"admin": True,
"http_status": response.status_code,
"message": message,
}
def check_key(key, proxy, timeout, model="claude-opus-4-6", include_models=False):
if key.startswith(ANTHROPIC_ADMIN_PREFIX):
return check_admin_key(key, proxy, timeout)
url = "https://api.anthropic.com/v1/messages"
headers = {
"content-type": "application/json",
"anthropic-version": "2023-06-01",
"x-api-key": key,
}
payload = {
"model": model,
"messages": [{"role": "user", "content": "ping"}],
"max_tokens": 1,
}
try:
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc)}
if response.status_code == 200:
rpm = 0
try:
rpm = int(response.headers.get("anthropic-ratelimit-requests-limit", "0"))
except ValueError:
rpm = 0
rate_data = anthropic_rate_headers(response)
result = {
"status": "VALID",
"model": model,
"rpm": rpm,
"tier": tier_from_rpm(rpm),
**rate_data,
}
if include_models:
result.update(list_models(key, proxy, timeout))
token_limit = rate_data.get("rate_tokens_limit") or ""
token_remaining = rate_data.get("rate_tokens_remaining") or ""
token_part = f" tokens={token_remaining}/{token_limit}" if token_limit or token_remaining else ""
result["message"] = f"model={model}; rpm={rpm}; tier={result['tier']}{token_part}"
return result
if response.status_code == 429:
return {"status": "LIMITED", "http_status": 429, "message": request_error_message(response)}
message = request_error_message(response)
lower = message.lower()
if "credit balance is too low" in lower or "usage limits" in lower:
return {"status": "NO_QUOTA", "http_status": response.status_code, "message": message}
if response.status_code in (401, 403):
status = "RESTRICTED" if "disabled" in lower or response.status_code == 403 else "DEAD"
return {"status": status, "http_status": response.status_code, "message": message}
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message}
def write_result(key, result, source, finding):
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "Anthropic")
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), source,
)
record_validation_result(SERVICE, key, result, source, finding, "Anthropic")
def parse_args():
parser = argparse.ArgumentParser(description="Anthropic key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=20)
parser.add_argument("--model", default=os.getenv("ANTHROPIC_CHECK_MODEL", "claude-opus-4-6"))
parser.add_argument("--list-models", action="store_true")
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-no-balance", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = set()
if args.retry_network:
retry_statuses.add("NETWORK")
if args.retry_limited:
retry_statuses.add("LIMITED")
if args.retry_unknown:
retry_statuses.add("UNKNOWN")
if args.retry_no_balance:
retry_statuses.add("NO_QUOTA")
processed = 0
skipped = 0
for key, source, finding in extract_candidates(args.input, args.plain):
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="Anthropic"):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] Anthropic candidate {mask_secret(key)} from {source}")
proxy = next(proxy_cycler) if proxy_cycler else None
result = check_key(key, proxy, args.timeout, args.model, args.list_models)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+607
View File
@@ -0,0 +1,607 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
load_checked_statuses,
load_known_keys,
load_known_statuses,
load_proxies,
mask_secret,
read_plain_keys,
recover_status_transaction,
record_cached_keycheck_occurrence,
record_validation_result,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
SERVICE = "aws"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "awsChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "awsResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "awsAlive.txt"),
"BEDROCK": os.path.join(OUTPUT_DIR, "awsBedrock.txt"),
"ADMIN": os.path.join(OUTPUT_DIR, "awsAdmin.txt"),
"CANARY": os.path.join(OUTPUT_DIR, "awsCanary.txt"),
"QUARANTINED": os.path.join(OUTPUT_DIR, "awsQuarantined.txt"),
"ACCESS_DENIED": os.path.join(OUTPUT_DIR, "awsAccessDenied.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "awsDead.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "awsNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "awsUnknown.txt"),
}
BEDROCK_REGIONS = ["us-east-1", "us-west-2", "eu-west-1", "eu-north-1", "ap-northeast-1", "ap-southeast-4"]
ANTHROPIC_MESSAGES_PROBE = {
"anthropic_version": "bedrock-2023-05-31",
"messages": [{"role": "user", "content": "ping"}],
"max_tokens": -1,
}
ANTHROPIC_MESSAGES_LIVE_PING = {
"anthropic_version": "bedrock-2023-05-31",
"messages": [{"role": "user", "content": "ping"}],
"max_tokens": 1,
}
BEDROCK_MODEL_TESTS = {
# Current Anthropic Bedrock runtime IDs. The default probe intentionally uses
# invalid max_tokens to validate auth/model access without generating tokens.
"anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE,
"us.anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE,
"global.anthropic.claude-fable-5": ANTHROPIC_MESSAGES_PROBE,
"anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE,
"us.anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE,
"global.anthropic.claude-sonnet-5": ANTHROPIC_MESSAGES_PROBE,
"anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE,
"us.anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE,
"global.anthropic.claude-opus-4-8": ANTHROPIC_MESSAGES_PROBE,
"anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE,
"us.anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE,
"global.anthropic.claude-opus-4-7": ANTHROPIC_MESSAGES_PROBE,
"anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE,
"us.anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE,
"global.anthropic.claude-sonnet-4-6": ANTHROPIC_MESSAGES_PROBE,
"anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE,
"us.anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE,
"global.anthropic.claude-haiku-4-5-20251001-v1:0": ANTHROPIC_MESSAGES_PROBE,
"anthropic.claude-3-5-sonnet-20241022-v2:0": ANTHROPIC_MESSAGES_PROBE,
"anthropic.claude-3-5-haiku-20241022-v1:0": ANTHROPIC_MESSAGES_PROBE,
"anthropic.claude-3-haiku-20240307-v1:0": ANTHROPIC_MESSAGES_PROBE,
"anthropic.claude-v2": {"prompt": "\n\nHuman:\n\nAssistant:", "max_tokens_to_sample": -1},
"anthropic.claude-instant-v1": {"prompt": "\n\nHuman:\n\nAssistant:", "max_tokens_to_sample": -1},
}
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def extract_candidates(input_file, plain_files):
seen_plain = set()
for item in iter_findings(input_file, ["AWS"]):
key = item["raw_v2"] or item["raw"]
if key and ":" in key:
yield key, item["source"], item["finding"]
import re
regex = re.compile(r"AKIA[0-9A-Z]{16}:[A-Za-z0-9+/]{40}")
for item in read_plain_keys(plain_files, regex):
key = item["key"]
if key not in seen_plain:
seen_plain.add(key)
yield key, item["source"], {}
def is_dead_aws_error(code):
return code in {"InvalidClientTokenId", "SignatureDoesNotMatch", "AuthFailure", "UnrecognizedClientException"}
def is_canary_text(value):
value = str(value or "").lower()
return "canarytokens" in value or "canary token" in value or "is_canary" in value
def is_canary_finding(finding):
if not isinstance(finding, dict):
return False
extra = finding.get("ExtraData") or {}
if isinstance(extra, dict):
if str(extra.get("is_canary", "")).lower() == "true":
return True
if any(is_canary_text(value) for value in extra.values()):
return True
return is_canary_text(finding.get("Raw")) or is_canary_text(finding.get("RawV2"))
def is_canary_arn(arn):
return is_canary_text(arn)
def aws_client(session, service, proxy=None, region_name=None, timeout=20):
kwargs = {}
if region_name:
kwargs["region_name"] = region_name
from botocore.config import Config
kwargs["config"] = Config(
proxies=proxy or None,
connect_timeout=timeout,
read_timeout=timeout,
retries={"max_attempts": 1},
)
return session.client(service, **kwargs)
def bedrock_validation_allows_invoke(exc):
text = str(exc or "").lower()
if any(item in text for item in ("operation not allowed", "not authorized", "access denied")):
return False
# The default probe sends deliberately invalid token limits. If Bedrock only
# rejects the payload shape after auth, InvokeModel reached the model path.
return any(item in text for item in ("max_tokens", "max_tokens_to_sample", "malformed input", "schema"))
def client_error_code(exc):
try:
return exc.response.get("Error", {}).get("Code", "ClientError")
except Exception:
return "ClientError"
def client_error_message(exc):
try:
return exc.response.get("Error", {}).get("Message", str(exc))
except Exception:
return str(exc)
def model_arn(region, model_id):
# Cross-region inference profile IDs are not foundation-model ARNs.
if model_id.startswith(("us.", "eu.", "jp.", "au.", "global.")):
return "*"
return f"arn:aws:bedrock:{region}::foundation-model/{model_id}"
def iam_policy_source_arn(sts_arn, account):
arn = str(sts_arn or "")
if ":assumed-role/" in arn:
role_part = arn.split(":assumed-role/", 1)[1].split("/", 1)[0]
return f"arn:aws:iam::{account}:role/{role_part}"
return arn if ":iam::" in arn else ""
def simulate_bedrock_activation(session, arn, account, region, model_id, proxy=None, timeout=20):
import botocore.exceptions
source_arn = iam_policy_source_arn(arn, account)
if not source_arn:
return {"status": "not_available", "message": "unsupported principal arn for IAM simulation"}
actions = [
"bedrock:GetFoundationModelAvailability",
"bedrock:ListFoundationModelAgreementOffers",
"bedrock:GetUseCaseForModelAccess",
"bedrock:PutUseCaseForModelAccess",
"bedrock:CreateFoundationModelAgreement",
"bedrock:GetInferenceProfile",
"bedrock:InvokeModel",
]
try:
iam = aws_client(session, "iam", proxy, timeout=timeout)
response = iam.simulate_principal_policy(
PolicySourceArn=source_arn,
ActionNames=actions,
ResourceArns=[model_arn(region, model_id)],
)
except botocore.exceptions.ClientError as exc:
return {
"status": "access_denied" if client_error_code(exc) == "AccessDenied" else "error",
"code": client_error_code(exc),
"message": client_error_message(exc)[:500],
}
decisions = {}
for item in response.get("EvaluationResults", []):
action = str(item.get("EvalActionName") or "")
decisions[action] = str(item.get("EvalDecision") or "")
activation_actions = ["bedrock:PutUseCaseForModelAccess", "bedrock:CreateFoundationModelAgreement"]
can_activate = all(decisions.get(action) == "allowed" for action in activation_actions)
return {"status": "ok", "source_arn": source_arn, "can_activate": can_activate, "decisions": decisions}
def check_bedrock_management(session, arn, account, proxy=None, timeout=20, regions=None, models=None, max_attempts=12, debug=False):
import botocore.exceptions
attempts = []
findings = []
tried = 0
for region in (regions or BEDROCK_REGIONS):
bedrock = aws_client(session, "bedrock", proxy, region, timeout)
use_case = None
try:
use_case = bedrock.get_use_case_for_model_access()
except botocore.exceptions.ClientError as exc:
use_case = {"error_code": client_error_code(exc), "message": client_error_message(exc)[:300]}
try:
profiles = bedrock.list_inference_profiles(typeEquals="SYSTEM_DEFINED", maxResults=20).get("inferenceProfileSummaries", [])
except botocore.exceptions.ClientError as exc:
profiles = {"error_code": client_error_code(exc), "message": client_error_message(exc)[:300]}
for model_id in (models or list(BEDROCK_MODEL_TESTS.keys())):
if max_attempts and tried >= max_attempts:
return {"enabled": bool(findings), "findings": findings, "message": "; ".join(attempts[:10])}
tried += 1
if debug:
print(f" BEDROCK MGMT TRY: region={region}, model={model_id}")
item = {"region": region, "model": model_id, "use_case": use_case, "profiles": profiles}
try:
item["foundation_model"] = bedrock.get_foundation_model(modelIdentifier=model_id).get("modelDetails", {})
except botocore.exceptions.ClientError as exc:
item["foundation_model_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]}
try:
item["availability"] = bedrock.get_foundation_model_availability(modelId=model_id)
except botocore.exceptions.ClientError as exc:
item["availability_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]}
try:
item["agreement_offers"] = bedrock.list_foundation_model_agreement_offers(modelId=model_id, offerType="ALL")
except botocore.exceptions.ClientError as exc:
item["agreement_offers_error"] = {"code": client_error_code(exc), "message": client_error_message(exc)[:300]}
item["iam_simulation"] = simulate_bedrock_activation(session, arn, account, region, model_id, proxy, timeout)
availability = item.get("availability") or {}
simulation = item.get("iam_simulation") or {}
can_activate = bool(simulation.get("can_activate"))
authorized = str(availability.get("authorizationStatus") or "").lower() in ("authorized", "available")
if can_activate or authorized:
findings.append(item)
else:
code = (item.get("availability_error") or item.get("foundation_model_error") or {}).get("code") or "checked"
attempts.append(f"{region}:{model_id}:can_activate={can_activate}:authorization={availability.get('authorizationStatus') or code}")
return {"enabled": bool(findings), "findings": findings, "message": "; ".join(attempts[:10])}
def check_bedrock(session, proxy=None, timeout=20, debug=False, regions=None, models=None, max_attempts=12, live_invoke=False):
import json
import botocore.exceptions
attempts = []
accepted = []
tried = 0
model_ids = models or list(BEDROCK_MODEL_TESTS.keys())
for region in (regions or BEDROCK_REGIONS):
for model_id in model_ids:
if max_attempts and tried >= max_attempts:
if accepted:
first = accepted[0]
return {
"enabled": True,
"region": first.get("region", ""),
"model": first.get("model", ""),
"available_models": [f"{item['region']}/{item['model']}" for item in accepted],
"message": "Bedrock InvokeModel accepted",
}
return {"enabled": False, "region": "", "model": "", "available_models": [], "message": "; ".join(attempts[:10]) or "Bedrock probe attempt limit reached"}
tried += 1
data = BEDROCK_MODEL_TESTS.get(model_id)
if data is None:
data = ANTHROPIC_MESSAGES_PROBE
if live_invoke and data is ANTHROPIC_MESSAGES_PROBE:
data = ANTHROPIC_MESSAGES_LIVE_PING
client = aws_client(session, "bedrock-runtime", proxy, region, timeout)
if debug:
print(f" BEDROCK TRY: region={region}, model={model_id}")
try:
client.invoke_model(body=json.dumps(data), modelId=model_id)
if debug:
print(" BEDROCK RESULT: invoke_model succeeded")
accepted.append({"region": region, "model": model_id, "message": "invoke_model succeeded"})
continue
except client.exceptions.ValidationException as exc:
message = str(exc)
if bedrock_validation_allows_invoke(exc):
# ValidationException for the intentional bad payload means auth/model access passed.
if debug:
print(f" BEDROCK RESULT: validation_exception_after_auth: {message[:200]}")
accepted.append({"region": region, "model": model_id, "message": message[:300]})
else:
if debug:
print(f" BEDROCK RESULT: validation_rejected: {message[:200]}")
attempts.append(f"{region}:{model_id}:validation:{message[:120]}")
continue
except client.exceptions.AccessDeniedException:
if debug:
print(" BEDROCK RESULT: access_denied")
attempts.append(f"{region}:{model_id}:access_denied")
continue
except client.exceptions.ResourceNotFoundException:
if debug:
print(" BEDROCK RESULT: model_not_found")
attempts.append(f"{region}:{model_id}:not_found")
continue
except botocore.exceptions.EndpointConnectionError as exc:
if debug:
print(f" BEDROCK RESULT: network_error: {str(exc)[:120]}")
attempts.append(f"{region}:{model_id}:network:{str(exc)[:80]}")
continue
except botocore.exceptions.ClientError as exc:
code = exc.response.get("Error", {}).get("Code", "ClientError")
if debug:
print(f" BEDROCK RESULT: {code}: {str(exc)[:160]}")
attempts.append(f"{region}:{model_id}:{code}")
continue
if accepted:
first = accepted[0]
return {
"enabled": True,
"region": first.get("region", ""),
"model": first.get("model", ""),
"available_models": [f"{item['region']}/{item['model']}" for item in accepted],
"message": "Bedrock InvokeModel accepted",
}
return {"enabled": False, "region": "", "model": "", "available_models": [], "message": "; ".join(attempts[:10])}
def inspect_iam(session, arn, proxy=None, timeout=20):
import botocore.exceptions
output = {"admin": False, "quarantined": False, "policy_check": "not_checked", "message": ""}
if ":user/" not in arn:
output["policy_check"] = "not_user_arn"
return output
username = arn.rsplit("/", 1)[1]
try:
iam = aws_client(session, "iam", proxy, timeout=timeout)
policies = iam.list_attached_user_policies(UserName=username).get("AttachedPolicies", [])
except botocore.exceptions.ClientError as exc:
code = exc.response.get("Error", {}).get("Code", "")
output["policy_check"] = "access_denied" if code == "AccessDenied" else "error"
output["message"] = str(exc)
return output
output["policy_check"] = "ok"
for policy in policies:
name = policy.get("PolicyName", "")
if "AWSCompromisedKeyQuarantine" in name:
output["quarantined"] = True
if name == "AdministratorAccess":
output["admin"] = True
return output
def check_key(
key,
probe_bedrock=False,
bedrock_debug=False,
proxy=None,
timeout=20,
bedrock_regions=None,
bedrock_models=None,
bedrock_max_attempts=12,
bedrock_live_invoke=False,
probe_bedrock_management=False,
):
try:
import boto3
import botocore.exceptions
except ImportError as exc:
return {"status": "UNKNOWN", "message": f"boto3/botocore missing: {exc}"}
access_key, secret = key.split(":", 1)
session = boto3.Session(aws_access_key_id=access_key, aws_secret_access_key=secret)
try:
identity = aws_client(session, "sts", proxy, timeout=timeout).get_caller_identity()
except botocore.exceptions.EndpointConnectionError as exc:
return {"status": "NETWORK", "message": str(exc)}
except botocore.exceptions.ClientError as exc:
code = exc.response.get("Error", {}).get("Code", "")
status = "DEAD" if is_dead_aws_error(code) else "ACCESS_DENIED"
return {"status": status, "code": code, "message": str(exc)}
except Exception as exc:
return {"status": "UNKNOWN", "message": str(exc)}
arn = identity.get("Arn", "")
if is_canary_arn(arn):
return {
"status": "CANARY",
"account": identity.get("Account", ""),
"arn": arn,
"admin": False,
"quarantined": False,
"iam_policy_check": "skipped_canary",
"bedrock_enabled": False,
"bedrock_region": "",
"bedrock_model": "",
"bedrock_message": "skipped_canary",
"bedrock_management_enabled": False,
"bedrock_management_message": "skipped_canary",
"message": "canary credential detected from STS arn; skipped IAM/Bedrock probes",
}
iam_info = inspect_iam(session, arn, proxy, timeout)
bedrock_info = {"enabled": False, "region": "", "model": "", "message": "not_checked"}
bedrock_management_info = {"enabled": False, "findings": [], "message": "not_checked"}
if probe_bedrock:
bedrock_info = check_bedrock(session, proxy, timeout, bedrock_debug, bedrock_regions, bedrock_models, bedrock_max_attempts, bedrock_live_invoke)
if probe_bedrock_management:
bedrock_management_info = check_bedrock_management(
session,
arn,
identity.get("Account", ""),
proxy,
timeout,
bedrock_regions,
bedrock_models,
bedrock_max_attempts,
bedrock_debug,
)
if iam_info.get("quarantined"):
status = "QUARANTINED"
elif bedrock_info.get("enabled"):
status = "BEDROCK"
else:
status = "ADMIN" if iam_info.get("admin") else "VALID"
message_parts = [
"sts_ok",
f"iam_policy_check={iam_info.get('policy_check')}",
]
if probe_bedrock:
message_parts.append(f"bedrock_enabled={bedrock_info.get('enabled')}")
if bedrock_info.get("region"):
message_parts.append(f"bedrock_region={bedrock_info.get('region')}")
if bedrock_info.get("model"):
message_parts.append(f"bedrock_model={bedrock_info.get('model')}")
if probe_bedrock_management:
findings = bedrock_management_info.get("findings") or []
can_activate = any((item.get("iam_simulation") or {}).get("can_activate") for item in findings)
message_parts.append(f"bedrock_mgmt_enabled={bedrock_management_info.get('enabled')}")
message_parts.append(f"bedrock_can_activate={can_activate}")
if iam_info.get("message") and iam_info.get("policy_check") != "access_denied":
message_parts.append(iam_info.get("message")[:300])
return {
"status": status,
"account": identity.get("Account", ""),
"arn": arn,
"admin": iam_info.get("admin", False),
"quarantined": iam_info.get("quarantined", False),
"iam_policy_check": iam_info.get("policy_check"),
"bedrock_enabled": bedrock_info.get("enabled"),
"bedrock_region": bedrock_info.get("region"),
"bedrock_model": bedrock_info.get("model"),
"bedrock_available_models": bedrock_info.get("available_models") or [],
"bedrock_message": bedrock_info.get("message"),
"bedrock_management_enabled": bedrock_management_info.get("enabled"),
"bedrock_management_findings": bedrock_management_info.get("findings") or [],
"bedrock_management_message": bedrock_management_info.get("message"),
"message": "; ".join(message_parts),
}
def write_result(key, result, source, finding):
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "AWS")
message = result.get("message", "")
if result.get("status") == "BEDROCK":
models = result.get("bedrock_available_models") or []
model_text = ",".join(str(item) for item in models) or f"{result.get('bedrock_region', '')}/{result.get('bedrock_model', '')}".strip("/")
message = f"{message}; models={model_text}"
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, result["status"], message, result.get("arn", source),
)
record_validation_result(SERVICE, key, result, source, finding, "AWS")
def parse_args():
parser = argparse.ArgumentParser(description="AWS key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--timeout", type=int, default=20)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--probe-bedrock", action="store_true")
parser.add_argument("--bedrock-debug", action="store_true")
parser.add_argument("--bedrock-regions", default=",".join(BEDROCK_REGIONS))
parser.add_argument("--bedrock-models", default=",".join(BEDROCK_MODEL_TESTS))
parser.add_argument("--bedrock-max-attempts", type=int, default=12)
parser.add_argument("--bedrock-live-invoke", action="store_true")
parser.add_argument("--probe-bedrock-management", action="store_true")
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
known = set(known_statuses)
retry_statuses = set()
if args.retry_network:
retry_statuses.add("NETWORK")
if args.retry_unknown:
retry_statuses.add("UNKNOWN")
if args.retry_valid:
retry_statuses.update({"VALID", "BEDROCK", "ADMIN"})
processed = 0
skipped = 0
bedrock_regions = [item.strip() for item in str(args.bedrock_regions or "").split(",") if item.strip()]
bedrock_models = [item.strip() for item in str(args.bedrock_models or "").split(",") if item.strip()]
for key, source, finding in extract_candidates(args.input, args.plain):
if should_skip_key(
key, checked, known, args, retry_statuses,
service=SERVICE, source=source, finding=finding, detector="AWS", known_statuses=known_statuses,
):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] AWS candidate {mask_secret(key)} from {source}")
if is_canary_finding(finding):
result = {
"status": "CANARY",
"message": "canary credential detected in TruffleHog ExtraData; skipped AWS API probes",
}
else:
proxy = next(proxy_cycler) if proxy_cycler else None
result = check_key(
key,
args.probe_bedrock,
args.bedrock_debug,
proxy,
args.timeout,
bedrock_regions,
bedrock_models,
args.bedrock_max_attempts,
args.bedrock_live_invoke,
args.probe_bedrock_management,
)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
if args.probe_bedrock:
print(
" BEDROCK PING: "
f"enabled={result.get('bedrock_enabled')}, "
f"region={result.get('bedrock_region') or '-'}, "
f"model={result.get('bedrock_model') or '-'}"
)
if result.get('bedrock_message'):
print(f" BEDROCK RESPONSE: {str(result.get('bedrock_message'))[:300]}")
if args.probe_bedrock_management:
findings = result.get("bedrock_management_findings") or []
can_activate = any((item.get("iam_simulation") or {}).get("can_activate") for item in findings)
print(
" BEDROCK MGMT: "
f"enabled={result.get('bedrock_management_enabled')}, "
f"can_activate={can_activate}, "
f"findings={len(findings)}"
)
if result.get("bedrock_management_message"):
print(f" BEDROCK MGMT RESPONSE: {str(result.get('bedrock_management_message'))[:300]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+946
View File
@@ -0,0 +1,946 @@
import sys
sys.dont_write_bytecode = True
import argparse
import json
import os
import re
from urllib.parse import urlsplit
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
sys.path.append(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
from keycheck_candidates import extract_azure_foundry_parts
from keycheck_common import (
append_jsonl,
append_status,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
keycheck_input_mode,
iter_findings,
iter_bounded_text_lines,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
recover_status_transaction,
request_error_message,
record_validation_result,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
SERVICE = "azure"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "azureChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "azureResults.jsonl")
AZURE_OPENAI_LLM_FILE = os.path.join(OUTPUT_DIR, "azureOpenAILLM.txt")
AZURE_OPENAI_PLAIN_FILE = os.path.join(OUTPUT_DIR, "azureOpenAI.txt")
AZURE_FOUNDRY_PLAIN_FILE = os.path.join(OUTPUT_DIR, "azureFoundry.txt")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "azureAlive.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "azureDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "azureRestricted.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "azureNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "azureUnknown.txt"),
"OPENAI_UNRESOLVED": os.path.join(OUTPUT_DIR, "azureOpenAIUnresolved.txt"),
"OPENAI_BAD_ENDPOINT": os.path.join(OUTPUT_DIR, "azureOpenAIBadEndpoint.txt"),
"FOUNDRY": os.path.join(OUTPUT_DIR, "azureFoundryLLM.txt"),
"FOUNDRY_UNRESOLVED": os.path.join(OUTPUT_DIR, "azureFoundryUnresolved.txt"),
"FOUNDRY_BAD_ENDPOINT": os.path.join(OUTPUT_DIR, "azureFoundryBadEndpoint.txt"),
}
AZURE_OPENAI_ENDPOINT_RE = re.compile(r"([a-z0-9-]+\.openai\.azure\.com)", re.IGNORECASE)
AZURE_FOUNDRY_HOST_RE = r"[a-z0-9-]+(?:\.[a-z0-9-]+)*\.(?:models\.ai\.azure\.com|services\.ai\.azure\.com|inference\.ai\.azure\.com)"
AZURE_FOUNDRY_ENDPOINT_RE = re.compile(r"((?:https?://)?" + AZURE_FOUNDRY_HOST_RE + r"(?:/[^\s:\"'<>\\]*)?)", re.IGNORECASE)
AZURE_OPENAI_DEPLOYMENTS_API_VERSION = "2023-03-15-preview"
AZURE_OPENAI_CHAT_API_VERSION = "2024-02-15-preview"
AZURE_FOUNDRY_API_VERSION = "2024-05-01-preview"
AZURE_FOUNDRY_KEY_ASSIGNMENT_RE = re.compile(
r"(?is)(?:authorization|api[_-]?key|key|token|secret|credential|bearer)[^\n:=]{0,80}[:=]\s*[\"']?(?:bearer\s+)?([A-Za-z0-9_./+=\-]{20,512})"
)
NON_FOUNDRY_KEY_PREFIXES = (
"sk-", "sk_", "sk-or-", "xai-", "ghp_", "gho_", "ghu_", "ghs_", "ghr_", "github_pat_",
"glpat-", "glrt-", "hf_", "AIza", "AQ.", "zai-", "gsk_", "r8_", "nvapi-",
)
def transaction_status_files():
return {**STATUS_FILES, "AUX_OPENAI_LLM": AZURE_OPENAI_LLM_FILE}
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, AZURE_OPENAI_LLM_FILE, AZURE_OPENAI_PLAIN_FILE, AZURE_FOUNDRY_PLAIN_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, transaction_status_files())
def parse_azure_sp(raw_v2):
try:
data = json.loads(raw_v2)
except (TypeError, ValueError):
return None
client_secret = data.get("clientSecret") or data.get("client_secret")
client_id = data.get("clientId") or data.get("client_id")
tenant_id = data.get("tenantId") or data.get("tenant_id")
if not all([client_secret, client_id, tenant_id]):
return None
return {"client_secret": client_secret, "client_id": client_id, "tenant_id": tenant_id}
def scanner_context_text(finding):
context = finding.get("ScannerContext") if isinstance(finding, dict) else None
if isinstance(context, dict):
return str(context.get("nearby") or "")
return ""
def foundry_keyish(value):
text = re.sub(r"(?i)^bearer\s+", "", str(value or "").strip().strip('"\'`,;')).strip()
lower = text.lower()
if not (20 <= len(text) <= 512):
return False
if any(marker in lower for marker in ("http://", "https://", "{{", "${", "<", "azure.com")):
return False
if any(ch.isspace() for ch in text):
return False
if re.match(r"(?i)^(?:authorization|api[_-]?key|key|token|secret|credential|bearer)\s*[:=]", text):
return False
if text.startswith(NON_FOUNDRY_KEY_PREFIXES):
return False
return bool(re.search(r"[A-Za-z]", text) and re.search(r"[0-9]", text))
def normalize_foundry_endpoint(value):
text = str(value or "").strip().strip('"\'`,;')
if not text:
return ""
split_text = text if re.match(r"(?i)^https?://", text) else "https://" + text
try:
parsed = urlsplit(split_text)
host = parsed.netloc or parsed.path.split("/", 1)[0]
path = parsed.path if parsed.netloc else ("/" + parsed.path.split("/", 1)[1] if "/" in parsed.path else "")
except Exception:
host, path = re.sub(r"(?i)^https?://", "", text).split("/", 1)[0], ""
path = path.rstrip(".,;:)]}/")
terminal_routes = (
("/models/chat/completions", ""),
("/openai/v1/chat/completions", "/openai/v1"),
("/v1/chat/completions", "/v1"),
("/chat/completions", ""),
("/v1/models", "/v1"),
("/models", ""),
)
lower_path = path.lower()
for suffix, replacement in terminal_routes:
if lower_path.endswith(suffix):
path = path[:-len(suffix)] + replacement
break
return (host + path).strip("/").lower()
def split_foundry_endpoint_key(text):
candidate = str(text or "").strip().split("\t", 1)[0].strip()
if not candidate:
return None
endpoint_match = AZURE_FOUNDRY_ENDPOINT_RE.search(candidate)
if not endpoint_match:
return None
endpoint = normalize_foundry_endpoint(endpoint_match.group(1))
before = candidate[:endpoint_match.start()].replace("https://", " ").replace("http://", " ").strip(" \t:=,;'\"/")
after = candidate[endpoint_match.end():].replace("https://", " ").replace("http://", " ").strip(" \t:=,;'\"/")
for key in (after, before):
if foundry_keyish(key):
return {"key": key, "endpoint": endpoint}
return None
def foundry_context_values(finding):
values = []
for value in (finding.get("Raw"), finding.get("RawV2")) if isinstance(finding, dict) else ():
if value:
text = str(value)
values.append(text)
values.extend(str(item) for item in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(text))
context = scanner_context_text(finding)
if context:
values.append(context)
values.extend(str(item) for item in AZURE_FOUNDRY_KEY_ASSIGNMENT_RE.findall(context or ""))
extra = finding.get("ExtraData") if isinstance(finding, dict) else None
if isinstance(extra, dict):
values.extend(str(value) for value in extra.values() if isinstance(value, str))
return values
def parse_azure_openai(raw, raw_v2, finding):
key = raw or ""
endpoint = ""
raw_v2 = raw_v2 or ""
match = re.match(r"^([a-f0-9]{32}):(.+\.openai\.azure\.com)$", raw_v2, re.IGNORECASE)
if match:
key = match.group(1)
endpoint = match.group(2)
if not endpoint:
context_match = AZURE_OPENAI_ENDPOINT_RE.search(scanner_context_text(finding))
if context_match:
endpoint = context_match.group(1)
if not key:
return None
return {"key": key, "endpoint": endpoint}
def parse_azure_openai_line(line):
text = str(line or "").strip()
if not text:
return None
text = text.split("\t", 1)[0].strip()
if ":" in text:
endpoint, key = text.split(":", 1)
if AZURE_OPENAI_ENDPOINT_RE.fullmatch(endpoint.strip()) and key.strip():
return {"endpoint": endpoint.strip(), "key": key.strip()}
endpoint_match = AZURE_OPENAI_ENDPOINT_RE.search(text)
key_match = re.search(r"\b[a-f0-9]{32}\b", text, re.IGNORECASE)
if endpoint_match and key_match:
return {"endpoint": endpoint_match.group(1), "key": key_match.group(0)}
if key_match:
return {"endpoint": "", "key": key_match.group(0)}
return None
def parse_azure_foundry(raw, raw_v2, finding):
key = raw or ""
endpoint = ""
raw_v2 = raw_v2 or ""
split = split_foundry_endpoint_key(raw_v2) or split_foundry_endpoint_key(raw)
if not split:
split = extract_azure_foundry_parts(raw, raw_v2)
if split:
key = split["key"]
endpoint = split["endpoint"]
if not endpoint:
for endpoint_text in (raw, raw_v2, scanner_context_text(finding)):
context_match = AZURE_FOUNDRY_ENDPOINT_RE.search(str(endpoint_text or ""))
if context_match:
endpoint = normalize_foundry_endpoint(context_match.group(1))
break
if not foundry_keyish(key):
for value in foundry_context_values(finding):
if foundry_keyish(value):
key = value.strip().strip('"\'`,;')
break
if not key or not foundry_keyish(key):
return None
return {"key": key.strip().strip('"\'`,;'), "endpoint": normalize_foundry_endpoint(endpoint)}
def parse_azure_foundry_line(line):
parts = str(line or "").strip().split("\t", 1)
text = parts[0].strip()
if not text:
return None
split = split_foundry_endpoint_key(text)
parsed = split if split else ({"key": text, "endpoint": ""} if foundry_keyish(text) else None)
if not parsed:
return None
if len(parts) > 1:
try:
metadata = json.loads(parts[1])
except ValueError:
metadata = {}
if isinstance(metadata, dict):
parsed["finding_uid"] = metadata.get("finding_uid") or ""
parsed["origin"] = metadata.get("origin") or ""
return parsed
def parse_azure_acr(raw_v2):
try:
data = json.loads(raw_v2)
except (TypeError, ValueError):
return None
username = data.get("username")
password = data.get("password")
if not username or not password:
return None
return {"username": username, "password": password}
def azure_openai_key(parsed):
endpoint = parsed.get("endpoint") or ""
return f"{endpoint}:{parsed['key']}" if endpoint else parsed["key"]
def azure_acr_key(parsed):
return f"{parsed['username']}:{parsed['password']}"
def azure_sp_key(parsed):
return f"{parsed['tenant_id']}:{parsed['client_id']}:{parsed['client_secret']}"
def azure_foundry_key(parsed):
endpoint = parsed.get("endpoint") or ""
return f"{endpoint}:{parsed['key']}" if endpoint else parsed["key"]
def extract_candidates(input_file):
seen_plain = set()
foundry_detectors = {"AzureFoundryEndpointBeforeKey", "AzureFoundryKeyBeforeEndpoint"}
detector_names = ["AzureOpenAI", "AzureContainerRegistry", "Azure", *sorted(foundry_detectors)]
for item in iter_findings(input_file, detector_names):
candidate_kind = item.get('candidate_kind') or ''
if keycheck_input_mode() == 'postgres' and candidate_kind:
secret_text = item.get('credential_secret_text') or ''
secret_json = item.get('credential_secret_json') or ''
endpoint = item.get('credential_endpoint') or ''
parsed = None
detector = ''
if candidate_kind == 'azure_service_principal':
parsed = parse_azure_sp(secret_json)
detector = 'Azure'
key = azure_sp_key(parsed) if parsed else ''
elif candidate_kind == 'azure_container_registry':
parsed = parse_azure_acr(secret_json)
detector = 'AzureContainerRegistry'
key = azure_acr_key(parsed) if parsed else ''
elif candidate_kind == 'azure_openai':
parsed = {'key': secret_text, 'endpoint': endpoint} if secret_text else None
detector = 'AzureOpenAI'
key = azure_openai_key(parsed) if parsed else ''
elif candidate_kind == 'azure_foundry':
parsed = (
{'key': secret_text, 'endpoint': normalize_foundry_endpoint(endpoint)}
if foundry_keyish(secret_text) and endpoint else
parse_azure_foundry(secret_text, '', item.get('finding') or {})
)
if parsed and endpoint:
parsed['endpoint'] = normalize_foundry_endpoint(endpoint)
detector = 'AzureFoundry'
key = azure_foundry_key(parsed) if parsed else ''
else:
key = ''
if parsed and key:
yield key, detector, item['source'], item['finding'], parsed
continue
unresolved_detector = {
'azure_openai': 'AzureOpenAI',
'azure_foundry': 'AzureFoundry',
'azure_container_registry': 'AzureContainerRegistry',
'azure_service_principal': 'Azure',
}.get(candidate_kind, 'Azure')
unresolved_key = secret_text or secret_json or candidate_kind
if unresolved_key:
yield unresolved_key, unresolved_detector, item['source'], item['finding'], {
'_unresolved_candidate': True,
'candidate_kind': candidate_kind,
}
continue
if item["detector"] == "AzureOpenAI":
foundry = parse_azure_foundry(item["raw"], item["raw_v2"], item["finding"])
if foundry and foundry.get("endpoint"):
key = azure_foundry_key(foundry)
yield key, "AzureFoundry", item["source"], item["finding"], foundry
parsed = parse_azure_openai(item["raw"], item["raw_v2"], item["finding"])
if not parsed:
continue
key = azure_openai_key(parsed)
yield key, "AzureOpenAI", item["source"], item["finding"], parsed
continue
if item["detector"] == "AzureContainerRegistry":
parsed = parse_azure_acr(item["raw_v2"])
if not parsed:
continue
key = azure_acr_key(parsed)
yield key, "AzureContainerRegistry", item["source"], item["finding"], parsed
continue
if item["detector"] == "Azure":
parsed = parse_azure_sp(item["raw_v2"])
if not parsed:
continue
key = azure_sp_key(parsed)
yield key, "Azure", item["source"], item["finding"], parsed
continue
if item["detector"] in foundry_detectors:
foundry = parse_azure_foundry(item.get("raw"), item.get("raw_v2"), item.get("finding"))
if not foundry or not foundry.get("endpoint"):
continue
key = azure_foundry_key(foundry)
yield key, "AzureFoundry", item["source"], item["finding"], foundry
if keycheck_input_mode() == 'jsonl' and os.path.exists(AZURE_FOUNDRY_PLAIN_FILE):
for line_num, line in enumerate(iter_bounded_text_lines(AZURE_FOUNDRY_PLAIN_FILE), 1):
parsed = parse_azure_foundry_line(line)
if not parsed:
continue
key = azure_foundry_key(parsed)
if key not in seen_plain:
seen_plain.add(key)
yield key, "AzureFoundry", f"{AZURE_FOUNDRY_PLAIN_FILE}:{line_num}", {}, parsed
if keycheck_input_mode() == 'jsonl' and os.path.exists(AZURE_OPENAI_PLAIN_FILE):
for line_num, line in enumerate(iter_bounded_text_lines(AZURE_OPENAI_PLAIN_FILE), 1):
parsed = parse_azure_openai_line(line)
if not parsed:
continue
key = azure_openai_key(parsed)
if key not in seen_plain:
seen_plain.add(key)
yield key, "AzureOpenAI", f"{AZURE_OPENAI_PLAIN_FILE}:{line_num}", {}, parsed
def permission_matches(action, pattern):
action = str(action or "").lower()
pattern = str(pattern or "").lower()
if pattern == "*":
return True
if pattern.endswith("/*"):
return action.startswith(pattern[:-1])
return action == pattern
def has_action(actions, wanted):
return any(permission_matches(wanted, action) for action in actions)
def probe_azure_rbac(access_token, proxy, timeout, max_subscriptions=3):
if not access_token:
return {"azure_rbac_level": "unknown", "message": "rbac_probe=no_access_token"}
headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"}
try:
response = requests.get(
"https://management.azure.com/subscriptions?api-version=2020-01-01",
headers=headers,
proxies=proxy,
timeout=timeout,
)
except requests.RequestException as exc:
return {"azure_rbac_level": "network", "message": f"rbac_probe_network={str(exc)[:200]}"}
if response.status_code == 403:
return {"azure_rbac_level": "token_only", "azure_subscription_count": 0, "message": "rbac_probe=subscriptions_forbidden"}
if response.status_code >= 400:
return {"azure_rbac_level": "unknown", "azure_rbac_http_status": response.status_code, "message": f"rbac_probe_http={response.status_code}:{request_error_message(response)[:200]}"}
payload = response.json()
subscriptions = payload.get("value") if isinstance(payload, dict) else []
subscriptions = subscriptions or []
sub_ids = [item.get("subscriptionId") for item in subscriptions if isinstance(item, dict) and item.get("subscriptionId")]
if not sub_ids:
return {"azure_rbac_level": "token_only", "azure_subscription_count": 0, "message": "rbac_probe=no_subscriptions"}
all_actions = set()
all_not_actions = set()
permission_errors = []
for sub_id in sub_ids[:max_subscriptions]:
url = f"https://management.azure.com/subscriptions/{sub_id}/providers/Microsoft.Authorization/permissions?api-version=2022-04-01"
try:
perms_response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
permission_errors.append(f"{sub_id}:network:{str(exc)[:120]}")
continue
if perms_response.status_code >= 400:
permission_errors.append(f"{sub_id}:http_{perms_response.status_code}:{request_error_message(perms_response)[:120]}")
continue
data = perms_response.json().get("value") or []
for item in data:
for action in item.get("actions") or []:
all_actions.add(str(action))
for action in item.get("notActions") or []:
all_not_actions.add(str(action))
can_all = has_action(all_actions, "*")
can_assign_roles = has_action(all_actions, "Microsoft.Authorization/roleAssignments/write") and not has_action(all_not_actions, "Microsoft.Authorization/roleAssignments/write")
can_manage_cognitive = any(
has_action(all_actions, action) for action in (
"Microsoft.CognitiveServices/accounts/write",
"Microsoft.CognitiveServices/accounts/deployments/write",
"Microsoft.CognitiveServices/*",
)
) or can_all
can_manage_ml = any(
has_action(all_actions, action) for action in (
"Microsoft.MachineLearningServices/workspaces/write",
"Microsoft.MachineLearningServices/*",
)
) or can_all
can_deploy_resources = has_action(all_actions, "Microsoft.Resources/deployments/write") or can_all
can_manage_ai = can_manage_cognitive or can_manage_ml
if can_all and can_assign_roles:
level = "owner_like"
elif can_all:
level = "contributor_like"
elif can_manage_ai:
level = "ai_manager"
elif all_actions:
level = "limited"
else:
level = "subscriptions_visible"
message = (
f"rbac_probe={level}; subscriptions={len(sub_ids)}; "
f"can_manage_ai={can_manage_ai}; can_assign_roles={can_assign_roles}; can_deploy_resources={can_deploy_resources}"
)
if permission_errors and not all_actions:
message += "; permission_errors=" + " | ".join(permission_errors[:3])
return {
"azure_rbac_level": level,
"azure_subscription_count": len(sub_ids),
"azure_subscription_ids": sub_ids[:10],
"azure_can_manage_ai": can_manage_ai,
"azure_can_manage_cognitive": can_manage_cognitive,
"azure_can_manage_ml": can_manage_ml,
"azure_can_assign_roles": can_assign_roles,
"azure_can_deploy_resources": can_deploy_resources,
"azure_permission_actions_sample": sorted(all_actions)[:40],
"message": message,
}
def check_service_principal(parsed, proxy, timeout):
url = f"https://login.microsoftonline.com/{parsed['tenant_id']}/oauth2/v2.0/token"
payload = {
"client_id": parsed["client_id"],
"client_secret": parsed["client_secret"],
"scope": "https://management.azure.com/.default",
"grant_type": "client_credentials",
}
try:
response = requests.post(url, data=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc)}
if response.status_code == 200:
data = response.json()
rbac = probe_azure_rbac(data.get("access_token"), proxy, timeout)
message = "token issued"
if rbac.get("message"):
message = f"{message}; {rbac.get('message')}"
return {
"status": "VALID",
"tenant_id": parsed["tenant_id"],
"client_id": parsed["client_id"],
"expires_in": data.get("expires_in"),
"message": message,
**rbac,
}
message = request_error_message(response)
lower = message.lower()
if response.status_code in (400, 401) and ("invalid_client" in lower or "invalid_grant" in lower):
return {"status": "DEAD", "http_status": response.status_code, "message": message}
if response.status_code in (401, 403):
return {"status": "RESTRICTED", "http_status": response.status_code, "message": message}
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message}
def azure_openai_deployment_ids(payload):
data = payload.get("data") if isinstance(payload, dict) else None
if data is None and isinstance(payload, dict):
data = payload.get("value")
deployments = []
for item in data or []:
if not isinstance(item, dict):
continue
deployment_id = item.get("id") or item.get("name")
model = item.get("model") or item.get("modelName") or ""
if deployment_id:
deployments.append({"id": deployment_id, "model": model})
return deployments
def endpoint_url(endpoint, path):
endpoint = str(endpoint or "").strip().rstrip("/")
if not endpoint.startswith("http://") and not endpoint.startswith("https://"):
endpoint = "https://" + endpoint
path = "/" + str(path or "").lstrip("/")
parsed = urlsplit(endpoint)
endpoint_path = parsed.path.rstrip("/")
if endpoint_path and path.lower().startswith(endpoint_path.lower() + "/"):
path = path[len(endpoint_path):]
return endpoint + path
def azure_model_ids(payload):
data = payload.get("data") if isinstance(payload, dict) else payload if isinstance(payload, list) else []
if data is None and isinstance(payload, dict):
data = payload.get("value") or payload.get("models")
models = []
for item in data or []:
if isinstance(item, str):
models.append(item)
elif isinstance(item, dict):
model_id = item.get("id") or item.get("name") or item.get("model") or item.get("modelName")
if model_id:
models.append(str(model_id))
return models
def endpoint_failure_status(error_text, status):
lower = str(error_text or "").lower()
if any(item in lower for item in (
"name resolution", "no such host", "failed to resolve", "getaddrinfo",
"unexpected_eof", "eof occurred in violation of protocol", "ssleoferror",
)):
return status
return "NETWORK"
def probe_azure_openai_chat_route(endpoint, key, deployments, proxy, timeout):
if not deployments:
return {"route_probe": "no_deployments"}
preferred = None
for item in deployments:
text = f"{item.get('id', '')} {item.get('model', '')}".lower()
if any(marker in text for marker in ("gpt", "chat", "turbo", "4o")):
preferred = item
break
deployment = preferred or deployments[0]
deployment_id = deployment["id"]
url = f"https://{endpoint}/openai/deployments/{deployment_id}/chat/completions?api-version={AZURE_OPENAI_CHAT_API_VERSION}"
headers = {"api-key": key, "Content-Type": "application/json"}
# Empty messages should fail validation after auth/deployment routing, without generating content.
payload = {"messages": [], "max_tokens": 1}
try:
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"route_probe": "network", "route_deployment": deployment_id, "route_message": str(exc)[:500]}
message = request_error_message(response)
if response.status_code in (200, 400):
return {
"route_probe": "accepted_auth_route",
"route_deployment": deployment_id,
"route_model": deployment.get("model", ""),
"route_http_status": response.status_code,
"route_message": message,
}
if response.status_code in (401, 403):
return {"route_probe": "auth_failed", "route_deployment": deployment_id, "route_http_status": response.status_code, "route_message": message}
if response.status_code == 404:
return {"route_probe": "not_found", "route_deployment": deployment_id, "route_http_status": 404, "route_message": message}
return {"route_probe": "unknown", "route_deployment": deployment_id, "route_http_status": response.status_code, "route_message": message}
def check_azure_openai(parsed, proxy, timeout, probe_openai_route=False):
endpoint = (parsed.get("endpoint") or "").strip().strip("/")
key = parsed.get("key")
if not endpoint:
return {"status": "OPENAI_UNRESOLVED", "message": "AzureOpenAI key found without endpoint/resource name"}
url = f"https://{endpoint}/openai/deployments?api-version={AZURE_OPENAI_DEPLOYMENTS_API_VERSION}"
headers = {"api-key": key, "Content-Type": "application/json"}
try:
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
first_error = str(exc)
def endpoint_failure_status(error_text):
lower = str(error_text or "").lower()
if any(item in lower for item in (
"name resolution", "no such host", "failed to resolve", "getaddrinfo",
"unexpected_eof", "eof occurred in violation of protocol", "ssleoferror",
)):
return {"status": "OPENAI_BAD_ENDPOINT", "endpoint": endpoint, "message": error_text}
return None
endpoint_status = endpoint_failure_status(first_error)
if endpoint_status:
return endpoint_status
return {"status": "NETWORK", "endpoint": endpoint, "message": first_error}
if response.status_code == 200:
deployments = azure_openai_deployment_ids(response.json())
result = {
"status": "VALID",
"endpoint": endpoint,
"deployment_count": len(deployments),
"deployments": [item.get("id") for item in deployments[:20]],
"message": f"deployments endpoint accepted key; deployments={len(deployments)}",
}
if probe_openai_route:
result.update(probe_azure_openai_chat_route(endpoint, key, deployments, proxy, timeout))
return result
message = request_error_message(response)
if response.status_code in (401, 403):
return {"status": "DEAD", "endpoint": endpoint, "http_status": response.status_code, "message": message}
if response.status_code == 404:
return {"status": "UNKNOWN", "endpoint": endpoint, "http_status": 404, "message": message}
if response.status_code >= 500:
return {"status": "NETWORK", "endpoint": endpoint, "http_status": response.status_code, "message": message}
return {"status": "UNKNOWN", "endpoint": endpoint, "http_status": response.status_code, "message": message}
def foundry_model_routes(endpoint):
return [
endpoint_url(endpoint, f"/models?api-version={AZURE_FOUNDRY_API_VERSION}"),
endpoint_url(endpoint, "/models"),
endpoint_url(endpoint, "/v1/models"),
]
def foundry_auth_headers(key):
return [
{"api-key": key, "Content-Type": "application/json"},
{"Authorization": f"Bearer {key}", "Content-Type": "application/json"},
]
def check_foundry_models(endpoint, key, proxy, timeout):
attempts = []
auth_failures = 0
attempted = 0
for url in foundry_model_routes(endpoint):
for headers in foundry_auth_headers(key):
attempted += 1
auth_kind = "bearer" if "Authorization" in headers else "api-key"
try:
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
status = endpoint_failure_status(str(exc), "FOUNDRY_BAD_ENDPOINT")
attempts.append(f"{url}:{auth_kind}:network:{str(exc)[:180]}")
if status == "FOUNDRY_BAD_ENDPOINT":
return {"status": status, "endpoint": endpoint, "message": str(exc)}
continue
message = request_error_message(response)
if response.status_code == 200:
models = azure_model_ids(response.json())
return {
"status": "FOUNDRY",
"endpoint": endpoint,
"auth_scheme": auth_kind,
"model_count": len(models),
"models": models[:50],
"message": f"models endpoint accepted key; auth={auth_kind}; models={len(models)}",
}
if response.status_code == 429:
return {
"status": "FOUNDRY",
"endpoint": endpoint,
"auth_scheme": auth_kind,
"model_count": 0,
"models": [],
"message": f"models endpoint rate limited after auth; auth={auth_kind}; {message[:200]}",
}
if response.status_code in (401, 403):
auth_failures += 1
attempts.append(f"{url}:{auth_kind}:auth_{response.status_code}:{message[:160]}")
continue
if response.status_code == 404:
attempts.append(f"{url}:{auth_kind}:http_404:{message[:160]}")
continue
if response.status_code >= 500:
attempts.append(f"{url}:{auth_kind}:server_{response.status_code}:{message[:160]}")
continue
attempts.append(f"{url}:{auth_kind}:http_{response.status_code}:{message[:160]}")
if attempted and auth_failures == attempted:
return {"status": "DEAD", "endpoint": endpoint, "message": "; ".join(attempts[:4])}
return {"status": "UNKNOWN", "endpoint": endpoint, "message": "; ".join(attempts[:4])}
def probe_foundry_route(endpoint, key, models, proxy, timeout):
configured = [item.strip() for item in (models or []) if item.strip()]
if not configured:
return {"foundry_route_probe": "not_configured"}
attempts = []
accepted = []
for model in configured:
route_specs = [
(endpoint_url(endpoint, f"/models/chat/completions?api-version={AZURE_FOUNDRY_API_VERSION}"), {"model": model, "messages": [], "max_tokens": 1}),
(endpoint_url(endpoint, "/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}),
(endpoint_url(endpoint, "/v1/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}),
(endpoint_url(endpoint, "/openai/v1/chat/completions"), {"model": model, "messages": [], "max_tokens": 1}),
]
for url, payload in route_specs:
for headers in foundry_auth_headers(key):
auth_kind = "bearer" if "Authorization" in headers else "api-key"
try:
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
attempts.append(f"{model}:{auth_kind}:network:{str(exc)[:120]}")
continue
message = request_error_message(response)
if response.status_code == 200:
accepted.append(model)
break
if response.status_code == 400 and any(item in message.lower() for item in ("messages", "content", "validation", "empty")):
accepted.append(model)
break
if response.status_code == 429:
accepted.append(model)
break
if response.status_code in (401, 403, 404):
attempts.append(f"{model}:{auth_kind}:http_{response.status_code}:{message[:160]}")
continue
attempts.append(f"{model}:{auth_kind}:http_{response.status_code}:{message[:160]}")
if model in accepted:
break
if accepted:
return {"foundry_route_probe": "accepted", "foundry_route_models": accepted, "foundry_route_message": "route accepted"}
return {"foundry_route_probe": "not_accepted", "foundry_route_models": [], "foundry_route_message": "; ".join(attempts[:8])}
def check_azure_foundry(parsed, proxy, timeout, probe_foundry_route_enabled=False, foundry_models=None):
endpoint = (parsed.get("endpoint") or "").strip().strip("/")
key = parsed.get("key")
if not endpoint:
return {"status": "FOUNDRY_UNRESOLVED", "message": "Azure Foundry key found without endpoint"}
result = check_foundry_models(endpoint, key, proxy, timeout)
if probe_foundry_route_enabled:
route = probe_foundry_route(endpoint, key, foundry_models or [], proxy, timeout)
if result.get("status") != "FOUNDRY" and route.get("foundry_route_probe") == "accepted":
result = {"status": "FOUNDRY", "endpoint": endpoint, "model_count": 0, "models": [], "message": "route accepted without model-list support"}
result.update(route)
return result
def is_azure_openai_llm(result):
if result.get("status") != "VALID":
return False
if int(result.get("deployment_count") or 0) <= 0:
return False
route_probe = result.get("route_probe")
if route_probe and route_probe != "accepted_auth_route":
return False
return True
def append_azure_openai_llm(key, result, source):
deployments = result.get("deployments") or []
deployment_text = ",".join(str(item) for item in deployments[:20])
details = " ".join(part for part in [
f"deployments={int(result.get('deployment_count') or 0)}",
f"route_probe={result.get('route_probe') or ''}" if result.get("route_probe") else "",
f"route_deployment={result.get('route_deployment') or ''}" if result.get("route_deployment") else "",
f"route_model={result.get('route_model') or ''}" if result.get("route_model") else "",
f"deployment_ids={deployment_text}" if deployment_text else "",
] if part)
append_status(AZURE_OPENAI_LLM_FILE, key, result.get("status", "VALID"), details, source)
def check_azure_acr(parsed, proxy, timeout):
username = parsed["username"]
password = parsed["password"]
url = f"https://{username}.azurecr.io/v2/"
try:
response = requests.get(url, auth=(username, password), proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
text = str(exc)
if "no such host" in text.lower():
return {"status": "DEAD", "registry": username, "message": text}
return {"status": "NETWORK", "registry": username, "message": text}
if response.status_code == 200:
return {"status": "VALID", "registry": username, "message": "ACR /v2 accepted basic auth"}
message = request_error_message(response)
if response.status_code == 401:
return {"status": "DEAD", "registry": username, "http_status": 401, "message": message}
if response.status_code == 403:
return {"status": "RESTRICTED", "registry": username, "http_status": 403, "message": message}
if response.status_code >= 500:
return {"status": "NETWORK", "registry": username, "http_status": response.status_code, "message": message}
return {"status": "UNKNOWN", "registry": username, "http_status": response.status_code, "message": message}
def write_result(key, detector, result, source, finding):
safe_finding = strip_finding_nearby_context(finding)
write_keycheck_event(SERVICE, RESULTS_FILE, key, {"detector": detector, **result}, source, safe_finding, detector)
commit_status_transaction(
CHECKED_FILE,
transaction_status_files(),
key,
result["status"],
result.get("message", ""),
source,
)
if detector == "AzureOpenAI" and is_azure_openai_llm(result):
append_azure_openai_llm(key, result, source)
record_validation_result(SERVICE, key, {"detector": detector, **result}, source, safe_finding, detector)
def strip_finding_nearby_context(finding):
if not isinstance(finding, dict):
return finding
output = dict(finding)
context = output.get("ScannerContext")
if isinstance(context, dict) and "nearby" in context:
output["ScannerContext"] = {key: value for key, value in context.items() if key != "nearby"}
return output
def parse_args():
parser = argparse.ArgumentParser(description="Azure key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=20)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
parser.add_argument("--probe-openai-route", action="store_true", help="Probe Azure OpenAI chat route with an invalid no-generation request after deployment listing succeeds")
parser.add_argument("--probe-foundry-route", action="store_true", help="Probe Azure Foundry/MaaS chat route for configured models")
parser.add_argument("--foundry-models", default="", help="Comma-separated Azure Foundry model IDs to route-probe")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = set()
if args.retry_network:
retry_statuses.update({"NETWORK", "FOUNDRY_BAD_ENDPOINT", "OPENAI_BAD_ENDPOINT"})
if args.retry_unknown:
retry_statuses.update({"UNKNOWN", "FOUNDRY_UNRESOLVED", "OPENAI_UNRESOLVED"})
if args.retry_valid:
retry_statuses.update({"VALID", "FOUNDRY"})
foundry_models = [item.strip() for item in str(args.foundry_models or "").split(",") if item.strip()]
processed = 0
skipped = 0
for key, detector, source, finding, parsed in extract_candidates(args.input):
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=detector):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] {detector} candidate {mask_secret(key)} from {source}")
proxy = next(proxy_cycler) if proxy_cycler else None
if parsed.get('_unresolved_candidate'):
result = {
'status': (
'FOUNDRY_UNRESOLVED'
if detector == 'AzureFoundry' else
'OPENAI_UNRESOLVED'
if detector == 'AzureOpenAI' else
'UNKNOWN'
),
'message': f"normalized {parsed.get('candidate_kind') or 'azure'} candidate is incomplete",
}
elif detector == "AzureOpenAI":
result = check_azure_openai(parsed, proxy, args.timeout, args.probe_openai_route)
elif detector == "AzureFoundry":
result = check_azure_foundry(parsed, proxy, args.timeout, args.probe_foundry_route, foundry_models)
if parsed.get("finding_uid"):
result["finding_uid"] = parsed.get("finding_uid")
elif detector == "AzureContainerRegistry":
result = check_azure_acr(parsed, proxy, args.timeout)
else:
result = check_service_principal(parsed, proxy, args.timeout)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
write_result(key, detector, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
@@ -0,0 +1,292 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
import re
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl,
classify_common_http_status,
combined_provider_routing_hint,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
keycheck_input_mode,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
provider_routing_database_failed,
recover_status_transaction,
request_error_message,
record_validation_result,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
from keycheckers.provider_resolution import resolve_provider_key
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
SERVICE = "deepseek"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "deepseekChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "deepseekResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "deepseekAlive.txt"),
"NO_BALANCE": os.path.join(OUTPUT_DIR, "deepseekNoBalance.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "deepseekDead.txt"),
"LIMITED": os.path.join(OUTPUT_DIR, "deepseekLimited.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "deepseekNetwork.txt"),
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "deepseekNoContext.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "deepseekUnknown.txt"),
}
DEEPSEEK_REGEX = re.compile(r"sk-[a-z0-9]{32}")
DEEPSEEK_DETECTOR_NAMES = {"deepseek", "deepseekapikey", "deepseek_api_key"}
DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"}
QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"}
KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"}
QWEN_CONTEXT_REGEX = re.compile(
r"(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)",
re.IGNORECASE,
)
DEEPSEEK_CONTEXT_REGEX = re.compile(r"(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)", re.IGNORECASE)
KIMI_CONTEXT_REGEX = re.compile(
r"(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))",
re.IGNORECASE,
)
AMBIGUOUS_PROVIDER_HINT = "ambiguous_qwen_deepseek"
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"}
EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment"
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def iter_candidate_decisions(input_file, plain_files):
detector_names = ["DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key", "CustomRegex"]
for item in iter_findings(input_file, detector_names):
finding = item.get("finding") or {}
if not finding_has_deepseek_detector(finding):
continue
key = item.get("credential_secret_text") or item["raw"]
if key and DEEPSEEK_REGEX.fullmatch(key):
hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding))
if provider_routing_database_failed():
raise RuntimeError("provider routing evidence lookup failed closed")
yield key, item["source"], finding, hint
def extract_candidates(input_file, plain_files):
for key, source, finding, hint in iter_candidate_decisions(input_file, plain_files):
if hint == "deepseek":
yield key, source, finding
def route_rejection_result(hint):
normalized = str(hint or "missing").strip().lower()
return {
"status": "NO_CONTEXT",
"routing_hint": normalized,
"message": f"candidate is not safely attributable to DeepSeek; routing_hint={normalized}",
}
def finding_detector_names(finding):
if not isinstance(finding, dict):
return set()
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
names = {
str(finding.get("DetectorName") or finding.get("detector") or "").strip().lower(),
str(extra.get("name") or "").strip().lower(),
}
return {name for name in names if name}
def finding_has_deepseek_detector(finding):
return bool(finding_detector_names(finding) & DEEPSEEK_DETECTOR_NAMES)
def finding_has_explicit_detector(finding, detector_names):
return bool(finding_detector_names(finding) & set(detector_names))
def finding_provider_routing_hint(finding):
if not isinstance(finding, dict):
return ""
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
persisted_hint = context.get("provider_hint")
if (
context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE
and persisted_hint in (*GENERIC_SK_PROVIDERS, AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
):
return persisted_hint
parts = []
for key in ("nearby", "file"):
if context.get(key):
parts.append(str(context.get(key)))
metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {}
data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {}
for details in data.values():
if not isinstance(details, dict):
continue
for key in ("file", "repository", "repo", "link", "image"):
if details.get(key):
parts.append(str(details.get(key)))
text = "\n".join(parts)
evidence = set()
if QWEN_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES):
evidence.add("qwen")
if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES):
evidence.add("deepseek")
if KIMI_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES):
evidence.add("kimi")
if persisted_hint == AMBIGUOUS_PROVIDER_HINT:
evidence.update(("qwen", "deepseek"))
elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT:
evidence.update(GENERIC_SK_PROVIDERS)
elif persisted_hint in GENERIC_SK_PROVIDERS:
evidence.add(persisted_hint)
if len(evidence) > 1:
return AMBIGUOUS_PROVIDER_HINT if evidence == {"qwen", "deepseek"} else AMBIGUOUS_GENERIC_SK_HINT
return next(iter(evidence)) if evidence else ""
def finding_has_ambiguous_provider_hint(finding):
return finding_provider_routing_hint(finding) in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
def finding_looks_like_qwen_context(finding):
return finding_provider_routing_hint(finding) == "qwen"
def check_key(key, proxy, timeout):
url = "https://api.deepseek.com/user/balance"
headers = {"Authorization": f"Bearer {key}"}
try:
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc)}
if response.status_code == 200:
data = response.json()
balance_infos = data.get("balance_infos", [])
total_usd = 0.0
for balance in balance_infos:
amount = float(balance.get("total_balance", "0") or 0)
currency = balance.get("currency", "USD")
if currency == "CNY":
amount *= 0.14
total_usd += amount
available = bool(data.get("is_available", False))
status = "VALID" if available and total_usd > 0 else "NO_BALANCE"
return {
"status": status,
"authenticated": True,
"available": available,
"balance_usd": round(total_usd, 4),
"message": f"available={available}; balance=${total_usd:.4f}",
}
status = classify_common_http_status(response.status_code)
return {"status": status, "http_status": response.status_code, "message": request_error_message(response)}
def write_result(key, result, source, finding):
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "DeepSeek")
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), source,
)
record_validation_result(SERVICE, key, result, source, finding, "DeepSeek")
def parse_args():
parser = argparse.ArgumentParser(description="DeepSeek key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=20)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-no-balance", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = set()
if args.retry_network:
retry_statuses.add("NETWORK")
if args.retry_limited:
retry_statuses.add("LIMITED")
if args.retry_unknown:
retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
if args.retry_no_balance:
retry_statuses.add("NO_BALANCE")
if args.retry_valid:
retry_statuses.add("VALID")
processed = 0
skipped = 0
postgres_mode = keycheck_input_mode() == "postgres"
for key, source, finding, routing_hint in iter_candidate_decisions(args.input, args.plain):
route_rejected = routing_hint != "deepseek"
ambiguous_route = routing_hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
if route_rejected and not postgres_mode:
skipped += 1
continue
if not route_rejected and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="DeepSeek"):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] DeepSeek candidate {mask_secret(key)} from {source}")
if route_rejected and ambiguous_route:
proxy = next(proxy_cycler) if proxy_cycler else None
result = resolve_provider_key(
key, finding, proxy, args.timeout,
hint=routing_hint, origin_service=SERVICE,
)
elif route_rejected:
result = route_rejection_result(routing_hint)
else:
proxy = next(proxy_cycler) if proxy_cycler else None
result = check_key(key, proxy, args.timeout)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
@@ -0,0 +1,240 @@
import sys
sys.dont_write_bytecode = True
import argparse
import base64
import json
import os
import re
import time
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
keycheck_input_mode,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
recover_status_transaction,
request_error_message,
record_validation_result,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
SERVICE = "dockerhub"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
PLAIN_FILE = os.path.join(OUTPUT_DIR, "dockerhub.txt")
CHECKED_FILE = os.path.join(OUTPUT_DIR, "dockerhubChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "dockerhubResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "dockerhubAlive.txt"),
"VALID_2FA": os.path.join(OUTPUT_DIR, "dockerhubAlive.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "dockerhubDead.txt"),
"NO_USERNAME": os.path.join(OUTPUT_DIR, "dockerhubNoUsername.txt"),
"RATE_LIMITED": os.path.join(OUTPUT_DIR, "dockerhubRateLimited.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "dockerhubNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "dockerhubUnknown.txt"),
}
DOCKER_PAT_RE = re.compile(r"\bdckr_pat_[A-Za-z0-9_-]{27}\b")
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *set(STATUS_FILES.values()), PLAIN_FILE])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def candidate_key(username, token):
return f"{username}:{token}" if username else token
def parse_username_token(raw, raw_v2, finding):
raw = raw or ""
raw_v2 = raw_v2 or ""
token = ""
username = ""
if ":" in raw_v2:
maybe_user, maybe_token = raw_v2.split(":", 1)
if DOCKER_PAT_RE.fullmatch(maybe_token):
username, token = maybe_user.strip(), maybe_token.strip()
if not token:
match = DOCKER_PAT_RE.search(raw_v2) or DOCKER_PAT_RE.search(raw)
if match:
token = match.group(0)
extra = finding.get("ExtraData") if isinstance(finding, dict) else {}
analysis = finding.get("AnalysisInfo") if isinstance(finding, dict) else {}
if isinstance(extra, dict):
username = username or extra.get("hub_username") or ""
if isinstance(analysis, dict):
username = username or analysis.get("username") or ""
return username, token
def extract_candidates(input_file, plain_file):
seen_plain = set()
for item in iter_findings(input_file, ["Dockerhub"]):
username, token = parse_username_token(item.get("raw"), item.get("raw_v2"), item.get("finding") or {})
if not token:
continue
key = candidate_key(username, token)
yield key, username, token, item["source"], item["finding"]
if keycheck_input_mode() == 'jsonl' and os.path.exists(plain_file):
with open(plain_file, "r", encoding="utf-8") as f:
for line_num, line in enumerate(f, 1):
line = line.strip()
if not line:
continue
username = ""
token = ""
if ":" in line:
maybe_user, rest = line.split(":", 1)
match = DOCKER_PAT_RE.search(rest)
if match:
username, token = maybe_user.strip(), match.group(0)
else:
match = DOCKER_PAT_RE.search(line)
if match:
token = match.group(0)
if not token:
continue
key = candidate_key(username, token)
if key not in seen_plain:
seen_plain.add(key)
yield key, username, token, f"{plain_file}:{line_num}", {}
def decode_jwt_payload(jwt_token):
try:
payload = jwt_token.split(".")[1]
payload += "=" * (-len(payload) % 4)
return json.loads(base64.urlsafe_b64decode(payload.encode()).decode())
except Exception:
return {}
def check_token(username, token, proxy, timeout):
if not username:
return {"status": "NO_USERNAME", "message": "DockerHub PAT requires username/email for login check"}
url = "https://hub.docker.com/v2/users/login"
payload = {"username": username, "password": token}
try:
response = requests.post(url, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc), "username": username}
message = request_error_message(response)
if response.status_code == 200:
data = response.json()
hub_token = data.get("token", "")
claims = decode_jwt_payload(hub_token) if hub_token else {}
hub_claims = claims.get("https://hub.docker.com", {}) if isinstance(claims, dict) else {}
return {
"status": "VALID",
"message": "login accepted",
"username": username,
"hub_username": hub_claims.get("username", username),
"hub_email": hub_claims.get("email", ""),
"scope": claims.get("scope", "") if isinstance(claims, dict) else "",
}
if response.status_code == 401:
try:
data = response.json()
except ValueError:
data = {}
if data.get("login_2fa_token"):
return {"status": "VALID_2FA", "message": "credentials accepted; 2FA required", "username": username}
return {"status": "DEAD", "http_status": 401, "message": message, "username": username}
if response.status_code == 429:
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "username": username}
if response.status_code >= 500:
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "username": username}
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "username": username}
def write_result(key, result, source, finding):
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, "Dockerhub")
extra = result.get("hub_username") or result.get("username") or source
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra,
)
record_validation_result(SERVICE, key, result, source, finding, "Dockerhub")
def parse_args():
parser = argparse.ArgumentParser(description="DockerHub PAT checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", default=PLAIN_FILE)
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=20)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-no-username", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = set()
if args.retry_network:
retry_statuses.add("NETWORK")
if args.retry_limited:
retry_statuses.add("RATE_LIMITED")
if args.retry_unknown:
retry_statuses.add("UNKNOWN")
if args.retry_no_username:
retry_statuses.add("NO_USERNAME")
processed = 0
skipped = 0
for key, username, token, source, finding in extract_candidates(args.input, args.plain):
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector="Dockerhub"):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] DockerHub candidate {mask_secret(key)} from {source}")
proxy = next(proxy_cycler) if proxy_cycler else None
result = check_token(username, token, proxy, args.timeout)
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
time.sleep(0.1)
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+789
View File
@@ -0,0 +1,789 @@
import sys
sys.dont_write_bytecode = True
import argparse
import base64
import binascii
import hashlib
import json
import os
import re
import time
from urllib.parse import urlsplit
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl,
append_status,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
keycheck_input_mode,
load_checked_statuses,
load_known_keys,
load_known_statuses,
load_proxies,
mask_secret,
recover_status_transaction,
request_error_message,
record_cached_keycheck_occurrence,
record_validation_result,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
SERVICE = "gcp"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
PLAIN_FILE = os.path.join(OUTPUT_DIR, "gcp.txt")
CHECKED_FILE = os.path.join(OUTPUT_DIR, "gcpChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "gcpResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "gcpAlive.txt"),
"VERTEX": os.path.join(OUTPUT_DIR, "gcpVertex.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "gcpDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "gcpRestricted.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "gcpNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "gcpUnknown.txt"),
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "gcpNoContext.txt"),
}
VERTEX_GEMINI_FILE = os.path.join(OUTPUT_DIR, "gcpVertexGemini.txt")
VERTEX_ANTHROPIC_FILE = os.path.join(OUTPUT_DIR, "gcpVertexAnthropic.txt")
GOOGLE_TOKEN_URL = "https://oauth2.googleapis.com/token"
TRUSTED_GOOGLE_TOKEN_ENDPOINTS = frozenset({
GOOGLE_TOKEN_URL,
"https://accounts.google.com/o/oauth2/token",
})
TOKEN_REDIRECT_STATUSES = {301, 302, 303, 307, 308}
SA_SCOPE = "https://www.googleapis.com/auth/cloud-platform"
MAX_PEM_BYTES = 24 * 1024
MAX_DER_BYTES = 16 * 1024
MAX_DER_LENGTH_BYTES = 2
MIN_RSA_BITS = 2048
MAX_RSA_BITS = 8192
MAX_RSA_INTEGER_BYTES = MAX_RSA_BITS // 8
RSA_ENCRYPTION_OID = bytes.fromhex("2a864886f70d010101")
VERTEX_LOCATIONS = ["global", "us", "eu"]
VERTEX_MODELS = ["gemini-3.6-flash", "gemini-3.1-pro-preview"]
VERTEX_ANTHROPIC_LOCATIONS = ["global", "us", "eu", "us-east5", "europe-west1"]
VERTEX_ANTHROPIC_MODELS = ["claude-opus-5", "claude-opus-4-7", "claude-opus-4-6", "claude-fable-5"]
def transaction_status_files():
return {
**STATUS_FILES,
"RATE_LIMITED": STATUS_FILES["UNKNOWN"],
"AUX_VERTEX_GEMINI": VERTEX_GEMINI_FILE,
"AUX_VERTEX_ANTHROPIC": VERTEX_ANTHROPIC_FILE,
}
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), VERTEX_GEMINI_FILE, VERTEX_ANTHROPIC_FILE, PLAIN_FILE])
recover_status_transaction(CHECKED_FILE, transaction_status_files())
def b64url(data):
return base64.urlsafe_b64encode(data).rstrip(b"=").decode()
def validate_google_token_uri(value):
token_uri = str(value or GOOGLE_TOKEN_URL).strip()
try:
parsed = urlsplit(token_uri)
port = parsed.port
except (TypeError, ValueError) as exc:
raise ValueError("invalid Google OAuth token_uri") from exc
if (
parsed.scheme != "https"
or parsed.username is not None
or parsed.password is not None
or port not in (None, 443)
or parsed.query
or parsed.fragment
or token_uri not in TRUSTED_GOOGLE_TOKEN_ENDPOINTS
):
raise ValueError("untrusted Google OAuth token_uri")
return token_uri
class InvalidRSAPrivateKey(ValueError):
pass
class DERReader:
def __init__(self, data):
if not isinstance(data, (bytes, bytearray, memoryview)):
raise InvalidRSAPrivateKey("DER value is not binary")
if len(data) > MAX_DER_BYTES:
raise InvalidRSAPrivateKey("DER value exceeds size limit")
self.data = data
self.pos = 0
def read_tlv(self):
if len(self.data) - self.pos < 2:
raise InvalidRSAPrivateKey("truncated DER tag or length")
tag = self.data[self.pos]
self.pos += 1
first_len = self.data[self.pos]
self.pos += 1
if first_len & 0x80:
length_len = first_len & 0x7F
if length_len == 0:
raise InvalidRSAPrivateKey("indefinite DER length is not allowed")
if length_len > MAX_DER_LENGTH_BYTES:
raise InvalidRSAPrivateKey("DER length-of-length exceeds limit")
if len(self.data) - self.pos < length_len:
raise InvalidRSAPrivateKey("truncated DER length")
length_bytes = self.data[self.pos:self.pos + length_len]
if length_bytes[0] == 0:
raise InvalidRSAPrivateKey("non-minimal DER length")
length = int.from_bytes(length_bytes, "big")
self.pos += length_len
if length < 0x80:
raise InvalidRSAPrivateKey("non-minimal DER length")
else:
length = first_len
if length > MAX_DER_BYTES:
raise InvalidRSAPrivateKey("DER value length exceeds limit")
if length > len(self.data) - self.pos:
raise InvalidRSAPrivateKey("truncated DER value")
value = self.data[self.pos:self.pos + length]
self.pos += length
return tag, value
def expect(self, tag):
actual, value = self.read_tlv()
if actual != tag:
raise InvalidRSAPrivateKey(f"expected DER tag {tag:#x}, got {actual:#x}")
return value
def at_end(self):
return self.pos == len(self.data)
def require_eof(self, context="DER structure"):
if not self.at_end():
raise InvalidRSAPrivateKey(f"trailing data in {context}")
def der_int(value, name="integer", max_bytes=MAX_RSA_INTEGER_BYTES):
if not value:
raise InvalidRSAPrivateKey(f"empty RSA {name}")
if len(value) > max_bytes + 1:
raise InvalidRSAPrivateKey(f"RSA {name} exceeds size limit")
if value[0] & 0x80:
raise InvalidRSAPrivateKey(f"negative RSA {name}")
if value[0] == 0:
if len(value) > 1 and not value[1] & 0x80:
raise InvalidRSAPrivateKey(f"non-minimal RSA {name}")
unsigned = value[1:]
else:
unsigned = value
if len(unsigned) > max_bytes:
raise InvalidRSAPrivateKey(f"RSA {name} exceeds size limit")
return int.from_bytes(unsigned, "big") if unsigned else 0
def parse_pkcs1_rsa_private_key(data):
rsa = DERReader(data)
version = der_int(rsa.expect(0x02), "version", 1)
if version != 0:
raise InvalidRSAPrivateKey("unsupported RSA private key version")
n = der_int(rsa.expect(0x02), "modulus")
public_exponent = der_int(rsa.expect(0x02), "public exponent")
d = der_int(rsa.expect(0x02), "private exponent")
for name in ("prime1", "prime2", "exponent1", "exponent2", "coefficient"):
der_int(rsa.expect(0x02), name)
rsa.require_eof("RSA private key")
modulus_bits = n.bit_length()
if not MIN_RSA_BITS <= modulus_bits <= MAX_RSA_BITS:
raise InvalidRSAPrivateKey(
f"RSA modulus must be between {MIN_RSA_BITS} and {MAX_RSA_BITS} bits"
)
if public_exponent == 0:
raise InvalidRSAPrivateKey("RSA public exponent is zero")
if d == 0 or d >= n:
raise InvalidRSAPrivateKey("RSA private exponent is out of range")
return n, d
def validate_rsa_algorithm_identifier(data):
algorithm = DERReader(data)
if algorithm.expect(0x06) != RSA_ENCRYPTION_OID:
raise InvalidRSAPrivateKey("PKCS#8 key does not use rsaEncryption")
if not algorithm.at_end() and algorithm.expect(0x05):
raise InvalidRSAPrivateKey("invalid rsaEncryption parameters")
algorithm.require_eof("PKCS#8 algorithm identifier")
def parse_rsa_private_key_from_pem(pem):
pem = str(pem or "")
if len(pem) > MAX_PEM_BYTES:
raise InvalidRSAPrivateKey("PEM private key exceeds size limit")
try:
pem_bytes = pem.encode("utf-8")
except UnicodeEncodeError as exc:
raise InvalidRSAPrivateKey("PEM private key is not valid UTF-8") from exc
if len(pem_bytes) > MAX_PEM_BYTES:
raise InvalidRSAPrivateKey("PEM private key exceeds size limit")
pem = pem.replace("\\n", "\n")
match = re.fullmatch(
r"\s*-----BEGIN (RSA PRIVATE KEY|PRIVATE KEY)-----\s*(.*?)\s*-----END \1-----\s*",
pem,
re.DOTALL,
)
if not match:
raise InvalidRSAPrivateKey("missing complete PEM private key block")
body = re.sub(r"\s+", "", match.group(2))
if len(body) < 256:
raise InvalidRSAPrivateKey("PEM private key body is too short")
try:
der = base64.b64decode(body + ("=" * (-len(body) % 4)), validate=True)
except (binascii.Error, ValueError) as exc:
raise InvalidRSAPrivateKey("invalid PEM base64") from exc
if len(der) > MAX_DER_BYTES:
raise InvalidRSAPrivateKey("DER private key exceeds size limit")
reader = DERReader(der)
top_bytes = reader.expect(0x30)
reader.require_eof("DER private key")
top = DERReader(top_bytes)
# PKCS#8 PrivateKeyInfo: SEQUENCE(version, alg, OCTET STRING(RSAPrivateKey))
first_tag, first_val = top.read_tlv()
if first_tag != 0x02:
raise InvalidRSAPrivateKey("unexpected private key structure")
second_tag, second_val = top.read_tlv()
if second_tag == 0x30:
if der_int(first_val, "PKCS#8 version", 1) != 0:
raise InvalidRSAPrivateKey("unsupported PKCS#8 version")
validate_rsa_algorithm_identifier(second_val)
private_octet = top.expect(0x04)
top.require_eof("PKCS#8 private key")
wrapped = DERReader(private_octet)
rsa_bytes = wrapped.expect(0x30)
wrapped.require_eof("PKCS#8 private key octets")
return parse_pkcs1_rsa_private_key(rsa_bytes)
if second_tag != 0x02:
raise InvalidRSAPrivateKey("unexpected private key structure")
return parse_pkcs1_rsa_private_key(top_bytes)
def rsa_pkcs1v15_sha256_sign(message, pem):
n, d = parse_rsa_private_key_from_pem(pem)
digest = hashlib.sha256(message).digest()
digest_info = bytes.fromhex("3031300d060960864801650304020105000420") + digest
key_len = (n.bit_length() + 7) // 8
if key_len < len(digest_info) + 11:
raise InvalidRSAPrivateKey("RSA key too small")
encoded = b"\x00\x01" + b"\xff" * (key_len - len(digest_info) - 3) + b"\x00" + digest_info
sig = pow(int.from_bytes(encoded, "big"), d, n).to_bytes(key_len, "big")
return sig
def make_service_account_assertion(creds):
now = int(time.time())
token_uri = validate_google_token_uri(creds.get("token_uri"))
header = {"alg": "RS256", "typ": "JWT", "kid": creds.get("private_key_id")}
payload = {
"iss": creds["client_email"],
"scope": SA_SCOPE,
"aud": token_uri,
"iat": now,
"exp": now + 3600,
}
signing_input = (b64url(json.dumps(header, separators=(",", ":")).encode()) + "." + b64url(json.dumps(payload, separators=(",", ":")).encode())).encode()
signature = rsa_pkcs1v15_sha256_sign(signing_input, creds["private_key"])
return signing_input.decode() + "." + b64url(signature), token_uri
def compact_json(data):
return json.dumps(data, ensure_ascii=False, separators=(",", ":"), sort_keys=True)
def scanner_context_text(finding):
context = finding.get("ScannerContext") if isinstance(finding, dict) else None
if isinstance(context, dict):
return str(context.get("nearby") or "")
return ""
def parse_json_object(text):
try:
return json.loads(text)
except (TypeError, ValueError):
pass
match = re.search(r"\{.*\}", str(text or ""), re.DOTALL)
if not match:
return None
try:
return json.loads(match.group(0))
except ValueError:
return None
def parse_service_account(raw_v2):
data = parse_json_object(raw_v2)
if not isinstance(data, dict):
return None
required = ["client_email", "private_key", "private_key_id"]
if not all(data.get(item) for item in required):
return None
private_key = str(data.get("private_key") or "")
if "-----BEGIN" not in private_key or "-----END" not in private_key or len(private_key) < 800:
return None
return data
def parse_adc(raw_v2, finding):
data = parse_json_object(scanner_context_text(finding)) or parse_json_object(raw_v2)
if not isinstance(data, dict):
return None
required = ["client_id", "client_secret", "refresh_token"]
if not all(data.get(item) for item in required):
return None
return data
def extract_candidates(input_file, plain_file):
seen_plain = set()
for item in iter_findings(input_file, ["GCP", "GCPApplicationDefaultCredentials"]):
if item["detector"] == "GCP":
parsed = parse_service_account(item["raw_v2"])
detector = "GCP"
else:
parsed = parse_adc(item["raw_v2"], item["finding"])
detector = "GCPApplicationDefaultCredentials"
if not parsed:
key = f"{detector}:no_context:{item['source']}"
yield key, detector, item["source"], item["finding"], None
continue
key = compact_json(parsed)
yield key, detector, item["source"], item["finding"], parsed
if keycheck_input_mode() == 'jsonl' and os.path.exists(plain_file):
with open(plain_file, "r", encoding="utf-8", errors="replace") as f:
content = f.read()
candidates = []
whole_file = parse_json_object(content)
if whole_file:
candidates.append((plain_file, whole_file))
for line_num, line in enumerate(content.splitlines(), 1):
key_text = line.split("\t", 1)[0].strip()
parsed = parse_json_object(key_text) or parse_json_object(line)
if parsed:
candidates.append((f"{plain_file}:{line_num}", parsed))
for source, parsed in candidates:
detector = "GCP" if parsed.get("private_key") else "GCPApplicationDefaultCredentials"
key = compact_json(parsed)
if key not in seen_plain:
seen_plain.add(key)
yield key, detector, source, {}, parsed
def vertex_api_host(location):
location = str(location or "").strip().lower()
if location == "global":
return "aiplatform.googleapis.com"
if location in ("us", "eu"):
return f"aiplatform.{location}.rep.googleapis.com"
return f"{location}-aiplatform.googleapis.com"
def probe_vertex_llm(access_token, project_id, proxy, timeout, locations=None, models=None, max_attempts=2):
if not project_id:
return {"enabled": False, "message": "project_id unavailable"}
headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"}
payload = {"contents": [{"role": "user", "parts": [{"text": "ping"}]}]}
attempts = []
accepted = []
tried = 0
for location in (locations or VERTEX_LOCATIONS):
for model in (models or VERTEX_MODELS):
if max_attempts and tried >= max_attempts:
if accepted:
first = accepted[0]
return {
"enabled": True,
"location": first.get("location", ""),
"model": first.get("model", ""),
"total_tokens": first.get("total_tokens"),
"available_models": [f"google/{item['location']}/{item['model']}" for item in accepted],
"message": "Vertex countTokens accepted",
}
return {"enabled": False, "message": "; ".join(attempts[:8]) or "Vertex probe attempt limit reached"}
tried += 1
url = f"https://{vertex_api_host(location)}/v1/projects/{project_id}/locations/{location}/publishers/google/models/{model}:countTokens"
try:
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
attempts.append(f"{location}:{model}:network:{str(exc)[:120]}")
continue
if response.status_code == 200:
data = response.json()
accepted.append({
"location": location,
"model": model,
"total_tokens": data.get("totalTokens") or data.get("total_tokens"),
})
continue
message = request_error_message(response)
if response.status_code in (400, 401, 403, 404, 429):
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
continue
if response.status_code >= 500:
attempts.append(f"{location}:{model}:server_{response.status_code}:{message[:160]}")
continue
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
if accepted:
first = accepted[0]
return {
"enabled": True,
"location": first.get("location", ""),
"model": first.get("model", ""),
"total_tokens": first.get("total_tokens"),
"available_models": [f"google/{item['location']}/{item['model']}" for item in accepted],
"message": "Vertex countTokens accepted",
}
return {"enabled": False, "message": "; ".join(attempts[:8])}
def probe_vertex_anthropic(access_token, project_id, proxy, timeout, locations=None, models=None, max_attempts=2):
if not project_id:
return {"enabled": False, "message": "project_id unavailable"}
models = models or []
if not models:
return {"enabled": False, "message": "no Anthropic models configured"}
headers = {"Authorization": f"Bearer {access_token}", "Content-Type": "application/json"}
payload = {
"anthropic_version": "vertex-2023-10-16",
"messages": [{"role": "user", "content": "ping"}],
"max_tokens": 1,
}
attempts = []
accepted = []
tried = 0
for location in (locations or VERTEX_ANTHROPIC_LOCATIONS):
for model in models:
if max_attempts and tried >= max_attempts:
if accepted:
first = accepted[0]
return {
"enabled": True,
"location": first.get("location", ""),
"model": first.get("model", ""),
"available_models": [f"anthropic/{item['location']}/{item['model']}" for item in accepted],
"message": "Vertex Anthropic rawPredict accepted",
}
return {"enabled": False, "message": "; ".join(attempts[:8]) or "Vertex Anthropic probe attempt limit reached"}
tried += 1
url = f"https://{vertex_api_host(location)}/v1/projects/{project_id}/locations/{location}/publishers/anthropic/models/{model}:rawPredict"
try:
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
attempts.append(f"{location}:{model}:network:{str(exc)[:120]}")
continue
if response.status_code == 200:
accepted.append({"location": location, "model": model})
continue
message = request_error_message(response)
if response.status_code in (400, 401, 403, 404, 429):
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
continue
if response.status_code >= 500:
attempts.append(f"{location}:{model}:server_{response.status_code}:{message[:160]}")
continue
attempts.append(f"{location}:{model}:http_{response.status_code}:{message[:160]}")
if accepted:
first = accepted[0]
return {
"enabled": True,
"location": first.get("location", ""),
"model": first.get("model", ""),
"available_models": [f"anthropic/{item['location']}/{item['model']}" for item in accepted],
"message": "Vertex Anthropic rawPredict accepted",
}
return {"enabled": False, "message": "; ".join(attempts[:8])}
def merge_vertex_results(google_vertex, anthropic_vertex):
google_vertex = google_vertex or {"enabled": False, "message": ""}
anthropic_vertex = anthropic_vertex or {"enabled": False, "message": ""}
available = []
available.extend(google_vertex.get("available_models") or [])
available.extend(anthropic_vertex.get("available_models") or [])
first = google_vertex if google_vertex.get("enabled") else anthropic_vertex if anthropic_vertex.get("enabled") else {}
messages = []
if google_vertex.get("message"):
messages.append(f"google: {google_vertex.get('message')}")
if anthropic_vertex.get("message"):
messages.append(f"anthropic: {anthropic_vertex.get('message')}")
return {
"enabled": bool(available),
"location": first.get("location", ""),
"model": first.get("model", ""),
"total_tokens": first.get("total_tokens"),
"available_models": available,
"google_enabled": bool(google_vertex.get("enabled")),
"anthropic_enabled": bool(anthropic_vertex.get("enabled")),
"message": "; ".join(messages),
}
def check_service_account(creds, proxy, timeout, probe_vertex=False, vertex_timeout=6, vertex_locations=None, vertex_models=None, vertex_max_attempts=2, vertex_anthropic_locations=None, vertex_anthropic_models=None, vertex_anthropic_max_attempts=2):
try:
assertion, token_uri = make_service_account_assertion(creds)
except InvalidRSAPrivateKey as exc:
return {
"status": "DEAD",
"classification": "invalid_private_key",
"message": f"invalid RSA private key: {exc}",
"project_id": creds.get("project_id"),
"client_email": creds.get("client_email"),
}
except Exception as exc:
return {"status": "UNKNOWN", "message": f"failed to build JWT assertion: {exc}"}
data = {"grant_type": "urn:ietf:params:oauth:grant-type:jwt-bearer", "assertion": assertion}
try:
response = requests.post(
token_uri, data=data, proxies=proxy, timeout=timeout, allow_redirects=False,
)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc), "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
if response.status_code in TOKEN_REDIRECT_STATUSES:
return {
"status": "UNKNOWN",
"http_status": response.status_code,
"message": "Google OAuth token endpoint redirect refused",
"project_id": creds.get("project_id"),
"client_email": creds.get("client_email"),
}
if response.status_code == 200:
payload = response.json()
result = {
"status": "VALID",
"message": "OAuth token issued",
"project_id": creds.get("project_id"),
"client_email": creds.get("client_email"),
"private_key_id": creds.get("private_key_id"),
"expires_in": payload.get("expires_in"),
}
if probe_vertex:
google_vertex = probe_vertex_llm(
payload.get("access_token"), creds.get("project_id"), proxy,
vertex_timeout, vertex_locations, vertex_models, vertex_max_attempts,
)
anthropic_vertex = probe_vertex_anthropic(
payload.get("access_token"), creds.get("project_id"), proxy,
vertex_timeout, vertex_anthropic_locations, vertex_anthropic_models, vertex_anthropic_max_attempts,
) if vertex_anthropic_models else {"enabled": False, "message": ""}
vertex = merge_vertex_results(google_vertex, anthropic_vertex)
result.update({
"vertex_enabled": vertex.get("enabled"),
"vertex_location": vertex.get("location", ""),
"vertex_model": vertex.get("model", ""),
"vertex_available_models": vertex.get("available_models") or [],
"vertex_google_enabled": vertex.get("google_enabled"),
"vertex_anthropic_enabled": vertex.get("anthropic_enabled"),
"vertex_total_tokens": vertex.get("total_tokens"),
"vertex_message": vertex.get("message", ""),
})
if vertex.get("enabled"):
result["status"] = "VERTEX"
result["message"] = "OAuth token issued; Vertex countTokens accepted"
return result
message = request_error_message(response)
lower = message.lower()
if response.status_code in (400, 401) and any(item in lower for item in ("invalid_grant", "invalid_client", "invalid jwt", "invalid signature")):
return {"status": "DEAD", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
if response.status_code in (401, 403):
return {"status": "RESTRICTED", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
if response.status_code == 429:
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
if response.status_code >= 500:
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "project_id": creds.get("project_id"), "client_email": creds.get("client_email")}
def check_adc(creds, proxy, timeout, probe_vertex=False, vertex_timeout=6, vertex_locations=None, vertex_models=None, vertex_max_attempts=2, vertex_anthropic_locations=None, vertex_anthropic_models=None, vertex_anthropic_max_attempts=2):
try:
token_uri = validate_google_token_uri(creds.get("token_uri"))
except ValueError as exc:
return {"status": "UNKNOWN", "message": str(exc), "client_id": creds.get("client_id")}
data = {
"client_id": creds["client_id"],
"client_secret": creds["client_secret"],
"refresh_token": creds["refresh_token"],
"grant_type": "refresh_token",
}
try:
response = requests.post(
token_uri, data=data, proxies=proxy, timeout=timeout, allow_redirects=False,
)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc), "client_id": creds.get("client_id")}
if response.status_code in TOKEN_REDIRECT_STATUSES:
return {
"status": "UNKNOWN",
"http_status": response.status_code,
"message": "Google OAuth token endpoint redirect refused",
"client_id": creds.get("client_id"),
}
if response.status_code == 200:
payload = response.json()
project_id = creds.get("quota_project_id") or creds.get("project_id")
result = {"status": "VALID", "message": "refresh token accepted", "client_id": creds.get("client_id"), "project_id": project_id, "expires_in": payload.get("expires_in")}
if probe_vertex:
google_vertex = probe_vertex_llm(
payload.get("access_token"), project_id, proxy,
vertex_timeout, vertex_locations, vertex_models, vertex_max_attempts,
)
anthropic_vertex = probe_vertex_anthropic(
payload.get("access_token"), project_id, proxy,
vertex_timeout, vertex_anthropic_locations, vertex_anthropic_models, vertex_anthropic_max_attempts,
) if vertex_anthropic_models else {"enabled": False, "message": ""}
vertex = merge_vertex_results(google_vertex, anthropic_vertex)
result.update({
"vertex_enabled": vertex.get("enabled"),
"vertex_location": vertex.get("location", ""),
"vertex_model": vertex.get("model", ""),
"vertex_available_models": vertex.get("available_models") or [],
"vertex_google_enabled": vertex.get("google_enabled"),
"vertex_anthropic_enabled": vertex.get("anthropic_enabled"),
"vertex_total_tokens": vertex.get("total_tokens"),
"vertex_message": vertex.get("message", ""),
})
if vertex.get("enabled"):
result["status"] = "VERTEX"
result["message"] = "refresh token accepted; Vertex countTokens accepted"
return result
message = request_error_message(response)
lower = message.lower()
if response.status_code in (400, 401) and any(item in lower for item in ("invalid_grant", "invalid_client", "unauthorized_client")):
return {"status": "DEAD", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
if response.status_code in (401, 403):
return {"status": "RESTRICTED", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
if response.status_code == 429:
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "client_id": creds.get("client_id")}
if response.status_code >= 500:
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "client_id": creds.get("client_id")}
def write_result(key, detector, result, source, finding):
write_keycheck_event(SERVICE, RESULTS_FILE, key, {"detector": detector, **result}, source, finding, detector)
extra = result.get("client_email") or result.get("client_id") or result.get("project_id") or source
message = result.get("message", "")
if result.get("status") == "VERTEX":
models = result.get("vertex_available_models") or []
model_text = ",".join(str(item) for item in models) or f"{result.get('vertex_location', '')}/{result.get('vertex_model', '')}".strip("/")
message = f"{message}; models={model_text}"
commit_status_transaction(
CHECKED_FILE, transaction_status_files(), key, result["status"], message, extra,
)
if result.get("status") == "VERTEX" and result.get("vertex_google_enabled"):
append_status(VERTEX_GEMINI_FILE, key, result["status"], message, extra)
if result.get("status") == "VERTEX" and result.get("vertex_anthropic_enabled"):
append_status(VERTEX_ANTHROPIC_FILE, key, result["status"], message, extra)
record_validation_result(SERVICE, key, {"detector": detector, **result}, source, finding, detector)
def parse_args():
parser = argparse.ArgumentParser(description="GCP credential checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", default=PLAIN_FILE)
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=25)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
parser.add_argument("--probe-vertex", action="store_true", help="After OAuth succeeds, probe Vertex AI Gemini with countTokens through the configured proxy")
parser.add_argument("--vertex-timeout", type=int, default=6, help="Seconds per Vertex countTokens request")
parser.add_argument("--vertex-max-attempts", type=int, default=6, help="Maximum location/model countTokens attempts per credential")
parser.add_argument("--vertex-locations", default=",".join(VERTEX_LOCATIONS), help="Comma-separated Vertex locations to probe")
parser.add_argument("--vertex-models", default=",".join(VERTEX_MODELS), help="Comma-separated Vertex models to probe")
parser.add_argument("--vertex-anthropic-locations", default=",".join(VERTEX_ANTHROPIC_LOCATIONS), help="Comma-separated Vertex Anthropic locations to probe")
parser.add_argument("--vertex-anthropic-models", default=",".join(VERTEX_ANTHROPIC_MODELS), help="Comma-separated Vertex Anthropic model IDs to probe with rawPredict")
parser.add_argument("--vertex-anthropic-max-attempts", type=int, default=20, help="Maximum Anthropic location/model attempts per credential")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
known = set(known_statuses)
retry_statuses = set()
if args.retry_network:
retry_statuses.add("NETWORK")
if args.retry_limited:
retry_statuses.add("RATE_LIMITED")
if args.retry_unknown:
retry_statuses.add("UNKNOWN")
if args.retry_valid:
retry_statuses.update({"VALID", "VERTEX"})
vertex_locations = [item.strip() for item in str(args.vertex_locations or "").split(",") if item.strip()]
vertex_models = [item.strip() for item in str(args.vertex_models or "").split(",") if item.strip()]
vertex_anthropic_locations = [item.strip() for item in str(args.vertex_anthropic_locations or "").split(",") if item.strip()]
vertex_anthropic_models = [item.strip() for item in str(args.vertex_anthropic_models or "").split(",") if item.strip()]
processed = 0
skipped = 0
for key, detector, source, finding, parsed in extract_candidates(args.input, args.plain):
if should_skip_key(
key, checked, known, args, retry_statuses,
service=SERVICE, source=source, finding=finding, detector=detector, known_statuses=known_statuses,
):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] {detector} candidate {mask_secret(key)} from {source}", flush=True)
proxy = next(proxy_cycler) if proxy_cycler else None
if not parsed:
result = {"status": "NO_CONTEXT", "message": "credential JSON is incomplete or unavailable"}
elif detector == "GCP":
result = check_service_account(
parsed, proxy, args.timeout, args.probe_vertex,
args.vertex_timeout, vertex_locations, vertex_models, args.vertex_max_attempts,
vertex_anthropic_locations, vertex_anthropic_models, args.vertex_anthropic_max_attempts,
)
else:
result = check_adc(
parsed, proxy, args.timeout, args.probe_vertex,
args.vertex_timeout, vertex_locations, vertex_models, args.vertex_max_attempts,
vertex_anthropic_locations, vertex_anthropic_models, args.vertex_anthropic_max_attempts,
)
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}", flush=True)
write_result(key, detector, result, source, finding)
known.add(key)
checked[key] = result["status"]
time.sleep(0.1)
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+738
View File
@@ -0,0 +1,738 @@
import sys
sys.dont_write_bytecode = True
import argparse
import json
import os
import re
import time
from datetime import datetime, timezone
from itertools import cycle
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
acquire_file_lock,
append_checked,
default_input_file,
default_proxy_file,
env_int,
ensure_output_files as ensure_private_output_files,
iter_findings,
iter_bounded_text_lines,
keycheck_input_mode,
load_known_statuses,
private_atomic_writer,
record_cached_keycheck_occurrence,
record_validation_result,
release_file_lock,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
from runtime_security import durable_replace, reject_reparse_components, require_private_directory, require_private_file
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
SERVICE = "gemini"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
def here(*parts):
return os.path.join(SCRIPT_DIR, *parts)
def out(*parts):
return os.path.join(OUTPUT_DIR, *parts)
def parent(*parts):
return os.path.join(PARENT_DIR, *parts)
# --- Configuration ---
DEFAULT_INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
DEFAULT_PLAIN_INPUT_FILES = [out("gem.txt")]
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = out("geminiChecked.txt")
RESULTS_FILE = out("geminiResults.jsonl")
STATUS_FILES = {
"VALID": out("geminiAlive.txt"),
"VALID_RATE_LIMITED": out("geminiAliveRateLimited.txt"),
"INVALID": out("geminiDead.txt"),
"EXPIRED": out("geminiExpired.txt"),
"LEAKED_REVOKED": out("geminiLeaked.txt"),
"API_DISABLED": out("geminiDisabled.txt"),
"RESTRICTED": out("geminiRestricted.txt"),
"RATE_LIMITED": out("geminiRateLimited.txt"),
"NETWORK_ERROR": out("geminiNetwork.txt"),
"UNKNOWN": out("geminiUnknown.txt"),
}
GEMINI_KEY_REGEX = re.compile(r"(?:AIza[0-9A-Za-z\-_]{35}|AQ\.[0-9A-Za-z\-_]{50})")
GEMINI_DETECTOR_NAMES = {"googleai", "googleaistudio"}
MODELS_URL = "https://generativelanguage.googleapis.com/v1beta/models"
PROBE_MODEL_PRIORITY = [
"gemini-3.1-pro-preview",
"gemini-3.7-flash",
]
MODEL_PRIORITY = [
"gemini-3",
"gemini-2.5-pro",
"gemini-2.5-flash",
"gemini-2.0-flash",
"gemini-1.5-pro",
"gemini-1.5-flash",
"imagen",
"embedding",
]
def now_iso():
return datetime.now(timezone.utc).isoformat(timespec="seconds")
def mask_key(key):
if not key or len(key) < 12:
return key
return f"{key[:8]}...{key[-4:]}"
def redact_key_text(text, key):
if not isinstance(text, str):
return text
redacted = text.replace(key, "***REDACTED***") if key else text
return GEMINI_KEY_REGEX.sub("***REDACTED***", redacted)
def redact_result_text(result, key):
if isinstance(result, dict):
return {k: redact_result_text(v, key) for k, v in result.items()}
if isinstance(result, list):
return [redact_result_text(v, key) for v in result]
return redact_key_text(result, key)
def key_from_line(line):
line = line.strip()
if not line:
return None
if "\t" in line:
return line.split("\t", 1)[0].strip()
return line.split(":", 1)[0].strip()
def load_keys_from_file(filepath):
if keycheck_input_mode() == 'postgres':
return set()
if not os.path.exists(filepath):
return set()
keys = set()
for line in iter_bounded_text_lines(filepath):
key = key_from_line(line)
if key:
keys.add(key)
return keys
def load_checked_statuses(filepath=CHECKED_FILE):
statuses = {}
if keycheck_input_mode() == 'postgres':
return statuses
if not os.path.exists(filepath):
return statuses
for line in iter_bounded_text_lines(filepath):
parts = line.rstrip("\n").split("\t")
if not parts or not parts[0]:
continue
key = parts[0]
status = parts[1] if len(parts) > 1 else "UNKNOWN"
statuses[key] = status
return statuses
def load_all_known_keys():
known = set(load_checked_statuses().keys())
for path in STATUS_FILES.values():
known.update(load_keys_from_file(path))
return known
def ensure_output_files():
if keycheck_input_mode() == 'postgres':
return
require_private_directory(OUTPUT_DIR, create=True)
legacy_rate_limited = out("geminiLimited.txt")
rate_limited = STATUS_FILES["RATE_LIMITED"]
if os.path.exists(legacy_rate_limited) and not os.path.exists(rate_limited):
require_private_file(legacy_rate_limited)
durable_replace(legacy_rate_limited, rate_limited)
require_private_file(rate_limited)
paths = {CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()}
ensure_private_output_files(paths)
migrate_legacy_alive_rate_limited()
def effective_status(result):
status = result.get("status")
probe_status = (result.get("probe") or {}).get("status")
if status == "VALID" and probe_status == "RATE_LIMITED":
return "VALID_RATE_LIMITED"
return status
def _gemini_status_layout():
paths_by_status = {
status: os.path.abspath(os.fspath(path))
for status, path in STATUS_FILES.items()
}
paths = list(dict.fromkeys(paths_by_status.values()))
directories = {os.path.normcase(os.path.dirname(path)) for path in paths}
if len(paths) != len(paths_by_status) or len(directories) != 1:
raise RuntimeError("Gemini status files must be unique files in one directory")
directory = os.path.dirname(paths[0])
require_private_directory(directory, create=True)
return paths_by_status, paths, os.path.join(directory, "geminiStatus.lock")
def migrate_legacy_alive_rate_limited():
paths_by_status, _, lock_path = _gemini_status_layout()
alive_path = paths_by_status["VALID"]
limited_path = paths_by_status["VALID_RATE_LIMITED"]
lock = acquire_file_lock(lock_path, timeout_sec=30)
try:
if not os.path.lexists(alive_path):
return
require_private_file(alive_path)
keep = []
moved = {}
for line in iter_bounded_text_lines(alive_path):
key = key_from_line(line)
if key and line.rstrip("\r\n").endswith(":RATE_LIMITED"):
moved.setdefault(key, line if line.endswith("\n") else f"{line}\n")
else:
keep.append(line)
if not moved:
return
if os.path.lexists(limited_path):
require_private_file(limited_path)
existing = _normalized_status_lines(list(iter_bounded_text_lines(limited_path)))
limited = []
published = set()
for line in existing:
key = key_from_line(line)
if key in moved:
if key in published:
continue
published.add(key)
limited.append(line)
for key, line in moved.items():
if key not in published:
limited.append(line)
published.add(key)
_validate_status_snapshot(limited_path, limited)
_validate_status_snapshot(alive_path, keep)
# Make every moved key durable before publishing the source snapshot
# that removes it. An interruption can therefore only leave duplicates.
_replace_status_snapshot(limited_path, limited)
confirmed = {key: 0 for key in moved}
for line in iter_bounded_text_lines(limited_path):
key = key_from_line(line)
if key in confirmed:
confirmed[key] += 1
if any(count != 1 for count in confirmed.values()):
raise RuntimeError("Gemini legacy rate-limited status publication was incomplete")
_replace_status_snapshot(alive_path, keep)
finally:
release_file_lock(lock, lock_path)
def load_proxies(proxy_file):
if not os.path.exists(proxy_file):
print(f"Info: {proxy_file} not found. Requests will go directly.")
return None
proxies = []
with open(proxy_file, "r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line:
continue
try:
ip, port, login, password = line.split(":")
proxy_url = f"http://{login}:{password}@{ip}:{port}"
proxies.append({"http": proxy_url, "https": proxy_url})
except ValueError:
print(f"Warning: bad proxy format: {line}. Skipping.")
if not proxies:
print(f"Warning: {proxy_file} is empty. Requests will go directly.")
return None
print(f"Loaded proxies: {len(proxies)}")
return cycle(proxies)
def status_file_line(key, result, status):
if status in ("VALID", "VALID_RATE_LIMITED"):
models_str = ",".join(result.get("notable_models", [])) or "models-only"
probe_status = result.get("probe", {}).get("status", "not_probed")
return f"{key}:[{models_str}]:{result.get('model_class', 'unknown')}:{probe_status}\n"
message = (result.get("error", {}).get("message") or "").replace("\n", " ")[:300]
return f"{key}\t{status}\t{message}\n"
def _normalized_status_lines(lines):
return [line if line.endswith("\n") else f"{line}\n" for line in lines]
def _validate_status_snapshot(path, lines):
max_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_BYTES", 32 * 1024 * 1024))
max_items = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_ITEMS", 100000))
max_line_bytes = max(1, env_int("KEYCHECK_INPUT_LIST_MAX_LINE_BYTES", 8192))
if len(lines) > max_items:
raise RuntimeError(f"Gemini status file exceeds its item bound: {path}")
total = 0
for index, line in enumerate(lines, 1):
encoded = line.encode("utf-8")
if len(encoded) > max_line_bytes:
raise RuntimeError(f"Gemini status line exceeds its byte bound: {path}:{index}")
total += len(encoded)
if total > max_bytes:
raise RuntimeError(f"Gemini status file exceeds its aggregate byte bound: {path}")
def _replace_status_snapshot(path, lines):
with private_atomic_writer(path, binary=True, suffix=".status.tmp") as handle:
for line in lines:
handle.write(line.encode("utf-8"))
def append_status_file(key, result):
if keycheck_input_mode() == 'postgres':
return
paths_by_status, paths, lock_path = _gemini_status_layout()
lock = acquire_file_lock(lock_path, timeout_sec=30)
try:
status = effective_status(result)
target_path = paths_by_status.get(status, paths_by_status["UNKNOWN"])
new_line = status_file_line(key, result, status)
snapshots = {}
for path in paths:
if os.path.lexists(path):
reject_reparse_components(path)
snapshots[path] = _normalized_status_lines(list(iter_bounded_text_lines(path)))
rewritten = {
path: [line for line in lines if key_from_line(line) != key]
for path, lines in snapshots.items()
}
rewritten[target_path].insert(0, new_line)
for path, lines in rewritten.items():
_validate_status_snapshot(path, lines)
# Publish the new classification before removing any old copies. A
# failure after this point can leave duplicates, but never no status.
_replace_status_snapshot(target_path, rewritten[target_path])
for path in paths:
if path == target_path or rewritten[path] == snapshots[path]:
continue
_replace_status_snapshot(path, rewritten[path])
finally:
release_file_lock(lock, lock_path)
def append_checked_file(key, result):
if keycheck_input_mode() == 'postgres':
return
append_checked(CHECKED_FILE, key, effective_status(result))
def is_gemini_detector(detector):
return str(detector or "").lower() in GEMINI_DETECTOR_NAMES
def custom_detector_name(data):
if not isinstance(data, dict):
return ""
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
name = extra.get("name") or ""
if str(data.get("DetectorName") or "").lower() == "customregex" and is_gemini_detector(name):
return name
return ""
def detector_name_from_finding(data):
if not isinstance(data, dict):
return ""
if is_gemini_detector(data.get("DetectorName")):
return data.get("DetectorName")
custom_name = custom_detector_name(data)
if custom_name:
return custom_name
# Old wrapped format from earlier scanner versions.
if is_gemini_detector(data.get("detector")):
return data.get("detector")
finding = data.get("finding")
if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")):
return finding.get("DetectorName")
custom_name = custom_detector_name(finding)
if custom_name:
return custom_name
return ""
def extract_key_from_finding(data):
if is_gemini_detector(data.get("DetectorName")):
return data.get("Raw") or data.get("RawV2")
if custom_detector_name(data):
return data.get("Raw") or data.get("RawV2")
# Old wrapped format from earlier scanner versions.
if is_gemini_detector(data.get("detector")):
return data.get("raw") or data.get("raw_v2")
finding = data.get("finding")
if isinstance(finding, dict) and is_gemini_detector(finding.get("DetectorName")):
return finding.get("Raw") or finding.get("RawV2")
if custom_detector_name(finding):
return finding.get("Raw") or finding.get("RawV2")
return None
def iter_candidate_keys(input_file, plain_files):
for item in iter_findings(input_file, ["GoogleAI", "GoogleAIStudio", "CustomRegex"]):
finding = item.get("finding") or {}
key = item.get("raw") or extract_key_from_finding(finding)
if key and GEMINI_KEY_REGEX.fullmatch(key) and detector_name_from_finding(finding):
yield item.get("source") or input_file, key, finding
if keycheck_input_mode() == 'postgres':
return
for path in plain_files:
if not os.path.exists(path):
print(f"Info: plain input {path} not found. Skipping.")
continue
try:
keys = set()
for line in iter_bounded_text_lines(path):
keys.update(GEMINI_KEY_REGEX.findall(line))
except (OSError, RuntimeError) as e:
print(f"Warning: cannot read {path}: {e}")
continue
for idx, key in enumerate(sorted(keys), 1):
yield f"{path}:plain:{idx}", key, {}
def parse_error_response(response):
try:
payload = response.json()
except json.JSONDecodeError:
payload = {}
error = payload.get("error", {}) if isinstance(payload, dict) else {}
return {
"http_status": response.status_code,
"code": error.get("code", response.status_code),
"status": error.get("status", ""),
"message": error.get("message", response.text[:500]),
}
def classify_error(error):
http_status = int(error.get("http_status") or 0)
status = str(error.get("status") or "").lower()
message = str(error.get("message") or "").lower()
if "reported as leaked" in message or "leaked" in message:
return "LEAKED_REVOKED"
if "api key expired" in message or "expired" in message:
return "EXPIRED"
if "api key not valid" in message or "invalid api key" in message:
return "INVALID"
if "has not been used" in message or "it is disabled" in message or "api is disabled" in message:
return "API_DISABLED"
if "requests to this api" in message and "blocked" in message:
return "RESTRICTED"
if "api key restrictions" in message or "permission_denied" in status:
return "RESTRICTED"
if http_status == 429 or "resource_exhausted" in status or "quota" in message:
return "RATE_LIMITED"
if http_status in (400, 401):
return "INVALID"
if http_status == 403:
return "RESTRICTED"
return "UNKNOWN"
def fetch_models(key, proxy, timeout, debug=False):
try:
response = requests.get(MODELS_URL, params={"key": key}, proxies=proxy, timeout=timeout)
except requests.RequestException as e:
return {
"status": "NETWORK_ERROR",
"error": {"message": str(e)},
"models": [],
"model_infos": [],
}
if debug:
print(f" DEBUG /models: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}")
if response.status_code != 200:
error = parse_error_response(response)
return {
"status": classify_error(error),
"error": error,
"models": [],
"model_infos": [],
}
payload = response.json()
model_infos = payload.get("models", [])
models = sorted({m.get("name", "").replace("models/", "") for m in model_infos if m.get("name")})
return {
"status": "VALID",
"error": {},
"models": models,
"model_infos": model_infos,
}
def supported_methods_by_model(model_infos):
output = {}
for model in model_infos:
name = model.get("name", "").replace("models/", "")
if not name:
continue
output[name] = sorted(model.get("supportedGenerationMethods", []))
return output
def classify_models(models, methods_by_model):
notable = []
lower_models = {m.lower(): m for m in models}
for marker in MODEL_PRIORITY:
for lower, original in lower_models.items():
if marker in lower and original not in notable:
notable.append(original)
generation_models = sorted([
model for model, methods in methods_by_model.items()
if "generateContent" in methods
])
if any("gemini-2.5-pro" in m.lower() for m in generation_models):
model_class = "pro_generation"
elif generation_models:
model_class = "generation"
elif models:
model_class = "models_only"
else:
model_class = "no_models"
return notable[:20], generation_models, model_class
def choose_probe_model(generation_models):
available = set(generation_models)
for model in PROBE_MODEL_PRIORITY:
if model in available:
return model
return generation_models[0] if generation_models else None
def probe_generation(key, model, proxy, timeout, debug=False):
if not model:
return {"status": "NO_GENERATION_MODEL", "model": None}
url = f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent"
headers = {"x-goog-api-key": key, "Content-Type": "application/json"}
payload = {
"contents": [{"parts": [{"text": "ping"}]}],
"generationConfig": {"maxOutputTokens": 1},
}
try:
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as e:
return {"status": "NETWORK_ERROR", "model": model, "error": {"message": str(e)}}
if debug:
print(f" DEBUG probe {model}: HTTP {response.status_code}: {redact_key_text(response.text[:500], key)}")
if response.status_code == 200:
return {"status": "GENERATION_OK", "model": model}
error = parse_error_response(response)
return {"status": classify_error(error), "model": model, "error": error}
def check_key(key, proxy, args):
result = fetch_models(key, proxy, args.timeout, args.debug)
result = redact_result_text(result, key)
result.update({
"checked_at": now_iso(),
"key_masked": mask_key(key),
"model_count": len(result.get("models", [])),
})
if result["status"] != "VALID":
result["notable_models"] = []
result["generation_models"] = []
result["model_class"] = "none"
return result
methods_by_model = supported_methods_by_model(result.get("model_infos", []))
notable, generation_models, model_class = classify_models(result["models"], methods_by_model)
result["methods_by_model"] = methods_by_model
result["notable_models"] = notable
result["generation_models"] = generation_models[:50]
result["model_class"] = model_class
result["billing_status"] = "unknown"
if args.probe_generation:
probe_model = choose_probe_model(generation_models)
result["probe"] = redact_result_text(probe_generation(key, probe_model, proxy, args.timeout, args.debug), key)
else:
result["probe"] = {"status": "not_probed", "model": None}
return result
def print_result(index, source, key, result):
print(f"\n[{index}] Candidate {mask_key(key)} from {source}")
print(f" STATUS: {result['status']}")
if result["status"] == "VALID":
print(f" MODELS: {result.get('model_count', 0)} total; class={result.get('model_class')}")
notable = result.get("notable_models", [])[:8]
if notable:
print(f" NOTABLE: {', '.join(notable)}")
probe = result.get("probe", {})
print(f" PROBE: {probe.get('status')} ({probe.get('model')})")
if effective_status(result) == "VALID_RATE_LIMITED":
print(f" OUT: {STATUS_FILES['VALID_RATE_LIMITED']}")
else:
error = result.get("error", {})
message = (error.get("message") or "").replace("\n", " ")[:300]
if message:
print(f" MESSAGE: {message}")
print(f" OUT: {STATUS_FILES.get(result['status'], STATUS_FILES['UNKNOWN'])}")
def parse_args():
parser = argparse.ArgumentParser(description="Gemini / Google AI API key classifier")
parser.add_argument("--input", default=DEFAULT_INPUT_FILE)
parser.add_argument("--plain", action="append", default=None, help="Plain text file with Gemini keys. Can be repeated.")
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=20)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--probe-generation", action="store_true", help="Optionally call generateContent, preferring gemini-3.1-pro-preview when available.")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
parser.add_argument("--debug", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
plain_files = args.plain if args.plain is not None else DEFAULT_PLAIN_INPUT_FILES
ensure_output_files()
print("--- Gemini key checker ---")
print("Default mode: /models only. Use --probe-generation for runtime/billing probe.")
print(f"Workspace: {SCRIPT_DIR}")
proxy_cycler = load_proxies(args.proxy_file)
checked_statuses = load_checked_statuses()
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
known_keys = set(known_statuses)
retry_statuses = set()
if args.retry_limited:
retry_statuses.update(("RATE_LIMITED", "VALID_RATE_LIMITED"))
if args.retry_unknown:
retry_statuses.add("UNKNOWN")
if args.retry_network:
retry_statuses.add("NETWORK_ERROR")
if args.retry_valid:
retry_statuses.update(("VALID", "VALID_RATE_LIMITED"))
print(f"Loaded known keys: {len(known_keys)}; checked records: {len(checked_statuses)}")
seen_this_run = set()
processed = 0
skipped = 0
for source, key, finding in iter_candidate_keys(args.input, plain_files):
if keycheck_input_mode() != 'postgres' and key in seen_this_run:
cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN'
detector = detector_name_from_finding(finding) or "GoogleAI"
record_cached_keycheck_occurrence(SERVICE, key, cached_status, source, finding, detector)
skipped += 1
continue
seen_this_run.add(key)
detector = detector_name_from_finding(finding) or "GoogleAI"
if should_skip_key(
key, checked_statuses, known_keys, args, retry_statuses,
service=SERVICE, source=source, finding=finding, detector=detector,
known_statuses=known_statuses,
):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
proxy = next(proxy_cycler) if proxy_cycler else None
result = check_key(key, proxy, args)
result["source"] = source
event_result = {**result, "status": effective_status(result)}
write_keycheck_event(SERVICE, RESULTS_FILE, key, event_result, source, finding, detector)
print_result(processed, source, key, result)
append_status_file(key, result)
append_checked_file(key, result)
record_validation_result(SERVICE, key, {**result, "status": effective_status(result)}, source, finding, detector)
known_keys.add(key)
checked_statuses[key] = effective_status(result)
# Small pause helps when many keys hit the same API/proxy.
time.sleep(0.1)
print("\n--- Done ---")
print(f"Processed: {processed}")
print(f"Skipped: {skipped}")
print(f"Results: {RESULTS_FILE}")
if __name__ == "__main__":
main()
+212
View File
@@ -0,0 +1,212 @@
import sys
sys.dont_write_bytecode = True
import argparse
import json
import os
import re
import time
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
read_plain_keys,
recover_status_transaction,
request_error_message,
record_validation_result,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
SERVICE = "github"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
PLAIN_FILE = os.path.join(OUTPUT_DIR, "github.txt")
CHECKED_FILE = os.path.join(OUTPUT_DIR, "githubChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "githubResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "githubAlive.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "githubDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "githubRestricted.txt"),
"RATE_LIMITED": os.path.join(OUTPUT_DIR, "githubRateLimited.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "githubNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "githubUnknown.txt"),
"REFRESH_TOKEN": os.path.join(OUTPUT_DIR, "githubRefreshToken.txt"),
}
GITHUB_TOKEN_RE = re.compile(r"\b(?:gh[pousr]_[A-Za-z0-9_]{20,}|github_pat_[A-Za-z0-9_]{20,})\b")
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), PLAIN_FILE])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def extract_candidates(input_file, plain_files):
seen_plain = set()
for item in iter_findings(input_file, ["Github", "GitHubOauth2"]):
text = "\n".join(str(value or "") for value in [item.get("raw"), item.get("raw_v2")])
for match in GITHUB_TOKEN_RE.findall(text):
yield match, item["source"], item["finding"]
for item in read_plain_keys(plain_files, GITHUB_TOKEN_RE):
key = item["key"]
if key not in seen_plain:
seen_plain.add(key)
yield key, item["source"], {}
def is_rate_limited(response):
remaining = response.headers.get("X-RateLimit-Remaining")
return response.status_code in (403, 429) and remaining == "0"
def check_token(token, proxy, timeout):
if token.startswith("ghr_"):
return {
"status": "REFRESH_TOKEN",
"message": "GitHub refresh tokens cannot be checked directly as bearer API tokens",
}
if token.startswith("ghs_"):
url = "https://api.github.com/installation/repositories"
token_kind = "installation"
else:
url = "https://api.github.com/user"
token_kind = "user"
headers = {
"Authorization": f"Bearer {token}",
"Accept": "application/vnd.github+json",
"X-GitHub-Api-Version": "2022-11-28",
"User-Agent": "local-keycheck-github",
}
try:
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc), "token_kind": token_kind}
message = request_error_message(response)
scopes = response.headers.get("X-OAuth-Scopes", "")
accepted_scopes = response.headers.get("X-Accepted-OAuth-Scopes", "")
rate_remaining = response.headers.get("X-RateLimit-Remaining", "")
rate_reset = response.headers.get("X-RateLimit-Reset", "")
if response.status_code == 200:
payload = response.json()
extra = {
"token_kind": token_kind,
"scopes": scopes,
"accepted_scopes": accepted_scopes,
"rate_remaining": rate_remaining,
"rate_reset": rate_reset,
}
if token_kind == "installation":
extra["repo_count"] = payload.get("total_count")
return {"status": "VALID", "message": "installation token accepted", **extra}
return {
"status": "VALID",
"message": "user token accepted",
"login": payload.get("login"),
"account_type": payload.get("type"),
**extra,
}
if response.status_code == 401:
return {"status": "DEAD", "http_status": 401, "message": message, "token_kind": token_kind}
if is_rate_limited(response):
return {"status": "RATE_LIMITED", "http_status": response.status_code, "message": message, "token_kind": token_kind, "rate_reset": rate_reset}
if response.status_code == 403:
return {"status": "RESTRICTED", "http_status": 403, "message": message, "token_kind": token_kind, "scopes": scopes}
if response.status_code in (404, 422):
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "token_kind": token_kind}
if response.status_code >= 500:
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "token_kind": token_kind}
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "token_kind": token_kind}
def write_result(key, result, source, finding):
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding)
extra = result.get("login") or result.get("repo_count") or result.get("token_kind") or source
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra,
)
record_validation_result(SERVICE, key, result, source, finding)
def parse_args():
parser = argparse.ArgumentParser(description="GitHub token checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=20)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-restricted", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = set()
if args.retry_network:
retry_statuses.add("NETWORK")
if args.retry_limited:
retry_statuses.add("RATE_LIMITED")
if args.retry_unknown:
retry_statuses.add("UNKNOWN")
if args.retry_restricted:
retry_statuses.add("RESTRICTED")
plain_files = args.plain or [PLAIN_FILE]
processed = 0
skipped = 0
for key, source, finding in extract_candidates(args.input, plain_files):
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] GitHub candidate {mask_secret(key)} from {source}")
proxy = next(proxy_cycler) if proxy_cycler else None
result = check_token(key, proxy, args.timeout)
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
time.sleep(0.1)
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+178
View File
@@ -0,0 +1,178 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
import re
import time
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
read_plain_keys,
recover_status_transaction,
request_error_message,
record_validation_result,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(SCRIPT_DIR)
SERVICE = "gitlab"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
PLAIN_FILE = os.path.join(OUTPUT_DIR, "gitlab.txt")
CHECKED_FILE = os.path.join(OUTPUT_DIR, "gitlabChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "gitlabResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "gitlabAlive.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "gitlabDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "gitlabRestricted.txt"),
"RATE_LIMITED": os.path.join(OUTPUT_DIR, "gitlabRateLimited.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "gitlabNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "gitlabUnknown.txt"),
}
GITLAB_TOKEN_RE = re.compile(r"\b(?:glpat|gloas|glcbt|glimt|glrt|glft|glsoat)-[A-Za-z0-9_\-=]{20,}\b")
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values(), PLAIN_FILE])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def extract_candidates(input_file, plain_files):
seen_plain = set()
for item in iter_findings(input_file, ["Gitlab"]):
text = "\n".join(str(value or "") for value in [item.get("raw"), item.get("raw_v2")])
for match in GITLAB_TOKEN_RE.findall(text):
yield match, item["source"], item["finding"]
for item in read_plain_keys(plain_files, GITLAB_TOKEN_RE):
key = item["key"]
if key not in seen_plain:
seen_plain.add(key)
yield key, item["source"], {}
def check_token(token, base_url, proxy, timeout):
base_url = base_url.rstrip("/")
url = f"{base_url}/api/v4/user"
headers = {"Authorization": f"Bearer {token}", "User-Agent": "local-keycheck-gitlab"}
try:
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc), "base_url": base_url}
message = request_error_message(response)
retry_after = response.headers.get("Retry-After", "")
rate_remaining = response.headers.get("RateLimit-Remaining") or response.headers.get("X-RateLimit-Remaining") or ""
if response.status_code == 200:
payload = response.json()
return {
"status": "VALID",
"message": "token accepted",
"username": payload.get("username"),
"name": payload.get("name"),
"user_id": payload.get("id"),
"base_url": base_url,
"rate_remaining": rate_remaining,
}
if response.status_code == 401:
return {"status": "DEAD", "http_status": 401, "message": message, "base_url": base_url}
if response.status_code == 403:
# TruffleHog treats 403 as a live token with insufficient scope or blocked account.
return {"status": "RESTRICTED", "http_status": 403, "message": message, "base_url": base_url}
if response.status_code == 429:
return {"status": "RATE_LIMITED", "http_status": 429, "message": message, "base_url": base_url, "retry_after": retry_after}
if response.status_code >= 500:
return {"status": "NETWORK", "http_status": response.status_code, "message": message, "base_url": base_url}
return {"status": "UNKNOWN", "http_status": response.status_code, "message": message, "base_url": base_url}
def write_result(key, result, source, finding):
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding)
extra = result.get("username") or result.get("user_id") or source
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, result["status"], result.get("message", ""), extra,
)
record_validation_result(SERVICE, key, result, source, finding)
def parse_args():
parser = argparse.ArgumentParser(description="GitLab token checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--base-url", default="https://gitlab.com")
parser.add_argument("--timeout", type=int, default=20)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-restricted", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = set()
if args.retry_network:
retry_statuses.add("NETWORK")
if args.retry_limited:
retry_statuses.add("RATE_LIMITED")
if args.retry_unknown:
retry_statuses.add("UNKNOWN")
if args.retry_restricted:
retry_statuses.add("RESTRICTED")
plain_files = args.plain or [PLAIN_FILE]
processed = 0
skipped = 0
for key, source, finding in extract_candidates(args.input, plain_files):
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] GitLab candidate {mask_secret(key)} from {source}")
proxy = next(proxy_cycler) if proxy_cycler else None
result = check_token(key, args.base_url, proxy, args.timeout)
print(f" STATUS: {result['status']} | {str(result.get('message', ''))[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
time.sleep(0.1)
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+275
View File
@@ -0,0 +1,275 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
import re
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl,
classify_common_http_status,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
read_plain_keys,
record_validation_result,
recover_status_transaction,
request_error_message,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
SERVICE = "groq"
DETECTOR = "Groq"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "groqChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "groqResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "groqAlive.txt"),
"NO_BALANCE": os.path.join(OUTPUT_DIR, "groqNoBalance.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "groqDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "groqRestricted.txt"),
"LIMITED": os.path.join(OUTPUT_DIR, "groqLimited.txt"),
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "groqNoContext.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "groqNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "groqUnknown.txt"),
}
GROQ_KEY_REGEX = re.compile(r"\bgsk_[A-Za-z0-9_-]{20,}\b")
MODELS_URL = "https://api.groq.com/openai/v1/models"
CHAT_URL = "https://api.groq.com/openai/v1/chat/completions"
CHAT_MODEL_PRIORITY = (
"llama-3.1-8b-instant",
"llama-3.3-70b-versatile",
"llama3-8b-8192",
"llama3-70b-8192",
"mixtral-8x7b-32768",
"gemma2-9b-it",
)
NO_BALANCE_MARKERS = (
"quota",
"billing",
"balance",
"credit",
"payment",
"insufficient",
"depleted",
)
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def extract_candidates(input_file, plain_files):
seen_plain = set()
for item in iter_findings(input_file, [DETECTOR]):
key = item["raw"]
if key and GROQ_KEY_REGEX.fullmatch(key):
yield key, item["source"], item["finding"]
for item in read_plain_keys(plain_files, GROQ_KEY_REGEX):
key = item["key"]
if key not in seen_plain:
seen_plain.add(key)
yield key, item["source"], {}
def classify_groq_response(response):
message = request_error_message(response).lower()
if response.status_code == 401:
return "DEAD"
if response.status_code == 403:
return "RESTRICTED"
if response.status_code == 429:
if any(marker in message for marker in NO_BALANCE_MARKERS):
return "NO_BALANCE"
return "LIMITED"
return classify_common_http_status(response.status_code)
def notable_models(payload):
models = payload.get("data", []) if isinstance(payload, dict) else []
ids = []
for item in models:
if isinstance(item, dict) and item.get("id"):
ids.append(str(item.get("id")))
priority = []
for marker in ("llama", "mixtral", "gemma", "whisper"):
for model in ids:
if marker in model.lower() and model not in priority:
priority.append(model)
return priority[:20], len(ids), ids
def choose_chat_model(model_ids):
model_ids = [str(model or "") for model in model_ids if model]
by_lower = {model.lower(): model for model in model_ids}
for model in CHAT_MODEL_PRIORITY:
if model.lower() in by_lower:
return by_lower[model.lower()]
for marker in ("llama", "mixtral", "gemma"):
for model in model_ids:
lowered = model.lower()
if marker in lowered and "whisper" not in lowered and "guard" not in lowered:
return model
return ""
def probe_chat_completion(key, model, proxy, timeout, debug=False):
if not model:
return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""}
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1}
try:
response = requests.post(CHAT_URL, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc), "model": model}
if debug:
print(f" DEBUG chat ping {model}: HTTP {response.status_code}: {response.text[:500].replace(key, '***REDACTED***')}")
if response.status_code == 200:
return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model}
return {
"status": classify_groq_response(response),
"http_status": response.status_code,
"message": request_error_message(response).replace(key, "***REDACTED***"),
"model": model,
}
def check_key(key, proxy, timeout, debug=False):
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
try:
response = requests.get(MODELS_URL, headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc)}
if debug:
print(f" DEBUG /models: HTTP {response.status_code}: {response.text[:500].replace(key, '***REDACTED***')}")
if response.status_code == 200:
try:
payload = response.json()
except ValueError:
payload = {}
models, model_count, model_ids = notable_models(payload)
chat_model = choose_chat_model(model_ids)
probe = probe_chat_completion(key, chat_model, proxy, timeout, debug)
if probe.get("status") != "GENERATION_OK":
return {
"status": probe.get("status") or "UNKNOWN",
"message": probe.get("message", ""),
"model_count": model_count,
"models": models,
"llm_probe_status": probe.get("status"),
"llm_probe_model": probe.get("model", chat_model),
"llm_probe_http_status": probe.get("http_status"),
}
return {
"status": "VALID",
"message": f"chat ping ok; model={chat_model}; models={model_count}",
"model_count": model_count,
"models": models,
"llm_probe_status": probe.get("status"),
"llm_probe_model": chat_model,
}
return {
"status": classify_groq_response(response),
"http_status": response.status_code,
"message": request_error_message(response).replace(key, "***REDACTED***"),
}
def write_result(key, result, source, finding):
status = result.get("status") or "UNKNOWN"
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
extra = ",".join(result.get("models") or [])[:500] if status == "VALID" else source
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra,
)
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
def parse_args():
parser = argparse.ArgumentParser(description="Groq key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=15)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-restricted", action="store_true")
parser.add_argument("--retry-no-balance", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
parser.add_argument("--debug", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = set()
if args.retry_network:
retry_statuses.add("NETWORK")
if args.retry_limited:
retry_statuses.add("LIMITED")
if args.retry_unknown:
retry_statuses.add("UNKNOWN")
if args.retry_restricted:
retry_statuses.add("RESTRICTED")
if args.retry_no_balance:
retry_statuses.add("NO_BALANCE")
if args.retry_valid:
retry_statuses.add("VALID")
print("--- Groq key checker ---")
processed = 0
skipped = 0
for key, source, finding in extract_candidates(args.input, args.plain):
if should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] Groq candidate {mask_secret(key)} from {source}")
proxy = next(proxy_cycler) if proxy_cycler else None
result = check_key(key, proxy, args.timeout, args.debug)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
@@ -0,0 +1,142 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
import re
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl, classify_common_http_status, commit_status_transaction,
default_input_file, default_proxy_file, ensure_output_files, iter_findings,
keycheck_input_mode,
load_checked_statuses, load_known_keys, load_proxies, mask_secret,
read_plain_keys, record_validation_result, recover_status_transaction,
request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event,
)
SERVICE = "huggingface"
DETECTOR_NAMES = ["HuggingFace", "Huggingface"]
DETECTOR = "HuggingFace"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "huggingfaceChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "huggingfaceResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "huggingfaceAlive.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "huggingfaceDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "huggingfaceRestricted.txt"),
"LIMITED": os.path.join(OUTPUT_DIR, "huggingfaceLimited.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "huggingfaceNetwork.txt"),
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "huggingfaceNoContext.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "huggingfaceUnknown.txt"),
}
KEY_REGEX = re.compile(r"\bhf_[A-Za-z0-9]{20,}\b")
WHOAMI_URL = "https://huggingface.co/api/whoami-v2"
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def iter_candidate_decisions(input_file, plain_files):
seen_plain = set()
for item in iter_findings(input_file, DETECTOR_NAMES):
key = item.get("credential_secret_text") or item["raw"]
if key:
yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key))
for item in read_plain_keys(plain_files, KEY_REGEX):
key = item["key"]
if key not in seen_plain:
seen_plain.add(key)
yield key, item["source"], {}, True
def extract_candidates(input_file, plain_files):
for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files):
if valid_format:
yield key, source, finding
def check_key(key, proxy, timeout):
try:
response = requests.get(WHOAMI_URL, headers={"Authorization": f"Bearer {key}"}, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc)}
if response.status_code == 200:
data = response.json() if response.text else {}
return {"status": "VALID", "message": "whoami accepted", "username": data.get("name") or data.get("fullname") or ""}
status = "RESTRICTED" if response.status_code == 403 else classify_common_http_status(response.status_code)
return {"status": status, "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***")}
def write_result(key, result, source, finding):
status = result.get("status") or "UNKNOWN"
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), source,
)
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
def parse_args():
parser = argparse.ArgumentParser(description="HuggingFace key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=15)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-restricted", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = set()
if args.retry_network: retry_statuses.add("NETWORK")
if args.retry_limited: retry_statuses.add("LIMITED")
if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
if args.retry_restricted: retry_statuses.add("RESTRICTED")
processed = skipped = 0
print("--- HuggingFace key checker ---")
postgres_mode = keycheck_input_mode() == "postgres"
for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain):
if not valid_format and not postgres_mode:
skipped += 1
continue
if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] HuggingFace candidate {mask_secret(key)} from {source}")
result = (
check_key(key, next(proxy_cycler) if proxy_cycler else None, args.timeout)
if valid_format else
{"status": "NO_CONTEXT", "message": "candidate does not match canonical Hugging Face token format"}
)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
File diff suppressed because it is too large Load Diff
+499
View File
@@ -0,0 +1,499 @@
import sys
sys.dont_write_bytecode = True
import argparse
import json
import os
import re
from urllib.parse import urlparse
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
combined_provider_routing_hint,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
keycheck_input_mode,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
provider_routing_database_failed,
read_plain_keys,
record_validation_result,
recover_status_transaction,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
from keycheckers.provider_resolution import resolve_provider_key
SERVICE = "kimi"
DETECTOR = "KimiMoonshot"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "kimiChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "kimiResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "kimiAlive.txt"),
"NO_BALANCE": os.path.join(OUTPUT_DIR, "kimiNoBalance.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "kimiDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "kimiRestricted.txt"),
"LIMITED": os.path.join(OUTPUT_DIR, "kimiLimited.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "kimiNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "kimiUnknown.txt"),
}
KIMI_DETECTOR_NAMES = {"kimimoonshot", "moonshotai", "moonshot", "kimi"}
KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"}
QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"}
DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"}
GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"}
AMBIGUOUS_QWEN_DEEPSEEK_HINT = "ambiguous_qwen_deepseek"
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment"
CANDIDATE_PROVIDER_ROUTE_FIELD = "_keycheck_candidate_provider_route"
KIMI_KEY_MAX_BYTES = 512
KIMI_KEY_REGEX = re.compile(
r"(?<![A-Za-z0-9_-])sk-[A-Za-z0-9][A-Za-z0-9_-]{20,505}(?![A-Za-z0-9_-])"
)
FOREIGN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
DEFAULT_BASE_URLS = (
"https://api.moonshot.ai/v1",
"https://api.moonshot.cn/v1",
)
QWEN_CONTEXT_REGEX = re.compile(
r"(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)",
re.IGNORECASE,
)
DEEPSEEK_CONTEXT_REGEX = re.compile(
r"(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)", re.IGNORECASE,
)
KIMI_CONTEXT_REGEX = re.compile(
r"(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))",
re.IGNORECASE,
)
def normalize_base_url(value):
return str(value or "").strip().rstrip("/")
def split_csv(value):
if not value:
return []
if isinstance(value, str):
return [item.strip() for item in value.split(",") if item.strip()]
return [str(item).strip() for item in value if str(item).strip()]
def unique_ordered(values):
output = []
seen = set()
for value in values:
normalized = normalize_base_url(value)
if normalized and normalized not in seen:
seen.add(normalized)
output.append(normalized)
return output
def endpoint_label(base_url):
parsed = urlparse(base_url)
return parsed.netloc or base_url
def finding_detector_names(finding):
if not isinstance(finding, dict):
return set()
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
names = {
str(finding.get("DetectorName") or finding.get("detector") or "").strip().lower(),
str(extra.get("name") or "").strip().lower(),
}
return {name for name in names if name}
def finding_has_detector(finding, detector_names):
return bool(finding_detector_names(finding) & set(detector_names))
def key_from_text(*values):
for value in values:
for match in KIMI_KEY_REGEX.finditer(str(value or "")):
key = match.group(0)
if not key.startswith(FOREIGN_KEY_PREFIXES):
return key
return ""
def key_rejection_reason(key):
value = str(key or "")
try:
key_bytes = len(value.encode("utf-8", errors="strict"))
except UnicodeEncodeError:
return "candidate is not valid UTF-8"
if key_bytes > KIMI_KEY_MAX_BYTES:
return f"candidate exceeds the {KIMI_KEY_MAX_BYTES}-byte key limit"
if value.count("sk-") != 1:
return "candidate contains multiple key prefixes"
if not KIMI_KEY_REGEX.fullmatch(value) or value.startswith(FOREIGN_KEY_PREFIXES):
return "candidate does not match the bounded Kimi/Moonshot key format"
return ""
def finding_provider_routing_hint(finding):
if not isinstance(finding, dict):
return ""
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
persisted_hint = context.get("provider_hint")
if context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE and persisted_hint in (
*GENERIC_SK_PROVIDERS, AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT,
):
return persisted_hint
parts = [str(context.get(key) or "") for key in ("nearby", "file")]
metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {}
data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {}
for details in data.values():
if isinstance(details, dict):
parts.extend(str(details.get(key) or "") for key in ("file", "repository", "repo", "link", "image"))
text = "\n".join(parts)
evidence = set()
if QWEN_CONTEXT_REGEX.search(text) or finding_has_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES):
evidence.add("qwen")
if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES):
evidence.add("deepseek")
if KIMI_CONTEXT_REGEX.search(text) or finding_has_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES):
evidence.add("kimi")
if persisted_hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT:
evidence.update(("qwen", "deepseek"))
elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT:
evidence.update(GENERIC_SK_PROVIDERS)
elif persisted_hint in GENERIC_SK_PROVIDERS:
evidence.add(persisted_hint)
if len(evidence) > 1:
return (
AMBIGUOUS_QWEN_DEEPSEEK_HINT
if evidence == {"qwen", "deepseek"}
else AMBIGUOUS_GENERIC_SK_HINT
)
return next(iter(evidence)) if evidence else ""
def iter_candidate_keys(input_file, plain_files, trusted_retry_files=None):
detector_names = [
"KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi",
"kimimoonshot", "moonshotai", "moonshot", "kimi", "CustomRegex",
]
routing_decisions = {}
seen = set()
for item in iter_findings(input_file, detector_names):
finding = dict(item.get("finding") or {})
finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, None)
candidate_metadata = item.get("candidate_metadata")
persisted_route = ""
if keycheck_input_mode() == "postgres" and isinstance(candidate_metadata, dict):
persisted_route = str(candidate_metadata.get("provider_hint") or "").lower()
if persisted_route == SERVICE:
finding[CANDIDATE_PROVIDER_ROUTE_FIELD] = SERVICE
if persisted_route != SERVICE and not finding_has_detector(finding, KIMI_DETECTOR_NAMES):
continue
key = key_from_text(item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2"))
if not key or key_rejection_reason(key):
continue
if persisted_route == SERVICE:
hint = SERVICE
else:
local_hint = finding_provider_routing_hint(finding)
if key not in routing_decisions:
routing_decisions[key] = combined_provider_routing_hint(key, local_hint)
hint = routing_decisions[key]
if provider_routing_database_failed():
raise RuntimeError("provider routing evidence lookup failed closed")
if hint == "kimi" or (
keycheck_input_mode() == "postgres"
and hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT)
):
seen.add(key)
yield key, item.get("source") or input_file, finding
owned_retry_paths = {
os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values()
}
retry_files = [
path for path in (trusted_retry_files or [])
if os.path.normcase(os.path.abspath(path)) in owned_retry_paths
]
for item in read_plain_keys(retry_files, KIMI_KEY_REGEX):
key = item["key"]
if key_rejection_reason(key) or key in seen:
continue
hint = combined_provider_routing_hint(key, "kimi")
if provider_routing_database_failed():
raise RuntimeError("provider routing evidence lookup failed closed")
if hint == "kimi":
seen.add(key)
yield key, item["source"], {}
def redact_text(value, key):
text = str(value or "")[:1000]
if key:
text = text.replace(key, "***REDACTED***")
return KIMI_KEY_REGEX.sub("***REDACTED***", text)
def parse_error(response, key):
try:
payload = response.json()
except ValueError:
payload = {}
error = payload.get("error") if isinstance(payload, dict) else {}
if not isinstance(error, dict):
error = {}
message = error.get("message") or response.text[:500]
return {
"http_status": response.status_code,
"code": error.get("code") or error.get("type") or "",
"message": redact_text(message, key),
}
def classify_error(error):
http_status = int(error.get("http_status") or 0)
code = str(error.get("code") or "").lower()
message = str(error.get("message") or "").lower()
if http_status == 401 or any(marker in code for marker in ("invalid_authentication", "invalid_api_key")):
return "DEAD"
if http_status == 403:
return "RESTRICTED"
if http_status == 429:
if any(marker in code + " " + message for marker in ("quota", "balance", "payment")):
return "NO_BALANCE"
return "LIMITED"
if 500 <= http_status <= 599:
return "NETWORK"
return "UNKNOWN"
def check_base_url(key, base_url, proxy, timeout, debug=False):
url = f"{normalize_base_url(base_url)}/users/me/balance"
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
try:
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {
"status": "NETWORK", "base_url": base_url,
"region": endpoint_label(base_url), "message": redact_text(exc, key),
}
if debug:
print(
f" DEBUG {endpoint_label(base_url)} balance: HTTP {response.status_code}: "
f"{redact_text(response.text[:500], key)}"
)
if response.status_code == 200:
try:
payload = response.json()
data = payload.get("data") if isinstance(payload, dict) else None
if not isinstance(data, dict) or "available_balance" not in data:
raise ValueError("missing data.available_balance")
available = float(data.get("available_balance"))
voucher = float(data.get("voucher_balance", 0) or 0)
cash = float(data.get("cash_balance", 0) or 0)
except (TypeError, ValueError, json.JSONDecodeError) as exc:
return {
"status": "UNKNOWN", "base_url": base_url,
"region": endpoint_label(base_url),
"message": f"invalid balance response: {exc}",
}
status = "VALID" if available > 0 else "NO_BALANCE"
return {
"status": status,
"authenticated": True,
"base_url": base_url,
"region": endpoint_label(base_url),
"balance_usd": round(available, 6),
"voucher_balance_usd": round(voucher, 6),
"cash_balance_usd": round(cash, 6),
"message": f"available_balance=${available:.6f}",
}
error = parse_error(response, key)
return {
"status": classify_error(error),
"base_url": base_url,
"region": endpoint_label(base_url),
"http_status": response.status_code,
"error": error,
"message": error.get("message") or "",
}
def choose_final_status(attempts):
statuses = [attempt.get("status") for attempt in attempts]
for status in ("VALID", "NO_BALANCE", "LIMITED", "RESTRICTED", "UNKNOWN", "NETWORK"):
if status in statuses:
return status
return "DEAD"
def check_key(key, base_urls, proxy, timeout, debug=False):
rejection = key_rejection_reason(key)
if rejection:
return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True}
attempts = []
for base_url in base_urls:
result = check_base_url(key, base_url, proxy, timeout, debug)
attempts.append(result)
if result.get("status") in ("VALID", "NO_BALANCE"):
return {**result, "attempts": attempts}
status = choose_final_status(attempts)
selected = next((attempt for attempt in attempts if attempt.get("status") == status), {})
return {**selected, "status": status, "attempts": attempts}
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def write_result(key, result, source, finding):
status = result.get("status") or "UNKNOWN"
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, status,
result.get("message", ""), result.get("region") or source,
)
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
def retry_statuses_from_args(args):
statuses = set()
if args.retry_network:
statuses.add("NETWORK")
if args.retry_limited:
statuses.add("LIMITED")
if args.retry_unknown:
statuses.add("UNKNOWN")
if args.retry_restricted:
statuses.add("RESTRICTED")
if args.retry_no_balance:
statuses.add("NO_BALANCE")
if args.retry_valid:
statuses.add("VALID")
return statuses
def retry_input_files_from_args(args):
if args.recheck_all:
statuses = list(STATUS_FILES)
else:
statuses = [
status for flag, status in (
(args.retry_network, "NETWORK"),
(args.retry_limited, "LIMITED"),
(args.retry_unknown, "UNKNOWN"),
(args.retry_restricted, "RESTRICTED"),
(args.retry_no_balance, "NO_BALANCE"),
(args.retry_valid, "VALID"),
) if flag
]
return [STATUS_FILES[status] for status in statuses]
def base_urls_from_args(args):
custom = []
for value in args.base_url:
custom.extend(split_csv(value))
custom.extend(split_csv(os.getenv("KIMI_BASE_URLS") or os.getenv("MOONSHOT_BASE_URLS")))
defaults = [] if args.no_default_base_urls else DEFAULT_BASE_URLS
return unique_ordered([*custom, *defaults])
def parse_args():
parser = argparse.ArgumentParser(description="Kimi / Moonshot AI key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=15)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--base-url", action="append", default=[])
parser.add_argument("--no-default-base-urls", action="store_true")
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-restricted", action="store_true")
parser.add_argument("--retry-no-balance", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
parser.add_argument("--debug", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = retry_statuses_from_args(args)
retry_files = retry_input_files_from_args(args)
base_urls = base_urls_from_args(args)
if not base_urls:
raise SystemExit("No Kimi/Moonshot base URLs configured")
print("--- Kimi / Moonshot AI key checker ---")
print("base_urls: " + ", ".join(endpoint_label(url) for url in base_urls))
processed = 0
skipped = 0
for key, source, finding in iter_candidate_keys(args.input, args.plain, retry_files):
finding = dict(finding or {})
candidate_route = str(finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, "") or "").lower()
if should_skip_key(
key, checked, known, args, retry_statuses, service=SERVICE,
source=source, finding=finding, detector=DETECTOR,
):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] Kimi/Moonshot candidate {mask_secret(key)} from {source}")
proxy = next(proxy_cycler) if proxy_cycler else None
routing_hint = "kimi"
if keycheck_input_mode() == "postgres":
if candidate_route == SERVICE:
routing_hint = SERVICE
else:
routing_hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding))
if provider_routing_database_failed():
raise RuntimeError("provider routing evidence lookup failed closed")
if routing_hint in (AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT):
result = resolve_provider_key(
key, finding, proxy, args.timeout, args.debug,
hint=routing_hint, origin_service=SERVICE,
)
else:
result = check_key(key, base_urls, proxy, args.timeout, args.debug)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+572
View File
@@ -0,0 +1,572 @@
import sys
sys.dont_write_bytecode = True
import requests
import json
import os
import argparse
import re
from itertools import cycle
import time
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files as ensure_private_output_files,
iter_findings,
iter_bounded_text_lines,
keycheck_input_mode,
load_known_statuses,
private_atomic_writer,
record_cached_keycheck_occurrence,
record_validation_result,
recover_status_transaction,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
try:
sys.stdout.reconfigure(encoding='utf-8', errors='replace')
sys.stderr.reconfigure(encoding='utf-8', errors='replace')
except Exception:
pass
# --- Конфигурация ---
SERVICE = "openai"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
ALIVE_FILE = os.path.join(OUTPUT_DIR, "openaiAlive.txt")
DEAD_FILE = os.path.join(OUTPUT_DIR, "openaiDead.txt")
NETWORK_FILE = os.path.join(OUTPUT_DIR, "openaiNetwork.txt")
LIMITED_FILE = os.path.join(OUTPUT_DIR, "openaiLimited.txt")
RESTRICTED_FILE = os.path.join(OUTPUT_DIR, "openaiRestricted.txt")
UNKNOWN_FILE = os.path.join(OUTPUT_DIR, "openaiUnknown.txt")
NO_TARGET_FILE = os.path.join(OUTPUT_DIR, "openaiNoTarget.txt")
CHECKED_FILE = os.path.join(OUTPUT_DIR, "openaiChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "openaiResults.jsonl")
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
MODEL_PRIORITY_FOR_TEST = (
'gpt-5.6-sol',
'gpt-5.6',
'gpt-5.6-luna',
'gpt-5.6-terra',
'gpt-5',
'o3-pro',
'o3',
)
TARGET_MODELS = {'gpt-5', 'o3-pro', 'o3'}
NON_CHAT_MODEL_MARKERS = (
'embedding', 'image', 'audio', 'tts', 'transcribe', 'realtime', 'search', 'moderation',
)
PROBE_MAX_COMPLETION_TOKENS = 16
STATUS_FILES = [ALIVE_FILE, DEAD_FILE, NETWORK_FILE, LIMITED_FILE, RESTRICTED_FILE, UNKNOWN_FILE, NO_TARGET_FILE]
STATUS_BY_FILE = {
ALIVE_FILE: 'ALIVE',
DEAD_FILE: 'DEAD',
NETWORK_FILE: 'NETWORK',
LIMITED_FILE: 'LIMITED',
RESTRICTED_FILE: 'RESTRICTED',
UNKNOWN_FILE: 'UNKNOWN',
NO_TARGET_FILE: 'NO_TARGET_MODELS',
}
OPENAI_KEY_REGEX = re.compile(r'sk-[A-Za-z0-9_-]{20,}')
# --- Вспомогательные функции ---
def key_from_line(line):
line = line.strip()
if not line:
return None
if '\t' in line:
return line.split('\t', 1)[0].strip()
if ':[' in line:
return line.split(':[', 1)[0].strip()
match = OPENAI_KEY_REGEX.search(line)
if match:
return match.group(0)
return line.split()[0].strip()
def load_set_from_file(filepath):
if keycheck_input_mode() == 'postgres':
return set()
if not os.path.exists(filepath): return set()
return {key for key in (key_from_line(line) for line in iter_bounded_text_lines(filepath)) if key}
def iter_plain_openai_keys(paths):
if keycheck_input_mode() == 'postgres':
return
seen = set()
items = []
for path in paths or []:
if not path or not os.path.exists(path):
continue
for line in iter_bounded_text_lines(path):
key = key_from_line(line)
if not key or key in seen:
continue
seen.add(key)
items.append({'key': key, 'source': path, 'finding': {}})
for item in items:
yield item
def retry_plain_files(args):
if keycheck_input_mode() == 'postgres':
return []
files = list(args.plain or [])
if args.recheck_all:
files.extend(STATUS_FILES)
else:
if args.retry_limited:
files.append(LIMITED_FILE)
if args.retry_network:
files.append(NETWORK_FILE)
if args.retry_unknown:
files.append(UNKNOWN_FILE)
if args.retry_restricted:
files.append(RESTRICTED_FILE)
if args.retry_no_balance:
files.append(LIMITED_FILE)
out = []
seen = set()
for path in files:
if path and path not in seen:
seen.add(path)
out.append(path)
return out
def load_checked_statuses():
if keycheck_input_mode() == 'postgres':
return {}
statuses = {}
if not os.path.exists(CHECKED_FILE):
return statuses
for line in iter_bounded_text_lines(CHECKED_FILE):
parts = line.rstrip('\n').split('\t')
if parts and parts[0]:
statuses[parts[0]] = parts[1] if len(parts) > 1 else 'UNKNOWN'
return statuses
def load_known_keys():
if keycheck_input_mode() == 'postgres':
return set()
known = set(load_checked_statuses().keys())
for path in STATUS_FILES:
known.update(load_set_from_file(path))
return known
def ensure_output_files():
ensure_private_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES])
recover_status_transaction(CHECKED_FILE, STATUS_BY_FILE)
def compact_status_file(path):
if keycheck_input_mode() == 'postgres':
return
if not os.path.exists(path):
return
last_by_key = {}
order = []
for line in iter_bounded_text_lines(path):
key = key_from_line(line)
if not key:
continue
if key not in last_by_key:
order.append(key)
last_by_key[key] = line
for attempt in range(6):
try:
with private_atomic_writer(path) as f:
for key in order:
f.write(last_by_key[key])
except PermissionError:
if attempt == 5:
print(f"Warning: unable to compact {path}; leaving existing file as-is")
return
time.sleep(0.1 * (attempt + 1))
else:
return
def compact_all_status_files():
for path in [CHECKED_FILE, *STATUS_FILES]:
compact_status_file(path)
def backfill_checked_file():
if keycheck_input_mode() == 'postgres':
return
checked = load_checked_statuses()
changed = False
for path, status in STATUS_BY_FILE.items():
for key in load_set_from_file(path):
if key not in checked:
checked[key] = status
changed = True
if not changed:
return
with private_atomic_writer(CHECKED_FILE) as f:
for key, status in sorted(checked.items()):
f.write(f"{key}\t{status}\tbackfilled\n")
def load_proxies(proxy_file=None):
proxy_file = proxy_file or PROXY_FILE
if not os.path.exists(proxy_file): return None
proxies = []
with open(proxy_file, 'r') as f:
for line in f:
line = line.strip()
if not line: continue
try:
ip, port, login, password = line.split(':')
proxy_url = f"http://{login}:{password}@{ip}:{port}"
proxies.append({"http": proxy_url, "https": proxy_url})
except ValueError:
print(f"⚠️ Неверный формат прокси: '{line}'. Пропускаем.")
if not proxies:
print("⚠️ Файл proxy.txt пуст. Запросы будут идти напрямую.")
return None
print(f"✅ Загружено {len(proxies)} прокси.")
return cycle(proxies)
def move_key_to_alive(
key, available_target_models, service_tier, source='', finding=None,
model_inventory=None, probe_model='',
):
"""
Перемещает ключ из DEAD_FILE в ALIVE_FILE, записывая модели и service_tier.
"""
models_str = ",".join(sorted(list(available_target_models)))
tier_str = str(service_tier or 'unknown').replace('\r', ' ').replace('\n', ' ')[:1000]
print(f" -> ✅ Ключ рабочий! Модели: {models_str}, Тир: {tier_str}. Перемещаем в {ALIVE_FILE}")
model_inventory = sorted(set(model_inventory or available_target_models))
result = {
'status': 'ALIVE',
'models': sorted(list(available_target_models)),
'model_inventory': model_inventory,
'model_count': len(model_inventory),
'llm_probe_model': probe_model,
'llm_probe_status': 'GENERATION_OK',
'service_tier': service_tier,
'message': f'chat ping ok; model={probe_model}; models={len(model_inventory)}',
}
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, 'OpenAI')
commit_status_transaction(
CHECKED_FILE,
STATUS_BY_FILE,
key,
'ALIVE',
status_line=f"{key}:[{models_str}]:{tier_str}\n",
checked_line=f"{key}\tALIVE\t{time.strftime('%Y-%m-%dT%H:%M:%S')}\n",
)
record_validation_result(SERVICE, key, result, source, finding, 'OpenAI')
def redact_message(message, key):
message = str(message).replace('\r', ' ').replace('\n', ' ')[:1000]
if key:
message = message.replace(key, '***REDACTED***')
return OPENAI_KEY_REGEX.sub('***REDACTED***', message)
def write_key_status(key, path, status, message='', source='', finding=None, metadata=None):
status_upper = status.upper()
projection_status = STATUS_BY_FILE.get(path)
if not projection_status:
raise ValueError(f'unknown OpenAI status projection: {path}')
message = redact_message(message, key)
result = {**(metadata or {}), 'status': status_upper, 'message': message}
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, 'OpenAI')
commit_status_transaction(
CHECKED_FILE,
STATUS_BY_FILE,
key,
projection_status,
message,
source,
status_line=f"{key}\t{status}\t{message}\n",
checked_line=f"{key}\t{projection_status}\t{time.strftime('%Y-%m-%dT%H:%M:%S')}\n",
)
record_validation_result(SERVICE, key, result, source, finding, 'OpenAI')
def extract_openai_key(data):
if data.get("DetectorName") == "OpenAI":
return data.get("Raw") or data.get("RawV2")
if data.get("detector") == "OpenAI":
return data.get("raw") or data.get("raw_v2")
finding = data.get("finding")
if isinstance(finding, dict) and finding.get("DetectorName") == "OpenAI":
return finding.get("Raw") or finding.get("RawV2")
return None
# --- Функции проверки ---
def check_authentication(key, proxy):
print(f" [1/2] Проверка аутентификации...")
url = "https://api.openai.com/v1/models"
headers = {"Authorization": f"Bearer {key}"}
try:
response = requests.get(url, headers=headers, proxies=proxy, timeout=15)
if response.status_code == 200:
print(" -> Аутентификация пройдена.")
return 'valid', response.json().get('data', [])
elif response.status_code == 401:
print(" -> Ошибка 401: Ключ недействителен или отозван.")
return 'dead', response.text
elif response.status_code == 403:
print(f" -> Ошибка 403: ключ ограничен/заблокирован: {response.text[:300]}")
return 'restricted', response.text
elif response.status_code == 429:
print(f" -> Ошибка 429: rate limit / quota: {response.text[:300]}")
return 'limited', response.text
else:
print(f" -> Ошибка {response.status_code}: {response.text}")
return 'unknown', response.text
except requests.RequestException as e:
print(f" -> Ошибка сети: {e}")
return 'network', str(e)
def reportable_target_models(model_ids):
return {
model for model in model_ids
if model in TARGET_MODELS or model.startswith('gpt-5.6-') or model == 'gpt-5.6'
}
def choose_probe_model(model_ids):
models = [str(model or '') for model in model_ids if model]
by_lower = {model.lower(): model for model in models}
for model in MODEL_PRIORITY_FOR_TEST:
if model.lower() in by_lower:
return by_lower[model.lower()]
for model in models:
lowered = model.lower()
if lowered.startswith(('gpt-', 'o')) and not any(
marker in lowered for marker in NON_CHAT_MODEL_MARKERS
):
return model
return ''
def check_balance_and_tier(key, model_to_test, proxy):
"""
Проверяет баланс и возвращает service_tier в случае успеха.
"""
print(f" [2/2] Проверка баланса и тира...")
if not model_to_test:
print(" -> Не найдено подходящих моделей для теста баланса.")
return 'unknown', 'no chat-capable model from /models'
print(f" -> Используем модель для теста: {model_to_test}")
url = "https://api.openai.com/v1/chat/completions"
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
payload = {
"model": model_to_test,
"messages": [{"role": "user", "content": "Reply with one digit."}],
"max_completion_tokens": PROBE_MAX_COMPLETION_TOKENS,
}
try:
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=20)
if response.status_code == 200:
# Успех, извлекаем service_tier
response_data = response.json()
service_tier = response_data.get('service_tier')
return 'ok', service_tier
elif response.status_code == 429:
print(f" -> Ошибка 429: Нет баланса или превышен лимит.")
return 'limited', response.text
elif response.status_code == 401:
print(f" -> Ошибка 401: ключ недействителен или отозван.")
return 'dead', response.text
elif response.status_code == 403:
print(f" -> Ошибка 403: ключ ограничен/заблокирован: {response.text[:300]}")
return 'restricted', response.text
else:
print(f" -> Ошибка {response.status_code}: {response.text}")
return 'unknown', response.text
except requests.RequestException as e:
print(f" -> Ошибка сети: {e}")
return 'network', str(e)
# --- Основной процесс ---
def parse_args():
parser = argparse.ArgumentParser(description='OpenAI key checker')
parser.add_argument('--input', default=INPUT_FILE)
parser.add_argument('--plain', action='append', default=[])
parser.add_argument('--proxy-file', default=PROXY_FILE)
parser.add_argument('--max-keys', type=int, default=0)
parser.add_argument('--retry-network', action='store_true')
parser.add_argument('--retry-limited', action='store_true')
parser.add_argument('--retry-unknown', action='store_true')
parser.add_argument('--retry-restricted', action='store_true')
parser.add_argument('--retry-no-balance', action='store_true')
parser.add_argument('--recheck-all', action='store_true')
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_output_files()
compact_all_status_files()
backfill_checked_file()
print("--- 🚀 Запуск чекера ключей OpenAI 🚀 ---")
proxy_cycler = load_proxies(args.proxy_file)
checked_statuses = load_checked_statuses()
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_BY_FILE)
known_keys = set(known_statuses)
alive_keys = load_set_from_file(ALIVE_FILE)
retry_statuses = set()
if args.retry_network:
retry_statuses.add('NETWORK')
if args.retry_limited:
retry_statuses.update(('LIMITED', 'LIMITED_OR_NO_BALANCE', 'LIMITED_OR_QUOTA'))
if args.retry_unknown:
retry_statuses.add('UNKNOWN')
if args.retry_restricted:
retry_statuses.add('RESTRICTED')
if args.retry_no_balance:
retry_statuses.update(('NO_BALANCE', 'NO_QUOTA', 'LIMITED_OR_NO_BALANCE', 'LIMITED_OR_QUOTA'))
print(f"📖 Загружено: {len(alive_keys)} живых ключей, {len(known_keys)} уже классифицированных ключей, {len(checked_statuses)} checked.")
if keycheck_input_mode() == 'jsonl' and not os.path.exists(args.input):
print(f"❌ Файл с секретами {args.input} не найден. Завершение.")
return
processed = 0
skipped = 0
seen_this_run = set()
def candidates():
for item in iter_findings(args.input, ['OpenAI']):
data = item.get('finding') or {}
key = item.get('raw') or extract_openai_key(data)
if key:
yield {'key': key, 'source': item.get('source') or args.input, 'finding': data}
seen_plain = set()
for item in iter_plain_openai_keys(retry_plain_files(args)):
key = item.get('key')
if key and key not in seen_plain:
seen_plain.add(key)
yield item
for item in candidates():
data = item.get('finding') or {}
source_line = item.get('source') or args.input
key = item.get('key')
if not key:
continue
if keycheck_input_mode() != 'postgres' and key in seen_this_run:
cached_status = checked_statuses.get(key) or known_statuses.get(key) or 'UNKNOWN'
record_cached_keycheck_occurrence(SERVICE, key, cached_status, source_line, data, 'OpenAI')
skipped += 1
continue
seen_this_run.add(key)
if should_skip_key(
key, checked_statuses, known_keys, args, retry_statuses,
service=SERVICE, source=source_line, finding=data, detector='OpenAI',
known_statuses=known_statuses,
):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] 🎯 Новый кандидат: {key[:8]}...{key[-4:]} from {source_line}")
current_proxy = next(proxy_cycler) if proxy_cycler else None
auth_status, auth_data = check_authentication(key, current_proxy)
if auth_status == 'valid' and auth_data:
all_available_models_data = auth_data
all_model_ids = sorted({
str(model.get('id') or '') for model in all_available_models_data
if isinstance(model, dict) and model.get('id')
})
found_target_models = reportable_target_models(all_model_ids)
model_to_test = choose_probe_model(all_model_ids)
if not model_to_test:
print(" -> Ключ валиден, но не имеет подходящей chat-модели. Пропускаем.")
write_key_status(
key, NO_TARGET_FILE, 'no_target_models', ','.join(all_model_ids)[:500],
source_line, data, {
'models': sorted(found_target_models),
'model_inventory': all_model_ids,
'model_count': len(all_model_ids),
'llm_probe_status': 'NO_CONTEXT',
},
)
known_keys.add(key)
checked_statuses[key] = 'NO_TARGET_MODELS'
continue
balance_status, balance_data = check_balance_and_tier(key, model_to_test, current_proxy)
probe_metadata = {
'models': sorted(found_target_models or {model_to_test}),
'model_inventory': all_model_ids,
'model_count': len(all_model_ids),
'llm_probe_model': model_to_test,
'llm_probe_status': {
'ok': 'GENERATION_OK',
'limited': 'LIMITED',
'network': 'NETWORK',
'restricted': 'RESTRICTED',
'dead': 'DEAD',
}.get(balance_status, 'UNKNOWN'),
}
# Проверяем, что результат не None (успешная проверка баланса)
if balance_status == 'ok':
move_key_to_alive(
key, found_target_models or {model_to_test}, balance_data, source_line, data,
model_inventory=all_model_ids, probe_model=model_to_test,
)
alive_keys.add(key)
final_status = 'ALIVE'
elif balance_status == 'limited':
write_key_status(key, LIMITED_FILE, 'limited_or_no_balance', balance_data, source_line, data, probe_metadata)
final_status = 'LIMITED_OR_NO_BALANCE'
elif balance_status == 'network':
write_key_status(key, NETWORK_FILE, 'network_error', balance_data, source_line, data, probe_metadata)
final_status = 'NETWORK'
elif balance_status == 'restricted':
write_key_status(key, RESTRICTED_FILE, 'restricted', balance_data, source_line, data, probe_metadata)
final_status = 'RESTRICTED'
elif balance_status == 'dead':
write_key_status(key, DEAD_FILE, 'invalid_or_revoked', balance_data, source_line, data, probe_metadata)
final_status = 'DEAD'
else:
write_key_status(key, UNKNOWN_FILE, 'unknown', balance_data, source_line, data, probe_metadata)
final_status = 'UNKNOWN'
known_keys.add(key)
checked_statuses[key] = final_status
elif auth_status == 'network':
write_key_status(key, NETWORK_FILE, 'network_error', auth_data, source_line, data)
known_keys.add(key)
checked_statuses[key] = 'NETWORK'
elif auth_status == 'limited':
write_key_status(key, LIMITED_FILE, 'limited_or_quota', auth_data, source_line, data)
known_keys.add(key)
checked_statuses[key] = 'LIMITED'
elif auth_status == 'restricted':
write_key_status(key, RESTRICTED_FILE, 'restricted', auth_data, source_line, data)
known_keys.add(key)
checked_statuses[key] = 'RESTRICTED'
elif auth_status == 'dead':
write_key_status(key, DEAD_FILE, 'invalid_or_revoked', auth_data, source_line, data)
known_keys.add(key)
checked_statuses[key] = 'DEAD'
else:
write_key_status(key, UNKNOWN_FILE, 'unknown', auth_data, source_line, data)
known_keys.add(key)
checked_statuses[key] = 'UNKNOWN'
print("\n--- ✅ Проверка завершена. ---")
print(f"Processed={processed}, skipped={skipped}")
if __name__ == "__main__":
main()
@@ -0,0 +1,498 @@
import sys
sys.dont_write_bytecode = True
import requests
import json
import os
import argparse
from itertools import cycle
from datetime import datetime, timezone
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
try:
sys.stdout.reconfigure(encoding='utf-8', errors='replace')
sys.stderr.reconfigure(encoding='utf-8', errors='replace')
except Exception:
pass
from keycheck_common import (
acquire_file_lock,
append_checked,
append_jsonl,
commit_status_transaction,
default_input_file,
default_proxy_file,
env_int,
ensure_output_files as ensure_private_output_files,
finding_detector_secret_hash,
iter_findings,
iter_bounded_text_lines,
keycheck_input_mode,
load_known_statuses,
load_checked_statuses,
mask_secret,
now_iso,
physical_jsonl_segments,
private_append_writer,
private_atomic_writer,
reconcile_keycheck_jsonl_segments,
record_validation_result,
recover_status_transaction,
release_file_lock,
repair_keycheck_jsonl_tail,
require_provider_authority,
rotate_jsonl_if_needed,
service_output_dir,
should_skip_key,
sha256_text,
write_keycheck_event,
)
from runtime_security import reject_reparse_components, require_private_file
# --- Конфигурация ---
SERVICE = "openrouter"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
ALIVE_FILE = os.path.join(OUTPUT_DIR, "openrouterAlive.txt")
DEAD_FILE = os.path.join(OUTPUT_DIR, "openrouterDead.txt")
LIMITED_FILE = os.path.join(OUTPUT_DIR, "openrouterLimited.txt")
NO_BALANCE_FILE = os.path.join(OUTPUT_DIR, "openrouterNoBalance.txt")
NETWORK_FILE = os.path.join(OUTPUT_DIR, "openrouterNetwork.txt")
UNKNOWN_FILE = os.path.join(OUTPUT_DIR, "openrouterUnknown.txt")
CHECKED_FILE = os.path.join(OUTPUT_DIR, "openrouterChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "openrouterResults.jsonl")
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
STATUS_FILES = {
"VALID": ALIVE_FILE,
"NO_BALANCE": NO_BALANCE_FILE,
"DEAD": DEAD_FILE,
"LIMITED": LIMITED_FILE,
"NETWORK": NETWORK_FILE,
"UNKNOWN": UNKNOWN_FILE,
}
CREDITS_URL = "https://openrouter.ai/api/v1/credits"
# --- Вспомогательные функции ---
def load_set_from_file(filepath):
"""Загружает ключи из файла в set для быстрой проверки."""
if not os.path.exists(filepath):
return set()
return {line.strip().split(':')[0] for line in iter_bounded_text_lines(filepath) if line.strip()}
def ensure_output_files():
ensure_private_output_files((CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()))
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def legacy_status_key(line):
value = line.strip()
if not value:
return None
if '\t' in value:
return value.split('\t', 1)[0].strip()
return value.split(':', 1)[0].strip()
def load_openrouter_keys(path):
if keycheck_input_mode() == 'postgres':
return set()
if not os.path.exists(path):
return set()
return {key for key in (legacy_status_key(line) for line in iter_bounded_text_lines(path)) if key}
def iter_plain_openrouter_keys(paths):
if keycheck_input_mode() == 'postgres':
return
seen = set()
items = []
for path in paths or []:
if not path or not os.path.exists(path):
continue
for line in iter_bounded_text_lines(path):
key = legacy_status_key(line)
if not key or key in seen:
continue
seen.add(key)
items.append({'raw': key, 'source': path, 'finding': {}})
for item in items:
yield item
def retry_plain_files(args):
if keycheck_input_mode() == 'postgres':
return []
files = list(args.plain or [])
if args.recheck_all:
files.extend(STATUS_FILES.values())
else:
if args.retry_valid:
files.append(ALIVE_FILE)
if args.retry_no_balance:
files.append(NO_BALANCE_FILE)
if args.retry_limited:
files.append(LIMITED_FILE)
if args.retry_network:
files.append(NETWORK_FILE)
if args.retry_unknown:
files.append(UNKNOWN_FILE)
out = []
seen = set()
for path in files:
if path and path not in seen:
seen.add(path)
out.append(path)
return out
def migrate_legacy_checked():
if keycheck_input_mode() == 'postgres':
return
checked = load_checked_statuses(CHECKED_FILE)
for key in sorted(load_openrouter_keys(ALIVE_FILE)):
if key not in checked:
write_keycheck_event(SERVICE, RESULTS_FILE, key, {'status': 'VALID', 'message': 'legacy alive status migration'}, 'legacy:openrouterAlive.txt', {}, 'OpenRouter', 'legacy_status')
append_checked(CHECKED_FILE, key, 'VALID')
checked[key] = 'VALID'
for key in sorted(load_openrouter_keys(DEAD_FILE)):
if key not in checked:
write_keycheck_event(SERVICE, RESULTS_FILE, key, {'status': 'DEAD', 'message': 'legacy dead status migration'}, 'legacy:openrouterDead.txt', {}, 'OpenRouter', 'legacy_status')
append_checked(CHECKED_FILE, key, 'DEAD')
checked[key] = 'DEAD'
def _legacy_alive_checked_at(path):
details = os.stat(path, follow_symlinks=False)
return datetime.fromtimestamp(details.st_mtime, timezone.utc).isoformat(timespec='seconds')
def _legacy_balance_event_payload(key, balance, checked_at):
balance_text = f'{balance:.6f}'
key_hash = sha256_text(key)
event_id = sha256_text('|'.join([
SERVICE,
'legacy_status',
'openrouterAlive.txt',
key_hash,
'NO_BALANCE',
balance_text,
]))
return {
'key_masked': mask_secret(key),
'key_hash': key_hash,
'secret_hash': key_hash,
'detector_secret_hash': finding_detector_secret_hash({}),
'finding_uid': '',
'detector': 'OpenRouter',
'source': 'legacy:openrouterAlive.txt',
'finding': {},
'checked_at': checked_at,
'result_source': 'legacy_status',
'status': 'NO_BALANCE',
'message': f'credits={balance_text}',
'event_id': event_id,
}
def _publish_legacy_no_balance(moved):
lock_path = f'{NO_BALANCE_FILE}.lock'
lock = acquire_file_lock(lock_path, timeout_sec=30)
try:
require_private_file(NO_BALANCE_FILE)
existing = load_openrouter_keys(NO_BALANCE_FILE)
with private_append_writer(NO_BALANCE_FILE) as handle:
for key, balance in moved:
if key in existing:
continue
balance_text = f'{balance:.6f}'
handle.write(f'{key}\tNO_BALANCE\tcredits={balance_text}\tmigrated_from_alive\n')
existing.add(key)
finally:
release_file_lock(lock, lock_path)
def _existing_legacy_event_ids(expected):
found = set()
max_line_bytes = max(1024, env_int('KEYCHECK_INPUT_MAX_LINE_BYTES', 16 * 1024 * 1024))
max_file_bytes = max(
max_line_bytes,
max(1, env_int('KEYCHECK_INPUT_LIST_MAX_BYTES', 32 * 1024 * 1024)),
max(0, env_int('KEYCHECK_RESULTS_MAX_MB', 32)) * 1024 * 1024 + max_line_bytes,
)
paths = [path for _, path in physical_jsonl_segments(RESULTS_FILE)]
if os.path.isfile(RESULTS_FILE):
paths.append(os.path.abspath(RESULTS_FILE))
for path in paths:
reject_reparse_components(path)
if os.path.getsize(path) > max_file_bytes:
raise RuntimeError(f'OpenRouter result file exceeds the bounded migration scan size: {path}')
with open(path, 'rb') as handle:
while True:
raw_line = handle.readline(max_line_bytes + 1)
if not raw_line:
break
if len(raw_line) > max_line_bytes:
raise RuntimeError(f'OpenRouter result line exceeds the bounded migration scan size: {path}')
if not raw_line.endswith(b'\n'):
raise RuntimeError(f'torn OpenRouter result line during legacy migration: {path}')
try:
payload = json.loads(raw_line.decode('utf-8', errors='replace'))
except (TypeError, ValueError):
continue
event_id = str(payload.get('event_id') or '') if isinstance(payload, dict) else ''
if event_id not in expected:
continue
wanted = expected[event_id]
for field in ('key_hash', 'status', 'source', 'result_source'):
if str(payload.get(field) or '') != str(wanted.get(field) or ''):
raise RuntimeError(f'conflicting OpenRouter legacy migration event: {event_id}')
found.add(event_id)
return found
def _publish_legacy_balance_events(moved, checked_at):
payloads = [_legacy_balance_event_payload(key, balance, checked_at) for key, balance in moved]
expected = {payload['event_id']: payload for payload in payloads}
lock_path = f'{RESULTS_FILE}.lock'
lock = acquire_file_lock(lock_path, timeout_sec=30)
try:
require_private_file(RESULTS_FILE)
repair_keycheck_jsonl_tail(RESULTS_FILE)
reconcile_keycheck_jsonl_segments(RESULTS_FILE)
existing = _existing_legacy_event_ids(expected)
max_bytes = max(0, env_int('KEYCHECK_RESULTS_MAX_MB', 32)) * 1024 * 1024
for payload in payloads:
event_id = payload['event_id']
if event_id in existing:
continue
rotate_jsonl_if_needed(RESULTS_FILE, max_bytes)
with private_append_writer(RESULTS_FILE) as handle:
handle.write(json.dumps(payload, ensure_ascii=False, default=str) + '\n')
existing.add(event_id)
finally:
release_file_lock(lock, lock_path)
def _publish_legacy_checked(moved, checked_at):
lock_path = f'{CHECKED_FILE}.lock'
lock = acquire_file_lock(lock_path, timeout_sec=30)
try:
require_private_file(CHECKED_FILE)
existing = set()
for line in iter_bounded_text_lines(CHECKED_FILE):
parts = line.rstrip('\r\n').split('\t')
if len(parts) >= 2:
existing.add((parts[0], parts[1]))
with private_append_writer(CHECKED_FILE) as handle:
for key, _ in moved:
identity = (key, 'NO_BALANCE')
if identity in existing:
continue
handle.write(f'{key}\tNO_BALANCE\t{checked_at}\n')
existing.add(identity)
finally:
release_file_lock(lock, lock_path)
def _rewrite_legacy_alive(keep):
with private_atomic_writer(ALIVE_FILE, binary=True, suffix='.legacy.tmp') as handle:
for line in keep:
handle.write(line.encode('utf-8'))
def migrate_legacy_alive_balances():
if keycheck_input_mode() == 'postgres':
return
lock_path = f'{ALIVE_FILE}.lock'
lock = acquire_file_lock(lock_path, timeout_sec=30)
try:
if not os.path.exists(ALIVE_FILE):
return
require_private_file(ALIVE_FILE)
checked_at = _legacy_alive_checked_at(ALIVE_FILE)
keep = []
moved = []
seen = set()
for line in iter_bounded_text_lines(ALIVE_FILE):
text = line.strip()
if not text:
keep.append(line)
continue
key = legacy_status_key(text)
balance = None
if ':' in text and '\t' not in text:
try:
balance = float(text.rsplit(':', 1)[1])
except ValueError:
balance = None
if key and balance is not None and balance <= 0:
if key not in seen:
moved.append((key, balance))
seen.add(key)
else:
keep.append(line)
if not moved:
return
_publish_legacy_no_balance(moved)
_publish_legacy_balance_events(moved, checked_at)
_publish_legacy_checked(moved, checked_at)
_rewrite_legacy_alive(keep)
finally:
release_file_lock(lock, lock_path)
def load_proxies(proxy_file=None):
"""Загружает и подготавливает прокси."""
proxy_file = proxy_file or PROXY_FILE
if not os.path.exists(proxy_file):
print("ℹ️ Файл proxy.txt не найден, запросы будут идти напрямую.")
return None
proxies = []
with open(proxy_file, 'r') as f:
for line in f:
line = line.strip()
if not line: continue
try:
ip, port, login, password = line.split(':')
proxy_url = f"http://{login}:{password}@{ip}:{port}"
proxies.append({"http": proxy_url, "https": proxy_url})
except ValueError:
print(f"⚠️ Неверный формат прокси: '{line}'. Пропускаем.")
if not proxies:
print("⚠️ Файл proxy.txt пуст. Запросы будут идти напрямую.")
return None
print(f"✅ Загружено {len(proxies)} прокси.")
return cycle(proxies)
def write_result(key, result, source_line, finding=None, previous_status=None):
status = result.get('status') or 'UNKNOWN'
credits = result.get('remaining_credits')
extra = f"credits={credits:.6f}" if isinstance(credits, (int, float)) else source_line
checked_at = now_iso()
result = {**result, 'checked_at': result.get('checked_at') or checked_at}
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source_line, finding, 'OpenRouter')
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, status, result.get('message', ''), extra,
)
record_validation_result(SERVICE, key, result, source_line, finding, 'OpenRouter')
# --- Функция проверки ---
def check_openrouter_key(key, proxy):
"""Проверяет один ключ OpenRouter и возвращает normalized result."""
headers = {"Authorization": f"Bearer {key}"}
try:
resp = requests.get(CREDITS_URL, headers=headers, proxies=proxy, timeout=15)
except requests.exceptions.RequestException as e:
return {'status': 'NETWORK', 'message': str(e)}
if resp.status_code == 200:
try:
data = resp.json().get("data", {})
total = float(data.get("total_credits", 0.0) or 0.0)
used = float(data.get("total_usage", 0.0) or 0.0)
except (TypeError, ValueError, json.JSONDecodeError) as exc:
return {'status': 'UNKNOWN', 'http_status': resp.status_code, 'message': f'invalid credits response: {exc}'}
remaining = total - used
if remaining > 0:
return {'status': 'VALID', 'remaining_credits': remaining, 'message': f'credits={remaining:.6f}'}
return {'status': 'NO_BALANCE', 'remaining_credits': remaining, 'message': f'credits={remaining:.6f}'}
if resp.status_code in (401, 403):
return {'status': 'DEAD', 'http_status': resp.status_code, 'message': resp.text[:1000]}
if resp.status_code == 429:
return {'status': 'LIMITED', 'http_status': resp.status_code, 'message': resp.text[:1000]}
if 500 <= resp.status_code <= 599:
return {'status': 'NETWORK', 'http_status': resp.status_code, 'message': resp.text[:1000]}
return {'status': 'UNKNOWN', 'http_status': resp.status_code, 'message': resp.text[:1000]}
def parse_args():
parser = argparse.ArgumentParser(description='OpenRouter key checker')
parser.add_argument('--input', default=INPUT_FILE)
parser.add_argument('--plain', action='append', default=[])
parser.add_argument('--proxy-file', default=PROXY_FILE)
parser.add_argument('--max-keys', type=int, default=0)
parser.add_argument('--retry-network', action='store_true')
parser.add_argument('--retry-limited', action='store_true')
parser.add_argument('--retry-unknown', action='store_true')
parser.add_argument('--retry-no-balance', action='store_true')
parser.add_argument('--retry-valid', action='store_true')
parser.add_argument('--recheck-all', action='store_true')
return parser.parse_args()
# --- Основной процесс ---
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_output_files()
migrate_legacy_alive_balances()
migrate_legacy_checked()
print("--- 🚀 Запуск чекера ключей OpenRouter 🚀 ---")
proxy_cycler = load_proxies(args.proxy_file)
checked_statuses = load_checked_statuses(CHECKED_FILE)
known_statuses = load_known_statuses(CHECKED_FILE, STATUS_FILES)
known_keys = set(known_statuses)
for path in STATUS_FILES.values():
known_keys.update(load_openrouter_keys(path))
retry_statuses = set()
if args.retry_network:
retry_statuses.add('NETWORK')
if args.retry_limited:
retry_statuses.add('LIMITED')
if args.retry_unknown:
retry_statuses.add('UNKNOWN')
if args.retry_no_balance:
retry_statuses.add('NO_BALANCE')
if args.retry_valid:
retry_statuses.add('VALID')
print(f"📖 Загружено: {len(load_openrouter_keys(ALIVE_FILE))} живых ключей, {len(known_keys)} классифицированных ключей.")
if keycheck_input_mode() == 'jsonl' and not os.path.exists(args.input):
print(f"❌ Файл с секретами {args.input} не найден. Завершение.")
return
processed = 0
def candidates():
for item in iter_findings(args.input, ["OpenRouter"]):
key = item.get("raw") or ""
if key:
yield item
seen = set()
for item in iter_plain_openrouter_keys(retry_plain_files(args)):
key = item.get("raw") or ""
if key and key not in seen:
seen.add(key)
yield item
for item in candidates():
key = item.get("raw") or ""
if not key:
continue
source = item.get("source") or args.input
finding = item.get("finding") or {}
if should_skip_key(
key, checked_statuses, known_keys, args, retry_statuses,
service=SERVICE, source=source, finding=finding, detector='OpenRouter', known_statuses=known_statuses,
):
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] 🎯 Новый кандидат: {key[:8]}...{key[-4:]} from {source}")
current_proxy = next(proxy_cycler) if proxy_cycler else None
result = check_openrouter_key(key, current_proxy)
print(f" STATUS: {result.get('status')} | {str(result.get('message', ''))[:200]}")
previous_status = known_statuses.get(key) or checked_statuses.get(key)
write_result(key, result, source, finding, previous_status)
known_keys.add(key)
checked_statuses[key] = result.get('status')
print("\n--- ✅ Проверка завершена. ---")
if __name__ == "__main__":
main()
+189
View File
@@ -0,0 +1,189 @@
import sys
sys.dont_write_bytecode = True
import os
SUPPORTED_PROVIDERS = ("deepseek", "zai", "qwen", "kimi")
DEFAULT_PROVIDER_ORDER = SUPPORTED_PROVIDERS
AMBIGUOUS_QWEN_DEEPSEEK_HINT = "ambiguous_qwen_deepseek"
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
def split_csv(value):
if not value:
return []
if isinstance(value, str):
values = value.split(",")
else:
values = value
return [str(item).strip().lower() for item in values if str(item).strip()]
def unique_supported(values):
output = []
seen = set()
for value in values:
provider = str(value or "").strip().lower()
if provider in SUPPORTED_PROVIDERS and provider not in seen:
seen.add(provider)
output.append(provider)
return output
def providers_for_hint(hint):
hint = str(hint or "").strip().lower()
if hint == AMBIGUOUS_QWEN_DEEPSEEK_HINT:
return ["qwen", "deepseek"]
if hint == AMBIGUOUS_GENERIC_SK_HINT:
return list(SUPPORTED_PROVIDERS)
return [hint] if hint in SUPPORTED_PROVIDERS else []
def detector_provider(finding):
if not isinstance(finding, dict):
return ""
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
names = {
str(finding.get("DetectorName") or finding.get("DetectorType") or "").strip().lower(),
str(extra.get("name") or "").strip().lower(),
}
mappings = (
("deepseek", {"deepseek", "deepseekapikey", "deepseek_api_key"}),
("zai", {"zaiglm"}),
("qwen", {"qwendashscope", "qwen_dashscope", "qwen", "dashscope"}),
("kimi", {"kimimoonshot", "moonshotai", "moonshot", "kimi"}),
)
for provider, detectors in mappings:
if names & detectors:
return provider
return ""
def ordered_providers(finding=None, hint="", origin_service="", configured_order=None):
context = finding.get("ScannerContext") if isinstance(finding, dict) and isinstance(
finding.get("ScannerContext"), dict
) else {}
hint = str(hint or context.get("provider_hint") or "").strip().lower()
compatible = unique_supported(context.get("provider_candidates") or providers_for_hint(hint))
if not compatible:
compatible = providers_for_hint(hint)
if not compatible:
compatible = list(SUPPORTED_PROVIDERS)
configured = unique_supported(
configured_order
if configured_order is not None
else split_csv(os.getenv("KEYCHECK_PROVIDER_RESOLUTION_ORDER"))
)
base_order = configured or list(DEFAULT_PROVIDER_ORDER)
origin = str(origin_service or detector_provider(finding)).strip().lower()
ordered = []
if origin in compatible:
ordered.append(origin)
ordered.extend(provider for provider in base_order if provider in compatible)
ordered.extend(provider for provider in compatible if provider not in ordered)
return unique_supported(ordered)
def provider_result_outcome(result):
result = result if isinstance(result, dict) else {}
status = str(result.get("status") or "UNKNOWN").strip().upper()
if result.get("authenticated") is True or status in ("VALID", "ALIVE"):
return "match"
if result.get("candidate_rejected") or status in (
"DEAD", "INVALID", "EXPIRED", "LEAKED_REVOKED", "INVALID_OR_REVOKED",
):
return "no_match"
return "retry"
def bounded_attempt(provider, result, outcome):
result = result if isinstance(result, dict) else {}
error = result.get("error") if isinstance(result.get("error"), dict) else {}
return {
"provider": provider,
"outcome": outcome,
"status": str(result.get("status") or "UNKNOWN").upper(),
"authenticated": bool(result.get("authenticated")),
"http_status": int(result.get("http_status") or error.get("http_status") or 0),
"business_code": str(result.get("business_code") or error.get("code") or "")[:80],
"region": str(result.get("region") or "")[:160],
"message": str(result.get("message") or "").replace("\r", " ").replace("\n", " ")[:300],
}
def default_provider_probe(provider, key, proxy, timeout, debug=False):
if provider == "deepseek":
from keycheckers.deepseek import deepseekKeycheck
return deepseekKeycheck.check_key(key, proxy, timeout)
if provider == "zai":
from keycheckers.zai import zaiKeycheck
return zaiKeycheck.check_key(
key, zaiKeycheck.base_urls_from_environment(), proxy, timeout, debug,
)
if provider == "qwen":
from keycheckers.qwen import qwenKeycheck
custom = qwenKeycheck.split_csv(
os.getenv("QWEN_BASE_URLS") or os.getenv("DASHSCOPE_BASE_URLS")
)
base_urls = qwenKeycheck.unique_ordered([*custom, *qwenKeycheck.DEFAULT_BASE_URLS])
return qwenKeycheck.check_key(key, base_urls, bool(custom), proxy, timeout, debug)
if provider == "kimi":
from keycheckers.kimi import kimiKeycheck
custom = kimiKeycheck.split_csv(
os.getenv("KIMI_BASE_URLS") or os.getenv("MOONSHOT_BASE_URLS")
)
base_urls = kimiKeycheck.unique_ordered([*custom, *kimiKeycheck.DEFAULT_BASE_URLS])
return kimiKeycheck.check_key(key, base_urls, proxy, timeout, debug)
raise ValueError(f"unsupported provider resolution adapter: {provider}")
def resolve_provider_key(
key, finding=None, proxy=None, timeout=15, debug=False, hint="",
origin_service="", configured_order=None, probe=None,
):
providers = ordered_providers(finding, hint, origin_service, configured_order)
probe = probe or default_provider_probe
attempts = []
retry_results = []
for provider in providers:
result = probe(provider, key, proxy, timeout, debug)
result = result if isinstance(result, dict) else {"status": "UNKNOWN"}
outcome = provider_result_outcome(result)
attempts.append(bounded_attempt(provider, result, outcome))
if outcome == "match":
return {
**result,
"resolved_provider": provider,
"provider_resolution": "matched",
"provider_resolution_order": providers,
"provider_resolution_attempts": attempts,
"result_source": "provider_resolution",
}
if outcome == "retry":
retry_results.append(result)
if retry_results:
selected = retry_results[0]
return {
**selected,
"provider_resolution": "retry",
"provider_resolution_order": providers,
"provider_resolution_attempts": attempts,
"result_source": "provider_resolution",
"message": str(selected.get("message") or "provider resolution remains inconclusive")[:1000],
}
return {
"status": "DEAD",
"provider_resolution": "exhausted",
"provider_resolution_order": providers,
"provider_resolution_attempts": attempts,
"result_source": "provider_resolution",
"message": "all compatible providers rejected the credential",
}
@@ -0,0 +1,203 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
import re
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
record_validation_result,
recover_status_transaction,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
from keycheckers.provider_resolution import (
AMBIGUOUS_GENERIC_SK_HINT,
AMBIGUOUS_QWEN_DEEPSEEK_HINT,
resolve_provider_key,
)
SERVICE = "provider_resolver"
DETECTOR = "ProviderResolver"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "providerResolverChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "providerResolverResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "providerResolverAlive.txt"),
"NO_BALANCE": os.path.join(OUTPUT_DIR, "providerResolverNoBalance.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "providerResolverDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "providerResolverRestricted.txt"),
"LIMITED": os.path.join(OUTPUT_DIR, "providerResolverLimited.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "providerResolverNetwork.txt"),
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "providerResolverNoContext.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "providerResolverUnknown.txt"),
}
RESOLVABLE_KEY_REGEX = re.compile(
r"(?<![A-Za-z0-9_.-])(?:"
r"(?:zai|sk)-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|"
r"[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}"
r")(?![A-Za-z0-9_.-])"
)
FOREIGN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
AMBIGUOUS_HINTS = {AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT}
def key_rejection_reason(key):
value = str(key or "")
try:
encoded = value.encode("utf-8", errors="strict")
except UnicodeEncodeError:
return "candidate is not valid UTF-8"
if len(encoded) > 512:
return "candidate exceeds the 512-byte key limit"
if value.startswith(FOREIGN_KEY_PREFIXES):
return "candidate has a foreign provider prefix"
if not RESOLVABLE_KEY_REGEX.fullmatch(value):
return "candidate does not match a bounded resolvable provider-key format"
return ""
def iter_candidate_keys(input_file):
detector_names = [
"ProviderResolver", "CustomRegex", "QwenDashScope", "Qwen_DashScope",
"Qwen", "DashScope", "DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key",
"KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi", "ZaiGLM",
"qwendashscope", "qwen_dashscope", "qwen", "dashscope", "deepseek",
"deepseekapikey", "deepseek_api_key", "kimimoonshot", "moonshotai",
"moonshot", "kimi", "zaiglm",
]
for item in iter_findings(input_file, detector_names):
finding = item.get("finding") or {}
key = item.get("credential_secret_text") or ""
if not key:
for value in (item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2")):
match = RESOLVABLE_KEY_REGEX.search(str(value or ""))
if match:
key = match.group(0)
break
if not key or key_rejection_reason(key):
continue
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
metadata_hint = ""
active_metadata = item.get("candidate_metadata")
if isinstance(active_metadata, dict):
metadata_hint = str(active_metadata.get("provider_hint") or "")
metadata_candidates = active_metadata.get("provider_candidates")
if metadata_hint or isinstance(metadata_candidates, list):
context = dict(context)
if metadata_hint:
context.setdefault("provider_hint", metadata_hint)
if isinstance(metadata_candidates, list):
context.setdefault("provider_candidates", metadata_candidates)
finding = dict(finding)
finding["ScannerContext"] = context
hint = str(context.get("provider_hint") or metadata_hint or AMBIGUOUS_GENERIC_SK_HINT)
if hint not in AMBIGUOUS_HINTS:
hint = AMBIGUOUS_GENERIC_SK_HINT
yield key, item.get("source") or input_file, finding, hint
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def write_result(key, result, source, finding):
status = result.get("status") or "UNKNOWN"
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, status,
result.get("message", ""), result.get("resolved_provider") or source,
)
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
def retry_statuses_from_args(args):
statuses = set()
for enabled, status in (
(args.retry_network, "NETWORK"),
(args.retry_limited, "LIMITED"),
(args.retry_unknown, "UNKNOWN"),
(args.retry_restricted, "RESTRICTED"),
(args.retry_no_balance, "NO_BALANCE"),
(args.retry_valid, "VALID"),
):
if enabled:
statuses.add(status)
return statuses
def parse_args():
parser = argparse.ArgumentParser(description="Ambiguous generic provider key resolver")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=15)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-restricted", action="store_true")
parser.add_argument("--retry-no-balance", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
parser.add_argument("--debug", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = retry_statuses_from_args(args)
processed = 0
skipped = 0
for key, source, finding, hint in iter_candidate_keys(args.input):
if should_skip_key(
key, checked, known, args, retry_statuses, service=SERVICE,
source=source, finding=finding, detector=DETECTOR,
):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] Ambiguous provider candidate {mask_secret(key)} from {source}")
proxy = next(proxy_cycler) if proxy_cycler else None
result = resolve_provider_key(
key, finding, proxy, args.timeout, args.debug, hint=hint,
)
print(
f" STATUS: {result['status']} provider={result.get('resolved_provider', '')} "
f"| {result.get('message', '')[:200]}"
)
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+773
View File
@@ -0,0 +1,773 @@
import sys
sys.dont_write_bytecode = True
import argparse
import json
import os
import re
from urllib.parse import urlparse
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
try:
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
sys.stderr.reconfigure(encoding="utf-8", errors="replace")
except (AttributeError, OSError, ValueError):
pass
from keycheck_common import (
combined_provider_routing_hint,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
keycheck_input_mode,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
provider_routing_database_failed,
read_plain_keys,
record_validation_result,
recover_status_transaction,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
from keycheckers.provider_resolution import resolve_provider_key
SERVICE = "qwen"
DETECTOR = "QwenDashScope"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "qwenChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "qwenResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "qwenAlive.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "qwenDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "qwenRestricted.txt"),
"LIMITED": os.path.join(OUTPUT_DIR, "qwenLimited.txt"),
"NO_BALANCE": os.path.join(OUTPUT_DIR, "qwenNoBalance.txt"),
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "qwenNoContext.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "qwenNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "qwenUnknown.txt"),
}
QWEN_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope", "dashscope", "qwen"}
QWEN_EXPLICIT_DETECTOR_NAMES = {"qwendashscope", "qwen_dashscope"}
DEEPSEEK_EXPLICIT_DETECTOR_NAMES = {"deepseekapikey", "deepseek_api_key"}
KIMI_EXPLICIT_DETECTOR_NAMES = {"kimimoonshot", "moonshotai"}
QWEN_KEY_MAX_BYTES = 512
QWEN_KEY_REGEX = re.compile(
r"(?<![A-Za-z0-9_-])sk-(?:sp-)?[A-Za-z0-9][A-Za-z0-9_-]{20,505}(?![A-Za-z0-9_-])"
)
QWEN_OVERSIZED_KEY_PREFIX_REGEX = re.compile(
r"(?<![A-Za-z0-9_-])sk-(?:sp-)?[A-Za-z0-9][A-Za-z0-9_-]{506}"
)
OVERLAPPING_QWEN_DEEPSEEK_REGEX = re.compile(r"sk-[a-z0-9]{32}")
FOREIGN_QWEN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
OPENAI_LEGACY_KEY_MARKER = "T3BlbkFJ"
QWEN_CONTEXT_REGEX = re.compile(
r"(?:DASHSCOPE_API_KEY|QWEN_API_KEY|dashscope|qwen|model[_-]?studio|bailian)",
re.IGNORECASE,
)
DEEPSEEK_CONTEXT_REGEX = re.compile(r"(?:DEEPSEEK_API_KEY|deepseek|api\.deepseek\.com)", re.IGNORECASE)
KIMI_CONTEXT_REGEX = re.compile(
r"(?:MOONSHOT_API_KEY|KIMI_API_KEY|api\.moonshot\.(?:ai|cn)|platform\.kimi\.(?:ai|com))",
re.IGNORECASE,
)
AMBIGUOUS_PROVIDER_HINT = "ambiguous_qwen_deepseek"
AMBIGUOUS_GENERIC_SK_HINT = "ambiguous_generic_sk"
GENERIC_SK_PROVIDERS = {"qwen", "deepseek", "kimi", "zai"}
EXPLICIT_ASSIGNMENT_HINT_SOURCE = "explicit_assignment"
CANDIDATE_PROVIDER_ROUTE_FIELD = "_keycheck_candidate_provider_route"
DEFAULT_BASE_URLS = [
"https://coding-intl.dashscope.aliyuncs.com/v1",
"https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
"https://dashscope-us.aliyuncs.com/compatible-mode/v1",
"https://dashscope.aliyuncs.com/compatible-mode/v1",
"https://cn-hongkong.dashscope.aliyuncs.com/compatible-mode/v1",
]
MODEL_MARKERS = ("qwen", "qwq", "qvq", "wan", "text-embedding", "multimodal-embedding")
CHAT_MODEL_PRIORITY = (
"qwen-plus",
"qwen-turbo",
"qwen-max",
"qwen3-235b-a22b",
"qwen3-32b",
"qwen2.5-72b-instruct",
"qwen2.5-32b-instruct",
"qwen2.5-14b-instruct",
"qwen2.5-7b-instruct",
"qwq-32b",
)
NON_CHAT_MODEL_MARKERS = ("embedding", "rerank", "wan", "image", "audio", "tts", "asr", "vision", "vl")
def normalize_base_url(value):
value = str(value or "").strip()
if not value:
return ""
return value.rstrip("/")
def split_csv(value):
if not value:
return []
if isinstance(value, str):
return [item.strip() for item in value.split(",") if item.strip()]
return [str(item).strip() for item in value if str(item).strip()]
def unique_ordered(values):
seen = set()
output = []
for value in values:
normalized = normalize_base_url(value)
if normalized and normalized not in seen:
seen.add(normalized)
output.append(normalized)
return output
def endpoint_label(base_url):
parsed = urlparse(base_url)
return parsed.netloc or base_url
def is_qwen_detector(value):
return str(value or "").lower() in QWEN_DETECTOR_NAMES
def custom_detector_name(data):
if not isinstance(data, dict):
return ""
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
name = str(extra.get("name") or "")
if str(data.get("DetectorName") or "").lower() == "customregex" and is_qwen_detector(name):
return name
return ""
def finding_detector_names(data):
if not isinstance(data, dict):
return set()
extra = data.get("ExtraData") if isinstance(data.get("ExtraData"), dict) else {}
names = {
str(data.get("DetectorName") or data.get("detector") or "").strip().lower(),
str(extra.get("name") or "").strip().lower(),
}
return {name for name in names if name}
def finding_has_explicit_detector(data, detector_names):
if not isinstance(data, dict):
return False
if finding_detector_names(data) & set(detector_names):
return True
nested = data.get("finding")
return isinstance(nested, dict) and bool(finding_detector_names(nested) & set(detector_names))
def detector_name_from_finding(data):
if not isinstance(data, dict):
return ""
if is_qwen_detector(data.get("DetectorName")):
return data.get("DetectorName")
custom_name = custom_detector_name(data)
if custom_name:
return custom_name
if is_qwen_detector(data.get("detector")):
return data.get("detector")
finding = data.get("finding")
if isinstance(finding, dict):
if is_qwen_detector(finding.get("DetectorName")):
return finding.get("DetectorName")
custom_name = custom_detector_name(finding)
if custom_name:
return custom_name
return ""
def key_from_text(*values):
for value in values:
for match in QWEN_KEY_REGEX.finditer(str(value or "")):
key = match.group(0)
if not key.startswith(FOREIGN_QWEN_KEY_PREFIXES) and OPENAI_LEGACY_KEY_MARKER not in key:
return key
return ""
def qwen_key_rejection_reason(key):
value = str(key or "")
try:
key_bytes = len(value.encode("utf-8", errors="strict"))
except UnicodeEncodeError:
return "candidate is not valid UTF-8"
if key_bytes > QWEN_KEY_MAX_BYTES:
return f"candidate exceeds the {QWEN_KEY_MAX_BYTES}-byte key limit"
if OPENAI_LEGACY_KEY_MARKER in value:
return "candidate is a recognizable OpenAI legacy key"
if value.count("sk-") != 1:
return "candidate contains multiple concatenated key prefixes"
if not QWEN_KEY_REGEX.fullmatch(value) or value.startswith(FOREIGN_QWEN_KEY_PREFIXES):
return "candidate does not match the bounded Qwen key format"
return ""
def is_qwen_key(key):
return not qwen_key_rejection_reason(key)
def finding_has_oversized_qwen_key(finding, *raw_values):
values = list(raw_values)
if isinstance(finding, dict):
values.extend((finding.get("Raw"), finding.get("RawV2"), finding.get("raw"), finding.get("raw_v2")))
nested = finding.get("finding")
if isinstance(nested, dict):
values.extend((nested.get("Raw"), nested.get("RawV2"), nested.get("raw"), nested.get("raw_v2")))
return any(
QWEN_OVERSIZED_KEY_PREFIX_REGEX.search(str(value or ""))
for value in values
)
def warn_rejected_candidate(reason, source, key=""):
reason = str(reason or "candidate rejected")
source = str(source or "unknown")
if key:
reason = reason.replace(key, "***REDACTED***")
source = source.replace(key, "***REDACTED***")
reason = reason.replace("\r", " ").replace("\n", " ")[:300]
source = source.replace("\r", " ").replace("\n", " ")[:300]
print(f"Warning: skipped Qwen candidate from {source}: {reason}", flush=True)
def warn_candidate_failure(reason, source, key=""):
reason = str(reason or "candidate failure")
source = str(source or "unknown")
if key:
reason = reason.replace(key, "***REDACTED***")
source = source.replace(key, "***REDACTED***")
reason = QWEN_KEY_REGEX.sub("***REDACTED***", reason).replace("\r", " ").replace("\n", " ")[:300]
source = QWEN_KEY_REGEX.sub("***REDACTED***", source).replace("\r", " ").replace("\n", " ")[:300]
print(f"Warning: Qwen candidate failure from {source}: {reason}", flush=True)
def extract_key_from_finding(data):
if not detector_name_from_finding(data):
return ""
if data.get("Raw") or data.get("RawV2"):
return key_from_text(data.get("Raw"), data.get("RawV2"))
if data.get("raw") or data.get("raw_v2"):
return key_from_text(data.get("raw"), data.get("raw_v2"))
finding = data.get("finding")
if isinstance(finding, dict):
return key_from_text(finding.get("Raw"), finding.get("RawV2"))
return ""
def finding_provider_routing_hint(finding):
if not isinstance(finding, dict):
return ""
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
persisted_hint = context.get("provider_hint")
if (
context.get("provider_hint_source") == EXPLICIT_ASSIGNMENT_HINT_SOURCE
and persisted_hint in (*GENERIC_SK_PROVIDERS, AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
):
return persisted_hint
parts = [str(context.get(key) or "") for key in ("nearby", "file")]
metadata = finding.get("SourceMetadata") if isinstance(finding.get("SourceMetadata"), dict) else {}
data = metadata.get("Data") if isinstance(metadata.get("Data"), dict) else {}
for details in data.values():
if not isinstance(details, dict):
continue
parts.extend(str(details.get(key) or "") for key in ("file", "repository", "repo", "link", "image"))
text = "\n".join(parts)
evidence = set()
if QWEN_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, QWEN_EXPLICIT_DETECTOR_NAMES):
evidence.add("qwen")
if DEEPSEEK_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, DEEPSEEK_EXPLICIT_DETECTOR_NAMES):
evidence.add("deepseek")
if KIMI_CONTEXT_REGEX.search(text) or finding_has_explicit_detector(finding, KIMI_EXPLICIT_DETECTOR_NAMES):
evidence.add("kimi")
if persisted_hint == AMBIGUOUS_PROVIDER_HINT:
evidence.update(("qwen", "deepseek"))
elif persisted_hint == AMBIGUOUS_GENERIC_SK_HINT:
evidence.update(GENERIC_SK_PROVIDERS)
elif persisted_hint in GENERIC_SK_PROVIDERS:
evidence.add(persisted_hint)
if len(evidence) > 1:
return AMBIGUOUS_PROVIDER_HINT if evidence == {"qwen", "deepseek"} else AMBIGUOUS_GENERIC_SK_HINT
return next(iter(evidence)) if evidence else ""
def finding_has_ambiguous_provider_hint(finding):
return finding_provider_routing_hint(finding) in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def iter_candidate_keys(input_file, plain_files, trusted_retry_files=None):
seen_plain = set()
seen_candidates = set()
routing_decisions = {}
detector_names = [
"QwenDashScope", "Qwen_DashScope", "qwendashscope", "qwen_dashscope",
"Qwen", "DashScope", "qwen", "dashscope", "CustomRegex",
]
for item in iter_findings(input_file, detector_names):
data = dict(item.get("finding") or {})
data.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, None)
candidate_metadata = item.get("candidate_metadata")
persisted_route = ""
if keycheck_input_mode() == "postgres" and isinstance(candidate_metadata, dict):
persisted_route = str(candidate_metadata.get("provider_hint") or "").lower()
if persisted_route == SERVICE:
data[CANDIDATE_PROVIDER_ROUTE_FIELD] = SERVICE
if finding_has_oversized_qwen_key(data, item.get("raw"), item.get("raw_v2")):
warn_rejected_candidate(
f"candidate exceeds the {QWEN_KEY_MAX_BYTES}-byte key limit",
item.get("source") or input_file,
)
continue
if persisted_route == SERVICE:
key = key_from_text(
item.get("raw"), item.get("raw_v2"),
data.get("Raw"), data.get("RawV2"),
)
else:
key = extract_key_from_finding(data)
if key and is_qwen_key(key):
if not key.startswith("sk-sp-"):
if persisted_route == SERVICE:
hint, lookup_failed = SERVICE, False
else:
local_hint = finding_provider_routing_hint(data)
if key in routing_decisions:
hint, lookup_failed = routing_decisions[key]
if local_hint == AMBIGUOUS_PROVIDER_HINT or (
local_hint and hint and local_hint != hint
):
hint = AMBIGUOUS_PROVIDER_HINT
elif not hint:
hint = local_hint
routing_decisions[key] = (hint, lookup_failed)
else:
hint = combined_provider_routing_hint(key, local_hint)
lookup_failed = provider_routing_database_failed()
routing_decisions[key] = (hint, lookup_failed)
if lookup_failed:
message = "provider routing evidence lookup failed closed"
warn_candidate_failure(message, item.get("source") or input_file, key)
raise RuntimeError(message)
if hint != "qwen" and not (
keycheck_input_mode() == "postgres"
and hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT)
):
continue
seen_candidates.add(key)
yield key, item.get("source") or input_file, data
for item in read_plain_keys(plain_files, QWEN_KEY_REGEX):
key = item["key"]
if not is_qwen_key(key) or not key.startswith("sk-sp-"):
continue
if key not in seen_plain:
seen_plain.add(key)
seen_candidates.add(key)
yield key, item["source"], {}
owned_retry_paths = {
os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values()
}
retry_files = [
path for path in (trusted_retry_files or [])
if os.path.normcase(os.path.abspath(path)) in owned_retry_paths
]
for item in read_plain_keys(retry_files, QWEN_KEY_REGEX):
key = item["key"]
if not is_qwen_key(key) or key in seen_candidates:
continue
if not key.startswith("sk-sp-"):
hint = combined_provider_routing_hint(key, "qwen")
if provider_routing_database_failed():
message = "provider routing evidence lookup failed closed"
warn_candidate_failure(message, item["source"], key)
raise RuntimeError(message)
if hint != "qwen":
continue
seen_candidates.add(key)
yield key, item["source"], {}
def redact_text(text, key):
redacted = str(text or "")[:1000]
if key:
redacted = redacted.replace(key, "***REDACTED***")
return QWEN_KEY_REGEX.sub("***REDACTED***", redacted)
def parse_error_response(response, key):
try:
payload = response.json()
except ValueError:
payload = {}
error = payload.get("error") if isinstance(payload, dict) else {}
if not isinstance(error, dict):
error = {}
message = error.get("message") or response.text[:500]
return {
"http_status": response.status_code,
"code": error.get("code") or error.get("type") or "",
"type": error.get("type") or "",
"message": redact_text(message, key),
}
def classify_error(error):
http_status = int(error.get("http_status") or 0)
code = str(error.get("code") or "").lower()
message = str(error.get("message") or "").lower()
if http_status == 401 or "invalid_api_key" in code or "incorrect api key" in message:
return "DEAD"
if http_status == 402 or "arrearage" in code or any(item in message for item in (
"arrearage", "arrears", "billing", "balance", "overdue", "payment",
"insufficient credit", "credit balance",
)):
return "NO_BALANCE"
if http_status == 403:
return "RESTRICTED"
if http_status == 429:
return "LIMITED"
if 500 <= http_status <= 599:
return "NETWORK"
return "UNKNOWN"
def choose_chat_model(models):
models = [str(model or "").replace("models/", "") for model in models if model]
by_lower = {model.lower(): model for model in models}
for model in CHAT_MODEL_PRIORITY:
if model.lower() in by_lower:
return by_lower[model.lower()]
for model in models:
lowered = model.lower()
if any(marker in lowered for marker in NON_CHAT_MODEL_MARKERS):
continue
if any(marker in lowered for marker in ("qwen", "qwq", "qvq")):
return model
return ""
def probe_chat_completion(key, base_url, model, proxy, timeout, debug=False):
if not model:
return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""}
url = f"{normalize_base_url(base_url)}/chat/completions"
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1}
try:
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc)[:1000], "model": model}
if debug:
print(f" DEBUG {endpoint_label(base_url)} chat ping {model}: HTTP {response.status_code}: {redact_text(response.text[:500], key)}")
if response.status_code == 200:
return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model}
error = parse_error_response(response, key)
return {"status": classify_error(error), "error": error, "message": error.get("message") or "", "model": model}
def parse_models(payload):
if not isinstance(payload, dict):
return [], []
model_infos = payload.get("data")
if not isinstance(model_infos, list):
model_infos = payload.get("models") if isinstance(payload.get("models"), list) else []
models = []
for item in model_infos:
if not isinstance(item, dict):
continue
model_id = item.get("id") or item.get("model") or item.get("name")
if model_id:
models.append(str(model_id).replace("models/", ""))
return sorted(set(models)), model_infos
def notable_models(models):
notable = []
for model in models:
lowered = model.lower()
if any(marker in lowered for marker in MODEL_MARKERS):
notable.append(model)
return notable[:30]
def check_base_url(key, base_url, proxy, timeout, debug=False):
url = f"{normalize_base_url(base_url)}/models"
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
try:
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {
"base_url": base_url,
"region": endpoint_label(base_url),
"status": "NETWORK",
"message": str(exc)[:1000],
}
if debug:
print(f" DEBUG {endpoint_label(base_url)} /models: HTTP {response.status_code}: {redact_text(response.text[:500], key)}")
if response.status_code == 200:
try:
payload = response.json()
except ValueError:
payload = {}
models, model_infos = parse_models(payload)
chat_model = choose_chat_model(models)
probe = probe_chat_completion(key, base_url, chat_model, proxy, timeout, debug)
probe_status = probe.get("status") or "UNKNOWN"
status = "VALID" if probe_status in ("GENERATION_OK", "NO_CONTEXT") else probe_status
probe_message = probe.get("message") or json.dumps(probe.get("error") or {}, ensure_ascii=False)[:1000]
return {
"base_url": base_url,
"region": endpoint_label(base_url),
"status": status,
"authenticated": True,
"model_count": len(models),
"models": notable_models(models),
"all_model_count": len(models),
"model_infos_count": len(model_infos),
"llm_probe_status": probe_status,
"llm_probe_model": probe.get("model", chat_model),
"message": (
f"chat ping ok; model={chat_model}; models={len(models)}"
if probe_status == "GENERATION_OK"
else f"models authenticated; generation_probe={probe_status}; models={len(models)}; {probe_message}"
)[:1000],
"error": probe.get("error") or {},
}
error = parse_error_response(response, key)
return {
"base_url": base_url,
"region": endpoint_label(base_url),
"status": classify_error(error),
"error": error,
"message": error.get("message") or "",
}
def choose_final_status(key, attempts, has_custom_base_urls):
statuses = [attempt.get("status") for attempt in attempts]
for status in ("VALID", "NO_BALANCE", "LIMITED", "RESTRICTED", "UNKNOWN", "NO_CONTEXT", "NETWORK"):
if status in statuses:
return status
return "DEAD"
def check_key(key, base_urls, has_custom_base_urls, proxy, timeout, debug=False):
rejection = qwen_key_rejection_reason(key)
if rejection:
return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True}
attempts = []
for base_url in base_urls:
result = check_base_url(key, base_url, proxy, timeout, debug)
attempts.append(result)
if result.get("status") == "VALID":
return {
"status": "VALID",
"region": result.get("region"),
"base_url": result.get("base_url"),
"model_count": result.get("model_count", 0),
"models": result.get("models", []),
"llm_probe_status": result.get("llm_probe_status", ""),
"llm_probe_model": result.get("llm_probe_model", ""),
"authenticated": bool(result.get("authenticated")),
"attempts": attempts,
"message": f"models={result.get('model_count', 0)} region={result.get('region')}",
}
status = choose_final_status(key, attempts, has_custom_base_urls)
message = ""
for attempt in attempts:
if attempt.get("status") == status:
message = attempt.get("message") or json.dumps(attempt.get("error") or {}, ensure_ascii=False)[:1000]
break
return {"status": status, "attempts": attempts, "message": message}
def write_result(key, result, source, finding):
rejection = qwen_key_rejection_reason(key)
if rejection:
raise ValueError(rejection)
status = result.get("status") or "UNKNOWN"
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
if status == "VALID":
extra = f"{result.get('region', '')};probe_model={result.get('llm_probe_model', '')};models={','.join(result.get('models', []))[:500]}"
else:
extra = source
commit_status_transaction(
CHECKED_FILE,
STATUS_FILES,
key,
status,
result.get("message", ""),
extra,
)
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
def retry_statuses_from_args(args):
retry_statuses = set()
if args.retry_network:
retry_statuses.add("NETWORK")
if args.retry_limited:
retry_statuses.add("LIMITED")
if args.retry_unknown:
retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
if args.retry_restricted:
retry_statuses.add("RESTRICTED")
if args.retry_no_balance:
retry_statuses.add("NO_BALANCE")
if args.retry_valid:
retry_statuses.add("VALID")
return retry_statuses
def retry_input_files_from_args(args):
if args.recheck_all:
statuses = list(STATUS_FILES)
else:
statuses = []
if args.retry_network:
statuses.append("NETWORK")
if args.retry_limited:
statuses.append("LIMITED")
if args.retry_unknown:
statuses.extend(("UNKNOWN", "NO_CONTEXT"))
if args.retry_restricted:
statuses.append("RESTRICTED")
if args.retry_no_balance:
statuses.append("NO_BALANCE")
if args.retry_valid:
statuses.append("VALID")
return list(dict.fromkeys(STATUS_FILES[status] for status in statuses))
def base_urls_from_args(args):
env_urls = split_csv(os.getenv("QWEN_BASE_URLS") or os.getenv("DASHSCOPE_BASE_URLS"))
custom_urls = []
for value in args.base_url:
custom_urls.extend(split_csv(value))
custom_urls.extend(env_urls)
default_urls = [] if args.no_default_base_urls else DEFAULT_BASE_URLS
return unique_ordered(custom_urls + default_urls), bool(custom_urls)
def parse_args():
parser = argparse.ArgumentParser(description="Qwen/DashScope key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=15)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--base-url", action="append", default=[], help="Extra DashScope/OpenAI-compatible base URL; can be repeated")
parser.add_argument("--no-default-base-urls", action="store_true")
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-restricted", action="store_true")
parser.add_argument("--retry-no-balance", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
parser.add_argument("--debug", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = retry_statuses_from_args(args)
retry_input_files = retry_input_files_from_args(args)
base_urls, has_custom_base_urls = base_urls_from_args(args)
if not base_urls:
raise SystemExit("No Qwen/DashScope base URLs configured")
print("--- Qwen/DashScope key checker ---")
print("base_urls: " + ", ".join(endpoint_label(url) for url in base_urls))
processed = 0
skipped = 0
for key, source, finding in iter_candidate_keys(args.input, args.plain, retry_input_files):
finding = dict(finding or {})
candidate_route = str(finding.pop(CANDIDATE_PROVIDER_ROUTE_FIELD, "") or "").lower()
rejection = qwen_key_rejection_reason(key)
if rejection:
skipped += 1
warn_rejected_candidate(rejection, source, key)
continue
try:
should_skip = should_skip_key(
key, checked, known, args, retry_statuses, service=SERVICE,
source=source, finding=finding, detector=DETECTOR,
)
except Exception as exc:
warn_candidate_failure(f"candidate preparation failed: {exc}", source, key)
raise
if should_skip:
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
try:
print(f"\n[{processed}] Qwen/DashScope candidate {mask_secret(key)} from {source}")
proxy = next(proxy_cycler) if proxy_cycler else None
routing_hint = "qwen"
if keycheck_input_mode() == "postgres":
if candidate_route == SERVICE:
routing_hint = SERVICE
else:
routing_hint = combined_provider_routing_hint(key, finding_provider_routing_hint(finding))
if provider_routing_database_failed():
raise RuntimeError("provider routing evidence lookup failed closed")
if routing_hint in (AMBIGUOUS_PROVIDER_HINT, AMBIGUOUS_GENERIC_SK_HINT):
result = resolve_provider_key(
key, finding, proxy, args.timeout, args.debug,
hint=routing_hint, origin_service=SERVICE,
)
else:
result = check_key(key, base_urls, has_custom_base_urls, proxy, args.timeout, args.debug)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
except Exception as exc:
warn_candidate_failure(f"candidate processing failed: {exc}", source, key)
raise
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
@@ -0,0 +1,403 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
import re
from collections import Counter
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl, classify_common_http_status, commit_status_transaction,
default_input_file, default_proxy_file, ensure_output_files, iter_findings,
keycheck_input_mode,
load_checked_statuses, load_known_keys, load_proxies, mask_secret,
read_plain_keys, record_validation_result, recover_status_transaction,
request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event,
)
SERVICE = "replicate"
DETECTOR = "Replicate"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "replicateChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "replicateResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "replicateAlive.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "replicateDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "replicateRestricted.txt"),
"LIMITED": os.path.join(OUTPUT_DIR, "replicateLimited.txt"),
"NO_BALANCE": os.path.join(OUTPUT_DIR, "replicateNoBalance.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "replicateNetwork.txt"),
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "replicateNoContext.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "replicateUnknown.txt"),
}
KEY_REGEX = re.compile(r"\br8_[A-Za-z0-9]{30,}\b")
API_BASE = "https://api.replicate.com/v1"
ACCOUNT_URL = f"{API_BASE}/account"
RESOURCE_ENDPOINTS = {
"predictions": f"{API_BASE}/predictions",
"deployments": f"{API_BASE}/deployments",
"trainings": f"{API_BASE}/trainings",
}
NO_BALANCE_MARKERS = (
"balance",
"billing",
"credit",
"credits",
"payment",
"insufficient",
"depleted",
"no credits",
"out of credit",
"run out of credit",
)
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def auth_headers(key):
return {"Authorization": f"Bearer {key}", "Accept": "application/json"}
def redacted_error_message(response, key):
return request_error_message(response).replace(key, "***REDACTED***")
def classify_replicate_response(response, key):
message = redacted_error_message(response, key).lower()
if response.status_code == 402 or any(marker in message for marker in NO_BALANCE_MARKERS):
return "NO_BALANCE"
if response.status_code == 403:
return "RESTRICTED"
return classify_common_http_status(response.status_code)
def api_get(key, url, proxy, timeout, debug=False):
try:
response = requests.get(url, headers=auth_headers(key), proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc)[:1000], "payload": None}
if debug:
detail = "ok" if response.status_code == 200 else redacted_error_message(response, key)[:500]
print(f" DEBUG GET {url}: HTTP {response.status_code}: {detail}")
if response.status_code != 200:
return {
"status": classify_replicate_response(response, key),
"http_status": response.status_code,
"message": redacted_error_message(response, key),
"payload": None,
}
try:
payload = response.json() if response.text else {}
except ValueError:
payload = {}
return {"status": "OK", "http_status": 200, "message": "ok", "payload": payload}
def paginated_items(payload):
if isinstance(payload, list):
return payload
if not isinstance(payload, dict):
return []
for key in ("results", "data", "items"):
value = payload.get(key)
if isinstance(value, list):
return value
return []
def text_value(value):
return str(value or "").strip()
def compact_model_ref(value):
if isinstance(value, str):
return value.strip()
if not isinstance(value, dict):
return ""
owner = text_value(value.get("owner") or value.get("model_owner"))
name = text_value(value.get("name") or value.get("model_name"))
if owner and name:
return f"{owner}/{name}"
for key in ("model", "id", "slug"):
item = text_value(value.get(key))
if item:
return item
url = text_value(value.get("url") or value.get("web_url"))
if "replicate.com/" in url:
return url.rstrip("/").split("replicate.com/", 1)[-1]
return ""
def model_refs_from_item(item):
if not isinstance(item, dict):
return []
refs = []
for key in ("model", "destination", "source_model", "base_model"):
ref = compact_model_ref(item.get(key))
if ref:
refs.append(ref)
for key in ("version", "latest_version", "current_release"):
value = item.get(key)
if isinstance(value, dict):
ref = compact_model_ref(value.get("model") or value.get("destination"))
if ref:
refs.append(ref)
return sorted(set(refs))
def summarize_predictions(payload, limit=10):
items = paginated_items(payload)
models = sorted({text_value(item.get("model")) for item in items if isinstance(item, dict) and item.get("model")})
statuses = Counter(text_value(item.get("status")) for item in items if isinstance(item, dict) and item.get("status"))
samples = []
for item in items[:limit]:
if not isinstance(item, dict):
continue
samples.append({
"id": text_value(item.get("id"))[:80],
"status": text_value(item.get("status")),
"model": text_value(item.get("model")),
"source": text_value(item.get("source")),
"data_removed": bool(item.get("data_removed")),
"created_at": text_value(item.get("created_at")),
"completed_at": text_value(item.get("completed_at")),
})
return {
"prediction_count_sample": len(items),
"prediction_has_next_page": bool(isinstance(payload, dict) and payload.get("next")),
"prediction_status_counts": dict(statuses),
"prediction_models": models[:50],
"prediction_samples": samples,
}
def deployment_name(item):
owner = text_value(item.get("owner") or item.get("deployment_owner"))
name = text_value(item.get("name") or item.get("deployment_name"))
if owner and name and "/" not in name:
return f"{owner}/{name}"
return name or owner
def summarize_deployments(payload, limit=20):
items = paginated_items(payload)
models = sorted({ref for item in items for ref in model_refs_from_item(item)})
deployments = []
for item in items[:limit]:
if not isinstance(item, dict):
continue
current_release = item.get("current_release") if isinstance(item.get("current_release"), dict) else {}
deployments.append({
"name": deployment_name(item),
"model": next(iter(model_refs_from_item(item)), ""),
"version": text_value(item.get("version") or current_release.get("version"))[:80],
"hardware": text_value(item.get("hardware") or current_release.get("hardware")),
"min_instances": item.get("min_instances"),
"max_instances": item.get("max_instances"),
})
return {
"deployment_count": len(items),
"deployment_has_next_page": bool(isinstance(payload, dict) and payload.get("next")),
"deployment_models": models[:50],
"deployments": deployments,
}
def summarize_trainings(payload, limit=10):
items = paginated_items(payload)
models = sorted({ref for item in items for ref in model_refs_from_item(item)})
statuses = Counter(text_value(item.get("status")) for item in items if isinstance(item, dict) and item.get("status"))
samples = []
for item in items[:limit]:
if not isinstance(item, dict):
continue
samples.append({
"id": text_value(item.get("id"))[:80],
"status": text_value(item.get("status")),
"model": next(iter(model_refs_from_item(item)), ""),
"created_at": text_value(item.get("created_at")),
"completed_at": text_value(item.get("completed_at")),
})
return {
"training_count_sample": len(items),
"training_has_next_page": bool(isinstance(payload, dict) and payload.get("next")),
"training_status_counts": dict(statuses),
"training_models": models[:50],
"training_samples": samples,
}
def probe_account_resources(key, proxy, timeout, debug=False):
summaries = {}
endpoint_statuses = {}
model_refs = set()
ok_count = 0
total_items = 0
summarizers = {
"predictions": summarize_predictions,
"deployments": summarize_deployments,
"trainings": summarize_trainings,
}
for name, url in RESOURCE_ENDPOINTS.items():
result = api_get(key, url, proxy, timeout, debug)
endpoint_statuses[name] = {k: v for k, v in result.items() if k in ("status", "http_status", "message")}
if result.get("status") != "OK":
continue
ok_count += 1
summary = summarizers[name](result.get("payload"))
summaries.update(summary)
for key_name, value in summary.items():
if key_name.endswith("_models") and isinstance(value, list):
model_refs.update(value)
total_items += sum(
int(summary.get(field, 0) or 0)
for field in ("prediction_count_sample", "deployment_count", "training_count_sample")
)
if ok_count == len(RESOURCE_ENDPOINTS):
probe_status = "RESOURCE_OK" if total_items else "NO_RESOURCES"
elif ok_count:
probe_status = "PARTIAL"
else:
probe_status = next((item.get("status") for item in endpoint_statuses.values() if item.get("status")), "UNKNOWN")
return {
"probe": {"status": probe_status, "endpoints": endpoint_statuses},
"models": sorted(model_refs)[:50],
"model_count": len(model_refs),
**summaries,
}
def iter_candidate_decisions(input_file, plain_files):
seen_plain = set()
for item in iter_findings(input_file, [DETECTOR]):
key = item.get("credential_secret_text") or item["raw"]
if key:
yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key))
for item in read_plain_keys(plain_files, KEY_REGEX):
key = item["key"]
if key not in seen_plain:
seen_plain.add(key)
yield key, item["source"], {}, True
def extract_candidates(input_file, plain_files):
for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files):
if valid_format:
yield key, source, finding
def check_key(key, proxy, args):
account = api_get(key, ACCOUNT_URL, proxy, args.timeout, args.debug)
if account.get("status") != "OK":
return {k: v for k, v in account.items() if k != "payload"}
data = account.get("payload") if isinstance(account.get("payload"), dict) else {}
result = {
"status": "VALID",
"message": "account endpoint accepted",
"account": data.get("username") or data.get("name") or "",
"account_type": data.get("type") or "",
}
if not args.no_resource_probe:
result.update(probe_account_resources(key, proxy, args.timeout, args.debug))
result["message"] = (
f"account endpoint accepted; probe={result.get('probe', {}).get('status')}; "
f"models={result.get('model_count', 0)}; "
f"deployments={result.get('deployment_count', 0)}; "
f"predictions={result.get('prediction_count_sample', 0)}; "
f"trainings={result.get('training_count_sample', 0)}"
)
else:
result.update({"probe": {"status": "not_probed"}, "models": [], "model_count": 0})
return result
def write_result(key, result, source, finding):
status = result.get("status") or "UNKNOWN"
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
extra = ",".join(result.get("models") or [])[:1000] if status == "VALID" else source
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra or source,
)
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
def parse_args():
parser = argparse.ArgumentParser(description="Replicate key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=15)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-restricted", action="store_true")
parser.add_argument("--retry-no-balance", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
parser.add_argument("--no-resource-probe", action="store_true", help="Only call /account; skip read-only predictions/deployments/trainings probes.")
parser.add_argument("--debug", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = set()
if args.retry_network: retry_statuses.add("NETWORK")
if args.retry_limited: retry_statuses.add("LIMITED")
if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
if args.retry_restricted: retry_statuses.add("RESTRICTED")
if args.retry_no_balance: retry_statuses.add("NO_BALANCE")
if args.retry_valid: retry_statuses.add("VALID")
processed = skipped = 0
print("--- Replicate key checker ---")
print("Default mode: /account plus read-only /predictions, /deployments and /trainings probes. Use --no-resource-probe for /account only.")
print(f"proxy: {args.proxy_file}")
postgres_mode = keycheck_input_mode() == "postgres"
for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain):
if not valid_format and not postgres_mode:
skipped += 1
continue
if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] Replicate candidate {mask_secret(key)} from {source}")
result = (
check_key(key, next(proxy_cycler) if proxy_cycler else None, args)
if valid_format else
{"status": "NO_CONTEXT", "message": "candidate does not match canonical Replicate token format"}
)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
if result.get("status") == "VALID":
print(f" ACCOUNT: {result.get('account') or 'unknown'}")
print(f" MODELS: {result.get('model_count', 0)} from account resources")
notable = result.get("models") or []
if notable:
print(f" MODEL REFS: {', '.join(notable[:8])}")
print(f" PROBE: {(result.get('probe') or {}).get('status')}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+231
View File
@@ -0,0 +1,231 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
import re
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
append_jsonl, classify_common_http_status, commit_status_transaction,
default_input_file, default_proxy_file, ensure_output_files, iter_findings,
keycheck_input_mode,
load_checked_statuses, load_known_keys, load_proxies, mask_secret,
read_plain_keys, record_validation_result, recover_status_transaction,
request_error_message, require_provider_authority, service_output_dir, should_skip_key, write_keycheck_event,
)
SERVICE = "xai"
DETECTOR_NAMES = ["XAI", "XAi", "Xai"]
DETECTOR = "XAI"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "xaiChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "xaiResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "xaiAlive.txt"),
"NO_BALANCE": os.path.join(OUTPUT_DIR, "xaiNoBalance.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "xaiDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "xaiRestricted.txt"),
"LIMITED": os.path.join(OUTPUT_DIR, "xaiLimited.txt"),
"NO_CONTEXT": os.path.join(OUTPUT_DIR, "xaiNoContext.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "xaiNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "xaiUnknown.txt"),
}
KEY_REGEX = re.compile(r"\bxai-[A-Za-z0-9_-]{20,}\b")
MODELS_URL = "https://api.x.ai/v1/models"
CHAT_URL = "https://api.x.ai/v1/chat/completions"
CHAT_MODEL_PRIORITY = (
"grok-4.6",
"grok-4.5",
"grok-4.3",
"grok-4.20-0309-reasoning",
"grok-4.20-0309-non-reasoning",
)
NO_BALANCE_MARKERS = (
"quota",
"billing",
"balance",
"credit",
"credits",
"payment",
"insufficient",
"depleted",
"spending limit",
"no credits",
"used all available credits",
"doesn't have any credits",
)
DEAD_MARKERS = ("incorrect api key", "invalid api key", "api key provided", "invalid-argument")
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def iter_candidate_decisions(input_file, plain_files):
seen_plain = set()
for item in iter_findings(input_file, DETECTOR_NAMES):
key = item.get("credential_secret_text") or item["raw"]
if key:
yield key, item["source"], item["finding"], bool(KEY_REGEX.fullmatch(key))
for item in read_plain_keys(plain_files, KEY_REGEX):
key = item["key"]
if key not in seen_plain:
seen_plain.add(key)
yield key, item["source"], {}, True
def extract_candidates(input_file, plain_files):
for key, source, finding, valid_format in iter_candidate_decisions(input_file, plain_files):
if valid_format:
yield key, source, finding
def check_key(key, proxy, timeout):
try:
response = requests.get(MODELS_URL, headers={"Authorization": f"Bearer {key}", "Accept": "application/json"}, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc)}
if response.status_code == 200:
data = response.json() if response.text else {}
models = [item.get("id") for item in data.get("data", []) if isinstance(item, dict) and item.get("id")]
model_inventory = sorted(set(models))
model = choose_chat_model(models)
probe = probe_chat_completion(key, model, proxy, timeout)
if probe.get("status") != "GENERATION_OK":
return {
"status": probe.get("status") or "UNKNOWN",
"message": probe.get("message", ""),
"model_count": len(models),
"models": models[:20],
"model_inventory": model_inventory,
"llm_probe_status": probe.get("status"),
"llm_probe_model": probe.get("model", model),
"llm_probe_http_status": probe.get("http_status"),
}
return {
"status": "VALID", "message": f"chat ping ok; model={model}; models={len(models)}",
"model_count": len(models), "models": models[:20], "model_inventory": model_inventory,
"llm_probe_status": probe.get("status"), "llm_probe_model": model,
}
status = classify_xai_response(response)
return {"status": status, "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***")}
def classify_xai_response(response):
message = request_error_message(response).lower()
if any(marker in message for marker in DEAD_MARKERS):
return "DEAD"
if any(marker in message for marker in NO_BALANCE_MARKERS):
return "NO_BALANCE"
if response.status_code == 401:
return "DEAD"
if response.status_code == 403:
return "RESTRICTED"
if response.status_code == 429:
return "LIMITED"
return classify_common_http_status(response.status_code)
def choose_chat_model(models):
models = [str(model or "") for model in models if model]
by_lower = {model.lower(): model for model in models}
for model in CHAT_MODEL_PRIORITY:
if model.lower() in by_lower:
return by_lower[model.lower()]
for model in models:
if "grok" in model.lower():
return model
return models[0] if models else ""
def probe_chat_completion(key, model, proxy, timeout):
if not model:
return {"status": "NO_CONTEXT", "message": "no chat-capable model from /models", "model": ""}
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
payload = {"model": model, "messages": [{"role": "user", "content": "ping"}], "max_tokens": 1}
try:
response = requests.post(CHAT_URL, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {"status": "NETWORK", "message": str(exc), "model": model}
if response.status_code == 200:
return {"status": "GENERATION_OK", "message": "chat completion accepted", "model": model}
return {"status": classify_xai_response(response), "http_status": response.status_code, "message": request_error_message(response).replace(key, "***REDACTED***"), "model": model}
def write_result(key, result, source, finding):
status = result.get("status") or "UNKNOWN"
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
extra = ",".join(result.get("models") or [])[:500] if status == "VALID" else source
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, status, result.get("message", ""), extra,
)
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
def parse_args():
parser = argparse.ArgumentParser(description="xAI key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=15)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-restricted", action="store_true")
parser.add_argument("--retry-no-balance", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = set()
if args.retry_network: retry_statuses.add("NETWORK")
if args.retry_limited: retry_statuses.add("LIMITED")
if args.retry_unknown: retry_statuses.update({"UNKNOWN", "NO_CONTEXT"})
if args.retry_restricted: retry_statuses.add("RESTRICTED")
if args.retry_no_balance: retry_statuses.add("NO_BALANCE")
if args.retry_valid: retry_statuses.add("VALID")
processed = skipped = 0
print("--- xAI key checker ---")
postgres_mode = keycheck_input_mode() == "postgres"
for key, source, finding, valid_format in iter_candidate_decisions(args.input, args.plain):
if not valid_format and not postgres_mode:
skipped += 1
continue
if valid_format and should_skip_key(key, checked, known, args, retry_statuses, service=SERVICE, source=source, finding=finding, detector=DETECTOR):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] xAI candidate {mask_secret(key)} from {source}")
result = (
check_key(key, next(proxy_cycler) if proxy_cycler else None, args.timeout)
if valid_format else
{"status": "NO_CONTEXT", "message": "candidate does not match canonical xAI token format"}
)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+508
View File
@@ -0,0 +1,508 @@
import sys
sys.dont_write_bytecode = True
import argparse
import json
import os
import re
from urllib.parse import urlparse
import requests
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from keycheck_common import (
combined_provider_routing_hint,
commit_status_transaction,
default_input_file,
default_proxy_file,
ensure_output_files,
iter_findings,
keycheck_input_mode,
load_checked_statuses,
load_known_keys,
load_proxies,
mask_secret,
provider_routing_database_failed,
read_plain_keys,
record_validation_result,
recover_status_transaction,
require_provider_authority,
service_output_dir,
should_skip_key,
write_keycheck_event,
)
from keycheckers.provider_resolution import (
AMBIGUOUS_GENERIC_SK_HINT,
AMBIGUOUS_QWEN_DEEPSEEK_HINT,
resolve_provider_key,
)
SERVICE = "zai"
DETECTOR = "ZaiGLM"
OUTPUT_DIR = os.getenv("KEYCHECK_OUTPUT_DIR") or service_output_dir(SERVICE)
INPUT_FILE = os.getenv("KEYCHECK_INPUT_FILE") or default_input_file()
PROXY_FILE = os.getenv("KEYCHECK_PROXY_FILE") or default_proxy_file()
CHECKED_FILE = os.path.join(OUTPUT_DIR, "zaiChecked.txt")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "zaiResults.jsonl")
STATUS_FILES = {
"VALID": os.path.join(OUTPUT_DIR, "zaiAlive.txt"),
"NO_BALANCE": os.path.join(OUTPUT_DIR, "zaiNoBalance.txt"),
"DEAD": os.path.join(OUTPUT_DIR, "zaiDead.txt"),
"RESTRICTED": os.path.join(OUTPUT_DIR, "zaiRestricted.txt"),
"LIMITED": os.path.join(OUTPUT_DIR, "zaiLimited.txt"),
"NETWORK": os.path.join(OUTPUT_DIR, "zaiNetwork.txt"),
"UNKNOWN": os.path.join(OUTPUT_DIR, "zaiUnknown.txt"),
}
DEFAULT_BASE_URLS = (
"https://api.z.ai/api/paas/v4",
"https://open.bigmodel.cn/api/paas/v4",
)
ZAI_KEY_REGEX = re.compile(
r"(?<![A-Za-z0-9_.-])(?:"
r"(?:zai|sk)-[A-Za-z0-9][A-Za-z0-9_-]{20,505}|"
r"[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}"
r")(?![A-Za-z0-9_.-])"
)
ZAI_DOTTED_KEY_REGEX = re.compile(r"^[A-Fa-f0-9]{32}\.[A-Za-z0-9_-]{16,128}$")
ZAI_CONTEXT_REGEX = re.compile(
r"(?:ZAI_API_KEY|GLM_API_KEY|ZHIPUAI_API_KEY|BIGMODEL_API_KEY|api\.z\.ai|"
r"open\.bigmodel\.cn|zhipuai|chatglm)",
re.IGNORECASE,
)
FOREIGN_KEY_PREFIXES = ("sk-ant-", "sk-or-", "sk-proj-", "sk-svcacct-", "sk-admin-")
AMBIGUOUS_HINTS = {AMBIGUOUS_QWEN_DEEPSEEK_HINT, AMBIGUOUS_GENERIC_SK_HINT}
AUTH_FAILURE_CODES = {"1000", "1001", "1003"}
AUTHENTICATED_NO_BALANCE_CODES = {"1113"}
AUTHENTICATED_LIMIT_CODES = {"1302", "1308", "1309", "1310", "1311"}
AUTHENTICATED_RESTRICTED_CODES = {"1005", "1220"}
PROBE_MODEL = "glm-5.2"
def normalize_base_url(value):
return str(value or "").strip().rstrip("/")
def split_csv(value):
if not value:
return []
values = value.split(",") if isinstance(value, str) else value
return [str(item).strip() for item in values if str(item).strip()]
def unique_ordered(values):
output = []
seen = set()
for value in values:
normalized = normalize_base_url(value)
if normalized and normalized not in seen:
seen.add(normalized)
output.append(normalized)
return output
def base_urls_from_environment(extra=None, include_defaults=True):
configured = []
for value in extra or ():
configured.extend(split_csv(value))
configured.extend(split_csv(os.getenv("ZAI_BASE_URLS") or os.getenv("ZHIPU_BASE_URLS")))
defaults = DEFAULT_BASE_URLS if include_defaults else ()
return unique_ordered([*configured, *defaults])
def endpoint_label(base_url):
parsed = urlparse(base_url)
return parsed.netloc or base_url
def key_from_text(*values):
for value in values:
match = ZAI_KEY_REGEX.search(str(value or ""))
if match:
return match.group(0)
return ""
def key_rejection_reason(key):
value = str(key or "")
try:
encoded = value.encode("utf-8", errors="strict")
except UnicodeEncodeError:
return "candidate is not valid UTF-8"
if len(encoded) > 512:
return "candidate exceeds the 512-byte key limit"
if value.startswith(FOREIGN_KEY_PREFIXES):
return "candidate has a foreign provider prefix"
if not ZAI_KEY_REGEX.fullmatch(value):
return "candidate does not match a bounded ZAI key format"
return ""
def finding_detector_names(finding):
if not isinstance(finding, dict):
return set()
extra = finding.get("ExtraData") if isinstance(finding.get("ExtraData"), dict) else {}
return {
name for name in (
str(finding.get("DetectorName") or finding.get("DetectorType") or "").strip().lower(),
str(extra.get("name") or "").strip().lower(),
) if name
}
def finding_provider_routing_hint(finding, key=""):
if not isinstance(finding, dict):
return "zai" if ZAI_DOTTED_KEY_REGEX.fullmatch(str(key or "")) else ""
context = finding.get("ScannerContext") if isinstance(finding.get("ScannerContext"), dict) else {}
persisted = str(context.get("provider_hint") or "").strip().lower()
if persisted:
return persisted
if "zaiglm" in finding_detector_names(finding) or ZAI_DOTTED_KEY_REGEX.fullmatch(str(key or "")):
return "zai"
text = "\n".join(str(context.get(name) or "") for name in ("nearby", "file"))
return "zai" if ZAI_CONTEXT_REGEX.search(text) else ""
def iter_candidate_decisions(input_file, plain_files, trusted_retry_files=None):
detectors = [
"ZaiGLM", "zaiglm", "CustomRegex", "QwenDashScope", "Qwen_DashScope",
"Qwen", "DashScope", "DeepSeek", "DeepSeekApiKey", "DeepSeek_API_Key",
"KimiMoonshot", "MoonshotAI", "Moonshot", "Kimi",
]
seen = set()
for item in iter_findings(input_file, detectors):
finding = item.get("finding") or {}
key = item.get("credential_secret_text") or key_from_text(
item.get("raw"), item.get("raw_v2"), finding.get("Raw"), finding.get("RawV2"),
)
if not key or key_rejection_reason(key):
continue
local_hint = finding_provider_routing_hint(finding, key)
hint = combined_provider_routing_hint(key, local_hint)
if provider_routing_database_failed():
raise RuntimeError("provider routing evidence lookup failed closed")
seen.add(key)
yield key, item.get("source") or input_file, finding, hint
owned_retry_paths = {
os.path.normcase(os.path.abspath(path)) for path in STATUS_FILES.values()
}
retry_paths = [
path for path in trusted_retry_files or ()
if os.path.normcase(os.path.abspath(path)) in owned_retry_paths
]
for item in read_plain_keys([*plain_files, *retry_paths], ZAI_KEY_REGEX):
key = item["key"]
if key in seen or key_rejection_reason(key):
continue
yield key, item["source"], {}, "zai"
def redact_text(value, key):
text = str(value or "")[:1000]
if key:
text = text.replace(key, "***REDACTED***")
return ZAI_KEY_REGEX.sub("***REDACTED***", text)
def parse_error(response, key):
try:
payload = response.json()
except ValueError:
payload = {}
error = payload.get("error") if isinstance(payload, dict) else {}
if not isinstance(error, dict):
error = {}
return {
"http_status": int(response.status_code),
"code": str(error.get("code") or (payload.get("code") if isinstance(payload, dict) else "") or ""),
"message": redact_text(
error.get("message") or error.get("msg") or (
payload.get("message") or payload.get("msg") if isinstance(payload, dict) else ""
) or response.text[:500],
key,
),
}
def classify_error(error):
http_status = int(error.get("http_status") or 0)
code = str(error.get("code") or "")
message = str(error.get("message") or "").lower()
if http_status == 402 or any(marker in message for marker in (
"insufficient balance", "balance is insufficient", "no balance", "account balance",
"recharge", "payment required", "billing arrears", "credit balance",
)):
return "NO_BALANCE", True
if code in AUTHENTICATED_NO_BALANCE_CODES:
return "NO_BALANCE", True
if any(marker in message for marker in (
"quota", "rate limit", "rate-limit", "too many requests", "resource exhausted",
"concurrency limit", "usage limit",
)):
return "LIMITED", True
if code in AUTHENTICATED_LIMIT_CODES:
return "LIMITED", True
if any(marker in message for marker in (
"permission denied", "access denied", "not authorized for", "model access", "forbidden",
)):
return "RESTRICTED", True
if code in AUTHENTICATED_RESTRICTED_CODES:
return "RESTRICTED", True
if http_status == 401 or code in AUTH_FAILURE_CODES:
return "DEAD", False
if 500 <= http_status <= 599 or code in {"1200", "1230", "1234", "1305"}:
return "NETWORK", False
if http_status == 429:
return "LIMITED", False
if http_status == 403:
return "RESTRICTED", False
return "UNKNOWN", False
def probe_chat_completion(key, base_url, model, proxy, timeout, debug=False):
if not model:
return {
"status": "UNKNOWN", "model": "",
"message": "no chat-capable model returned by /models",
}
url = f"{normalize_base_url(base_url)}/chat/completions"
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
payload = {
"model": model,
"messages": [{"role": "user", "content": "ping"}],
"max_tokens": 1,
"stream": False,
}
try:
response = requests.post(url, headers=headers, json=payload, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {
"status": "NETWORK", "model": model,
"message": redact_text(exc, key),
}
if debug:
print(
f" DEBUG {endpoint_label(base_url)} chat probe {model}: HTTP {response.status_code}: "
f"{redact_text(response.text[:500], key)}"
)
if response.status_code == 200:
try:
response_payload = response.json()
except ValueError as exc:
return {
"status": "UNKNOWN", "model": model, "http_status": response.status_code,
"message": f"invalid chat completion response: {exc}",
}
if isinstance(response_payload, dict) and response_payload.get("choices"):
return {
"status": "GENERATION_OK", "model": model,
"http_status": response.status_code, "message": "chat completion accepted",
}
error = parse_error(response, key)
status, authenticated = classify_error(error)
return {
"status": status, "authenticated": authenticated, "model": model,
"http_status": response.status_code, "business_code": error.get("code") or "",
"error": error, "message": error.get("message") or "",
}
def check_base_url(key, base_url, proxy, timeout, debug=False):
url = f"{normalize_base_url(base_url)}/models"
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
try:
response = requests.get(url, headers=headers, proxies=proxy, timeout=timeout)
except requests.RequestException as exc:
return {
"status": "NETWORK", "base_url": base_url,
"region": endpoint_label(base_url), "message": redact_text(exc, key),
}
if debug:
print(
f" DEBUG {endpoint_label(base_url)} /models: HTTP {response.status_code}: "
f"{redact_text(response.text[:500], key)}"
)
if response.status_code == 200:
try:
payload = response.json()
data = payload.get("data") if isinstance(payload, dict) else None
if not isinstance(data, list):
raise ValueError("missing data model list")
models = sorted({
str(item.get("id") or item.get("name") or "")
for item in data if isinstance(item, dict) and (item.get("id") or item.get("name"))
})
except (TypeError, ValueError, json.JSONDecodeError) as exc:
return {
"status": "UNKNOWN", "base_url": base_url,
"region": endpoint_label(base_url), "message": f"invalid models response: {exc}",
}
probe_model = PROBE_MODEL
probe = probe_chat_completion(key, base_url, probe_model, proxy, timeout, debug)
probe_status = probe.get("status") or "UNKNOWN"
if probe_status == "GENERATION_OK":
status = "VALID"
elif probe_status in {"NO_BALANCE", "LIMITED", "RESTRICTED", "NETWORK"}:
status = probe_status
elif probe_status == "DEAD":
status = "RESTRICTED"
else:
status = "UNKNOWN"
probe_message = probe.get("message") or json.dumps(probe.get("error") or {}, ensure_ascii=False)
message = (
f"chat probe ok; model={probe_model}; models={len(data)}"
if status == "VALID"
else f"models authenticated; generation_probe={probe_status}; model={probe_model}; "
f"models={len(data)}; {probe_message}"
)
return {
"status": status, "authenticated": True, "base_url": base_url,
"region": endpoint_label(base_url), "model_count": len(data),
"models": models[:30], "llm_probe_status": probe_status,
"llm_probe_model": probe.get("model") or probe_model,
"llm_probe_http_status": probe.get("http_status"),
"business_code": probe.get("business_code") or "",
"probe": probe, "error": probe.get("error") or {},
"message": message[:1000],
}
error = parse_error(response, key)
status, authenticated = classify_error(error)
return {
"status": status, "authenticated": authenticated, "base_url": base_url,
"region": endpoint_label(base_url), "http_status": response.status_code,
"business_code": error.get("code") or "", "error": error,
"message": error.get("message") or "",
}
def check_key(key, base_urls, proxy, timeout, debug=False):
rejection = key_rejection_reason(key)
if rejection:
return {"status": "UNKNOWN", "message": rejection, "candidate_rejected": True}
attempts = []
for base_url in base_urls:
result = check_base_url(key, base_url, proxy, timeout, debug)
attempts.append(result)
if result.get("authenticated"):
return {**result, "attempts": attempts}
statuses = [attempt.get("status") for attempt in attempts]
status = next(
(candidate for candidate in ("NETWORK", "LIMITED", "RESTRICTED", "UNKNOWN", "DEAD") if candidate in statuses),
"UNKNOWN",
)
selected = next((attempt for attempt in attempts if attempt.get("status") == status), {})
return {**selected, "status": status, "attempts": attempts}
def ensure_files():
ensure_output_files([CHECKED_FILE, RESULTS_FILE, *STATUS_FILES.values()])
recover_status_transaction(CHECKED_FILE, STATUS_FILES)
def write_result(key, result, source, finding):
status = result.get("status") or "UNKNOWN"
write_keycheck_event(SERVICE, RESULTS_FILE, key, result, source, finding, DETECTOR)
commit_status_transaction(
CHECKED_FILE, STATUS_FILES, key, status,
result.get("message", ""), result.get("region") or source,
)
record_validation_result(SERVICE, key, result, source, finding, DETECTOR)
def retry_statuses_from_args(args):
statuses = set()
for enabled, status in (
(args.retry_network, "NETWORK"),
(args.retry_limited, "LIMITED"),
(args.retry_unknown, "UNKNOWN"),
(args.retry_restricted, "RESTRICTED"),
(args.retry_no_balance, "NO_BALANCE"),
(args.retry_valid, "VALID"),
):
if enabled:
statuses.add(status)
return statuses
def retry_input_files_from_args(args):
if args.recheck_all:
statuses = list(STATUS_FILES)
else:
statuses = list(retry_statuses_from_args(args))
return [STATUS_FILES[status] for status in statuses]
def parse_args():
parser = argparse.ArgumentParser(description="ZAI / Zhipu GLM key checker")
parser.add_argument("--input", default=INPUT_FILE)
parser.add_argument("--plain", action="append", default=[])
parser.add_argument("--proxy-file", default=PROXY_FILE)
parser.add_argument("--timeout", type=int, default=15)
parser.add_argument("--max-keys", type=int, default=0)
parser.add_argument("--base-url", action="append", default=[])
parser.add_argument("--no-default-base-urls", action="store_true")
parser.add_argument("--retry-network", action="store_true")
parser.add_argument("--retry-limited", action="store_true")
parser.add_argument("--retry-unknown", action="store_true")
parser.add_argument("--retry-restricted", action="store_true")
parser.add_argument("--retry-no-balance", action="store_true")
parser.add_argument("--retry-valid", action="store_true")
parser.add_argument("--recheck-all", action="store_true")
parser.add_argument("--debug", action="store_true")
return parser.parse_args()
def main():
require_provider_authority(SERVICE)
args = parse_args()
ensure_files()
proxy_cycler = load_proxies(args.proxy_file)
checked = load_checked_statuses(CHECKED_FILE)
known = load_known_keys(CHECKED_FILE, STATUS_FILES)
retry_statuses = retry_statuses_from_args(args)
retry_files = retry_input_files_from_args(args)
base_urls = base_urls_from_environment(args.base_url, not args.no_default_base_urls)
if not base_urls:
raise SystemExit("No ZAI base URLs configured")
processed = 0
skipped = 0
postgres_mode = keycheck_input_mode() == "postgres"
for key, source, finding, routing_hint in iter_candidate_decisions(args.input, args.plain, retry_files):
is_ambiguous = routing_hint in AMBIGUOUS_HINTS
if routing_hint != "zai" and not (postgres_mode and is_ambiguous):
skipped += 1
continue
if should_skip_key(
key, checked, known, args, retry_statuses, service=SERVICE,
source=source, finding=finding, detector=DETECTOR,
):
skipped += 1
continue
if args.max_keys and processed >= args.max_keys:
break
processed += 1
print(f"\n[{processed}] ZAI candidate {mask_secret(key)} from {source}")
proxy = next(proxy_cycler) if proxy_cycler else None
if is_ambiguous:
result = resolve_provider_key(
key, finding, proxy, args.timeout, args.debug,
hint=routing_hint, origin_service=SERVICE,
)
else:
result = check_key(key, base_urls, proxy, args.timeout, args.debug)
print(f" STATUS: {result['status']} | {result.get('message', '')[:200]}")
write_result(key, result, source, finding)
known.add(key)
checked[key] = result["status"]
print(f"\nDone. Processed={processed}, skipped={skipped}, results={RESULTS_FILE}")
if __name__ == "__main__":
main()
+768
View File
@@ -0,0 +1,768 @@
import hashlib
import hmac
import json
import os
import shutil
import socket
import stat
from db_backend import parse_postgres_url
from process_identity import verify_retained_process
from runtime_security import (
canonical_path,
private_file_ready,
read_private_json,
reject_reparse_components,
require_trusted_native_executable,
sha256_file,
)
CODE_MANIFEST_SCHEMA = 5
APPLICATION_IMPORT_SUFFIXES = ('.py', '.pyw', '.pyc', '.pyd')
CONTROL_SCHEMA = 1
DISCOVERY_PRODUCER_ROLE = 'discovery-producer'
DISCOVERY_PRODUCER_SOURCES = ('gitlab', 'dockerhub', 'huggingface')
PHASE_INACTIVE = 'INACTIVE'
PHASE_ACTIVATING = 'ACTIVATING'
PHASE_ACTIVE = 'ACTIVE'
PHASE_STOPPING = 'STOPPING'
PHASE_FAILED_HOLD = 'FAILED_HOLD'
LIFECYCLE_PHASES = {
PHASE_INACTIVE,
PHASE_ACTIVATING,
PHASE_ACTIVE,
PHASE_STOPPING,
PHASE_FAILED_HOLD,
}
# These files collectively decide process ownership, database authority, and
# what data may be launched or persisted by the supervisor.
CODE_AUTHORITY_FILES = (
'owned_process.py',
'supervisor.py',
'supervisor_instance.py',
'console_runner.py',
'scanner.py',
'docker_shadow.py',
'keycheck_runner.py',
'dashboard.py',
'postgres_runtime.py',
'process_identity.py',
'runtime_security.py',
'scanner_db.py',
'db_backend.py',
'result_spool.py',
'janitor.py',
'result_bundle.py',
'result_ingester.py',
'jsonl_projector.py',
'admin_api.py',
'worker_api.py',
'worker_assignment.py',
'worker_package.py',
'scan_execution.py',
'keycheck_candidates.py',
'paths.py',
'target_identity.py',
'lifecycle_authority.py',
'audit_github_tokens.py',
'sync_alive_github_tokens.py',
'child_bootstrap.py',
'runtime_bootstrap.py',
'runtime_document.py',
'capacity_model.py',
'runtime_document_io.py',
'managed_files.py',
'host_agent_client.py',
'host_agent_protocol.py',
'host_agent_reconcile.py',
'host_agent_server.py',
'host_agent_apply.py',
'host_agent_lifecycle.py',
'host_agent_runtime.py',
'host_agent_state.py',
'worker_contracts.py',
'worker_assignment_runner.py',
)
REMOTE_WORKER_CODE_AUTHORITY_FILES = (
'db_backend.py',
'docker_depth_experiment.py',
'janitor.py',
'keycheck_candidates.py',
'lifecycle_authority.py',
'owned_process.py',
'paths.py',
'process_identity.py',
'query_policy.py',
'remote_worker_bootstrap.py',
'remote_worker_client.py',
'result_bundle.py',
'result_spool.py',
'runtime_security.py',
'scan_execution.py',
'scanner.py',
'scanner_db.py',
'supervisor_instance.py',
'target_identity.py',
'worker_contracts.py',
'worker_assignment_runner.py',
'worker_cli.py',
'worker_local_state.py',
'worker_supervisor.py',
'worker_package.py',
)
EXTERNAL_CODE_AUTHORITY_FILES = (
'../runtime/check-openrouter-keys.ps1',
'../start_core_runtime.ps1',
'../start_runtime.ps1',
'../stop_runtime.ps1',
) if os.name == 'nt' else ()
# The launchers execute before a manifest can be captured. First-launch trust
# therefore requires offline ACL hardening; manifests only detect later drift.
TRUFFLEHOG_MANIFEST_NAME = 'trufflehog'
GIT_MANIFEST_NAME = 'git'
CHILD_INSTANCE_FILE_ENV = 'TRUF_SUPERVISOR_INSTANCE_FILE'
CHILD_INSTANCE_ID_ENV = 'TRUF_SUPERVISOR_INSTANCE_ID'
CHILD_TOKEN_ENV = 'TRUF_SUPERVISOR_TOKEN'
CHILD_CONFIG_HASH_ENV = 'TRUF_SUPERVISOR_CONFIG_SHA256'
CHILD_SCRIPT_HASH_ENV = 'TRUF_SUPERVISOR_SHA256'
CHILD_MANIFEST_HASH_ENV = 'TRUF_SUPERVISOR_CODE_MANIFEST_SHA256'
CHILD_DSN_HASH_ENV = 'TRUF_SUPERVISOR_DSN_SHA256'
CHILD_KIND_ENV = 'TRUF_SUPERVISOR_CHILD_KIND'
PRIVATE_CHILD_ENV_KEYS = (
CHILD_INSTANCE_FILE_ENV,
CHILD_INSTANCE_ID_ENV,
CHILD_TOKEN_ENV,
CHILD_CONFIG_HASH_ENV,
CHILD_SCRIPT_HASH_ENV,
CHILD_MANIFEST_HASH_ENV,
CHILD_DSN_HASH_ENV,
CHILD_KIND_ENV,
'SCANNER_SUPERVISED',
'TRUF_MANAGED_POSTGRES_DSN',
'SCANNER_DB_URL',
'DATABASE_URL',
'SCANNER_DASHBOARD_DB_URL',
'KEYCHECK_DB_URL',
'TRUF_DASHBOARD_CANONICAL_LAUNCH',
'TRUF_DASHBOARD_HOST',
)
_LIBPQ_PRIVATE_ENV_KEYS = frozenset({
'PGPASSWORD',
'PGUSER',
'PGDATABASE',
'PGHOST',
'PGHOSTADDR',
'PGPORT',
'PGSERVICE',
'PGSERVICEFILE',
'PGPASSFILE',
'PGOPTIONS',
'PGSSLMODE',
'PGSSLKEY',
'PGSSLCERT',
'PGSSLROOTCERT',
})
_PRIVATE_EXTERNAL_ENV_KEYS = frozenset(PRIVATE_CHILD_ENV_KEYS) | _LIBPQ_PRIVATE_ENV_KEYS
class LifecycleAuthorityError(ValueError):
pass
def _manifest_payload(manifest):
return json.dumps(
manifest,
ensure_ascii=True,
sort_keys=True,
separators=(',', ':'),
).encode('utf-8')
def code_manifest_sha256(manifest):
return hashlib.sha256(_manifest_payload(manifest)).hexdigest()
def _is_reparse_point(path):
details = os.lstat(path)
if stat.S_ISLNK(details.st_mode):
return True
attributes = getattr(details, 'st_file_attributes', 0)
reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0)
return bool(attributes & reparse_attribute) or getattr(os.path, 'isjunction', lambda _path: False)(path)
def _application_root(app_dir=None):
candidate = os.path.abspath(os.fspath(app_dir or os.path.dirname(os.path.abspath(__file__))))
try:
details = os.lstat(candidate)
if _is_reparse_point(candidate):
raise LifecycleAuthorityError(f'application root reparse point is forbidden: {candidate}')
except OSError as exc:
raise LifecycleAuthorityError(f'application root is unavailable: {candidate}') from exc
if not stat.S_ISDIR(details.st_mode):
raise LifecycleAuthorityError(f'application root is not a directory: {candidate}')
return canonical_path(candidate)
def _external_authority_expected_path(root, name):
if name not in EXTERNAL_CODE_AUTHORITY_FILES:
raise LifecycleAuthorityError(f'code authority path is not an allowed external: {name}')
project_root = os.path.normcase(os.path.abspath(os.path.dirname(root)))
candidate = os.path.normcase(os.path.abspath(os.path.join(root, *name.split('/'))))
try:
contained = candidate != project_root and os.path.commonpath((project_root, candidate)) == project_root
except ValueError:
contained = False
if not contained:
raise LifecycleAuthorityError(f'external code authority path escapes the project root: {name}')
return candidate
def _require_external_authority_file(root, name):
path = _external_authority_expected_path(root, name)
try:
reject_reparse_components(path)
except OSError as exc:
raise LifecycleAuthorityError(f'external code authority reparse point is forbidden: {name}') from exc
try:
details = os.stat(path, follow_symlinks=False)
except OSError as exc:
raise LifecycleAuthorityError(f'required external code authority file is absent: {name}') from exc
if not stat.S_ISREG(details.st_mode):
raise LifecycleAuthorityError(f'external code authority file is not regular: {name}')
if canonical_path(path) != path:
raise LifecycleAuthorityError(f'external code authority path is not exact: {name}')
return path
def _application_code_files(root):
"""Return the exact importable application code surface."""
names = []
def raise_walk_error(exc):
raise LifecycleAuthorityError(f'unable to inspect the application root: {exc}') from exc
for current, directories, files in os.walk(root, followlinks=False, onerror=raise_walk_error):
for name in directories:
candidate = os.path.join(current, name)
try:
linked = _is_reparse_point(candidate)
except OSError as exc:
raise LifecycleAuthorityError(f'unable to inspect application directory: {candidate}') from exc
if linked:
relative = os.path.relpath(candidate, root).replace(os.sep, '/')
raise LifecycleAuthorityError(f'application directory reparse point is forbidden: {relative}')
relative_current = os.path.relpath(current, root)
in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep))
suffixes = ('.pyc',) if in_cache else APPLICATION_IMPORT_SUFFIXES
for name in files:
source_path = os.path.abspath(os.path.join(current, name))
try:
linked = _is_reparse_point(source_path)
except OSError as exc:
raise LifecycleAuthorityError(f'unable to inspect application file: {source_path}') from exc
if linked:
relative = os.path.relpath(source_path, root).replace(os.sep, '/')
raise LifecycleAuthorityError(f'application file reparse point is forbidden: {relative}')
if not name.lower().endswith(suffixes):
continue
path = canonical_path(source_path)
try:
if os.path.commonpath((root, path)) != root:
raise LifecycleAuthorityError(f'code authority path escapes the application root: {path}')
except ValueError as exc:
raise LifecycleAuthorityError(f'code authority path escapes the application root: {path}') from exc
names.append(os.path.relpath(source_path, root).replace(os.sep, '/'))
return sorted(names)
def code_authority_file_names(app_dir=None, existing_only=False):
root = _application_root(app_dir)
names = list(_application_code_files(root))
for name in (*CODE_AUTHORITY_FILES, *EXTERNAL_CODE_AUTHORITY_FILES):
path = canonical_path(os.path.join(root, name))
if not existing_only or os.path.isfile(path):
names.append(name)
return tuple(dict.fromkeys(names))
def resolve_manifest_executable(value, *, name=TRUFFLEHOG_MANIFEST_NAME, app_dir=None):
if name not in {TRUFFLEHOG_MANIFEST_NAME, GIT_MANIFEST_NAME}:
raise LifecycleAuthorityError('unsupported manifested executable')
label = 'TruffleHog' if name == TRUFFLEHOG_MANIFEST_NAME else 'Git'
text = str(value or '').strip()
if not text:
if name == GIT_MANIFEST_NAME:
text = 'git'
if os.name == 'nt':
private_git = os.path.join(os.path.dirname(_application_root(app_dir)), 'runtime', 'git', 'cmd', 'git.exe')
try:
reject_reparse_components(private_git)
except OSError as exc:
raise LifecycleAuthorityError('project-private Git path contains a reparse point') from exc
text = private_git if os.path.lexists(private_git) else text
else:
from paths import default_trufflehog_path
text = default_trufflehog_path()
candidate = shutil.which(text) if not os.path.isabs(text) and not any(sep in text for sep in ('/', '\\')) else text
if not candidate:
raise LifecycleAuthorityError(f'configured {label} executable is unavailable: {text}')
if os.name != 'nt':
if not os.path.isabs(candidate) or candidate != os.path.normpath(candidate):
raise LifecycleAuthorityError(f'configured {label} executable path must be exact and absolute: {candidate}')
try:
reject_reparse_components(candidate)
except (OSError, ValueError) as exc:
raise LifecycleAuthorityError(f'configured {label} executable path is unsafe: {candidate}') from exc
path = canonical_path(candidate)
if not os.path.isfile(path):
raise LifecycleAuthorityError(f'configured {label} executable is not a regular file: {path}')
return path
def manifest_authority_paths(
app_dir=None, trufflehog_path=None, policy_paths=None, existing_only=False, *,
git_path=None, include_executables=True,
):
"""List authority paths; exclude executables when applying private-file policy."""
root = _application_root(app_dir)
paths = []
names = list(code_authority_file_names(root, existing_only=existing_only))
# Required externals may never disappear from read-only preflight or an
# offline hardening plan, even when optional paths use existing_only.
names.extend(name for name in EXTERNAL_CODE_AUTHORITY_FILES if name not in names)
for name in names:
if name in EXTERNAL_CODE_AUTHORITY_FILES:
paths.append(_require_external_authority_file(root, name))
continue
path = canonical_path(os.path.join(root, *name.split('/')))
if os.path.isfile(path):
paths.append(path)
elif not existing_only:
raise LifecycleAuthorityError(f'code authority file is absent or outside the application root: {name}')
if include_executables:
if trufflehog_path or not existing_only:
try:
paths.append(resolve_manifest_executable(trufflehog_path))
except LifecycleAuthorityError:
if not existing_only or trufflehog_path:
raise
# Git is required even in preflight/offline hardening's existing-only mode.
paths.append(resolve_manifest_executable(git_path, name=GIT_MANIFEST_NAME, app_dir=root))
for value in policy_paths or ():
if not value:
continue
path = canonical_path(value)
if os.path.isfile(path):
paths.append(path)
elif not existing_only:
raise LifecycleAuthorityError(f'configured policy authority file is unavailable: {path}')
return tuple(dict.fromkeys(paths))
def build_code_manifest(
app_dir=None, trufflehog_path=None, policy_paths=None, *, git_path=None,
include_trufflehog=True,
):
root = _application_root(app_dir)
files = {}
for name in code_authority_file_names(root):
path = (
_require_external_authority_file(root, name)
if name in EXTERNAL_CODE_AUTHORITY_FILES
else canonical_path(os.path.join(root, *name.split('/')))
)
try:
contained = os.path.commonpath((root, path)) == root
except ValueError:
contained = False
explicitly_external = (
name in EXTERNAL_CODE_AUTHORITY_FILES
and path == _external_authority_expected_path(root, name)
)
if (not contained and not explicitly_external) or not os.path.isfile(path):
raise LifecycleAuthorityError(f'code authority file is absent or outside the application root: {name}')
files[name] = {'path': path, 'sha256': sha256_file(path)}
git_executable = resolve_manifest_executable(git_path, name=GIT_MANIFEST_NAME, app_dir=root)
executables = {
GIT_MANIFEST_NAME: {'path': git_executable, 'sha256': sha256_file(git_executable)},
}
if include_trufflehog:
executable = resolve_manifest_executable(trufflehog_path)
executables[TRUFFLEHOG_MANIFEST_NAME] = {
'path': executable,
'sha256': sha256_file(executable),
}
assets = {}
for value in policy_paths or ():
if not value:
continue
path = canonical_path(value)
if not os.path.isfile(path):
raise LifecycleAuthorityError(f'configured policy authority file is unavailable: {path}')
assets[path] = {'path': path, 'sha256': sha256_file(path)}
return {
'schema': CODE_MANIFEST_SCHEMA,
'root': root,
'files': files,
'executables': executables,
'assets': assets,
}
def normalize_code_manifest(manifest, *, required_names=None, external_names=None):
if not isinstance(manifest, dict) or manifest.get('schema') != CODE_MANIFEST_SCHEMA:
raise LifecycleAuthorityError('unsupported code authority manifest schema')
root_value = manifest.get('root') or ''
root = _application_root(root_value) if root_value else ''
values = manifest.get('files')
expected_names = set(values) if isinstance(values, dict) else set()
external_names = set(
EXTERNAL_CODE_AUTHORITY_FILES if external_names is None else external_names
)
if not external_names <= set(EXTERNAL_CODE_AUTHORITY_FILES):
raise LifecycleAuthorityError('code authority manifest has unsupported external files')
required_names = set(CODE_AUTHORITY_FILES if required_names is None else required_names)
required_names.update(external_names)
if not root or not isinstance(values, dict) or not required_names.issubset(expected_names):
raise LifecycleAuthorityError('code authority manifest has an incomplete file set')
files = {}
for name in sorted(expected_names):
value = values.get(name)
if not isinstance(value, dict):
raise LifecycleAuthorityError(f'invalid code authority entry: {name}')
if name in external_names:
path = os.path.normcase(os.path.abspath(os.fspath(value.get('path') or '')))
expected_path = _external_authority_expected_path(root, name)
else:
path = canonical_path(value.get('path') or '')
expected_path = canonical_path(os.path.join(root, *name.split('/')))
try:
contained = os.path.commonpath((root, expected_path)) == root
except ValueError:
contained = False
if not contained:
raise LifecycleAuthorityError(f'code authority path is not an allowed external: {name}')
digest = str(value.get('sha256') or '')
if path != expected_path:
raise LifecycleAuthorityError(f'code authority path mismatch: {name}')
if len(digest) != 64 or any(ch not in '0123456789abcdef' for ch in digest):
raise LifecycleAuthorityError(f'invalid code authority digest: {name}')
files[name] = {'path': path, 'sha256': digest}
executables = manifest.get('executables')
executable_names = set(executables) if isinstance(executables, dict) else set()
if executable_names not in (
{GIT_MANIFEST_NAME},
{TRUFFLEHOG_MANIFEST_NAME, GIT_MANIFEST_NAME},
):
raise LifecycleAuthorityError('code authority manifest has an incomplete executable set')
normalized_executables = {}
for name, label in ((TRUFFLEHOG_MANIFEST_NAME, 'TruffleHog'), (GIT_MANIFEST_NAME, 'Git')):
if name not in executables:
continue
executable = executables[name]
if not isinstance(executable, dict):
raise LifecycleAuthorityError(f'invalid {label} authority entry')
raw_path = executable.get('path')
executable_digest = str(executable.get('sha256') or '')
if not isinstance(raw_path, str) or not os.path.isabs(raw_path) or '\x00' in raw_path or len(executable_digest) != 64 or any(ch not in '0123456789abcdef' for ch in executable_digest):
raise LifecycleAuthorityError(f'invalid {label} authority identity')
executable_path = canonical_path(raw_path)
if os.name != 'nt' and executable_path != raw_path:
raise LifecycleAuthorityError(f'{label} authority path is not exact')
normalized_executables[name] = {'path': executable_path, 'sha256': executable_digest}
assets_value = manifest.get('assets')
if not isinstance(assets_value, dict):
raise LifecycleAuthorityError('code authority manifest has an invalid asset set')
assets = {}
for name in sorted(assets_value):
value = assets_value[name]
if not isinstance(value, dict):
raise LifecycleAuthorityError(f'invalid policy authority entry: {name}')
path = canonical_path(value.get('path') or '')
digest = str(value.get('sha256') or '')
if name != path or not os.path.isabs(path) or len(digest) != 64 or any(ch not in '0123456789abcdef' for ch in digest):
raise LifecycleAuthorityError(f'invalid policy authority identity: {name}')
assets[name] = {'path': path, 'sha256': digest}
return {
'schema': CODE_MANIFEST_SCHEMA,
'root': root,
'files': files,
'executables': normalized_executables,
'assets': assets,
}
def verify_code_manifest(
manifest, expected_sha256=None, require_private_acl=False, *,
required_names=None, external_names=None,
):
normalized = normalize_code_manifest(
manifest, required_names=required_names, external_names=external_names,
)
digest = code_manifest_sha256(normalized)
if expected_sha256 and not hmac.compare_digest(digest, str(expected_sha256)):
raise LifecycleAuthorityError('code authority manifest digest mismatch')
for name in (EXTERNAL_CODE_AUTHORITY_FILES if external_names is None else external_names):
if normalized['files'][name]['path'] != _require_external_authority_file(normalized['root'], name):
raise LifecycleAuthorityError(f'code authority path mismatch: {name}')
current_code = set(_application_code_files(normalized['root']))
manifested_code = {
name for name, value in normalized['files'].items()
if name.lower().endswith(APPLICATION_IMPORT_SUFFIXES)
and os.path.commonpath((normalized['root'], value['path'])) == normalized['root']
}
if current_code != manifested_code:
added = sorted(current_code - manifested_code)
removed = sorted(manifested_code - current_code)
detail = added[0] if added else removed[0] if removed else 'unknown'
raise LifecycleAuthorityError(f'application code authority file set drifted: {detail}')
entries = [(name, value, False) for name, value in normalized['files'].items()] + [
(f'executable:{name}', value, True) for name, value in normalized['executables'].items()
] + [(f'asset:{name}', value, False) for name, value in normalized['assets'].items()]
for name, value, native in entries:
if require_private_acl:
if native and os.name != 'nt':
try:
require_trusted_native_executable(value['path'])
except (OSError, ValueError) as exc:
raise LifecycleAuthorityError(f'code authority executable is not trusted: {name}: {exc}') from exc
elif not private_file_ready(value['path']):
raise LifecycleAuthorityError(f'code authority ACL is not exact-private: {name}')
try:
current = sha256_file(value['path'])
except OSError as exc:
raise LifecycleAuthorityError(f'unable to verify code authority file: {name}') from exc
if not hmac.compare_digest(current, value['sha256']):
raise LifecycleAuthorityError(f'code authority drifted: {name}')
return normalized
def dsn_sha256(dsn):
value = str(dsn or '')
return hashlib.sha256(value.encode('utf-8')).hexdigest() if value else ''
def supervised_child_environment(metadata, canonical_dsn, child_kind):
return {
'SCANNER_SUPERVISED': '1',
CHILD_INSTANCE_FILE_ENV: str(metadata['instance_file']),
CHILD_INSTANCE_ID_ENV: str(metadata['instance_id']),
CHILD_TOKEN_ENV: str(metadata['token']),
CHILD_CONFIG_HASH_ENV: str(metadata['config_sha256']),
CHILD_SCRIPT_HASH_ENV: str(metadata['supervisor_sha256']),
CHILD_MANIFEST_HASH_ENV: str(metadata['code_manifest_sha256']),
CHILD_DSN_HASH_ENV: str(metadata.get('canonical_dsn_sha256') or ''),
CHILD_KIND_ENV: str(child_kind),
'TRUF_MANAGED_POSTGRES_DSN': str(canonical_dsn or ''),
}
def strip_supervisor_credentials(env):
for key in list(env):
normalized = str(key).upper()
if normalized.startswith('TRUF_POSTGRES_') or normalized in _PRIVATE_EXTERNAL_ENV_KEYS:
env.pop(key, None)
return env
def _send_handshake(metadata, timeout=3):
control = metadata.get('control') or {}
request = {
'schema': CONTROL_SCHEMA,
'instance_id': metadata['instance_id'],
'token': metadata['token'],
'action': 'handshake',
}
payload = json.dumps(request, ensure_ascii=True, separators=(',', ':')).encode('utf-8') + b'\n'
with socket.create_connection((control.get('host'), int(control.get('port') or 0)), timeout=timeout) as sock:
sock.settimeout(timeout)
sock.sendall(payload)
sock.shutdown(socket.SHUT_WR)
chunks = []
total = 0
while True:
chunk = sock.recv(65536)
if not chunk:
break
total += len(chunk)
if total > 1024 * 1024:
raise LifecycleAuthorityError('supervisor handshake response is too large')
chunks.append(chunk)
try:
response = json.loads(b''.join(chunks).decode('utf-8'))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise LifecycleAuthorityError('invalid supervisor handshake response') from exc
if (
not isinstance(response, dict)
or response.get('schema') != CONTROL_SCHEMA
or response.get('instance_id') != metadata['instance_id']
or response.get('ok') is not True
or not isinstance(response.get('result'), dict)
):
raise LifecycleAuthorityError('authenticated supervisor handshake failed')
return response['result']
def _send_handshake_with_timeout_retry(metadata, timeout_retries=0):
retries = min(1, max(0, int(timeout_retries or 0)))
for attempt in range(retries + 1):
try:
return _send_handshake(metadata)
except TimeoutError:
if attempt >= retries:
raise
def verify_supervisor_command_line(arguments, supervisor_path, config_path):
"""Verify the unique script binding and config value retained by the OS."""
arguments = [str(argument) for argument in arguments]
options = {
'--runtime-bootstrap-entrypoint': [],
'--config': [],
}
option_value_indices = set()
for index, argument in enumerate(arguments):
for option in options:
if argument == option:
value = arguments[index + 1] if index + 1 < len(arguments) else ''
options[option].append(value)
if index + 1 < len(arguments):
option_value_indices.add(index + 1)
elif argument.startswith(option + '='):
options[option].append(argument.split('=', 1)[1])
expected_supervisor = canonical_path(supervisor_path)
direct_bindings = []
for index, argument in enumerate(arguments):
if index in option_value_indices or not argument or argument.startswith('-'):
continue
try:
if canonical_path(argument) == expected_supervisor:
direct_bindings.append(argument)
except (OSError, TypeError, ValueError):
continue
bindings = options['--runtime-bootstrap-entrypoint'] + direct_bindings
if len(bindings) != 1:
raise LifecycleAuthorityError('supervisor command line must contain exactly one bound supervisor script')
try:
binding_matches = canonical_path(bindings[0]) == expected_supervisor
except (OSError, TypeError, ValueError):
binding_matches = False
if not binding_matches:
raise LifecycleAuthorityError('supervisor command line bound supervisor script mismatch')
configs = options['--config']
if len(configs) != 1:
raise LifecycleAuthorityError('supervisor command line must contain exactly one bound config argument')
try:
config_matches = canonical_path(configs[0]) == canonical_path(config_path)
except (OSError, TypeError, ValueError):
config_matches = False
if not config_matches:
raise LifecycleAuthorityError('supervisor command line bound config argument mismatch')
def _verify_supervisor_process(metadata):
process = verify_retained_process(
metadata['pid'],
metadata['process_creation_time'],
metadata['executable'],
)
try:
arguments = process.command_line()
verify_supervisor_command_line(
arguments,
metadata['supervisor_path'],
metadata['config_path'],
)
finally:
process.close()
def require_active_supervisor_child(
config_path=None, child_kind=None, require_dsn=True, handshake_timeout_retries=0,
):
"""Authenticate a mutating child before it reads application inputs."""
instance_file = os.getenv(CHILD_INSTANCE_FILE_ENV) or ''
inherited_id = os.getenv(CHILD_INSTANCE_ID_ENV) or ''
inherited_token = os.getenv(CHILD_TOKEN_ENV) or ''
inherited_config_hash = os.getenv(CHILD_CONFIG_HASH_ENV) or ''
inherited_script_hash = os.getenv(CHILD_SCRIPT_HASH_ENV) or ''
inherited_manifest_hash = os.getenv(CHILD_MANIFEST_HASH_ENV) or ''
inherited_dsn_hash = os.getenv(CHILD_DSN_HASH_ENV) or ''
inherited_kind = os.getenv(CHILD_KIND_ENV) or ''
if not all((instance_file, inherited_id, inherited_token, inherited_config_hash, inherited_script_hash, inherited_manifest_hash)):
raise LifecycleAuthorityError('direct mutation is retired; use an authenticated active supervisor command')
if child_kind and inherited_kind != str(child_kind):
raise LifecycleAuthorityError('supervised child kind does not match the requested mutation entrypoint')
# Imported lazily to avoid a module cycle while supervisor metadata support
# itself imports the manifest helpers above.
from supervisor_instance import load_instance_metadata
try:
metadata = load_instance_metadata(instance_file)
except (OSError, ValueError) as exc:
raise LifecycleAuthorityError('private supervisor instance metadata is unavailable') from exc
if not hmac.compare_digest(metadata['instance_id'], inherited_id):
raise LifecycleAuthorityError('supervisor child instance identity mismatch')
if not hmac.compare_digest(metadata['token'], inherited_token):
raise LifecycleAuthorityError('supervisor child credential mismatch')
if metadata.get('activation_state') != PHASE_ACTIVE:
raise LifecycleAuthorityError('supervisor is not ACTIVE; mutation is refused')
if canonical_path(instance_file) != canonical_path(metadata.get('instance_file') or instance_file):
raise LifecycleAuthorityError('supervisor child instance path mismatch')
if config_path and canonical_path(config_path) != metadata['config_path']:
raise LifecycleAuthorityError('supervisor child config path mismatch')
expected_pairs = (
('config_sha256', inherited_config_hash),
('supervisor_sha256', inherited_script_hash),
('code_manifest_sha256', inherited_manifest_hash),
)
for key, inherited in expected_pairs:
if not hmac.compare_digest(str(metadata.get(key) or ''), inherited):
raise LifecycleAuthorityError(f'supervisor child {key} authority mismatch')
if not hmac.compare_digest(sha256_file(metadata['config_path']), metadata['config_sha256']):
raise LifecycleAuthorityError('supervisor config authority drifted')
verify_code_manifest(
metadata['code_manifest'],
metadata['code_manifest_sha256'],
require_private_acl=True,
)
_verify_supervisor_process(metadata)
dsn = os.getenv('TRUF_MANAGED_POSTGRES_DSN') or ''
if require_dsn:
try:
parsed = parse_postgres_url(dsn)
except ValueError as exc:
raise LifecycleAuthorityError('a canonical managed PostgreSQL DSN is required') from exc
if parsed['host'] != '127.0.0.1':
raise LifecycleAuthorityError('managed PostgreSQL DSN is not loopback-bound')
actual_dsn_hash = dsn_sha256(dsn)
if not inherited_dsn_hash or not hmac.compare_digest(actual_dsn_hash, inherited_dsn_hash):
raise LifecycleAuthorityError('managed PostgreSQL DSN authority mismatch')
if not hmac.compare_digest(str(metadata.get('canonical_dsn_sha256') or ''), inherited_dsn_hash):
raise LifecycleAuthorityError('private metadata PostgreSQL DSN authority mismatch')
for key in ('SCANNER_DB_URL', 'DATABASE_URL'):
if not hmac.compare_digest(str(os.getenv(key) or ''), dsn):
raise LifecycleAuthorityError(f'{key} does not match the managed PostgreSQL DSN')
if any(key.upper().startswith('PG') for key in os.environ):
raise LifecycleAuthorityError('libpq PG* environment overrides are forbidden for managed children')
handshake = _send_handshake_with_timeout_retry(metadata, handshake_timeout_retries)
if handshake.get('instance_id') != metadata['instance_id'] or handshake.get('activation_state') != PHASE_ACTIVE:
raise LifecycleAuthorityError('supervisor handshake did not confirm ACTIVE authority')
for key, value in expected_pairs:
if not hmac.compare_digest(str(handshake.get(key) or ''), value):
raise LifecycleAuthorityError(f'supervisor handshake {key} mismatch')
if require_dsn and not hmac.compare_digest(str(handshake.get('canonical_dsn_sha256') or ''), inherited_dsn_hash):
raise LifecycleAuthorityError('supervisor handshake PostgreSQL DSN authority mismatch')
return metadata
+1440
View File
File diff suppressed because it is too large Load Diff
+324
View File
@@ -0,0 +1,324 @@
import sys
sys.dont_write_bytecode = True
import argparse
import fnmatch
import hashlib
import os
import shutil
from db_backend import database_url_from_env, is_postgres_url
from paths import apply_path_config, default_project_paths
from migrate_runtime_safety import require_runtime_hardening_stopped
from postgres_runtime import load_postgres_environment
from runtime_security import ClusterAuthorityLock, reject_reparse_components
DESKTOP_HF = r"C:\Users\pro100noob\Desktop\HugginFace"
EXCLUDED_DIRS = {"__pycache__", ".git", ".opencode", "node_modules", "tmp", "runtime"}
APP_FILES = [
"app.py",
"console_runner.py",
"dashboard.py",
"keycheck_runner.py",
"migrate_layout.py",
"paths.py",
"scan_manager.py",
"scanner.py",
"scanner_db.py",
"supervisor.py",
"ui_components.py",
"config.yaml",
"secrets.yaml",
"requirements.txt",
"CHEATSHEET.md",
"DETECTOR_NOTES.md",
]
APP_DIRS = [".streamlit", "keycheckers"]
RESULT_FILES = ["found_secrets.jsonl", "scan_results.jsonl", "scan_errors.log", "scanner.db", "scanner.db-wal", "scanner.db-shm"]
KEYCHECK_SERVICE_SCRIPTS = {
"anthropic": [os.path.join("anthropic", "anthropicKeycheck.py")],
"aws": [os.path.join("aws", "awsKeycheck.py")],
"azure": [os.path.join("azure", "azureKeycheck.py")],
"deepseek": [os.path.join("deepseek", "deepseekKeycheck.py")],
"dockerhub": [os.path.join("dockerhub", "dockerhubKeycheck.py"), os.path.join("dockerhub", "dockerhub.txt")],
"gcp": [os.path.join("gcp", "gcpKeycheck.py"), os.path.join("gcp", "gcp.txt")],
"gemini": [os.path.join("gemini", "geminiKeycheck.py"), os.path.join("gemini", "gem.txt")],
"groq": [os.path.join("groq", "groqKeycheck.py")],
"github": [os.path.join("github", "githubKeycheck.py"), os.path.join("github", "github.txt")],
"gitlab": [os.path.join("gitlab", "gitlabKeycheck.py"), os.path.join("gitlab", "gitlab.txt")],
"kimi": [os.path.join("kimi", "kimiKeycheck.py")],
"openai": ["Keycheck.py"],
"openrouter": ["OpenrouterKeycheck.py"],
"provider_resolver": [os.path.join("provider_resolver", "providerResolverKeycheck.py")],
"qwen": [os.path.join("qwen", "qwenKeycheck.py")],
"replicate": [os.path.join("replicate", "replicateKeycheck.py")],
"xai": [os.path.join("xai", "xaiKeycheck.py")],
"huggingface": [os.path.join("huggingface", "huggingfaceKeycheck.py")],
"zai": [os.path.join("zai", "zaiKeycheck.py")],
}
KEYCHECK_TOP_LEVEL_PREFIXES = {
"openai": "openai",
"openrouter": "openrouter",
}
def load_config(config_path):
if not config_path:
return {'global': default_project_paths()}
try:
import yaml
except ImportError as e:
raise SystemExit("PyYAML is required for --config") from e
with open(config_path, "r", encoding="utf-8") as f:
config = apply_path_config(yaml.safe_load(f) or {}, config_path)
return config
def load_layout(config_path):
return load_config(config_path).get('global') or {}
def mkdir(path, dry_run=False):
if dry_run:
print(f"mkdir {path}")
return
os.makedirs(path, exist_ok=True)
def copy_file(src, dst, overwrite=False, dry_run=False):
if not os.path.exists(src):
return False
if os.path.lexists(dst) and not overwrite:
print(f"skip existing {dst}")
return False
parent = os.path.dirname(dst)
if parent:
mkdir(parent, dry_run)
if dry_run:
print(f"copy {src} -> {dst}")
return True
try:
reject_reparse_components(parent or os.path.dirname(os.path.abspath(dst)))
if os.path.lexists(dst):
reject_reparse_components(dst)
shutil.copy2(src, dst)
except OSError as e:
print(f"skip locked/unavailable {src}: {e}")
return False
print(f"copied {src} -> {dst}")
return True
def ignore_app_dir(_dir, names):
ignored = set()
for name in names:
if name in EXCLUDED_DIRS:
ignored.add(name)
if fnmatch.fnmatch(name, "*.pyc"):
ignored.add(name)
return ignored
def copy_dir(src, dst, overwrite=False, dry_run=False, verified_apply=False):
if not os.path.isdir(src):
return False
if os.path.lexists(dst) and not overwrite:
print(f"skip existing {dst}")
return False
if dry_run:
print(f"copytree {src} -> {dst}")
return True
if os.path.lexists(dst) and overwrite:
raise RuntimeError('legacy directory replacement is retired; existing directories are never replaced')
shutil.copytree(src, dst, ignore=ignore_app_dir, dirs_exist_ok=False)
print(f"copied {src} -> {dst}")
return True
def create_layout(layout, dry_run=False):
for key in ("project_dir", "runtime_dir", "results_dir", "queue_dir", "log_dir", "state_dir", "keycheck_dir", "work_dir"):
mkdir(layout[key], dry_run)
mkdir(os.path.join(layout["runtime_dir"], "imports"), dry_run)
def copy_app_files(source_dir, layout, overwrite=False, dry_run=False, verified_apply=False):
project_dir = layout["project_dir"]
if os.path.abspath(source_dir) == os.path.abspath(project_dir):
print("app source is already project_dir; app copy skipped")
return
for name in APP_FILES:
copy_file(os.path.join(source_dir, name), os.path.join(project_dir, name), overwrite, dry_run)
for name in APP_DIRS:
copy_dir(os.path.join(source_dir, name), os.path.join(project_dir, name), overwrite, dry_run, verified_apply)
def copy_scanner_runtime(old_root, layout, overwrite=False, dry_run=False, verified_apply=False):
for name in RESULT_FILES:
copy_file(os.path.join(old_root, name), os.path.join(layout["results_dir"], name), overwrite, dry_run)
for pattern in ("todo_*.txt", "checked_*.txt"):
if not os.path.isdir(old_root):
continue
for name in os.listdir(old_root):
if fnmatch.fnmatch(name, pattern):
copy_file(os.path.join(old_root, name), os.path.join(layout["queue_dir"], name), overwrite, dry_run)
copy_dir(os.path.join(old_root, "logs"), layout["log_dir"], overwrite, dry_run, verified_apply)
copy_dir(os.path.join(old_root, "state"), layout["state_dir"], overwrite, dry_run, verified_apply)
copy_file(os.path.join(old_root, "runner_state.json"), os.path.join(layout["state_dir"], "runner_state.json"), overwrite, dry_run)
def line_hash(line):
return hashlib.sha256(line.strip().encode("utf-8", errors="replace")).hexdigest()
def existing_line_hashes(path):
hashes = set()
if not os.path.exists(path):
return hashes
with open(path, "r", encoding="utf-8", errors="replace") as f:
for line in f:
if line.strip():
hashes.add(line_hash(line))
return hashes
def import_jsonl_dedupe(inputs, output, dry_run=False):
hashes = existing_line_hashes(output)
added = 0
if dry_run:
print(f"dedupe import {len(inputs)} file(s) -> {output}")
return 0
mkdir(os.path.dirname(output), dry_run=False)
with open(output, "a", encoding="utf-8") as dst:
for path in inputs:
if not os.path.exists(path):
continue
with open(path, "r", encoding="utf-8", errors="replace") as src:
for line in src:
if not line.strip():
continue
digest = line_hash(line)
if digest in hashes:
continue
dst.write(line if line.endswith("\n") else line + "\n")
hashes.add(digest)
added += 1
print(f"imported {added} unique finding line(s) into {output}")
return added
def copy_legacy_keychecker_outputs(layout, desktop_dir=DESKTOP_HF, overwrite=False, dry_run=False):
if not os.path.isdir(desktop_dir):
return
for service in KEYCHECK_SERVICE_SCRIPTS:
source_dir = os.path.join(desktop_dir, service)
target_dir = os.path.join(layout["keycheck_dir"], service)
if os.path.isdir(source_dir):
for name in os.listdir(source_dir):
if name.lower().endswith((".txt", ".jsonl")):
copy_file(os.path.join(source_dir, name), os.path.join(target_dir, name), overwrite, dry_run)
for service, prefix in KEYCHECK_TOP_LEVEL_PREFIXES.items():
target_dir = os.path.join(layout["keycheck_dir"], service)
for name in os.listdir(desktop_dir):
lower = name.lower()
if lower.startswith(prefix) and lower.endswith((".txt", ".jsonl")):
copy_file(os.path.join(desktop_dir, name), os.path.join(target_dir, name), overwrite, dry_run)
def merge_unique_lines(inputs, output, dry_run=False):
values = []
seen = existing_values = set()
if os.path.exists(output):
with open(output, "r", encoding="utf-8", errors="replace") as f:
existing_values = {line.strip().lstrip("\ufeff") for line in f if line.strip()}
seen = set(existing_values)
for path in inputs:
if not os.path.exists(path):
continue
with open(path, "r", encoding="utf-8", errors="replace") as f:
for line in f:
value = line.strip().lstrip("\ufeff")
if value and value not in seen:
values.append(value)
seen.add(value)
if dry_run:
print(f"merge {len(values)} unique line(s) -> {output}")
return
mkdir(os.path.dirname(output), dry_run=False)
with open(output, "a", encoding="utf-8") as f:
for value in values:
f.write(value + "\n")
print(f"appended {len(values)} unique line(s) -> {output}")
def import_huggingface_desktop(layout, desktop_dir=DESKTOP_HF, overwrite=False, dry_run=False):
if not os.path.isdir(desktop_dir):
print(f"Desktop HugginFace directory not found: {desktop_dir}")
return
jsonl_inputs = [os.path.join(desktop_dir, name) for name in os.listdir(desktop_dir) if fnmatch.fnmatch(name, "found_secrets*.jsonl")]
import_jsonl_dedupe(jsonl_inputs, os.path.join(layout["results_dir"], "found_secrets.jsonl"), dry_run)
merge_unique_lines([os.path.join(desktop_dir, "checked.txt")], os.path.join(layout["queue_dir"], "checked_huggingface.txt"), dry_run)
merge_unique_lines([os.path.join(desktop_dir, "todo.txt")], os.path.join(layout["queue_dir"], "todo_huggingface.txt"), dry_run)
copy_file(os.path.join(desktop_dir, "proxy.txt"), layout["proxy_file"], overwrite=False, dry_run=dry_run)
copy_file(os.path.join(desktop_dir, "requirements-keycheckers.txt"), os.path.join(layout["project_dir"], "requirements-keycheckers.txt"), overwrite, dry_run)
copy_file(os.path.join(desktop_dir, "KEYCHECKERS.md"), os.path.join(layout["project_dir"], "KEYCHECKERS.md"), overwrite, dry_run)
for service, rel_paths in KEYCHECK_SERVICE_SCRIPTS.items():
for rel_path in rel_paths:
src = os.path.join(desktop_dir, rel_path)
dst = os.path.join(layout["project_dir"], "keycheckers", service, os.path.basename(rel_path))
copy_file(src, dst, overwrite, dry_run)
copy_legacy_keychecker_outputs(layout, desktop_dir, overwrite, dry_run)
def parse_args():
parser = argparse.ArgumentParser(description="Copy/import legacy scanner files into the unified D:\\truf layout.")
parser.add_argument("--config", default="config.yaml")
parser.add_argument("--source-app", default=os.path.dirname(os.path.abspath(__file__)))
parser.add_argument("--target-app", help="Destination app directory. Defaults to <root_dir>\\app.")
parser.add_argument("--in-place", action="store_true", help="Use project_dir from config instead of copying to <root_dir>\\app.")
parser.add_argument("--old-root", default=r"D:\truf")
parser.add_argument("--desktop-hf", default=DESKTOP_HF)
parser.add_argument("--overwrite", action="store_true")
parser.add_argument("--dry-run", action="store_true")
parser.add_argument("--apply", action="store_true", help="Retired; production layout mutation is disabled")
parser.add_argument("--no-app-copy", action="store_true")
parser.add_argument("--no-desktop-import", action="store_true")
return parser.parse_args()
def main():
args = parse_args()
if args.apply and args.dry_run:
raise SystemExit('--apply and --dry-run are mutually exclusive')
if args.apply:
raise SystemExit(
'migrate_layout --apply is retired because the production layout is already migrated. '
'Use reviewed offline backup/restore tooling for any future relocation.'
)
config = load_config(args.config)
layout = config.get('global') or {}
if not args.in_place:
layout["project_dir"] = args.target_app or os.path.join(layout["root_dir"], "app")
dry_run = not args.apply
load_postgres_environment(os.path.abspath(args.config), config)
endpoint_dsn = database_url_from_env() or layout.get('database_url')
if not is_postgres_url(endpoint_dsn):
raise SystemExit('A caller-selected canonical PostgreSQL DSN is required for maintenance authority')
with ClusterAuthorityLock(config, endpoint_dsn=endpoint_dsn):
require_runtime_hardening_stopped(config)
create_layout(layout, dry_run)
if not args.no_app_copy:
copy_app_files(args.source_app, layout, args.overwrite, dry_run, verified_apply=args.apply)
copy_scanner_runtime(args.old_root, layout, args.overwrite, dry_run, verified_apply=args.apply)
if not args.no_desktop_import:
import_huggingface_desktop(layout, args.desktop_hf, args.overwrite, dry_run)
if dry_run:
print("Dry-run complete. --apply is retired; use reviewed offline backup/restore tooling for relocation.")
return 0
print("Migration copy/import finished. Originals were left in place.")
return 0
if __name__ == "__main__":
main()
+137
View File
@@ -0,0 +1,137 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
import sqlite3
from scanner_db import ScannerDB
TABLES = [
'runs',
'source_cycles',
'target_scans',
'target_queue',
'scan_publication_outbox',
'findings',
'errors',
'queue_snapshots',
'config_snapshots',
'package_repo_candidates',
'keycheck_results',
'keycheck_event_map',
'finding_uid_map',
]
def sqlite_connect_ro(path):
uri = 'file:' + os.path.abspath(path).replace('\\', '/') + '?mode=ro'
conn = sqlite3.connect(uri, uri=True)
conn.row_factory = sqlite3.Row
return conn
def sqlite_columns(conn, table):
return [row['name'] for row in conn.execute(f'PRAGMA table_info({table})').fetchall()]
def sqlite_count(conn, table):
return int(conn.execute(f'SELECT COUNT(*) AS count FROM {table}').fetchone()['count'])
def sqlite_row_estimate(conn, table, exact=False):
if exact:
return sqlite_count(conn, table)
try:
row = conn.execute('SELECT seq FROM sqlite_sequence WHERE name = ?', (table,)).fetchone()
if row and row['seq'] is not None:
return int(row['seq'])
except sqlite3.Error:
pass
return None
def pg_reset_identity(db, table):
db.conn.execute(f'''
SELECT setval(
pg_get_serial_sequence('{table}', 'id'),
COALESCE((SELECT MAX(id) FROM {table}), 1),
(SELECT MAX(id) IS NOT NULL FROM {table})
)
''')
def postgres_safe_value(value):
if isinstance(value, str) and '\x00' in value:
return value.replace('\x00', '\\u0000')
return value
def copy_table(source, target, table, batch_size, dry_run=False, exact_counts=False):
columns = sqlite_columns(source, table)
if not columns:
print(f'{table}: missing or empty schema in SQLite, skipped', flush=True)
return 0
total = sqlite_row_estimate(source, table, exact=exact_counts)
total_label = total if total is not None else 'unknown'
print(f'{table}: source_rows={total_label}' + ('' if exact_counts else ' estimated'), flush=True)
if dry_run or total == 0:
return int(total or 0)
column_sql = ', '.join(columns)
placeholders = ', '.join('?' for _ in columns)
conflict_sql = ' ON CONFLICT (id) DO NOTHING' if 'id' in columns else ''
insert_sql = f'INSERT INTO {table} ({column_sql}) VALUES ({placeholders}){conflict_sql}'
copied = 0
cursor = source.execute(f'SELECT {column_sql} FROM {table} ORDER BY id' if 'id' in columns else f'SELECT {column_sql} FROM {table}')
while True:
rows = cursor.fetchmany(batch_size)
if not rows:
break
values = [[postgres_safe_value(row[column]) for column in columns] for row in rows]
if target.conn.is_postgres:
with target.conn._conn.cursor() as pg_cursor:
with pg_cursor.copy(f'COPY {table} ({column_sql}) FROM STDIN') as copy:
for value_row in values:
copy.write_row(value_row)
else:
for value_row in values:
target.conn.execute(insert_sql, value_row)
copied += len(rows)
target.conn.commit()
print(f'{table}: copied={copied}/{total_label}', flush=True)
if 'id' in columns:
pg_reset_identity(target, table)
target.conn.commit()
return copied
def truncate_target(db):
table_sql = ', '.join(TABLES)
db.conn.execute(f'TRUNCATE TABLE {table_sql} RESTART IDENTITY CASCADE')
db.conn.commit()
def parse_args():
parser = argparse.ArgumentParser(description='Offline migrate scanner observability SQLite DB to PostgreSQL.')
parser.add_argument('--sqlite', required=True, help='Path to scanner_active.db or archived SQLite DB')
parser.add_argument('--db-url', default=os.getenv('SCANNER_DB_URL') or os.getenv('DATABASE_URL'), help='PostgreSQL DSN')
parser.add_argument('--batch-size', type=int, default=1000)
parser.add_argument('--exact-counts', action='store_true', help='Use exact COUNT(*) per table; slow on multi-GB SQLite files')
parser.add_argument('--truncate', action='store_true', help='Delete existing Postgres observability rows before import')
parser.add_argument('--dry-run', action='store_true')
return parser.parse_args()
def main():
raise SystemExit(
'migrate_observability_db.py is retired for PostgreSQL. Use migrate_runtime_safety.py '
'--apply --sources-stopped with the bound cluster; perform legacy data import only with a reviewed, cluster-bound tool.'
)
if __name__ == '__main__':
sys.exit(main())
File diff suppressed because it is too large Load Diff
+21
View File
@@ -0,0 +1,21 @@
"""Retired direct database mutation entrypoint.
Dashboard indexes are part of the authority-locked offline runtime-safety
migration. Keeping a second live DDL path would bypass cluster identity and
stopped-source checks.
"""
import sys
sys.dont_write_bytecode = True
def main():
raise SystemExit(
'optimize_dashboard_db is retired; run migrate_runtime_safety.py '
'offline under cluster authority'
)
if __name__ == '__main__':
main()
+1346
View File
File diff suppressed because it is too large Load Diff
+237
View File
@@ -0,0 +1,237 @@
import ntpath
import os
import re
from query_policy import validate_rejected_query_policy
APP_DIR = os.path.dirname(os.path.abspath(__file__))
CANONICAL_ROOT = os.path.dirname(APP_DIR)
DEFAULT_TRUFFLEHOG = r"C:\Tools\trufflehog.exe"
PLACEHOLDER_RE = re.compile(r"\{([A-Za-z_][A-Za-z0-9_]*)\}")
class PathResolutionError(ValueError):
pass
def _norm(path):
return os.path.normpath(str(path))
def _config_dir(config_path=None):
if config_path:
return os.path.dirname(os.path.abspath(config_path))
return APP_DIR
def _expand(value, context):
text = str(value)
missing = sorted({name for name in PLACEHOLDER_RE.findall(text) if name not in context})
if missing:
raise PathResolutionError(f"Unknown path placeholder(s): {', '.join(missing)} in {text!r}")
for name in PLACEHOLDER_RE.findall(text):
text = text.replace("{" + name + "}", str(context[name]))
return os.path.expandvars(os.path.expanduser(text))
def is_command_name(value):
text = str(value or "")
return bool(text) and not os.path.isabs(text) and "\\" not in text and "/" not in text
def is_database_url(value):
return str(value or "").strip().lower().startswith(("postgresql://", "postgres://"))
def resolve_path(value, context=None, base_dir=None, allow_command=False, required=False):
if value is None or str(value).strip() == "":
if required:
raise PathResolutionError("Required path is empty")
return value
context = context or {}
text = _expand(value, context)
if os.name != 'nt' and (ntpath.splitdrive(text)[0] or '\\' in text):
raise PathResolutionError(f"Windows path is not supported on this platform: {text!r}")
if allow_command and is_command_name(text):
return text
if os.path.isabs(text):
return _norm(text)
base = base_dir or context.get("project_dir") or context.get("config_dir") or os.getcwd()
if os.name != 'nt' and (ntpath.splitdrive(str(base))[0] or '\\' in str(base)):
raise PathResolutionError(f"Windows base path is not supported on this platform: {base!r}")
return _norm(os.path.join(base, text))
def default_trufflehog_path():
return DEFAULT_TRUFFLEHOG if os.name == 'nt' and os.path.exists(DEFAULT_TRUFFLEHOG) else "trufflehog"
def resolve_project_paths(global_config=None, config_path=None):
global_config = global_config or {}
context = {"config_dir": _config_dir(config_path)}
root_raw = (
global_config.get("root_dir")
or os.getenv("SCANNER_ROOT_DIR")
or os.getenv("SCANNER_PROJECT_ROOT")
or CANONICAL_ROOT
)
context["root_dir"] = resolve_path(root_raw, context, base_dir=context["config_dir"], required=True)
project_raw = global_config.get("project_dir") or os.getenv("SCANNER_PROJECT_DIR") or context["config_dir"]
context["project_dir"] = resolve_path(project_raw, context, base_dir=context["config_dir"], required=True)
ordered_defaults = [
("runtime_dir", os.getenv("SCANNER_RUNTIME_DIR") or os.path.join(context["root_dir"], "runtime")),
("result_bundle_dir", os.getenv("SCANNER_RESULT_BUNDLE_DIR") or "{runtime_dir}/result_bundles"),
("result_spool_dir", "{runtime_dir}/result_spool"),
("results_dir", os.getenv("SCAN_RESULTS_DIR") or "{runtime_dir}/results"),
("queue_dir", "{runtime_dir}/queues"),
("state_dir", "{runtime_dir}/state"),
("log_dir", "{runtime_dir}/logs"),
("control_dir", "{runtime_dir}/control"),
("keycheck_dir", "{runtime_dir}/keychecks"),
("postman_cache_dir", "{runtime_dir}/postman_cache"),
("gharchive_cache_dir", "{state_dir}/gharchive_cache"),
("work_dir", os.getenv("TRUFFLEHOG_WORK_DIR") or os.path.join(context["root_dir"], "tmp")),
("proxy_file", "{runtime_dir}/proxy.txt"),
("database_path", os.getenv("SCANNER_DB_PATH") or os.getenv("SCAN_DB_PATH") or "{results_dir}/scanner.db"),
("state_file", "{state_dir}/runner_state.json"),
("secrets_file", "{project_dir}/secrets.yaml"),
]
for key, default in ordered_defaults:
raw = global_config.get(key) or default
context[key] = resolve_path(raw, context, base_dir=context["project_dir"], required=True)
managed_database_url = os.getenv("TRUF_MANAGED_POSTGRES_DSN") or ""
database_url = managed_database_url or global_config.get("database_url") or os.getenv("SCANNER_DB_URL") or os.getenv("DATABASE_URL") or ""
context["database_url"] = _expand(database_url, context) if database_url else ""
dashboard_db_url = managed_database_url or global_config.get("dashboard_db_url") or os.getenv("SCANNER_DASHBOARD_DB_URL") or context["database_url"]
context["dashboard_db_url"] = _expand(dashboard_db_url, context) if dashboard_db_url else ""
trufflehog_raw = global_config.get("trufflehog_path") or os.getenv("TRUFFLEHOG_PATH") or default_trufflehog_path()
context["trufflehog_path"] = resolve_path(
trufflehog_raw,
context,
base_dir=context["project_dir"],
allow_command=True,
required=True,
)
return context
def default_project_paths():
return resolve_project_paths({}, None)
def resolve_optional_path(value, path_context, base_dir=None, allow_command=False):
if not value:
return value
return resolve_path(value, path_context, base_dir=base_dir or path_context.get("project_dir"), allow_command=allow_command)
def resolve_postgres_data_dir(global_config=None, runtime_dir=None, base_dir=None):
global_config = global_config or {}
runtime_dir = runtime_dir or global_config.get('runtime_dir')
if not runtime_dir:
root_dir = global_config.get('root_dir') or CANONICAL_ROOT
runtime_dir = os.path.join(root_dir, 'runtime')
context = dict(global_config)
context['runtime_dir'] = runtime_dir
raw = global_config.get('postgres_data_dir') or os.path.join(runtime_dir, 'postgres', 'data')
return resolve_path(
raw,
context,
base_dir=base_dir or global_config.get('project_dir') or global_config.get('root_dir'),
required=True,
)
def resolve_postgres_bin_dir(global_config=None, runtime_dir=None, base_dir=None):
global_config = global_config or {}
runtime_dir = runtime_dir or global_config.get('runtime_dir')
if not runtime_dir:
runtime_dir = os.path.join(global_config.get('root_dir') or CANONICAL_ROOT, 'runtime')
context = dict(global_config, runtime_dir=runtime_dir)
return resolve_path(
global_config.get('postgres_bin_dir') or os.path.join(runtime_dir, 'postgres', 'pgsql', 'bin'),
context,
base_dir=base_dir or global_config.get('project_dir') or global_config.get('root_dir'),
required=True,
)
def apply_path_config(config, config_path=None):
config = config or {}
validate_rejected_query_policy(config)
global_config = config.setdefault("global", {})
path_context = resolve_project_paths(global_config, config_path)
for key, value in path_context.items():
global_config[key] = value
if global_config.get('legacy_result_spool_dir'):
global_config['legacy_result_spool_dir'] = resolve_path(
global_config['legacy_result_spool_dir'],
path_context,
base_dir=path_context['project_dir'],
required=True,
)
if global_config.get('postgres_data_dir'):
global_config['postgres_data_dir'] = resolve_postgres_data_dir(
global_config,
path_context['runtime_dir'],
base_dir=path_context['project_dir'],
)
if global_config.get('postgres_bin_dir'):
global_config['postgres_bin_dir'] = resolve_postgres_bin_dir(
global_config,
path_context['runtime_dir'],
base_dir=path_context['project_dir'],
)
for key in (
'api_proxy_file', 'download_proxy_file', 'trufflehog_config',
'dashboard_db_path', 'scan_limiter_db', 'dockerhub_tag_cache_path',
):
if global_config.get(key):
global_config[key] = resolve_path(
global_config[key],
path_context,
base_dir=path_context['project_dir'],
required=True,
)
supervisor = config.setdefault("supervisor", {})
supervisor_defaults = {
"log_dir": "{log_dir}",
"control_dir": "{control_dir}",
"instance_file": "{control_dir}/supervisor.instance.json",
"lock_file": "{control_dir}/supervisor.lock",
"supervisor_log": "{log_dir}/supervisor.log",
"status_file": "{log_dir}/supervisor.status.txt",
"dashboard_log": "{log_dir}/dashboard.log",
"state_dir": "{state_dir}",
}
for key, default in supervisor_defaults.items():
supervisor[key] = resolve_path(
supervisor.get(key) or default,
path_context,
base_dir=path_context["project_dir"],
required=True,
)
return config
def ensure_directories(paths, keys):
for key in keys:
path = paths.get(key)
if path:
os.makedirs(path, exist_ok=True)
File diff suppressed because it is too large Load Diff
+449
View File
@@ -0,0 +1,449 @@
import ctypes
import os
import select
import signal
import time
from dataclasses import dataclass
from runtime_security import canonical_path
if os.name == 'nt':
from ctypes import wintypes
class _FILETIME(ctypes.Structure):
_fields_ = [('dwLowDateTime', wintypes.DWORD), ('dwHighDateTime', wintypes.DWORD)]
class _UNICODE_STRING(ctypes.Structure):
_fields_ = [
('Length', wintypes.USHORT),
('MaximumLength', wintypes.USHORT),
('Buffer', ctypes.c_void_p),
]
_P_DWORD = ctypes.POINTER(wintypes.DWORD)
_P_ULONG = ctypes.POINTER(wintypes.ULONG)
_P_BOOL = ctypes.POINTER(wintypes.BOOL)
_P_FILETIME = ctypes.POINTER(_FILETIME)
_P_UNICODE_STRING = ctypes.POINTER(_UNICODE_STRING)
_P_INT = ctypes.POINTER(ctypes.c_int)
_P_LPWSTR = ctypes.POINTER(wintypes.LPWSTR)
_KERNEL32 = ctypes.WinDLL('kernel32', use_last_error=True)
_NTDLL = ctypes.WinDLL('ntdll', use_last_error=True)
_SHELL32 = ctypes.WinDLL('shell32', use_last_error=True)
_GET_EXIT_CODE_PROCESS = _KERNEL32.GetExitCodeProcess
_GET_EXIT_CODE_PROCESS.argtypes = [wintypes.HANDLE, _P_DWORD]
_GET_EXIT_CODE_PROCESS.restype = wintypes.BOOL
_WAIT_FOR_SINGLE_OBJECT = _KERNEL32.WaitForSingleObject
_WAIT_FOR_SINGLE_OBJECT.argtypes = [wintypes.HANDLE, wintypes.DWORD]
_WAIT_FOR_SINGLE_OBJECT.restype = wintypes.DWORD
_CLOSE_HANDLE = _KERNEL32.CloseHandle
_CLOSE_HANDLE.argtypes = [wintypes.HANDLE]
_CLOSE_HANDLE.restype = wintypes.BOOL
_GET_PROCESS_TIMES = _KERNEL32.GetProcessTimes
_GET_PROCESS_TIMES.argtypes = [
wintypes.HANDLE, _P_FILETIME, _P_FILETIME, _P_FILETIME, _P_FILETIME,
]
_GET_PROCESS_TIMES.restype = wintypes.BOOL
_QUERY_FULL_PROCESS_IMAGE_NAME = _KERNEL32.QueryFullProcessImageNameW
_QUERY_FULL_PROCESS_IMAGE_NAME.argtypes = [
wintypes.HANDLE, wintypes.DWORD, wintypes.LPWSTR, _P_DWORD,
]
_QUERY_FULL_PROCESS_IMAGE_NAME.restype = wintypes.BOOL
_IS_PROCESS_IN_JOB = _KERNEL32.IsProcessInJob
_IS_PROCESS_IN_JOB.argtypes = [wintypes.HANDLE, wintypes.HANDLE, _P_BOOL]
_IS_PROCESS_IN_JOB.restype = wintypes.BOOL
_OPEN_PROCESS = _KERNEL32.OpenProcess
_OPEN_PROCESS.argtypes = [wintypes.DWORD, wintypes.BOOL, wintypes.DWORD]
_OPEN_PROCESS.restype = wintypes.HANDLE
_GET_CURRENT_PROCESS = _KERNEL32.GetCurrentProcess
_GET_CURRENT_PROCESS.argtypes = []
_GET_CURRENT_PROCESS.restype = wintypes.HANDLE
_TERMINATE_PROCESS = _KERNEL32.TerminateProcess
_TERMINATE_PROCESS.argtypes = [wintypes.HANDLE, wintypes.UINT]
_TERMINATE_PROCESS.restype = wintypes.BOOL
_LOCAL_FREE = _KERNEL32.LocalFree
_LOCAL_FREE.argtypes = [wintypes.HLOCAL]
_LOCAL_FREE.restype = wintypes.HLOCAL
_NT_QUERY_INFORMATION_PROCESS = _NTDLL.NtQueryInformationProcess
_NT_QUERY_INFORMATION_PROCESS.argtypes = [
wintypes.HANDLE, wintypes.ULONG, ctypes.c_void_p, wintypes.ULONG, _P_ULONG,
]
_NT_QUERY_INFORMATION_PROCESS.restype = ctypes.c_long
_COMMAND_LINE_TO_ARGV = _SHELL32.CommandLineToArgvW
_COMMAND_LINE_TO_ARGV.argtypes = [wintypes.LPCWSTR, _P_INT]
_COMMAND_LINE_TO_ARGV.restype = _P_LPWSTR
else:
_FILETIME = _UNICODE_STRING = None
_KERNEL32 = _NTDLL = _SHELL32 = None
class ProcessIdentityError(OSError):
pass
class ProcessExitedError(ProcessIdentityError):
pass
@dataclass(frozen=True)
class ProcessIdentity:
pid: int
creation_time: str
creation_time_unix: float
executable: str
in_job: object
def as_dict(self):
return {
'pid': int(self.pid),
'creation_time': str(self.creation_time),
'creation_time_unix': float(self.creation_time_unix),
'executable': str(self.executable),
'in_job': self.in_job,
}
class RetainedProcess:
def __init__(self, identity, handle=None, pidfd=None):
self.identity = identity
self._handle = handle
self._pidfd = pidfd
self._closed = False
@property
def pid(self):
return self.identity.pid
def is_running(self):
if self._closed:
return False
if os.name == 'nt':
result = _WAIT_FOR_SINGLE_OBJECT(self._handle, 0)
if result == 258:
return True
if result == 0:
return False
raise ctypes.WinError(ctypes.get_last_error())
try:
current = _posix_identity(self.pid)
return current.creation_time == self.identity.creation_time
except ProcessIdentityError:
return False
def wait(self, timeout):
timeout = max(0.0, float(timeout))
if os.name == 'nt':
milliseconds = min(int(timeout * 1000), 0xFFFFFFFE)
result = _WAIT_FOR_SINGLE_OBJECT(self._handle, milliseconds)
if result == 0:
return True
if result == 258:
return False
raise ctypes.WinError(ctypes.get_last_error())
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
if not self.is_running():
return True
time.sleep(min(0.05, max(0.0, deadline - time.monotonic())))
return not self.is_running()
def exit_code(self):
if self._closed:
raise ProcessIdentityError('retained process handle is closed')
if os.name != 'nt':
return None
code = wintypes.DWORD()
if not _GET_EXIT_CODE_PROCESS(self._handle, ctypes.byref(code)):
raise ProcessIdentityError(f'unable to read process exit status: {ctypes.WinError(ctypes.get_last_error())}')
if code.value == 259:
return None
return int(code.value)
def terminate(self):
if self._closed:
raise ProcessIdentityError('retained process handle is closed')
if os.name == 'nt':
if not _TERMINATE_PROCESS(self._handle, 1):
raise ctypes.WinError(ctypes.get_last_error())
return
sender = getattr(signal, 'pidfd_send_signal', None)
if self._pidfd is not None and sender is not None:
sender(self._pidfd, signal.SIGTERM, None, 0)
return
current = _posix_identity(self.pid)
if (
current.creation_time != self.identity.creation_time
or current.executable != self.identity.executable
):
raise ProcessIdentityError(f'process identity changed before signaling PID {self.pid}')
os.kill(self.pid, signal.SIGTERM)
def command_line(self):
if self._closed:
raise ProcessIdentityError('retained process handle is closed')
if os.name != 'nt':
try:
with open(f'/proc/{self.pid}/cmdline', 'rb') as handle:
return [item.decode(errors='surrogateescape') for item in handle.read().split(b'\0') if item]
except OSError as exc:
raise ProcessIdentityError(f'unable to read process {self.pid} command line') from exc
needed = wintypes.ULONG()
_NT_QUERY_INFORMATION_PROCESS(self._handle, 60, None, 0, ctypes.byref(needed))
if not needed.value:
raise ProcessIdentityError(f'unable to size process {self.pid} command line')
buffer = ctypes.create_string_buffer(needed.value)
status = _NT_QUERY_INFORMATION_PROCESS(
self._handle, 60, buffer, needed.value, ctypes.byref(needed),
)
if status < 0:
raise ProcessIdentityError(f'unable to read process {self.pid} command line (NTSTATUS 0x{status & 0xFFFFFFFF:08X})')
value = ctypes.cast(buffer, _P_UNICODE_STRING).contents
command = ctypes.wstring_at(value.Buffer, value.Length // ctypes.sizeof(ctypes.c_wchar))
argc = ctypes.c_int()
argv = _COMMAND_LINE_TO_ARGV(command, ctypes.byref(argc))
if not argv:
raise ProcessIdentityError(f'unable to parse process {self.pid} command line')
try:
return [argv[index] for index in range(argc.value)]
finally:
_LOCAL_FREE(argv)
def close(self):
if self._closed:
return
self._closed = True
if os.name == 'nt' and self._handle:
_CLOSE_HANDLE(self._handle)
elif self._pidfd is not None:
try:
os.close(self._pidfd)
except OSError:
pass
def __enter__(self):
return self
def __exit__(self, exc_type, value, traceback):
self.close()
def __del__(self):
try:
self.close()
except BaseException:
pass
def _windows_identity(handle, pid):
creation = _FILETIME()
ignored_exit = _FILETIME()
ignored_kernel = _FILETIME()
ignored_user = _FILETIME()
if not _GET_PROCESS_TIMES(
handle, ctypes.byref(creation), ctypes.byref(ignored_exit),
ctypes.byref(ignored_kernel), ctypes.byref(ignored_user),
):
raise ctypes.WinError(ctypes.get_last_error())
filetime = (int(creation.dwHighDateTime) << 32) | int(creation.dwLowDateTime)
path_buffer = ctypes.create_unicode_buffer(32768)
path_size = wintypes.DWORD(len(path_buffer))
if not _QUERY_FULL_PROCESS_IMAGE_NAME(handle, 0, path_buffer, ctypes.byref(path_size)):
raise ctypes.WinError(ctypes.get_last_error())
in_job = wintypes.BOOL()
if not _IS_PROCESS_IN_JOB(handle, None, ctypes.byref(in_job)):
raise ctypes.WinError(ctypes.get_last_error())
unix_time = (filetime - 116444736000000000) / 10000000.0
return ProcessIdentity(
pid=int(pid),
creation_time=f'windows-filetime:{filetime}',
creation_time_unix=unix_time,
executable=canonical_path(path_buffer.value),
in_job=bool(in_job.value),
)
def _posix_identity(pid):
stat_path = f'/proc/{int(pid)}/stat'
try:
with open(stat_path, 'r', encoding='ascii') as handle:
value = handle.read()
close_paren = value.rfind(')')
fields = value[close_paren + 2:].split()
start_ticks = int(fields[19])
executable = canonical_path(os.readlink(f'/proc/{int(pid)}/exe'))
clock_ticks = int(os.sysconf('SC_CLK_TCK'))
boot_time = None
with open('/proc/stat', 'r', encoding='ascii') as handle:
for line in handle:
if line.startswith('btime '):
boot_time = float(line.split()[1])
break
if boot_time is None:
raise ValueError('boot time unavailable')
except (OSError, ValueError, IndexError) as exc:
raise ProcessIdentityError(f'unable to inspect process {pid}') from exc
return ProcessIdentity(
pid=int(pid),
creation_time=f'proc-start-ticks:{start_ticks}',
creation_time_unix=boot_time + (start_ticks / float(clock_ticks)),
executable=executable,
in_job=False,
)
def _pidfd_live(pidfd):
poller = select.poll()
poller.register(pidfd, select.POLLIN)
return not bool(poller.poll(0))
def open_process(pid, *, terminate=False):
pid = int(pid)
if pid <= 0:
raise ProcessIdentityError(f'invalid process ID: {pid}')
if os.name == 'nt':
rights = 0x00100000 | 0x00001000
if terminate:
rights |= 0x00000001
handle = _OPEN_PROCESS(rights, False, pid)
if not handle:
native_error = ctypes.WinError(ctypes.get_last_error())
raise ProcessIdentityError(f'unable to open process {pid}: {native_error}') from native_error
try:
wait_result = _WAIT_FOR_SINGLE_OBJECT(handle, 0)
if wait_result == 0:
exit_code = wintypes.DWORD()
code = int(exit_code.value) if _GET_EXIT_CODE_PROCESS(
handle, ctypes.byref(exit_code),
) else -1
raise ProcessExitedError(
f'process {pid} has already exited with code {code}'
)
if wait_result != 258:
raise ProcessIdentityError(
f'unable to wait on process {pid}: {ctypes.WinError(ctypes.get_last_error())}'
)
exit_code = wintypes.DWORD()
if not _GET_EXIT_CODE_PROCESS(handle, ctypes.byref(exit_code)):
native_error = ctypes.WinError(ctypes.get_last_error())
raise ProcessIdentityError(
f'unable to read process {pid} exit status: {native_error}'
) from native_error
try:
identity = _windows_identity(handle, pid)
except OSError as exc:
retry_exit_code = wintypes.DWORD()
if (
_GET_EXIT_CODE_PROCESS(handle, ctypes.byref(retry_exit_code))
and retry_exit_code.value != 259
):
raise ProcessExitedError(
f'process {pid} exited during identity inspection '
f'with code {int(retry_exit_code.value)}'
) from exc
raise ProcessIdentityError(f'unable to inspect process {pid}') from exc
final_wait = _WAIT_FOR_SINGLE_OBJECT(handle, 0)
if final_wait == 0:
raise ProcessExitedError(
f'process {pid} exited during identity inspection'
)
if final_wait != 258:
raise ProcessIdentityError(
f'unable to confirm process {pid} liveness: '
f'{ctypes.WinError(ctypes.get_last_error())}'
)
return RetainedProcess(identity, handle=handle)
except BaseException:
_CLOSE_HANDLE(handle)
raise
pidfd = None
if hasattr(os, 'pidfd_open'):
try:
pidfd = os.pidfd_open(pid, 0)
except ProcessLookupError as exc:
raise ProcessExitedError(f'process {pid} has already exited') from exc
except OSError as exc:
raise ProcessIdentityError(
f'unable to pin process {pid} with pidfd',
) from exc
try:
if pidfd is not None and not _pidfd_live(pidfd):
raise ProcessExitedError(f'process {pid} exited before identity binding')
identity = _posix_identity(pid)
if pidfd is not None and not _pidfd_live(pidfd):
raise ProcessExitedError(f'process {pid} exited during identity binding')
verified = _posix_identity(pid)
if (
verified.creation_time != identity.creation_time
or verified.executable != identity.executable
):
raise ProcessIdentityError(
f'process {pid} identity changed during pidfd binding',
)
if pidfd is not None and not _pidfd_live(pidfd):
raise ProcessExitedError(f'process {pid} exited after identity binding')
return RetainedProcess(identity, pidfd=pidfd)
except BaseException:
if pidfd is not None:
os.close(pidfd)
raise
def current_process_identity():
if os.name == 'nt':
return _windows_identity(_GET_CURRENT_PROCESS(), os.getpid())
return _posix_identity(os.getpid())
def verify_retained_process(pid, creation_time, executable, *, terminate=False):
process = open_process(pid, terminate=True) if terminate else open_process(pid)
expected_executable = canonical_path(executable)
if process.identity.creation_time != str(creation_time) or process.identity.executable != expected_executable:
process.close()
raise ProcessIdentityError(f'process identity mismatch for PID {pid}')
return process
def serialize_process_identity(identity):
if isinstance(identity, ProcessIdentity):
return identity.as_dict()
raise TypeError('expected ProcessIdentity')
def exact_process_identity_state(pid, creation_time, executable):
"""Return alive, dead, reused, or unknown without PID-only inference."""
try:
pid = int(pid)
except (TypeError, ValueError):
return 'unknown'
if pid <= 0 or not creation_time or not executable:
return 'unknown'
try:
process = open_process(pid)
except ProcessExitedError:
return 'dead'
except ProcessIdentityError as exc:
cause = exc.__cause__
winerror = getattr(cause, 'winerror', None) or getattr(exc, 'winerror', None)
errno_value = getattr(cause, 'errno', None) or getattr(exc, 'errno', None)
if os.name == 'nt' and winerror in (87, 1168):
return 'dead'
if os.name != 'nt' and errno_value in (2, 3):
return 'dead'
return 'unknown'
try:
if not process.is_running():
return 'dead'
if (
str(process.identity.creation_time) != str(creation_time)
or canonical_path(process.identity.executable) != canonical_path(executable)
):
return 'reused'
return 'alive'
except (OSError, ValueError):
return 'unknown'
finally:
process.close()
+126
View File
@@ -0,0 +1,126 @@
from datetime import datetime
import re
REJECTED_QUERY_STATUS = 'rejected_zero_alive'
REJECTED_QUERY_KEYS = {
'source', 'query', 'status', 'evidence_cutoff', 'successful_scans',
'findings', 'unique_credentials', 'pending_candidates',
'ever_alive_credentials', 'reviewed_queue_rows',
}
REJECTED_QUERY_COUNT_KEYS = {
'successful_scans', 'findings', 'unique_credentials',
'pending_candidates', 'ever_alive_credentials', 'reviewed_queue_rows',
}
OPERATIONAL_QUERY_SENTINELS = {
('github_archive', 'gharchive'),
('github_archive_files', 'gharchive-files'),
('github_gists', 'gists'),
('github_actions', 'logs'),
('gitlab_ci', 'logs'),
('huggingface', 'spaces'),
}
SOURCE_RE = re.compile(r'[a-z][a-z0-9_]{0,63}')
MAX_QUERY_LENGTH = 512
MAX_EVIDENCE_COUNT = (1 << 63) - 1
class QueryPolicyError(ValueError):
pass
def _active_queries(source, source_config):
raw_queries = source_config.get('queries', [])
if isinstance(raw_queries, str):
raw_queries = raw_queries.split(',')
if not isinstance(raw_queries, (list, tuple)):
raise QueryPolicyError(f'active query policy is invalid for source {source}')
queries = set()
for raw_query in raw_queries:
query = str(raw_query or '').strip()
if not query:
raise QueryPolicyError(f'active query policy contains an empty query for source {source}')
queries.add(query)
return queries
def validate_rejected_query_policy(config):
config = config or {}
raw_policy = config.get('query_policy')
if raw_policy is None:
return ()
if not isinstance(raw_policy, dict) or set(raw_policy) != {'rejected'}:
raise QueryPolicyError('query_policy must contain only the rejected registry')
raw_entries = raw_policy.get('rejected')
if not isinstance(raw_entries, list):
raise QueryPolicyError('query_policy.rejected must be a list')
sources = config.get('sources') or {}
if not isinstance(sources, dict):
raise QueryPolicyError('configured sources must be a mapping')
normalized = []
seen = set()
for raw_entry in raw_entries:
if not isinstance(raw_entry, dict) or set(raw_entry) != REJECTED_QUERY_KEYS:
raise QueryPolicyError('rejected query evidence shape is invalid')
source = raw_entry.get('source')
query = raw_entry.get('query')
if not isinstance(source, str) or not SOURCE_RE.fullmatch(source):
raise QueryPolicyError('rejected query source is invalid')
if source not in sources or not isinstance(sources[source], dict):
raise QueryPolicyError(f'rejected query source is not configured: {source}')
if (
not isinstance(query, str)
or query != query.strip()
or not query
or len(query) > MAX_QUERY_LENGTH
):
raise QueryPolicyError(f'rejected query text is invalid for source {source}')
pair = (source, query)
if pair in seen:
raise QueryPolicyError('rejected query registry contains a duplicate pair')
if pair in OPERATIONAL_QUERY_SENTINELS:
raise QueryPolicyError('operational query sentinel cannot be rejected')
if query in _active_queries(source, sources[source]):
raise QueryPolicyError('active and rejected query policy overlap')
if raw_entry.get('status') != REJECTED_QUERY_STATUS:
raise QueryPolicyError('rejected query status is invalid')
cutoff = raw_entry.get('evidence_cutoff')
if not isinstance(cutoff, str) or not cutoff or cutoff != cutoff.strip():
raise QueryPolicyError('rejected query evidence cutoff is invalid')
try:
parsed_cutoff = datetime.fromisoformat(cutoff.replace('Z', '+00:00'))
except ValueError as exc:
raise QueryPolicyError('rejected query evidence cutoff is invalid') from exc
if parsed_cutoff.tzinfo is None or parsed_cutoff.utcoffset() is None:
raise QueryPolicyError('rejected query evidence cutoff must include a timezone')
counts = {}
for name in REJECTED_QUERY_COUNT_KEYS:
value = raw_entry.get(name)
if (
isinstance(value, bool)
or not isinstance(value, int)
or value < 0
or value > MAX_EVIDENCE_COUNT
):
raise QueryPolicyError(f'rejected query {name} is invalid')
counts[name] = value
if counts['successful_scans'] < 1000:
raise QueryPolicyError('rejected query has fewer than 1000 successful scans')
if counts['pending_candidates'] != 0:
raise QueryPolicyError('rejected query still has pending candidates')
if counts['ever_alive_credentials'] != 0:
raise QueryPolicyError('rejected query has historical alive credentials')
seen.add(pair)
normalized.append({
'source': source,
'query': query,
'status': REJECTED_QUERY_STATUS,
'evidence_cutoff': cutoff,
**{name: counts[name] for name in sorted(REJECTED_QUERY_COUNT_KEYS)},
})
return tuple(sorted(normalized, key=lambda entry: (entry['source'], entry['query'])))
+162
View File
@@ -0,0 +1,162 @@
"""Stdlib-only integrity boundary for packaged remote worker clients."""
import hashlib
import json
import os
import runpy
import stat
import sys
sys.dont_write_bytecode = True
if not sys.dont_write_bytecode:
raise RuntimeError('worker bootstrap could not disable bytecode writes')
MAX_MANIFEST_BYTES = 1024 * 1024
MAX_MANIFEST_FILES = 512
WORKER_PACKAGE_SCHEMA = 3
WORKER_PROTOCOL_VERSION = 2
def _canonical(path):
return os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(path))))
def _is_reparse_point(path):
details = os.lstat(path)
if stat.S_ISLNK(details.st_mode):
return True
attributes = getattr(details, 'st_file_attributes', 0)
reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0)
return bool(attributes & reparse_attribute) or getattr(
os.path, 'isjunction', lambda _path: False,
)(path)
def _relative(value, label):
value = str(value or '')
if (
not value or len(value) > 512 or '\\' in value or '\x00' in value
or value.startswith('/') or value.endswith('/')
):
raise RuntimeError(f'invalid worker package {label} path')
if any(part in ('', '.', '..') for part in value.split('/')):
raise RuntimeError(f'invalid worker package {label} path')
return value
def _sha256(path):
digest = hashlib.sha256()
with open(path, 'rb', buffering=0) as handle:
for block in iter(lambda: handle.read(1024 * 1024), b''):
digest.update(block)
return digest.hexdigest()
def _load_manifest(path):
details = os.stat(path, follow_symlinks=False)
if _is_reparse_point(path) or not stat.S_ISREG(details.st_mode):
raise RuntimeError('worker package manifest is not a regular file')
with open(path, 'rb') as handle:
payload = handle.read(MAX_MANIFEST_BYTES + 1)
if len(payload) > MAX_MANIFEST_BYTES:
raise RuntimeError('worker package manifest exceeds its byte bound')
value = json.loads(payload.decode('utf-8', errors='strict'))
if (
not isinstance(value, dict)
or type(value.get('schema')) is not int
or value['schema'] != WORKER_PACKAGE_SCHEMA
or type(value.get('protocol_version')) is not int
or value['protocol_version'] != WORKER_PROTOCOL_VERSION
):
raise RuntimeError('worker package manifest is invalid')
return value
def _application_files(app_dir):
files = set()
def raise_walk_error(exc):
raise RuntimeError(f'unable to inspect worker application root: {exc}') from exc
for current, directories, names in os.walk(
app_dir, followlinks=False, onerror=raise_walk_error,
):
for name in directories:
candidate = os.path.join(current, name)
if _is_reparse_point(candidate) or name.lower() == '__pycache__':
raise RuntimeError('worker application directory is unsupported')
for name in names:
candidate = os.path.join(current, name)
details = os.stat(candidate, follow_symlinks=False)
if _is_reparse_point(candidate) or not stat.S_ISREG(details.st_mode):
raise RuntimeError('worker application file is not regular')
files.add(os.path.relpath(candidate, app_dir).replace(os.sep, '/'))
return files
def _verify_application(package_root, manifest):
app_root = _relative(manifest.get('app_root'), 'application root')
app_dir = _canonical(os.path.join(package_root, *app_root.split('/')))
if app_dir != _canonical(os.path.dirname(__file__)) or _is_reparse_point(app_dir):
raise RuntimeError('worker package application root is not canonical')
values = manifest.get('files')
if not isinstance(values, dict) or not 1 <= len(values) <= MAX_MANIFEST_FILES:
raise RuntimeError('worker package file set is invalid')
expected = set()
for name, entry in values.items():
name = _relative(name, 'file name')
if not isinstance(entry, dict) or set(entry) != {'path', 'sha256'}:
raise RuntimeError('worker package file entry is invalid')
relative = _relative(entry.get('path'), f'file {name}')
if relative != f'{app_root}/{name}':
raise RuntimeError('worker package file path is not canonical')
digest = str(entry.get('sha256') or '')
if len(digest) != 64 or any(ch not in '0123456789abcdef' for ch in digest):
raise RuntimeError('worker package file digest is invalid')
path = _canonical(os.path.join(package_root, *relative.split('/')))
try:
contained = os.path.commonpath((app_dir, path)) == app_dir
except ValueError:
contained = False
if not contained or _is_reparse_point(path) or _sha256(path) != digest:
raise RuntimeError('worker package application integrity check failed')
expected.add(name)
if _application_files(app_dir) != expected:
raise RuntimeError('worker package application file set drifted')
return app_dir
def main():
if not (
sys.flags.isolated and sys.flags.no_site and sys.flags.dont_write_bytecode
):
raise RuntimeError(
'remote worker bootstrap requires isolated no-site bytecode-free startup (-I -S -B)'
)
sys.dont_write_bytecode = True
if len(sys.argv) < 2 or sys.argv[1] != '--':
raise RuntimeError('usage: remote_worker_bootstrap.py -- <worker args>')
package_root = _canonical(os.path.dirname(os.path.dirname(__file__)))
manifest = _load_manifest(os.path.join(package_root, 'worker-package.json'))
app_dir = _verify_application(package_root, manifest)
dependency_dir = os.path.join(app_dir, 'dependencies')
if (
not os.path.isdir(dependency_dir) or _is_reparse_point(dependency_dir)
or _canonical(dependency_dir) == app_dir
):
raise RuntimeError('package-local worker dependencies are unavailable')
entrypoint = os.path.join(app_dir, 'worker_cli.py')
sys.path.insert(0, dependency_dir)
sys.path.insert(0, app_dir)
sys.argv = [entrypoint, *sys.argv[2:]]
runpy.run_path(entrypoint, run_name='__main__')
if __name__ == '__main__':
try:
main()
except Exception as exc:
raise SystemExit('remote worker bootstrap rejected launch') from exc
File diff suppressed because it is too large Load Diff
+4
View File
@@ -0,0 +1,4 @@
requests
boto3
botocore
psycopg[binary]>=3.2
+10
View File
@@ -0,0 +1,10 @@
streamlit
starlette>=0.47.3,<1
uvicorn>=0.53,<1
python-multipart>=0.0.10
pandas
requests
plotly
PyYAML
psycopg[binary]>=3.2
zstandard==0.23.0
+757
View File
@@ -0,0 +1,757 @@
import hashlib
import json
import os
import re
import struct
from dataclasses import dataclass
from runtime_security import (
PrivatePathState,
durable_publish,
ensure_private_directory,
harden_private_file,
inspect_private_relative_path,
private_file_ready,
reject_reparse_components,
require_private_directory,
)
from worker_contracts import (
AssignmentOutcome,
MAX_DIAGNOSTIC_AGGREGATE_BYTES,
MAX_DIAGNOSTICS_PER_ASSIGNMENT,
decode_diagnostic_envelope,
build_legacy_error_frame_diagnostics,
encode_diagnostic_envelope,
)
MAGIC = b'TRUF-RB2\n'
FORMAT_VERSION = 2
FRAME_HEADER = struct.Struct('!cI')
FRAME_TYPES = frozenset((b'H', b'F', b'E', b'D', b'K', b'M', b'C'))
FRAME_ORDER = {name: index for index, name in enumerate((b'H', b'F', b'E', b'D', b'K', b'M', b'C'))}
ID_RE = re.compile(r'^[a-f0-9]{32,64}$')
DEFAULT_MAX_EVENT_BYTES = 64 * 1024 * 1024
DEFAULT_MAX_FRAME_BYTES = 16 * 1024 * 1024
FRAME_BOUNDS = {
b'H': 1024 * 1024,
b'F': 16 * 1024 * 1024,
b'E': 1024 * 1024,
b'D': 64 * 1024,
b'K': 2 * 1024 * 1024,
b'M': 16 * 1024 * 1024,
b'C': 1024 * 1024,
}
MAX_FINDING_FRAMES = 20000
MAX_ERROR_FRAMES = 2000
MAX_CANDIDATE_FRAMES = 2000
MAX_TOTAL_FRAMES = 24003 + MAX_DIAGNOSTICS_PER_ASSIGNMENT
class ResultBundleError(ValueError):
pass
class ResultBundleConflictError(ResultBundleError):
pass
class ResultBundleUnavailableError(OSError):
pass
def canonical_json_bytes(value):
try:
return json.dumps(
value, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
).encode('utf-8')
except (TypeError, ValueError) as exc:
raise ResultBundleError('bundle frame is not canonical JSON data') from exc
def _validated_id(value, name):
text = str(value or '').lower()
if not ID_RE.fullmatch(text):
raise ResultBundleError(f'invalid {name}')
return text
def _relative_ready_path(bundle_id):
bundle_id = _validated_id(bundle_id, 'bundle_id')
return os.path.join('ready', bundle_id[:2], f'{bundle_id}.trb')
def bundle_ready_path(root, bundle_id):
return os.path.join(os.path.abspath(root), _relative_ready_path(bundle_id))
def bundle_partial_relative_path(bundle_id, reservation_token):
bundle_id = _validated_id(bundle_id, 'bundle_id')
producer_token = hashlib.sha256(str(reservation_token).encode('utf-8')).hexdigest()[:24]
return os.path.join('tmp', bundle_id[:2], f'{bundle_id}.{producer_token}.partial')
def bundle_partial_path(root, bundle_id, reservation_token):
return os.path.join(
os.path.abspath(root), bundle_partial_relative_path(bundle_id, reservation_token),
)
def ensure_bundle_reservation_paths(root, reservation):
root = require_private_directory(os.path.abspath(root), create=False)
reservation = (
reservation if isinstance(reservation, BundleReservation)
else BundleReservation.from_mapping(reservation)
)
for name in ('tmp', 'ready', 'quarantine'):
ensure_private_directory(
os.path.join(root, name, reservation.bundle_id[:2]),
reject_reparse=True,
)
return reservation
@dataclass(frozen=True)
class BundleReservation:
reservation_id: int
reservation_token: str
bundle_id: str
scan_event_id: str
queue_id: int
claim_lease_token: str
declared_bytes: int
ready_path: str
source: str = ''
platform: str = ''
query: str = ''
target: str = ''
normalized_target: str = ''
run_id: int | None = None
cycle_id: int | None = None
producer_instance_id: str = ''
producer_pid: int = 0
producer_creation_time: str = ''
producer_executable: str = ''
@classmethod
def from_mapping(cls, value):
data = dict(value or {})
ready_path = data.get('ready_path') or data.get('ready_relative_path') or ''
return cls(
reservation_id=int(data['reservation_id'] if 'reservation_id' in data else data['id']),
reservation_token=str(data['reservation_token']),
bundle_id=_validated_id(data['bundle_id'], 'bundle_id'),
scan_event_id=_validated_id(data['scan_event_id'], 'scan_event_id'),
queue_id=int(data['queue_id']),
claim_lease_token=str(data.get('claim_lease_token') or data.get('claim_lease_token_value') or ''),
declared_bytes=int(data.get('declared_bytes') or data.get('declared_bundle_bytes') or 0),
ready_path=str(ready_path),
source=str(data.get('source') or ''),
platform=str(data.get('platform') or ''),
query=str(data.get('query') or ''),
target=str(data.get('target') or ''),
normalized_target=str(data.get('normalized_target') or ''),
run_id=data.get('run_id'),
cycle_id=data.get('cycle_id'),
producer_instance_id=str(data.get('producer_instance_id') or ''),
producer_pid=int(data.get('producer_pid') or 0),
producer_creation_time=str(data.get('producer_creation_time') or ''),
producer_executable=str(data.get('producer_executable') or ''),
)
def header(self):
return {
'format_version': FORMAT_VERSION,
'reservation_id': self.reservation_id,
'reservation_token': self.reservation_token,
'bundle_id': self.bundle_id,
'scan_event_id': self.scan_event_id,
'queue_id': self.queue_id,
'claim_lease_token': self.claim_lease_token,
'declared_bytes': self.declared_bytes,
'ready_relative_path': self.ready_path.replace('\\', '/'),
'source': self.source,
'platform': self.platform,
'query': self.query,
'target': self.target,
'normalized_target': self.normalized_target,
'run_id': self.run_id,
'cycle_id': self.cycle_id,
'producer_instance_id': self.producer_instance_id,
'producer_pid': self.producer_pid,
'producer_creation_time': self.producer_creation_time,
'producer_executable': self.producer_executable,
}
@dataclass(frozen=True)
class BundleCommit:
reservation_id: int
bundle_id: str
scan_event_id: str
scan_event_hash: str
relative_path: str
actual_bytes: int
frame_count: int
finding_count: int
error_count: int
candidate_count: int
def as_dict(self):
return dict(self.__dict__)
@dataclass(frozen=True)
class BundleMetadata(BundleCommit):
header: dict
result_metadata: dict
diagnostic_count: int
diagnostic_bytes: int
class ResultBundleWriter:
def __init__(self, root, reservation, handle, partial_path, ready_path, fault=None):
self.root = root
self.reservation = reservation
self.handle = handle
self.partial_path = partial_path
self.ready_path = ready_path
self.fault = fault
self.digest = hashlib.sha256()
self.bytes_written = 0
self.frame_count = 0
self.finding_count = 0
self.error_count = 0
self.candidate_count = 0
self.diagnostic_count = 0
self.diagnostic_bytes = 0
self.diagnostic_aggregate_bytes = 0
self._diagnostic_uids = set()
self._last_frame_order = FRAME_ORDER[b'H']
self.finished = False
self.ready_published = False
self._write_bytes(MAGIC)
self._write_frame(b'H', reservation.header())
@classmethod
def open(cls, root, reservation, fault=None, require_s_drive=False):
root = require_private_directory(os.path.abspath(root), create=False)
drive = os.path.splitdrive(root)[0].upper()
if require_s_drive and drive != 'S:':
raise ResultBundleError('production result bundle root must be on S:')
reservation = ensure_bundle_reservation_paths(root, reservation)
if reservation.declared_bytes <= len(MAGIC) or reservation.declared_bytes > DEFAULT_MAX_EVENT_BYTES:
raise ResultBundleError('declared bundle byte bound is invalid')
expected_relative = _relative_ready_path(reservation.bundle_id)
supplied_relative = str(reservation.ready_path or expected_relative).replace('/', os.sep)
if os.path.normcase(os.path.normpath(supplied_relative)) != os.path.normcase(os.path.normpath(expected_relative)):
raise ResultBundleError('reservation ready path is not deterministic for its bundle ID')
tmp_dir = os.path.join(root, 'tmp', reservation.bundle_id[:2])
ready_dir = os.path.join(root, 'ready', reservation.bundle_id[:2])
partial_path = bundle_partial_path(
root, reservation.bundle_id, reservation.reservation_token,
)
ready_path = os.path.join(ready_dir, f'{reservation.bundle_id}.trb')
if os.path.lexists(ready_path):
raise ResultBundleConflictError('deterministic ready bundle path already exists')
descriptor = os.open(
partial_path,
os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0),
0o600,
)
try:
os.close(descriptor)
descriptor = None
harden_private_file(partial_path)
handle = open(partial_path, 'w+b', buffering=0)
return cls(root, reservation, handle, partial_path, ready_path, fault=fault)
except BaseException:
if descriptor is not None:
os.close(descriptor)
try:
os.remove(partial_path)
except OSError:
pass
raise
def _inject(self, stage):
if self.fault is not None:
self.fault(stage, self)
def _write_bytes(self, payload):
if self.bytes_written + len(payload) > self.reservation.declared_bytes:
raise ResultBundleError('bundle exceeded its pre-reserved byte bound')
self.handle.write(payload)
self.digest.update(payload)
self.bytes_written += len(payload)
def _write_frame(self, frame_type, value):
if self.finished or frame_type not in FRAME_TYPES or frame_type == b'C':
raise ResultBundleError('invalid bundle frame write')
if self.frame_count >= MAX_TOTAL_FRAMES - 1:
raise ResultBundleError('bundle frame count exceeds its bound')
if FRAME_ORDER[frame_type] < self._last_frame_order:
raise ResultBundleError('bundle frame order is not deterministic')
payload = canonical_json_bytes(value)
bound = min(FRAME_BOUNDS[frame_type], self.reservation.declared_bytes)
if len(payload) > bound:
raise ResultBundleError(f'{frame_type.decode()} frame exceeds its byte bound')
framed = FRAME_HEADER.pack(frame_type, len(payload)) + payload
self._inject(f'before_frame_{frame_type.decode()}')
self._write_bytes(framed)
self.frame_count += 1
self._last_frame_order = FRAME_ORDER[frame_type]
self._inject(f'after_frame_{frame_type.decode()}')
return len(payload)
def write_finding(self, finding):
if self.finding_count >= MAX_FINDING_FRAMES:
raise ResultBundleError('bundle finding count exceeds its bound')
self._write_frame(b'F', finding)
self.finding_count += 1
def write_error(self, error):
if self.error_count >= MAX_ERROR_FRAMES:
raise ResultBundleError('bundle error count exceeds its bound')
self._write_frame(b'E', {'error': str(error)})
self.error_count += 1
def write_diagnostic(self, diagnostic):
if self.diagnostic_count >= MAX_DIAGNOSTICS_PER_ASSIGNMENT:
raise ResultBundleError('bundle diagnostic count exceeds its bound')
try:
if isinstance(diagnostic, dict):
envelope = decode_diagnostic_envelope(canonical_json_bytes(diagnostic))
else:
envelope = decode_diagnostic_envelope(
encode_diagnostic_envelope(diagnostic)
)
payload = encode_diagnostic_envelope(envelope)
value = json.loads(payload.decode('ascii'))
except (TypeError, ValueError, UnicodeError) as exc:
raise ResultBundleError('bundle diagnostic frame is invalid') from exc
if envelope.diagnostic_uid in self._diagnostic_uids:
raise ResultBundleError('bundle diagnostic identity is duplicated')
if (
envelope.reservation_id != self.reservation.reservation_id
or envelope.scan_event_id != self.reservation.scan_event_id
or envelope.source != self.reservation.source
):
raise ResultBundleError('bundle diagnostic identity conflicts with its reservation')
if (
self.diagnostic_aggregate_bytes + len(payload) + 1
> MAX_DIAGNOSTIC_AGGREGATE_BYTES
):
raise ResultBundleError('bundle diagnostic aggregate exceeds its byte bound')
self._write_frame(b'D', value)
self._diagnostic_uids.add(envelope.diagnostic_uid)
self.diagnostic_count += 1
self.diagnostic_bytes += FRAME_HEADER.size + len(payload)
self.diagnostic_aggregate_bytes += len(payload) + 1
def write_candidate(self, candidate):
if self.candidate_count >= MAX_CANDIDATE_FRAMES:
raise ResultBundleError('bundle candidate count exceeds its bound')
self._write_frame(b'K', candidate)
self.candidate_count += 1
def finish(self, metadata):
if self.finished:
raise ResultBundleError('bundle writer is already finished')
self._write_frame(b'M', metadata)
content_hash = self.digest.hexdigest()
footer = {
'format_version': FORMAT_VERSION,
'content_sha256': content_hash,
'content_bytes': self.bytes_written,
'byte_count': 0,
'frame_count': self.frame_count + 1,
'finding_count': self.finding_count,
'error_count': self.error_count,
'candidate_count': self.candidate_count,
}
if self.diagnostic_count:
footer.update({
'diagnostic_count': self.diagnostic_count,
'diagnostic_bytes': self.diagnostic_bytes,
})
while True:
payload = canonical_json_bytes(footer)
framed = FRAME_HEADER.pack(b'C', len(payload)) + payload
total = self.bytes_written + len(framed)
if footer['byte_count'] == total:
break
footer['byte_count'] = total
if total > self.reservation.declared_bytes:
raise ResultBundleError('bundle footer exceeds its pre-reserved byte bound')
self._inject('before_footer')
self.handle.write(framed)
self.bytes_written = total
self.frame_count += 1
self._inject('after_footer')
self._inject('before_fsync')
self.handle.flush()
os.fsync(self.handle.fileno())
self._inject('after_fsync')
self.handle.close()
self.handle = None
if not private_file_ready(self.partial_path):
raise ResultBundleError('private bundle ACL verification failed before publication')
if os.path.getsize(self.partial_path) != self.bytes_written:
raise ResultBundleError('bundle size changed before publication')
self._inject('before_rename')
durable_publish(self.partial_path, self.ready_path)
self.ready_published = True
self._inject('after_rename')
if not private_file_ready(self.ready_path):
inspection = inspect_private_relative_path(
self.root, os.path.relpath(self.ready_path, self.root),
)
if inspection.state == PrivatePathState.UNKNOWN:
raise ResultBundleUnavailableError(
'ready bundle state is unavailable after publication'
)
if inspection.state == PrivatePathState.PRESENT:
raise ResultBundleError('ready bundle ACL verification failed')
self.finished = True
relative = os.path.relpath(self.ready_path, self.root)
return BundleCommit(
reservation_id=self.reservation.reservation_id,
bundle_id=self.reservation.bundle_id,
scan_event_id=self.reservation.scan_event_id,
scan_event_hash=content_hash,
relative_path=relative.replace(os.sep, '/'),
actual_bytes=self.bytes_written,
frame_count=self.frame_count,
finding_count=self.finding_count,
error_count=self.error_count,
candidate_count=self.candidate_count,
)
def abort(self):
if self.handle is not None:
self.handle.close()
self.handle = None
if not self.ready_published:
try:
os.remove(self.partial_path)
except FileNotFoundError:
pass
def __enter__(self):
return self
def __exit__(self, exc_type, value, traceback):
if not self.finished:
self.abort()
class ResultBundleReader:
def __init__(self, path, max_event_bytes=DEFAULT_MAX_EVENT_BYTES):
self.path = os.path.abspath(path)
self.max_event_bytes = max(1, int(max_event_bytes))
self._validated = None
self._validated_fingerprint = None
@classmethod
def from_reservation(cls, root, reservation, max_event_bytes=DEFAULT_MAX_EVENT_BYTES):
reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation)
return cls(bundle_ready_path(root, reservation.bundle_id), max_event_bytes=max_event_bytes)
@staticmethod
def _stat_fingerprint(value):
return (
int(value.st_dev), int(value.st_ino), int(value.st_mode),
int(value.st_size), int(value.st_mtime_ns),
)
def _file_fingerprint(self):
reject_reparse_components(self.path)
if not private_file_ready(self.path):
raise ResultBundleUnavailableError('bundle path is not currently available as an exact private regular file')
return self._stat_fingerprint(os.stat(self.path, follow_symlinks=False))
def _iter_frames(self, expected_fingerprint=None):
fingerprint = self._file_fingerprint()
if expected_fingerprint is not None and fingerprint != expected_fingerprint:
raise ResultBundleError('bundle changed after validation')
size = fingerprint[3]
if size <= len(MAGIC) or size > self.max_event_bytes:
raise ResultBundleError('bundle aggregate byte bound is invalid')
with open(self.path, 'rb', buffering=0) as handle:
opened_fingerprint = self._stat_fingerprint(os.fstat(handle.fileno()))
if opened_fingerprint != fingerprint:
raise ResultBundleUnavailableError('bundle identity changed while it was opened')
magic = handle.read(len(MAGIC))
if magic != MAGIC:
raise ResultBundleError('bundle magic/version mismatch')
offset = len(MAGIC)
while offset < size:
header = handle.read(FRAME_HEADER.size)
if len(header) != FRAME_HEADER.size:
raise ResultBundleError('truncated bundle frame header')
frame_type, payload_length = FRAME_HEADER.unpack(header)
if frame_type not in FRAME_TYPES:
raise ResultBundleError('unknown bundle frame type')
bound = min(FRAME_BOUNDS[frame_type], self.max_event_bytes)
if payload_length > bound or offset + FRAME_HEADER.size + payload_length > size:
raise ResultBundleError('bundle frame length exceeds its bound')
payload = handle.read(payload_length)
if len(payload) != payload_length:
raise ResultBundleError('truncated bundle frame payload')
try:
value = json.loads(payload.decode('utf-8', errors='strict'))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise ResultBundleError('bundle frame contains invalid UTF-8 JSON') from exc
if canonical_json_bytes(value) != payload:
raise ResultBundleError('bundle frame JSON is not canonical')
offset += FRAME_HEADER.size + payload_length
yield frame_type, value, header + payload, offset
if offset != size:
raise ResultBundleError('bundle byte count is inconsistent')
if self._stat_fingerprint(os.fstat(handle.fileno())) != opened_fingerprint:
raise ResultBundleError('bundle changed while it was read')
if self._file_fingerprint() != fingerprint:
raise ResultBundleError('bundle path changed while it was read')
def validate(self):
if self._validated is not None:
return self._validated
digest = hashlib.sha256(MAGIC)
header_value = None
metadata_value = None
footer = None
counts = {b'F': 0, b'E': 0, b'D': 0, b'K': 0}
diagnostic_bytes = 0
diagnostic_aggregate_bytes = 0
diagnostic_uids = set()
diagnostic_scan_outcomes = []
frame_count = 0
last_frame_order = -1
final_offset = len(MAGIC)
content_bytes = None
fingerprint = self._file_fingerprint()
for frame_type, value, framed, offset in self._iter_frames(fingerprint):
frame_count += 1
if frame_count > MAX_TOTAL_FRAMES:
raise ResultBundleError('bundle frame count exceeds its bound')
if FRAME_ORDER[frame_type] < last_frame_order:
raise ResultBundleError('bundle frame order is not deterministic')
last_frame_order = FRAME_ORDER[frame_type]
final_offset = offset
if footer is not None:
raise ResultBundleError('commit footer is not the final frame')
if metadata_value is not None and frame_type != b'C':
raise ResultBundleError('result metadata is not immediately before the commit footer')
if frame_count == 1 and frame_type != b'H':
raise ResultBundleError('bundle header is not the first frame')
if frame_type in FRAME_TYPES and not isinstance(value, dict):
raise ResultBundleError('bundle typed frame must contain a JSON object')
if frame_type == b'E' and not isinstance(value.get('error'), str):
raise ResultBundleError('bundle error frame is invalid')
if frame_type == b'H':
if header_value is not None:
raise ResultBundleError('bundle contains duplicate headers')
header_value = value
elif frame_type == b'M':
if metadata_value is not None:
raise ResultBundleError('bundle contains duplicate metadata')
metadata_value = value
elif frame_type == b'C':
footer = value
content_bytes = offset - len(framed)
continue
elif frame_type == b'D':
try:
envelope = decode_diagnostic_envelope(canonical_json_bytes(value))
except (TypeError, ValueError, UnicodeError) as exc:
raise ResultBundleError('bundle diagnostic frame is invalid') from exc
if envelope.diagnostic_uid in diagnostic_uids:
raise ResultBundleError('bundle diagnostic identity is duplicated')
if header_value is None or (
envelope.reservation_id != int(header_value.get('reservation_id') or 0)
or envelope.scan_event_id != str(header_value.get('scan_event_id') or '')
or envelope.source != str(header_value.get('source') or '')
):
raise ResultBundleError('bundle diagnostic identity conflicts with its header')
diagnostic_uids.add(envelope.diagnostic_uid)
if envelope.assignment_outcome is not AssignmentOutcome.ACCEPTED:
raise ResultBundleError(
'bundle diagnostic assignment outcome is invalid'
)
diagnostic_scan_outcomes.append(envelope.scan_outcome.value)
counts[b'D'] += 1
if counts[b'D'] > MAX_DIAGNOSTICS_PER_ASSIGNMENT:
raise ResultBundleError('bundle typed frame count exceeds its bound')
envelope_bytes = len(encode_diagnostic_envelope(envelope))
diagnostic_bytes += len(framed)
diagnostic_aggregate_bytes += envelope_bytes + 1
if (
diagnostic_aggregate_bytes > MAX_DIAGNOSTIC_AGGREGATE_BYTES
):
raise ResultBundleError('bundle diagnostic aggregate exceeds its byte bound')
elif frame_type in counts:
counts[frame_type] += 1
limit = {
b'F': MAX_FINDING_FRAMES,
b'E': MAX_ERROR_FRAMES,
b'D': MAX_DIAGNOSTICS_PER_ASSIGNMENT,
b'K': MAX_CANDIDATE_FRAMES,
}[frame_type]
if counts[frame_type] > limit:
raise ResultBundleError('bundle typed frame count exceeds its bound')
digest.update(framed)
if not isinstance(header_value, dict) or not isinstance(metadata_value, dict) or not isinstance(footer, dict):
raise ResultBundleError('bundle is missing required header, metadata, or footer')
if frame_count < 3 or footer.get('format_version') != FORMAT_VERSION:
raise ResultBundleError('bundle footer version is invalid')
expected_scan_outcome = {
'clean': 'clean',
'found': 'found',
'degraded': 'degraded',
'error': 'error',
'skipped': 'skipped',
}.get(str(metadata_value.get('status') or 'clean'), 'error')
if any(
outcome != expected_scan_outcome
for outcome in diagnostic_scan_outcomes
):
raise ResultBundleError(
'bundle diagnostic scan outcome conflicts with result metadata'
)
base_footer_fields = {
'format_version', 'content_sha256', 'content_bytes', 'byte_count',
'frame_count', 'finding_count', 'error_count', 'candidate_count',
}
expected_footer_fields = (
base_footer_fields | {'diagnostic_count', 'diagnostic_bytes'}
if counts[b'D'] else base_footer_fields
)
if set(footer) != expected_footer_fields:
raise ResultBundleError('bundle footer shape is invalid')
expected = {
'content_sha256': digest.hexdigest(),
'content_bytes': content_bytes,
'byte_count': final_offset,
'frame_count': frame_count,
'finding_count': counts[b'F'],
'error_count': counts[b'E'],
'candidate_count': counts[b'K'],
}
if counts[b'D']:
expected.update({
'diagnostic_count': counts[b'D'],
'diagnostic_bytes': diagnostic_bytes,
})
if footer.get('content_sha256') != expected['content_sha256']:
raise ResultBundleError('bundle content hash mismatch')
for key in (
'content_bytes', 'byte_count', 'frame_count', 'finding_count',
'error_count', 'candidate_count',
):
value = footer.get(key)
if isinstance(value, bool) or not isinstance(value, int) or value != expected[key]:
raise ResultBundleError(f'bundle footer {key} mismatch')
diagnostic_footer_fields = {'diagnostic_count', 'diagnostic_bytes'} & set(footer)
if counts[b'D']:
if diagnostic_footer_fields != {'diagnostic_count', 'diagnostic_bytes'}:
raise ResultBundleError('bundle footer diagnostic accounting is missing')
for key in ('diagnostic_count', 'diagnostic_bytes'):
value = footer.get(key)
if isinstance(value, bool) or not isinstance(value, int) or value != expected[key]:
raise ResultBundleError(f'bundle footer {key} mismatch')
elif diagnostic_footer_fields:
raise ResultBundleError('D-less bundle has unexpected diagnostic accounting')
bundle_id = _validated_id(header_value.get('bundle_id'), 'bundle_id')
event_id = _validated_id(header_value.get('scan_event_id'), 'scan_event_id')
reservation_id = header_value.get('reservation_id')
if isinstance(reservation_id, bool) or not isinstance(reservation_id, int) or reservation_id <= 0:
raise ResultBundleError('invalid reservation_id')
self._validated = BundleMetadata(
reservation_id=reservation_id,
bundle_id=bundle_id,
scan_event_id=event_id,
scan_event_hash=expected['content_sha256'],
relative_path='',
actual_bytes=final_offset,
frame_count=frame_count,
finding_count=counts[b'F'],
error_count=counts[b'E'],
candidate_count=counts[b'K'],
header=header_value,
result_metadata=metadata_value,
diagnostic_count=counts[b'D'],
diagnostic_bytes=diagnostic_bytes,
)
self._validated_fingerprint = fingerprint
return self._validated
def _values(self, wanted):
validated = self.validate()
digest = hashlib.sha256(MAGIC)
footer = None
for frame_type, value, framed, _ in self._iter_frames(
self._validated_fingerprint
):
if frame_type == b'C':
footer = value
else:
digest.update(framed)
if frame_type == wanted:
yield value
if (
digest.hexdigest() != validated.scan_event_hash
or not isinstance(footer, dict)
or footer.get('content_sha256') != validated.scan_event_hash
):
raise ResultBundleError('bundle content changed after validation')
def iter_findings(self):
return self._values(b'F')
def iter_errors(self):
for value in self._values(b'E'):
yield value.get('error') if isinstance(value, dict) else value
def iter_candidates(self):
return self._values(b'K')
def iter_diagnostics(self):
for value in self._values(b'D'):
envelope = decode_diagnostic_envelope(canonical_json_bytes(value))
yield json.loads(encode_diagnostic_envelope(envelope).decode('ascii'))
def effective_diagnostics(self):
validated = self.validate()
if validated.diagnostic_count:
return tuple(self.iter_diagnostics())
metadata = self.metadata()
header = self.header()
envelopes = build_legacy_error_frame_diagnostics(
reservation_id=validated.reservation_id,
scan_event_id=validated.scan_event_id,
slot_id=0,
source=str(header.get('source') or ''),
timestamp=(
metadata.get('timestamp') or metadata.get('scan_started_at')
),
errors=tuple(self.iter_errors()),
retryable=bool(metadata.get('retryable', False)),
attempt=max(1, int(metadata.get('attempt') or 1)),
)
return tuple(
json.loads(encode_diagnostic_envelope(envelope).decode('ascii'))
for envelope in envelopes
)
def metadata(self):
self.validate()
if self._file_fingerprint() != self._validated_fingerprint:
raise ResultBundleError('bundle changed after validation')
return dict(self.validate().result_metadata)
def header(self):
self.validate()
if self._file_fingerprint() != self._validated_fingerprint:
raise ResultBundleError('bundle changed after validation')
return dict(self.validate().header)
+605
View File
@@ -0,0 +1,605 @@
import sys
sys.dont_write_bytecode = True
if not sys.dont_write_bytecode:
raise RuntimeError('result ingester could not disable bytecode writes')
import argparse
import json
import logging
import os
import time
from lifecycle_authority import require_active_supervisor_child
from paths import apply_path_config
from process_identity import current_process_identity, exact_process_identity_state
from result_bundle import (
ResultBundleError,
ResultBundleReader,
bundle_partial_relative_path,
)
from runtime_security import (
PrivatePathState,
durable_publish,
durable_unlink,
ensure_private_directory,
private_file_ready,
inspect_private_relative_path,
require_private_directory,
sha256_file,
)
from scanner_db import (
DockerCoverageDispositionConflictError,
DockerFindingAttributionLimitError,
ScanEventConflictError,
ScannerDB,
)
logger = logging.getLogger(__name__)
class ResultIngester:
def __init__(
self, db, bundle_root, supervisor_instance_id, lease_seconds=300, fault=None,
quarantine_max_items=10000, quarantine_max_bytes=1024 * 1024 * 1024,
metadata_retention_days=30, metadata_retirement_batch=100,
recover_expired_ready=False,
):
self.db = db
self.bundle_root = require_private_directory(bundle_root, create=False)
self.supervisor_instance_id = str(supervisor_instance_id)
self.lease_seconds = max(30, int(lease_seconds))
self.fault = fault
self.lease = None
self.recovery_after_id = 0
self.quarantine_max_items = max(0, int(quarantine_max_items))
self.quarantine_max_bytes = max(0, int(quarantine_max_bytes))
self.metadata_retention_seconds = max(1, int(metadata_retention_days)) * 86400
self.metadata_retirement_batch = min(500, max(1, int(metadata_retirement_batch)))
self.next_metadata_retirement = 0.0
self.recover_expired_ready = recover_expired_ready is True
def _inject(self, stage, value=None):
if self.fault is not None:
self.fault(stage, value)
def start(self):
self.db.require_runtime_safety_schema()
self.db.require_final_cutover()
identity = current_process_identity()
self.lease = self.db.acquire_pipeline_lease(
'result_ingester', self.supervisor_instance_id, identity,
lease_seconds=self.lease_seconds, initial_state='recovering',
)
if not self.lease:
raise RuntimeError('another result ingester owns the singleton advisory lock')
self.reconcile_terminal_artifacts()
self.retire_terminal_metadata()
self.recover()
if not self.heartbeat('ready'):
raise RuntimeError('result ingester ready lease publication failed')
return self
def heartbeat(self, state='ready', error=''):
return self.db.heartbeat_pipeline_lease(
'result_ingester', self.lease['generation'], self.lease['lease_token'],
lease_seconds=self.lease_seconds, state=state, error=error,
)
def stop(self, error=''):
if self.lease:
released = self.db.release_pipeline_lease(
'result_ingester', self.lease['generation'], self.lease['lease_token'],
state='failed' if error else 'released', error=error,
)
self.lease = None
return released
return True
def _path(self, relative):
normalized = str(relative or '').replace('/', os.sep)
path = os.path.abspath(os.path.join(self.bundle_root, normalized))
if os.path.commonpath((self.bundle_root, path)) != self.bundle_root or path == self.bundle_root:
raise ResultBundleError('bundle database path escapes its configured root')
return path
def _quarantine_path(self, reservation):
bundle_id = str(reservation['bundle_id'])
return os.path.join(
self.bundle_root, 'quarantine', bundle_id[:2], f'{bundle_id}.trb',
)
def _quarantine_relative_path(self, reservation):
bundle_id = str(reservation['bundle_id'])
return f'quarantine/{bundle_id[:2]}/{bundle_id}.trb'
def _ensure_quarantine_shard(self, reservation):
bundle_id = str(reservation['bundle_id'])
return ensure_private_directory(
os.path.join(self.bundle_root, 'quarantine', bundle_id[:2]),
reject_reparse=True,
)
def _inspect(self, relative_path):
return inspect_private_relative_path(self.bundle_root, relative_path)
def _defer_reservation_cleanup(self, reservation, error):
try:
self.db.defer_result_reservation_cleanup(reservation['id'], str(error))
except Exception:
logger.warning(
'reservation cleanup backoff could not be recorded: %s', reservation['id']
)
def reconcile_terminal_artifacts(self, max_pages=100):
if not hasattr(self.db, 'bundle_terminal_temp_artifacts'):
return
for _ in range(max(1, int(max_pages))):
rows = self.db.bundle_terminal_temp_artifacts(100)
if not rows:
return
progressed = False
for row in rows:
try:
inspection = self._inspect(row['relative_path'])
if inspection.state == PrivatePathState.UNKNOWN:
self.db.defer_pipeline_artifact_cleanup(
row['id'], inspection.detail or 'artifact storage state is unknown',
)
continue
if inspection.state == PrivatePathState.PRESENT:
if not private_file_ready(inspection.path):
self.db.defer_pipeline_artifact_cleanup(
row['id'], 'artifact is not an exact private file',
)
continue
durable_unlink(inspection.path)
inspection = self._inspect(row['relative_path'])
if inspection.state == PrivatePathState.ABSENT:
self.db.mark_pipeline_artifact_deleted(row['id'])
progressed = True
else:
self.db.defer_pipeline_artifact_cleanup(
row['id'], 'artifact unlink was not confirmed',
)
except OSError as exc:
try:
self.db.defer_pipeline_artifact_cleanup(row['id'], str(exc))
except Exception:
logger.warning(
'artifact cleanup backoff could not be recorded: %s', row['id']
)
continue
if not progressed:
return
if len(rows) < 100:
return
return
def retire_terminal_metadata(self):
if time.monotonic() < self.next_metadata_retirement:
return
self.next_metadata_retirement = time.monotonic() + 60
try:
self.db.retire_admission_intents(
self.metadata_retention_seconds, self.metadata_retirement_batch,
)
self.db.retire_deleted_pipeline_artifacts(
self.metadata_retention_seconds, self.metadata_retirement_batch,
)
except Exception as exc:
logger.warning('bounded terminal metadata retirement deferred: %s', type(exc).__name__)
def quarantine(self, reservation, ready_path, reason_code, detail):
ready_relative = str(reservation['ready_relative_path']).replace('\\', '/')
quarantine_relative = self._quarantine_relative_path(reservation)
self._ensure_quarantine_shard(reservation)
ready = self._inspect(ready_relative)
quarantine = self._inspect(quarantine_relative)
if ready.state == PrivatePathState.UNKNOWN or quarantine.state == PrivatePathState.UNKNOWN:
raise ResultBundleError('bundle quarantine path state is unknown')
if ready.state == PrivatePathState.PRESENT and quarantine.state == PrivatePathState.PRESENT:
raise ResultBundleError('both ready and quarantine paths exist for one reservation')
if ready.state == PrivatePathState.ABSENT and quarantine.state == PrivatePathState.ABSENT:
raise ResultBundleError('bundle disappeared before quarantine')
quarantine_path = quarantine.path
ensure_private_directory(os.path.dirname(quarantine_path), reject_reparse=True)
source_path = ready.path if ready.state == PrivatePathState.PRESENT else quarantine.path
if not private_file_ready(source_path):
raise ResultBundleError('bundle quarantine source is not an exact private file')
byte_count = os.path.getsize(source_path)
payload_hash = sha256_file(source_path) if byte_count else ''
relative = quarantine_relative
quarantine_id = self.db.quarantine_result_bundle(
reservation['id'], reason_code, detail, relative,
payload_sha256=payload_hash, byte_count=byte_count,
quarantine_max_items=self.quarantine_max_items,
quarantine_max_bytes=self.quarantine_max_bytes,
physical_confirmed=ready.state != PrivatePathState.PRESENT,
)
if ready.state == PrivatePathState.PRESENT:
durable_publish(ready.path, quarantine_path)
confirmed = self._inspect(quarantine_relative)
if confirmed.state != PrivatePathState.PRESENT:
raise ResultBundleError('bundle quarantine publication was not confirmed')
quarantine_id = self.db.quarantine_result_bundle(
reservation['id'], reason_code, detail, relative,
payload_sha256=payload_hash, byte_count=byte_count,
quarantine_max_items=self.quarantine_max_items,
quarantine_max_bytes=self.quarantine_max_bytes,
physical_confirmed=True,
)
return quarantine_id
def recover(self, page_size=100, max_pages=100):
pages = 0
while pages < max(1, int(max_pages)):
rows = self.db.active_result_reservations(self.recovery_after_id, page_size)
if not rows:
self.recovery_after_id = 0
return pages
pages += 1
for reservation in rows:
self.recovery_after_id = int(reservation['id'])
if (
reservation['state'] == 'scanning'
and str(reservation.get('assignment_kind') or 'local') == 'remote'
):
continue
ready_relative = str(reservation['ready_relative_path']).replace('\\', '/')
ready = self._inspect(ready_relative)
identity_state = None
if reservation['state'] == 'scanning':
identity_state = exact_process_identity_state(
reservation['producer_pid'], reservation['producer_creation_time'],
reservation['producer_executable'],
)
if (
ready.state == PrivatePathState.UNKNOWN
and identity_state in ('dead', 'reused')
):
try:
ensure_private_directory(
os.path.join(
self.bundle_root, 'ready',
str(reservation['bundle_id'])[:2],
),
reject_reparse=True,
)
ensure_private_directory(
os.path.join(
self.bundle_root, 'tmp',
str(reservation['bundle_id'])[:2],
),
reject_reparse=True,
)
ready = self._inspect(ready_relative)
except OSError as exc:
self._defer_reservation_cleanup(reservation, exc)
continue
quarantine_relative = self._quarantine_relative_path(reservation)
try:
self._ensure_quarantine_shard(reservation)
except OSError as exc:
self._defer_reservation_cleanup(reservation, exc)
continue
quarantine = self._inspect(quarantine_relative)
if quarantine.state == PrivatePathState.PRESENT:
try:
self.db.quarantine_result_bundle(
reservation['id'],
reservation.get('last_error_code') or 'recovered_quarantine',
reservation.get('last_error_detail') or 'recovered deterministic quarantine file',
quarantine_relative,
payload_sha256=sha256_file(quarantine.path),
byte_count=os.path.getsize(quarantine.path),
quarantine_max_items=self.quarantine_max_items,
quarantine_max_bytes=self.quarantine_max_bytes,
physical_confirmed=True,
)
except OSError as exc:
self._defer_reservation_cleanup(reservation, exc)
continue
if quarantine.state == PrivatePathState.UNKNOWN:
self._defer_reservation_cleanup(
reservation, quarantine.detail or 'quarantine state unknown',
)
continue
if ready.state == PrivatePathState.UNKNOWN:
self._defer_reservation_cleanup(
reservation, ready.detail or 'ready state unknown',
)
continue
prepared_quarantine = self.db.pending_result_bundle_quarantine(
reservation['id']
)
if prepared_quarantine and ready.state == PrivatePathState.PRESENT:
try:
self.quarantine(
reservation, ready.path,
prepared_quarantine['reason_code'],
prepared_quarantine.get('reason_detail') or 'recovered prepared quarantine',
)
except OSError as exc:
self._defer_reservation_cleanup(reservation, exc)
continue
if reservation['state'] == 'scanning':
if ready.state == PrivatePathState.PRESENT:
try:
metadata = ResultBundleReader(ready.path).validate()
recovered = metadata.as_dict()
recovered['relative_path'] = str(
reservation['ready_relative_path']
).replace('\\', '/')
marked_ready = self.db.mark_result_bundle_ready(
reservation['id'], recovered,
)
if (
not marked_ready
and self.recover_expired_ready
and identity_state in ('dead', 'reused')
):
self.db.recover_expired_result_bundle_ready(
reservation['id'], recovered,
)
except OSError as exc:
self._defer_reservation_cleanup(reservation, exc)
except (ValueError, ResultBundleError) as exc:
try:
self.quarantine(
reservation, ready.path, 'bundle_validation_failed', str(exc),
)
except OSError as cleanup_exc:
self._defer_reservation_cleanup(reservation, cleanup_exc)
continue
if identity_state in ('dead', 'reused'):
try:
ensure_private_directory(
os.path.join(self.bundle_root, 'ready', str(reservation['bundle_id'])[:2]),
reject_reparse=True,
)
ensure_private_directory(
os.path.join(self.bundle_root, 'tmp', str(reservation['bundle_id'])[:2]),
reject_reparse=True,
)
except OSError as exc:
self._defer_reservation_cleanup(reservation, exc)
continue
ready = self._inspect(ready_relative)
if ready.state != PrivatePathState.ABSENT:
continue
if identity_state in ('dead', 'reused'):
partial_relative = bundle_partial_relative_path(
reservation['bundle_id'], reservation['reservation_token'],
).replace(os.sep, '/')
partial = self._inspect(partial_relative)
if partial.state == PrivatePathState.UNKNOWN:
self._defer_reservation_cleanup(
reservation, partial.detail or 'partial state unknown',
)
continue
if partial.state == PrivatePathState.PRESENT:
if not private_file_ready(partial.path):
continue
try:
durable_unlink(partial.path)
except OSError as exc:
self._defer_reservation_cleanup(reservation, exc)
continue
partial = self._inspect(partial_relative)
if partial.state != PrivatePathState.ABSENT:
self._defer_reservation_cleanup(
reservation, 'partial unlink was not confirmed',
)
continue
self.db.refund_uncommitted_reservation(
reservation['id'],
{
'pid': reservation['producer_pid'],
'creation_time': reservation['producer_creation_time'],
'executable': reservation['producer_executable'],
},
f'producer identity is {identity_state} and exact ready path is absent',
partial_absence_confirmed=True,
)
elif reservation['state'] in ('ready', 'ingesting'):
if ready.state == PrivatePathState.ABSENT:
self.db.quarantine_result_bundle(
reservation['id'], 'ready_bundle_missing',
'database ready row has a definitively absent exact ready path',
'', byte_count=0,
quarantine_max_items=self.quarantine_max_items,
quarantine_max_bytes=self.quarantine_max_bytes,
physical_confirmed=True,
)
elif reservation['state'] == 'db_committed':
bundle = self.db.result_bundle_for_reservation(reservation['id'])
if not bundle:
continue
if ready.state == PrivatePathState.PRESENT:
if not private_file_ready(ready.path):
continue
try:
durable_unlink(ready.path)
except OSError as exc:
self._defer_reservation_cleanup(reservation, exc)
continue
ready = self._inspect(ready_relative)
if ready.state == PrivatePathState.ABSENT:
event = self.db.confirm_scan_event(
reservation['scan_event_id'], bundle['scan_event_hash'],
)
if event:
self.db.acknowledge_removed_bundle(
reservation['id'], reservation['scan_event_id'],
bundle['scan_event_hash'],
)
return pages
def process_one(self):
claimed = self.db.claim_ready_result_bundle(
self.lease['generation'], self.lease['lease_token'], self.lease_seconds,
)
if not claimed:
return False
reservation = claimed['reservation']
bundle = claimed['bundle']
ready_relative = str(bundle['relative_path']).replace('\\', '/')
ready = self._inspect(ready_relative)
try:
if bundle['state'] == 'db_committed':
event = self.db.confirm_scan_event(bundle['scan_event_id'], bundle['scan_event_hash'])
if not event:
raise ResultBundleError('db_committed bundle has no exact authoritative event')
else:
if ready.state == PrivatePathState.UNKNOWN:
return False
if ready.state == PrivatePathState.ABSENT:
self.db.quarantine_result_bundle(
reservation['id'], 'ready_bundle_missing',
'claimed ready bundle is definitively absent', '',
quarantine_max_items=self.quarantine_max_items,
quarantine_max_bytes=self.quarantine_max_bytes,
physical_confirmed=True,
)
return True
self._inject('before_validation', reservation)
reader = ResultBundleReader(ready.path)
validated = reader.validate()
self._inject('after_validation', validated)
if (
validated.bundle_id != str(bundle['bundle_id'])
or validated.scan_event_id != str(bundle['scan_event_id'])
or validated.scan_event_hash != str(bundle['scan_event_hash'])
or validated.actual_bytes != int(bundle['actual_bytes'])
):
raise ResultBundleError('validated bundle totals conflict with its claimed database row')
self._inject('before_db_commit', reservation)
self.db.ingest_result_bundle(reader, reservation, bundle)
self._inject('after_db_commit', reservation)
event = self.db.confirm_scan_event(bundle['scan_event_id'], bundle['scan_event_hash'])
self._inject('after_confirmation', event)
if not event:
raise ResultBundleError('database commit was not confirmed by exact event ID and hash')
ready = self._inspect(ready_relative)
if ready.state == PrivatePathState.UNKNOWN:
return False
if ready.state == PrivatePathState.PRESENT:
self._inject('before_unlink', reservation)
durable_unlink(ready.path)
self._inject('after_unlink', reservation)
ready = self._inspect(ready_relative)
if ready.state != PrivatePathState.ABSENT:
return False
self._inject('before_capacity_release', reservation)
if not self.db.acknowledge_removed_bundle(
reservation['id'], bundle['scan_event_id'], bundle['scan_event_hash'],
):
raise RuntimeError('bundle capacity acknowledgement was not fenced')
self._inject('after_capacity_release', reservation)
return True
except DockerCoverageDispositionConflictError as exc:
ready = self._inspect(ready_relative)
if ready.state == PrivatePathState.PRESENT:
try:
self.quarantine(
reservation, ready.path,
'docker_coverage_disposition_conflict', str(exc),
)
except OSError as cleanup_exc:
self._defer_reservation_cleanup(reservation, cleanup_exc)
return False
return True
except DockerFindingAttributionLimitError as exc:
ready = self._inspect(ready_relative)
if ready.state == PrivatePathState.PRESENT:
try:
self.quarantine(
reservation, ready.path,
'docker_attribution_limit_exceeded', str(exc),
)
except OSError as cleanup_exc:
self._defer_reservation_cleanup(reservation, cleanup_exc)
return False
return True
except ScanEventConflictError as exc:
ready = self._inspect(ready_relative)
if ready.state == PrivatePathState.PRESENT:
try:
self.quarantine(reservation, ready.path, 'scan_event_hash_conflict', str(exc))
except OSError as cleanup_exc:
self._defer_reservation_cleanup(reservation, cleanup_exc)
return False
return True
except (ResultBundleError, ValueError) as exc:
ready = self._inspect(ready_relative)
if ready.state == PrivatePathState.PRESENT:
try:
self.quarantine(reservation, ready.path, 'bundle_validation_failed', str(exc))
except OSError as cleanup_exc:
self._defer_reservation_cleanup(reservation, cleanup_exc)
return False
return True
raise
except OSError as exc:
self._defer_reservation_cleanup(reservation, exc)
return False
def parse_args():
parser = argparse.ArgumentParser(description='Singleton durable result bundle ingester')
parser.add_argument('--config', required=True)
return parser.parse_args()
def main():
metadata = require_active_supervisor_child(child_kind='result-ingester', require_dsn=True)
args = parse_args()
import yaml
with open(args.config, 'r', encoding='utf-8') as handle:
config = apply_path_config(yaml.safe_load(handle) or {}, args.config)
global_config = config.get('global') or {}
db = ScannerDB(db_url=global_config['database_url'], initialize=False)
if not db.enabled:
raise SystemExit('result ingester PostgreSQL connection is unavailable')
db.set_application_name('truf-result-ingester')
worker = ResultIngester(
db, global_config['result_bundle_dir'], metadata['instance_id'],
lease_seconds=int(((config.get('supervisor') or {}).get('result_ingester') or {}).get('lease_seconds', 300)),
quarantine_max_items=int(global_config.get('pipeline_quarantine_max_items', 10000)),
quarantine_max_bytes=int(global_config.get('pipeline_quarantine_max_bytes', 1024 * 1024 * 1024)),
metadata_retention_days=int(global_config.get('pipeline_metadata_retention_days', 30)),
metadata_retirement_batch=int(global_config.get('pipeline_metadata_retirement_batch', 100)),
)
error = ''
try:
worker.start()
idle = max(0.05, float(((config.get('supervisor') or {}).get('result_ingester') or {}).get('poll_sec', 0.2)))
next_heartbeat = time.monotonic() + worker.lease_seconds / 3
while True:
worker.retire_terminal_metadata()
worker.reconcile_terminal_artifacts(max_pages=1)
worker.recover(max_pages=1)
processed = worker.process_one()
if time.monotonic() >= next_heartbeat:
if not worker.heartbeat('ready'):
raise RuntimeError('result ingester heartbeat fence was lost')
next_heartbeat = time.monotonic() + worker.lease_seconds / 3
if not processed:
time.sleep(idle)
except KeyboardInterrupt:
pass
except BaseException as exc:
error = f'{type(exc).__name__}: {exc}'
raise
finally:
try:
worker.stop(error)
finally:
db.close()
if __name__ == '__main__':
main()
+1073
View File
File diff suppressed because it is too large Load Diff
+143
View File
@@ -0,0 +1,143 @@
"""Stdlib-only pre-import boundary for canonical runtime entrypoints."""
import sys
import os
if __name__ == '__main__':
if sys.platform != 'linux' or not os.path.isfile('/.dockerenv') or os.path.abspath(__file__) != '/opt/truf/app/runtime_bootstrap.py':
raise SystemExit('Docker development copy: runtime control is disabled outside the prepared container. See DOCKER_MIGRATION.md.')
import runpy
runpy.run_path('/opt/truf/app/container_runtime.py')['require_container']()
sys.dont_write_bytecode = True
if not sys.dont_write_bytecode:
raise RuntimeError('runtime bootstrap could not disable bytecode writes')
import runpy
import stat
RUNTIME_BOOTSTRAP_ENV = 'TRUF_RUNTIME_BOOTSTRAP'
RUNTIME_BOOTSTRAP_VALUE = '1'
APPLICATION_IMPORT_SUFFIXES = ('.py', '.pyw', '.pyc', '.pyd')
SUPERVISOR_ENTRYPOINT_FLAG = '--runtime-bootstrap-entrypoint'
TARGETS = {
'supervisor': 'supervisor.py',
'postgres-runtime': 'postgres_runtime.py',
'migrate-runtime-safety': 'migrate_runtime_safety.py',
}
def _require_isolated_startup():
if not (
sys.flags.isolated
and sys.flags.no_site
and sys.flags.dont_write_bytecode
and sys.dont_write_bytecode
):
raise RuntimeError('runtime bootstrap requires isolated no-site bytecode-free startup (-I -S -B)')
def _canonical(path):
return os.path.normcase(os.path.realpath(os.path.abspath(os.fspath(path))))
def _is_reparse_point(path):
details = os.lstat(path)
if stat.S_ISLNK(details.st_mode):
return True
attributes = getattr(details, 'st_file_attributes', 0)
reparse_attribute = getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0)
return bool(attributes & reparse_attribute) or getattr(os.path, 'isjunction', lambda _path: False)(path)
def _reject_cached_bytecode(app_dir):
def raise_walk_error(exc):
raise RuntimeError(f'unable to inspect the application root: {exc}') from exc
try:
root_details = os.lstat(app_dir)
except OSError as exc:
raise RuntimeError(f'application root is unavailable: {app_dir}') from exc
if _is_reparse_point(app_dir):
raise RuntimeError(f'application root reparse point is forbidden: {app_dir}')
if not stat.S_ISDIR(root_details.st_mode):
raise RuntimeError(f'application root is not a directory: {app_dir}')
canonical_root = _canonical(app_dir)
for current, directories, files in os.walk(app_dir, followlinks=False, onerror=raise_walk_error):
for name in directories:
candidate = os.path.join(current, name)
if _is_reparse_point(candidate):
relative = os.path.relpath(candidate, app_dir).replace(os.sep, '/')
if name.lower() == '__pycache__':
raise RuntimeError(f'application __pycache__ link is forbidden: {relative}')
raise RuntimeError(f'application directory reparse point is forbidden: {relative}')
relative_current = os.path.relpath(current, app_dir)
in_cache = any(part.lower() == '__pycache__' for part in relative_current.split(os.sep))
for name in files:
candidate = os.path.join(current, name)
relative = os.path.relpath(candidate, app_dir).replace(os.sep, '/')
if _is_reparse_point(candidate):
raise RuntimeError(f'application file reparse point is forbidden: {relative}')
if name.lower().endswith(APPLICATION_IMPORT_SUFFIXES):
try:
contained = os.path.commonpath((canonical_root, _canonical(candidate))) == canonical_root
except ValueError:
contained = False
if not contained:
raise RuntimeError(f'application Python authority escapes its root: {relative}')
if in_cache and name.lower().endswith('.pyc'):
raise RuntimeError(f'application __pycache__ bytecode is forbidden: {relative}')
def _require_supervisor_entrypoint_binding(arguments, entrypoint):
bindings = []
for index, argument in enumerate(arguments):
text = str(argument)
if text == SUPERVISOR_ENTRYPOINT_FLAG:
if index + 1 >= len(arguments):
raise RuntimeError('supervisor runtime bootstrap entrypoint binding has no path')
bindings.append(str(arguments[index + 1]))
elif text.startswith(SUPERVISOR_ENTRYPOINT_FLAG + '='):
bindings.append(text.split('=', 1)[1])
if len(bindings) != 1:
raise RuntimeError('supervisor runtime requires exactly one explicit bootstrap entrypoint binding')
binding = bindings[0]
if not os.path.isabs(binding) or _canonical(binding) != _canonical(entrypoint):
raise RuntimeError('supervisor runtime bootstrap entrypoint binding is not canonical supervisor.py')
def main():
_require_isolated_startup()
if len(sys.argv) < 3 or sys.argv[2] != '--':
raise RuntimeError('usage: runtime_bootstrap.py <supervisor|postgres-runtime|migrate-runtime-safety> -- <args>')
target_name = str(sys.argv[1]).strip().lower()
target_file = TARGETS.get(target_name)
if not target_file:
raise RuntimeError(f'unsupported canonical runtime target: {target_name}')
app_dir = os.path.dirname(os.path.abspath(__file__))
_reject_cached_bytecode(app_dir)
entrypoint = _canonical(os.path.join(app_dir, target_file))
arguments = list(sys.argv[3:])
if target_name == 'supervisor':
_require_supervisor_entrypoint_binding(arguments, entrypoint)
child_namespace = runpy.run_path(os.path.join(app_dir, 'child_bootstrap.py'))
enable_dependencies = child_namespace.get('_enable_dependency_paths')
if not callable(enable_dependencies):
raise RuntimeError('authenticated dependency path bootstrap is unavailable')
enable_dependencies(target_name)
os.environ[RUNTIME_BOOTSTRAP_ENV] = RUNTIME_BOOTSTRAP_VALUE
sys.path.insert(0, app_dir)
sys.argv = [entrypoint, *arguments]
runpy.run_path(entrypoint, run_name='__main__')
if __name__ == '__main__':
try:
main()
except Exception as exc:
raise SystemExit(f'canonical runtime bootstrap rejected launch: {exc}') from exc
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+941
View File
@@ -0,0 +1,941 @@
import hashlib
import hmac
import json
import platform as host_platform
import sys
from contextlib import nullcontext
from dataclasses import dataclass
from datetime import datetime, timedelta, timezone
from typing import Mapping
from result_bundle import BundleReservation, FORMAT_VERSION
from scanner import (
cleanup_assignment_work_dir,
client_remote_execution_binding,
client_scan_phase_events,
client_scan_execution_policy,
scan_slot_scope,
scan_target_result,
stage_result_bundle,
)
from scanner_db import normalize_target
from target_identity import normalize_huggingface_space_id, parse_dockerhub_digest_target
PROTOCOL_VERSION = 2
REMOTE_EXECUTION_SNAPSHOT_SCHEMA = 1
PACKAGE_DETECTOR_POLICY = '@package/detector_policy'
MAX_REMOTE_EXECUTION_SNAPSHOT_BYTES = 64 * 1024
_REMOTE_SCAN_POLICY_BOUNDS = {
'trufflehog_stdout_max_mb': (1, 4096),
'trufflehog_stderr_max_mb': (1, 4096),
'result_bundle_max_event_bytes': (1024, 4 * 1024 * 1024 * 1024),
'trufflehog_max_findings_per_target': (1, 1000000),
'trufflehog_job_memory_limit_bytes': (0, 1 << 50),
'trufflehog_windows_job_cpu_weight': (0, 10000),
'trufflehog_windows_memory_priority': (0, 5),
'trufflehog_diagnostic_max_lines': (1, 2000),
'trufflehog_diagnostic_max_line_chars': (1, 8192),
'trufflehog_diagnostic_max_line_bytes': (1, 8192),
'trufflehog_diagnostic_max_errors': (1, 200),
'trufflehog_diagnostic_max_warnings': (1, 200),
'trufflehog_diagnostic_max_unclassified': (1, 20),
}
class ScanExecutionError(RuntimeError):
pass
@dataclass(frozen=True)
class QueueDispositionPolicy:
target_retry_max_attempts: int = 3
target_retry_base_delay_sec: int = 3600
target_retry_max_delay_sec: int = 86400
target_timeout_retry_delay_sec: int = 21600
docker_layer_checkpoint_delay_sec: int = 60
ci_soft_cooldown_days: int = 7
soft_skip_reasons: tuple[str, ...] = ()
@dataclass(frozen=True)
class ScanCompatibility:
protocol_version: int
bundle_format_version: int
platform_tag: str
code_manifest_sha256: str
effective_config_sha256: str
detector_policy_sha256: str = ''
@classmethod
def from_mapping(cls, value):
value = dict(value or {})
return cls(
protocol_version=int(value.get('protocol_version') or 0),
bundle_format_version=int(value.get('bundle_format_version') or 0),
platform_tag=str(value.get('platform_tag') or ''),
code_manifest_sha256=_digest(value.get('code_manifest_sha256'), 'code manifest'),
effective_config_sha256=_digest(
value.get('effective_config_sha256'), 'effective config',
),
detector_policy_sha256=_digest(
value.get('detector_policy_sha256'), 'detector policy', optional=True,
),
)
def as_dict(self):
return dict(self.__dict__)
@dataclass(frozen=True)
class WorkerBuildCompatibility:
protocol_version: int
bundle_format_version: int
platform_tag: str
code_manifest_sha256: str
detector_policy_sha256: str
@classmethod
def from_mapping(cls, value):
value = dict(value or {})
if set(value) != {
'protocol_version', 'bundle_format_version', 'platform_tag',
'code_manifest_sha256', 'detector_policy_sha256',
}:
raise ValueError('worker build compatibility shape is invalid')
return cls(
protocol_version=int(value.get('protocol_version') or 0),
bundle_format_version=int(value.get('bundle_format_version') or 0),
platform_tag=str(value.get('platform_tag') or ''),
code_manifest_sha256=_digest(value.get('code_manifest_sha256'), 'code manifest'),
detector_policy_sha256=_digest(
value.get('detector_policy_sha256'), 'detector policy',
),
)
def as_dict(self):
return dict(self.__dict__)
def _digest(value, label, optional=False):
value = str(value or '')
if optional and not value:
return ''
if len(value) != 64 or any(char not in '0123456789abcdef' for char in value):
raise ValueError(f'invalid {label} digest')
return value
def local_platform_tag():
machine = host_platform.machine().strip().lower().replace('amd64', 'x86_64')
system = 'windows' if sys.platform == 'win32' else 'linux' if sys.platform.startswith('linux') else ''
if not system or machine not in {'x86_64', 'aarch64', 'arm64'}:
raise ScanExecutionError('unsupported worker platform')
return f'{system}-{machine.replace("arm64", "aarch64")}'
def validate_scan_compatibility(required, local):
required = required if isinstance(required, ScanCompatibility) else ScanCompatibility.from_mapping(required)
local = local if isinstance(local, ScanCompatibility) else ScanCompatibility.from_mapping(local)
if required.protocol_version != PROTOCOL_VERSION or local.protocol_version != PROTOCOL_VERSION:
raise ScanExecutionError('worker protocol is incompatible')
if required.bundle_format_version != FORMAT_VERSION or local.bundle_format_version != FORMAT_VERSION:
raise ScanExecutionError('result bundle format is incompatible')
for name in (
'platform_tag', 'code_manifest_sha256', 'effective_config_sha256',
'detector_policy_sha256',
):
if not hmac.compare_digest(str(getattr(required, name)), str(getattr(local, name))):
raise ScanExecutionError(f'worker {name.replace("_", " ")} is incompatible')
return required
def validate_worker_build_compatibility(
required, local, *, expected_protocol_version=PROTOCOL_VERSION,
):
required = (
required if isinstance(required, WorkerBuildCompatibility)
else WorkerBuildCompatibility.from_mapping(required)
)
local = (
local if isinstance(local, WorkerBuildCompatibility)
else WorkerBuildCompatibility.from_mapping(local)
)
if (
required.protocol_version != expected_protocol_version
or local.protocol_version != expected_protocol_version
):
raise ScanExecutionError('worker protocol is incompatible')
if required.bundle_format_version != FORMAT_VERSION or local.bundle_format_version != FORMAT_VERSION:
raise ScanExecutionError('result bundle format is incompatible')
for name in ('platform_tag', 'code_manifest_sha256', 'detector_policy_sha256'):
if not hmac.compare_digest(str(getattr(required, name)), str(getattr(local, name))):
raise ScanExecutionError(f'worker {name.replace("_", " ")} is incompatible')
return required
_COMMON_SCAN_KWARGS = {
'timeout_sec', 'detectors', 'exclude_detectors', 'no_verification',
'trufflehog_config', 'token',
}
_SOURCE_SCAN_KWARGS = {
'git': {'git_plan'},
'github': {'git_plan', 'max_depth', 'max_commit_age_days', 'commit_lookup_pages',
'skip_if_commit_lookup_fails'},
'github_archive': {'max_depth', 'max_commit_age_days', 'commit_lookup_pages',
'skip_if_commit_lookup_fails'},
'gitlab': {'git_plan', 'external_trufflehog_lifecycle', 'max_depth',
'max_commit_age_days', 'commit_lookup_pages', 'skip_if_commit_lookup_fails'},
'docker': {'docker_layer_work', 'trufflehog_concurrency', 'docker_recovery_limits',
'docker_recovery_min_free_bytes'},
'huggingface': set(),
'npm': {'max_artifact_size_mb'},
'pypi': {'max_artifact_size_mb'},
'package_git': {'max_depth', 'max_commit_age_days', 'commit_lookup_pages',
'skip_if_commit_lookup_fails'},
'postman': {'max_artifact_size_mb'},
'github_gists': {'max_artifact_size_mb'},
'github_archive_files': {'max_artifact_size_mb'},
'github_actions': {
'ci_runs_per_repo', 'ci_lookback_days', 'ci_max_log_archive_mb',
'ci_max_log_file_mb', 'ci_failed_first', 'ci_scan_artifacts',
'ci_max_artifacts_per_run', 'ci_max_artifact_archive_mb',
'ci_max_artifact_file_mb', 'ci_max_artifact_files',
'ci_target_max_download_mb', 'fetch_timeout',
},
'gitlab_ci': {
'ci_pipelines_per_project', 'ci_jobs_per_pipeline', 'ci_lookback_days',
'ci_max_trace_mb', 'ci_scan_artifacts', 'ci_max_artifacts_per_pipeline',
'ci_max_artifact_archive_mb', 'ci_max_artifact_file_mb',
'ci_max_artifact_files', 'ci_target_max_download_mb', 'fetch_timeout',
},
}
def validate_scan_kwargs(platform, scan_kwargs):
platform = str(platform or '').strip().lower()
if platform not in _SOURCE_SCAN_KWARGS:
raise ScanExecutionError('unsupported scan platform')
values = dict(scan_kwargs or {})
unknown = set(values) - _COMMON_SCAN_KWARGS - _SOURCE_SCAN_KWARGS[platform]
if unknown:
raise ScanExecutionError('scan settings contain unsupported fields')
timeout = values.get('timeout_sec')
if isinstance(timeout, bool):
raise ScanExecutionError('scan timeout is invalid')
try:
timeout = float(timeout)
except (TypeError, ValueError, OverflowError):
raise ScanExecutionError('scan timeout is invalid') from None
if not 1 <= timeout <= 86400:
raise ScanExecutionError('scan timeout is outside the worker bound')
values['timeout_sec'] = timeout
return values
def normalize_remote_scan_policy(value):
values = dict(value or {})
expected = {
'drop_detectors', 'strict_git_provider_token_filter',
*_REMOTE_SCAN_POLICY_BOUNDS,
}
if set(values) != expected:
raise ScanExecutionError('remote scan policy shape is invalid')
raw_drop = values['drop_detectors']
if isinstance(raw_drop, str):
raw_drop = raw_drop.split(',')
if not isinstance(raw_drop, (list, tuple)) or len(raw_drop) > 256:
raise ScanExecutionError('remote detector drop policy is invalid')
drop_detectors = []
for item in raw_drop:
if not isinstance(item, str):
raise ScanExecutionError('remote detector drop policy is invalid')
item = item.strip().lower()
if not item:
continue
if len(item) > 128 or any(ord(char) < 32 or ord(char) == 127 for char in item):
raise ScanExecutionError('remote detector drop policy is invalid')
drop_detectors.append(item)
strict = values['strict_git_provider_token_filter']
if not isinstance(strict, bool):
raise ScanExecutionError('remote Git provider token policy is invalid')
normalized = {
'drop_detectors': sorted(set(drop_detectors)),
'strict_git_provider_token_filter': strict,
}
for name, (minimum, maximum) in _REMOTE_SCAN_POLICY_BOUNDS.items():
raw = values[name]
if not isinstance(raw, int) or isinstance(raw, bool):
raise ScanExecutionError('remote scan policy limit is invalid')
number = raw
if number < minimum or number > maximum:
raise ScanExecutionError('remote scan policy limit is outside its bounds')
normalized[name] = number
return normalized
def remote_execution_identity(
platform, scan_kwargs, event_scan_options, queue_policy, limits, scan_policy,
):
normalized_scan = validate_scan_kwargs(platform, scan_kwargs)
event_options = dict(event_scan_options or {})
if 'token' in event_options or 'git_plan' in event_options:
raise ScanExecutionError('event scan settings contain private or planned fields')
expected_event = {
name: value for name, value in normalized_scan.items()
if name not in {'token', 'git_plan'}
}
if event_options != expected_event:
raise ScanExecutionError('event scan settings do not match execution settings')
try:
policy = (
queue_policy if isinstance(queue_policy, QueueDispositionPolicy)
else QueueDispositionPolicy(**dict(queue_policy or {}))
)
except (TypeError, ValueError) as exc:
raise ScanExecutionError('queue disposition policy is invalid') from exc
policy_value = {
'target_retry_max_attempts': int(policy.target_retry_max_attempts),
'target_retry_base_delay_sec': int(policy.target_retry_base_delay_sec),
'target_retry_max_delay_sec': int(policy.target_retry_max_delay_sec),
'target_timeout_retry_delay_sec': int(policy.target_timeout_retry_delay_sec),
'docker_layer_checkpoint_delay_sec': int(policy.docker_layer_checkpoint_delay_sec),
'ci_soft_cooldown_days': int(policy.ci_soft_cooldown_days),
'soft_skip_reasons': list(policy.soft_skip_reasons),
}
limit_values = dict(limits or {})
if set(limit_values) != {'candidate_max_items', 'candidate_max_bytes'}:
raise ScanExecutionError('worker assignment limits are invalid')
normalized_limits = {
'candidate_max_items': int(limit_values['candidate_max_items']),
'candidate_max_bytes': int(limit_values['candidate_max_bytes']),
}
if (
not 1 <= normalized_limits['candidate_max_items'] <= 100000
or not 1024 <= normalized_limits['candidate_max_bytes'] <= 64 * 1024 * 1024
):
raise ScanExecutionError('worker assignment limits are outside their bounds')
execution = {
'source': str(platform or '').strip().lower(),
'scan_kwargs': event_options,
'scan_policy': normalize_remote_scan_policy(scan_policy),
'queue_policy': policy_value,
'limits': normalized_limits,
}
return canonical_json_sha256(execution), execution
def validate_remote_assignment_compatibility(
required, local_build, platform, scan_kwargs, event_scan_options, queue_policy, limits,
scan_policy, *, expected_protocol_version=PROTOCOL_VERSION,
):
required = required if isinstance(required, ScanCompatibility) else ScanCompatibility.from_mapping(required)
validate_worker_build_compatibility({
'protocol_version': required.protocol_version,
'bundle_format_version': required.bundle_format_version,
'platform_tag': required.platform_tag,
'code_manifest_sha256': required.code_manifest_sha256,
'detector_policy_sha256': required.detector_policy_sha256,
}, local_build, expected_protocol_version=expected_protocol_version)
effective, _ = remote_execution_identity(
platform, scan_kwargs, event_scan_options, queue_policy, limits, scan_policy,
)
if not hmac.compare_digest(effective, required.effective_config_sha256):
raise ScanExecutionError('worker effective config is incompatible')
return required
def _remote_snapshot_envelope(value):
if not isinstance(value, dict):
raise ScanExecutionError('remote execution snapshot must be an object')
value = dict(value)
if set(value) != {
'schema', 'compatibility', 'execution', 'planning', 'credential_ref',
} or value.get('schema') != REMOTE_EXECUTION_SNAPSHOT_SCHEMA:
raise ScanExecutionError('remote execution snapshot shape is invalid')
compatibility = ScanCompatibility.from_mapping(value.get('compatibility'))
execution = dict(value.get('execution') or {})
if set(execution) != {
'source', 'scan_kwargs', 'scan_policy', 'queue_policy', 'limits',
}:
raise ScanExecutionError('remote execution snapshot settings are invalid')
source = str(execution.get('source') or '').strip().lower()
effective, normalized_execution = remote_execution_identity(
source, execution.get('scan_kwargs'), execution.get('scan_kwargs'),
execution.get('queue_policy'), execution.get('limits'),
execution.get('scan_policy'),
)
if normalized_execution['scan_kwargs'].get('trufflehog_config') != PACKAGE_DETECTOR_POLICY:
raise ScanExecutionError('remote execution snapshot policy path is invalid')
if not hmac.compare_digest(effective, compatibility.effective_config_sha256):
raise ScanExecutionError('remote execution snapshot effective config is invalid')
planning = dict(value.get('planning') or {})
credential_ref = dict(value.get('credential_ref') or {})
if set(credential_ref) != {'source', 'auth_entry'}:
raise ScanExecutionError('remote execution snapshot credential reference is invalid')
queue_source = str(credential_ref.get('source') or '').strip().lower()
auth_entry = str(credential_ref.get('auth_entry') or '')
if len(auth_entry) > 128 or '\x00' in auth_entry:
raise ScanExecutionError('remote execution snapshot credential reference is invalid')
return compatibility, normalized_execution, planning, queue_source, auth_entry
def _normalize_exact_git_v1_planning(planning):
planning = dict(planning or {})
if set(planning) != {
'kind', 'git_baseline_depth', 'git_ref_resolution_attempts',
'git_ref_resolution_timeout_sec', 'git_ref_resolution_max_bytes',
} or planning.get('kind') != 'exact_git_v1':
raise ScanExecutionError('remote execution snapshot planning is invalid')
try:
baseline_depth = int(planning['git_baseline_depth'])
attempts = int(planning['git_ref_resolution_attempts'])
timeout = float(planning['git_ref_resolution_timeout_sec'])
max_bytes = int(planning['git_ref_resolution_max_bytes'])
except (TypeError, ValueError, OverflowError) as exc:
raise ScanExecutionError('remote execution snapshot planning is invalid') from exc
if (
isinstance(planning['git_baseline_depth'], bool)
or isinstance(planning['git_ref_resolution_attempts'], bool)
or isinstance(planning['git_ref_resolution_timeout_sec'], bool)
or isinstance(planning['git_ref_resolution_max_bytes'], bool)
or not 1 <= baseline_depth <= 1000000
or not 1 <= attempts <= 20
or not 0.1 <= timeout <= 300
or not 1024 <= max_bytes <= 64 * 1024 * 1024
):
raise ScanExecutionError('remote execution snapshot planning is outside its bounds')
return {
'kind': 'exact_git_v1',
'git_baseline_depth': baseline_depth,
'git_ref_resolution_attempts': attempts,
'git_ref_resolution_timeout_sec': timeout,
'git_ref_resolution_max_bytes': max_bytes,
}
def _normalize_kind_only_planning(planning, kind):
planning = dict(planning or {})
if planning != {'kind': kind}:
raise ScanExecutionError('remote execution snapshot planning is invalid')
return {'kind': kind}
def _normalized_remote_snapshot(
value, *, queue_sources, worker_platform, planning_kind,
planning_normalizer, public_credential,
):
compatibility, execution, planning, queue_source, auth_entry = (
_remote_snapshot_envelope(value)
)
platform = execution['source']
if (
queue_source not in queue_sources
or (worker_platform is None and platform != queue_source)
or (worker_platform is not None and platform != worker_platform)
or (public_credential and auth_entry)
):
raise ScanExecutionError('remote execution snapshot source capability is invalid')
normalized_planning = planning_normalizer(planning)
if normalized_planning.get('kind') != planning_kind:
raise ScanExecutionError('remote execution snapshot planning kind is invalid')
normalized = {
'schema': REMOTE_EXECUTION_SNAPSHOT_SCHEMA,
'compatibility': compatibility.as_dict(),
'execution': execution,
'planning': normalized_planning,
'credential_ref': {'source': queue_source, 'auth_entry': auth_entry},
}
encoded = json.dumps(
normalized, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
allow_nan=False,
).encode('utf-8')
if len(encoded) > MAX_REMOTE_EXECUTION_SNAPSHOT_BYTES:
raise ScanExecutionError('remote execution snapshot exceeds its byte bound')
return normalized
def normalize_exact_git_execution_snapshot(value):
return _normalized_remote_snapshot(
value,
queue_sources=frozenset(('github', 'gitlab')),
worker_platform=None,
planning_kind='exact_git_v1',
planning_normalizer=_normalize_exact_git_v1_planning,
public_credential=False,
)
def normalize_docker_direct_execution_snapshot(value):
return _normalized_remote_snapshot(
value,
queue_sources=frozenset(('dockerhub',)),
worker_platform='docker',
planning_kind='docker_direct_v1',
planning_normalizer=lambda planning: _normalize_kind_only_planning(
planning, 'docker_direct_v1',
),
public_credential=True,
)
def normalize_huggingface_space_execution_snapshot(value):
return _normalized_remote_snapshot(
value,
queue_sources=frozenset(('huggingface',)),
worker_platform='huggingface',
planning_kind='huggingface_space_v1',
planning_normalizer=lambda planning: _normalize_kind_only_planning(
planning, 'huggingface_space_v1',
),
public_credential=True,
)
def normalize_remote_execution_snapshot(value):
if not isinstance(value, dict) or not isinstance(value.get('planning'), dict):
raise ScanExecutionError('remote execution snapshot planning is invalid')
kind = value['planning'].get('kind')
normalizer = {
'exact_git_v1': normalize_exact_git_execution_snapshot,
'docker_direct_v1': normalize_docker_direct_execution_snapshot,
'huggingface_space_v1': normalize_huggingface_space_execution_snapshot,
}.get(kind)
if normalizer is None:
raise ScanExecutionError('remote execution snapshot planning kind is unsupported')
return normalizer(value)
def normalize_docker_direct_execution_target(value):
try:
return parse_dockerhub_digest_target(value)
except (TypeError, ValueError) as exc:
raise ScanExecutionError('Docker direct target is invalid') from exc
def normalize_huggingface_space_execution_target(value):
try:
return normalize_huggingface_space_id(value)
except (TypeError, ValueError) as exc:
raise ScanExecutionError('HuggingFace Space target is invalid') from exc
def remote_execution_snapshot_sha256(value):
normalized = normalize_remote_execution_snapshot(value)
encoded = json.dumps(
normalized, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
allow_nan=False,
).encode('utf-8')
return hashlib.sha256(encoded).hexdigest()
def _normalize_remote_assignment_deadlines(value, reservation, scan_kwargs):
if not isinstance(value, dict) or set(value) != {
'target_scan_timeout_seconds', 'result_upload_body_timeout_seconds',
'assignment_ttl_seconds', 'assignment_issued_at',
'assignment_deadline_at',
}:
raise ScanExecutionError('worker assignment deadlines shape is invalid')
deadlines = dict(value)
for name in (
'target_scan_timeout_seconds', 'result_upload_body_timeout_seconds',
'assignment_ttl_seconds',
):
if type(deadlines[name]) is not int or deadlines[name] <= 0:
raise ScanExecutionError('worker assignment deadline value is invalid')
issued_at = deadlines['assignment_issued_at']
deadline_at = deadlines['assignment_deadline_at']
if (
type(issued_at) is not str
or type(deadline_at) is not str
or issued_at != reservation.get('remote_issued_at')
or deadline_at != reservation.get('remote_expires_at')
):
raise ScanExecutionError('worker assignment deadline changed after reservation')
try:
issued = datetime.fromisoformat(issued_at)
deadline = datetime.fromisoformat(deadline_at)
except ValueError as exc:
raise ScanExecutionError('worker assignment deadline timestamp is invalid') from exc
if (
issued.tzinfo is None
or deadline.tzinfo is None
or issued.utcoffset() != timedelta(0)
or deadline.utcoffset() != timedelta(0)
or issued.isoformat(timespec='seconds') != issued_at
or deadline.isoformat(timespec='seconds') != deadline_at
or deadline - issued != timedelta(seconds=deadlines['assignment_ttl_seconds'])
):
raise ScanExecutionError('worker assignment deadline timestamp is invalid')
if deadlines['target_scan_timeout_seconds'] != scan_kwargs.get('timeout_sec'):
raise ScanExecutionError('worker assignment target scan timeout changed')
return deadlines
def _validate_remote_assignment(
assignment, local_build, expected_protocol_version,
package_capabilities=None,
):
if not isinstance(assignment, dict) or set(assignment) != {
'reservation', 'deadlines', 'compatibility', 'scan_kwargs', 'event_scan_options',
'queue_policy', 'limits', 'scan_policy', 'execution_snapshot',
'execution_snapshot_sha256', 'execution_plan',
}:
raise ScanExecutionError('worker assignment shape is invalid')
reservation_value = dict(assignment.get('reservation') or {})
try:
reservation = BundleReservation.from_mapping(reservation_value)
except (KeyError, TypeError, ValueError) as exc:
raise ScanExecutionError('worker assignment reservation is invalid') from exc
source = str(reservation.source or '').strip().lower()
platform = str(reservation.platform or '').strip().lower()
if (
reservation_value.get('assignment_kind') != 'remote'
or not source or not platform
):
raise ScanExecutionError('worker assignment reservation is not remote')
snapshot = normalize_remote_execution_snapshot(
assignment.get('execution_snapshot'),
)
snapshot_sha256 = _digest(
assignment.get('execution_snapshot_sha256'), 'execution snapshot',
)
if not hmac.compare_digest(
snapshot_sha256, remote_execution_snapshot_sha256(snapshot),
):
raise ScanExecutionError('worker assignment execution snapshot hash changed')
if snapshot['compatibility']['protocol_version'] != expected_protocol_version:
raise ScanExecutionError('worker assignment snapshot protocol is incompatible')
required = ScanCompatibility.from_mapping(assignment.get('compatibility'))
if required.as_dict() != snapshot['compatibility']:
raise ScanExecutionError('worker assignment compatibility changed after admission')
validate_remote_assignment_compatibility(
required, local_build, platform, assignment.get('scan_kwargs'),
assignment.get('event_scan_options'), assignment.get('queue_policy'),
assignment.get('limits'), assignment.get('scan_policy'),
expected_protocol_version=expected_protocol_version,
)
effective, execution = remote_execution_identity(
platform, assignment.get('scan_kwargs'),
assignment.get('event_scan_options'), assignment.get('queue_policy'),
assignment.get('limits'), assignment.get('scan_policy'),
)
if execution != snapshot['execution']:
raise ScanExecutionError('worker assignment settings changed after admission')
if (
snapshot['credential_ref']['source'] != source
or snapshot['execution']['source'] != platform
or not hmac.compare_digest(effective, required.effective_config_sha256)
or not hmac.compare_digest(
str(reservation_value.get('remote_effective_config_sha256') or ''),
required.effective_config_sha256,
)
):
raise ScanExecutionError('worker assignment identity changed after admission')
planning_kind = snapshot['planning']['kind']
if expected_protocol_version == 1 and planning_kind != 'exact_git_v1':
raise ScanExecutionError('legacy worker assignment planning kind is invalid')
capability = (source, platform, planning_kind)
if package_capabilities is not None:
capabilities = {
tuple(value) for value in package_capabilities
if isinstance(value, (list, tuple)) and len(value) == 3
}
if capability not in capabilities:
raise ScanExecutionError(
'worker assignment capability is not supported by this package'
)
plan = assignment.get('execution_plan')
if not isinstance(plan, dict) or set(plan) != {
'kind', 'execution_target', 'bound_plan',
} or plan.get('kind') != planning_kind:
raise ScanExecutionError('worker assignment execution plan is invalid')
scan_kwargs = dict(assignment.get('scan_kwargs') or {})
event_scan_options = dict(assignment.get('event_scan_options') or {})
deadlines = _normalize_remote_assignment_deadlines(
assignment.get('deadlines'), reservation_value, scan_kwargs,
)
if planning_kind == 'exact_git_v1':
if (
source not in {'github', 'gitlab'} or platform != source
or str(plan.get('execution_target') or '') != reservation.target
or not isinstance(plan.get('bound_plan'), dict)
or scan_kwargs.get('git_plan') != plan['bound_plan']
or scan_kwargs.get('docker_layer_work') is not None
):
raise ScanExecutionError('worker assignment exact Git plan is invalid')
execution_target = reservation.target
elif planning_kind == 'docker_direct_v1':
if source != 'dockerhub' or platform != 'docker':
raise ScanExecutionError('worker assignment Docker capability is invalid')
parsed = normalize_docker_direct_execution_target(reservation.target)
execution_target = parsed['image']
if parsed['normalized_target'] != reservation.normalized_target:
raise ScanExecutionError('worker assignment Docker identity is invalid')
elif planning_kind == 'huggingface_space_v1':
if source != 'huggingface' or platform != 'huggingface':
raise ScanExecutionError('worker assignment HuggingFace capability is invalid')
execution_target = normalize_huggingface_space_execution_target(
reservation.target,
)
if normalize_target(execution_target, platform) != reservation.normalized_target:
raise ScanExecutionError('worker assignment HuggingFace identity is invalid')
else:
raise ScanExecutionError('worker assignment planning kind is unsupported')
if planning_kind != 'exact_git_v1' and (
plan.get('bound_plan') is not None
or str(plan.get('execution_target') or '') != execution_target
or scan_kwargs != event_scan_options
or any(name in scan_kwargs for name in (
'token', 'git_plan', 'docker_layer_work',
))
):
raise ScanExecutionError('worker assignment direct plan is not credential-free')
return {
'reservation': reservation,
'snapshot': snapshot,
'snapshot_sha256': snapshot_sha256,
'compatibility': required,
'planning_kind': planning_kind,
'execution_target': execution_target,
'deadlines': deadlines,
'execution_plan': {
'kind': planning_kind,
'execution_target': execution_target,
'bound_plan': plan.get('bound_plan'),
},
}
def validate_protocol1_remote_assignment(assignment, local_build):
return _validate_remote_assignment(assignment, local_build, 1)
def validate_protocol2_remote_assignment(
assignment, local_build, package_capabilities=None,
):
return _validate_remote_assignment(
assignment, local_build, PROTOCOL_VERSION, package_capabilities,
)
def _first_error_line(result):
for error in result.get('errors') or ():
for line in str(error).splitlines():
line = line.strip()
if not line:
continue
try:
payload = json.loads(line)
except (TypeError, ValueError):
return line[:300]
return str(payload.get('error') or payload.get('msg') or line)[:300]
return ''
def _docker_result_resets_attempts(result):
if result.get('docker_layer_plan') is None or not result.get('retryable', False):
return False
execution = result.get('docker_layer_execution')
records = execution.get('blobs') if isinstance(execution, dict) else None
descriptors = result['docker_layer_plan'].get('descriptors')
if not isinstance(records, list) or not isinstance(descriptors, list):
return False
if not any(
item.get('coverage_state') in ('selected', 'shared_pending')
for item in descriptors if isinstance(item, dict)
):
return False
return not any(
item.get('status') in ('retryable_failed', 'terminal_failed')
for item in records if isinstance(item, dict)
)
def queue_disposition_for_result(result, platform, attempts, policy, *, now=None):
policy = policy if isinstance(policy, QueueDispositionPolicy) else QueueDispositionPolicy(**policy)
attempts = max(0, int(attempts or 0))
max_attempts = max(1, int(policy.target_retry_max_attempts or 1))
now = now or datetime.now(timezone.utc)
skipped = str(result.get('skipped') or '')
if skipped in set(policy.soft_skip_reasons):
return {
'queue_status': 'deferred', 'queue_error': skipped,
'available_after': (now + timedelta(days=max(1, policy.ci_soft_cooldown_days))).isoformat(timespec='seconds'),
'reset_attempts': True,
}
if result.get('docker_layer_plan') is not None:
if not result.get('errors'):
status, available_after, reset = 'done', None, False
elif not bool(result.get('retryable', False)):
status, available_after, reset = 'failed', None, False
else:
reset = _docker_result_resets_attempts(result)
if not reset and attempts >= max_attempts:
status, available_after = 'failed', None
else:
status = 'deferred'
available_after = (now + timedelta(seconds=max(
1, int(policy.docker_layer_checkpoint_delay_sec or 1),
))).isoformat(timespec='seconds')
return {
'queue_status': status,
'queue_error': _first_error_line(result) if result.get('errors') else None,
'available_after': available_after, 'reset_attempts': reset,
}
if not result.get('errors'):
return {
'queue_status': 'done', 'queue_error': None,
'available_after': None, 'reset_attempts': False,
}
timed_out = bool((result.get('scan_meta') or {}).get('command_timed_out')) \
or result.get('error_class') == 'timeout'
if timed_out:
status = 'failed' if attempts >= max_attempts else 'deferred'
delay = max(60, int(policy.target_timeout_retry_delay_sec or 60))
elif result.get('source_failure'):
status = 'failed' if not result.get('retryable', True) and attempts >= max_attempts else 'deferred'
delay = max(1, int(policy.target_retry_max_delay_sec or 1))
elif not result.get('retryable', True) or attempts >= max_attempts:
status, delay = 'failed', 0
else:
status = 'deferred'
base = max(1, int(policy.target_retry_base_delay_sec or 1))
maximum = max(base, int(policy.target_retry_max_delay_sec or base))
delay = min(maximum, base * (2 ** max(0, attempts - 1)))
return {
'queue_status': status,
'queue_error': _first_error_line(result),
'available_after': (
(now + timedelta(seconds=delay)).isoformat(timespec='seconds')
if status == 'deferred' else None
),
'reset_attempts': bool(
(result.get('source_failure') and result.get('retryable', True))
or _docker_result_resets_attempts(result)
),
}
def stage_scan_result_in_scope(
result, reservation, bundle_root, event_scan_options, queue_policy, *, attempts,
candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024,
require_s_drive=False, fault=None, diagnostic_slot_id=0,
):
reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation)
disposition = queue_disposition_for_result(
result, reservation.platform, attempts, queue_policy,
)
return stage_result_bundle(
result, reservation, bundle_root, event_scan_options, disposition,
candidate_max_items=candidate_max_items,
candidate_max_bytes=candidate_max_bytes,
require_s_drive=require_s_drive, fault=fault,
diagnostic_slot_id=diagnostic_slot_id,
diagnostic_attempt=max(1, int(attempts or 1)),
)
def execute_planned_result_in_scope(
reservation, bundle_root, scan_kwargs, event_scan_options, queue_policy, *,
attempts, candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024,
execution_target=None, scan_meta_defaults=None, require_s_drive=False,
phase_callback=None, bundle_fault=None, diagnostic_slot_id=0,
):
reservation = reservation if isinstance(reservation, BundleReservation) else BundleReservation.from_mapping(reservation)
scan_kwargs = validate_scan_kwargs(reservation.platform, scan_kwargs)
target = reservation.target if execution_target is None else execution_target
expected = reservation.normalized_target or normalize_target(reservation.target, reservation.platform)
if normalize_target(target, reservation.platform) != expected:
raise ScanExecutionError('execution target does not match the reservation')
with client_scan_phase_events(phase_callback):
result = scan_target_result(
target, reservation.platform, reservation.scan_event_id, scan_kwargs,
)
result['target'] = reservation.target
result['scan_type'] = reservation.platform
if scan_meta_defaults:
metadata = result.setdefault('scan_meta', {})
if not isinstance(metadata, dict):
raise ScanExecutionError('scanner metadata is invalid')
for name, value in dict(scan_meta_defaults).items():
metadata.setdefault(name, value)
if phase_callback is not None:
phase_callback('cleaning')
cleanup = cleanup_assignment_work_dir()
phase_callback('cleaning', cleanup)
phase_callback('bundling')
return stage_scan_result_in_scope(
result, reservation, bundle_root, event_scan_options, queue_policy,
attempts=attempts, candidate_max_items=candidate_max_items,
candidate_max_bytes=candidate_max_bytes, require_s_drive=require_s_drive,
fault=bundle_fault, diagnostic_slot_id=diagnostic_slot_id,
)
def execute_planned_claim(
reservation, bundle_root, scan_kwargs, event_scan_options, queue_policy, scan_policy, *,
attempts, candidate_max_items=2000, candidate_max_bytes=2 * 1024 * 1024,
lease=None, execution_target=None, scan_meta_defaults=None,
require_s_drive=False, phase_callback=None, bundle_fault=None,
diagnostic_slot_id=0,
):
timeout = validate_scan_kwargs(
reservation.platform if isinstance(reservation, BundleReservation) else reservation.get('platform'),
scan_kwargs,
)['timeout_sec']
platform = reservation.platform if isinstance(reservation, BundleReservation) else reservation.get('platform')
policy = normalize_remote_scan_policy(scan_policy)
if phase_callback is not None:
phase_callback('waiting_permit', {'boundary': 'scan_slot_scope'})
with client_scan_execution_policy(policy):
with scan_slot_scope(['scan-target', platform], timeout, lease=lease):
return execute_planned_result_in_scope(
reservation, bundle_root, scan_kwargs, event_scan_options, queue_policy,
attempts=attempts, candidate_max_items=candidate_max_items,
candidate_max_bytes=candidate_max_bytes, execution_target=execution_target,
scan_meta_defaults=scan_meta_defaults, require_s_drive=require_s_drive,
phase_callback=phase_callback, bundle_fault=bundle_fault,
diagnostic_slot_id=diagnostic_slot_id,
)
def execute_protocol2_remote_claim(
validated_assignment, bundle_root, scan_kwargs, event_scan_options,
queue_policy, scan_policy, *, attempts, candidate_max_items=2000,
candidate_max_bytes=2 * 1024 * 1024, lease=None,
scan_meta_defaults=None, require_s_drive=False, phase_callback=None,
bundle_fault=None, diagnostic_slot_id=0,
):
kind = str(validated_assignment.get('planning_kind') or '')
authority = (
client_remote_execution_binding(kind)
if kind in {'docker_direct_v1', 'huggingface_space_v1'}
else nullcontext()
)
with authority:
return execute_planned_claim(
validated_assignment['reservation'], bundle_root, scan_kwargs,
event_scan_options, queue_policy, scan_policy,
attempts=attempts,
candidate_max_items=candidate_max_items,
candidate_max_bytes=candidate_max_bytes,
lease=lease,
execution_target=validated_assignment['execution_target'],
scan_meta_defaults={
**dict(scan_meta_defaults or {}), 'planning_kind': kind,
},
require_s_drive=require_s_drive,
phase_callback=phase_callback, bundle_fault=bundle_fault,
diagnostic_slot_id=diagnostic_slot_id,
)
def canonical_json_sha256(value):
return hashlib.sha256(json.dumps(
value, ensure_ascii=True, sort_keys=True, separators=(',', ':'),
).encode('utf-8')).hexdigest()
+71
View File
@@ -0,0 +1,71 @@
import json
import threading
RETIRED_MESSAGE = 'Legacy ScanManager mutation controls are retired; use the authenticated supervisor.'
class ScanManager:
"""Read-only compatibility shell for the retired legacy scanner UI."""
def __init__(self):
self.current_scan = None
self.scan_thread = None
self.results = []
self.progress = {
'total': 0,
'completed': 0,
'failed': 0,
'secrets_found': 0,
'current_target': None,
'status': 'retired',
}
self.scan_lock = threading.Lock()
self.scanned_targets = set()
self.scan_history = []
def start_scan(self, scan_type, targets, scan_options):
return False, RETIRED_MESSAGE
def pause_scan(self):
return False
def resume_scan(self):
return False
def stop_scan(self):
return False
def is_scanning(self):
return False
def get_progress(self):
with self.scan_lock:
return self.progress.copy()
def get_results(self):
with self.scan_lock:
return self.results.copy()
def clear_results(self):
return False
def get_scan_info(self):
with self.scan_lock:
return self.current_scan.copy() if self.current_scan else None
def export_results(self, format='json'):
with self.scan_lock:
if str(format).lower() == 'json':
return json.dumps(self.results, indent=2, default=str)
return None
def clear_scan_history(self):
return False
def get_scan_history(self):
with self.scan_lock:
return self.scan_history.copy()
def get_scanned_targets_count(self):
return len(self.scanned_targets)
+17045
View File
File diff suppressed because it is too large Load Diff
+30695
View File
File diff suppressed because it is too large Load Diff
+423
View File
@@ -0,0 +1,423 @@
import sys
sys.dont_write_bytecode = True
import json
import gzip
import os
import tarfile
import tempfile
import uuid
import zipfile
from datetime import datetime, timedelta, timezone
from types import SimpleNamespace
import scanner as scanner_module
from console_runner import queue_error_disposition, target_retry_delay_sec
from db_backend import redact_database_url
from scanner import (
apply_trufflehog_diagnostics,
build_authenticated_git_url,
cached_gharchive_hour,
convert_package_git_unavailable_to_skip,
docker_tag_platform_support,
fetch_dockerhub_images,
normalize_git_repo_candidate,
redact_command_args,
run_command,
safe_extract_package_zip,
safe_extract_tar,
)
from scanner_db import ScannerDB, normalize_target, target_status
from runtime_security import ensure_private_directory, harden_private_file
def diagnostic(message, error, **extra):
return json.dumps({'level': 'error', 'msg': message, 'error': error, **extra})
def assert_diagnostic_policy():
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(
result,
diagnostic('non-critical error processing chunk', 'error reading chunk: brotli: PADDING_2'),
0,
'pypi',
)
assert not result.get('errors')
assert result.get('warnings') and result.get('degraded')
assert target_status(result) == 'degraded'
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(result, diagnostic('Skipping result: invalid', 'empty raw'), 0, 'npm')
assert not result.get('errors') and result.get('warnings')
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(
result,
diagnostic('error reading chunk', 'read tcp: connection reset by peer'),
0,
'pypi',
)
assert result.get('errors') and result.get('retryable') is True
assert result.get('error_class') == 'network'
for detail in ('unexpected EOF', 'permission denied', 'no space left on device'):
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(result, diagnostic('error reading chunk', detail), 0, 'pypi')
assert result.get('errors'), detail
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(result, diagnostic('error reading chunk', 'brotli: PADDING_2'), 0, 'git')
assert result.get('errors')
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(result, diagnostic('space', 'no repo found for repo'), 0, 'huggingface')
assert result.get('skipped') and not result.get('errors')
assert target_status(result) == 'skipped'
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(
result,
diagnostic('error processing image', 'no child with platform linux/amd64 in index image:tag'),
1,
'docker',
)
assert result.get('skipped') and not result.get('errors')
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(
result,
diagnostic('error processing layer', 'gzip: invalid header') + '\n' + json.dumps({'level': 'info-0', 'msg': 'finished scanning'}),
0,
'docker',
)
assert result.get('degraded') and not result.get('errors')
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(
result,
diagnostic('a detector ignored the context timeout', 'context deadline exceeded') + '\n' + json.dumps({'level': 'info-0', 'msg': 'finished scanning'}),
0,
'docker',
)
assert result.get('degraded') and not result.get('errors')
assert result.get('warning_classes') == ['detector_timeout']
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(
result,
diagnostic('error processing layer', 'gzip: invalid header'),
1,
'docker',
)
assert result.get('errors') and not result.get('degraded')
result = {'findings': [], 'errors': []}
apply_trufflehog_diagnostics(result, '', 2, 'filesystem')
assert result.get('errors') and result.get('retryable') is True
result['findings'] = [{'DetectorName': 'Example'}]
assert target_status(result) == 'error'
result = {'errors': [diagnostic('error running scan', 'remote: Repository not found.')], 'findings': []}
convert_package_git_unavailable_to_skip(result)
assert result.get('skipped') and not result.get('errors')
def assert_docker_platform_policy():
assert docker_tag_platform_support({'images': [{'os': 'linux', 'architecture': 'amd64'}]}) is True
assert docker_tag_platform_support({'images': [{'os': 'linux', 'architecture': 'arm64'}]}) is False
assert docker_tag_platform_support({'images': []}) is None
assert docker_tag_platform_support({'images': [{'os': 'linux'}]}) is None
assert docker_tag_platform_support({'images': [{'os': 'unknown', 'architecture': 'unknown'}]}) is None
assert docker_tag_platform_support({'images': [{'os': 'linux', 'architecture': 'arm64'}, {'os': 'unknown', 'architecture': 'unknown'}]}) is None
assert docker_tag_platform_support({}) is None
assert normalize_target('Owner/Repo:Prod', 'docker') == 'owner/repo:prod'
assert normalize_git_repo_candidate('https://[invalid url, do not cite]/repo') is None
def assert_docker_partial_pagination_policy():
class Response:
def __init__(self, page):
self.page = page
def raise_for_status(self):
if self.page > 1:
raise RuntimeError('page outside result set')
def json(self):
return {'count': 1, 'results': [{'repo_name': 'owner/repo'}]}
original = scanner_module.api_request
try:
def fake_request(method, url, **kwargs):
page = int(url.split('page=', 1)[1].split('&', 1)[0])
return Response(page)
scanner_module.api_request = fake_request
assert fetch_dockerhub_images('test', 3, per_page=10, resolve_tags=False) == ['owner/repo']
finally:
scanner_module.api_request = original
def assert_retry_policy():
assert target_retry_delay_sec(1, 60, 3600) == 60
assert target_retry_delay_sec(2, 60, 3600) == 120
assert target_retry_delay_sec(10, 60, 3600) == 3600
args = SimpleNamespace(target_retry_max_attempts=3, target_timeout_retry_delay_sec=21600)
status, available_after, attempts, max_attempts = queue_error_disposition(
None, 'github', 'github', 'https://github.com/o/r',
{'errors': ['timeout'], 'error_class': 'timeout', 'scan_meta': {'command_timed_out': True}},
args, {'attempts': 3},
)
assert status == 'failed' and available_after is None and attempts == 3 and max_attempts == 3
def assert_timeout_output_policy():
marker = '{"DetectorName":"PartialFinding"}'
old_work_dir = scanner_module.scan_config.work_dir
old_min_free = scanner_module.scan_config.min_free_gb
old_authority_check = scanner_module.require_trufflehog_launch_authority
with tempfile.TemporaryDirectory() as temp_dir:
work_dir = os.path.join(temp_dir, 'work')
ensure_private_directory(work_dir, reject_reparse=True)
scanner_module.initialize_scanner_runtime(preflight_complete=True, register_cleanup=False)
scanner_module.scan_config.work_dir = work_dir
scanner_module.scan_config.min_free_gb = 0
scanner_module.require_trufflehog_launch_authority = lambda _command: None
try:
stdout, stderr, returncode = run_command(
[sys.executable, '-c', f'import time; print({marker!r}, flush=True); time.sleep(5)'],
1,
)
finally:
scanner_module.scan_config.work_dir = old_work_dir
scanner_module.scan_config.min_free_gb = old_min_free
scanner_module.require_trufflehog_launch_authority = old_authority_check
assert returncode == -1 and marker in stdout and 'timed out' in stderr.lower()
def assert_archive_and_secret_safety():
url, secrets = build_authenticated_git_url('https://attacker.example/repo.git', 'github', 'sentinel-token')
assert url == 'https://attacker.example/repo.git' and not secrets and 'sentinel-token' not in url
redacted = redact_command_args(['git', 'https://x-access-token:sentinel-token@github.com/org/repo.git', '--token', 'sentinel-token'])
assert all('sentinel-token' not in value for value in redacted)
with tempfile.TemporaryDirectory() as temp_dir:
tar_path = os.path.join(temp_dir, 'unsafe.tar')
with tarfile.open(tar_path, 'w') as archive:
link = tarfile.TarInfo('link')
link.type = tarfile.SYMTYPE
link.linkname = '..'
archive.addfile(link)
try:
safe_extract_tar(tar_path, os.path.join(temp_dir, 'tar-out'))
raise AssertionError('unsafe tar link was accepted')
except ValueError:
pass
zip_path = os.path.join(temp_dir, 'large.zip')
with zipfile.ZipFile(zip_path, 'w') as archive:
archive.writestr('large.txt', b'x' * 2048)
try:
safe_extract_package_zip(
zip_path, os.path.join(temp_dir, 'zip-out'),
max_total_size_mb=0, max_file_size_mb=0,
)
except ValueError:
raise AssertionError('disabled package zip bounds rejected a valid archive')
crowded_tar = os.path.join(temp_dir, 'crowded.tar')
with tarfile.open(crowded_tar, 'w') as archive:
archive.addfile(tarfile.TarInfo('one'))
archive.addfile(tarfile.TarInfo('two'))
try:
safe_extract_tar(crowded_tar, os.path.join(temp_dir, 'crowded-out'), max_files=1)
raise AssertionError('tar member bound was not enforced')
except ValueError:
pass
assert 'query-secret' not in redact_database_url('postgresql://u@localhost/db?password=query-secret')
malformed = ScannerDB(db_path=os.path.join(tempfile.gettempdir(), 'must-not-open.db'), db_url='postgreql://bad')
assert malformed.postgres_required and not malformed.enabled and malformed.path is None
with tempfile.TemporaryDirectory() as temp_dir:
runtime_dir = os.path.join(temp_dir, 'runtime')
cache_dir = os.path.join(runtime_dir, 'gharchive')
ensure_private_directory(runtime_dir)
ensure_private_directory(cache_dir)
hour = datetime(2026, 7, 11, 12, tzinfo=timezone.utc)
cache_path = os.path.join(cache_dir, '2026-07-11-12.json.gz')
with gzip.open(cache_path, 'wb') as archive:
archive.write(b'{"type":"PushEvent"}\n')
harden_private_file(cache_path)
old_runtime = scanner_module.scan_config.runtime_dir
old_free = scanner_module.scan_config.gharchive_cache_min_free_bytes
try:
scanner_module.scan_config.runtime_dir = runtime_dir
scanner_module.scan_config.gharchive_cache_min_free_bytes = 0
assert scanner_module.canonical_path(
cached_gharchive_hour(hour, cache_dir, request_timeout=1, retries=1)
) == scanner_module.canonical_path(cache_path)
finally:
scanner_module.scan_config.runtime_dir = old_runtime
scanner_module.scan_config.gharchive_cache_min_free_bytes = old_free
def assert_queue_policy():
old_urls = {key: os.environ.pop(key, None) for key in ('SCANNER_DB_URL', 'DATABASE_URL')}
try:
with tempfile.TemporaryDirectory() as temp_dir:
db = ScannerDB(db_path=os.path.join(temp_dir, 'scanner.db'))
try:
assert db.enabled
target = 'owner/repo:tag'
db.enqueue_targets('dockerhub', 'docker', 'q', [target])
claimed = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 3600, return_rows=True)
assert [row['target'] for row in claimed] == [target]
first_claim = claimed[0]
row = db.target_queue_item('dockerhub', 'docker', target)
assert row['status'] == 'in_progress' and int(row['attempts']) == 1
assert db.reclaim_target_leases('dockerhub', 'docker', 'owner-b') == 0
future = (datetime.now(timezone.utc) + timedelta(hours=1)).isoformat(timespec='seconds')
db.complete_target_queue_item(
'dockerhub', 'docker', target, None, 'deferred', 'network', future,
queue_id=first_claim['id'], lease_token=first_claim['lease_token'],
)
db.sync_target_queue_from_files('dockerhub', 'docker', [target], [], 'q')
row = db.target_queue_item('dockerhub', 'docker', target)
assert row['status'] == 'deferred' and row['available_after'] == future
updated_at = db.conn.execute(
'SELECT updated_at FROM target_queue WHERE source = ? AND normalized_target = ?',
('dockerhub', normalize_target(target, 'docker')),
).fetchone()['updated_at']
db.sync_target_queue_from_files('dockerhub', 'docker', [target], [], 'q')
assert db.conn.execute(
'SELECT updated_at FROM target_queue WHERE source = ? AND normalized_target = ?',
('dockerhub', normalize_target(target, 'docker')),
).fetchone()['updated_at'] == updated_at
assert db.claim_targets('dockerhub', 'docker', 1, 'owner-b', 3600) == []
db.conn.execute(
"UPDATE target_queue SET available_after = ? WHERE source = ?",
('2000-01-01T00:00:00+00:00', 'dockerhub'),
)
db.conn.commit()
reclaimed = db.claim_targets('dockerhub', 'docker', 1, 'owner-b', 3600, return_rows=True)
assert [item['target'] for item in reclaimed] == [target]
db.complete_target_queue_item(
'dockerhub', 'docker', target, None, 'failed', 'permanent',
queue_id=reclaimed[0]['id'], lease_token=reclaimed[0]['lease_token'],
)
db.sync_target_queue_from_files('dockerhub', 'docker', [], [target], 'q')
row = db.target_queue_item('dockerhub', 'docker', target)
assert row['status'] == 'failed' and row['available_after'] is None
done_target = 'owner/other:tag'
db.enqueue_targets('dockerhub', 'docker', 'q', [done_target])
done_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 3600, return_rows=True)[0]
db.complete_target_queue_item(
'dockerhub', 'docker', done_target, None, 'done',
queue_id=done_claim['id'], lease_token=done_claim['lease_token'],
)
db.enqueue_targets('dockerhub', 'docker', 'q', [done_target], requeue_done=True)
row = db.target_queue_item('dockerhub', 'docker', done_target)
assert row['status'] == 'pending' and int(row['attempts']) == 0
done_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 3600, return_rows=True)[0]
db.complete_target_queue_item(
'dockerhub', 'docker', done_target, None, 'done',
queue_id=done_claim['id'], lease_token=done_claim['lease_token'],
)
legacy_target = 'owner/legacy:tag'
db.enqueue_targets('dockerhub', 'docker', 'q', [legacy_target])
db.conn.execute(
"UPDATE target_queue SET status = 'pending', available_after = ? WHERE normalized_target = ?",
(future, normalize_target(legacy_target, 'docker')),
)
db.conn.commit()
db.sync_target_queue_from_files('dockerhub', 'docker', [legacy_target], [], 'q')
row = db.target_queue_item('dockerhub', 'docker', legacy_target)
assert row['status'] == 'deferred' and row['available_after'] == future
stale_target = 'owner/stale:tag'
db.enqueue_targets('dockerhub', 'docker', 'q', [stale_target])
stale_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-a', 60, max_attempts=3, return_rows=True)[0]
assert stale_claim['target'] == stale_target
db.conn.execute(
"UPDATE target_queue SET lease_expires_at = ? WHERE normalized_target = ?",
('2000-01-01T00:00:00+00:00', normalize_target(stale_target, 'docker')),
)
db.conn.commit()
newer_claim = db.claim_targets('dockerhub', 'docker', 1, 'owner-b', 60, max_attempts=3, return_rows=True)[0]
assert newer_claim['target'] == stale_target
assert newer_claim['lease_token'] != stale_claim['lease_token']
assert not db.complete_target_queue_item(
'dockerhub', 'docker', stale_target, None, 'done',
queue_id=stale_claim['id'], lease_token=stale_claim['lease_token'],
)
row = db.target_queue_item('dockerhub', 'docker', stale_target)
assert row['status'] == 'in_progress' and row['lease_owner'] == 'owner-b'
run_id = db.start_run('smoke', ['smoke'])
cycle_id = db.start_source_cycle(run_id, 'dockerhub', 'docker', 'search', 'q', 1, 1, None, {}, {})
result = {
'target': stale_target, 'scan_type': 'docker', 'findings': [], 'errors': [],
'scan_event_id': str(uuid.uuid4()),
'timestamp': datetime.now(timezone.utc).isoformat(timespec='seconds'),
}
before = db.conn.execute('SELECT COUNT(*) AS count FROM target_scans').fetchone()['count']
stale = db.record_and_complete_target_result(
run_id, cycle_id, 'dockerhub', 'q', stale_target, result, {},
row['id'], 'owner-a', 'done',
lease_token=stale_claim['lease_token'],
)
assert stale and stale['stale']
assert db.conn.execute('SELECT COUNT(*) AS count FROM target_scans').fetchone()['count'] == before + 1
row = db.target_queue_item('dockerhub', 'docker', stale_target)
assert row['status'] == 'in_progress' and row['lease_token'] == newer_claim['lease_token']
result = dict(result, scan_event_id=str(uuid.uuid4()))
owned = db.record_and_complete_target_result(
run_id, cycle_id, 'dockerhub', 'q', stale_target, result, {},
row['id'], 'owner-b', 'done',
lease_token=newer_claim['lease_token'],
)
assert owned and not owned['stale']
assert db.conn.execute('SELECT COUNT(*) AS count FROM target_scans').fetchone()['count'] == before + 2
exhausted_target = 'owner/exhausted:tag'
db.enqueue_targets('dockerhub', 'docker', 'q', [exhausted_target])
assert db.claim_targets('dockerhub', 'docker', 1, 'owner-x', 60, max_attempts=1) == [exhausted_target]
db.conn.execute(
"UPDATE target_queue SET lease_expires_at = ? WHERE normalized_target = ?",
('2000-01-01T00:00:00+00:00', normalize_target(exhausted_target, 'docker')),
)
db.conn.commit()
assert db.claim_targets('dockerhub', 'docker', 1, 'owner-y', 60, max_attempts=1) == []
assert db.target_queue_item('dockerhub', 'docker', exhausted_target)['status'] == 'failed'
finally:
db.close()
finally:
for key, value in old_urls.items():
if value is not None:
os.environ[key] = value
def main():
for key in ('SCANNER_DB_URL', 'DATABASE_URL', 'TRUF_MANAGED_POSTGRES_DSN', 'KEYCHECK_DB_URL'):
os.environ.pop(key, None)
assert_diagnostic_policy()
assert_docker_platform_policy()
assert_docker_partial_pagination_policy()
assert_retry_policy()
assert_timeout_output_policy()
assert_archive_and_secret_safety()
assert_queue_policy()
print('scanner error policy smoke: OK')
if __name__ == '__main__':
main()
+6562
View File
File diff suppressed because it is too large Load Diff
+378
View File
@@ -0,0 +1,378 @@
import hmac
import ipaddress
import os
import re
import secrets
from datetime import datetime, timezone
from process_identity import current_process_identity, verify_retained_process
from lifecycle_authority import (
LIFECYCLE_PHASES,
PHASE_ACTIVATING,
LifecycleAuthorityError,
build_code_manifest,
code_manifest_sha256,
normalize_code_manifest,
verify_code_manifest,
verify_supervisor_command_line,
)
from runtime_security import (
atomic_write_private_json,
canonical_path,
durable_unlink,
PrivateFileLock,
read_private_json,
sha256_file,
write_private_json_exclusive,
)
INSTANCE_SCHEMA = 2
CONTROL_SCHEMA = 1
SHUTDOWN_RECEIPT_SCHEMA = 1
class InstanceMetadataError(ValueError):
pass
class InstanceLockError(OSError):
pass
def instance_lock_path(instance_path):
return os.path.splitext(os.path.abspath(instance_path))[0] + '.lock'
def shutdown_receipt_path(instance_path):
return os.path.splitext(os.path.abspath(instance_path))[0] + '.exit.json'
class SupervisorInstanceLock:
"""Lifetime singleton lock keyed by the canonical private instance path."""
def __init__(self, instance_path, lock_path=None):
self.instance_path = canonical_path(instance_path)
self.path = os.path.normcase(os.path.abspath(lock_path or instance_lock_path(instance_path)))
self._lock = PrivateFileLock(self.path)
self._acquired = False
@property
def acquired(self):
return self._acquired
def acquire(self):
if self._acquired:
return self
try:
self._lock.acquire()
except BlockingIOError as exc:
raise InstanceLockError('another supervisor owns this private runtime control lock') from exc
except OSError as exc:
raise InstanceLockError(str(exc)) from exc
self._acquired = True
return self
def release(self):
if not self._acquired:
return
self._acquired = False
self._lock.release()
def __enter__(self):
return self.acquire()
def __exit__(self, exc_type, value, traceback):
self.release()
def utc_now_iso():
return datetime.now(timezone.utc).isoformat(timespec='seconds')
def is_loopback_host(host):
try:
return ipaddress.ip_address(str(host)).is_loopback
except ValueError:
return str(host).strip().lower() == 'localhost'
def build_instance_metadata(
launch_nonce,
supervisor_path,
config_path,
control_host,
control_port,
manages_postgres,
identity=None,
instance_id=None,
token=None,
activation_state=PHASE_ACTIVATING,
expected_config_sha256=None,
expected_supervisor_sha256=None,
code_manifest=None,
expected_code_manifest_sha256=None,
canonical_dsn_sha256='',
lifecycle_mode='background',
instance_file=None,
):
if not launch_nonce:
raise InstanceMetadataError('launch nonce is required')
if not is_loopback_host(control_host):
raise InstanceMetadataError('control endpoint must be loopback-only')
identity = identity or current_process_identity()
supervisor_path = canonical_path(supervisor_path)
config_path = canonical_path(config_path)
actual_supervisor_sha256 = sha256_file(supervisor_path)
actual_config_sha256 = sha256_file(config_path)
if expected_supervisor_sha256 and not hmac.compare_digest(actual_supervisor_sha256, str(expected_supervisor_sha256)):
raise InstanceMetadataError('supervisor script changed after parent authority capture')
if expected_config_sha256 and not hmac.compare_digest(actual_config_sha256, str(expected_config_sha256)):
raise InstanceMetadataError('supervisor config changed after parent authority capture')
code_manifest = normalize_code_manifest(code_manifest or build_code_manifest())
manifest_sha256 = code_manifest_sha256(code_manifest)
if expected_code_manifest_sha256 and not hmac.compare_digest(manifest_sha256, str(expected_code_manifest_sha256)):
raise InstanceMetadataError('code manifest changed after parent authority capture')
lifecycle_mode = str(lifecycle_mode)
if lifecycle_mode not in ('background', 'foreground'):
raise InstanceMetadataError('invalid supervisor lifecycle mode')
return {
'schema': INSTANCE_SCHEMA,
'instance_id': instance_id or secrets.token_urlsafe(24),
'token': token or secrets.token_urlsafe(48),
'launch_nonce': str(launch_nonce),
'pid': int(identity.pid),
'process_creation_time': str(identity.creation_time),
'executable': canonical_path(identity.executable),
'supervisor_path': supervisor_path,
'supervisor_sha256': actual_supervisor_sha256,
'config_path': config_path,
'config_sha256': actual_config_sha256,
'code_manifest': code_manifest,
'code_manifest_sha256': manifest_sha256,
'canonical_dsn_sha256': str(canonical_dsn_sha256 or ''),
'instance_file': canonical_path(instance_file) if instance_file else '',
'lifecycle_mode': lifecycle_mode,
'control': {'host': str(control_host), 'port': int(control_port)},
'startup_time': utc_now_iso(),
'manages_postgres': bool(manages_postgres),
'activation_state': str(activation_state),
'private_file_ready': True,
}
def validate_instance_metadata(value):
if not isinstance(value, dict) or value.get('schema') != INSTANCE_SCHEMA:
raise InstanceMetadataError('unsupported supervisor instance schema')
required_strings = (
'instance_id', 'token', 'launch_nonce', 'process_creation_time', 'executable',
'supervisor_path', 'supervisor_sha256', 'config_path', 'config_sha256', 'startup_time',
'code_manifest_sha256', 'lifecycle_mode',
)
for key in required_strings:
if not isinstance(value.get(key), str) or not value[key]:
raise InstanceMetadataError(f'invalid supervisor instance field: {key}')
for key in ('supervisor_sha256', 'config_sha256', 'code_manifest_sha256'):
if not re.fullmatch(r'[0-9a-f]{64}', value[key]):
raise InstanceMetadataError(f'invalid supervisor instance hash: {key}')
canonical_dsn_sha256 = str(value.get('canonical_dsn_sha256') or '')
if canonical_dsn_sha256 and not re.fullmatch(r'[0-9a-f]{64}', canonical_dsn_sha256):
raise InstanceMetadataError('invalid supervisor instance DSN authority')
try:
manifest = normalize_code_manifest(value.get('code_manifest'))
except (OSError, ValueError) as exc:
raise InstanceMetadataError(str(exc)) from exc
if not hmac.compare_digest(code_manifest_sha256(manifest), value['code_manifest_sha256']):
raise InstanceMetadataError('supervisor instance code manifest digest mismatch')
if len(value['instance_id']) > 256 or len(value['token']) < 32 or len(value['token']) > 512:
raise InstanceMetadataError('invalid supervisor instance credentials')
try:
pid = int(value.get('pid'))
except (TypeError, ValueError) as exc:
raise InstanceMetadataError('invalid supervisor instance PID') from exc
if pid <= 0:
raise InstanceMetadataError('invalid supervisor instance PID')
control = value.get('control')
if not isinstance(control, dict) or not is_loopback_host(control.get('host')):
raise InstanceMetadataError('invalid supervisor control endpoint')
try:
port = int(control.get('port'))
except (TypeError, ValueError) as exc:
raise InstanceMetadataError('invalid supervisor control port') from exc
if not 0 < port <= 65535:
raise InstanceMetadataError('invalid supervisor control port')
if value.get('private_file_ready') is not True:
raise InstanceMetadataError('supervisor instance is not marked private-file-ready')
activation_state = str(value.get('activation_state') or PHASE_ACTIVATING).upper()
if activation_state not in LIFECYCLE_PHASES:
raise InstanceMetadataError('invalid supervisor activation state')
lifecycle_mode = str(value.get('lifecycle_mode') or '')
if lifecycle_mode not in ('background', 'foreground'):
raise InstanceMetadataError('invalid supervisor lifecycle mode')
normalized = dict(value)
normalized['pid'] = pid
normalized['control'] = {'host': str(control['host']), 'port': port}
normalized['executable'] = canonical_path(value['executable'])
normalized['supervisor_path'] = canonical_path(value['supervisor_path'])
normalized['config_path'] = canonical_path(value['config_path'])
normalized['code_manifest'] = manifest
normalized['code_manifest_sha256'] = value['code_manifest_sha256']
normalized['canonical_dsn_sha256'] = canonical_dsn_sha256
normalized['instance_file'] = canonical_path(value['instance_file']) if value.get('instance_file') else ''
normalized['lifecycle_mode'] = lifecycle_mode
normalized['manages_postgres'] = bool(value.get('manages_postgres'))
normalized['activation_state'] = activation_state
return normalized
def write_instance_metadata(path, metadata):
write_private_json_exclusive(path, validate_instance_metadata(metadata))
def load_instance_metadata(path):
return validate_instance_metadata(read_private_json(path))
def update_instance_activation(path, instance_id, activation_state):
activation_state = str(activation_state).upper()
if activation_state not in LIFECYCLE_PHASES:
raise InstanceMetadataError('invalid supervisor activation state')
current = load_instance_metadata(path)
if not hmac.compare_digest(current['instance_id'], str(instance_id)):
raise InstanceMetadataError('supervisor activation instance mismatch')
current['activation_state'] = activation_state
atomic_write_private_json(path, validate_instance_metadata(current))
return current
def remove_instance_if_matches(path, instance_id, instance_lock=None, lock_path=None):
owned_lock = None
if instance_lock is None:
try:
owned_lock = SupervisorInstanceLock(path, lock_path=lock_path).acquire()
instance_lock = owned_lock
except OSError:
return False
try:
current = load_instance_metadata(path)
except (OSError, ValueError):
if owned_lock:
owned_lock.release()
return False
if not hmac.compare_digest(current['instance_id'], str(instance_id)):
if owned_lock:
owned_lock.release()
return False
try:
before = os.stat(path, follow_symlinks=False)
confirmed = load_instance_metadata(path)
after = os.stat(path, follow_symlinks=False)
identity_before = (before.st_dev, before.st_ino, before.st_size, before.st_mtime_ns)
identity_after = (after.st_dev, after.st_ino, after.st_size, after.st_mtime_ns)
if identity_before != identity_after or not hmac.compare_digest(confirmed['instance_id'], str(instance_id)):
return False
durable_unlink(path)
return True
except (OSError, ValueError):
return False
finally:
if owned_lock:
owned_lock.release()
def verify_instance_process(
metadata,
supervisor_path=None,
config_path=None,
allow_config_drift=False,
allow_code_drift=False,
):
metadata = validate_instance_metadata(metadata)
if supervisor_path and metadata['supervisor_path'] != canonical_path(supervisor_path):
raise InstanceMetadataError('supervisor path does not match instance metadata')
if config_path and metadata['config_path'] != canonical_path(config_path):
raise InstanceMetadataError('config path does not match instance metadata')
if not allow_code_drift:
try:
verify_code_manifest(metadata['code_manifest'], metadata['code_manifest_sha256'])
if sha256_file(metadata['supervisor_path']) != metadata['supervisor_sha256']:
raise InstanceMetadataError('supervisor script hash does not match instance metadata')
except OSError as exc:
raise InstanceMetadataError(f'unable to recompute supervisor code authority: {exc}') from exc
except ValueError as exc:
raise InstanceMetadataError(str(exc)) from exc
try:
config_matches = sha256_file(metadata['config_path']) == metadata['config_sha256']
except OSError as exc:
if not allow_config_drift:
raise InstanceMetadataError(f'unable to recompute supervisor config authority hash: {exc}') from exc
config_matches = False
if not config_matches and not allow_config_drift:
raise InstanceMetadataError('supervisor config hash drifted; only authenticated shutdown is allowed')
process = verify_retained_process(
metadata['pid'],
metadata['process_creation_time'],
metadata['executable'],
)
try:
arguments = process.command_line()
if metadata['lifecycle_mode'] == 'background' and '--background-child' not in arguments:
raise InstanceMetadataError('retained Python process is not a background supervisor child')
if metadata['lifecycle_mode'] == 'foreground' and '--background-child' in arguments:
raise InstanceMetadataError('retained Python process lifecycle mode mismatch')
try:
verify_supervisor_command_line(
arguments,
metadata['supervisor_path'],
metadata['config_path'],
)
except LifecycleAuthorityError as exc:
raise InstanceMetadataError(str(exc)) from exc
except BaseException:
process.close()
raise
return process
def write_shutdown_receipt(instance_path, instance_id, exit_code):
value = {
'schema': SHUTDOWN_RECEIPT_SCHEMA,
'instance_id': str(instance_id),
'exit_code': int(exit_code),
'completed_at': utc_now_iso(),
}
atomic_write_private_json(shutdown_receipt_path(instance_path), value)
def load_shutdown_receipt(instance_path, instance_id):
value = read_private_json(shutdown_receipt_path(instance_path))
if value.get('schema') != SHUTDOWN_RECEIPT_SCHEMA:
raise InstanceMetadataError('unsupported supervisor shutdown receipt schema')
if not hmac.compare_digest(str(value.get('instance_id') or ''), str(instance_id)):
raise InstanceMetadataError('supervisor shutdown receipt instance mismatch')
try:
code = int(value.get('exit_code'))
except (TypeError, ValueError) as exc:
raise InstanceMetadataError('invalid supervisor shutdown receipt exit code') from exc
return code
def remove_shutdown_receipt(instance_path, instance_id=None):
path = shutdown_receipt_path(instance_path)
try:
if instance_id is not None:
load_shutdown_receipt(instance_path, instance_id)
durable_unlink(path)
return True
except (OSError, ValueError):
return False
def authenticate_request(request, instance_id, token):
if not isinstance(request, dict) or request.get('schema') != CONTROL_SCHEMA:
return False
request_instance = request.get('instance_id')
request_token = request.get('token')
if not isinstance(request_instance, str) or not isinstance(request_token, str):
return False
return hmac.compare_digest(request_instance, str(instance_id)) and hmac.compare_digest(request_token, str(token))
+356
View File
@@ -0,0 +1,356 @@
import sys
sys.dont_write_bytecode = True
import argparse
import os
import re
import threading
import yaml
from migrate_runtime_safety import require_runtime_hardening_stopped
from db_backend import database_url_from_env, is_postgres_url
from paths import apply_path_config
from postgres_runtime import load_postgres_environment
from runtime_security import (
ClusterAuthorityLock,
canonical_path,
durable_replace,
harden_private_file,
PrivateFileLock,
private_file_ready,
require_private_directory,
require_private_file,
)
GITHUB_TOKEN_PREFIXES = ('ghp_', 'gho_', 'ghu_', 'ghs_', 'ghr_', 'github_pat_')
ACCEPTED_ALIVE_STATUSES = {'ALIVE', 'VALID', 'VALID_2FA'}
DOCKERHUB_TOKEN_RE = re.compile(r'^dckr_pat_[A-Za-z0-9_-]{27}$')
PROVIDER_DEFAULTS = {
'github': {
'alive_file': os.path.join('runtime', 'keychecks', 'github', 'githubAlive.txt'),
'pool': 'github_main',
'name_prefix': 'gh',
},
'dockerhub': {
'alive_file': os.path.join('runtime', 'keychecks', 'dockerhub', 'dockerhubAlive.txt'),
'pool': 'dockerhub_main',
'name_prefix': 'dockerhub',
},
}
def read_alive_tokens(path):
tokens = []
seen = set()
with open(path, 'r', encoding='utf-8', errors='replace') as f:
for line in f:
# Status files are TSV-like: token, status, message, extra.
fields = line.rstrip('\r\n').split('\t')
token = fields[0].strip() if fields else ''
if not token or not token.startswith(GITHUB_TOKEN_PREFIXES):
continue
if any(character.isspace() for character in token):
raise ValueError('alive token input contains whitespace in a token field')
status = fields[1].strip().upper() if len(fields) > 1 else ''
if status not in ACCEPTED_ALIVE_STATUSES:
raise ValueError(f'alive token input contains an unaccepted or missing status: {status or "(missing)"}')
if token in seen:
continue
seen.add(token)
tokens.append(token)
return tokens
def read_alive_credentials(path, provider):
provider = str(provider or 'github').strip().lower()
if provider == 'github':
return [{'token': token} for token in read_alive_tokens(path)], 0
if provider != 'dockerhub':
raise ValueError(f'unsupported alive credential provider: {provider}')
credentials = []
seen = {}
skipped_missing_username = 0
with open(path, 'r', encoding='utf-8', errors='replace') as handle:
for line in handle:
fields = line.rstrip('\r\n').split('\t')
identity = fields[0].strip() if fields else ''
if not identity:
continue
if ':' in identity:
username, token = identity.rsplit(':', 1)
username = username.strip()
token = token.strip()
else:
username = ''
token = identity
if not DOCKERHUB_TOKEN_RE.fullmatch(token):
continue
status = fields[1].strip().upper() if len(fields) > 1 else ''
if status not in {'VALID', 'VALID_2FA'}:
raise ValueError(f'alive DockerHub input contains an unaccepted or missing status: {status or "(missing)"}')
if not username:
skipped_missing_username += 1
continue
if len(username) > 256 or ':' in username or any(character.isspace() for character in username):
raise ValueError('alive DockerHub input contains an invalid username field')
previous = seen.get(token)
if previous is not None:
if previous.casefold() != username.casefold():
raise ValueError('alive DockerHub input contains conflicting usernames for one token')
continue
seen[token] = username
credentials.append({'username': username, 'token': token})
return credentials, skipped_missing_username
def next_name(existing_names, prefix):
pattern = re.compile(rf'^{re.escape(prefix)}_(\d+)$')
max_index = 0
for name in existing_names:
match = pattern.match(str(name or ''))
if match:
max_index = max(max_index, int(match.group(1)))
return f'{prefix}_{max_index + 1}'
def _atomic_write_private_yaml(path, value):
parent = require_private_directory(os.path.dirname(os.path.abspath(path)), create=False)
payload = yaml.safe_dump(value, allow_unicode=True, sort_keys=False, width=120).encode('utf-8')
temporary = f'{path}.{os.getpid()}.{threading.get_ident()}.tmp'
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, 'O_BINARY', 0) | getattr(os, 'O_NOFOLLOW', 0)
descriptor = os.open(temporary, flags, 0o600)
try:
os.close(descriptor)
descriptor = None
harden_private_file(temporary)
with open(temporary, 'wb') as handle:
handle.write(payload)
handle.flush()
os.fsync(handle.fileno())
if not private_file_ready(temporary):
raise OSError(f'private temporary secrets ACL changed: {temporary}')
durable_replace(temporary, path)
if not private_file_ready(path):
raise OSError(f'private secrets ACL changed during publication: {path}')
finally:
if descriptor is not None:
os.close(descriptor)
try:
if os.path.exists(temporary):
os.remove(temporary)
except OSError:
pass
def sync_tokens(
secrets_path,
alive_path,
pool_name,
name_prefix,
apply=False,
*,
provider='github',
replace_conflicting_usernames=False,
canonical_secrets_path=None,
authority_lock=None,
stopped_verified=False,
):
if authority_lock is None or not getattr(authority_lock, 'acquired', False):
raise RuntimeError('alive-token sync requires an acquired cluster authority lock')
if stopped_verified is not True:
raise RuntimeError('alive-token sync requires verified stopped runtime proof')
if not canonical_secrets_path or canonical_path(secrets_path) != canonical_path(canonical_secrets_path):
raise RuntimeError('alive-token sync secrets path must exactly match canonical global.secrets_file')
if not os.path.exists(alive_path):
raise SystemExit(f'alive token file not found: {alive_path}')
if not os.path.exists(secrets_path):
raise SystemExit(f'secrets file not found: {secrets_path}')
require_private_file(secrets_path)
require_private_file(alive_path)
lock = PrivateFileLock(f'{secrets_path}.sync.lock').acquire()
try:
require_private_file(secrets_path)
require_private_file(alive_path)
with open(secrets_path, 'r', encoding='utf-8') as f:
secrets = yaml.safe_load(f) or {}
auth_pools = secrets.setdefault('auth_pools', {})
pool = auth_pools.setdefault(pool_name, [])
if not isinstance(pool, list):
raise SystemExit(f'auth_pools.{pool_name} must be a list')
existing_before = len(pool)
provider = str(provider or 'github').strip().lower()
if provider not in PROVIDER_DEFAULTS:
raise ValueError(f'unsupported alive credential provider: {provider}')
existing_tokens = set()
existing_entries = {}
existing_names = set()
normalized_existing = 0
normalized_usernames = 0
for entry in pool:
if not isinstance(entry, dict):
continue
if entry.get('name'):
existing_names.add(str(entry.get('name')))
token = str(entry.get('token') or '')
stripped = token.strip()
if stripped != token:
normalized_existing += 1
if apply:
entry['token'] = stripped
if stripped:
existing_tokens.add(stripped)
existing_entries.setdefault(stripped, []).append(entry)
if provider == 'dockerhub':
username = str(entry.get('username') or '')
stripped_username = username.strip()
if stripped_username != username:
normalized_usernames += 1
if apply:
entry['username'] = stripped_username
alive_credentials, skipped_missing_username = read_alive_credentials(alive_path, provider)
added = []
username_filled = 0
username_conflicts = 0
username_replaced = 0
for credential in alive_credentials:
token = credential['token']
if token in existing_tokens:
if provider == 'dockerhub':
entries = existing_entries.get(token, [])
usernames = {
str(entry.get('username') or '').strip().casefold()
for entry in entries if str(entry.get('username') or '').strip()
}
expected = credential['username'].casefold()
if len(usernames) > 1 or (usernames and expected not in usernames):
if replace_conflicting_usernames:
username_replaced += 1
if apply:
for entry in entries:
entry['username'] = credential['username']
else:
username_conflicts += 1
continue
if not usernames:
username_filled += 1
if apply:
for entry in entries:
entry['username'] = credential['username']
continue
name = next_name(existing_names, name_prefix)
existing_names.add(name)
existing_tokens.add(token)
new_entry = {'name': name}
if provider == 'dockerhub':
new_entry['username'] = credential['username']
new_entry['token'] = token
added.append(new_entry)
if apply and (
added or normalized_existing or normalized_usernames
or username_filled or username_replaced
):
pool.extend(added)
_atomic_write_private_yaml(secrets_path, secrets)
return {
'alive_unique': len(alive_credentials),
'existing_before': existing_before,
'added': len(added),
'normalized_existing': normalized_existing,
'normalized_usernames': normalized_usernames,
'username_filled': username_filled,
'username_conflicts': username_conflicts,
'username_replaced': username_replaced,
'skipped_missing_username': skipped_missing_username,
'pool_after': len(pool) + (0 if apply else len(added)),
}
finally:
lock.release()
def load_config(path):
with open(path, 'r', encoding='utf-8') as handle:
return apply_path_config(yaml.safe_load(handle) or {}, path)
def parse_args():
parser = argparse.ArgumentParser(description='Sync alive provider credentials into a private auth pool without printing them.')
parser.add_argument('--provider', choices=sorted(PROVIDER_DEFAULTS), default='github')
parser.add_argument('--secrets', help='Must exactly match global.secrets_file from --config')
parser.add_argument('--alive-file')
parser.add_argument('--pool')
parser.add_argument('--name-prefix')
parser.add_argument('--config', default=os.path.join('app', 'config.yaml'))
parser.add_argument('--dry-run', action='store_true')
parser.add_argument('--apply', action='store_true', help='Apply under verified offline maintenance authority')
parser.add_argument(
'--replace-conflicting-usernames', action='store_true',
help='Replace an existing DockerHub username only when the same token has an authoritative alive pair',
)
return parser.parse_args()
def main():
args = parse_args()
if args.apply and args.dry_run:
raise SystemExit('--apply and --dry-run are mutually exclusive')
config_path = canonical_path(args.config)
require_private_file(config_path)
config = load_config(config_path)
configured_value = (config.get('global') or {}).get('secrets_file')
if not configured_value:
raise SystemExit('global.secrets_file is required')
configured_secrets = canonical_path(configured_value)
requested_secrets = canonical_path(args.secrets) if args.secrets else configured_secrets
if requested_secrets != configured_secrets:
raise SystemExit('--secrets must exactly match canonical global.secrets_file')
load_postgres_environment(config_path, config)
endpoint_dsn = database_url_from_env() or (config.get('global') or {}).get('database_url')
if not is_postgres_url(endpoint_dsn):
raise SystemExit('A caller-selected canonical PostgreSQL DSN is required for maintenance authority')
provider = str(getattr(args, 'provider', 'github') or 'github').strip().lower()
defaults = PROVIDER_DEFAULTS.get(provider)
if defaults is None:
raise SystemExit(f'unsupported provider: {provider}')
alive_file = args.alive_file or defaults['alive_file']
pool_name = args.pool or defaults['pool']
name_prefix = args.name_prefix or defaults['name_prefix']
with ClusterAuthorityLock(config, endpoint_dsn=endpoint_dsn) as authority_lock:
require_runtime_hardening_stopped(config)
result = sync_tokens(
configured_secrets,
alive_file,
pool_name,
name_prefix,
apply=args.apply,
provider=provider,
replace_conflicting_usernames=bool(getattr(args, 'replace_conflicting_usernames', False)),
canonical_secrets_path=configured_secrets,
authority_lock=authority_lock,
stopped_verified=True,
)
mode = 'updated' if args.apply else 'dry_run'
print(
f"{mode}: provider={provider} pool={pool_name} alive_unique={result['alive_unique']} "
f"existing_before={result['existing_before']} added={result['added']} "
f"username_filled={result.get('username_filled', 0)} "
f"username_conflicts={result.get('username_conflicts', 0)} "
f"username_replaced={result.get('username_replaced', 0)} "
f"skipped_missing_username={result.get('skipped_missing_username', 0)} "
f"normalized_existing={result['normalized_existing']} "
f"normalized_usernames={result.get('normalized_usernames', 0)} pool_after={result['pool_after']}"
)
return 0
if __name__ == '__main__':
main()

Some files were not shown because too many files have changed in this diff Show More