1160 lines
63 KiB
YAML
1160 lines
63 KiB
YAML
# Linux container profile; host runtime entrypoints remain deliberately disabled.
|
|
# See DOCKER_MIGRATION.md. The Windows configuration is preserved in the initial Git commit.
|
|
|
|
global:
|
|
loop: true # true = run forever; false = run one full pass over enabled sources
|
|
cooldown: 30 # seconds to sleep after one full pass over all enabled sources
|
|
backlog_poll_sec: 0.5 # immediately refill released scan slots while durable queue work remains
|
|
root_dir: "/opt/truf" # application image root, never the original Windows checkout
|
|
project_dir: "{root_dir}/app"
|
|
runtime_dir: "/data/runtime-linux" # new Linux state; do not reuse Windows control metadata
|
|
postgres_data_dir: "/data/postgres-linux" # independently initialized, never a Windows cluster copy
|
|
postgres_bin_dir: "/usr/lib/postgresql/16/bin"
|
|
result_bundle_dir: "/data/scanner-result-bundles"
|
|
result_bundle_max_event_bytes: 67108864 # hard limit per bundle; separate from remote reservation
|
|
remote_assignment_reserve_bytes: 2097152 # bundle and projection baseline per unresolved remote assignment
|
|
remote_assignment_max_active: 50 # global unresolved remote assignments across all users
|
|
result_bundle_max_items: 10000
|
|
result_bundle_max_total_bytes: 3221225472
|
|
result_bundle_min_free_bytes: 21474836480
|
|
projection_backlog_max_items: 10000
|
|
projection_backlog_max_bytes: 2147483648
|
|
projection_backlog_headroom_bytes: 402653184 # one worst-case aggregate scan projection beyond all 3 physical slots
|
|
keycheck_queue_max_items: 131072
|
|
keycheck_queue_max_bytes: 134217728
|
|
pipeline_quarantine_max_items: 10000
|
|
pipeline_quarantine_max_bytes: 1073741824
|
|
pipeline_metadata_retention_days: 30
|
|
pipeline_metadata_retirement_batch: 100
|
|
keycheck_candidates_per_event: 2000
|
|
keycheck_candidate_bytes_per_event: 2097152
|
|
keycheck_result_projection_reserve_bytes: 3145728
|
|
keycheck_recheck_batch_items: 10000
|
|
legacy_result_spool_dir: "{runtime_dir}/result_spool"
|
|
legacy_result_spool_max_event_bytes: 201326592
|
|
legacy_result_spool_max_events: 10000
|
|
legacy_result_spool_max_total_bytes: 3221225472
|
|
results_dir: "{runtime_dir}/results" # findings, scan result JSONL/logs, and scanner.db
|
|
queue_dir: "{runtime_dir}/queues" # todo_*.txt and checked_*.txt live here
|
|
keycheck_dir: "{runtime_dir}/keychecks" # keychecker outputs grouped by service
|
|
postman_cache_dir: "{runtime_dir}/postman_cache" # durable cached Postman collection/environment JSON
|
|
postman_cache_max_items: 100000 # aggregate content-addressed artifacts; capacity failure stops discovery
|
|
postman_cache_max_bytes: 21474836480 # artifact, metadata, and temporary bytes under the cache root
|
|
postman_cache_min_free_bytes: 21474836480 # preserve the shared 20 GiB work-volume reserve
|
|
postman_cache_lock_timeout_sec: 300 # match the bounded discovery window under concurrent cache publishers
|
|
postman_discovery_max_artifacts_per_cycle: 1000 # shared non-package download/cache attempt cap
|
|
postman_discovery_max_artifacts_per_page: 100 # bound one API page or GHArchive hour batch
|
|
postman_discovery_max_bytes_per_cycle: 1073741824 # aggregate downloaded artifact bytes
|
|
postman_discovery_max_elapsed_sec: 300 # includes network, cache scan, lock, and publication work
|
|
postman_package_harvest_max_artifacts: 100 # matching package artifacts examined/published per target
|
|
postman_package_harvest_max_bytes: 134217728 # aggregate matching artifact bytes considered per target
|
|
postman_package_harvest_max_elapsed_sec: 30 # optional package harvesting wall-clock deadline
|
|
postman_context_max_input_bytes: 16777216
|
|
postman_context_max_nodes: 100000
|
|
postman_context_max_depth: 64
|
|
postman_context_max_scalar_bytes: 16777216
|
|
postman_context_max_items: 50000
|
|
context_enrichment_max_source_bytes: 16777216 # aggregate optional-context source reads per target
|
|
context_enrichment_max_findings: 2000 # stop optional enrichment without dropping later findings
|
|
context_enrichment_max_postman_comparisons: 200000 # hard cap on fallback substring comparisons
|
|
context_enrichment_max_elapsed_sec: 5 # aggregate optional-context wall-clock budget per target
|
|
trufflehog_diagnostic_max_lines: 2000 # stop parsing stderr after bounded diagnostic work
|
|
trufflehog_diagnostic_max_line_chars: 8192
|
|
trufflehog_diagnostic_max_line_bytes: 8192
|
|
trufflehog_diagnostic_max_errors: 200
|
|
trufflehog_diagnostic_max_warnings: 200
|
|
trufflehog_diagnostic_max_unclassified: 20
|
|
keycheck_input_max_line_bytes: 16777216 # canonical found_secrets producer/consumer JSONL line limit
|
|
keycheck_candidate_artifact_max_items: 2000 # cap candidates derived from one scanned artifact
|
|
keycheck_candidate_artifact_max_bytes: 2097152
|
|
keycheck_candidate_file_max_items: 100000 # aggregate bounded loader/writer limits
|
|
keycheck_candidate_file_max_bytes: 33554432
|
|
keycheck_candidate_line_max_bytes: 8192
|
|
gharchive_cache_dir: "{state_dir}/gharchive_cache" # shared validated hourly .json.gz cache for both GHArchive sources
|
|
gharchive_cache_max_items: 48 # aggregate retained hourly archives
|
|
gharchive_cache_max_bytes: 8589934592 # compressed artifacts, lock metadata, and temporary bytes
|
|
gharchive_cache_min_free_bytes: 5368709120 # preserve 5 GiB free on the runtime volume
|
|
gharchive_download_max_bytes: 536870912 # per-hour compressed response cap
|
|
gharchive_decompressed_max_bytes: 8589934592 # per-hour gzip expansion cap
|
|
gharchive_max_events: 5000000 # per-hour event/line count cap
|
|
gharchive_max_line_bytes: 8388608 # reject oversized individual JSON event lines
|
|
gharchive_cache_lock_timeout_sec: 600 # bounded aggregate/per-hour cross-process lock wait
|
|
state_dir: "{runtime_dir}/state" # runner state files
|
|
log_dir: "{runtime_dir}/logs" # supervisor/source/dashboard logs
|
|
control_dir: "/run/truf/control" # ephemeral container identity, never persisted across recreation
|
|
proxy_file: "{runtime_dir}/proxy.txt" # shared proxy list for checkers/scanners
|
|
api_proxy_enabled: true # discovery/metadata use proxy_file; artifact bodies explicitly bypass proxies
|
|
api_proxy_file: "{proxy_file}" # host:port:user:pass or full proxy URL; resolved from proxy_file by default
|
|
api_proxy_timeout: 5 # proxy connect cap; caller read timeout remains unchanged (fallback: 5 s)
|
|
api_proxy_max_retries: 100 # default total attempts; source-specific attempts/deadlines take precedence
|
|
api_proxy_retry_delay: 5 # seconds between API proxy retries
|
|
download_proxy_enabled: false # reserved: keep heavy downloads/scans direct for now
|
|
download_proxy_file: "" # reserved proxy list for package/artifact downloads when enabled later
|
|
max_active_scans: 1 # conservative default within the container's shared memory budget
|
|
opportunistic_scan_slots: 0 # host-memory-based opportunistic admission is disabled in containers
|
|
opportunistic_scan_sources: [github, gitlab, huggingface]
|
|
opportunistic_scan_reserve_overhead_bytes: 1073741824 # reserve Job cap plus 1 GiB process/staging overhead
|
|
opportunistic_scan_min_available_after_reserve_bytes: 4294967296 # preserve 4 GiB physical RAM after admission
|
|
opportunistic_scan_min_commit_after_reserve_bytes: 6442450944 # preserve 6 GiB commit headroom after admission
|
|
scan_limiter_db: "{state_dir}/scan_limiter.db" # separate SQLite DB for cross-process scan slot leasing
|
|
scan_slot_wait_sec: 0.5 # sleep between slot-acquire attempts when all scan slots are busy
|
|
scan_slot_wait_log_sec: 30 # log long waits at this interval
|
|
scan_slot_stale_sec: 7200 # clean slots older than this or owned by dead scanner PIDs
|
|
target_retry_max_attempts: 3 # total target attempts before a terminal failed queue state
|
|
target_retry_base_delay_sec: 3600 # transient target retry delay; doubles after each failed attempt
|
|
target_retry_max_delay_sec: 86400 # cap exponential target retry delay at 24 hours
|
|
target_timeout_retry_delay_sec: 21600 # timed-out targets use a non-terminal slow retry after 6 hours
|
|
target_claim_batch_size: 1 # fallback only; PostgreSQL slot-first dispatch claims exactly acquired capacity
|
|
admission_resolution_attempts: 300 # exact-token probes after an ambiguous PostgreSQL admission response
|
|
admission_resolution_seconds: 300 # cover 45s loss grace, 60s stable-ready gate, and reconnect margin
|
|
admission_resolution_retry_delay_sec: 1 # bounded pause between fast failed recovery probes
|
|
sync_file_queues: false # legacy todo/checked import is complete; PostgreSQL is authoritative
|
|
dockerhub_tag_cache_path: "{state_dir}/dockerhub_tag_cache.sqlite" # cache Docker Hub tag resolutions/rate limits
|
|
dockerhub_tag_cache_ttl_sec: 21600 # successful tag resolutions are reused for 6 hours
|
|
dockerhub_tag_negative_cache_ttl_sec: 3600 # empty/not-found tag lookups are retried after 1 hour
|
|
dockerhub_tag_rate_limit_cache_ttl_sec: 1800 # global Docker tag API cooldown fallback
|
|
dockerhub_tag_cache_max_rows: 50000
|
|
dockerhub_tag_cache_max_age_sec: 604800
|
|
dockerhub_tag_cache_max_bytes: 268435456
|
|
dockerhub_tag_cache_min_free_bytes: 536870912
|
|
database_path: "{results_dir}/scanner_active.db" # SQLite fallback observability DB when SCANNER_DB_URL is empty
|
|
database_url: "" # Postgres DSN comes from SCANNER_DB_URL; keep real credentials out of config
|
|
dashboard_db_path: "{database_path}" # SQLite dashboard fallback
|
|
dashboard_db_url: "" # optional dashboard Postgres DSN override; defaults to SCANNER_DB_URL when empty
|
|
dashboard_immutable_db: false # active DB is read-only via SQLite mode=ro, but not immutable because WAL changes
|
|
jsonl_rotation_enabled: true # rotate large runtime JSONL files instead of growing multi-GB active files
|
|
found_secrets_max_mb: 128 # rotate found_secrets.jsonl after this active-file size
|
|
scan_results_max_mb: 256 # rotate scan_results.jsonl after this active-file size
|
|
scan_errors_max_mb: 32 # bound each scan_errors.log segment
|
|
scan_errors_keep: 5 # retain at most this many rotated scan error segments
|
|
jsonl_lock_stale_sec: 300 # stale lock cleanup for cross-process JSONL rotation
|
|
jsonl_max_segments: 16 # hard cap; publication pauses until registered consumers catch up
|
|
jsonl_ledger_max_rows: 1000000 # durable O(1) publication identity bound
|
|
jsonl_ledger_max_bytes: 536870912
|
|
jsonl_legacy_index_max_bytes: 16777216 # larger existing files require offline ledger reconciliation
|
|
jsonl_tail_scan_max_bytes: 8388608
|
|
jsonl_torn_quarantine_max_bytes: 65536
|
|
detectors: "" # empty = use all TruffleHog detectors; set IDs to limit intentionally
|
|
exclude_detectors: "github.v1,gitlab.v1,GitHubOauth2" # drop noisy legacy GitHub/GitLab detectors; keep modern prefixes
|
|
no_verification: true # pass --no-verification to TruffleHog; local checkers classify live/dead later
|
|
strict_git_provider_token_filter: true # drop unverified GitHub/GitLab detections that do not match known token prefixes
|
|
drop_detectors: "Privacy,URI,JDBC,Postgres,MongoDB,SQLServer,Box,ZohoCRM,Accuweather,Roaring,Flatio,LinkPreview,RailwayApp" # do not persist obvious non-keycheckable/generic noise detectors
|
|
versions_per_package: 3 # npm/PyPI: scan up to N recent versions per matching package
|
|
work_dir: "/data/scanner-work" # isolated scratch root for clone/download/extract folders
|
|
trufflehog_stdout_max_mb: 32 # hard file-backed streamed stdout byte bound per scan
|
|
trufflehog_stderr_max_mb: 8 # hard file-backed streamed diagnostic byte bound per scan
|
|
trufflehog_config: "{project_dir}/trufflehog-custom-detectors.yaml" # custom detectors loaded by TruffleHog --config
|
|
trufflehog_job_memory_limit_bytes: 4294967296 # aggregate Windows Job limit for each TruffleHog tree (4096 MiB)
|
|
trufflehog_windows_job_cpu_weight: 2 # normal priority with low relative Job weight; avoids broken BELOW_NORMAL startup
|
|
trufflehog_windows_memory_priority: 4 # reclaim scan pages before normal-priority desktop working sets
|
|
min_free_gb: 20 # minimum free space required in work_dir before starting new scans
|
|
state_file: "{state_dir}/runner_state.json" # stores current query index for each source
|
|
secrets_file: "/data/config/secrets.yaml" # runtime-owned private credentials, never baked into the image
|
|
stop_on_seen_pages: false # default false; enable per source where API pagination is sequential
|
|
seen_page_threshold: 2 # stop after this many consecutive all-known pages
|
|
min_pages_before_stop: 1 # always fetch at least this many pages before early stop
|
|
|
|
supervisor:
|
|
enabled_sources: [gitlab, dockerhub, huggingface] # exact distributed discovery-producer profile
|
|
interactive: false # canonical foreground container supervisor, no terminal dependency
|
|
autostart: true # start only the explicit core allowlist above
|
|
poll_sec: 1.0 # how often supervisor checks child process/log state
|
|
heartbeat_sec: 60 # rewrite background status snapshot at least this often; 0 disables heartbeat
|
|
authority_check_interval_sec: 5 # detect code/config authority drift within this bound
|
|
pipeline_status_refresh_sec: 2 # cache authenticated pipeline status queries between loop ticks
|
|
postgres_health_interval_sec: 15 # bounded authenticated controller probe interval
|
|
postgres_ready_loss_grace_sec: 45 # tolerate transient loss for the same live authenticated postmaster
|
|
postgres_stable_ready_sec: 60 # dependency gate opens only after readiness remains stable this long
|
|
postgres_connect_timeout_sec: 5 # bounded controller authentication connect timeout
|
|
postgres_query_timeout_ms: 5000 # bounded controller identity/readiness query timeout
|
|
postgres_stop_timeout_sec: 60 # pg_ctl's bounded identity-verified coordinated stop timeout
|
|
postgres_start_settle_timeout_sec: 30 # bound late postmaster publication checks after pg_ctl start -W
|
|
postgres_shutdown_timeout_sec: 120 # total supervisor wait for the controller during coordinated shutdown
|
|
postgres_log_max_mb: 64 # collector/startup log segment bound
|
|
postgres_log_keep: 24 # maximum retained collector and startup log segments
|
|
refresh_sec: 5 # non-interactive stdout/status-loop interval
|
|
log_dir: "{log_dir}" # one appended child-process log per source
|
|
log_max_mb: 64 # live source output rotates at this active-segment bound
|
|
log_keep: 5 # keep this many rotated log segments per log file
|
|
control_dir: "{control_dir}" # hardened directory; links/junctions are rejected
|
|
instance_file: "{control_dir}/supervisor.instance.json" # private authenticated process/control identity
|
|
lock_file: "{control_dir}/supervisor.lock" # secondary per-instance lock; cluster authority is data-dir-derived and non-configurable
|
|
supervisor_log: "{log_dir}/supervisor.log" # stdout/stderr for background supervisor
|
|
status_file: "{log_dir}/supervisor.status.txt" # latest background supervisor status table
|
|
control_host: "127.0.0.1" # local only; do not expose externally
|
|
control_port: 8765
|
|
background_start_timeout_sec: 20 # wait for matching private metadata and authenticated handshake
|
|
background_shutdown_timeout_sec: 180 # wait on the retained verified supervisor process handle
|
|
attach_poll_sec: 0.2 # --attach watch-mode poll interval
|
|
dashboard_log: "{log_dir}/dashboard.log" # stdout/stderr for supervisor-launched dashboard
|
|
state_dir: "{state_dir}" # per-source runner_state_*.json files to avoid parallel write races
|
|
per_source_state: true # true = supervisor sets RUNNER_STATE_FILE per child process
|
|
result_ingester:
|
|
enabled: true
|
|
poll_sec: 0.2
|
|
lease_seconds: 300
|
|
jsonl_projector:
|
|
enabled: true
|
|
poll_sec: 0.2
|
|
lease_seconds: 300
|
|
worker_api:
|
|
enabled: false # fail closed; configure profiles and private ingress before enabling
|
|
address: "127.0.0.1" # raw API must remain loopback/private and unexposed
|
|
port: 8766
|
|
sources: [] # optional narrowing of exact protocol-2 package capability triples
|
|
auth_entries: {} # GitLab issuance plus optional legacy GitHub reconciliation auth entries
|
|
compatibility_profiles: {} # trusted package manifests; never inferred from clients
|
|
assignment_ttl_seconds: 86400 # immutable server-time deadline, default 24 hours
|
|
assignment_ttl_seconds_by_source: {}
|
|
max_bundle_bytes: 67108864 # hard-capped again by global result_bundle_max_event_bytes
|
|
reaper_interval_seconds: 60
|
|
reaper_batch_size: 1000
|
|
limit_concurrency: 64
|
|
body_idle_timeout_seconds: 30
|
|
json_body_timeout_seconds: 60
|
|
bundle_body_timeout_seconds: 1800
|
|
admin:
|
|
enabled: false # fail closed; available only behind authenticated Caddy
|
|
origin: "" # exact public HTTPS origin when enabled
|
|
edge_marker: "" # independent 256-bit secret shared only with Caddy
|
|
max_body_bytes: 8192
|
|
snapshot_limit: 200
|
|
requeue_limit: 100
|
|
managed_file_roots:
|
|
runtime-keychecks:
|
|
path: "/data/runtime-linux/keychecks"
|
|
permissions:
|
|
list: true
|
|
read: true
|
|
create_replace: false
|
|
delete: false
|
|
limits:
|
|
max_relative_path_bytes: 1024
|
|
max_component_bytes: 255
|
|
max_path_depth: 16
|
|
max_listing_entries: 500
|
|
max_listing_bytes: 262144
|
|
max_file_bytes: 67108864
|
|
runtime-logs:
|
|
path: "/data/runtime-linux/logs"
|
|
permissions:
|
|
list: true
|
|
read: true
|
|
create_replace: false
|
|
delete: false
|
|
limits:
|
|
max_relative_path_bytes: 1024
|
|
max_component_bytes: 255
|
|
max_path_depth: 16
|
|
max_listing_entries: 500
|
|
max_listing_bytes: 262144
|
|
max_file_bytes: 67108864
|
|
runtime-results:
|
|
path: "/data/runtime-linux/results"
|
|
permissions:
|
|
list: true
|
|
read: true
|
|
create_replace: false
|
|
delete: false
|
|
limits:
|
|
max_relative_path_bytes: 1024
|
|
max_component_bytes: 255
|
|
max_path_depth: 16
|
|
max_listing_entries: 500
|
|
max_listing_bytes: 262144
|
|
max_file_bytes: 268435456
|
|
docker_shadow:
|
|
enabled: true # manual-only operator command; never autostarted
|
|
cohort_size: 50
|
|
lease_seconds: 3600
|
|
janitor:
|
|
enabled: true
|
|
interval_sec: 60
|
|
minimum_age_sec: 7200
|
|
max_candidates: 50
|
|
max_entries: 10000
|
|
max_bytes: 1073741824
|
|
max_seconds: 30
|
|
max_depth: 64
|
|
interval: 300 # default delay before repeating a --once child source
|
|
restart_delay: 30 # initial restart delay after failures or unexpected exits
|
|
max_restart_delay: 600 # cap for exponential restart backoff
|
|
restart_reset_after: 300 # clear failure streak/old exit after this many stable seconds
|
|
dashboard:
|
|
enabled: false # true = supervisor also starts dashboard.py
|
|
address: "127.0.0.1" # local-only dashboard bind address
|
|
port: 5000
|
|
startup_grace_sec: 30 # allow Streamlit to initialize before a failed health probe triggers restart
|
|
health_interval_sec: 2 # bounded asynchronous /_stcore/health probe interval
|
|
health_timeout_sec: 1
|
|
restart_base_sec: 2 # exponential dashboard-only restart backoff
|
|
restart_max_sec: 60
|
|
stable_health_sec: 60 # reset dashboard restart streak only after stable health
|
|
defaults:
|
|
enabled: true
|
|
once: false # false = child console_runner loops internally; true = one source cycle per child run
|
|
repeat: true # only relevant when once=true; repeat one-shot cycles after interval
|
|
restart: true # restart crashed/exited child processes
|
|
extra_args: [] # extra args passed to console_runner.py in config mode
|
|
sources:
|
|
github:
|
|
once: false
|
|
github_archive:
|
|
enabled: true # controlled broad GitHub discovery via GHArchive; start manually from supervisor
|
|
use_system_proxy: true
|
|
once: false
|
|
repeat: true
|
|
restart: true
|
|
interval: 3600
|
|
github_archive_files:
|
|
enabled: true # controlled GHArchive changed-file fetch; start manually from supervisor
|
|
use_system_proxy: true
|
|
once: false
|
|
repeat: true
|
|
restart: true
|
|
interval: 1800
|
|
github_gists:
|
|
enabled: true # controlled public gist discovery; start manually from supervisor
|
|
once: false
|
|
repeat: true
|
|
restart: true
|
|
interval: 1800
|
|
gitlab:
|
|
once: false
|
|
github_actions:
|
|
enabled: false # paused after fresh and retained cohorts produced no strict-usable yield
|
|
once: false
|
|
repeat: true
|
|
restart: true
|
|
interval: 3600
|
|
gitlab_ci:
|
|
enabled: true # show in supervisor; main source remains disabled for normal all-source runner
|
|
once: false
|
|
repeat: true
|
|
restart: true
|
|
interval: 3600
|
|
dockerhub:
|
|
once: false
|
|
npm:
|
|
once: false
|
|
pypi:
|
|
once: false
|
|
package_git:
|
|
enabled: false
|
|
once: false
|
|
huggingface:
|
|
once: false
|
|
postman:
|
|
enabled: true # allows `supervisor --sources postman`; main sources.postman.enabled controls default all-source inclusion
|
|
once: false
|
|
env:
|
|
PYTHONIOENCODING: utf-8 # avoid Windows console codec crashes on Unicode repository paths
|
|
|
|
keychecks:
|
|
enabled: true # supervisor manages this as pseudo-source "keychecks"
|
|
autostart: true # start keychecks when supervisor starts, even if scanner sources wait for manual start
|
|
input_mode: postgres # PostgreSQL candidate leases are authoritative; JSONL requires explicit compatibility mode
|
|
services: all # all or comma/list: openai,anthropic,qwen,kimi,github,...
|
|
interval: 3600 # repeat keycheck_runner every N seconds when in repeat/hourly mode
|
|
repeat: true
|
|
restart: false # do not auto-restart failed checker batch; wait for next interval/manual restart
|
|
max_keys: 0 # 0 = no per-run limit; set N for throttled hourly batches
|
|
scheduler_batch_keys: 1000 # per-provider process slice; max_keys=0 keeps rotating until empty/deadline
|
|
scheduler_workers: 1 # serialize provider handshakes; avoids transient control-plane startup failures
|
|
scheduler_deadline_sec: 1800 # aggregate work-conserving provider deadline
|
|
retry_network: true # retry transient network failures each scheduled run
|
|
retry_limited: false # set true to retry rate-limited/no-quota statuses hourly
|
|
retry_unknown: false
|
|
retry_restricted: false
|
|
retry_no_balance: false # retry no_balance/no_quota statuses when explicitly enabled
|
|
retry_valid: false # valid provider keys are re-probed only when explicitly requested
|
|
recheck_all: false # true forces all known keys to be checked again
|
|
env:
|
|
KEYCHECK_INPUT_TAIL_MB: "0" # first keycheck reads from offset 0; high-watermark state handles later appends
|
|
KEYCHECK_INPUT_MAX_LINE_BYTES: "16777216" # must match global.keycheck_input_max_line_bytes
|
|
KEYCHECK_CANDIDATE_MAX_UNCONSUMED_ATTEMPTS: "3"
|
|
KEYCHECK_DB_INGEST_TAIL_MB: "64" # optional tail size; initial DB ingest uses full bounded offsets unless KEYCHECK_DB_INGEST_TAIL_INITIAL=1
|
|
KEYCHECK_RESULTS_MAX_MB: "32" # rotate per-service *Results.jsonl files into manifest segments
|
|
KEYCHECK_PROVIDER_RESOLUTION_ORDER: "deepseek,zai,qwen,kimi"
|
|
KEYCHECK_EVENT_MAP_BACKFILL_ROWS: "0" # one-time legacy backfill is complete; live ingest maintains this map atomically
|
|
KEYCHECK_UID_MAP_BACKFILL_ROWS: "0" # scanner writes finding_uid_map for all new findings
|
|
db_ingest:
|
|
enabled: false # explicit JSONL compatibility import only
|
|
repair_links:
|
|
enabled: false # new DB candidates carry exact finding attribution
|
|
service_args: # provider-specific probe flags passed by keycheck_runner
|
|
gemini:
|
|
- --probe-generation # call generateContent; RATE_LIMITED valid keys go to geminiAliveRateLimited.txt
|
|
aws:
|
|
- --probe-bedrock # after STS, probe Bedrock access using safe validation-style calls
|
|
- --bedrock-max-attempts
|
|
- "12"
|
|
azure:
|
|
- --probe-openai-route # after deployments list, probe Azure OpenAI chat route without generation
|
|
- --probe-foundry-route # probe Azure AI Foundry/MaaS route for configured models
|
|
- --foundry-models
|
|
- claude-opus-4-6,claude-fable-5
|
|
- --timeout
|
|
- "8"
|
|
gcp:
|
|
- --probe-vertex # after OAuth, probe Vertex AI Gemini countTokens access
|
|
- --vertex-timeout
|
|
- "6"
|
|
- --vertex-max-attempts
|
|
- "6"
|
|
- --vertex-models
|
|
- gemini-3.6-flash,gemini-3.1-pro-preview
|
|
- --vertex-locations
|
|
- global,us,eu
|
|
- --vertex-anthropic-models
|
|
- claude-opus-5,claude-opus-4-7,claude-opus-4-6,claude-fable-5
|
|
- --vertex-anthropic-locations
|
|
- global,us,eu,us-east5,europe-west1
|
|
- --vertex-anthropic-max-attempts
|
|
- "20"
|
|
summary_tsv: "{keycheck_dir}/summary.tsv"
|
|
summary_json: "{keycheck_dir}/summary.json"
|
|
alive_summary_tsv: "{keycheck_dir}/alive_summary.tsv"
|
|
|
|
query_policy:
|
|
rejected: # reviewed source-specific zero-alive evidence; queue rows remain auditable and reversible
|
|
- {source: dockerhub, query: OR, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1631, findings: 29137, unique_credentials: 36, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 36}
|
|
- {source: dockerhub, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1009, findings: 69486, unique_credentials: 30, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 229}
|
|
- {source: github, query: coding, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1696, findings: 251, unique_credentials: 11, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 0}
|
|
- {source: github, query: memory, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3190, findings: 254, unique_credentials: 11, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 8}
|
|
- {source: npm, query: OR, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1985, findings: 152, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 598}
|
|
- {source: npm, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 9446, findings: 5833, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3957}
|
|
- {source: npm, query: agents, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 8936, findings: 4573, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3346}
|
|
- {source: npm, query: ai, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 7676, findings: 49499, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3663}
|
|
- {source: npm, query: assistant, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3888, findings: 14831, unique_credentials: 2, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 899}
|
|
- {source: npm, query: benchmark, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2371, findings: 298, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1292}
|
|
- {source: npm, query: bot, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1784, findings: 2081, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 951}
|
|
- {source: npm, query: chat, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2048, findings: 381, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2087}
|
|
- {source: npm, query: chats, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1297, findings: 107, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 719}
|
|
- {source: npm, query: completion, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2529, findings: 1351, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1263}
|
|
- {source: npm, query: completions, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3058, findings: 1169, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1466}
|
|
- {source: npm, query: conversation, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1454, findings: 475, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 954}
|
|
- {source: npm, query: gemini, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 4922, findings: 3830, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 891}
|
|
- {source: npm, query: groq, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2050, findings: 333, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 484}
|
|
- {source: npm, query: langchain, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1264, findings: 169, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 908}
|
|
- {source: npm, query: llm, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3682, findings: 17976, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2333}
|
|
- {source: npm, query: mcp, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 8704, findings: 5572, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3524}
|
|
- {source: npm, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 6262, findings: 3288, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1359}
|
|
- {source: npm, query: openrouter, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1590, findings: 355, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1014}
|
|
- {source: npm, query: prompt, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1899, findings: 91, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1161}
|
|
- {source: npm, query: rag, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2242, findings: 1912, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 544}
|
|
- {source: npm, query: semantic, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1911, findings: 164, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1168}
|
|
- {source: npm, query: studio, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 4158, findings: 484, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1349}
|
|
- {source: npm, query: xai, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1390, findings: 2395, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 513}
|
|
- {source: package_git, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1851, findings: 5395, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 8791}
|
|
- {source: package_git, query: bot, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1610, findings: 7260, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 4486}
|
|
- {source: package_git, query: completion, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3149, findings: 10028, unique_credentials: 9, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1507}
|
|
- {source: package_git, query: llm, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1285, findings: 2116, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 5572}
|
|
- {source: package_git, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2053, findings: 14375, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 9968}
|
|
- {source: package_git, query: semantic, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3895, findings: 24064, unique_credentials: 22, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2414}
|
|
- {source: package_git, query: studio, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1028, findings: 2666, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 5709}
|
|
- {source: postman, query: XAI_API_KEY, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2935, findings: 83, unique_credentials: 1, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 306}
|
|
- {source: pypi, query: langchain, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1095, findings: 47, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 16}
|
|
- {source: pypi, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1253, findings: 731, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 66}
|
|
|
|
sources:
|
|
github:
|
|
enabled: false # direct GitHub discovery is paused in favor of the Actions experiment
|
|
auth_pool: github_main # auth_pools.<name> in secrets.yaml; leave empty to use env/legacy token
|
|
auth_rotation: per_cycle # per_cycle = use next token each source cycle; none = first available
|
|
retry_with_next_auth_on_rate_limit: true # switch token and retry when GitHub API rate-limits
|
|
rate_limit_cooldown: 3600 # seconds to cool down a rate-limited token if API gives no reset time
|
|
mode: recent # recent = updated recently; search = paginated search; custom = target_file URLs
|
|
queries: # one query is used per source cycle; state rotates through this list
|
|
- llm
|
|
- ai
|
|
- agent
|
|
- agents
|
|
- assistant
|
|
- bot
|
|
- chatbot
|
|
- rag
|
|
- semantic
|
|
- prompt
|
|
- completion
|
|
- completions
|
|
- mcp
|
|
- claude
|
|
- anthropic
|
|
- opus
|
|
- sonnet
|
|
- haiku
|
|
- vertex
|
|
- aiplatform
|
|
- bedrock
|
|
- foundry
|
|
- azure-openai
|
|
- openrouter
|
|
- langchain
|
|
- langgraph
|
|
- litellm
|
|
- crewai
|
|
- autogen
|
|
- semantic-kernel
|
|
- mistral
|
|
- groq
|
|
- cohere
|
|
- xai
|
|
- together
|
|
- perplexity
|
|
- gemini
|
|
- qwen
|
|
- kimi
|
|
- dashscope
|
|
- aistudio
|
|
- studio
|
|
- OR
|
|
- open
|
|
- chat
|
|
- chats
|
|
- conversation
|
|
- conversational
|
|
- dialogue
|
|
- OPENAI_API_KEY # bounded high-signal README integration query
|
|
- api.openai.com # bounded high-signal README endpoint query
|
|
- openai-agents # bounded OpenAI agent SDK/ecosystem query
|
|
- copilot
|
|
- ai-assistant
|
|
- ai-agent
|
|
- multi-agent
|
|
- workflow
|
|
- workflows
|
|
- llmops
|
|
- benchmark
|
|
- tokens
|
|
- function-calling
|
|
query_overrides:
|
|
OPENAI_API_KEY:
|
|
pages: 1
|
|
per_page: 25
|
|
max_targets: 5
|
|
api.openai.com:
|
|
pages: 1
|
|
per_page: 25
|
|
max_targets: 5
|
|
openai-agents:
|
|
pages: 1
|
|
per_page: 50
|
|
max_targets: 5
|
|
pages: 5 # pages to fetch in search/recent mode
|
|
per_page: 100 # targets per API page, max is usually 100
|
|
workers: 1 # one shared non-Docker scan slot for this source
|
|
timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes
|
|
recent_hours: 24 # recent mode: only repos updated within this many hours are discovered
|
|
max_repo_age_days: 90 # skip repos older than this by repo_age_field before queueing
|
|
repo_age_field: updated_at # metadata field for repo age: created_at, updated_at, pushed_at
|
|
max_commit_age_days: 90 # scan only recent git history by passing --since-commit boundary
|
|
commit_lookup_pages: 3 # API pages of commits to inspect when finding --since-commit boundary
|
|
skip_if_commit_lookup_fails: true # true = do not scan if recent commit boundary cannot be found
|
|
max_depth: 100 # TruffleHog --max-depth; extra cap on commit count scanned
|
|
exact_git_planning_enabled: true # resolve and bind the exact default/explicit ref head after claim
|
|
git_baseline_depth: 100 # bounded first/reset scan depth; deltas use only the covered base
|
|
git_ref_resolution_timeout_sec: 10
|
|
git_ref_resolution_attempts: 2
|
|
git_ref_resolution_max_bytes: 1048576
|
|
scan_full_history: false # true = ignore max_depth; still uses max_commit_age_days when configured
|
|
sort_by: updated # GitHub search sort: updated, stars, forks, created
|
|
sort_order: desc # desc = newest/highest first; asc = oldest/lowest first
|
|
created_filter: any # optional GitHub created filter: any, today, week, month, year
|
|
stop_on_seen_pages: true # stop fetching pages after consecutive all-known pages
|
|
seen_page_threshold: 2
|
|
min_pages_before_stop: 1
|
|
updated_target_rescan_enabled: true # preserve pushed_at and rescan completed repos after newer pushes
|
|
updated_target_rescan_max_per_cycle: 1 # bounded rollout: at most one changed repo per discovery cycle
|
|
updated_target_rescan_cooldown_hours: 24
|
|
|
|
github_archive:
|
|
enabled: false # broad discovery source; supervisor exposes it for manual/controlled runs
|
|
use_system_proxy: true # direct GHArchive route is unstable; use inherited HTTP(S)_PROXY only for hourly dumps
|
|
auth_pool: github_main
|
|
auth_rotation: per_cycle
|
|
mode: archive # GHArchive hourly events -> GitHub repos -> recent git scan
|
|
queries:
|
|
- gharchive
|
|
archive_hours_back: 6 # read recent completed GHArchive hours
|
|
fetch_timeout: 120 # read timeout while downloading large hourly gzip archives
|
|
archive_max_repos_per_cycle: 500 # hard cap after event/repo dedupe/scoring
|
|
archive_rescan_cooldown_hours: 48 # allow rescanning active repos after cooldown; not forever-checked
|
|
archive_event_types:
|
|
- PushEvent
|
|
- CreateEvent
|
|
- PublicEvent
|
|
workers: 4
|
|
timeout: 1800
|
|
max_depth: 75
|
|
scan_full_history: false
|
|
max_commit_age_days: 0 # avoid GitHub commit-boundary API lookups for broad source
|
|
commit_lookup_pages: 0
|
|
skip_if_commit_lookup_fails: false
|
|
stop_on_seen_pages: false
|
|
|
|
github_archive_files:
|
|
enabled: false # fetch suspicious changed files from GHArchive PushEvents
|
|
use_system_proxy: true
|
|
auth_pool: github_main
|
|
auth_rotation: per_cycle
|
|
mode: archive
|
|
queries:
|
|
- gharchive-files
|
|
archive_hours_back: 6
|
|
archive_max_files_per_cycle: 100
|
|
archive_max_commit_lookups: 200
|
|
archive_event_types:
|
|
- PushEvent
|
|
workers: 1
|
|
timeout: 300
|
|
fetch_timeout: 120
|
|
max_artifact_size_mb: 2
|
|
stop_on_seen_pages: false
|
|
|
|
github_gists:
|
|
enabled: false # public gist source; supervisor exposes it for manual/controlled runs
|
|
auth_pool: github_main
|
|
auth_rotation: per_cycle
|
|
mode: search
|
|
queries:
|
|
- gists
|
|
pages: 3
|
|
per_page: 100
|
|
workers: 3
|
|
timeout: 300
|
|
fetch_timeout: 20
|
|
max_artifact_size_mb: 2
|
|
stop_on_seen_pages: true
|
|
seen_page_threshold: 2
|
|
min_pages_before_stop: 1
|
|
|
|
gitlab:
|
|
enabled: true # include the GitLab discovery producer in distributed runs
|
|
target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission
|
|
auth_pool: gitlab_main # auth_pools.<name> in secrets.yaml; leave empty to use env/legacy token
|
|
auth_rotation: per_cycle # per_cycle = use next token each source cycle; none = first available
|
|
retry_with_next_auth_on_rate_limit: true # switch token and retry when GitLab API returns 429
|
|
rate_limit_cooldown: 3600 # seconds to cool down a rate-limited token if API gives no reset time
|
|
discovery_request_attempts: 3 # bounded retries for idempotent project discovery transport failures
|
|
discovery_retry_delay: 5 # seconds between GitLab discovery transport attempts
|
|
external_trufflehog_lifecycle: true # bypass embedded overseer and require explicit completion for GitLab scans
|
|
mode: recent # recent = last_activity_after; search = paginated project search; custom = target_file URLs
|
|
queries: # one query is used per source cycle; state rotates through this list
|
|
- llm
|
|
- ai
|
|
- agent
|
|
- agents
|
|
- assistant
|
|
- bot
|
|
- chatbot
|
|
- rag
|
|
- openai-api # bounded metadata-friendly OpenAI API query
|
|
- openai-agents # bounded OpenAI agent SDK/ecosystem query
|
|
- librechat # bounded deployable chat application query
|
|
- semantic
|
|
- prompt
|
|
- completion
|
|
- completions
|
|
- mcp
|
|
- openrouter
|
|
- groq
|
|
- xai
|
|
- langchain
|
|
- litellm
|
|
- gemini
|
|
- qwen
|
|
- kimi
|
|
- dashscope
|
|
- aistudio
|
|
- studio
|
|
- OR
|
|
- open
|
|
- chat
|
|
- chats
|
|
- conversation
|
|
- conversational
|
|
- dialogue
|
|
- copilot
|
|
- coding
|
|
- ai-assistant
|
|
- ai-agent
|
|
- multi-agent
|
|
- workflow
|
|
- workflows
|
|
- llmops
|
|
- benchmark
|
|
- tokens
|
|
- memory
|
|
- function-calling
|
|
- hermes
|
|
- harnes
|
|
- flow
|
|
- helpdesk
|
|
- paperless
|
|
query_overrides:
|
|
openai-api:
|
|
pages: 1
|
|
per_page: 50
|
|
max_targets: 5
|
|
openai-agents:
|
|
pages: 1
|
|
per_page: 50
|
|
max_targets: 5
|
|
librechat:
|
|
pages: 1
|
|
per_page: 25
|
|
max_targets: 5
|
|
hermes:
|
|
pages: 1
|
|
per_page: 50
|
|
max_targets: 5
|
|
harnes:
|
|
pages: 1
|
|
per_page: 50
|
|
max_targets: 5
|
|
flow:
|
|
pages: 1
|
|
per_page: 50
|
|
max_targets: 5
|
|
helpdesk:
|
|
pages: 1
|
|
per_page: 50
|
|
max_targets: 5
|
|
paperless:
|
|
pages: 1
|
|
per_page: 50
|
|
max_targets: 5
|
|
pages: 10 # pages to fetch in search/recent mode
|
|
per_page: 100 # targets per API page, max is usually 100
|
|
workers: 1 # one shared non-Docker scan slot for this source
|
|
timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes
|
|
recent_hours: 96 # recent mode: only projects active within this many hours are discovered
|
|
max_repo_age_days: 90 # skip projects older than this by repo_age_field before queueing
|
|
repo_age_field: last_activity_at # metadata field: created_at, updated_at, last_activity_at
|
|
max_commit_age_days: 90 # scan only recent git history by passing --since-commit boundary
|
|
commit_lookup_pages: 3 # API pages of commits to inspect when finding --since-commit boundary
|
|
skip_if_commit_lookup_fails: true # true = do not scan if recent commit boundary cannot be found
|
|
max_depth: 100 # TruffleHog --max-depth; extra cap on commit count scanned
|
|
exact_git_planning_enabled: true # resolve and bind the exact default/explicit ref head after claim
|
|
git_baseline_depth: 100 # bounded first/reset scan depth; deltas use only the covered base
|
|
git_ref_resolution_timeout_sec: 10
|
|
git_ref_resolution_attempts: 2
|
|
git_ref_resolution_max_bytes: 1048576
|
|
scan_full_history: false # true = ignore max_depth; still uses max_commit_age_days when configured
|
|
sort_by: last_activity_at # GitLab order_by field: last_activity_at, created_at, updated_at, name, stars
|
|
sort_order: desc # desc = newest/highest first; asc = oldest/lowest first
|
|
visibility: public # public, internal, private; private/internal require token permissions
|
|
stop_on_seen_pages: true # stop fetching pages after consecutive all-known pages
|
|
seen_page_threshold: 2
|
|
min_pages_before_stop: 1
|
|
updated_target_rescan_enabled: true # preserve last_activity_at and rescan completed projects after new activity
|
|
updated_target_rescan_max_per_cycle: 1
|
|
updated_target_rescan_cooldown_hours: 24
|
|
|
|
github_actions:
|
|
enabled: false # paused; queue and historical evidence remain intact
|
|
auth_pool: github_main
|
|
auth_rotation: per_cycle
|
|
mode: recent
|
|
queries:
|
|
- logs
|
|
ci_seed_sources: github,github_archive,package_git
|
|
ci_max_repos_per_cycle: 25
|
|
ci_seed_scan_limit: 15000
|
|
ci_seed_query_batch_size: 250 # bound each PostgreSQL history page on HDD
|
|
ci_soft_cooldown_days: 7
|
|
ci_runs_per_repo: 5
|
|
ci_lookback_days: 30
|
|
ci_max_log_archive_mb: 150
|
|
ci_max_log_file_mb: 50
|
|
ci_scan_artifacts: true
|
|
ci_max_artifacts_per_run: 3
|
|
ci_max_artifact_archive_mb: 150
|
|
ci_max_artifact_file_mb: 50
|
|
ci_max_artifact_files: 1000
|
|
ci_failed_first: true
|
|
refresh_registry: true # keep discovering recent repositories while historical targets remain queued
|
|
target_claim_order: balanced # split work between fresh discoveries and the retained historical backlog
|
|
workers: 1
|
|
timeout: 180
|
|
|
|
gitlab_ci:
|
|
enabled: false # controlled experiment: run manually from supervisor
|
|
auth_pool: gitlab_main
|
|
auth_rotation: per_cycle
|
|
mode: recent
|
|
queries:
|
|
- logs
|
|
ci_seed_sources: gitlab,package_git
|
|
ci_max_repos_per_cycle: 25
|
|
ci_seed_scan_limit: 15000
|
|
ci_seed_query_batch_size: 250 # bound each PostgreSQL history page on HDD
|
|
ci_soft_cooldown_days: 7
|
|
ci_pipelines_per_project: 5
|
|
ci_jobs_per_pipeline: 20
|
|
ci_lookback_days: 30
|
|
ci_max_trace_mb: 100
|
|
ci_scan_artifacts: true
|
|
ci_max_artifacts_per_pipeline: 5
|
|
ci_max_artifact_archive_mb: 150
|
|
ci_max_artifact_file_mb: 50
|
|
ci_max_artifact_files: 1000
|
|
workers: 3
|
|
timeout: 180
|
|
|
|
huggingface:
|
|
enabled: true # discover newest HuggingFace Spaces for remote workers
|
|
target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission
|
|
auth_pool: huggingface_main # optional auth_pools.<name> in secrets.yaml or use HF_TOKEN/HUGGINGFACE_TOKEN
|
|
auth_rotation: per_cycle
|
|
mode: recent # recent/search both fetch newest spaces; custom = target_file space IDs
|
|
queries:
|
|
- spaces # placeholder query for state rotation; HuggingFace API fetch ignores it for now
|
|
pages: 100 # bounded cursor pages from the newest-lastModified Spaces API
|
|
per_page: 100 # newest-modified API page size is fixed at 100 by the runner
|
|
workers: 1
|
|
timeout: 1800
|
|
fetch_timeout: 15
|
|
discovery_request_attempts: 3 # one transient page timeout must not restart the whole source
|
|
discovery_retry_delay: 5
|
|
stop_on_seen_pages: true
|
|
seen_page_threshold: 2
|
|
min_pages_before_stop: 1
|
|
updated_target_rescan_enabled: true # preserve lastModified and revisit changed completed Spaces
|
|
updated_target_rescan_max_per_cycle: 1
|
|
updated_target_rescan_cooldown_hours: 24
|
|
|
|
dockerhub:
|
|
enabled: true # include the DockerHub discovery producer in distributed runs
|
|
target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission
|
|
require_digest: true # only immutable repo@sha256:... image targets may reach TruffleHog
|
|
auth_pool: dockerhub_main # auth_pools.<name> in secrets.yaml; all available accounts are rotated per image scan
|
|
auth_rotation: per_scan # DockerTokenManager rotates Docker accounts for each docker scan
|
|
retry_with_next_auth_on_rate_limit: true # rotate Hub/Registry requests to another account after 429/invalid auth
|
|
rate_limit_cooldown: 1800 # per-account cooldown; shared cooldown starts only after pool exhaustion
|
|
mode: search # search = Docker Hub search; recent = client-side recent tag filtering; custom = target_file images
|
|
refresh_registry: true # discover fresh images even while deferred targets remain queued
|
|
queries: # one query is used per source cycle; Docker Hub empty query returns no results
|
|
- llm
|
|
- ai
|
|
- agents
|
|
- assistant
|
|
- bot
|
|
- chatbot
|
|
- rag
|
|
- semantic
|
|
- prompt
|
|
- completion
|
|
- completions
|
|
- mcp
|
|
- openrouter
|
|
- groq
|
|
- xai
|
|
- langchain
|
|
- litellm
|
|
- gemini
|
|
- qwen
|
|
- kimi
|
|
- dashscope
|
|
- aistudio
|
|
- studio
|
|
- librechat # bounded deployable chat application query
|
|
- lobechat # bounded deployable chat application query
|
|
- openai-proxy # bounded OpenAI-compatible proxy query
|
|
- open
|
|
- chat
|
|
- chats
|
|
- conversation
|
|
- conversational
|
|
- dialogue
|
|
- copilot
|
|
- coding
|
|
- ai-assistant
|
|
- ai-agent
|
|
- multi-agent
|
|
- workflow
|
|
- workflows
|
|
- llmops
|
|
- benchmark
|
|
- tokens
|
|
- memory
|
|
- function-calling
|
|
- hermes
|
|
- harnes
|
|
- flow
|
|
- helpdesk
|
|
- paperless
|
|
- open-webui
|
|
- ragflow
|
|
- dify
|
|
- flowise
|
|
- crewai
|
|
- n8n
|
|
- langflow
|
|
- autogen
|
|
- browser-use
|
|
- openhands
|
|
- anythingllm
|
|
- agent-zero
|
|
query_overrides:
|
|
librechat:
|
|
max_targets: 10
|
|
lobechat:
|
|
max_targets: 10
|
|
openai-proxy:
|
|
max_targets: 10
|
|
hermes:
|
|
max_targets: 10
|
|
harnes:
|
|
max_targets: 10
|
|
flow:
|
|
max_targets: 10
|
|
helpdesk:
|
|
max_targets: 10
|
|
paperless:
|
|
max_targets: 10
|
|
pages: 30 # maximum Docker Hub search pages for every configured query
|
|
per_page: 100 # maximum Docker Hub results per search page
|
|
workers: 2 # allow two Docker image scans within the global three-slot limit
|
|
trufflehog_job_memory_limit_bytes: 6442450944 # Docker images get 6 GiB; other sources retain the 4 GiB global cap
|
|
timeout: 600 # bounded full-image scan window after disabling TruffleHog's embedded overseer
|
|
trufflehog_concurrency: 4 # bound per-image layer/chunk fan-out; two source workers remain available
|
|
recent_days: 7 # recent mode: keep images/tags updated within this many days
|
|
fetch_workers: 5 # parallel Docker Hub search page fetches before scanning
|
|
fetch_timeout: 15 # seconds per Docker Hub API request
|
|
tag_fetch_workers: 4 # bound the in-flight burst before a shared 429 cooldown becomes visible
|
|
tag_retry_count: 0 # do not retry individual transport failures during high-volume tag resolution
|
|
tag_retry_delay: 10 # base delay if bounded non-429 transport retries are enabled later
|
|
tag_resolve_limit: 100 # max old bare todo repos to resolve to tags per cycle; 0 = all
|
|
docker_platform_filter_enabled: true # skip tags only when complete metadata proves linux/amd64 is absent
|
|
docker_platform_os: linux
|
|
docker_platform_arch: amd64
|
|
docker_platform_candidate_tags: 20 # inspect extra recent tags so an ARM-only latest tag does not hide an amd64 tag
|
|
docker_images_per_repository: 3 # ordinary FIFO resolver stays at the reviewed production depth
|
|
docker_content_scan_mode: canary # only prior durable full-image timeouts are eligible for layer fallback
|
|
docker_layer_canary_basis_points: 10000 # all prior durable full-image timeouts use bounded layer fallback
|
|
docker_adaptive_canary_basis_points: 0 # disabled until an exact aggregate shadow report passes every gate
|
|
docker_adaptive_gate_max_age_sec: 604800 # matching shadow evidence expires after seven days
|
|
docker_layer_config_max_bytes: 1048576 # image config is selected independently from layer budget
|
|
docker_layer_max_bytes: 268435456 # max compressed bytes for one selected application layer
|
|
docker_layer_image_max_bytes: 1073741824 # cumulative compressed layer budget per immutable image
|
|
docker_layer_max_layers: 8 # select application layers from top to base within the byte budget
|
|
docker_layer_archive_max_size_bytes: 268435456 # TruffleHog per-member extraction bound
|
|
docker_layer_archive_max_depth: 4 # required for an OCI wrapper plus compressed layer archive
|
|
docker_layer_archive_timeout_sec: 30
|
|
docker_layer_blob_timeout_sec: 600 # shared transfer+filesystem scan deadline per durable checkpoint
|
|
docker_layer_filesystem_concurrency: 2
|
|
docker_layer_blob_max_attempts: 3 # blob budget is authoritative; parent checkpoint attempts reset
|
|
docker_layer_blob_lease_sec: 1800 # exceeds the bounded 600-second execution plus handoff margin
|
|
docker_layer_min_free_bytes: 21474836480 # retain 20 GiB on the private work volume
|
|
docker_layer_checkpoint_delay_sec: 60 # resume the next selected blob without a full-image restart
|
|
docker_adaptive_checkpoint_max_blobs: 4 # scheduling-only bound for a future gated adaptive checkpoint
|
|
docker_adaptive_checkpoint_max_bytes: 536870912 # aggregate compressed bytes per adaptive checkpoint
|
|
docker_repository_refresh_interval_sec: 86400 # revisit each resolved repository daily for new immutable digests
|
|
docker_repository_refresh_max_per_cycle: 0 # disable periodic refresh of completed repository anchors
|
|
sort_by: updated_at # Docker Hub search sort field
|
|
sort_order: desc # desc = newest/highest first; asc = oldest/lowest first
|
|
|
|
npm:
|
|
enabled: true # true = include npm registry in config-mode runs
|
|
mode: search # npm currently supports search mode
|
|
queries: # one query is used per source cycle; state rotates through this list
|
|
- chatbot
|
|
- litellm
|
|
- qwen
|
|
- kimi
|
|
- dashscope
|
|
- aistudio
|
|
- conversational
|
|
- dialogue
|
|
- copilot
|
|
- coding
|
|
- ai-assistant
|
|
- ai-agent
|
|
- multi-agent
|
|
- workflow
|
|
- workflows
|
|
- llmops
|
|
- tokens
|
|
- memory
|
|
- function-calling
|
|
pages: 30 # npm search pages to fetch for the current query
|
|
per_page: 50 # npm packages per search page
|
|
workers: 3 # parallel TruffleHog filesystem scans for downloaded packages
|
|
timeout: 300 # seconds before killing one package scan process tree
|
|
versions_per_package: 3 # scan latest N versions per package within max_version_age_days
|
|
max_version_age_days: 90 # skip package versions older than this many days
|
|
max_artifact_size_mb: 300 # skip/download-fail package tarballs larger than this
|
|
fetch_timeout: 20 # seconds per npm registry request
|
|
|
|
pypi:
|
|
enabled: true # true = include PyPI registry in config-mode runs
|
|
mode: search # PyPI currently supports search mode
|
|
queries: # one query is used per source cycle; state rotates through this list
|
|
- llm
|
|
- ai
|
|
- agent
|
|
- agents
|
|
- assistant
|
|
- bot
|
|
- chatbot
|
|
- rag
|
|
- semantic
|
|
- prompt
|
|
- completion
|
|
- completions
|
|
- mcp
|
|
- openrouter
|
|
- groq
|
|
- xai
|
|
- litellm
|
|
- gemini
|
|
- qwen
|
|
- kimi
|
|
- dashscope
|
|
- aistudio
|
|
- studio
|
|
- OR
|
|
- chat
|
|
- chats
|
|
- conversation
|
|
- conversational
|
|
- dialogue
|
|
- copilot
|
|
- coding
|
|
- ai-assistant
|
|
- ai-agent
|
|
- multi-agent
|
|
- workflow
|
|
- workflows
|
|
- llmops
|
|
- benchmark
|
|
- tokens
|
|
- memory
|
|
- function-calling
|
|
pages: 40 # PyPI search pages to fetch for the current query
|
|
per_page: 50 # max package names to process per PyPI search page
|
|
workers: 3 # parallel TruffleHog filesystem scans for downloaded packages
|
|
timeout: 300 # seconds before killing one package scan process tree
|
|
versions_per_package: 3 # scan latest N release artifacts per package within max_version_age_days
|
|
max_version_age_days: 90 # skip package releases older than this many days
|
|
max_artifact_size_mb: 300 # skip/download-fail package artifacts larger than this
|
|
fetch_timeout: 20 # seconds per PyPI request
|
|
|
|
package_git:
|
|
enabled: false # disabled after package-level discovery became duplicate-heavy
|
|
auth_pool: github_main # most package metadata points to GitHub; GitLab 401/403 falls back unauthenticated
|
|
auth_rotation: per_cycle
|
|
mode: search # package_git currently supports search mode and custom JSON/git URL targets
|
|
package_sources: # registries used for package -> repository discovery
|
|
- npm
|
|
- pypi
|
|
queries:
|
|
- ai
|
|
- agents
|
|
- assistant
|
|
- chatbot
|
|
- rag
|
|
- prompt
|
|
- completions
|
|
- mcp
|
|
- openrouter
|
|
- groq
|
|
- xai
|
|
- langchain
|
|
- litellm
|
|
- gemini
|
|
- qwen
|
|
- kimi
|
|
- dashscope
|
|
- aistudio
|
|
- OR
|
|
- chat
|
|
- chats
|
|
- conversation
|
|
- conversational
|
|
- dialogue
|
|
- copilot
|
|
- coding
|
|
- ai-assistant
|
|
- ai-agent
|
|
- multi-agent
|
|
- workflow
|
|
- workflows
|
|
- llmops
|
|
- benchmark
|
|
- tokens
|
|
- memory
|
|
- function-calling
|
|
pages: 10 # registry search pages per query for repo discovery
|
|
per_page: 50 # packages per search page
|
|
refresh_registry: true # merge cached repo candidates with fresh registry discovery each cycle
|
|
max_targets: 30 # bound one cycle so registry discovery cannot be starved by historical backlog
|
|
target_claim_order: balanced # split claims between oldest backlog and newest discovered repositories
|
|
workers: 3 # parallel TruffleHog git scans
|
|
timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes
|
|
versions_per_package: 3 # inspect repo metadata for up to N recent package versions
|
|
max_version_age_days: 90 # ignore package versions older than this during discovery
|
|
max_depth: 500 # git commit depth for package-derived repos
|
|
scan_full_history: false
|
|
max_commit_age_days: 0 # 0 avoids extra provider API commit-boundary lookup by default
|
|
commit_lookup_pages: 3
|
|
skip_if_commit_lookup_fails: false
|
|
fetch_timeout: 20 # seconds per registry metadata request
|
|
|
|
postman:
|
|
enabled: true # first run: enable manually for controlled backfill/tail scans
|
|
auth_pool: github_main # uses GitHub tokens for code search, commit lookup, and content download
|
|
auth_rotation: per_cycle # runner state still records a last auth; source-local pool rotates all tokens per request
|
|
mode: search # search = GitHub code search for Postman JSON artifacts; custom = target_file JSON targets
|
|
queries:
|
|
- anthropic
|
|
- gemini
|
|
- qwen
|
|
- kimi
|
|
- dashscope
|
|
- llm
|
|
- azure-openai
|
|
- openai.azure.com
|
|
- foundry
|
|
- services.ai.azure.com
|
|
- models.ai.azure.com
|
|
- DASHSCOPE_API_KEY
|
|
- QWEN_API_KEY
|
|
- MOONSHOT_API_KEY
|
|
- KIMI_API_KEY
|
|
- dashscope.aliyuncs.com
|
|
- api.moonshot.ai
|
|
- api.moonshot.cn
|
|
- GROQ_API_KEY
|
|
- api.groq.com
|
|
- OPENROUTER_API_KEY
|
|
- api.openrouter.ai
|
|
- REPLICATE_API_TOKEN
|
|
- api.replicate.com
|
|
- api.x.ai
|
|
- HF_TOKEN
|
|
- HUGGINGFACE_TOKEN
|
|
- ANTHROPIC_API_KEY
|
|
- api.anthropic.com
|
|
- rag
|
|
- agent
|
|
- assistant
|
|
search_kinds:
|
|
- all # Postman, Insomnia, Bruno, Thunder Client, Hoppscotch, and signature searches
|
|
pages: 2 # tail default; use 10 for one-time backfill up to GitHub's 1000-result cap
|
|
per_page: 100
|
|
workers: 3
|
|
timeout: 300
|
|
fetch_timeout: 20
|
|
max_targets: 0 # set a small value for smoke tests
|
|
max_file_age_days: 365 # backfill freshness filter by latest commit touching the file path; 0 disables
|
|
max_artifact_size_mb: 200
|
|
postman_cache_dir: "{postman_cache_dir}"
|
|
github_code_search_rpm: 8 # safe per-token code search request rate; GitHub limit is about 10/min/token
|
|
all_tokens_cooldown: 1800 # fallback sleep when all GitHub tokens are rate-limited and no reset is known
|
|
stop_on_seen_pages: true # tail mode: stop after consecutive fully known pages
|
|
seen_page_threshold: 2
|
|
min_pages_before_stop: 1
|