# Linux container profile; host runtime entrypoints remain deliberately disabled. # See DOCKER_MIGRATION.md. The Windows configuration is preserved in the initial Git commit. global: loop: true # true = run forever; false = run one full pass over enabled sources cooldown: 30 # seconds to sleep after one full pass over all enabled sources backlog_poll_sec: 0.5 # immediately refill released scan slots while durable queue work remains root_dir: "/opt/truf" # application image root, never the original Windows checkout project_dir: "{root_dir}/app" runtime_dir: "/data/runtime-linux" # new Linux state; do not reuse Windows control metadata postgres_data_dir: "/data/postgres-linux" # independently initialized, never a Windows cluster copy postgres_bin_dir: "/usr/lib/postgresql/16/bin" result_bundle_dir: "/data/scanner-result-bundles" result_bundle_max_event_bytes: 67108864 # hard limit per bundle; separate from remote reservation remote_assignment_reserve_bytes: 2097152 # bundle and projection baseline per unresolved remote assignment remote_assignment_max_active: 50 # global unresolved remote assignments across all users result_bundle_max_items: 10000 result_bundle_max_total_bytes: 3221225472 result_bundle_min_free_bytes: 21474836480 projection_backlog_max_items: 10000 projection_backlog_max_bytes: 2147483648 projection_backlog_headroom_bytes: 402653184 # one worst-case aggregate scan projection beyond all 3 physical slots keycheck_queue_max_items: 131072 keycheck_queue_max_bytes: 134217728 pipeline_quarantine_max_items: 10000 pipeline_quarantine_max_bytes: 1073741824 pipeline_metadata_retention_days: 30 pipeline_metadata_retirement_batch: 100 keycheck_candidates_per_event: 2000 keycheck_candidate_bytes_per_event: 2097152 keycheck_result_projection_reserve_bytes: 3145728 keycheck_recheck_batch_items: 10000 legacy_result_spool_dir: "{runtime_dir}/result_spool" legacy_result_spool_max_event_bytes: 201326592 legacy_result_spool_max_events: 10000 legacy_result_spool_max_total_bytes: 3221225472 results_dir: "{runtime_dir}/results" # findings, scan result JSONL/logs, and scanner.db queue_dir: "{runtime_dir}/queues" # todo_*.txt and checked_*.txt live here keycheck_dir: "{runtime_dir}/keychecks" # keychecker outputs grouped by service postman_cache_dir: "{runtime_dir}/postman_cache" # durable cached Postman collection/environment JSON postman_cache_max_items: 100000 # aggregate content-addressed artifacts; capacity failure stops discovery postman_cache_max_bytes: 21474836480 # artifact, metadata, and temporary bytes under the cache root postman_cache_min_free_bytes: 21474836480 # preserve the shared 20 GiB work-volume reserve postman_cache_lock_timeout_sec: 300 # match the bounded discovery window under concurrent cache publishers postman_discovery_max_artifacts_per_cycle: 1000 # shared non-package download/cache attempt cap postman_discovery_max_artifacts_per_page: 100 # bound one API page or GHArchive hour batch postman_discovery_max_bytes_per_cycle: 1073741824 # aggregate downloaded artifact bytes postman_discovery_max_elapsed_sec: 300 # includes network, cache scan, lock, and publication work postman_package_harvest_max_artifacts: 100 # matching package artifacts examined/published per target postman_package_harvest_max_bytes: 134217728 # aggregate matching artifact bytes considered per target postman_package_harvest_max_elapsed_sec: 30 # optional package harvesting wall-clock deadline postman_context_max_input_bytes: 16777216 postman_context_max_nodes: 100000 postman_context_max_depth: 64 postman_context_max_scalar_bytes: 16777216 postman_context_max_items: 50000 context_enrichment_max_source_bytes: 16777216 # aggregate optional-context source reads per target context_enrichment_max_findings: 2000 # stop optional enrichment without dropping later findings context_enrichment_max_postman_comparisons: 200000 # hard cap on fallback substring comparisons context_enrichment_max_elapsed_sec: 5 # aggregate optional-context wall-clock budget per target trufflehog_diagnostic_max_lines: 2000 # stop parsing stderr after bounded diagnostic work trufflehog_diagnostic_max_line_chars: 8192 trufflehog_diagnostic_max_line_bytes: 8192 trufflehog_diagnostic_max_errors: 200 trufflehog_diagnostic_max_warnings: 200 trufflehog_diagnostic_max_unclassified: 20 keycheck_input_max_line_bytes: 16777216 # canonical found_secrets producer/consumer JSONL line limit keycheck_candidate_artifact_max_items: 2000 # cap candidates derived from one scanned artifact keycheck_candidate_artifact_max_bytes: 2097152 keycheck_candidate_file_max_items: 100000 # aggregate bounded loader/writer limits keycheck_candidate_file_max_bytes: 33554432 keycheck_candidate_line_max_bytes: 8192 gharchive_cache_dir: "{state_dir}/gharchive_cache" # shared validated hourly .json.gz cache for both GHArchive sources gharchive_cache_max_items: 48 # aggregate retained hourly archives gharchive_cache_max_bytes: 8589934592 # compressed artifacts, lock metadata, and temporary bytes gharchive_cache_min_free_bytes: 5368709120 # preserve 5 GiB free on the runtime volume gharchive_download_max_bytes: 536870912 # per-hour compressed response cap gharchive_decompressed_max_bytes: 8589934592 # per-hour gzip expansion cap gharchive_max_events: 5000000 # per-hour event/line count cap gharchive_max_line_bytes: 8388608 # reject oversized individual JSON event lines gharchive_cache_lock_timeout_sec: 600 # bounded aggregate/per-hour cross-process lock wait state_dir: "{runtime_dir}/state" # runner state files log_dir: "{runtime_dir}/logs" # supervisor/source/dashboard logs control_dir: "/run/truf/control" # ephemeral container identity, never persisted across recreation proxy_file: "{runtime_dir}/proxy.txt" # shared proxy list for checkers/scanners api_proxy_enabled: true # discovery/metadata use proxy_file; artifact bodies explicitly bypass proxies api_proxy_file: "{proxy_file}" # host:port:user:pass or full proxy URL; resolved from proxy_file by default api_proxy_timeout: 5 # proxy connect cap; caller read timeout remains unchanged (fallback: 5 s) api_proxy_max_retries: 100 # default total attempts; source-specific attempts/deadlines take precedence api_proxy_retry_delay: 5 # seconds between API proxy retries download_proxy_enabled: false # reserved: keep heavy downloads/scans direct for now download_proxy_file: "" # reserved proxy list for package/artifact downloads when enabled later max_active_scans: 1 # conservative default within the container's shared memory budget opportunistic_scan_slots: 0 # host-memory-based opportunistic admission is disabled in containers opportunistic_scan_sources: [github, gitlab, huggingface] opportunistic_scan_reserve_overhead_bytes: 1073741824 # reserve Job cap plus 1 GiB process/staging overhead opportunistic_scan_min_available_after_reserve_bytes: 4294967296 # preserve 4 GiB physical RAM after admission opportunistic_scan_min_commit_after_reserve_bytes: 6442450944 # preserve 6 GiB commit headroom after admission scan_limiter_db: "{state_dir}/scan_limiter.db" # separate SQLite DB for cross-process scan slot leasing scan_slot_wait_sec: 0.5 # sleep between slot-acquire attempts when all scan slots are busy scan_slot_wait_log_sec: 30 # log long waits at this interval scan_slot_stale_sec: 7200 # clean slots older than this or owned by dead scanner PIDs target_retry_max_attempts: 3 # total target attempts before a terminal failed queue state target_retry_base_delay_sec: 3600 # transient target retry delay; doubles after each failed attempt target_retry_max_delay_sec: 86400 # cap exponential target retry delay at 24 hours target_timeout_retry_delay_sec: 21600 # timed-out targets use a non-terminal slow retry after 6 hours target_claim_batch_size: 1 # fallback only; PostgreSQL slot-first dispatch claims exactly acquired capacity admission_resolution_attempts: 300 # exact-token probes after an ambiguous PostgreSQL admission response admission_resolution_seconds: 300 # cover 45s loss grace, 60s stable-ready gate, and reconnect margin admission_resolution_retry_delay_sec: 1 # bounded pause between fast failed recovery probes sync_file_queues: false # legacy todo/checked import is complete; PostgreSQL is authoritative dockerhub_tag_cache_path: "{state_dir}/dockerhub_tag_cache.sqlite" # cache Docker Hub tag resolutions/rate limits dockerhub_tag_cache_ttl_sec: 21600 # successful tag resolutions are reused for 6 hours dockerhub_tag_negative_cache_ttl_sec: 3600 # empty/not-found tag lookups are retried after 1 hour dockerhub_tag_rate_limit_cache_ttl_sec: 1800 # global Docker tag API cooldown fallback dockerhub_tag_cache_max_rows: 50000 dockerhub_tag_cache_max_age_sec: 604800 dockerhub_tag_cache_max_bytes: 268435456 dockerhub_tag_cache_min_free_bytes: 536870912 database_path: "{results_dir}/scanner_active.db" # SQLite fallback observability DB when SCANNER_DB_URL is empty database_url: "" # Postgres DSN comes from SCANNER_DB_URL; keep real credentials out of config dashboard_db_path: "{database_path}" # SQLite dashboard fallback dashboard_db_url: "" # optional dashboard Postgres DSN override; defaults to SCANNER_DB_URL when empty dashboard_immutable_db: false # active DB is read-only via SQLite mode=ro, but not immutable because WAL changes jsonl_rotation_enabled: true # rotate large runtime JSONL files instead of growing multi-GB active files found_secrets_max_mb: 128 # rotate found_secrets.jsonl after this active-file size scan_results_max_mb: 256 # rotate scan_results.jsonl after this active-file size scan_errors_max_mb: 32 # bound each scan_errors.log segment scan_errors_keep: 5 # retain at most this many rotated scan error segments jsonl_lock_stale_sec: 300 # stale lock cleanup for cross-process JSONL rotation jsonl_max_segments: 16 # hard cap; publication pauses until registered consumers catch up jsonl_ledger_max_rows: 1000000 # durable O(1) publication identity bound jsonl_ledger_max_bytes: 536870912 jsonl_legacy_index_max_bytes: 16777216 # larger existing files require offline ledger reconciliation jsonl_tail_scan_max_bytes: 8388608 jsonl_torn_quarantine_max_bytes: 65536 detectors: "" # empty = use all TruffleHog detectors; set IDs to limit intentionally exclude_detectors: "github.v1,gitlab.v1,GitHubOauth2" # drop noisy legacy GitHub/GitLab detectors; keep modern prefixes no_verification: true # pass --no-verification to TruffleHog; local checkers classify live/dead later strict_git_provider_token_filter: true # drop unverified GitHub/GitLab detections that do not match known token prefixes drop_detectors: "Privacy,URI,JDBC,Postgres,MongoDB,SQLServer,Box,ZohoCRM,Accuweather,Roaring,Flatio,LinkPreview,RailwayApp" # do not persist obvious non-keycheckable/generic noise detectors versions_per_package: 3 # npm/PyPI: scan up to N recent versions per matching package work_dir: "/data/scanner-work" # isolated scratch root for clone/download/extract folders trufflehog_stdout_max_mb: 32 # hard file-backed streamed stdout byte bound per scan trufflehog_stderr_max_mb: 8 # hard file-backed streamed diagnostic byte bound per scan trufflehog_config: "{project_dir}/trufflehog-custom-detectors.yaml" # custom detectors loaded by TruffleHog --config trufflehog_job_memory_limit_bytes: 4294967296 # aggregate Windows Job limit for each TruffleHog tree (4096 MiB) trufflehog_windows_job_cpu_weight: 2 # normal priority with low relative Job weight; avoids broken BELOW_NORMAL startup trufflehog_windows_memory_priority: 4 # reclaim scan pages before normal-priority desktop working sets min_free_gb: 20 # minimum free space required in work_dir before starting new scans state_file: "{state_dir}/runner_state.json" # stores current query index for each source secrets_file: "/data/config/secrets.yaml" # runtime-owned private credentials, never baked into the image stop_on_seen_pages: false # default false; enable per source where API pagination is sequential seen_page_threshold: 2 # stop after this many consecutive all-known pages min_pages_before_stop: 1 # always fetch at least this many pages before early stop supervisor: enabled_sources: [gitlab, dockerhub, huggingface] # exact distributed discovery-producer profile interactive: false # canonical foreground container supervisor, no terminal dependency autostart: true # start only the explicit core allowlist above poll_sec: 1.0 # how often supervisor checks child process/log state heartbeat_sec: 60 # rewrite background status snapshot at least this often; 0 disables heartbeat authority_check_interval_sec: 5 # detect code/config authority drift within this bound pipeline_status_refresh_sec: 2 # cache authenticated pipeline status queries between loop ticks postgres_health_interval_sec: 15 # bounded authenticated controller probe interval postgres_ready_loss_grace_sec: 45 # tolerate transient loss for the same live authenticated postmaster postgres_stable_ready_sec: 60 # dependency gate opens only after readiness remains stable this long postgres_connect_timeout_sec: 5 # bounded controller authentication connect timeout postgres_query_timeout_ms: 5000 # bounded controller identity/readiness query timeout postgres_stop_timeout_sec: 60 # pg_ctl's bounded identity-verified coordinated stop timeout postgres_start_settle_timeout_sec: 30 # bound late postmaster publication checks after pg_ctl start -W postgres_shutdown_timeout_sec: 120 # total supervisor wait for the controller during coordinated shutdown postgres_log_max_mb: 64 # collector/startup log segment bound postgres_log_keep: 24 # maximum retained collector and startup log segments refresh_sec: 5 # non-interactive stdout/status-loop interval log_dir: "{log_dir}" # one appended child-process log per source log_max_mb: 64 # live source output rotates at this active-segment bound log_keep: 5 # keep this many rotated log segments per log file control_dir: "{control_dir}" # hardened directory; links/junctions are rejected instance_file: "{control_dir}/supervisor.instance.json" # private authenticated process/control identity lock_file: "{control_dir}/supervisor.lock" # secondary per-instance lock; cluster authority is data-dir-derived and non-configurable supervisor_log: "{log_dir}/supervisor.log" # stdout/stderr for background supervisor status_file: "{log_dir}/supervisor.status.txt" # latest background supervisor status table control_host: "127.0.0.1" # local only; do not expose externally control_port: 8765 background_start_timeout_sec: 20 # wait for matching private metadata and authenticated handshake background_shutdown_timeout_sec: 180 # wait on the retained verified supervisor process handle attach_poll_sec: 0.2 # --attach watch-mode poll interval dashboard_log: "{log_dir}/dashboard.log" # stdout/stderr for supervisor-launched dashboard state_dir: "{state_dir}" # per-source runner_state_*.json files to avoid parallel write races per_source_state: true # true = supervisor sets RUNNER_STATE_FILE per child process result_ingester: enabled: true poll_sec: 0.2 lease_seconds: 300 jsonl_projector: enabled: true poll_sec: 0.2 lease_seconds: 300 worker_api: enabled: false # fail closed; configure profiles and private ingress before enabling address: "127.0.0.1" # raw API must remain loopback/private and unexposed port: 8766 sources: [] # optional narrowing of exact protocol-2 package capability triples auth_entries: {} # GitLab issuance plus optional legacy GitHub reconciliation auth entries compatibility_profiles: {} # trusted package manifests; never inferred from clients assignment_ttl_seconds: 86400 # immutable server-time deadline, default 24 hours assignment_ttl_seconds_by_source: {} max_bundle_bytes: 67108864 # hard-capped again by global result_bundle_max_event_bytes reaper_interval_seconds: 60 reaper_batch_size: 1000 limit_concurrency: 64 body_idle_timeout_seconds: 30 json_body_timeout_seconds: 60 bundle_body_timeout_seconds: 1800 admin: enabled: false # fail closed; available only behind authenticated Caddy origin: "" # exact public HTTPS origin when enabled edge_marker: "" # independent 256-bit secret shared only with Caddy max_body_bytes: 8192 snapshot_limit: 200 requeue_limit: 100 managed_file_roots: runtime-keychecks: path: "/data/runtime-linux/keychecks" permissions: list: true read: true create_replace: false delete: false limits: max_relative_path_bytes: 1024 max_component_bytes: 255 max_path_depth: 16 max_listing_entries: 500 max_listing_bytes: 262144 max_file_bytes: 67108864 runtime-logs: path: "/data/runtime-linux/logs" permissions: list: true read: true create_replace: false delete: false limits: max_relative_path_bytes: 1024 max_component_bytes: 255 max_path_depth: 16 max_listing_entries: 500 max_listing_bytes: 262144 max_file_bytes: 67108864 runtime-results: path: "/data/runtime-linux/results" permissions: list: true read: true create_replace: false delete: false limits: max_relative_path_bytes: 1024 max_component_bytes: 255 max_path_depth: 16 max_listing_entries: 500 max_listing_bytes: 262144 max_file_bytes: 268435456 docker_shadow: enabled: true # manual-only operator command; never autostarted cohort_size: 50 lease_seconds: 3600 janitor: enabled: true interval_sec: 60 minimum_age_sec: 7200 max_candidates: 50 max_entries: 10000 max_bytes: 1073741824 max_seconds: 30 max_depth: 64 interval: 300 # default delay before repeating a --once child source restart_delay: 30 # initial restart delay after failures or unexpected exits max_restart_delay: 600 # cap for exponential restart backoff restart_reset_after: 300 # clear failure streak/old exit after this many stable seconds dashboard: enabled: false # true = supervisor also starts dashboard.py address: "127.0.0.1" # local-only dashboard bind address port: 5000 startup_grace_sec: 30 # allow Streamlit to initialize before a failed health probe triggers restart health_interval_sec: 2 # bounded asynchronous /_stcore/health probe interval health_timeout_sec: 1 restart_base_sec: 2 # exponential dashboard-only restart backoff restart_max_sec: 60 stable_health_sec: 60 # reset dashboard restart streak only after stable health defaults: enabled: true once: false # false = child console_runner loops internally; true = one source cycle per child run repeat: true # only relevant when once=true; repeat one-shot cycles after interval restart: true # restart crashed/exited child processes extra_args: [] # extra args passed to console_runner.py in config mode sources: github: once: false github_archive: enabled: true # controlled broad GitHub discovery via GHArchive; start manually from supervisor use_system_proxy: true once: false repeat: true restart: true interval: 3600 github_archive_files: enabled: true # controlled GHArchive changed-file fetch; start manually from supervisor use_system_proxy: true once: false repeat: true restart: true interval: 1800 github_gists: enabled: true # controlled public gist discovery; start manually from supervisor once: false repeat: true restart: true interval: 1800 gitlab: once: false github_actions: enabled: false # paused after fresh and retained cohorts produced no strict-usable yield once: false repeat: true restart: true interval: 3600 gitlab_ci: enabled: true # show in supervisor; main source remains disabled for normal all-source runner once: false repeat: true restart: true interval: 3600 dockerhub: once: false npm: once: false pypi: once: false package_git: enabled: false once: false huggingface: once: false postman: enabled: true # allows `supervisor --sources postman`; main sources.postman.enabled controls default all-source inclusion once: false env: PYTHONIOENCODING: utf-8 # avoid Windows console codec crashes on Unicode repository paths keychecks: enabled: true # supervisor manages this as pseudo-source "keychecks" autostart: true # start keychecks when supervisor starts, even if scanner sources wait for manual start input_mode: postgres # PostgreSQL candidate leases are authoritative; JSONL requires explicit compatibility mode services: all # all or comma/list: openai,anthropic,qwen,kimi,github,... interval: 3600 # repeat keycheck_runner every N seconds when in repeat/hourly mode repeat: true restart: false # do not auto-restart failed checker batch; wait for next interval/manual restart max_keys: 0 # 0 = no per-run limit; set N for throttled hourly batches scheduler_batch_keys: 1000 # per-provider process slice; max_keys=0 keeps rotating until empty/deadline scheduler_workers: 1 # serialize provider handshakes; avoids transient control-plane startup failures scheduler_deadline_sec: 1800 # aggregate work-conserving provider deadline retry_network: true # retry transient network failures each scheduled run retry_limited: false # set true to retry rate-limited/no-quota statuses hourly retry_unknown: false retry_restricted: false retry_no_balance: false # retry no_balance/no_quota statuses when explicitly enabled retry_valid: false # valid provider keys are re-probed only when explicitly requested recheck_all: false # true forces all known keys to be checked again env: KEYCHECK_INPUT_TAIL_MB: "0" # first keycheck reads from offset 0; high-watermark state handles later appends KEYCHECK_INPUT_MAX_LINE_BYTES: "16777216" # must match global.keycheck_input_max_line_bytes KEYCHECK_CANDIDATE_MAX_UNCONSUMED_ATTEMPTS: "3" KEYCHECK_DB_INGEST_TAIL_MB: "64" # optional tail size; initial DB ingest uses full bounded offsets unless KEYCHECK_DB_INGEST_TAIL_INITIAL=1 KEYCHECK_RESULTS_MAX_MB: "32" # rotate per-service *Results.jsonl files into manifest segments KEYCHECK_PROVIDER_RESOLUTION_ORDER: "deepseek,zai,qwen,kimi" KEYCHECK_EVENT_MAP_BACKFILL_ROWS: "0" # one-time legacy backfill is complete; live ingest maintains this map atomically KEYCHECK_UID_MAP_BACKFILL_ROWS: "0" # scanner writes finding_uid_map for all new findings db_ingest: enabled: false # explicit JSONL compatibility import only repair_links: enabled: false # new DB candidates carry exact finding attribution service_args: # provider-specific probe flags passed by keycheck_runner gemini: - --probe-generation # call generateContent; RATE_LIMITED valid keys go to geminiAliveRateLimited.txt aws: - --probe-bedrock # after STS, probe Bedrock access using safe validation-style calls - --bedrock-max-attempts - "12" azure: - --probe-openai-route # after deployments list, probe Azure OpenAI chat route without generation - --probe-foundry-route # probe Azure AI Foundry/MaaS route for configured models - --foundry-models - claude-opus-4-6,claude-fable-5 - --timeout - "8" gcp: - --probe-vertex # after OAuth, probe Vertex AI Gemini countTokens access - --vertex-timeout - "6" - --vertex-max-attempts - "6" - --vertex-models - gemini-3.6-flash,gemini-3.1-pro-preview - --vertex-locations - global,us,eu - --vertex-anthropic-models - claude-opus-5,claude-opus-4-7,claude-opus-4-6,claude-fable-5 - --vertex-anthropic-locations - global,us,eu,us-east5,europe-west1 - --vertex-anthropic-max-attempts - "20" summary_tsv: "{keycheck_dir}/summary.tsv" summary_json: "{keycheck_dir}/summary.json" alive_summary_tsv: "{keycheck_dir}/alive_summary.tsv" query_policy: rejected: # reviewed source-specific zero-alive evidence; queue rows remain auditable and reversible - {source: dockerhub, query: OR, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1631, findings: 29137, unique_credentials: 36, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 36} - {source: dockerhub, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1009, findings: 69486, unique_credentials: 30, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 229} - {source: github, query: coding, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1696, findings: 251, unique_credentials: 11, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 0} - {source: github, query: memory, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3190, findings: 254, unique_credentials: 11, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 8} - {source: npm, query: OR, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1985, findings: 152, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 598} - {source: npm, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 9446, findings: 5833, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3957} - {source: npm, query: agents, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 8936, findings: 4573, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3346} - {source: npm, query: ai, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 7676, findings: 49499, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3663} - {source: npm, query: assistant, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3888, findings: 14831, unique_credentials: 2, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 899} - {source: npm, query: benchmark, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2371, findings: 298, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1292} - {source: npm, query: bot, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1784, findings: 2081, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 951} - {source: npm, query: chat, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2048, findings: 381, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2087} - {source: npm, query: chats, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1297, findings: 107, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 719} - {source: npm, query: completion, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2529, findings: 1351, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1263} - {source: npm, query: completions, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3058, findings: 1169, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1466} - {source: npm, query: conversation, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1454, findings: 475, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 954} - {source: npm, query: gemini, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 4922, findings: 3830, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 891} - {source: npm, query: groq, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2050, findings: 333, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 484} - {source: npm, query: langchain, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1264, findings: 169, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 908} - {source: npm, query: llm, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3682, findings: 17976, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2333} - {source: npm, query: mcp, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 8704, findings: 5572, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 3524} - {source: npm, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 6262, findings: 3288, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1359} - {source: npm, query: openrouter, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1590, findings: 355, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1014} - {source: npm, query: prompt, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1899, findings: 91, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1161} - {source: npm, query: rag, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2242, findings: 1912, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 544} - {source: npm, query: semantic, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1911, findings: 164, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1168} - {source: npm, query: studio, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 4158, findings: 484, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1349} - {source: npm, query: xai, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1390, findings: 2395, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 513} - {source: package_git, query: agent, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1851, findings: 5395, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 8791} - {source: package_git, query: bot, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1610, findings: 7260, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 4486} - {source: package_git, query: completion, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3149, findings: 10028, unique_credentials: 9, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 1507} - {source: package_git, query: llm, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1285, findings: 2116, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 5572} - {source: package_git, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2053, findings: 14375, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 9968} - {source: package_git, query: semantic, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 3895, findings: 24064, unique_credentials: 22, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 2414} - {source: package_git, query: studio, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1028, findings: 2666, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 5709} - {source: postman, query: XAI_API_KEY, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 2935, findings: 83, unique_credentials: 1, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 306} - {source: pypi, query: langchain, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1095, findings: 47, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 16} - {source: pypi, query: open, status: rejected_zero_alive, evidence_cutoff: "2026-09-07T17:38:01+00:00", successful_scans: 1253, findings: 731, unique_credentials: 0, pending_candidates: 0, ever_alive_credentials: 0, reviewed_queue_rows: 66} sources: github: enabled: false # direct GitHub discovery is paused in favor of the Actions experiment auth_pool: github_main # auth_pools. in secrets.yaml; leave empty to use env/legacy token auth_rotation: per_cycle # per_cycle = use next token each source cycle; none = first available retry_with_next_auth_on_rate_limit: true # switch token and retry when GitHub API rate-limits rate_limit_cooldown: 3600 # seconds to cool down a rate-limited token if API gives no reset time mode: recent # recent = updated recently; search = paginated search; custom = target_file URLs queries: # one query is used per source cycle; state rotates through this list - llm - ai - agent - agents - assistant - bot - chatbot - rag - semantic - prompt - completion - completions - mcp - claude - anthropic - opus - sonnet - haiku - vertex - aiplatform - bedrock - foundry - azure-openai - openrouter - langchain - langgraph - litellm - crewai - autogen - semantic-kernel - mistral - groq - cohere - xai - together - perplexity - gemini - qwen - kimi - dashscope - aistudio - studio - OR - open - chat - chats - conversation - conversational - dialogue - OPENAI_API_KEY # bounded high-signal README integration query - api.openai.com # bounded high-signal README endpoint query - openai-agents # bounded OpenAI agent SDK/ecosystem query - copilot - ai-assistant - ai-agent - multi-agent - workflow - workflows - llmops - benchmark - tokens - function-calling query_overrides: OPENAI_API_KEY: pages: 1 per_page: 25 max_targets: 5 api.openai.com: pages: 1 per_page: 25 max_targets: 5 openai-agents: pages: 1 per_page: 50 max_targets: 5 pages: 5 # pages to fetch in search/recent mode per_page: 100 # targets per API page, max is usually 100 workers: 1 # one shared non-Docker scan slot for this source timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes recent_hours: 24 # recent mode: only repos updated within this many hours are discovered max_repo_age_days: 90 # skip repos older than this by repo_age_field before queueing repo_age_field: updated_at # metadata field for repo age: created_at, updated_at, pushed_at max_commit_age_days: 90 # scan only recent git history by passing --since-commit boundary commit_lookup_pages: 3 # API pages of commits to inspect when finding --since-commit boundary skip_if_commit_lookup_fails: true # true = do not scan if recent commit boundary cannot be found max_depth: 100 # TruffleHog --max-depth; extra cap on commit count scanned exact_git_planning_enabled: true # resolve and bind the exact default/explicit ref head after claim git_baseline_depth: 100 # bounded first/reset scan depth; deltas use only the covered base git_ref_resolution_timeout_sec: 10 git_ref_resolution_attempts: 2 git_ref_resolution_max_bytes: 1048576 scan_full_history: false # true = ignore max_depth; still uses max_commit_age_days when configured sort_by: updated # GitHub search sort: updated, stars, forks, created sort_order: desc # desc = newest/highest first; asc = oldest/lowest first created_filter: any # optional GitHub created filter: any, today, week, month, year stop_on_seen_pages: true # stop fetching pages after consecutive all-known pages seen_page_threshold: 2 min_pages_before_stop: 1 updated_target_rescan_enabled: true # preserve pushed_at and rescan completed repos after newer pushes updated_target_rescan_max_per_cycle: 1 # bounded rollout: at most one changed repo per discovery cycle updated_target_rescan_cooldown_hours: 24 github_archive: enabled: false # broad discovery source; supervisor exposes it for manual/controlled runs use_system_proxy: true # direct GHArchive route is unstable; use inherited HTTP(S)_PROXY only for hourly dumps auth_pool: github_main auth_rotation: per_cycle mode: archive # GHArchive hourly events -> GitHub repos -> recent git scan queries: - gharchive archive_hours_back: 6 # read recent completed GHArchive hours fetch_timeout: 120 # read timeout while downloading large hourly gzip archives archive_max_repos_per_cycle: 500 # hard cap after event/repo dedupe/scoring archive_rescan_cooldown_hours: 48 # allow rescanning active repos after cooldown; not forever-checked archive_event_types: - PushEvent - CreateEvent - PublicEvent workers: 4 timeout: 1800 max_depth: 75 scan_full_history: false max_commit_age_days: 0 # avoid GitHub commit-boundary API lookups for broad source commit_lookup_pages: 0 skip_if_commit_lookup_fails: false stop_on_seen_pages: false github_archive_files: enabled: false # fetch suspicious changed files from GHArchive PushEvents use_system_proxy: true auth_pool: github_main auth_rotation: per_cycle mode: archive queries: - gharchive-files archive_hours_back: 6 archive_max_files_per_cycle: 100 archive_max_commit_lookups: 200 archive_event_types: - PushEvent workers: 1 timeout: 300 fetch_timeout: 120 max_artifact_size_mb: 2 stop_on_seen_pages: false github_gists: enabled: false # public gist source; supervisor exposes it for manual/controlled runs auth_pool: github_main auth_rotation: per_cycle mode: search queries: - gists pages: 3 per_page: 100 workers: 3 timeout: 300 fetch_timeout: 20 max_artifact_size_mb: 2 stop_on_seen_pages: true seen_page_threshold: 2 min_pages_before_stop: 1 gitlab: enabled: true # include the GitLab discovery producer in distributed runs target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission auth_pool: gitlab_main # auth_pools. in secrets.yaml; leave empty to use env/legacy token auth_rotation: per_cycle # per_cycle = use next token each source cycle; none = first available retry_with_next_auth_on_rate_limit: true # switch token and retry when GitLab API returns 429 rate_limit_cooldown: 3600 # seconds to cool down a rate-limited token if API gives no reset time discovery_request_attempts: 3 # bounded retries for idempotent project discovery transport failures discovery_retry_delay: 5 # seconds between GitLab discovery transport attempts external_trufflehog_lifecycle: true # bypass embedded overseer and require explicit completion for GitLab scans mode: recent # recent = last_activity_after; search = paginated project search; custom = target_file URLs queries: # one query is used per source cycle; state rotates through this list - llm - ai - agent - agents - assistant - bot - chatbot - rag - openai-api # bounded metadata-friendly OpenAI API query - openai-agents # bounded OpenAI agent SDK/ecosystem query - librechat # bounded deployable chat application query - semantic - prompt - completion - completions - mcp - openrouter - groq - xai - langchain - litellm - gemini - qwen - kimi - dashscope - aistudio - studio - OR - open - chat - chats - conversation - conversational - dialogue - copilot - coding - ai-assistant - ai-agent - multi-agent - workflow - workflows - llmops - benchmark - tokens - memory - function-calling - hermes - harnes - flow - helpdesk - paperless query_overrides: openai-api: pages: 1 per_page: 50 max_targets: 5 openai-agents: pages: 1 per_page: 50 max_targets: 5 librechat: pages: 1 per_page: 25 max_targets: 5 hermes: pages: 1 per_page: 50 max_targets: 5 harnes: pages: 1 per_page: 50 max_targets: 5 flow: pages: 1 per_page: 50 max_targets: 5 helpdesk: pages: 1 per_page: 50 max_targets: 5 paperless: pages: 1 per_page: 50 max_targets: 5 pages: 10 # pages to fetch in search/recent mode per_page: 100 # targets per API page, max is usually 100 workers: 1 # one shared non-Docker scan slot for this source timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes recent_hours: 96 # recent mode: only projects active within this many hours are discovered max_repo_age_days: 90 # skip projects older than this by repo_age_field before queueing repo_age_field: last_activity_at # metadata field: created_at, updated_at, last_activity_at max_commit_age_days: 90 # scan only recent git history by passing --since-commit boundary commit_lookup_pages: 3 # API pages of commits to inspect when finding --since-commit boundary skip_if_commit_lookup_fails: true # true = do not scan if recent commit boundary cannot be found max_depth: 100 # TruffleHog --max-depth; extra cap on commit count scanned exact_git_planning_enabled: true # resolve and bind the exact default/explicit ref head after claim git_baseline_depth: 100 # bounded first/reset scan depth; deltas use only the covered base git_ref_resolution_timeout_sec: 10 git_ref_resolution_attempts: 2 git_ref_resolution_max_bytes: 1048576 scan_full_history: false # true = ignore max_depth; still uses max_commit_age_days when configured sort_by: last_activity_at # GitLab order_by field: last_activity_at, created_at, updated_at, name, stars sort_order: desc # desc = newest/highest first; asc = oldest/lowest first visibility: public # public, internal, private; private/internal require token permissions stop_on_seen_pages: true # stop fetching pages after consecutive all-known pages seen_page_threshold: 2 min_pages_before_stop: 1 updated_target_rescan_enabled: true # preserve last_activity_at and rescan completed projects after new activity updated_target_rescan_max_per_cycle: 1 updated_target_rescan_cooldown_hours: 24 github_actions: enabled: false # paused; queue and historical evidence remain intact auth_pool: github_main auth_rotation: per_cycle mode: recent queries: - logs ci_seed_sources: github,github_archive,package_git ci_max_repos_per_cycle: 25 ci_seed_scan_limit: 15000 ci_seed_query_batch_size: 250 # bound each PostgreSQL history page on HDD ci_soft_cooldown_days: 7 ci_runs_per_repo: 5 ci_lookback_days: 30 ci_max_log_archive_mb: 150 ci_max_log_file_mb: 50 ci_scan_artifacts: true ci_max_artifacts_per_run: 3 ci_max_artifact_archive_mb: 150 ci_max_artifact_file_mb: 50 ci_max_artifact_files: 1000 ci_failed_first: true refresh_registry: true # keep discovering recent repositories while historical targets remain queued target_claim_order: balanced # split work between fresh discoveries and the retained historical backlog workers: 1 timeout: 180 gitlab_ci: enabled: false # controlled experiment: run manually from supervisor auth_pool: gitlab_main auth_rotation: per_cycle mode: recent queries: - logs ci_seed_sources: gitlab,package_git ci_max_repos_per_cycle: 25 ci_seed_scan_limit: 15000 ci_seed_query_batch_size: 250 # bound each PostgreSQL history page on HDD ci_soft_cooldown_days: 7 ci_pipelines_per_project: 5 ci_jobs_per_pipeline: 20 ci_lookback_days: 30 ci_max_trace_mb: 100 ci_scan_artifacts: true ci_max_artifacts_per_pipeline: 5 ci_max_artifact_archive_mb: 150 ci_max_artifact_file_mb: 50 ci_max_artifact_files: 1000 workers: 3 timeout: 180 huggingface: enabled: true # discover newest HuggingFace Spaces for remote workers target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission auth_pool: huggingface_main # optional auth_pools. in secrets.yaml or use HF_TOKEN/HUGGINGFACE_TOKEN auth_rotation: per_cycle mode: recent # recent/search both fetch newest spaces; custom = target_file space IDs queries: - spaces # placeholder query for state rotation; HuggingFace API fetch ignores it for now pages: 100 # bounded cursor pages from the newest-lastModified Spaces API per_page: 100 # newest-modified API page size is fixed at 100 by the runner workers: 1 timeout: 1800 fetch_timeout: 15 discovery_request_attempts: 3 # one transient page timeout must not restart the whole source discovery_retry_delay: 5 stop_on_seen_pages: true seen_page_threshold: 2 min_pages_before_stop: 1 updated_target_rescan_enabled: true # preserve lastModified and revisit changed completed Spaces updated_target_rescan_max_per_cycle: 1 updated_target_rescan_cooldown_hours: 24 dockerhub: enabled: true # include the DockerHub discovery producer in distributed runs target_claim_order: oldest # oldest, newest, or balanced PostgreSQL queue admission require_digest: true # only immutable repo@sha256:... image targets may reach TruffleHog auth_pool: dockerhub_main # auth_pools. in secrets.yaml; all available accounts are rotated per image scan auth_rotation: per_scan # DockerTokenManager rotates Docker accounts for each docker scan retry_with_next_auth_on_rate_limit: true # rotate Hub/Registry requests to another account after 429/invalid auth rate_limit_cooldown: 1800 # per-account cooldown; shared cooldown starts only after pool exhaustion mode: search # search = Docker Hub search; recent = client-side recent tag filtering; custom = target_file images refresh_registry: true # discover fresh images even while deferred targets remain queued queries: # one query is used per source cycle; Docker Hub empty query returns no results - llm - ai - agents - assistant - bot - chatbot - rag - semantic - prompt - completion - completions - mcp - openrouter - groq - xai - langchain - litellm - gemini - qwen - kimi - dashscope - aistudio - studio - librechat # bounded deployable chat application query - lobechat # bounded deployable chat application query - openai-proxy # bounded OpenAI-compatible proxy query - open - chat - chats - conversation - conversational - dialogue - copilot - coding - ai-assistant - ai-agent - multi-agent - workflow - workflows - llmops - benchmark - tokens - memory - function-calling - hermes - harnes - flow - helpdesk - paperless - open-webui - ragflow - dify - flowise - crewai - n8n - langflow - autogen - browser-use - openhands - anythingllm - agent-zero query_overrides: librechat: max_targets: 10 lobechat: max_targets: 10 openai-proxy: max_targets: 10 hermes: max_targets: 10 harnes: max_targets: 10 flow: max_targets: 10 helpdesk: max_targets: 10 paperless: max_targets: 10 pages: 30 # maximum Docker Hub search pages for every configured query per_page: 100 # maximum Docker Hub results per search page workers: 2 # allow two Docker image scans within the global three-slot limit trufflehog_job_memory_limit_bytes: 6442450944 # Docker images get 6 GiB; other sources retain the 4 GiB global cap timeout: 600 # bounded full-image scan window after disabling TruffleHog's embedded overseer trufflehog_concurrency: 4 # bound per-image layer/chunk fan-out; two source workers remain available recent_days: 7 # recent mode: keep images/tags updated within this many days fetch_workers: 5 # parallel Docker Hub search page fetches before scanning fetch_timeout: 15 # seconds per Docker Hub API request tag_fetch_workers: 4 # bound the in-flight burst before a shared 429 cooldown becomes visible tag_retry_count: 0 # do not retry individual transport failures during high-volume tag resolution tag_retry_delay: 10 # base delay if bounded non-429 transport retries are enabled later tag_resolve_limit: 100 # max old bare todo repos to resolve to tags per cycle; 0 = all docker_platform_filter_enabled: true # skip tags only when complete metadata proves linux/amd64 is absent docker_platform_os: linux docker_platform_arch: amd64 docker_platform_candidate_tags: 20 # inspect extra recent tags so an ARM-only latest tag does not hide an amd64 tag docker_images_per_repository: 3 # ordinary FIFO resolver stays at the reviewed production depth docker_content_scan_mode: canary # only prior durable full-image timeouts are eligible for layer fallback docker_layer_canary_basis_points: 10000 # all prior durable full-image timeouts use bounded layer fallback docker_adaptive_canary_basis_points: 0 # disabled until an exact aggregate shadow report passes every gate docker_adaptive_gate_max_age_sec: 604800 # matching shadow evidence expires after seven days docker_layer_config_max_bytes: 1048576 # image config is selected independently from layer budget docker_layer_max_bytes: 268435456 # max compressed bytes for one selected application layer docker_layer_image_max_bytes: 1073741824 # cumulative compressed layer budget per immutable image docker_layer_max_layers: 8 # select application layers from top to base within the byte budget docker_layer_archive_max_size_bytes: 268435456 # TruffleHog per-member extraction bound docker_layer_archive_max_depth: 4 # required for an OCI wrapper plus compressed layer archive docker_layer_archive_timeout_sec: 30 docker_layer_blob_timeout_sec: 600 # shared transfer+filesystem scan deadline per durable checkpoint docker_layer_filesystem_concurrency: 2 docker_layer_blob_max_attempts: 3 # blob budget is authoritative; parent checkpoint attempts reset docker_layer_blob_lease_sec: 1800 # exceeds the bounded 600-second execution plus handoff margin docker_layer_min_free_bytes: 21474836480 # retain 20 GiB on the private work volume docker_layer_checkpoint_delay_sec: 60 # resume the next selected blob without a full-image restart docker_adaptive_checkpoint_max_blobs: 4 # scheduling-only bound for a future gated adaptive checkpoint docker_adaptive_checkpoint_max_bytes: 536870912 # aggregate compressed bytes per adaptive checkpoint docker_repository_refresh_interval_sec: 86400 # revisit each resolved repository daily for new immutable digests docker_repository_refresh_max_per_cycle: 0 # disable periodic refresh of completed repository anchors sort_by: updated_at # Docker Hub search sort field sort_order: desc # desc = newest/highest first; asc = oldest/lowest first npm: enabled: true # true = include npm registry in config-mode runs mode: search # npm currently supports search mode queries: # one query is used per source cycle; state rotates through this list - chatbot - litellm - qwen - kimi - dashscope - aistudio - conversational - dialogue - copilot - coding - ai-assistant - ai-agent - multi-agent - workflow - workflows - llmops - tokens - memory - function-calling pages: 30 # npm search pages to fetch for the current query per_page: 50 # npm packages per search page workers: 3 # parallel TruffleHog filesystem scans for downloaded packages timeout: 300 # seconds before killing one package scan process tree versions_per_package: 3 # scan latest N versions per package within max_version_age_days max_version_age_days: 90 # skip package versions older than this many days max_artifact_size_mb: 300 # skip/download-fail package tarballs larger than this fetch_timeout: 20 # seconds per npm registry request pypi: enabled: true # true = include PyPI registry in config-mode runs mode: search # PyPI currently supports search mode queries: # one query is used per source cycle; state rotates through this list - llm - ai - agent - agents - assistant - bot - chatbot - rag - semantic - prompt - completion - completions - mcp - openrouter - groq - xai - litellm - gemini - qwen - kimi - dashscope - aistudio - studio - OR - chat - chats - conversation - conversational - dialogue - copilot - coding - ai-assistant - ai-agent - multi-agent - workflow - workflows - llmops - benchmark - tokens - memory - function-calling pages: 40 # PyPI search pages to fetch for the current query per_page: 50 # max package names to process per PyPI search page workers: 3 # parallel TruffleHog filesystem scans for downloaded packages timeout: 300 # seconds before killing one package scan process tree versions_per_package: 3 # scan latest N release artifacts per package within max_version_age_days max_version_age_days: 90 # skip package releases older than this many days max_artifact_size_mb: 300 # skip/download-fail package artifacts larger than this fetch_timeout: 20 # seconds per PyPI request package_git: enabled: false # disabled after package-level discovery became duplicate-heavy auth_pool: github_main # most package metadata points to GitHub; GitLab 401/403 falls back unauthenticated auth_rotation: per_cycle mode: search # package_git currently supports search mode and custom JSON/git URL targets package_sources: # registries used for package -> repository discovery - npm - pypi queries: - ai - agents - assistant - chatbot - rag - prompt - completions - mcp - openrouter - groq - xai - langchain - litellm - gemini - qwen - kimi - dashscope - aistudio - OR - chat - chats - conversation - conversational - dialogue - copilot - coding - ai-assistant - ai-agent - multi-agent - workflow - workflows - llmops - benchmark - tokens - memory - function-calling pages: 10 # registry search pages per query for repo discovery per_page: 50 # packages per search page refresh_registry: true # merge cached repo candidates with fresh registry discovery each cycle max_targets: 30 # bound one cycle so registry discovery cannot be starved by historical backlog target_claim_order: balanced # split claims between oldest backlog and newest discovered repositories workers: 3 # parallel TruffleHog git scans timeout: 1800 # allow one low-priority Git clone/scan up to 30 minutes versions_per_package: 3 # inspect repo metadata for up to N recent package versions max_version_age_days: 90 # ignore package versions older than this during discovery max_depth: 500 # git commit depth for package-derived repos scan_full_history: false max_commit_age_days: 0 # 0 avoids extra provider API commit-boundary lookup by default commit_lookup_pages: 3 skip_if_commit_lookup_fails: false fetch_timeout: 20 # seconds per registry metadata request postman: enabled: true # first run: enable manually for controlled backfill/tail scans auth_pool: github_main # uses GitHub tokens for code search, commit lookup, and content download auth_rotation: per_cycle # runner state still records a last auth; source-local pool rotates all tokens per request mode: search # search = GitHub code search for Postman JSON artifacts; custom = target_file JSON targets queries: - anthropic - gemini - qwen - kimi - dashscope - llm - azure-openai - openai.azure.com - foundry - services.ai.azure.com - models.ai.azure.com - DASHSCOPE_API_KEY - QWEN_API_KEY - MOONSHOT_API_KEY - KIMI_API_KEY - dashscope.aliyuncs.com - api.moonshot.ai - api.moonshot.cn - GROQ_API_KEY - api.groq.com - OPENROUTER_API_KEY - api.openrouter.ai - REPLICATE_API_TOKEN - api.replicate.com - api.x.ai - HF_TOKEN - HUGGINGFACE_TOKEN - ANTHROPIC_API_KEY - api.anthropic.com - rag - agent - assistant search_kinds: - all # Postman, Insomnia, Bruno, Thunder Client, Hoppscotch, and signature searches pages: 2 # tail default; use 10 for one-time backfill up to GitHub's 1000-result cap per_page: 100 workers: 3 timeout: 300 fetch_timeout: 20 max_targets: 0 # set a small value for smoke tests max_file_age_days: 365 # backfill freshness filter by latest commit touching the file path; 0 disables max_artifact_size_mb: 200 postman_cache_dir: "{postman_cache_dir}" github_code_search_rpm: 8 # safe per-token code search request rate; GitHub limit is about 10/min/token all_tokens_cooldown: 1800 # fallback sleep when all GitHub tokens are rate-limited and no reset is known stop_on_seen_pages: true # tail mode: stop after consecutive fully known pages seen_page_threshold: 2 min_pages_before_stop: 1